diff --git a/ai/huawei_gateway.py b/ai/huawei_gateway.py index 23a7d33..df4cca1 100755 --- a/ai/huawei_gateway.py +++ b/ai/huawei_gateway.py @@ -45,9 +45,12 @@ TARGET_HOST = 'tokenhub.developer.huaweicloud.com' # ================= 请求体限制配置 ================= APIG_BODY_LIMIT = 1200 * 1024 # APIG 请求体限制 ~1.2MB (实测边界1260KB) +COMPACT_THRESHOLD = 350 * 1024 # 自动压缩阈值:超过350KB就压缩(避免上游504超时) GZIP_THRESHOLD = 200 * 1024 # 超过 200KB 时启用 gzip 压缩转发 UPSTREAM_TIMEOUT_MIN = 60 # 小请求超时 60s UPSTREAM_TIMEOUT_MAX = 300 # 大请求超时 300s +RETRY_ON_429 = 2 # 429限流重试次数 +RETRY_ON_504 = 1 # 504超时重试次数(之后再降级compact) # ================= 日志 ================= logging.basicConfig( @@ -299,8 +302,19 @@ def auto_compact(raw_body, real_token, orig_request): if not convo_msgs: return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400 + # 预截断:每条消息内容限制 500 字,大幅减少 compact 请求的 token 数,避免上游 504 + MAX_MSG_CHARS = 500 + truncated_msgs = [] + for m in convo_msgs: + content = m.get('content', '') + if len(content) > MAX_MSG_CHARS: + truncated_msgs.append({**m, 'content': content[:MAX_MSG_CHARS] + '...[截断]'}) + else: + truncated_msgs.append(m) + convo_msgs = truncated_msgs + # 按批次分割对话:每批控制在安全大小内 - SAFE_BATCH_BYTES = 800 * 1024 # 每批 800KB + SAFE_BATCH_BYTES = 350 * 1024 # 每批 350KB(避免上游504超时) batches = [] current_batch = [] current_size = 0 @@ -452,6 +466,15 @@ def auto_compact(raw_body, real_token, orig_request): } +# ================= 全局请求日志(捕获所有请求,包括404) ================= +@app.before_request +def log_every_request(): + body_preview = "" + if request.method in ('POST', 'PUT', 'PATCH') and request.content_length and request.content_length < 2048: + body_preview = request.get_data()[:200].decode('utf-8', errors='replace') + logger.info(f">>> {request.method} {request.full_path} | body={request.content_length or 0}bytes | from={request.remote_addr} | {body_preview}") + + @app.route('/health') def health(): """健康检查端点""" @@ -583,7 +606,7 @@ def compact_endpoint(): return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400 # 按批次分割对话:每批控制在安全大小内 - SAFE_BATCH_BYTES = 800 * 1024 # 每批 800KB(留余量给 JSON 开销) + SAFE_BATCH_BYTES = 350 * 1024 # 每批 350KB(避免上游504超时) batches = [] current_batch = [] current_size = 0 @@ -636,6 +659,20 @@ def compact_endpoint(): data=batch_body, allow_redirects=False, timeout=UPSTREAM_TIMEOUT_MAX, stream=False ) + + # compact 内部 429/504 重试 + for _retry in range(3): + if resp.status_code not in (429, 504): + break + retry_wait = (_retry + 1) * 5 + logger.warning(f"compact: 第{i+1}批收到 {resp.status_code},等待{retry_wait}s重试...") + time.sleep(retry_wait) + resp = http_session.request( + method='POST', url=target_url, headers=headers, + data=batch_body, allow_redirects=False, + timeout=UPSTREAM_TIMEOUT_MAX, stream=False + ) + data = resp.json() if resp.status_code == 200 and 'choices' in data: @@ -791,9 +828,12 @@ def proxy(subpath): else: upstream_timeout = UPSTREAM_TIMEOUT_MIN - # ============ 超限自动分批压缩(仅对 chat/completions) ============ - if body_size > APIG_BODY_LIMIT and subpath in ('chat/completions', 'chat/completions/'): - logger.info(f"chat/completions 请求体超限: {body_size//1024}KB > {APIG_BODY_LIMIT//1024}KB, 自动触发分批压缩") + # ============ 自动压缩(仅对 chat/completions) ============ + # 1. 超过 APIG 限制 → 必须压缩(否则请求会被截断) + # 2. 超过 COMPACT_THRESHOLD → 主动压缩(避免上游504超时) + if body_size > COMPACT_THRESHOLD and subpath in ('chat/completions', 'chat/completions/'): + reason = "超APIG限制" if body_size > APIG_BODY_LIMIT else "可能超时" + logger.info(f"chat/completions 请求体 {body_size//1024}KB > {COMPACT_THRESHOLD//1024}KB ({reason}), 自动触发分批压缩") return auto_compact(raw_body, real_token, request) # 非 chat/completions 超限,返回清晰错误 @@ -840,6 +880,55 @@ def proxy(subpath): stream=True ) + # ============ 429 限流重试 ============ + if resp.status_code == 429: + for retry_i in range(1, RETRY_ON_429 + 1): + resp.close() + wait = retry_i * 5 # 5s, 10s + logger.warning(f"上游 429 限流,等待{wait}s后重试({retry_i}/{RETRY_ON_429})...") + time.sleep(wait) + resp = http_session.request( + method=request.method, url=target_url, headers=headers, + data=raw_body, cookies=request.cookies, + allow_redirects=False, timeout=upstream_timeout, stream=True + ) + if resp.status_code != 429: + break + logger.warning(f"重试仍返回 429") + + # ============ 504 超时重试 + 降级compact ============ + if resp.status_code == 504 and subpath in ('chat/completions', 'chat/completions/'): + for retry_i in range(1, RETRY_ON_504 + 1): + resp.close() + wait = retry_i * 3 + logger.warning(f"上游 504 超时,等待{wait}s后重试({retry_i}/{RETRY_ON_504})...") + time.sleep(wait) + resp = http_session.request( + method=request.method, url=target_url, headers=headers, + data=raw_body, cookies=request.cookies, + allow_redirects=False, timeout=upstream_timeout, stream=True + ) + if resp.status_code != 504: + break + + # 重试仍504 → 降级为compact压缩后重试 + if resp.status_code == 504: + resp.close() + logger.warning(f"504重试仍失败,降级为auto_compact压缩后重试...") + try: + return auto_compact(raw_body, real_token, request) + except Exception as e: + logger.error(f"降级compact也失败: {e}") + # compact也失败,返回友好错误 + return { + "error": { + "message": "模型推理超时,已尝试压缩上下文但仍失败。请缩短对话后重试。", + "type": "server_error", + "code": "model_timeout", + "param": None + } + }, 504 + # 过滤 hop-by-hop 头和压缩编码头 skip_headers = {'transfer-encoding', 'content-encoding', 'content-length', 'connection', 'keep-alive', 'upgrade'}