fix: compact 504 超时修复

- 降低 auto_compact 阈值: 1.2MB → 350KB (426KB 请求也会导致上游 504)
- 预截断: compact 前每条消息限制 500 字, 大幅减少 token 数
- 降低批次大小: 800KB → 350KB 每批
- 429 限流自动重试: 最多 2 次, 递增等待 (5s, 10s)
- 504 超时自动重试: 1 次, 降级为 auto_compact
- compact 内部 429/504 重试: 最多 3 次
- 全局请求日志: before_request 捕获所有请求
This commit is contained in:
chaos committed 2026-07-08 15:12:29 +08:00
1 parent 368ffbe66c
commit 5899032aee
1 file changed
+94 -5
+94 -5
View File
@@ -45,9 +45,12 @@ TARGET_HOST = 'tokenhub.developer.huaweicloud.com'
# ================= 请求体限制配置 ================= # ================= 请求体限制配置 =================
APIG_BODY_LIMIT = 1200 * 1024 # APIG 请求体限制 ~1.2MB (实测边界1260KB) APIG_BODY_LIMIT = 1200 * 1024 # APIG 请求体限制 ~1.2MB (实测边界1260KB)
COMPACT_THRESHOLD = 350 * 1024 # 自动压缩阈值:超过350KB就压缩(避免上游504超时)
GZIP_THRESHOLD = 200 * 1024 # 超过 200KB 时启用 gzip 压缩转发 GZIP_THRESHOLD = 200 * 1024 # 超过 200KB 时启用 gzip 压缩转发
UPSTREAM_TIMEOUT_MIN = 60 # 小请求超时 60s UPSTREAM_TIMEOUT_MIN = 60 # 小请求超时 60s
UPSTREAM_TIMEOUT_MAX = 300 # 大请求超时 300s UPSTREAM_TIMEOUT_MAX = 300 # 大请求超时 300s
RETRY_ON_429 = 2 # 429限流重试次数
RETRY_ON_504 = 1 # 504超时重试次数(之后再降级compact)
# ================= 日志 ================= # ================= 日志 =================
logging.basicConfig( logging.basicConfig(
@@ -299,8 +302,19 @@ def auto_compact(raw_body, real_token, orig_request):
if not convo_msgs: if not convo_msgs:
return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400 return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400
# 预截断:每条消息内容限制 500 字,大幅减少 compact 请求的 token 数,避免上游 504
MAX_MSG_CHARS = 500
truncated_msgs = []
for m in convo_msgs:
content = m.get('content', '')
if len(content) > MAX_MSG_CHARS:
truncated_msgs.append({**m, 'content': content[:MAX_MSG_CHARS] + '...[截断]'})
else:
truncated_msgs.append(m)
convo_msgs = truncated_msgs
# 按批次分割对话:每批控制在安全大小内 # 按批次分割对话:每批控制在安全大小内
SAFE_BATCH_BYTES = 800 * 1024 # 每批 800KB SAFE_BATCH_BYTES = 350 * 1024 # 每批 350KB(避免上游504超时)
batches = [] batches = []
current_batch = [] current_batch = []
current_size = 0 current_size = 0
@@ -452,6 +466,15 @@ def auto_compact(raw_body, real_token, orig_request):
} }
# ================= 全局请求日志(捕获所有请求,包括404) =================
@app.before_request
def log_every_request():
body_preview = ""
if request.method in ('POST', 'PUT', 'PATCH') and request.content_length and request.content_length < 2048:
body_preview = request.get_data()[:200].decode('utf-8', errors='replace')
logger.info(f">>> {request.method} {request.full_path} | body={request.content_length or 0}bytes | from={request.remote_addr} | {body_preview}")
@app.route('/health') @app.route('/health')
def health(): def health():
"""健康检查端点""" """健康检查端点"""
@@ -583,7 +606,7 @@ def compact_endpoint():
return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400 return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400
# 按批次分割对话:每批控制在安全大小内 # 按批次分割对话:每批控制在安全大小内
SAFE_BATCH_BYTES = 800 * 1024 # 每批 800KB(留余量给 JSON 开销) SAFE_BATCH_BYTES = 350 * 1024 # 每批 350KB(避免上游504超时)
batches = [] batches = []
current_batch = [] current_batch = []
current_size = 0 current_size = 0
@@ -636,6 +659,20 @@ def compact_endpoint():
data=batch_body, allow_redirects=False, data=batch_body, allow_redirects=False,
timeout=UPSTREAM_TIMEOUT_MAX, stream=False timeout=UPSTREAM_TIMEOUT_MAX, stream=False
) )
# compact 内部 429/504 重试
for _retry in range(3):
if resp.status_code not in (429, 504):
break
retry_wait = (_retry + 1) * 5
logger.warning(f"compact: 第{i+1}批收到 {resp.status_code},等待{retry_wait}s重试...")
time.sleep(retry_wait)
resp = http_session.request(
method='POST', url=target_url, headers=headers,
data=batch_body, allow_redirects=False,
timeout=UPSTREAM_TIMEOUT_MAX, stream=False
)
data = resp.json() data = resp.json()
if resp.status_code == 200 and 'choices' in data: if resp.status_code == 200 and 'choices' in data:
@@ -791,9 +828,12 @@ def proxy(subpath):
else: else:
upstream_timeout = UPSTREAM_TIMEOUT_MIN upstream_timeout = UPSTREAM_TIMEOUT_MIN
# ============ 超限自动分批压缩(仅对 chat/completions) ============ # ============ 自动压缩(仅对 chat/completions) ============
if body_size > APIG_BODY_LIMIT and subpath in ('chat/completions', 'chat/completions/'): # 1. 超过 APIG 限制 → 必须压缩(否则请求会被截断)
logger.info(f"chat/completions 请求体超限: {body_size//1024}KB > {APIG_BODY_LIMIT//1024}KB, 自动触发分批压缩") # 2. 超过 COMPACT_THRESHOLD → 主动压缩(避免上游504超时)
if body_size > COMPACT_THRESHOLD and subpath in ('chat/completions', 'chat/completions/'):
reason = "超APIG限制" if body_size > APIG_BODY_LIMIT else "可能超时"
logger.info(f"chat/completions 请求体 {body_size//1024}KB > {COMPACT_THRESHOLD//1024}KB ({reason}), 自动触发分批压缩")
return auto_compact(raw_body, real_token, request) return auto_compact(raw_body, real_token, request)
# 非 chat/completions 超限,返回清晰错误 # 非 chat/completions 超限,返回清晰错误
@@ -840,6 +880,55 @@ def proxy(subpath):
stream=True stream=True
) )
# ============ 429 限流重试 ============
if resp.status_code == 429:
for retry_i in range(1, RETRY_ON_429 + 1):
resp.close()
wait = retry_i * 5 # 5s, 10s
logger.warning(f"上游 429 限流,等待{wait}s后重试({retry_i}/{RETRY_ON_429})...")
time.sleep(wait)
resp = http_session.request(
method=request.method, url=target_url, headers=headers,
data=raw_body, cookies=request.cookies,
allow_redirects=False, timeout=upstream_timeout, stream=True
)
if resp.status_code != 429:
break
logger.warning(f"重试仍返回 429")
# ============ 504 超时重试 + 降级compact ============
if resp.status_code == 504 and subpath in ('chat/completions', 'chat/completions/'):
for retry_i in range(1, RETRY_ON_504 + 1):
resp.close()
wait = retry_i * 3
logger.warning(f"上游 504 超时,等待{wait}s后重试({retry_i}/{RETRY_ON_504})...")
time.sleep(wait)
resp = http_session.request(
method=request.method, url=target_url, headers=headers,
data=raw_body, cookies=request.cookies,
allow_redirects=False, timeout=upstream_timeout, stream=True
)
if resp.status_code != 504:
break
# 重试仍504 → 降级为compact压缩后重试
if resp.status_code == 504:
resp.close()
logger.warning(f"504重试仍失败,降级为auto_compact压缩后重试...")
try:
return auto_compact(raw_body, real_token, request)
except Exception as e:
logger.error(f"降级compact也失败: {e}")
# compact也失败,返回友好错误
return {
"error": {
"message": "模型推理超时,已尝试压缩上下文但仍失败。请缩短对话后重试。",
"type": "server_error",
"code": "model_timeout",
"param": None
}
}, 504
# 过滤 hop-by-hop 头和压缩编码头 # 过滤 hop-by-hop 头和压缩编码头
skip_headers = {'transfer-encoding', 'content-encoding', 'content-length', skip_headers = {'transfer-encoding', 'content-encoding', 'content-length',
'connection', 'keep-alive', 'upgrade'} 'connection', 'keep-alive', 'upgrade'}