fix: compact 504 超时修复
- 降低 auto_compact 阈值: 1.2MB → 350KB (426KB 请求也会导致上游 504) - 预截断: compact 前每条消息限制 500 字, 大幅减少 token 数 - 降低批次大小: 800KB → 350KB 每批 - 429 限流自动重试: 最多 2 次, 递增等待 (5s, 10s) - 504 超时自动重试: 1 次, 降级为 auto_compact - compact 内部 429/504 重试: 最多 3 次 - 全局请求日志: before_request 捕获所有请求
This commit is contained in:
1 parent
368ffbe66c
commit
5899032aee
1 file changed
+94
-5
+94
-5
@@ -45,9 +45,12 @@ TARGET_HOST = 'tokenhub.developer.huaweicloud.com'
|
||||
|
||||
# ================= 请求体限制配置 =================
|
||||
APIG_BODY_LIMIT = 1200 * 1024 # APIG 请求体限制 ~1.2MB (实测边界1260KB)
|
||||
COMPACT_THRESHOLD = 350 * 1024 # 自动压缩阈值:超过350KB就压缩(避免上游504超时)
|
||||
GZIP_THRESHOLD = 200 * 1024 # 超过 200KB 时启用 gzip 压缩转发
|
||||
UPSTREAM_TIMEOUT_MIN = 60 # 小请求超时 60s
|
||||
UPSTREAM_TIMEOUT_MAX = 300 # 大请求超时 300s
|
||||
RETRY_ON_429 = 2 # 429限流重试次数
|
||||
RETRY_ON_504 = 1 # 504超时重试次数(之后再降级compact)
|
||||
|
||||
# ================= 日志 =================
|
||||
logging.basicConfig(
|
||||
@@ -299,8 +302,19 @@ def auto_compact(raw_body, real_token, orig_request):
|
||||
if not convo_msgs:
|
||||
return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400
|
||||
|
||||
# 预截断:每条消息内容限制 500 字,大幅减少 compact 请求的 token 数,避免上游 504
|
||||
MAX_MSG_CHARS = 500
|
||||
truncated_msgs = []
|
||||
for m in convo_msgs:
|
||||
content = m.get('content', '')
|
||||
if len(content) > MAX_MSG_CHARS:
|
||||
truncated_msgs.append({**m, 'content': content[:MAX_MSG_CHARS] + '...[截断]'})
|
||||
else:
|
||||
truncated_msgs.append(m)
|
||||
convo_msgs = truncated_msgs
|
||||
|
||||
# 按批次分割对话:每批控制在安全大小内
|
||||
SAFE_BATCH_BYTES = 800 * 1024 # 每批 800KB
|
||||
SAFE_BATCH_BYTES = 350 * 1024 # 每批 350KB(避免上游504超时)
|
||||
batches = []
|
||||
current_batch = []
|
||||
current_size = 0
|
||||
@@ -452,6 +466,15 @@ def auto_compact(raw_body, real_token, orig_request):
|
||||
}
|
||||
|
||||
|
||||
# ================= 全局请求日志(捕获所有请求,包括404) =================
|
||||
@app.before_request
|
||||
def log_every_request():
|
||||
body_preview = ""
|
||||
if request.method in ('POST', 'PUT', 'PATCH') and request.content_length and request.content_length < 2048:
|
||||
body_preview = request.get_data()[:200].decode('utf-8', errors='replace')
|
||||
logger.info(f">>> {request.method} {request.full_path} | body={request.content_length or 0}bytes | from={request.remote_addr} | {body_preview}")
|
||||
|
||||
|
||||
@app.route('/health')
|
||||
def health():
|
||||
"""健康检查端点"""
|
||||
@@ -583,7 +606,7 @@ def compact_endpoint():
|
||||
return {"error": {"message": "无对话内容可压缩", "type": "invalid_request_error"}}, 400
|
||||
|
||||
# 按批次分割对话:每批控制在安全大小内
|
||||
SAFE_BATCH_BYTES = 800 * 1024 # 每批 800KB(留余量给 JSON 开销)
|
||||
SAFE_BATCH_BYTES = 350 * 1024 # 每批 350KB(避免上游504超时)
|
||||
batches = []
|
||||
current_batch = []
|
||||
current_size = 0
|
||||
@@ -636,6 +659,20 @@ def compact_endpoint():
|
||||
data=batch_body, allow_redirects=False,
|
||||
timeout=UPSTREAM_TIMEOUT_MAX, stream=False
|
||||
)
|
||||
|
||||
# compact 内部 429/504 重试
|
||||
for _retry in range(3):
|
||||
if resp.status_code not in (429, 504):
|
||||
break
|
||||
retry_wait = (_retry + 1) * 5
|
||||
logger.warning(f"compact: 第{i+1}批收到 {resp.status_code},等待{retry_wait}s重试...")
|
||||
time.sleep(retry_wait)
|
||||
resp = http_session.request(
|
||||
method='POST', url=target_url, headers=headers,
|
||||
data=batch_body, allow_redirects=False,
|
||||
timeout=UPSTREAM_TIMEOUT_MAX, stream=False
|
||||
)
|
||||
|
||||
data = resp.json()
|
||||
|
||||
if resp.status_code == 200 and 'choices' in data:
|
||||
@@ -791,9 +828,12 @@ def proxy(subpath):
|
||||
else:
|
||||
upstream_timeout = UPSTREAM_TIMEOUT_MIN
|
||||
|
||||
# ============ 超限自动分批压缩(仅对 chat/completions) ============
|
||||
if body_size > APIG_BODY_LIMIT and subpath in ('chat/completions', 'chat/completions/'):
|
||||
logger.info(f"chat/completions 请求体超限: {body_size//1024}KB > {APIG_BODY_LIMIT//1024}KB, 自动触发分批压缩")
|
||||
# ============ 自动压缩(仅对 chat/completions) ============
|
||||
# 1. 超过 APIG 限制 → 必须压缩(否则请求会被截断)
|
||||
# 2. 超过 COMPACT_THRESHOLD → 主动压缩(避免上游504超时)
|
||||
if body_size > COMPACT_THRESHOLD and subpath in ('chat/completions', 'chat/completions/'):
|
||||
reason = "超APIG限制" if body_size > APIG_BODY_LIMIT else "可能超时"
|
||||
logger.info(f"chat/completions 请求体 {body_size//1024}KB > {COMPACT_THRESHOLD//1024}KB ({reason}), 自动触发分批压缩")
|
||||
return auto_compact(raw_body, real_token, request)
|
||||
|
||||
# 非 chat/completions 超限,返回清晰错误
|
||||
@@ -840,6 +880,55 @@ def proxy(subpath):
|
||||
stream=True
|
||||
)
|
||||
|
||||
# ============ 429 限流重试 ============
|
||||
if resp.status_code == 429:
|
||||
for retry_i in range(1, RETRY_ON_429 + 1):
|
||||
resp.close()
|
||||
wait = retry_i * 5 # 5s, 10s
|
||||
logger.warning(f"上游 429 限流,等待{wait}s后重试({retry_i}/{RETRY_ON_429})...")
|
||||
time.sleep(wait)
|
||||
resp = http_session.request(
|
||||
method=request.method, url=target_url, headers=headers,
|
||||
data=raw_body, cookies=request.cookies,
|
||||
allow_redirects=False, timeout=upstream_timeout, stream=True
|
||||
)
|
||||
if resp.status_code != 429:
|
||||
break
|
||||
logger.warning(f"重试仍返回 429")
|
||||
|
||||
# ============ 504 超时重试 + 降级compact ============
|
||||
if resp.status_code == 504 and subpath in ('chat/completions', 'chat/completions/'):
|
||||
for retry_i in range(1, RETRY_ON_504 + 1):
|
||||
resp.close()
|
||||
wait = retry_i * 3
|
||||
logger.warning(f"上游 504 超时,等待{wait}s后重试({retry_i}/{RETRY_ON_504})...")
|
||||
time.sleep(wait)
|
||||
resp = http_session.request(
|
||||
method=request.method, url=target_url, headers=headers,
|
||||
data=raw_body, cookies=request.cookies,
|
||||
allow_redirects=False, timeout=upstream_timeout, stream=True
|
||||
)
|
||||
if resp.status_code != 504:
|
||||
break
|
||||
|
||||
# 重试仍504 → 降级为compact压缩后重试
|
||||
if resp.status_code == 504:
|
||||
resp.close()
|
||||
logger.warning(f"504重试仍失败,降级为auto_compact压缩后重试...")
|
||||
try:
|
||||
return auto_compact(raw_body, real_token, request)
|
||||
except Exception as e:
|
||||
logger.error(f"降级compact也失败: {e}")
|
||||
# compact也失败,返回友好错误
|
||||
return {
|
||||
"error": {
|
||||
"message": "模型推理超时,已尝试压缩上下文但仍失败。请缩短对话后重试。",
|
||||
"type": "server_error",
|
||||
"code": "model_timeout",
|
||||
"param": None
|
||||
}
|
||||
}, 504
|
||||
|
||||
# 过滤 hop-by-hop 头和压缩编码头
|
||||
skip_headers = {'transfer-encoding', 'content-encoding', 'content-length',
|
||||
'connection', 'keep-alive', 'upgrade'}
|
||||
|
||||
Reference in new issue
Block a user