fix: auto_compact压缩后用摘要+用户问题重新请求模型,而非直接返回摘要

This commit is contained in:
chaos committed 2026-07-08 17:00:42 +08:00
1 parent 0ffbf20f6b
commit 34eb36623c
1 file changed
+90 -47
+90 -47
View File
@@ -412,58 +412,101 @@ def auto_compact(raw_body, real_token, orig_request):
except Exception: except Exception:
final_summary = "\n".join(summaries) final_summary = "\n".join(summaries)
logger.info(f"auto_compact: 完成, {body_size//1024}KB → {len(final_summary)}字 ({num_batches}批)") logger.info(f"auto_compact: 压缩完成, {body_size//1024}KB → {len(final_summary)}字 ({num_batches}批)")
# 返回标准 OpenAI 格式(客户端无感知) # ============ 用压缩后的摘要+用户最新问题,重新请求模型 ============
if stream: # 提取用户最后一条消息
chat_id = str(uuid.uuid4()) last_user_msg = None
created = int(time.time()) for m in reversed(convo_msgs):
if m.get('role') == 'user':
last_user_msg = m
break
def compact_sse(): # 构建压缩后的请求:system + 摘要(作为assistant上下文) + 用户最新问题
first = { compact_messages = system_msgs + [
"id": chat_id, "object": "chat.completion.chunk", "created": created, {"role": "assistant", "content": f"[上下文压缩摘要]\n{final_summary}"}
"model": model, ]
"choices": [{"index": 0, "delta": {"role": "assistant", "content": ""}, "finish_reason": None}] if last_user_msg:
} compact_messages.append(last_user_msg)
yield f"data: {_json.dumps(first, ensure_ascii=False)}\n\n"
content_chunk = { compact_payload = {
"id": chat_id, "object": "chat.completion.chunk", "created": created, "model": model,
"model": model, "messages": compact_messages,
"choices": [{"index": 0, "delta": {"content": final_summary}, "finish_reason": None}] "max_tokens": max_tokens,
} "stream": stream,
yield f"data: {_json.dumps(content_chunk, ensure_ascii=False)}\n\n" }
# 保留原始请求中的其他参数
for k in ('temperature', 'top_p', 'presence_penalty', 'frequency_penalty'):
if k in payload:
compact_payload[k] = payload[k]
done_chunk = { compact_body = _json.dumps(compact_payload, ensure_ascii=False).encode('utf-8')
"id": chat_id, "object": "chat.completion.chunk", "created": created, compact_headers = {
"model": model, 'Authorization': f'Bearer {real_token}',
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}] 'Host': TARGET_HOST,
} 'Content-Type': 'application/json',
yield f"data: {_json.dumps(done_chunk, ensure_ascii=False)}\n\n" }
yield "data: [DONE]\n\n" target_url = f'https://{TARGET_HOST}/v2/chat/completions'
return Response(compact_sse(), status=200, headers={ logger.info(f"auto_compact: 用压缩上下文重新请求模型 ({len(compact_body)//1024}KB, stream={stream})")
'Content-Type': 'text/event-stream;charset=UTF-8',
'Cache-Control': 'no-cache', try:
'X-Auto-Compact': f'batches={num_batches},original_kb={body_size//1024}', resp = http_session.request(
}) method='POST', url=target_url, headers=compact_headers,
else: data=compact_body, allow_redirects=False,
return { timeout=UPSTREAM_TIMEOUT_MAX, stream=True
"id": f"chatcmpl-compact-{int(time.time())}", )
"object": "chat.completion",
"created": int(time.time()), # 401 重试
"model": model, if resp.status_code == 401:
"choices": [{ resp.close()
"index": 0, cache.blacklist_current()
"message": {"role": "assistant", "content": final_summary}, new_token = find_token_in_memory()
"finish_reason": "stop" if new_token:
}], compact_headers['Authorization'] = f'Bearer {new_token}'
"usage": { resp = http_session.request(
"prompt_tokens": body_size // 4, method='POST', url=target_url, headers=compact_headers,
"completion_tokens": len(final_summary), data=compact_body, allow_redirects=False,
"total_tokens": body_size // 4 + len(final_summary) timeout=UPSTREAM_TIMEOUT_MAX, stream=True
} )
}
if resp.status_code != 200:
try:
err_body = resp.content[:500]
logger.error(f"auto_compact: 重新请求模型失败: HTTP {resp.status_code} - {err_body.decode('utf-8', errors='replace')}")
except:
logger.error(f"auto_compact: 重新请求模型失败: HTTP {resp.status_code}")
resp.close()
# 降级:返回摘要
return {"error": {"message": f"压缩后重新请求失败(HTTP {resp.status_code}),上下文摘要: {final_summary[:500]}", "type": "server_error"}}, 502
# 流式转发
content_type = resp.headers.get('Content-Type', '')
if 'text/event-stream' in content_type or stream:
skip_h = {'transfer-encoding', 'content-encoding', 'content-length', 'connection', 'keep-alive', 'upgrade'}
resp_headers = [(k, v) for k, v in resp.headers.items() if k.lower() not in skip_h]
resp_headers.append(('X-Auto-Compact', f'batches={num_batches},original_kb={body_size//1024}'))
def compact_stream():
try:
for chunk in resp.iter_content(chunk_size=16384):
if chunk:
yield chunk
finally:
resp.close()
return Response(compact_stream(), status=200, headers=resp_headers, direct_passthrough=True)
else:
# 非流式
content = resp.content
resp.close()
# 在响应头中标记经过了压缩
return Response(content, status=200,
headers={'Content-Type': 'application/json', 'X-Auto-Compact': f'batches={num_batches},original_kb={body_size//1024}'})
except Exception as e:
logger.error(f"auto_compact: 重新请求模型异常: {e}")
return {"error": {"message": f"压缩后请求异常: {str(e)}", "type": "server_error"}}, 500
# ================= 全局请求日志(捕获所有请求,包括404) ================= # ================= 全局请求日志(捕获所有请求,包括404) =================