fix: auto_compact压缩后用摘要+用户问题重新请求模型,而非直接返回摘要
This commit is contained in:
1 parent
0ffbf20f6b
commit
34eb36623c
1 file changed
+90
-47
+90
-47
@@ -412,58 +412,101 @@ def auto_compact(raw_body, real_token, orig_request):
|
|||||||
except Exception:
|
except Exception:
|
||||||
final_summary = "\n".join(summaries)
|
final_summary = "\n".join(summaries)
|
||||||
|
|
||||||
logger.info(f"auto_compact: 完成, {body_size//1024}KB → {len(final_summary)}字 ({num_batches}批)")
|
logger.info(f"auto_compact: 压缩完成, {body_size//1024}KB → {len(final_summary)}字 ({num_batches}批)")
|
||||||
|
|
||||||
# 返回标准 OpenAI 格式(客户端无感知)
|
# ============ 用压缩后的摘要+用户最新问题,重新请求模型 ============
|
||||||
if stream:
|
# 提取用户最后一条消息
|
||||||
chat_id = str(uuid.uuid4())
|
last_user_msg = None
|
||||||
created = int(time.time())
|
for m in reversed(convo_msgs):
|
||||||
|
if m.get('role') == 'user':
|
||||||
|
last_user_msg = m
|
||||||
|
break
|
||||||
|
|
||||||
def compact_sse():
|
# 构建压缩后的请求:system + 摘要(作为assistant上下文) + 用户最新问题
|
||||||
first = {
|
compact_messages = system_msgs + [
|
||||||
"id": chat_id, "object": "chat.completion.chunk", "created": created,
|
{"role": "assistant", "content": f"[上下文压缩摘要]\n{final_summary}"}
|
||||||
"model": model,
|
]
|
||||||
"choices": [{"index": 0, "delta": {"role": "assistant", "content": ""}, "finish_reason": None}]
|
if last_user_msg:
|
||||||
}
|
compact_messages.append(last_user_msg)
|
||||||
yield f"data: {_json.dumps(first, ensure_ascii=False)}\n\n"
|
|
||||||
|
|
||||||
content_chunk = {
|
compact_payload = {
|
||||||
"id": chat_id, "object": "chat.completion.chunk", "created": created,
|
"model": model,
|
||||||
"model": model,
|
"messages": compact_messages,
|
||||||
"choices": [{"index": 0, "delta": {"content": final_summary}, "finish_reason": None}]
|
"max_tokens": max_tokens,
|
||||||
}
|
"stream": stream,
|
||||||
yield f"data: {_json.dumps(content_chunk, ensure_ascii=False)}\n\n"
|
}
|
||||||
|
# 保留原始请求中的其他参数
|
||||||
|
for k in ('temperature', 'top_p', 'presence_penalty', 'frequency_penalty'):
|
||||||
|
if k in payload:
|
||||||
|
compact_payload[k] = payload[k]
|
||||||
|
|
||||||
done_chunk = {
|
compact_body = _json.dumps(compact_payload, ensure_ascii=False).encode('utf-8')
|
||||||
"id": chat_id, "object": "chat.completion.chunk", "created": created,
|
compact_headers = {
|
||||||
"model": model,
|
'Authorization': f'Bearer {real_token}',
|
||||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]
|
'Host': TARGET_HOST,
|
||||||
}
|
'Content-Type': 'application/json',
|
||||||
yield f"data: {_json.dumps(done_chunk, ensure_ascii=False)}\n\n"
|
}
|
||||||
yield "data: [DONE]\n\n"
|
target_url = f'https://{TARGET_HOST}/v2/chat/completions'
|
||||||
|
|
||||||
return Response(compact_sse(), status=200, headers={
|
logger.info(f"auto_compact: 用压缩上下文重新请求模型 ({len(compact_body)//1024}KB, stream={stream})")
|
||||||
'Content-Type': 'text/event-stream;charset=UTF-8',
|
|
||||||
'Cache-Control': 'no-cache',
|
try:
|
||||||
'X-Auto-Compact': f'batches={num_batches},original_kb={body_size//1024}',
|
resp = http_session.request(
|
||||||
})
|
method='POST', url=target_url, headers=compact_headers,
|
||||||
else:
|
data=compact_body, allow_redirects=False,
|
||||||
return {
|
timeout=UPSTREAM_TIMEOUT_MAX, stream=True
|
||||||
"id": f"chatcmpl-compact-{int(time.time())}",
|
)
|
||||||
"object": "chat.completion",
|
|
||||||
"created": int(time.time()),
|
# 401 重试
|
||||||
"model": model,
|
if resp.status_code == 401:
|
||||||
"choices": [{
|
resp.close()
|
||||||
"index": 0,
|
cache.blacklist_current()
|
||||||
"message": {"role": "assistant", "content": final_summary},
|
new_token = find_token_in_memory()
|
||||||
"finish_reason": "stop"
|
if new_token:
|
||||||
}],
|
compact_headers['Authorization'] = f'Bearer {new_token}'
|
||||||
"usage": {
|
resp = http_session.request(
|
||||||
"prompt_tokens": body_size // 4,
|
method='POST', url=target_url, headers=compact_headers,
|
||||||
"completion_tokens": len(final_summary),
|
data=compact_body, allow_redirects=False,
|
||||||
"total_tokens": body_size // 4 + len(final_summary)
|
timeout=UPSTREAM_TIMEOUT_MAX, stream=True
|
||||||
}
|
)
|
||||||
}
|
|
||||||
|
if resp.status_code != 200:
|
||||||
|
try:
|
||||||
|
err_body = resp.content[:500]
|
||||||
|
logger.error(f"auto_compact: 重新请求模型失败: HTTP {resp.status_code} - {err_body.decode('utf-8', errors='replace')}")
|
||||||
|
except:
|
||||||
|
logger.error(f"auto_compact: 重新请求模型失败: HTTP {resp.status_code}")
|
||||||
|
resp.close()
|
||||||
|
# 降级:返回摘要
|
||||||
|
return {"error": {"message": f"压缩后重新请求失败(HTTP {resp.status_code}),上下文摘要: {final_summary[:500]}", "type": "server_error"}}, 502
|
||||||
|
|
||||||
|
# 流式转发
|
||||||
|
content_type = resp.headers.get('Content-Type', '')
|
||||||
|
if 'text/event-stream' in content_type or stream:
|
||||||
|
skip_h = {'transfer-encoding', 'content-encoding', 'content-length', 'connection', 'keep-alive', 'upgrade'}
|
||||||
|
resp_headers = [(k, v) for k, v in resp.headers.items() if k.lower() not in skip_h]
|
||||||
|
resp_headers.append(('X-Auto-Compact', f'batches={num_batches},original_kb={body_size//1024}'))
|
||||||
|
|
||||||
|
def compact_stream():
|
||||||
|
try:
|
||||||
|
for chunk in resp.iter_content(chunk_size=16384):
|
||||||
|
if chunk:
|
||||||
|
yield chunk
|
||||||
|
finally:
|
||||||
|
resp.close()
|
||||||
|
|
||||||
|
return Response(compact_stream(), status=200, headers=resp_headers, direct_passthrough=True)
|
||||||
|
else:
|
||||||
|
# 非流式
|
||||||
|
content = resp.content
|
||||||
|
resp.close()
|
||||||
|
# 在响应头中标记经过了压缩
|
||||||
|
return Response(content, status=200,
|
||||||
|
headers={'Content-Type': 'application/json', 'X-Auto-Compact': f'batches={num_batches},original_kb={body_size//1024}'})
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"auto_compact: 重新请求模型异常: {e}")
|
||||||
|
return {"error": {"message": f"压缩后请求异常: {str(e)}", "type": "server_error"}}, 500
|
||||||
|
|
||||||
|
|
||||||
# ================= 全局请求日志(捕获所有请求,包括404) =================
|
# ================= 全局请求日志(捕获所有请求,包括404) =================
|
||||||
|
|||||||
Reference in new issue
Block a user