Commit 53638176 authored by wangteng's avatar wangteng

首页 分析 功能上线

parent 8efd6c9d
......@@ -230,24 +230,29 @@ class LLMClient:
# 部分兼容服务会先输出 thinking 块。第一次额度不够时,继续请求一次,
# 让模型有足够预算输出最终文本,而不是把 thinking 当作最终结果返回。
# 注意:思考型模型(如 glm-5.3)的 thinking 也计入 max_tokens,
# 小额度(调用方常传 512)极易被思考耗尽;重试阶梯须放得足够大:
# 初始值 → 至少 2048 → 8192 硬顶(去重升序),且不受 cfg.max_tokens 封顶。
initial_budget = max_tokens or self.cfg.max_tokens
retry_budget = min(initial_budget * 2, self.cfg.max_tokens)
token_budgets = (
(initial_budget, retry_budget)
if retry_budget > initial_budget
else (initial_budget,)
)
token_budgets = tuple(sorted(
{b for b in (initial_budget, max(initial_budget * 2, 2048), 8192) if b <= 8192}
))
for attempt, max_tokens in enumerate(token_budgets, start=1):
kwargs = {**base_kwargs, "max_tokens": max_tokens}
msg = self._provider.messages.create(**kwargs)
for block in msg.content:
text = getattr(block, "text", None)
if isinstance(text, str):
if isinstance(text, str) and text.strip():
return text
if attempt == 1 and len(token_budgets) > 1:
if attempt < len(token_budgets):
logger.info(
"LLM 仅返回思考块,使用更高输出额度重试 "
f"(max_tokens={token_budgets[1]})"
f"LLM 仅返回思考块(stop_reason={getattr(msg, 'stop_reason', '?')}),"
f"提高输出额度重试 (max_tokens={token_budgets[attempt]})"
)
logger.warning(
"LLM 响应中没有文本块: "
f"stop_reason={getattr(msg, 'stop_reason', '?')}, "
f"blocks={[getattr(b, 'type', type(b).__name__) for b in msg.content]}"
)
raise LLMUnavailable("LLM 响应中没有可用的文本内容")
except LLMUnavailable:
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment