feat: Token Plan 迁移与 .env 热加载,并修复日报 AI 摘要为空
Token Plan 迁移 / 配置热加载:
- configs/llm_models.yaml: 各场景切到 Token Plan(deepseek-v4.1-flash / qwen3.6-flash)
- 新增 configs/runtime_env.py: .env 按 (mtime_ns, size) 热加载并同步 os.environ,
统一 env_get 取值;llm / embedding / vectorstore / mcp / pipeline 改用 env_get
- configs/loader.py / scripts/run_scheduler.py 等配套调整
- 新增 tests/test_hot_reload.py
日报 AI 摘要为空修复(2026-09-25):
- 根因: 推理模型的 reasoning token 与正文共用 max_tokens, 预算 1500 被"思考"
占满 -> text_tokens=0 / finish_reason=length, 摘要静默为空且不重试
- daily_report 场景新增 max_tokens(默认 4000, YAML 保存即热生效);
LLMConfig 支持可选 max_tokens; 分块预算 800 -> 2000
- _llm_call 拆出 _call_once, 正文为空时自动加倍预算重试(上限 16000),
用尽才降级返回空串; 网络异常重试语义不变
- docs/user-guide.md 新增 FAQ; continuation.md 记录本次排查
- 已重跑 2026-09-25 日报(report_id=357)补回 466 字摘要
测试: 相关用例 56 passed(test_hot_reload 12 passed);
ruff 无新增问题; 3 个 crawler 既有失败与本改动无关
This commit is contained in:
@@ -132,8 +132,8 @@ class TestLlmCallRetry:
|
||||
|
||||
def test_retry_then_success(self, monkeypatch) -> None:
|
||||
import scheduler.reporter as rep
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 3)
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01)
|
||||
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 3)
|
||||
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
|
||||
client, n = self._fake_client(2) # 前 2 次失败,第 3 次成功
|
||||
out = rep._llm_call(client, self._cfg(), "p")
|
||||
assert out == "今日要点摘要"
|
||||
@@ -141,8 +141,8 @@ class TestLlmCallRetry:
|
||||
|
||||
def test_exhausts_retries_raises(self, monkeypatch) -> None:
|
||||
import scheduler.reporter as rep
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 2)
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01)
|
||||
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 2)
|
||||
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
|
||||
client, n = self._fake_client(99) # 一直失败
|
||||
with pytest.raises(ConnectionError):
|
||||
rep._llm_call(client, self._cfg(), "p")
|
||||
@@ -171,6 +171,84 @@ class TestLlmCallRetry:
|
||||
assert n["count"] == 3 # 2 块 + 1 次合并
|
||||
|
||||
|
||||
class TestReasoningBudgetEscalation:
|
||||
"""推理模型占满 max_tokens 导致正文为空时的预算升级(回归 9-25 无 AI 摘要)。"""
|
||||
|
||||
@staticmethod
|
||||
def _empty_then_ok_client(empty_times: int):
|
||||
"""前 empty_times 次返回空正文 + finish_reason=length,之后返回正常摘要。"""
|
||||
from types import SimpleNamespace
|
||||
|
||||
seen: list[int] = []
|
||||
|
||||
class Completions:
|
||||
def create(self, **kwargs):
|
||||
seen.append(kwargs.get("max_tokens"))
|
||||
if len(seen) <= empty_times:
|
||||
return SimpleNamespace(
|
||||
choices=[SimpleNamespace(
|
||||
message=SimpleNamespace(content=""),
|
||||
finish_reason="length",
|
||||
)]
|
||||
)
|
||||
return SimpleNamespace(
|
||||
choices=[SimpleNamespace(
|
||||
message=SimpleNamespace(content="恢复后的摘要"),
|
||||
finish_reason="stop",
|
||||
)]
|
||||
)
|
||||
|
||||
return SimpleNamespace(chat=SimpleNamespace(completions=Completions())), seen
|
||||
|
||||
@staticmethod
|
||||
def _cfg(max_tokens: int | None = None):
|
||||
from llm.client import LLMConfig
|
||||
|
||||
return LLMConfig(
|
||||
provider="qwen", model="deepseek-v4.1-flash",
|
||||
api_key="sk-test", base_url="https://example.invalid/v1",
|
||||
temperature=0.3, max_tokens=max_tokens,
|
||||
)
|
||||
|
||||
def test_empty_content_escalates_and_recovers(self, monkeypatch) -> None:
|
||||
"""正文为空时自动加倍预算并最终拿到摘要(不再静默返回空)。"""
|
||||
import scheduler.reporter as rep
|
||||
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 3)
|
||||
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
|
||||
client, seen = self._empty_then_ok_client(1)
|
||||
out = rep._llm_call(client, self._cfg(4000), "p")
|
||||
assert out == "恢复后的摘要"
|
||||
assert seen == [4000, 8000] # 首次失败后预算翻倍
|
||||
|
||||
def test_scene_max_tokens_wins_over_default(self, monkeypatch) -> None:
|
||||
"""场景 max_tokens 生效;未配置时回退代码默认 4000。"""
|
||||
import scheduler.reporter as rep
|
||||
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 1)
|
||||
client, seen = self._empty_then_ok_client(0)
|
||||
rep._llm_call(client, self._cfg(6000), "p")
|
||||
assert seen == [6000]
|
||||
client, seen = self._empty_then_ok_client(0)
|
||||
rep._llm_call(client, self._cfg(), "p")
|
||||
assert seen == [rep.DEFAULT_SUMMARY_MAX_TOKENS]
|
||||
|
||||
def test_all_empty_returns_blank_without_raising(self, monkeypatch) -> None:
|
||||
"""预算升级用尽仍为空时降级返回空串(日报仍可入库,不抛异常)。"""
|
||||
import scheduler.reporter as rep
|
||||
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 3)
|
||||
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
|
||||
client, seen = self._empty_then_ok_client(99)
|
||||
out = rep._llm_call(client, self._cfg(4000), "p")
|
||||
assert out == ""
|
||||
# 4000 → 8000 → 16000(受 MAX_SUMMARY_MAX_TOKENS 上限约束)
|
||||
assert seen == [4000, 8000, 16000]
|
||||
|
||||
def test_budget_never_exceeds_cap(self) -> None:
|
||||
"""升级预算不超过 MAX_SUMMARY_MAX_TOKENS,避免无限放大。"""
|
||||
import scheduler.reporter as rep
|
||||
assert rep.MAX_SUMMARY_MAX_TOKENS == 16000
|
||||
assert rep.DEFAULT_SUMMARY_MAX_TOKENS > rep.DEFAULT_SUMMARY_CHUNK_MAX_TOKENS
|
||||
|
||||
|
||||
class TestCollectXwlb:
|
||||
"""_collect_xwlb 取数逻辑:应查询日报前一日(已播出的联播),并跳过内容提要。"""
|
||||
|
||||
|
||||
Reference in New Issue
Block a user