feat: Token Plan 迁移与 .env 热加载,并修复日报 AI 摘要为空
Token Plan 迁移 / 配置热加载:
- configs/llm_models.yaml: 各场景切到 Token Plan(deepseek-v4.1-flash / qwen3.6-flash)
- 新增 configs/runtime_env.py: .env 按 (mtime_ns, size) 热加载并同步 os.environ,
统一 env_get 取值;llm / embedding / vectorstore / mcp / pipeline 改用 env_get
- configs/loader.py / scripts/run_scheduler.py 等配套调整
- 新增 tests/test_hot_reload.py
日报 AI 摘要为空修复(2026-09-25):
- 根因: 推理模型的 reasoning token 与正文共用 max_tokens, 预算 1500 被"思考"
占满 -> text_tokens=0 / finish_reason=length, 摘要静默为空且不重试
- daily_report 场景新增 max_tokens(默认 4000, YAML 保存即热生效);
LLMConfig 支持可选 max_tokens; 分块预算 800 -> 2000
- _llm_call 拆出 _call_once, 正文为空时自动加倍预算重试(上限 16000),
用尽才降级返回空串; 网络异常重试语义不变
- docs/user-guide.md 新增 FAQ; continuation.md 记录本次排查
- 已重跑 2026-09-25 日报(report_id=357)补回 466 字摘要
测试: 相关用例 56 passed(test_hot_reload 12 passed);
ruff 无新增问题; 3 个 crawler 既有失败与本改动无关
This commit is contained in:
+58
-1
@@ -356,7 +356,23 @@ def test_load_llm_config_deepseek_from_env(monkeypatch: pytest.MonkeyPatch) -> N
|
||||
assert "deepseek" in cfg.base_url
|
||||
|
||||
|
||||
def test_load_llm_config_qwen_from_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def test_load_llm_config_qwen_from_env(
|
||||
monkeypatch: pytest.MonkeyPatch, tmp_path
|
||||
) -> None:
|
||||
"""qwen 的 key 兜底链:QWEN_API_KEY 缺失时回退 DASHSCOPE_API_KEY。
|
||||
|
||||
真实 .env 里带有 QWEN_API_KEY,会盖过 DASHSCOPE_API_KEY,所以本用例先把
|
||||
热加载指向一个空 .env,测完再切回真实 .env。
|
||||
"""
|
||||
from configs import runtime_env
|
||||
|
||||
empty = tmp_path / ".env"
|
||||
empty.write_text("", encoding="utf-8")
|
||||
monkeypatch.setenv(runtime_env.ENV_FILE_OVERRIDE, str(empty))
|
||||
runtime_env.reset_cache()
|
||||
monkeypatch.delenv("QWEN_API_KEY", raising=False)
|
||||
monkeypatch.delenv("QWEN_BASE_URL", raising=False)
|
||||
|
||||
monkeypatch.setenv("LLM_PROVIDER", "qwen")
|
||||
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test-qwen")
|
||||
monkeypatch.setenv("QWEN_MODEL", "qwen-plus")
|
||||
@@ -367,6 +383,11 @@ def test_load_llm_config_qwen_from_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
assert cfg.model == "qwen-plus"
|
||||
assert "dashscope" in cfg.base_url or "aliyuncs" in cfg.base_url
|
||||
|
||||
# 切回真实 .env,避免影响后续用例(DB / Qdrant 等直接读 os.environ 的测试)
|
||||
monkeypatch.delenv(runtime_env.ENV_FILE_OVERRIDE, raising=False)
|
||||
runtime_env.reset_cache()
|
||||
runtime_env.ensure_env_loaded(force=True)
|
||||
|
||||
|
||||
def test_load_llm_config_unknown_provider_raises(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
with pytest.raises(ValueError):
|
||||
@@ -498,3 +519,39 @@ def test_load_llm_config_scene_max_attempts_zero() -> None:
|
||||
assert _pick_int({"max_attempts": 0}, "max_attempts", 3) == 0
|
||||
assert _pick_int({"max_attempts": ""}, "max_attempts", 3) == 3
|
||||
assert _pick_int({}, "max_attempts", 3) == 3
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# max_tokens 场景配置(推理模型 reasoning 与正文共用预算,回归 9-25 无 AI 摘要)
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def test_load_llm_config_scene_max_tokens(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""scenes.<scene>.max_tokens 应写入 LLMConfig;未配置则为 None(调用方走默认)。"""
|
||||
monkeypatch.setenv("QWEN_API_KEY", "sk-qwen")
|
||||
_patch_scene(monkeypatch, {"provider": "qwen", "model": "deepseek-v4.1-flash",
|
||||
"max_tokens": 4000})
|
||||
cfg = load_llm_config(scene="daily_report")
|
||||
assert cfg.max_tokens == 4000
|
||||
|
||||
_patch_scene(monkeypatch, {"provider": "qwen", "model": "deepseek-v4.1-flash"})
|
||||
assert load_llm_config(scene="daily_report").max_tokens is None
|
||||
|
||||
|
||||
def test_pick_optional_int() -> None:
|
||||
"""_pick_optional_int:未配置/空值/非法 → None;数字字符串可解析。"""
|
||||
from llm.client import _pick_optional_int
|
||||
|
||||
assert _pick_optional_int({"max_tokens": 4000}, "max_tokens") == 4000
|
||||
assert _pick_optional_int({"max_tokens": "4000"}, "max_tokens") == 4000
|
||||
assert _pick_optional_int({"max_tokens": ""}, "max_tokens") is None
|
||||
assert _pick_optional_int({"max_tokens": "abc"}, "max_tokens") is None
|
||||
assert _pick_optional_int({}, "max_tokens") is None
|
||||
|
||||
|
||||
def test_real_yaml_daily_report_max_tokens_has_reasoning_headroom() -> None:
|
||||
"""真实配置的日报预算必须大于旧的 1500(为 reasoning token 留余量)。"""
|
||||
from configs.loader import load_scene_config
|
||||
|
||||
scene = load_scene_config("daily_report")
|
||||
max_tokens = int(scene.get("max_tokens") or 0)
|
||||
assert max_tokens >= 4000, f"daily_report.max_tokens 过小: {max_tokens}"
|
||||
|
||||
Reference in New Issue
Block a user