feat: Token Plan 迁移与 .env 热加载,并修复日报 AI 摘要为空

Token Plan 迁移 / 配置热加载:
- configs/llm_models.yaml: 各场景切到 Token Plan(deepseek-v4.1-flash / qwen3.6-flash)
- 新增 configs/runtime_env.py: .env 按 (mtime_ns, size) 热加载并同步 os.environ,
  统一 env_get 取值;llm / embedding / vectorstore / mcp / pipeline 改用 env_get
- configs/loader.py / scripts/run_scheduler.py 等配套调整
- 新增 tests/test_hot_reload.py

日报 AI 摘要为空修复(2026-09-25):
- 根因: 推理模型的 reasoning token 与正文共用 max_tokens, 预算 1500 被"思考"
  占满 -> text_tokens=0 / finish_reason=length, 摘要静默为空且不重试
- daily_report 场景新增 max_tokens(默认 4000, YAML 保存即热生效);
  LLMConfig 支持可选 max_tokens; 分块预算 800 -> 2000
- _llm_call 拆出 _call_once, 正文为空时自动加倍预算重试(上限 16000),
  用尽才降级返回空串; 网络异常重试语义不变
- docs/user-guide.md 新增 FAQ; continuation.md 记录本次排查
- 已重跑 2026-09-25 日报(report_id=357)补回 466 字摘要

测试: 相关用例 56 passed(test_hot_reload 12 passed);
      ruff 无新增问题; 3 个 crawler 既有失败与本改动无关
This commit is contained in:
2026-09-25 11:13:37 +08:00
parent 2eaea2ee81
commit ff911cf6f7
19 changed files with 1024 additions and 200 deletions
+217
View File
@@ -0,0 +1,217 @@
"""配置热加载测试:改 .env / llm_models.yaml 后无需重启即生效。
覆盖三层:
- configs/loader.py YAML 场景((mtime, size) 失效缓存)
- configs/runtime_env .env(热更新 / 删除语义 / 外部显式覆盖优先)
- scripts/run_scheduler 定时任务热同步(改 SCHEDULE_TIMES 等)
"""
from __future__ import annotations
import os
from pathlib import Path
import pytest
from configs import loader, runtime_env
@pytest.fixture(autouse=True)
def _clean_caches() -> None:
"""每个用例前后都清干净,避免托管的环境变量污染其它测试。"""
loader.clear_cache()
runtime_env.reset_cache()
yield
loader.clear_cache()
runtime_env.reset_cache()
def _use_env_file(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, content: str) -> Path:
"""把 runtime_env 指向临时 .env,返回其路径。"""
env = tmp_path / ".env"
env.write_text(content, encoding="utf-8")
monkeypatch.setenv(runtime_env.ENV_FILE_OVERRIDE, str(env))
runtime_env.reset_cache()
return env
# --------------------------------------------------------------------------- #
# YAML 场景热加载
# --------------------------------------------------------------------------- #
def test_yaml_scene_hot_reload(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""改 llm_models.yaml 后,下一次 load_scene_config 立即读到新值。"""
cfg = tmp_path / "llm_models.yaml"
cfg.write_text(
"scenes:\n daily_report:\n provider: qwen\n model: qwen3.6-flash\n",
encoding="utf-8",
)
monkeypatch.setenv(loader.MODELS_CONFIG_OVERRIDE, str(cfg))
assert loader.load_scene_config("daily_report")["model"] == "qwen3.6-flash"
# 内容长度不同,保证 (mtime, size) 签名一定变化
cfg.write_text(
"scenes:\n daily_report:\n provider: qwen\n model: deepseek-v4.1-flash\n",
encoding="utf-8",
)
# 不重启、不 clear_cache 也应生效
assert loader.load_scene_config("daily_report")["model"] == "deepseek-v4.1-flash"
def test_yaml_scene_removed_falls_back(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""场景被删掉后返回空 dict(由调用方回退 .env)。"""
cfg = tmp_path / "llm_models.yaml"
cfg.write_text("scenes:\n stock_report:\n model: x\n", encoding="utf-8")
monkeypatch.setenv(loader.MODELS_CONFIG_OVERRIDE, str(cfg))
assert loader.load_scene_config("stock_report")["model"] == "x"
cfg.write_text("scenes: {}\n", encoding="utf-8")
assert loader.load_scene_config("stock_report") == {}
def test_yaml_missing_file_returns_empty(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv(loader.MODELS_CONFIG_OVERRIDE, str(tmp_path / "nope.yaml"))
assert loader.load_scene_config("daily_report") == {}
assert loader.load_defaults() == {}
# --------------------------------------------------------------------------- #
# .env 热加载
# --------------------------------------------------------------------------- #
def test_env_hot_reload(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""改 .env 后 env_get 立即读到新值。"""
env = _use_env_file(tmp_path, monkeypatch, "HOT_RELOAD_KEY=one\n")
monkeypatch.delenv("HOT_RELOAD_KEY", raising=False)
assert runtime_env.env_get("HOT_RELOAD_KEY") == "one"
env.write_text("HOT_RELOAD_KEY=two-longer-value\n", encoding="utf-8")
assert runtime_env.env_get("HOT_RELOAD_KEY") == "two-longer-value"
def test_env_removed_key_is_unset(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""从 .env 删掉的托管键要同步从 os.environ 移除。"""
env = _use_env_file(tmp_path, monkeypatch, "HOT_REMOVE_KEY=abc\n")
monkeypatch.delenv("HOT_REMOVE_KEY", raising=False)
assert runtime_env.env_get("HOT_REMOVE_KEY") == "abc"
assert os.environ["HOT_REMOVE_KEY"] == "abc"
env.write_text("OTHER_KEY=1\n", encoding="utf-8")
assert runtime_env.env_get("HOT_REMOVE_KEY") is None
assert "HOT_REMOVE_KEY" not in os.environ
def test_env_empty_value_is_unset(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
_use_env_file(tmp_path, monkeypatch, "HOT_EMPTY_KEY=\n")
monkeypatch.delenv("HOT_EMPTY_KEY", raising=False)
assert runtime_env.env_get("HOT_EMPTY_KEY", "fallback") == "fallback"
def test_explicit_process_env_wins(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""进程环境显式设置且与文件不同 → 不被 .env 覆盖(一次性覆盖语义)。"""
env = _use_env_file(tmp_path, monkeypatch, "HOT_OVERRIDE_KEY=from-file\n")
monkeypatch.setenv("HOT_OVERRIDE_KEY", "from-shell")
assert runtime_env.env_get("HOT_OVERRIDE_KEY") == "from-shell"
env.write_text("HOT_OVERRIDE_KEY=from-file-changed\n", encoding="utf-8")
assert runtime_env.env_get("HOT_OVERRIDE_KEY") == "from-shell"
def test_llm_config_follows_env_hot_reload(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""端到端:改 .env 里的 QWEN_MODEL,load_llm_config 立即用新模型。"""
from llm.client import load_llm_config
contents = (
"LLM_PROVIDER=qwen\n"
"QWEN_API_KEY=sk-test\n"
"QWEN_BASE_URL=https://token-plan.example/compatible-mode/v1\n"
"QWEN_MODEL=qwen3.6-flash\n"
)
env = _use_env_file(tmp_path, monkeypatch, contents)
for key in ("LLM_PROVIDER", "QWEN_API_KEY", "QWEN_BASE_URL", "QWEN_MODEL"):
monkeypatch.delenv(key, raising=False)
cfg = load_llm_config(scene="")
assert cfg.provider == "qwen"
assert cfg.model == "qwen3.6-flash"
env.write_text(contents.replace("qwen3.6-flash", "deepseek-v4.1-flash"), encoding="utf-8")
assert load_llm_config(scene="").model == "deepseek-v4.1-flash"
# --------------------------------------------------------------------------- #
# 调度任务热同步
# --------------------------------------------------------------------------- #
def test_desired_jobs_tracks_env(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""SCHEDULE_TIMES / CNINFO_SCHEDULE_TIME / STOCK_REPORT_TIME 改完立即反映。"""
from scripts.run_scheduler import _desired_jobs
_use_env_file(
tmp_path,
monkeypatch,
"SCHEDULE_TIMES=09:15\nCNINFO_SCHEDULE_TIME=05:45\nSTOCK_REPORT_TIME=\n",
)
jobs = _desired_jobs()
assert set(jobs) == {"pipeline_0915", "pipeline_cninfo"}
# 当天唯一次数 → 附带日报步骤
assert jobs["pipeline_0915"]["steps"][-1] == "report"
assert jobs["pipeline_cninfo"]["hour"] == 5
assert jobs["pipeline_cninfo"]["minute"] == 45
def test_sync_jobs_adds_and_removes(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""_sync_jobs 只按需增删:加时间点、启停个股日报、幂等。"""
from apscheduler.schedulers.background import BackgroundScheduler
import scripts.run_scheduler as rs
env = _use_env_file(
tmp_path,
monkeypatch,
"SCHEDULE_TIMES=09:15\nCNINFO_SCHEDULE_TIME=05:45\nSTOCK_REPORT_TIME=\n",
)
rs._jobs_sig = None
sched = BackgroundScheduler()
assert rs._sync_jobs(sched) is True
assert {j.id for j in sched.get_jobs()} == {"pipeline_0915", "pipeline_cninfo"}
# 新增一个时间点 + 启用个股日报
env.write_text(
"SCHEDULE_TIMES=09:15,16:40\nCNINFO_SCHEDULE_TIME=05:45\nSTOCK_REPORT_TIME=07:30\n",
encoding="utf-8",
)
assert rs._sync_jobs(sched) is True
ids = {j.id for j in sched.get_jobs()}
assert ids == {"pipeline_0915", "pipeline_1640", "pipeline_cninfo", "stock_report"}
# 配置未变 → 不重复变更
assert rs._sync_jobs(sched) is False
def test_parse_hhmm_falls_back_on_bad_input() -> None:
from scripts.run_scheduler import _parse_hhmm
assert _parse_hhmm("06:30", (0, 0)) == (6, 30)
assert _parse_hhmm("99:99", (6, 30)) == (6, 30)
assert _parse_hhmm("garbage", (7, 0)) == (7, 0)
# --------------------------------------------------------------------------- #
# 其它模块复用热读取
# --------------------------------------------------------------------------- #
def test_reporter_days_back_is_lazy(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""日报的 STOCK_REPORT_DAYS 走懒读取,不再冻结在模块常量里。"""
from scheduler import reporter
_use_env_file(tmp_path, monkeypatch, "STOCK_REPORT_DAYS=15\n")
monkeypatch.delenv("STOCK_REPORT_DAYS", raising=False)
assert reporter._cninfo_days_back() == 15
+58 -1
View File
@@ -356,7 +356,23 @@ def test_load_llm_config_deepseek_from_env(monkeypatch: pytest.MonkeyPatch) -> N
assert "deepseek" in cfg.base_url
def test_load_llm_config_qwen_from_env(monkeypatch: pytest.MonkeyPatch) -> None:
def test_load_llm_config_qwen_from_env(
monkeypatch: pytest.MonkeyPatch, tmp_path
) -> None:
"""qwen 的 key 兜底链:QWEN_API_KEY 缺失时回退 DASHSCOPE_API_KEY。
真实 .env 里带有 QWEN_API_KEY,会盖过 DASHSCOPE_API_KEY,所以本用例先把
热加载指向一个空 .env,测完再切回真实 .env。
"""
from configs import runtime_env
empty = tmp_path / ".env"
empty.write_text("", encoding="utf-8")
monkeypatch.setenv(runtime_env.ENV_FILE_OVERRIDE, str(empty))
runtime_env.reset_cache()
monkeypatch.delenv("QWEN_API_KEY", raising=False)
monkeypatch.delenv("QWEN_BASE_URL", raising=False)
monkeypatch.setenv("LLM_PROVIDER", "qwen")
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test-qwen")
monkeypatch.setenv("QWEN_MODEL", "qwen-plus")
@@ -367,6 +383,11 @@ def test_load_llm_config_qwen_from_env(monkeypatch: pytest.MonkeyPatch) -> None:
assert cfg.model == "qwen-plus"
assert "dashscope" in cfg.base_url or "aliyuncs" in cfg.base_url
# 切回真实 .env,避免影响后续用例(DB / Qdrant 等直接读 os.environ 的测试)
monkeypatch.delenv(runtime_env.ENV_FILE_OVERRIDE, raising=False)
runtime_env.reset_cache()
runtime_env.ensure_env_loaded(force=True)
def test_load_llm_config_unknown_provider_raises(monkeypatch: pytest.MonkeyPatch) -> None:
with pytest.raises(ValueError):
@@ -498,3 +519,39 @@ def test_load_llm_config_scene_max_attempts_zero() -> None:
assert _pick_int({"max_attempts": 0}, "max_attempts", 3) == 0
assert _pick_int({"max_attempts": ""}, "max_attempts", 3) == 3
assert _pick_int({}, "max_attempts", 3) == 3
# --------------------------------------------------------------------------- #
# max_tokens 场景配置(推理模型 reasoning 与正文共用预算,回归 9-25 无 AI 摘要)
# --------------------------------------------------------------------------- #
def test_load_llm_config_scene_max_tokens(monkeypatch: pytest.MonkeyPatch) -> None:
"""scenes.<scene>.max_tokens 应写入 LLMConfig;未配置则为 None(调用方走默认)。"""
monkeypatch.setenv("QWEN_API_KEY", "sk-qwen")
_patch_scene(monkeypatch, {"provider": "qwen", "model": "deepseek-v4.1-flash",
"max_tokens": 4000})
cfg = load_llm_config(scene="daily_report")
assert cfg.max_tokens == 4000
_patch_scene(monkeypatch, {"provider": "qwen", "model": "deepseek-v4.1-flash"})
assert load_llm_config(scene="daily_report").max_tokens is None
def test_pick_optional_int() -> None:
"""_pick_optional_int:未配置/空值/非法 → None;数字字符串可解析。"""
from llm.client import _pick_optional_int
assert _pick_optional_int({"max_tokens": 4000}, "max_tokens") == 4000
assert _pick_optional_int({"max_tokens": "4000"}, "max_tokens") == 4000
assert _pick_optional_int({"max_tokens": ""}, "max_tokens") is None
assert _pick_optional_int({"max_tokens": "abc"}, "max_tokens") is None
assert _pick_optional_int({}, "max_tokens") is None
def test_real_yaml_daily_report_max_tokens_has_reasoning_headroom() -> None:
"""真实配置的日报预算必须大于旧的 1500(为 reasoning token 留余量)。"""
from configs.loader import load_scene_config
scene = load_scene_config("daily_report")
max_tokens = int(scene.get("max_tokens") or 0)
assert max_tokens >= 4000, f"daily_report.max_tokens 过小: {max_tokens}"
+82 -4
View File
@@ -132,8 +132,8 @@ class TestLlmCallRetry:
def test_retry_then_success(self, monkeypatch) -> None:
import scheduler.reporter as rep
monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 3)
monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01)
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 3)
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
client, n = self._fake_client(2) # 前 2 次失败,第 3 次成功
out = rep._llm_call(client, self._cfg(), "p")
assert out == "今日要点摘要"
@@ -141,8 +141,8 @@ class TestLlmCallRetry:
def test_exhausts_retries_raises(self, monkeypatch) -> None:
import scheduler.reporter as rep
monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 2)
monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01)
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 2)
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
client, n = self._fake_client(99) # 一直失败
with pytest.raises(ConnectionError):
rep._llm_call(client, self._cfg(), "p")
@@ -171,6 +171,84 @@ class TestLlmCallRetry:
assert n["count"] == 3 # 2 块 + 1 次合并
class TestReasoningBudgetEscalation:
"""推理模型占满 max_tokens 导致正文为空时的预算升级(回归 9-25 无 AI 摘要)。"""
@staticmethod
def _empty_then_ok_client(empty_times: int):
"""前 empty_times 次返回空正文 + finish_reason=length,之后返回正常摘要。"""
from types import SimpleNamespace
seen: list[int] = []
class Completions:
def create(self, **kwargs):
seen.append(kwargs.get("max_tokens"))
if len(seen) <= empty_times:
return SimpleNamespace(
choices=[SimpleNamespace(
message=SimpleNamespace(content=""),
finish_reason="length",
)]
)
return SimpleNamespace(
choices=[SimpleNamespace(
message=SimpleNamespace(content="恢复后的摘要"),
finish_reason="stop",
)]
)
return SimpleNamespace(chat=SimpleNamespace(completions=Completions())), seen
@staticmethod
def _cfg(max_tokens: int | None = None):
from llm.client import LLMConfig
return LLMConfig(
provider="qwen", model="deepseek-v4.1-flash",
api_key="sk-test", base_url="https://example.invalid/v1",
temperature=0.3, max_tokens=max_tokens,
)
def test_empty_content_escalates_and_recovers(self, monkeypatch) -> None:
"""正文为空时自动加倍预算并最终拿到摘要(不再静默返回空)。"""
import scheduler.reporter as rep
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 3)
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
client, seen = self._empty_then_ok_client(1)
out = rep._llm_call(client, self._cfg(4000), "p")
assert out == "恢复后的摘要"
assert seen == [4000, 8000] # 首次失败后预算翻倍
def test_scene_max_tokens_wins_over_default(self, monkeypatch) -> None:
"""场景 max_tokens 生效;未配置时回退代码默认 4000。"""
import scheduler.reporter as rep
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 1)
client, seen = self._empty_then_ok_client(0)
rep._llm_call(client, self._cfg(6000), "p")
assert seen == [6000]
client, seen = self._empty_then_ok_client(0)
rep._llm_call(client, self._cfg(), "p")
assert seen == [rep.DEFAULT_SUMMARY_MAX_TOKENS]
def test_all_empty_returns_blank_without_raising(self, monkeypatch) -> None:
"""预算升级用尽仍为空时降级返回空串(日报仍可入库,不抛异常)。"""
import scheduler.reporter as rep
monkeypatch.setattr(rep, "_llm_retry_times", lambda: 3)
monkeypatch.setattr(rep, "_llm_retry_backoff_sec", lambda: 0.01)
client, seen = self._empty_then_ok_client(99)
out = rep._llm_call(client, self._cfg(4000), "p")
assert out == ""
# 4000 → 8000 → 16000(受 MAX_SUMMARY_MAX_TOKENS 上限约束)
assert seen == [4000, 8000, 16000]
def test_budget_never_exceeds_cap(self) -> None:
"""升级预算不超过 MAX_SUMMARY_MAX_TOKENS,避免无限放大。"""
import scheduler.reporter as rep
assert rep.MAX_SUMMARY_MAX_TOKENS == 16000
assert rep.DEFAULT_SUMMARY_MAX_TOKENS > rep.DEFAULT_SUMMARY_CHUNK_MAX_TOKENS
class TestCollectXwlb:
"""_collect_xwlb 取数逻辑:应查询日报前一日(已播出的联播),并跳过内容提要。"""