feat: 大模型使用场景化配置与去重多源记录
- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
@@ -113,10 +113,20 @@ class TestLlmCallRetry:
|
||||
|
||||
return SimpleNamespace(chat=SimpleNamespace(completions=Completions())), n
|
||||
|
||||
@staticmethod
|
||||
def _cfg():
|
||||
from llm.client import LLMConfig
|
||||
|
||||
return LLMConfig(
|
||||
provider="deepseek", model="deepseek-v4-flash",
|
||||
api_key="sk-test", base_url="https://api.deepseek.com",
|
||||
temperature=0.3,
|
||||
)
|
||||
|
||||
def test_success_first_try(self) -> None:
|
||||
from scheduler.reporter import _llm_call
|
||||
client, n = self._fake_client(0)
|
||||
out = _llm_call(client, "deepseek-v4-flash", "p")
|
||||
out = _llm_call(client, self._cfg(), "p")
|
||||
assert out == "今日要点摘要"
|
||||
assert n["count"] == 1
|
||||
|
||||
@@ -125,7 +135,7 @@ class TestLlmCallRetry:
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 3)
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01)
|
||||
client, n = self._fake_client(2) # 前 2 次失败,第 3 次成功
|
||||
out = rep._llm_call(client, "deepseek-v4-flash", "p")
|
||||
out = rep._llm_call(client, self._cfg(), "p")
|
||||
assert out == "今日要点摘要"
|
||||
assert n["count"] == 3
|
||||
|
||||
@@ -135,7 +145,7 @@ class TestLlmCallRetry:
|
||||
monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01)
|
||||
client, n = self._fake_client(99) # 一直失败
|
||||
with pytest.raises(ConnectionError):
|
||||
rep._llm_call(client, "deepseek-v4-flash", "p")
|
||||
rep._llm_call(client, self._cfg(), "p")
|
||||
assert n["count"] == 2 # 重试 2 次后放弃
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user