feat: AI 模型按场景独立配置(llm_scenes)+ 去重多来源合并

- system.yaml 新增 llm_scenes(translation / daily_report,含用途/方法/模型要求说明)
- load_llm_config(scene=...) 场景覆盖;日报摘要 temperature 0.3 硬编码 → 配置
- M3 去重:唯一篇记录 source_ids(跨源重复合并,首个来源为 source_id)
- source_ids 经翻译透传至 events,日报事件 source 多来源拼接展示(≤3 个)
- 已部署 pi5:merge 实证 10 源合并;日报 report_id=224 正常入库
This commit is contained in:
2026-08-12 09:56:55 +08:00
parent 9aa44e610d
commit c72a5ed13a
13 changed files with 331 additions and 20 deletions
+22 -4
View File
@@ -96,6 +96,23 @@ def _url_source_label(url: str, source_id: str = "") -> str:
return domain_map.get(domain, domain)
def _article_source_label(article: dict) -> str | None:
"""文章来源展示:去重合并后多来源时拼接展示名,否则回退单源逻辑。
多来源(ProcessedArticle.source_ids 长度 > 1)时展示如 "Reuters, CNBC",
最多取前 3 个来源,截断至 64 字符(news_event.source 为 VARCHAR(64))。
"""
src_ids = list(dict.fromkeys(
s for s in (article.get("source_ids") or []) if s and s != "?"
))
if len(src_ids) > 1:
names = list(dict.fromkeys(
(_source_name(s) or s) for s in src_ids[:3]
))
return ", ".join(names)[:64]
return _url_source_label(article.get("url"), article.get("source_id", ""))
# 日报覆盖时间窗口(小时)
_REPORT_WINDOW_HOURS = 25
@@ -299,11 +316,11 @@ def _call_llm_simple(
"""
import time as _time
# 复用客户端(同 provider/model 只创建一次)
# 复用客户端(同 provider/model 只创建一次;日报摘要场景见 system.yaml llm_scenes.daily_report)
cache_key = "default"
if cache_key not in _llm_client_cache:
from llm.client import load_llm_config, make_sync_client
_llm_client_cache["config"] = load_llm_config()
_llm_client_cache["config"] = load_llm_config(scene="daily_report")
_llm_client_cache[cache_key] = make_sync_client(_llm_client_cache["config"])
config = _llm_client_cache["config"]
@@ -321,7 +338,7 @@ def _call_llm_simple(
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_prompt},
],
temperature=0.3,
temperature=config.temperature,
max_tokens=max_tokens,
)
content = (resp.choices[0].message.content or "").strip()
@@ -581,7 +598,8 @@ def _build_report_data(
article = ev.get("article", {})
title = (article.get("title_zh") or article.get("title") or "").strip()[:512]
url = article.get("url") or None
src = _url_source_label(url, article.get("source_id", ""))
# 去重合并后的多来源(如 "Reuters, CNBC"),否则回退单源展示
src = _article_source_label(article)
# 归一化:"" / "?" 不入库,留 None(DB 仅存 positive/negative/neutral)
sentiment = ev.get("sentiment") or None
if sentiment in ("", "?"):