feat: AI 模型按场景独立配置(llm_scenes)+ 去重多来源合并
- system.yaml 新增 llm_scenes(translation / daily_report,含用途/方法/模型要求说明) - load_llm_config(scene=...) 场景覆盖;日报摘要 temperature 0.3 硬编码 → 配置 - M3 去重:唯一篇记录 source_ids(跨源重复合并,首个来源为 source_id) - source_ids 经翻译透传至 events,日报事件 source 多来源拼接展示(≤3 个) - 已部署 pi5:merge 实证 10 源合并;日报 report_id=224 正常入库
This commit is contained in:
@@ -103,6 +103,8 @@ def dedup_source(
|
||||
|
||||
if result.is_duplicate:
|
||||
dup_count += 1
|
||||
# 跨源重复:把来源合并进已保留的唯一篇(记录多个来源)
|
||||
_merge_duplicate_source(article, result)
|
||||
logger.debug("[%s] 🔁 %s → L%d: %s",
|
||||
source_id,
|
||||
article.title[:40],
|
||||
@@ -110,6 +112,9 @@ def dedup_source(
|
||||
result.short_summary())
|
||||
else:
|
||||
unique_count += 1
|
||||
# 初始化来源列表(首个来源 = 本篇文章来源)
|
||||
if not article.source_ids:
|
||||
article.source_ids = [article.source_id]
|
||||
# 写入唯一条目
|
||||
out_file = out_dir / f"{article.url_hash}.json"
|
||||
out_file.write_text(
|
||||
@@ -130,6 +135,38 @@ def dedup_source(
|
||||
}
|
||||
|
||||
|
||||
def _merge_duplicate_source(article: ProcessedArticle, result: DedupResult) -> None:
|
||||
"""重复篇:把来源 ID 追加进已保留的唯一篇 JSON(最终显示的新闻记录多个来源)。
|
||||
|
||||
唯一篇文件按 url_hash 定位(跨日期目录搜索,因指纹窗口为 ±30 天);
|
||||
文件不存在(超窗口被清理)时仅记录日志,不阻塞去重流程。
|
||||
"""
|
||||
if not result.matched_url_hash:
|
||||
return
|
||||
candidates = sorted(Path("data/deduped").glob(f"*/uniques/{result.matched_url_hash}.json"))
|
||||
if not candidates:
|
||||
logger.warning(
|
||||
"重复篇唯一文件不存在(可能已超窗口): %s(重复来源 %s 未合并)",
|
||||
result.matched_url_hash, article.source_id,
|
||||
)
|
||||
return
|
||||
target = candidates[0]
|
||||
try:
|
||||
data = json.loads(target.read_text(encoding="utf-8"))
|
||||
# 旧格式文件可能无 source_ids:以主来源 source_id 兜底
|
||||
merged = list(dict.fromkeys(
|
||||
[*(data.get("source_ids") or [data.get("source_id")]), article.source_id]
|
||||
))
|
||||
data["source_ids"] = merged
|
||||
target.write_text(
|
||||
json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8"
|
||||
)
|
||||
logger.debug("来源合并: %s → %s (sources=%s)",
|
||||
article.source_id, result.matched_url_hash, merged)
|
||||
except Exception as e:
|
||||
logger.exception("来源合并失败 %s: %s", target, e)
|
||||
|
||||
|
||||
def _layer_num(result: DedupResult) -> int:
|
||||
"""DedupResult → 命中层编号。"""
|
||||
if result.matched_layer is None:
|
||||
|
||||
Reference in New Issue
Block a user