feat: AI 模型按场景独立配置(llm_scenes)+ 去重多来源合并
- system.yaml 新增 llm_scenes(translation / daily_report,含用途/方法/模型要求说明) - load_llm_config(scene=...) 场景覆盖;日报摘要 temperature 0.3 硬编码 → 配置 - M3 去重:唯一篇记录 source_ids(跨源重复合并,首个来源为 source_id) - source_ids 经翻译透传至 events,日报事件 source 多来源拼接展示(≤3 个) - 已部署 pi5:merge 实证 10 源合并;日报 report_id=224 正常入库
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
"""M3 三层去重模块单元测试。"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
@@ -17,6 +18,7 @@ from dedup import (
|
||||
normalize_content,
|
||||
simhash64,
|
||||
)
|
||||
from dedup.pipeline import dedup_source
|
||||
from extractor.models import ProcessedArticle
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
@@ -531,3 +533,100 @@ class TestDedupResult:
|
||||
summary = r.short_summary()
|
||||
assert "[DUP/simhash]" in summary
|
||||
assert "hd=2" in summary
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# dedup_source 跨源来源合并(M9.2:最终显示新闻记录多个来源)
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class TestMergeSources:
|
||||
"""dedup_source 在跨源重复时把来源合并进唯一篇。"""
|
||||
|
||||
DATE_STR = "20260805"
|
||||
|
||||
def _write_processed(
|
||||
self,
|
||||
base: Path,
|
||||
source_id: str,
|
||||
article: ProcessedArticle,
|
||||
) -> None:
|
||||
"""写入 data/processed/{source_id}/{date}/{url_hash}.json。"""
|
||||
d = base / "data" / "processed" / source_id / self.DATE_STR
|
||||
d.mkdir(parents=True, exist_ok=True)
|
||||
(d / f"{article.url_hash}.json").write_text(
|
||||
article.model_dump_json(indent=2, ensure_ascii=False), encoding="utf-8"
|
||||
)
|
||||
|
||||
def test_cross_source_merge(self, tmp_path, monkeypatch):
|
||||
"""同内容两源报道 → 唯一篇 source_ids 记录两个来源,主来源不变。"""
|
||||
monkeypatch.chdir(tmp_path) # 隔离 data/ 相对路径与默认指纹库
|
||||
|
||||
content = ("The Federal Reserve kept interest rates unchanged on Wednesday. "
|
||||
"Markets rallied in response.")
|
||||
art_a = _make_article(
|
||||
url="https://www.reuters.com/business/1", url_hash="aaaa111111111111",
|
||||
source_id="reuters", content=content,
|
||||
)
|
||||
art_b = _make_article(
|
||||
url="https://www.cnbc.com/2026/1", url_hash="bbbb222222222222",
|
||||
source_id="cnbc", source_name="CNBC", content=content,
|
||||
)
|
||||
self._write_processed(tmp_path, "reuters", art_a)
|
||||
self._write_processed(tmp_path, "cnbc", art_b)
|
||||
|
||||
with Deduper() as deduper:
|
||||
dedup_source("reuters", deduper, self.DATE_STR)
|
||||
dedup_source("cnbc", deduper, self.DATE_STR)
|
||||
|
||||
# 唯一篇 = reuters(先处理),跨源重复后 source_ids 合并
|
||||
uniq = tmp_path / "data" / "deduped" / self.DATE_STR / "uniques" / "aaaa111111111111.json"
|
||||
assert uniq.exists()
|
||||
data = json.loads(uniq.read_text(encoding="utf-8"))
|
||||
assert data["source_id"] == "reuters" # 主来源不变
|
||||
assert data["source_ids"] == ["reuters", "cnbc"]
|
||||
assert (tmp_path / "data" / "deduped" / self.DATE_STR / "uniques"
|
||||
/ "bbbb222222222222.json").exists() is False # 重复篇不单独落盘
|
||||
|
||||
def test_unique_initializes_source_ids(self, tmp_path, monkeypatch):
|
||||
"""无重复时唯一篇 source_ids 初始化为 [source_id]。"""
|
||||
monkeypatch.chdir(tmp_path)
|
||||
art = _make_article(url="https://x.com/1", url_hash="cccc333333333333",
|
||||
source_id="ft", source_name="Financial Times")
|
||||
self._write_processed(tmp_path, "ft", art)
|
||||
|
||||
with Deduper() as deduper:
|
||||
dedup_source("ft", deduper, self.DATE_STR)
|
||||
|
||||
uniq = tmp_path / "data" / "deduped" / self.DATE_STR / "uniques" / "cccc333333333333.json"
|
||||
data = json.loads(uniq.read_text(encoding="utf-8"))
|
||||
assert data["source_ids"] == ["ft"]
|
||||
|
||||
def test_merge_idempotent(self, tmp_path, monkeypatch):
|
||||
"""同一来源重复出现多次合并时去重(不产生重复来源)。"""
|
||||
monkeypatch.chdir(tmp_path)
|
||||
content = "Identical content across sources for idempotent test."
|
||||
art_a = _make_article(
|
||||
url="https://www.reuters.com/business/2", url_hash="dddd444444444444",
|
||||
source_id="reuters", content=content,
|
||||
)
|
||||
art_b = _make_article(
|
||||
url="https://www.cnbc.com/2026/2", url_hash="eeee555555555555",
|
||||
source_id="cnbc", source_name="CNBC", content=content,
|
||||
)
|
||||
art_c = _make_article(
|
||||
url="https://www.marketwatch.com/2", url_hash="ffff666666666666",
|
||||
source_id="marketwatch", source_name="MarketWatch", content=content,
|
||||
)
|
||||
self._write_processed(tmp_path, "reuters", art_a)
|
||||
self._write_processed(tmp_path, "cnbc", art_b)
|
||||
self._write_processed(tmp_path, "marketwatch", art_c)
|
||||
|
||||
with Deduper() as deduper:
|
||||
dedup_source("reuters", deduper, self.DATE_STR)
|
||||
dedup_source("cnbc", deduper, self.DATE_STR)
|
||||
dedup_source("marketwatch", deduper, self.DATE_STR)
|
||||
|
||||
uniq = tmp_path / "data" / "deduped" / self.DATE_STR / "uniques" / "dddd444444444444.json"
|
||||
data = json.loads(uniq.read_text(encoding="utf-8"))
|
||||
assert data["source_ids"] == ["reuters", "cnbc", "marketwatch"]
|
||||
|
||||
Reference in New Issue
Block a user