feat: AI 模型按场景独立配置(llm_scenes)+ 去重多来源合并
- system.yaml 新增 llm_scenes(translation / daily_report,含用途/方法/模型要求说明) - load_llm_config(scene=...) 场景覆盖;日报摘要 temperature 0.3 硬编码 → 配置 - M3 去重:唯一篇记录 source_ids(跨源重复合并,首个来源为 source_id) - source_ids 经翻译透传至 events,日报事件 source 多来源拼接展示(≤3 个) - 已部署 pi5:merge 实证 10 源合并;日报 report_id=224 正常入库
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
"""M3 三层去重模块单元测试。"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
@@ -17,6 +18,7 @@ from dedup import (
|
||||
normalize_content,
|
||||
simhash64,
|
||||
)
|
||||
from dedup.pipeline import dedup_source
|
||||
from extractor.models import ProcessedArticle
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
@@ -531,3 +533,100 @@ class TestDedupResult:
|
||||
summary = r.short_summary()
|
||||
assert "[DUP/simhash]" in summary
|
||||
assert "hd=2" in summary
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# dedup_source 跨源来源合并(M9.2:最终显示新闻记录多个来源)
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class TestMergeSources:
|
||||
"""dedup_source 在跨源重复时把来源合并进唯一篇。"""
|
||||
|
||||
DATE_STR = "20260805"
|
||||
|
||||
def _write_processed(
|
||||
self,
|
||||
base: Path,
|
||||
source_id: str,
|
||||
article: ProcessedArticle,
|
||||
) -> None:
|
||||
"""写入 data/processed/{source_id}/{date}/{url_hash}.json。"""
|
||||
d = base / "data" / "processed" / source_id / self.DATE_STR
|
||||
d.mkdir(parents=True, exist_ok=True)
|
||||
(d / f"{article.url_hash}.json").write_text(
|
||||
article.model_dump_json(indent=2, ensure_ascii=False), encoding="utf-8"
|
||||
)
|
||||
|
||||
def test_cross_source_merge(self, tmp_path, monkeypatch):
|
||||
"""同内容两源报道 → 唯一篇 source_ids 记录两个来源,主来源不变。"""
|
||||
monkeypatch.chdir(tmp_path) # 隔离 data/ 相对路径与默认指纹库
|
||||
|
||||
content = ("The Federal Reserve kept interest rates unchanged on Wednesday. "
|
||||
"Markets rallied in response.")
|
||||
art_a = _make_article(
|
||||
url="https://www.reuters.com/business/1", url_hash="aaaa111111111111",
|
||||
source_id="reuters", content=content,
|
||||
)
|
||||
art_b = _make_article(
|
||||
url="https://www.cnbc.com/2026/1", url_hash="bbbb222222222222",
|
||||
source_id="cnbc", source_name="CNBC", content=content,
|
||||
)
|
||||
self._write_processed(tmp_path, "reuters", art_a)
|
||||
self._write_processed(tmp_path, "cnbc", art_b)
|
||||
|
||||
with Deduper() as deduper:
|
||||
dedup_source("reuters", deduper, self.DATE_STR)
|
||||
dedup_source("cnbc", deduper, self.DATE_STR)
|
||||
|
||||
# 唯一篇 = reuters(先处理),跨源重复后 source_ids 合并
|
||||
uniq = tmp_path / "data" / "deduped" / self.DATE_STR / "uniques" / "aaaa111111111111.json"
|
||||
assert uniq.exists()
|
||||
data = json.loads(uniq.read_text(encoding="utf-8"))
|
||||
assert data["source_id"] == "reuters" # 主来源不变
|
||||
assert data["source_ids"] == ["reuters", "cnbc"]
|
||||
assert (tmp_path / "data" / "deduped" / self.DATE_STR / "uniques"
|
||||
/ "bbbb222222222222.json").exists() is False # 重复篇不单独落盘
|
||||
|
||||
def test_unique_initializes_source_ids(self, tmp_path, monkeypatch):
|
||||
"""无重复时唯一篇 source_ids 初始化为 [source_id]。"""
|
||||
monkeypatch.chdir(tmp_path)
|
||||
art = _make_article(url="https://x.com/1", url_hash="cccc333333333333",
|
||||
source_id="ft", source_name="Financial Times")
|
||||
self._write_processed(tmp_path, "ft", art)
|
||||
|
||||
with Deduper() as deduper:
|
||||
dedup_source("ft", deduper, self.DATE_STR)
|
||||
|
||||
uniq = tmp_path / "data" / "deduped" / self.DATE_STR / "uniques" / "cccc333333333333.json"
|
||||
data = json.loads(uniq.read_text(encoding="utf-8"))
|
||||
assert data["source_ids"] == ["ft"]
|
||||
|
||||
def test_merge_idempotent(self, tmp_path, monkeypatch):
|
||||
"""同一来源重复出现多次合并时去重(不产生重复来源)。"""
|
||||
monkeypatch.chdir(tmp_path)
|
||||
content = "Identical content across sources for idempotent test."
|
||||
art_a = _make_article(
|
||||
url="https://www.reuters.com/business/2", url_hash="dddd444444444444",
|
||||
source_id="reuters", content=content,
|
||||
)
|
||||
art_b = _make_article(
|
||||
url="https://www.cnbc.com/2026/2", url_hash="eeee555555555555",
|
||||
source_id="cnbc", source_name="CNBC", content=content,
|
||||
)
|
||||
art_c = _make_article(
|
||||
url="https://www.marketwatch.com/2", url_hash="ffff666666666666",
|
||||
source_id="marketwatch", source_name="MarketWatch", content=content,
|
||||
)
|
||||
self._write_processed(tmp_path, "reuters", art_a)
|
||||
self._write_processed(tmp_path, "cnbc", art_b)
|
||||
self._write_processed(tmp_path, "marketwatch", art_c)
|
||||
|
||||
with Deduper() as deduper:
|
||||
dedup_source("reuters", deduper, self.DATE_STR)
|
||||
dedup_source("cnbc", deduper, self.DATE_STR)
|
||||
dedup_source("marketwatch", deduper, self.DATE_STR)
|
||||
|
||||
uniq = tmp_path / "data" / "deduped" / self.DATE_STR / "uniques" / "dddd444444444444.json"
|
||||
data = json.loads(uniq.read_text(encoding="utf-8"))
|
||||
assert data["source_ids"] == ["reuters", "cnbc", "marketwatch"]
|
||||
|
||||
@@ -463,6 +463,29 @@ class TestLoadLLMConfig:
|
||||
config = load_llm_config(provider="deepseek")
|
||||
assert config.max_attempts == 3
|
||||
|
||||
def test_scene_translation(self, monkeypatch):
|
||||
"""translation 场景覆盖 llm 默认段(模型/温度)。"""
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test-key")
|
||||
config = load_llm_config(provider="deepseek", scene="translation")
|
||||
assert config.model == "deepseek-v4-flash"
|
||||
assert config.temperature == 0.1
|
||||
assert config.max_attempts == 3
|
||||
|
||||
def test_scene_daily_report(self, monkeypatch):
|
||||
"""daily_report 场景独立配置(温度 0.3 / max_tokens 1500)。"""
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test-key")
|
||||
config = load_llm_config(scene="daily_report")
|
||||
assert config.provider == "deepseek"
|
||||
assert config.model == "deepseek-v4-flash"
|
||||
assert config.temperature == 0.3
|
||||
assert config.max_tokens == 1500
|
||||
|
||||
def test_scene_unknown_falls_back_to_default(self, monkeypatch):
|
||||
"""未定义的场景名回退 llm 默认段,不报错。"""
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test-key")
|
||||
config = load_llm_config(provider="deepseek", scene="not_exists")
|
||||
assert config.temperature == 0.1
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# translate_and_extract(mock LLM)
|
||||
|
||||
@@ -177,3 +177,25 @@ class TestBuildReportData:
|
||||
Counter(), "")
|
||||
# "?" 不写入 DB,留 None
|
||||
assert r.events[0].sentiment is None
|
||||
|
||||
def test_multi_source_label(self) -> None:
|
||||
"""去重合并后的多来源 → source 拼接展示(Reuters, CNBC)。"""
|
||||
now = datetime(2026, 8, 4, 8, 0, 0)
|
||||
ev = _fake_high_event("多来源事件", 5, source_id="reuters",
|
||||
url="https://reuters.com/news/9")
|
||||
# 模拟 M3 去重合并:source_ids 含两个来源
|
||||
ev["article"]["source_ids"] = ["reuters", "cnbc"]
|
||||
r = _build_report_data(now, {}, [ev], Counter(), Counter(), Counter(),
|
||||
Counter(), "")
|
||||
assert r.events[0].source == "Reuters, CNBC"
|
||||
assert r.events[0].url == "https://reuters.com/news/9"
|
||||
|
||||
def test_single_source_falls_back(self) -> None:
|
||||
"""source_ids 为空/单一时回退单源逻辑(不拼接)。"""
|
||||
now = datetime(2026, 8, 4, 8, 0, 0)
|
||||
ev = _fake_high_event("单来源事件", 4, source_id="investinglive",
|
||||
url="https://investinglive.com/news/3")
|
||||
ev["article"]["source_ids"] = ["investinglive"]
|
||||
r = _build_report_data(now, {}, [ev], Counter(), Counter(), Counter(),
|
||||
Counter(), "")
|
||||
assert r.events[0].source == "InvestingLive"
|
||||
|
||||
Reference in New Issue
Block a user