Files
news/tests/test_run_dedup.py
T
simon 8fa27ad65b fix: 修复 dedup 高重复率返回码被掩蔽的问题 (P1-1)
- run_dedup: 生产模式重复率仅作 WARNING 告警,不影响退出码(执行成功即 0)
- run_dedup: 新增 --strict 验收模式(重复率 > 5% 返回 1),保留 M3 验收门槛
- run_dedup: 新增统计快照 data/deduped/{date}/stats.json(原子写)
- run_dedup: 顺带修复空日场景 sources.json 写入 FileNotFoundError
- pipeline: 移除 dedup rc=1 特判,恢复'非 0 即失败'统一语义
- 新增 tests/test_run_dedup.py 7 个测试;全量 266 passed
2026-08-22 19:50:14 +08:00

174 lines
6.1 KiB
Python

"""run_dedup 脚本测试(生产/验收返回码语义 + stats.json 统计快照)。
覆盖:
- 生产模式(默认):高重复率不影响退出码,返回 0;重复率仅作告警
- 验收模式(--strict):高重复率返回 1(保留 M3 验收门槛 ≤ 5%)
- 低重复率场景两种模式均返回 0
- stats.json 统计快照结构正确(原子写),含重复率与指纹库总量
"""
from __future__ import annotations
import hashlib
import json
import sys
from pathlib import Path
import pytest
def _url_hash(url: str) -> str:
return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16]
def _write_article(proc_dir: Path, url: str, title: str, content: str) -> None:
"""写一个 Article JSON,文件名为 url_hash。"""
h = _url_hash(url)
article = {
"source_id": "testsrc",
"url": url,
"url_hash": h,
"title": title,
"content": content,
"word_count": len(content),
}
(proc_dir / f"{h}.json").write_text(
json.dumps(article, ensure_ascii=False), encoding="utf-8"
)
def _setup_fixture(tmp_path: Path, *, dup_count: int, uniq_count: int) -> Path:
"""构造 processed 夹具:uniq_count 篇互不相同 + dup_count 篇与第 1 篇内容相同。"""
day = "20260616"
proc_dir = tmp_path / "processed" / "testsrc" / day
proc_dir.mkdir(parents=True)
base_content = "这是一条足够长的测试新闻正文内容,用于触发去重判定逻辑。" * 3
# 有重复篇时才写基准篇(重复目标);否则唯一数 = uniq_count
if dup_count > 0:
_write_article(proc_dir, "https://x.example/base", "基准新闻", base_content)
# 与基准篇内容相同、URL 不同 → 判为重复
for i in range(dup_count):
_write_article(
proc_dir, f"https://x.example/dup{i}", f"重复新闻{i}", base_content
)
# 互不相同的其它唯一文章
for i in range(uniq_count):
_write_article(
proc_dir, f"https://x.example/uniq{i}", f"独立新闻{i}",
f"完全不同的正文内容片段编号 {i},讲述另一件事。" * 3,
)
return proc_dir
def _run_dedup_main(tmp_path: Path, *extra_args: str) -> int:
"""以指定参数调用 scripts.run_dedup.main,返回退出码。"""
import scripts.run_dedup as mod
day = "20260616"
argv = [
"run_dedup",
"--processed-root", str(tmp_path / "processed"),
"--out-root", str(tmp_path / "deduped"),
"--db", str(tmp_path / "fingerprints.sqlite3"),
"--date", day,
"--log-level", "ERROR",
*extra_args,
]
old_argv = sys.argv
sys.argv = argv
try:
return mod.main()
finally:
sys.argv = old_argv
def _read_stats(tmp_path: Path) -> dict:
p = tmp_path / "deduped" / "20260616" / "stats.json"
assert p.is_file(), "stats.json 未生成"
return json.loads(p.read_text(encoding="utf-8"))
# --------------------------------------------------------------------------- #
# 返回码语义
# --------------------------------------------------------------------------- #
def test_production_mode_high_dup_returns_0(tmp_path: Path) -> None:
"""生产模式:高重复率(90%)不影响退出码,返回 0。"""
_setup_fixture(tmp_path, dup_count=9, uniq_count=0)
rc = _run_dedup_main(tmp_path)
assert rc == 0, "生产模式下重复率仅作告警,不应改变退出码"
def test_strict_mode_high_dup_returns_1(tmp_path: Path) -> None:
"""验收模式 --strict:重复率超过 5% 门槛时返回 1。"""
_setup_fixture(tmp_path, dup_count=9, uniq_count=0)
rc = _run_dedup_main(tmp_path, "--strict")
assert rc == 1
def test_strict_mode_low_dup_returns_0(tmp_path: Path) -> None:
"""验收模式:重复率 ≤ 5% 时返回 0。"""
_setup_fixture(tmp_path, dup_count=0, uniq_count=10)
rc = _run_dedup_main(tmp_path, "--strict")
assert rc == 0
def test_empty_day_returns_0(tmp_path: Path) -> None:
"""当日无文章时返回 0(两种模式)。"""
proc_dir = tmp_path / "processed" / "testsrc" / "20260616"
proc_dir.mkdir(parents=True)
assert _run_dedup_main(tmp_path) == 0
assert _run_dedup_main(tmp_path, "--strict") == 0
# --------------------------------------------------------------------------- #
# stats.json 统计快照
# --------------------------------------------------------------------------- #
def test_stats_json_written_and_correct(tmp_path: Path) -> None:
"""stats.json 结构正确:重复率、唯一/重复数、指纹库总量。"""
_setup_fixture(tmp_path, dup_count=9, uniq_count=0)
_run_dedup_main(tmp_path)
s = _read_stats(tmp_path)
assert s["date"] == "20260616"
assert s["unique"] == 1
assert s["duplicates"] == 9
assert s["total"] == 10
assert s["dup_rate"] == pytest.approx(0.9, abs=1e-6)
assert s["dup_rate_threshold"] == pytest.approx(0.05)
# 指纹库只保留唯一文章
assert s["fingerprint_total"] == 1
assert "generated_at" in s and "layers" in s
def test_stats_json_low_dup_rate(tmp_path: Path) -> None:
"""低重复率场景:dup_rate 为 0。"""
_setup_fixture(tmp_path, dup_count=0, uniq_count=5)
_run_dedup_main(tmp_path)
s = _read_stats(tmp_path)
assert s["unique"] == 5
assert s["duplicates"] == 0
assert s["dup_rate"] == 0.0
assert s["fingerprint_total"] == 5
# --------------------------------------------------------------------------- #
# pipeline: dedup 不再特判(真实失败现在会正确上报)
# --------------------------------------------------------------------------- #
def test_pipeline_dedup_failure_no_longer_masked(monkeypatch: pytest.MonkeyPatch) -> None:
"""移除特判后:dedup 非 0 退出码应如实标记为失败。"""
from datetime import date
from types import SimpleNamespace
from scheduler import pipeline
def _fake_run(cmd, timeout=None): # noqa: ARG001
return SimpleNamespace(returncode=1)
monkeypatch.setattr(pipeline.subprocess, "run", _fake_run)
sr = pipeline.run_step("dedup", date.today().strftime("%Y%m%d"))
assert sr.success is False, "dedup 返回非 0 应标记失败(特判已移除)"
assert "rc=1" in sr.tail_msg