- run_dedup: 生产模式重复率仅作 WARNING 告警,不影响退出码(执行成功即 0)
- run_dedup: 新增 --strict 验收模式(重复率 > 5% 返回 1),保留 M3 验收门槛
- run_dedup: 新增统计快照 data/deduped/{date}/stats.json(原子写)
- run_dedup: 顺带修复空日场景 sources.json 写入 FileNotFoundError
- pipeline: 移除 dedup rc=1 特判,恢复'非 0 即失败'统一语义
- 新增 tests/test_run_dedup.py 7 个测试;全量 266 passed
174 lines
6.1 KiB
Python
174 lines
6.1 KiB
Python
"""run_dedup 脚本测试(生产/验收返回码语义 + stats.json 统计快照)。
|
|
|
|
覆盖:
|
|
- 生产模式(默认):高重复率不影响退出码,返回 0;重复率仅作告警
|
|
- 验收模式(--strict):高重复率返回 1(保留 M3 验收门槛 ≤ 5%)
|
|
- 低重复率场景两种模式均返回 0
|
|
- stats.json 统计快照结构正确(原子写),含重复率与指纹库总量
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
|
|
def _url_hash(url: str) -> str:
|
|
return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16]
|
|
|
|
|
|
def _write_article(proc_dir: Path, url: str, title: str, content: str) -> None:
|
|
"""写一个 Article JSON,文件名为 url_hash。"""
|
|
h = _url_hash(url)
|
|
article = {
|
|
"source_id": "testsrc",
|
|
"url": url,
|
|
"url_hash": h,
|
|
"title": title,
|
|
"content": content,
|
|
"word_count": len(content),
|
|
}
|
|
(proc_dir / f"{h}.json").write_text(
|
|
json.dumps(article, ensure_ascii=False), encoding="utf-8"
|
|
)
|
|
|
|
|
|
def _setup_fixture(tmp_path: Path, *, dup_count: int, uniq_count: int) -> Path:
|
|
"""构造 processed 夹具:uniq_count 篇互不相同 + dup_count 篇与第 1 篇内容相同。"""
|
|
day = "20260616"
|
|
proc_dir = tmp_path / "processed" / "testsrc" / day
|
|
proc_dir.mkdir(parents=True)
|
|
|
|
base_content = "这是一条足够长的测试新闻正文内容,用于触发去重判定逻辑。" * 3
|
|
# 有重复篇时才写基准篇(重复目标);否则唯一数 = uniq_count
|
|
if dup_count > 0:
|
|
_write_article(proc_dir, "https://x.example/base", "基准新闻", base_content)
|
|
# 与基准篇内容相同、URL 不同 → 判为重复
|
|
for i in range(dup_count):
|
|
_write_article(
|
|
proc_dir, f"https://x.example/dup{i}", f"重复新闻{i}", base_content
|
|
)
|
|
# 互不相同的其它唯一文章
|
|
for i in range(uniq_count):
|
|
_write_article(
|
|
proc_dir, f"https://x.example/uniq{i}", f"独立新闻{i}",
|
|
f"完全不同的正文内容片段编号 {i},讲述另一件事。" * 3,
|
|
)
|
|
return proc_dir
|
|
|
|
|
|
def _run_dedup_main(tmp_path: Path, *extra_args: str) -> int:
|
|
"""以指定参数调用 scripts.run_dedup.main,返回退出码。"""
|
|
import scripts.run_dedup as mod
|
|
|
|
day = "20260616"
|
|
argv = [
|
|
"run_dedup",
|
|
"--processed-root", str(tmp_path / "processed"),
|
|
"--out-root", str(tmp_path / "deduped"),
|
|
"--db", str(tmp_path / "fingerprints.sqlite3"),
|
|
"--date", day,
|
|
"--log-level", "ERROR",
|
|
*extra_args,
|
|
]
|
|
old_argv = sys.argv
|
|
sys.argv = argv
|
|
try:
|
|
return mod.main()
|
|
finally:
|
|
sys.argv = old_argv
|
|
|
|
|
|
def _read_stats(tmp_path: Path) -> dict:
|
|
p = tmp_path / "deduped" / "20260616" / "stats.json"
|
|
assert p.is_file(), "stats.json 未生成"
|
|
return json.loads(p.read_text(encoding="utf-8"))
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# 返回码语义
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def test_production_mode_high_dup_returns_0(tmp_path: Path) -> None:
|
|
"""生产模式:高重复率(90%)不影响退出码,返回 0。"""
|
|
_setup_fixture(tmp_path, dup_count=9, uniq_count=0)
|
|
rc = _run_dedup_main(tmp_path)
|
|
assert rc == 0, "生产模式下重复率仅作告警,不应改变退出码"
|
|
|
|
|
|
def test_strict_mode_high_dup_returns_1(tmp_path: Path) -> None:
|
|
"""验收模式 --strict:重复率超过 5% 门槛时返回 1。"""
|
|
_setup_fixture(tmp_path, dup_count=9, uniq_count=0)
|
|
rc = _run_dedup_main(tmp_path, "--strict")
|
|
assert rc == 1
|
|
|
|
|
|
def test_strict_mode_low_dup_returns_0(tmp_path: Path) -> None:
|
|
"""验收模式:重复率 ≤ 5% 时返回 0。"""
|
|
_setup_fixture(tmp_path, dup_count=0, uniq_count=10)
|
|
rc = _run_dedup_main(tmp_path, "--strict")
|
|
assert rc == 0
|
|
|
|
|
|
def test_empty_day_returns_0(tmp_path: Path) -> None:
|
|
"""当日无文章时返回 0(两种模式)。"""
|
|
proc_dir = tmp_path / "processed" / "testsrc" / "20260616"
|
|
proc_dir.mkdir(parents=True)
|
|
assert _run_dedup_main(tmp_path) == 0
|
|
assert _run_dedup_main(tmp_path, "--strict") == 0
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# stats.json 统计快照
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def test_stats_json_written_and_correct(tmp_path: Path) -> None:
|
|
"""stats.json 结构正确:重复率、唯一/重复数、指纹库总量。"""
|
|
_setup_fixture(tmp_path, dup_count=9, uniq_count=0)
|
|
_run_dedup_main(tmp_path)
|
|
s = _read_stats(tmp_path)
|
|
assert s["date"] == "20260616"
|
|
assert s["unique"] == 1
|
|
assert s["duplicates"] == 9
|
|
assert s["total"] == 10
|
|
assert s["dup_rate"] == pytest.approx(0.9, abs=1e-6)
|
|
assert s["dup_rate_threshold"] == pytest.approx(0.05)
|
|
# 指纹库只保留唯一文章
|
|
assert s["fingerprint_total"] == 1
|
|
assert "generated_at" in s and "layers" in s
|
|
|
|
|
|
def test_stats_json_low_dup_rate(tmp_path: Path) -> None:
|
|
"""低重复率场景:dup_rate 为 0。"""
|
|
_setup_fixture(tmp_path, dup_count=0, uniq_count=5)
|
|
_run_dedup_main(tmp_path)
|
|
s = _read_stats(tmp_path)
|
|
assert s["unique"] == 5
|
|
assert s["duplicates"] == 0
|
|
assert s["dup_rate"] == 0.0
|
|
assert s["fingerprint_total"] == 5
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# pipeline: dedup 不再特判(真实失败现在会正确上报)
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def test_pipeline_dedup_failure_no_longer_masked(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
"""移除特判后:dedup 非 0 退出码应如实标记为失败。"""
|
|
from datetime import date
|
|
from types import SimpleNamespace
|
|
|
|
from scheduler import pipeline
|
|
|
|
def _fake_run(cmd, timeout=None): # noqa: ARG001
|
|
return SimpleNamespace(returncode=1)
|
|
|
|
monkeypatch.setattr(pipeline.subprocess, "run", _fake_run)
|
|
sr = pipeline.run_step("dedup", date.today().strftime("%Y%m%d"))
|
|
assert sr.success is False, "dedup 返回非 0 应标记失败(特判已移除)"
|
|
assert "rc=1" in sr.tail_msg
|