- run_xwlb: 新增 --date(处理日)参数,落盘改为处理日目录,下游 M2-M6 打通
- run_xwlb: 幂等落盘(按 url_hash 去重,重写 index 替代裸追加)
- pipeline: xwlb 步骤传 --date {date}
- run_extractor: xwlb publish_time 从 url 兜底解析真实播出日
- 新增 tests/test_run_xwlb.py 9 个测试
- 清理 60 天历史死数据(用户决策 A1)
- docs: 文档重构收尾(删除 deployment.md,已在服务器直接修改)
164 lines
6.3 KiB
Python
164 lines
6.3 KiB
Python
"""run_xwlb 脚本测试(处理日目录语义 + 幂等去重)。
|
|
|
|
覆盖:
|
|
- 落盘目录 = 处理日(而非数据日): 处理日 20260823, 抓数据日 20260822 的联播,
|
|
写入 data/raw/xwlb/20260823/
|
|
- 处理日目录下 index.jsonl 的 source_id/url_hash 正确
|
|
- 幂等: 同一处理日重复运行,index.jsonl 不产生重复行(url_hash 唯一)
|
|
- 日期格式错误返回 2,API 无数据返回 0
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from datetime import date, timedelta
|
|
from pathlib import Path
|
|
from unittest import mock
|
|
|
|
from scripts.run_xwlb import _load_existing_index, _url_hash
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# 幂等辅助函数
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def test_url_hash_stable() -> None:
|
|
"""url_hash 稳定且为 SHA1 前 16 位。"""
|
|
h1 = _url_hash("xwlb://2026-08-22/3")
|
|
h2 = _url_hash("xwlb://2026-08-22/3")
|
|
assert h1 == h2
|
|
assert len(h1) == 16
|
|
assert _url_hash("xwlb://2026-08-22/3") != _url_hash("xwlb://2026-08-22/4")
|
|
|
|
|
|
def test_load_existing_index_empty(tmp_path: Path) -> None:
|
|
"""无 index.jsonl 时返回空 dict。"""
|
|
assert _load_existing_index(tmp_path / "index.jsonl") == {}
|
|
|
|
|
|
def test_load_existing_index_parses(tmp_path: Path) -> None:
|
|
"""能读取已有行,key=url_hash;跳过损坏行。"""
|
|
p = tmp_path / "index.jsonl"
|
|
p.write_text('{"url_hash": "abc", "url": "x"}\nnot-json\n{"url_hash": "def", "url": "y"}\n', encoding="utf-8")
|
|
existing = _load_existing_index(p)
|
|
assert set(existing) == {"abc", "def"}
|
|
|
|
|
|
def test_load_existing_index_fallback_url(tmp_path: Path) -> None:
|
|
"""无 url_hash 时用 url 兜底。"""
|
|
p = tmp_path / "index.jsonl"
|
|
p.write_text('{"url": "xwlb://2026-08-22/1"}\n', encoding="utf-8")
|
|
existing = _load_existing_index(p)
|
|
assert "xwlb://2026-08-22/1" in existing
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# 主流程(以 mock 网络方式调用 main)
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def _make_api_body() -> dict:
|
|
"""构造 doorcome xwlb API 响应体(3 条 + 1 条内容提要)。"""
|
|
news = []
|
|
for sid in range(1, 5):
|
|
news.append({
|
|
"daily_sub_id": sid,
|
|
"news_title": f"联播标题{sid}",
|
|
"news_improve": f"联播正文内容{sid}",
|
|
"news_days": "2026-08-22",
|
|
})
|
|
return {"data": {"news": news}}
|
|
|
|
|
|
def _run_main(tmp_path: Path, process_day: str, body: dict | None = None) -> int:
|
|
"""以 mock API 响应调用 scripts.run_xwlb.main。"""
|
|
import scripts.run_xwlb as mod
|
|
|
|
body = body if body is not None else _make_api_body()
|
|
with mock.patch.object(mod.urllib.request, "urlopen") as m_urlopen:
|
|
resp = mock.MagicMock()
|
|
resp.read.return_value = json.dumps(body).encode("utf-8")
|
|
m_urlopen.return_value.__enter__.return_value = resp
|
|
# 直接调用 main 前的 argparse 不方便,这里调用内部逻辑等价于 main:
|
|
# 通过 subprocess 太重,改为构造 args 后调用主流程函数(main 内联,故复制流程)。
|
|
# 简化:直接验证 _load_existing_index + 手写落盘逻辑的核心——用真实 main 需要 sys.argv。
|
|
# 因此此处通过 mock sys.argv 调用 main。
|
|
import sys
|
|
old_argv = sys.argv
|
|
sys.argv = [
|
|
"run_xwlb",
|
|
"--date", process_day,
|
|
"--output-root", str(tmp_path),
|
|
"--log-level", "ERROR",
|
|
]
|
|
try:
|
|
rc = mod.main()
|
|
finally:
|
|
sys.argv = old_argv
|
|
return rc
|
|
|
|
|
|
def test_main_writes_process_day_dir(tmp_path: Path) -> None:
|
|
"""落盘目录 = 处理日(data/raw/xwlb/20260823),而非数据日 20260822。"""
|
|
rc = _run_main(tmp_path, "20260823")
|
|
assert rc == 0
|
|
|
|
out_dir = tmp_path / "xwlb" / "20260823"
|
|
assert out_dir.is_dir()
|
|
# 数据日目录不应存在
|
|
assert not (tmp_path / "xwlb" / "20260822").is_dir()
|
|
|
|
lines = (out_dir / "index.jsonl").read_text(encoding="utf-8").splitlines()
|
|
assert len(lines) == 4 # 4 条(含提要,但脚本不去重提要,只跳过 sid<=1 的逻辑在 reporter)
|
|
# 校验字段
|
|
recs = [json.loads(l) for l in lines]
|
|
assert all(r["source_id"] == "xwlb" for r in recs)
|
|
assert all(r["stage"] == "article" for r in recs)
|
|
assert all(r["url_hash"] for r in recs)
|
|
# html 文件落盘
|
|
html_files = list(out_dir.glob("*.html"))
|
|
assert len(html_files) == 4
|
|
|
|
|
|
def test_main_idempotent_no_duplicate_rows(tmp_path: Path) -> None:
|
|
"""同一处理日重复运行: index.jsonl 不产生重复行(url_hash 唯一)。"""
|
|
rc1 = _run_main(tmp_path, "20260823")
|
|
rc2 = _run_main(tmp_path, "20260823")
|
|
assert rc1 == 0 and rc2 == 0
|
|
|
|
lines = (tmp_path / "xwlb" / "20260823" / "index.jsonl").read_text(encoding="utf-8").splitlines()
|
|
hashes = [json.loads(l)["url_hash"] for l in lines]
|
|
assert len(hashes) == len(set(hashes)), "重复运行产生重复行"
|
|
assert len(lines) == 4, "第二次运行应全部跳过,行数不变"
|
|
|
|
|
|
def test_main_different_process_days_isolated(tmp_path: Path) -> None:
|
|
"""不同处理日使用不同目录,互不污染。"""
|
|
_run_main(tmp_path, "20260823")
|
|
_run_main(tmp_path, "20260824")
|
|
assert (tmp_path / "xwlb" / "20260823").is_dir()
|
|
assert (tmp_path / "xwlb" / "20260824").is_dir()
|
|
# 20260824 的数据日是 20260823,若 mock 固定返回 2026-08-22 数据,
|
|
# 两个目录内容应相同字段结构,但互不影响
|
|
n1 = len((tmp_path / "xwlb" / "20260823" / "index.jsonl").read_text(encoding="utf-8").splitlines())
|
|
n2 = len((tmp_path / "xwlb" / "20260824" / "index.jsonl").read_text(encoding="utf-8").splitlines())
|
|
assert n1 == 4 and n2 == 4
|
|
|
|
|
|
def test_main_empty_news_returns_0(tmp_path: Path) -> None:
|
|
"""API 无数据: 返回 0,不落盘。"""
|
|
rc = _run_main(tmp_path, "20260823", body={"data": {"news": []}})
|
|
assert rc == 0
|
|
assert not (tmp_path / "xwlb" / "20260823").is_dir()
|
|
|
|
|
|
def test_main_bad_date_returns_2(tmp_path: Path) -> None:
|
|
"""--date 格式错误: 返回 2。"""
|
|
import sys
|
|
import scripts.run_xwlb as mod
|
|
old_argv = sys.argv
|
|
sys.argv = ["run_xwlb", "--date", "2026-13-99", "--output-root", str(tmp_path)]
|
|
try:
|
|
rc = mod.main()
|
|
finally:
|
|
sys.argv = old_argv
|
|
assert rc == 2 |