fix: 修复 xwlb 日期错位与重复行问题(方案 A)

- run_xwlb: 新增 --date(处理日)参数,落盘改为处理日目录,下游 M2-M6 打通
- run_xwlb: 幂等落盘(按 url_hash 去重,重写 index 替代裸追加)
- pipeline: xwlb 步骤传 --date {date}
- run_extractor: xwlb publish_time 从 url 兜底解析真实播出日
- 新增 tests/test_run_xwlb.py 9 个测试
- 清理 60 天历史死数据(用户决策 A1)
- docs: 文档重构收尾(删除 deployment.md,已在服务器直接修改)
This commit is contained in:
2026-08-22 19:11:32 +08:00
parent 65ead54b4f
commit 7ea8925209
5 changed files with 306 additions and 19 deletions
+164
View File
@@ -0,0 +1,164 @@
"""run_xwlb 脚本测试(处理日目录语义 + 幂等去重)。
覆盖:
- 落盘目录 = 处理日(而非数据日): 处理日 20260823, 抓数据日 20260822 的联播,
写入 data/raw/xwlb/20260823/
- 处理日目录下 index.jsonl 的 source_id/url_hash 正确
- 幂等: 同一处理日重复运行,index.jsonl 不产生重复行(url_hash 唯一)
- 日期格式错误返回 2,API 无数据返回 0
"""
from __future__ import annotations
import json
from datetime import date, timedelta
from pathlib import Path
from unittest import mock
from scripts.run_xwlb import _load_existing_index, _url_hash
# --------------------------------------------------------------------------- #
# 幂等辅助函数
# --------------------------------------------------------------------------- #
def test_url_hash_stable() -> None:
"""url_hash 稳定且为 SHA1 前 16 位。"""
h1 = _url_hash("xwlb://2026-08-22/3")
h2 = _url_hash("xwlb://2026-08-22/3")
assert h1 == h2
assert len(h1) == 16
assert _url_hash("xwlb://2026-08-22/3") != _url_hash("xwlb://2026-08-22/4")
def test_load_existing_index_empty(tmp_path: Path) -> None:
"""无 index.jsonl 时返回空 dict。"""
assert _load_existing_index(tmp_path / "index.jsonl") == {}
def test_load_existing_index_parses(tmp_path: Path) -> None:
"""能读取已有行,key=url_hash;跳过损坏行。"""
p = tmp_path / "index.jsonl"
p.write_text('{"url_hash": "abc", "url": "x"}\nnot-json\n{"url_hash": "def", "url": "y"}\n', encoding="utf-8")
existing = _load_existing_index(p)
assert set(existing) == {"abc", "def"}
def test_load_existing_index_fallback_url(tmp_path: Path) -> None:
"""无 url_hash 时用 url 兜底。"""
p = tmp_path / "index.jsonl"
p.write_text('{"url": "xwlb://2026-08-22/1"}\n', encoding="utf-8")
existing = _load_existing_index(p)
assert "xwlb://2026-08-22/1" in existing
# --------------------------------------------------------------------------- #
# 主流程(以 mock 网络方式调用 main)
# --------------------------------------------------------------------------- #
def _make_api_body() -> dict:
"""构造 doorcome xwlb API 响应体(3 条 + 1 条内容提要)。"""
news = []
for sid in range(1, 5):
news.append({
"daily_sub_id": sid,
"news_title": f"联播标题{sid}",
"news_improve": f"联播正文内容{sid}",
"news_days": "2026-08-22",
})
return {"data": {"news": news}}
def _run_main(tmp_path: Path, process_day: str, body: dict | None = None) -> int:
"""以 mock API 响应调用 scripts.run_xwlb.main。"""
import scripts.run_xwlb as mod
body = body if body is not None else _make_api_body()
with mock.patch.object(mod.urllib.request, "urlopen") as m_urlopen:
resp = mock.MagicMock()
resp.read.return_value = json.dumps(body).encode("utf-8")
m_urlopen.return_value.__enter__.return_value = resp
# 直接调用 main 前的 argparse 不方便,这里调用内部逻辑等价于 main:
# 通过 subprocess 太重,改为构造 args 后调用主流程函数(main 内联,故复制流程)。
# 简化:直接验证 _load_existing_index + 手写落盘逻辑的核心——用真实 main 需要 sys.argv。
# 因此此处通过 mock sys.argv 调用 main。
import sys
old_argv = sys.argv
sys.argv = [
"run_xwlb",
"--date", process_day,
"--output-root", str(tmp_path),
"--log-level", "ERROR",
]
try:
rc = mod.main()
finally:
sys.argv = old_argv
return rc
def test_main_writes_process_day_dir(tmp_path: Path) -> None:
"""落盘目录 = 处理日(data/raw/xwlb/20260823),而非数据日 20260822。"""
rc = _run_main(tmp_path, "20260823")
assert rc == 0
out_dir = tmp_path / "xwlb" / "20260823"
assert out_dir.is_dir()
# 数据日目录不应存在
assert not (tmp_path / "xwlb" / "20260822").is_dir()
lines = (out_dir / "index.jsonl").read_text(encoding="utf-8").splitlines()
assert len(lines) == 4 # 4 条(含提要,但脚本不去重提要,只跳过 sid<=1 的逻辑在 reporter)
# 校验字段
recs = [json.loads(l) for l in lines]
assert all(r["source_id"] == "xwlb" for r in recs)
assert all(r["stage"] == "article" for r in recs)
assert all(r["url_hash"] for r in recs)
# html 文件落盘
html_files = list(out_dir.glob("*.html"))
assert len(html_files) == 4
def test_main_idempotent_no_duplicate_rows(tmp_path: Path) -> None:
"""同一处理日重复运行: index.jsonl 不产生重复行(url_hash 唯一)。"""
rc1 = _run_main(tmp_path, "20260823")
rc2 = _run_main(tmp_path, "20260823")
assert rc1 == 0 and rc2 == 0
lines = (tmp_path / "xwlb" / "20260823" / "index.jsonl").read_text(encoding="utf-8").splitlines()
hashes = [json.loads(l)["url_hash"] for l in lines]
assert len(hashes) == len(set(hashes)), "重复运行产生重复行"
assert len(lines) == 4, "第二次运行应全部跳过,行数不变"
def test_main_different_process_days_isolated(tmp_path: Path) -> None:
"""不同处理日使用不同目录,互不污染。"""
_run_main(tmp_path, "20260823")
_run_main(tmp_path, "20260824")
assert (tmp_path / "xwlb" / "20260823").is_dir()
assert (tmp_path / "xwlb" / "20260824").is_dir()
# 20260824 的数据日是 20260823,若 mock 固定返回 2026-08-22 数据,
# 两个目录内容应相同字段结构,但互不影响
n1 = len((tmp_path / "xwlb" / "20260823" / "index.jsonl").read_text(encoding="utf-8").splitlines())
n2 = len((tmp_path / "xwlb" / "20260824" / "index.jsonl").read_text(encoding="utf-8").splitlines())
assert n1 == 4 and n2 == 4
def test_main_empty_news_returns_0(tmp_path: Path) -> None:
"""API 无数据: 返回 0,不落盘。"""
rc = _run_main(tmp_path, "20260823", body={"data": {"news": []}})
assert rc == 0
assert not (tmp_path / "xwlb" / "20260823").is_dir()
def test_main_bad_date_returns_2(tmp_path: Path) -> None:
"""--date 格式错误: 返回 2。"""
import sys
import scripts.run_xwlb as mod
old_argv = sys.argv
sys.argv = ["run_xwlb", "--date", "2026-13-99", "--output-root", str(tmp_path)]
try:
rc = mod.main()
finally:
sys.argv = old_argv
assert rc == 2