fix: 修复补跑历史日期的静默空转与日报日期错位 (P0-3/P0-4)

- run_step: 补跑保护——历史日期跳过 crawler(首页只含当天内容)并告警,复用已有 raw 数据
- run_step: report 步骤改用传入的 date_str,不再硬编码 date.today()
- 新增 4 个补跑相关测试,修正 3 个既有测试适配新行为
- 全量 259 passed(3 个 crawler 基线失败与本次无关)
This commit is contained in:
2026-08-22 19:20:21 +08:00
parent 7ea8925209
commit 85ba218f2f
4 changed files with 129 additions and 6 deletions
+23 -1
View File
@@ -4,6 +4,27 @@
--- ---
## 本次完成 (2026-08-22) — P0-3/P0-4 补跑日期语义修复
**P0-3(crawler 补跑历史日期静默空转)**:
- 原因:`crawler/storage.py` 落盘硬编码 `date.today()`(engine 不传 day),`pipeline --once --date {历史}` 时下游读历史目录为空,且 `run_extractor` 对空目录返回成功 → 全链路静默空转
- 修复:`run_step()` 开头加补跑保护——`date_str != 今天` 时跳过 crawler(首页只含当天内容,历史文章已滚走,补抓不可能),WARNING 日志 + `tail_msg` 说明,返回 success(补跑复用已有 raw 数据是正确行为);xwlb 不跳过(API 支持任意历史日期)
**P0-4(report 硬编码 date.today() 忽略传入日期)**:
- 原因:`run_step("report")` 未使用入参 `date_str`,取当前时间;补跑历史日期+--report 会生成今天的空日报并覆盖当天已有日报(唯一键 file_name='')
- 修复:一行改动 `report_date = date_str`;`reporter.generate_report(day_str)` 本身已支持任意日期,无需其它改动
**测试**:
- 新增 4 个(补跑跳过不执行子进程/当天正常执行/report 收到传入日期/全链路仅 crawler 跳过),并入 `tests/test_incremental.py`
- 修正 3 个既有测试(`tests/test_scheduler.py`)适配新行为(用当天日期测正常执行路径)
- 全量 **259 passed**,3 个 crawler 基线失败与本次无关
**待办/遗留**:
- P1:dedup rc=1 静默转成功(天天重复率>5% 被掩蔽)、守护进程补跑 steps 含 cninfo_*、调度时区与 date.today() 一致性,待用户决策
- 既有小问题:`tests/test_scheduler.py` 部分 run_pipeline 测试未传 state_path,会写真实 `data/pipeline/state.json`(既有行为,未改)
---
## 本次完成 (2026-08-22) — xwlb 日期错位修复(方案 A) ## 本次完成 (2026-08-22) — xwlb 日期错位修复(方案 A)
**背景问题**: **背景问题**:
@@ -33,7 +54,8 @@
**运维规则新增**: 每次代码升级完成后必须执行 `sudo systemctl restart a-share-research.service` **运维规则新增**: 每次代码升级完成后必须执行 `sudo systemctl restart a-share-research.service`
**待办/遗留**: **待办/遗留**:
- P0-3(crawler 无 --date,补跑历史日期静默空转)、P0-4(report 硬编码 date.today())、P1(dedup rc=1 静默/补跑含 cninfo/时区)未修,待用户决策 - ~~P0-3(crawler 无 --date,补跑历史日期静默空转)、P0-4(report 硬编码 date.today())~~ 已修复(见上方小节)
- P1(dedup rc=1 静默/补跑含 cninfo/时区)未修,待用户决策
- `run_extractor --force` 重提取会重复追加 `data/processed/{src}/{date}/index.jsonl`(仅影响该辅助索引,下游读 *.json 不受影响),暂不修 - `run_extractor --force` 重提取会重复追加 `data/processed/{src}/{date}/index.jsonl`(仅影响该辅助索引,下游读 *.json 不受影响),暂不修
- 历史 60 天联播未入库(按用户决策 A1 清理,不回补) - 历史 60 天联播未入库(按用户决策 A1 清理,不回补)
+16 -2
View File
@@ -220,15 +220,29 @@ def run_step(name: str, date_str: str) -> StepResult:
返回: StepResult。 返回: StepResult。
""" """
# 补跑保护:抓取类步骤只能产生当天数据(网站首页只含当前内容,历史文章已滚走),
# 补跑历史日期时跳过抓取并告警,复用已有 data/raw/*/{date_str} 数据。
# 注意:仅保护 crawler; xwlb 走 API 支持任意历史日期,无需跳过。
if name == "crawler" and date_str != date.today().strftime("%Y%m%d"):
started = datetime.now()
logger.warning(
"补跑模式:首页只含当天内容,无法补抓 {};跳过抓取,复用已有 data/raw/*/{}",
date_str, date_str,
)
return StepResult(
name=name, success=True, elapsed_sec=0.0,
tail_msg="补跑跳过(抓取只能产生当天数据)", started_at=started,
)
# report 步骤:内部函数,不走子进程 # report 步骤:内部函数,不走子进程
# 日报按当天日期生成: 新闻由 _collect_news_events 回溯过去 30 小时, # 日报按 date_str 日期生成: 新闻由 _collect_news_events 回溯过去 30 小时,
# xwlb 由 _collect_xwlb 固定取前一日(已播出)联播。 # xwlb 由 _collect_xwlb 固定取前一日(已播出)联播。
if name == "report": if name == "report":
started = datetime.now() started = datetime.now()
try: try:
from .reporter import generate_report # noqa: E402 from .reporter import generate_report # noqa: E402
report_date = date.today().strftime("%Y%m%d") report_date = date_str
logger.info("日报: report_date={} (新闻 30h 回溯, xwlb 前一日)", report_date) logger.info("日报: report_date={} (新闻 30h 回溯, xwlb 前一日)", report_date)
path = generate_report(report_date, upload=True) path = generate_report(report_date, upload=True)
+80
View File
@@ -266,3 +266,83 @@ def test_once_rejects_resume_with_steps(monkeypatch: pytest.MonkeyPatch) -> None
date="20260616") date="20260616")
rc = rs._once(args) rc = rs._once(args)
assert rc == 2 assert rc == 2
# --------------------------------------------------------------------------- #
# P0-3: 补跑模式跳过抓取 / P0-4: report 使用传入日期
# --------------------------------------------------------------------------- #
def test_crawler_backfill_skips_past_date(monkeypatch: pytest.MonkeyPatch) -> None:
"""补跑历史日期时跳过 crawler(首页只含当天内容),不执行子进程。"""
from scheduler import pipeline
def _fail_run(cmd, timeout=None): # noqa: ARG001
raise AssertionError("补跑模式不应执行子进程")
monkeypatch.setattr(pipeline.subprocess, "run", _fail_run)
sr = pipeline.run_step("crawler", "20260101")
assert sr.success is True
assert "跳过" in sr.tail_msg
assert sr.elapsed_sec == 0.0
def test_crawler_today_runs_normally(monkeypatch: pytest.MonkeyPatch) -> None:
"""当天日期正常执行 crawler(不受补跑保护影响)。"""
from datetime import date
from scheduler import pipeline
called: list[str] = []
def _fake_run(cmd, timeout=None): # noqa: ARG001
called.append(" ".join(cmd))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(pipeline.subprocess, "run", _fake_run)
sr = pipeline.run_step("crawler", date.today().strftime("%Y%m%d"))
assert sr.success is True
assert called and "run_crawler" in called[0]
def test_report_step_uses_date_str(monkeypatch: pytest.MonkeyPatch) -> None:
"""report 步骤必须使用传入的 date_str,而非 date.today()(P0-4)。"""
from scheduler import pipeline
from scheduler import reporter
received: list[str] = []
def _fake_generate(day_str, upload=True): # noqa: ARG001
received.append(day_str)
return 12345
monkeypatch.setattr(reporter, "generate_report", _fake_generate)
sr = pipeline.run_step("report", "20260101")
assert sr.success is True
assert received == ["20260101"]
def test_pipeline_backfill_skips_crawler_keeps_rest(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch,
) -> None:
"""补跑历史日期:全链路中仅 crawler 跳过,其余步骤照常执行。"""
from scheduler import pipeline
calls: list[str] = []
def _fake_run(cmd, timeout=None): # noqa: ARG001
name = next(c.split(".")[-1] for c in cmd if "scripts.run_" in c)
calls.append(name)
return SimpleNamespace(returncode=0)
monkeypatch.setattr(pipeline.subprocess, "run", _fake_run)
state_path = tmp_path / "state.json"
result = pipeline.run_pipeline(
"20260101",
steps=["crawler", "extractor", "dedup"],
state_path=state_path,
)
# crawler 跳过(未执行子进程),extractor/dedup 正常
assert calls == ["run_extractor", "run_dedup"]
assert all(s.success for s in result.steps)
state = pipeline._load_pipeline_state(state_path)
assert state["20260101"]["crawler"]["status"] == "ok"
+10 -3
View File
@@ -56,9 +56,12 @@ def test_pipeline_result_partial_failure() -> None:
# --------------------------------------------------------------------------- # # --------------------------------------------------------------------------- #
def test_run_step_success() -> None: def test_run_step_success() -> None:
# crawler 补跑保护:仅当天日期才执行子进程,故用今天的日期
from datetime import date
today = date.today().strftime("%Y%m%d")
with patch("subprocess.run", return_value=_mock_proc(returncode=0, with patch("subprocess.run", return_value=_mock_proc(returncode=0,
stderr="INFO | 完成: 成功 20/20")): stderr="INFO | 完成: 成功 20/20")):
sr = run_step("crawler", "20260616") sr = run_step("crawler", today)
assert sr.success is True assert sr.success is True
assert sr.exit_code == 0 assert sr.exit_code == 0
@@ -109,6 +112,8 @@ def test_run_pipeline_all_success() -> None:
def test_run_pipeline_continues_on_failure() -> None: def test_run_pipeline_continues_on_failure() -> None:
"""中间步骤失败,后续继续执行(不阻断)。""" """中间步骤失败,后续继续执行(不阻断)。"""
from datetime import date
today = date.today().strftime("%Y%m%d") # crawler 补跑保护:需当天日期才执行
call_count = {"n": 0} call_count = {"n": 0}
def _side_effect(*args, **kwargs): def _side_effect(*args, **kwargs):
@@ -118,7 +123,7 @@ def test_run_pipeline_continues_on_failure() -> None:
return _mock_proc(returncode=0, stderr="ok") return _mock_proc(returncode=0, stderr="ok")
with patch("subprocess.run", side_effect=_side_effect): with patch("subprocess.run", side_effect=_side_effect):
result = run_pipeline("20260616", steps=["crawler", "extractor", "dedup", "llm"]) result = run_pipeline(today, steps=["crawler", "extractor", "dedup", "llm"])
assert len(result.steps) == 4 assert len(result.steps) == 4
# extractor 失败,但后续仍执行 # extractor 失败,但后续仍执行
assert result.steps[1].success is False assert result.steps[1].success is False
@@ -126,7 +131,9 @@ def test_run_pipeline_continues_on_failure() -> None:
def test_run_pipeline_custom_steps() -> None: def test_run_pipeline_custom_steps() -> None:
from datetime import date
today = date.today().strftime("%Y%m%d")
with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="ok")): with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="ok")):
result = run_pipeline("20260616", steps=["crawler", "extractor"]) result = run_pipeline(today, steps=["crawler", "extractor"])
assert len(result.steps) == 2 assert len(result.steps) == 2
assert result.all_success is True assert result.all_success is True