From 85ba218f2f061a1e9b33e17ee86fe09a24bc2d55 Mon Sep 17 00:00:00 2001 From: simon Date: Sat, 22 Aug 2026 19:20:21 +0800 Subject: [PATCH] =?UTF-8?q?fix:=20=E4=BF=AE=E5=A4=8D=E8=A1=A5=E8=B7=91?= =?UTF-8?q?=E5=8E=86=E5=8F=B2=E6=97=A5=E6=9C=9F=E7=9A=84=E9=9D=99=E9=BB=98?= =?UTF-8?q?=E7=A9=BA=E8=BD=AC=E4=B8=8E=E6=97=A5=E6=8A=A5=E6=97=A5=E6=9C=9F?= =?UTF-8?q?=E9=94=99=E4=BD=8D=20(P0-3/P0-4)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - run_step: 补跑保护——历史日期跳过 crawler(首页只含当天内容)并告警,复用已有 raw 数据 - run_step: report 步骤改用传入的 date_str,不再硬编码 date.today() - 新增 4 个补跑相关测试,修正 3 个既有测试适配新行为 - 全量 259 passed(3 个 crawler 基线失败与本次无关) --- continuation.md | 24 +++++++++++- scheduler/pipeline.py | 18 ++++++++- tests/test_incremental.py | 80 +++++++++++++++++++++++++++++++++++++++ tests/test_scheduler.py | 13 +++++-- 4 files changed, 129 insertions(+), 6 deletions(-) diff --git a/continuation.md b/continuation.md index e6543c3..5a4e1da 100644 --- a/continuation.md +++ b/continuation.md @@ -4,6 +4,27 @@ --- +## 本次完成 (2026-08-22) — P0-3/P0-4 补跑日期语义修复 + +**P0-3(crawler 补跑历史日期静默空转)**: +- 原因:`crawler/storage.py` 落盘硬编码 `date.today()`(engine 不传 day),`pipeline --once --date {历史}` 时下游读历史目录为空,且 `run_extractor` 对空目录返回成功 → 全链路静默空转 +- 修复:`run_step()` 开头加补跑保护——`date_str != 今天` 时跳过 crawler(首页只含当天内容,历史文章已滚走,补抓不可能),WARNING 日志 + `tail_msg` 说明,返回 success(补跑复用已有 raw 数据是正确行为);xwlb 不跳过(API 支持任意历史日期) + +**P0-4(report 硬编码 date.today() 忽略传入日期)**: +- 原因:`run_step("report")` 未使用入参 `date_str`,取当前时间;补跑历史日期+--report 会生成今天的空日报并覆盖当天已有日报(唯一键 file_name='') +- 修复:一行改动 `report_date = date_str`;`reporter.generate_report(day_str)` 本身已支持任意日期,无需其它改动 + +**测试**: +- 新增 4 个(补跑跳过不执行子进程/当天正常执行/report 收到传入日期/全链路仅 crawler 跳过),并入 `tests/test_incremental.py` +- 修正 3 个既有测试(`tests/test_scheduler.py`)适配新行为(用当天日期测正常执行路径) +- 全量 **259 passed**,3 个 crawler 基线失败与本次无关 + +**待办/遗留**: +- P1:dedup rc=1 静默转成功(天天重复率>5% 被掩蔽)、守护进程补跑 steps 含 cninfo_*、调度时区与 date.today() 一致性,待用户决策 +- 既有小问题:`tests/test_scheduler.py` 部分 run_pipeline 测试未传 state_path,会写真实 `data/pipeline/state.json`(既有行为,未改) + +--- + ## 本次完成 (2026-08-22) — xwlb 日期错位修复(方案 A) **背景问题**: @@ -33,7 +54,8 @@ **运维规则新增**: 每次代码升级完成后必须执行 `sudo systemctl restart a-share-research.service` **待办/遗留**: -- P0-3(crawler 无 --date,补跑历史日期静默空转)、P0-4(report 硬编码 date.today())、P1(dedup rc=1 静默/补跑含 cninfo/时区)未修,待用户决策 +- ~~P0-3(crawler 无 --date,补跑历史日期静默空转)、P0-4(report 硬编码 date.today())~~ 已修复(见上方小节) +- P1(dedup rc=1 静默/补跑含 cninfo/时区)未修,待用户决策 - `run_extractor --force` 重提取会重复追加 `data/processed/{src}/{date}/index.jsonl`(仅影响该辅助索引,下游读 *.json 不受影响),暂不修 - 历史 60 天联播未入库(按用户决策 A1 清理,不回补) diff --git a/scheduler/pipeline.py b/scheduler/pipeline.py index ea303b9..ec8c7e0 100644 --- a/scheduler/pipeline.py +++ b/scheduler/pipeline.py @@ -220,15 +220,29 @@ def run_step(name: str, date_str: str) -> StepResult: 返回: StepResult。 """ + # 补跑保护:抓取类步骤只能产生当天数据(网站首页只含当前内容,历史文章已滚走), + # 补跑历史日期时跳过抓取并告警,复用已有 data/raw/*/{date_str} 数据。 + # 注意:仅保护 crawler; xwlb 走 API 支持任意历史日期,无需跳过。 + if name == "crawler" and date_str != date.today().strftime("%Y%m%d"): + started = datetime.now() + logger.warning( + "补跑模式:首页只含当天内容,无法补抓 {};跳过抓取,复用已有 data/raw/*/{}", + date_str, date_str, + ) + return StepResult( + name=name, success=True, elapsed_sec=0.0, + tail_msg="补跑跳过(抓取只能产生当天数据)", started_at=started, + ) + # report 步骤:内部函数,不走子进程 - # 日报按当天日期生成: 新闻由 _collect_news_events 回溯过去 30 小时, + # 日报按 date_str 日期生成: 新闻由 _collect_news_events 回溯过去 30 小时, # xwlb 由 _collect_xwlb 固定取前一日(已播出)联播。 if name == "report": started = datetime.now() try: from .reporter import generate_report # noqa: E402 - report_date = date.today().strftime("%Y%m%d") + report_date = date_str logger.info("日报: report_date={} (新闻 30h 回溯, xwlb 前一日)", report_date) path = generate_report(report_date, upload=True) diff --git a/tests/test_incremental.py b/tests/test_incremental.py index 38833cd..6b0eda3 100644 --- a/tests/test_incremental.py +++ b/tests/test_incremental.py @@ -266,3 +266,83 @@ def test_once_rejects_resume_with_steps(monkeypatch: pytest.MonkeyPatch) -> None date="20260616") rc = rs._once(args) assert rc == 2 + + +# --------------------------------------------------------------------------- # +# P0-3: 补跑模式跳过抓取 / P0-4: report 使用传入日期 +# --------------------------------------------------------------------------- # + +def test_crawler_backfill_skips_past_date(monkeypatch: pytest.MonkeyPatch) -> None: + """补跑历史日期时跳过 crawler(首页只含当天内容),不执行子进程。""" + from scheduler import pipeline + + def _fail_run(cmd, timeout=None): # noqa: ARG001 + raise AssertionError("补跑模式不应执行子进程") + + monkeypatch.setattr(pipeline.subprocess, "run", _fail_run) + sr = pipeline.run_step("crawler", "20260101") + assert sr.success is True + assert "跳过" in sr.tail_msg + assert sr.elapsed_sec == 0.0 + + +def test_crawler_today_runs_normally(monkeypatch: pytest.MonkeyPatch) -> None: + """当天日期正常执行 crawler(不受补跑保护影响)。""" + from datetime import date + + from scheduler import pipeline + + called: list[str] = [] + + def _fake_run(cmd, timeout=None): # noqa: ARG001 + called.append(" ".join(cmd)) + return SimpleNamespace(returncode=0) + + monkeypatch.setattr(pipeline.subprocess, "run", _fake_run) + sr = pipeline.run_step("crawler", date.today().strftime("%Y%m%d")) + assert sr.success is True + assert called and "run_crawler" in called[0] + + +def test_report_step_uses_date_str(monkeypatch: pytest.MonkeyPatch) -> None: + """report 步骤必须使用传入的 date_str,而非 date.today()(P0-4)。""" + from scheduler import pipeline + from scheduler import reporter + + received: list[str] = [] + + def _fake_generate(day_str, upload=True): # noqa: ARG001 + received.append(day_str) + return 12345 + + monkeypatch.setattr(reporter, "generate_report", _fake_generate) + sr = pipeline.run_step("report", "20260101") + assert sr.success is True + assert received == ["20260101"] + + +def test_pipeline_backfill_skips_crawler_keeps_rest( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + """补跑历史日期:全链路中仅 crawler 跳过,其余步骤照常执行。""" + from scheduler import pipeline + + calls: list[str] = [] + + def _fake_run(cmd, timeout=None): # noqa: ARG001 + name = next(c.split(".")[-1] for c in cmd if "scripts.run_" in c) + calls.append(name) + return SimpleNamespace(returncode=0) + + monkeypatch.setattr(pipeline.subprocess, "run", _fake_run) + state_path = tmp_path / "state.json" + result = pipeline.run_pipeline( + "20260101", + steps=["crawler", "extractor", "dedup"], + state_path=state_path, + ) + # crawler 跳过(未执行子进程),extractor/dedup 正常 + assert calls == ["run_extractor", "run_dedup"] + assert all(s.success for s in result.steps) + state = pipeline._load_pipeline_state(state_path) + assert state["20260101"]["crawler"]["status"] == "ok" diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index 80b34ff..63ca1e6 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -56,9 +56,12 @@ def test_pipeline_result_partial_failure() -> None: # --------------------------------------------------------------------------- # def test_run_step_success() -> None: + # crawler 补跑保护:仅当天日期才执行子进程,故用今天的日期 + from datetime import date + today = date.today().strftime("%Y%m%d") with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="INFO | 完成: 成功 20/20")): - sr = run_step("crawler", "20260616") + sr = run_step("crawler", today) assert sr.success is True assert sr.exit_code == 0 @@ -109,6 +112,8 @@ def test_run_pipeline_all_success() -> None: def test_run_pipeline_continues_on_failure() -> None: """中间步骤失败,后续继续执行(不阻断)。""" + from datetime import date + today = date.today().strftime("%Y%m%d") # crawler 补跑保护:需当天日期才执行 call_count = {"n": 0} def _side_effect(*args, **kwargs): @@ -118,7 +123,7 @@ def test_run_pipeline_continues_on_failure() -> None: return _mock_proc(returncode=0, stderr="ok") with patch("subprocess.run", side_effect=_side_effect): - result = run_pipeline("20260616", steps=["crawler", "extractor", "dedup", "llm"]) + result = run_pipeline(today, steps=["crawler", "extractor", "dedup", "llm"]) assert len(result.steps) == 4 # extractor 失败,但后续仍执行 assert result.steps[1].success is False @@ -126,7 +131,9 @@ def test_run_pipeline_continues_on_failure() -> None: def test_run_pipeline_custom_steps() -> None: + from datetime import date + today = date.today().strftime("%Y%m%d") with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="ok")): - result = run_pipeline("20260616", steps=["crawler", "extractor"]) + result = run_pipeline(today, steps=["crawler", "extractor"]) assert len(result.steps) == 2 assert result.all_success is True