fix: 修复补跑历史日期的静默空转与日报日期错位 (P0-3/P0-4)
- run_step: 补跑保护——历史日期跳过 crawler(首页只含当天内容)并告警,复用已有 raw 数据 - run_step: report 步骤改用传入的 date_str,不再硬编码 date.today() - 新增 4 个补跑相关测试,修正 3 个既有测试适配新行为 - 全量 259 passed(3 个 crawler 基线失败与本次无关)
This commit is contained in:
+23
-1
@@ -4,6 +4,27 @@
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## 本次完成 (2026-08-22) — P0-3/P0-4 补跑日期语义修复
|
||||||
|
|
||||||
|
**P0-3(crawler 补跑历史日期静默空转)**:
|
||||||
|
- 原因:`crawler/storage.py` 落盘硬编码 `date.today()`(engine 不传 day),`pipeline --once --date {历史}` 时下游读历史目录为空,且 `run_extractor` 对空目录返回成功 → 全链路静默空转
|
||||||
|
- 修复:`run_step()` 开头加补跑保护——`date_str != 今天` 时跳过 crawler(首页只含当天内容,历史文章已滚走,补抓不可能),WARNING 日志 + `tail_msg` 说明,返回 success(补跑复用已有 raw 数据是正确行为);xwlb 不跳过(API 支持任意历史日期)
|
||||||
|
|
||||||
|
**P0-4(report 硬编码 date.today() 忽略传入日期)**:
|
||||||
|
- 原因:`run_step("report")` 未使用入参 `date_str`,取当前时间;补跑历史日期+--report 会生成今天的空日报并覆盖当天已有日报(唯一键 file_name='')
|
||||||
|
- 修复:一行改动 `report_date = date_str`;`reporter.generate_report(day_str)` 本身已支持任意日期,无需其它改动
|
||||||
|
|
||||||
|
**测试**:
|
||||||
|
- 新增 4 个(补跑跳过不执行子进程/当天正常执行/report 收到传入日期/全链路仅 crawler 跳过),并入 `tests/test_incremental.py`
|
||||||
|
- 修正 3 个既有测试(`tests/test_scheduler.py`)适配新行为(用当天日期测正常执行路径)
|
||||||
|
- 全量 **259 passed**,3 个 crawler 基线失败与本次无关
|
||||||
|
|
||||||
|
**待办/遗留**:
|
||||||
|
- P1:dedup rc=1 静默转成功(天天重复率>5% 被掩蔽)、守护进程补跑 steps 含 cninfo_*、调度时区与 date.today() 一致性,待用户决策
|
||||||
|
- 既有小问题:`tests/test_scheduler.py` 部分 run_pipeline 测试未传 state_path,会写真实 `data/pipeline/state.json`(既有行为,未改)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## 本次完成 (2026-08-22) — xwlb 日期错位修复(方案 A)
|
## 本次完成 (2026-08-22) — xwlb 日期错位修复(方案 A)
|
||||||
|
|
||||||
**背景问题**:
|
**背景问题**:
|
||||||
@@ -33,7 +54,8 @@
|
|||||||
**运维规则新增**: 每次代码升级完成后必须执行 `sudo systemctl restart a-share-research.service`
|
**运维规则新增**: 每次代码升级完成后必须执行 `sudo systemctl restart a-share-research.service`
|
||||||
|
|
||||||
**待办/遗留**:
|
**待办/遗留**:
|
||||||
- P0-3(crawler 无 --date,补跑历史日期静默空转)、P0-4(report 硬编码 date.today())、P1(dedup rc=1 静默/补跑含 cninfo/时区)未修,待用户决策
|
- ~~P0-3(crawler 无 --date,补跑历史日期静默空转)、P0-4(report 硬编码 date.today())~~ 已修复(见上方小节)
|
||||||
|
- P1(dedup rc=1 静默/补跑含 cninfo/时区)未修,待用户决策
|
||||||
- `run_extractor --force` 重提取会重复追加 `data/processed/{src}/{date}/index.jsonl`(仅影响该辅助索引,下游读 *.json 不受影响),暂不修
|
- `run_extractor --force` 重提取会重复追加 `data/processed/{src}/{date}/index.jsonl`(仅影响该辅助索引,下游读 *.json 不受影响),暂不修
|
||||||
- 历史 60 天联播未入库(按用户决策 A1 清理,不回补)
|
- 历史 60 天联播未入库(按用户决策 A1 清理,不回补)
|
||||||
|
|
||||||
|
|||||||
+16
-2
@@ -220,15 +220,29 @@ def run_step(name: str, date_str: str) -> StepResult:
|
|||||||
|
|
||||||
返回: StepResult。
|
返回: StepResult。
|
||||||
"""
|
"""
|
||||||
|
# 补跑保护:抓取类步骤只能产生当天数据(网站首页只含当前内容,历史文章已滚走),
|
||||||
|
# 补跑历史日期时跳过抓取并告警,复用已有 data/raw/*/{date_str} 数据。
|
||||||
|
# 注意:仅保护 crawler; xwlb 走 API 支持任意历史日期,无需跳过。
|
||||||
|
if name == "crawler" and date_str != date.today().strftime("%Y%m%d"):
|
||||||
|
started = datetime.now()
|
||||||
|
logger.warning(
|
||||||
|
"补跑模式:首页只含当天内容,无法补抓 {};跳过抓取,复用已有 data/raw/*/{}",
|
||||||
|
date_str, date_str,
|
||||||
|
)
|
||||||
|
return StepResult(
|
||||||
|
name=name, success=True, elapsed_sec=0.0,
|
||||||
|
tail_msg="补跑跳过(抓取只能产生当天数据)", started_at=started,
|
||||||
|
)
|
||||||
|
|
||||||
# report 步骤:内部函数,不走子进程
|
# report 步骤:内部函数,不走子进程
|
||||||
# 日报按当天日期生成: 新闻由 _collect_news_events 回溯过去 30 小时,
|
# 日报按 date_str 日期生成: 新闻由 _collect_news_events 回溯过去 30 小时,
|
||||||
# xwlb 由 _collect_xwlb 固定取前一日(已播出)联播。
|
# xwlb 由 _collect_xwlb 固定取前一日(已播出)联播。
|
||||||
if name == "report":
|
if name == "report":
|
||||||
started = datetime.now()
|
started = datetime.now()
|
||||||
try:
|
try:
|
||||||
from .reporter import generate_report # noqa: E402
|
from .reporter import generate_report # noqa: E402
|
||||||
|
|
||||||
report_date = date.today().strftime("%Y%m%d")
|
report_date = date_str
|
||||||
logger.info("日报: report_date={} (新闻 30h 回溯, xwlb 前一日)", report_date)
|
logger.info("日报: report_date={} (新闻 30h 回溯, xwlb 前一日)", report_date)
|
||||||
|
|
||||||
path = generate_report(report_date, upload=True)
|
path = generate_report(report_date, upload=True)
|
||||||
|
|||||||
@@ -266,3 +266,83 @@ def test_once_rejects_resume_with_steps(monkeypatch: pytest.MonkeyPatch) -> None
|
|||||||
date="20260616")
|
date="20260616")
|
||||||
rc = rs._once(args)
|
rc = rs._once(args)
|
||||||
assert rc == 2
|
assert rc == 2
|
||||||
|
|
||||||
|
|
||||||
|
# --------------------------------------------------------------------------- #
|
||||||
|
# P0-3: 补跑模式跳过抓取 / P0-4: report 使用传入日期
|
||||||
|
# --------------------------------------------------------------------------- #
|
||||||
|
|
||||||
|
def test_crawler_backfill_skips_past_date(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
|
"""补跑历史日期时跳过 crawler(首页只含当天内容),不执行子进程。"""
|
||||||
|
from scheduler import pipeline
|
||||||
|
|
||||||
|
def _fail_run(cmd, timeout=None): # noqa: ARG001
|
||||||
|
raise AssertionError("补跑模式不应执行子进程")
|
||||||
|
|
||||||
|
monkeypatch.setattr(pipeline.subprocess, "run", _fail_run)
|
||||||
|
sr = pipeline.run_step("crawler", "20260101")
|
||||||
|
assert sr.success is True
|
||||||
|
assert "跳过" in sr.tail_msg
|
||||||
|
assert sr.elapsed_sec == 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_crawler_today_runs_normally(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
|
"""当天日期正常执行 crawler(不受补跑保护影响)。"""
|
||||||
|
from datetime import date
|
||||||
|
|
||||||
|
from scheduler import pipeline
|
||||||
|
|
||||||
|
called: list[str] = []
|
||||||
|
|
||||||
|
def _fake_run(cmd, timeout=None): # noqa: ARG001
|
||||||
|
called.append(" ".join(cmd))
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(pipeline.subprocess, "run", _fake_run)
|
||||||
|
sr = pipeline.run_step("crawler", date.today().strftime("%Y%m%d"))
|
||||||
|
assert sr.success is True
|
||||||
|
assert called and "run_crawler" in called[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_report_step_uses_date_str(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
|
"""report 步骤必须使用传入的 date_str,而非 date.today()(P0-4)。"""
|
||||||
|
from scheduler import pipeline
|
||||||
|
from scheduler import reporter
|
||||||
|
|
||||||
|
received: list[str] = []
|
||||||
|
|
||||||
|
def _fake_generate(day_str, upload=True): # noqa: ARG001
|
||||||
|
received.append(day_str)
|
||||||
|
return 12345
|
||||||
|
|
||||||
|
monkeypatch.setattr(reporter, "generate_report", _fake_generate)
|
||||||
|
sr = pipeline.run_step("report", "20260101")
|
||||||
|
assert sr.success is True
|
||||||
|
assert received == ["20260101"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_pipeline_backfill_skips_crawler_keeps_rest(
|
||||||
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch,
|
||||||
|
) -> None:
|
||||||
|
"""补跑历史日期:全链路中仅 crawler 跳过,其余步骤照常执行。"""
|
||||||
|
from scheduler import pipeline
|
||||||
|
|
||||||
|
calls: list[str] = []
|
||||||
|
|
||||||
|
def _fake_run(cmd, timeout=None): # noqa: ARG001
|
||||||
|
name = next(c.split(".")[-1] for c in cmd if "scripts.run_" in c)
|
||||||
|
calls.append(name)
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(pipeline.subprocess, "run", _fake_run)
|
||||||
|
state_path = tmp_path / "state.json"
|
||||||
|
result = pipeline.run_pipeline(
|
||||||
|
"20260101",
|
||||||
|
steps=["crawler", "extractor", "dedup"],
|
||||||
|
state_path=state_path,
|
||||||
|
)
|
||||||
|
# crawler 跳过(未执行子进程),extractor/dedup 正常
|
||||||
|
assert calls == ["run_extractor", "run_dedup"]
|
||||||
|
assert all(s.success for s in result.steps)
|
||||||
|
state = pipeline._load_pipeline_state(state_path)
|
||||||
|
assert state["20260101"]["crawler"]["status"] == "ok"
|
||||||
|
|||||||
+10
-3
@@ -56,9 +56,12 @@ def test_pipeline_result_partial_failure() -> None:
|
|||||||
# --------------------------------------------------------------------------- #
|
# --------------------------------------------------------------------------- #
|
||||||
|
|
||||||
def test_run_step_success() -> None:
|
def test_run_step_success() -> None:
|
||||||
|
# crawler 补跑保护:仅当天日期才执行子进程,故用今天的日期
|
||||||
|
from datetime import date
|
||||||
|
today = date.today().strftime("%Y%m%d")
|
||||||
with patch("subprocess.run", return_value=_mock_proc(returncode=0,
|
with patch("subprocess.run", return_value=_mock_proc(returncode=0,
|
||||||
stderr="INFO | 完成: 成功 20/20")):
|
stderr="INFO | 完成: 成功 20/20")):
|
||||||
sr = run_step("crawler", "20260616")
|
sr = run_step("crawler", today)
|
||||||
assert sr.success is True
|
assert sr.success is True
|
||||||
assert sr.exit_code == 0
|
assert sr.exit_code == 0
|
||||||
|
|
||||||
@@ -109,6 +112,8 @@ def test_run_pipeline_all_success() -> None:
|
|||||||
|
|
||||||
def test_run_pipeline_continues_on_failure() -> None:
|
def test_run_pipeline_continues_on_failure() -> None:
|
||||||
"""中间步骤失败,后续继续执行(不阻断)。"""
|
"""中间步骤失败,后续继续执行(不阻断)。"""
|
||||||
|
from datetime import date
|
||||||
|
today = date.today().strftime("%Y%m%d") # crawler 补跑保护:需当天日期才执行
|
||||||
call_count = {"n": 0}
|
call_count = {"n": 0}
|
||||||
|
|
||||||
def _side_effect(*args, **kwargs):
|
def _side_effect(*args, **kwargs):
|
||||||
@@ -118,7 +123,7 @@ def test_run_pipeline_continues_on_failure() -> None:
|
|||||||
return _mock_proc(returncode=0, stderr="ok")
|
return _mock_proc(returncode=0, stderr="ok")
|
||||||
|
|
||||||
with patch("subprocess.run", side_effect=_side_effect):
|
with patch("subprocess.run", side_effect=_side_effect):
|
||||||
result = run_pipeline("20260616", steps=["crawler", "extractor", "dedup", "llm"])
|
result = run_pipeline(today, steps=["crawler", "extractor", "dedup", "llm"])
|
||||||
assert len(result.steps) == 4
|
assert len(result.steps) == 4
|
||||||
# extractor 失败,但后续仍执行
|
# extractor 失败,但后续仍执行
|
||||||
assert result.steps[1].success is False
|
assert result.steps[1].success is False
|
||||||
@@ -126,7 +131,9 @@ def test_run_pipeline_continues_on_failure() -> None:
|
|||||||
|
|
||||||
|
|
||||||
def test_run_pipeline_custom_steps() -> None:
|
def test_run_pipeline_custom_steps() -> None:
|
||||||
|
from datetime import date
|
||||||
|
today = date.today().strftime("%Y%m%d")
|
||||||
with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="ok")):
|
with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="ok")):
|
||||||
result = run_pipeline("20260616", steps=["crawler", "extractor"])
|
result = run_pipeline(today, steps=["crawler", "extractor"])
|
||||||
assert len(result.steps) == 2
|
assert len(result.steps) == 2
|
||||||
assert result.all_success is True
|
assert result.all_success is True
|
||||||
|
|||||||
Reference in New Issue
Block a user