feat: 增量处理与 pipeline 断点续跑

- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
  增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
  run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
2026-08-12 10:19:27 +08:00
parent c1a803968a
commit 2b4efea219
8 changed files with 518 additions and 35 deletions
+86 -2
View File
@@ -2,17 +2,27 @@
编排 M1→M6 全链路,每一步调用已有脚本。
单步失败记录日志但不阻断后续(后续步骤可能使用旧缓存数据,降级继续)。
断点恢复:
每次运行把各步骤结果写入 data/pipeline/state.json(按日期隔离);
run_pipeline(resume=True) 时跳过连续成功的步骤,从第一个失败/未执行
的步骤继续,实现 `pipeline --once --resume` 断点续跑。
"""
from __future__ import annotations
import json
import subprocess
import time
from dataclasses import dataclass, field
from datetime import date, datetime
from pathlib import Path
from loguru import logger
# 断点状态文件(按日期隔离,记录每步骤结果)
DEFAULT_STATE_PATH = Path("data/pipeline/state.json")
# 步骤超时(秒)
STEP_TIMEOUTS: dict[str, int] = {
"crawler": 900, # M1 抓取(含 Playwright 浏览器,13 源约 8-12 min)
@@ -64,6 +74,54 @@ class PipelineResult:
return all(s.success for s in self.steps)
def _load_pipeline_state(path: Path = DEFAULT_STATE_PATH) -> dict:
"""读取断点状态文件;不存在或损坏时返回空 dict。"""
if not path.is_file():
return {}
try:
data = json.loads(path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError) as e:
logger.warning("pipeline 状态文件损坏,忽略: {} ({})", path, e)
return {}
return data if isinstance(data, dict) else {}
def _save_pipeline_state(state: dict, path: Path = DEFAULT_STATE_PATH) -> None:
"""原子写状态文件(tmp + rename,避免中断写坏)。"""
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".json.tmp")
tmp.write_text(json.dumps(state, ensure_ascii=False, indent=2), encoding="utf-8")
tmp.replace(path)
def _update_step_state(state: dict, date_str: str, sr: StepResult) -> None:
"""把单步结果写入状态(ok/failed,含退出码与耗时)。"""
day_state = state.setdefault(date_str, {})
day_state[sr.name] = {
"status": "ok" if sr.success else "failed",
"exit_code": sr.exit_code,
"started_at": sr.started_at.isoformat() if sr.started_at else None,
"elapsed_sec": round(sr.elapsed_sec, 1),
}
def _resume_start_index(
names: list[str],
date_str: str,
state: dict,
) -> int:
"""计算断点续跑起始下标:跳过连续 ok 前缀,从首个失败/未记录步骤开始。
返回 0..len(names)-1;全部成功时返回 len(names)(表示无需续跑)。
"""
day_state = state.get(date_str, {})
for i, name in enumerate(names):
rec = day_state.get(name)
if rec is None or rec.get("status") != "ok":
return i
return len(names)
def run_step(name: str, date_str: str) -> StepResult:
"""执行单个 pipeline 步骤。
@@ -158,19 +216,45 @@ def run_step(name: str, date_str: str) -> StepResult:
tail_msg=str(e)[:200], started_at=started)
def run_pipeline(date_str: str, *, steps: list[str] | None = None) -> PipelineResult:
def run_pipeline(
date_str: str,
*,
steps: list[str] | None = None,
resume: bool = False,
state_path: Path = DEFAULT_STATE_PATH,
) -> PipelineResult:
"""串联执行全链路(M1→M6)。
参数:
date_str: YYYYMMDD。
steps: 可选步骤列表,默认全部 6 步。
resume: True 时断点续跑——读取 data/pipeline/state.json 中该日期的
记录,跳过连续成功的步骤,从第一个失败/未执行步骤继续。
state_path: 断点状态文件路径(测试可注入)。
"""
names = steps or [k for k in STEP_COMMANDS if k not in ("report", "cninfo_crawl", "cninfo_extract", "cninfo_pdf")]
result = PipelineResult(started_at=datetime.now())
for name in names:
state = _load_pipeline_state(state_path)
start_idx = 0
if resume:
start_idx = _resume_start_index(names, date_str, state)
if start_idx >= len(names):
logger.info("resume: {} 的所有步骤均已完成,无需续跑", date_str)
result.finished_at = datetime.now()
return result
logger.info(
"resume: 从步骤 {} 继续{}",
names[start_idx],
f" (跳过已成功 {names[:start_idx]})" if start_idx > 0 else "",
)
for name in names[start_idx:]:
sr = run_step(name, date_str)
result.steps.append(sr)
# 记录断点状态(无论成败,便于下次 resume)
_update_step_state(state, date_str, sr)
_save_pipeline_state(state, state_path)
if not sr.success:
logger.warning("步骤 {} 失败,后续步骤继续(可能降级)", name)
# 步间留一点缓冲