feat: 增量处理与 pipeline 断点续跑
- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
+86
-2
@@ -2,17 +2,27 @@
|
||||
|
||||
编排 M1→M6 全链路,每一步调用已有脚本。
|
||||
单步失败记录日志但不阻断后续(后续步骤可能使用旧缓存数据,降级继续)。
|
||||
|
||||
断点恢复:
|
||||
每次运行把各步骤结果写入 data/pipeline/state.json(按日期隔离);
|
||||
run_pipeline(resume=True) 时跳过连续成功的步骤,从第一个失败/未执行
|
||||
的步骤继续,实现 `pipeline --once --resume` 断点续跑。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import date, datetime
|
||||
from pathlib import Path
|
||||
|
||||
from loguru import logger
|
||||
|
||||
# 断点状态文件(按日期隔离,记录每步骤结果)
|
||||
DEFAULT_STATE_PATH = Path("data/pipeline/state.json")
|
||||
|
||||
# 步骤超时(秒)
|
||||
STEP_TIMEOUTS: dict[str, int] = {
|
||||
"crawler": 900, # M1 抓取(含 Playwright 浏览器,13 源约 8-12 min)
|
||||
@@ -64,6 +74,54 @@ class PipelineResult:
|
||||
return all(s.success for s in self.steps)
|
||||
|
||||
|
||||
def _load_pipeline_state(path: Path = DEFAULT_STATE_PATH) -> dict:
|
||||
"""读取断点状态文件;不存在或损坏时返回空 dict。"""
|
||||
if not path.is_file():
|
||||
return {}
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (json.JSONDecodeError, OSError) as e:
|
||||
logger.warning("pipeline 状态文件损坏,忽略: {} ({})", path, e)
|
||||
return {}
|
||||
return data if isinstance(data, dict) else {}
|
||||
|
||||
|
||||
def _save_pipeline_state(state: dict, path: Path = DEFAULT_STATE_PATH) -> None:
|
||||
"""原子写状态文件(tmp + rename,避免中断写坏)。"""
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".json.tmp")
|
||||
tmp.write_text(json.dumps(state, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
tmp.replace(path)
|
||||
|
||||
|
||||
def _update_step_state(state: dict, date_str: str, sr: StepResult) -> None:
|
||||
"""把单步结果写入状态(ok/failed,含退出码与耗时)。"""
|
||||
day_state = state.setdefault(date_str, {})
|
||||
day_state[sr.name] = {
|
||||
"status": "ok" if sr.success else "failed",
|
||||
"exit_code": sr.exit_code,
|
||||
"started_at": sr.started_at.isoformat() if sr.started_at else None,
|
||||
"elapsed_sec": round(sr.elapsed_sec, 1),
|
||||
}
|
||||
|
||||
|
||||
def _resume_start_index(
|
||||
names: list[str],
|
||||
date_str: str,
|
||||
state: dict,
|
||||
) -> int:
|
||||
"""计算断点续跑起始下标:跳过连续 ok 前缀,从首个失败/未记录步骤开始。
|
||||
|
||||
返回 0..len(names)-1;全部成功时返回 len(names)(表示无需续跑)。
|
||||
"""
|
||||
day_state = state.get(date_str, {})
|
||||
for i, name in enumerate(names):
|
||||
rec = day_state.get(name)
|
||||
if rec is None or rec.get("status") != "ok":
|
||||
return i
|
||||
return len(names)
|
||||
|
||||
|
||||
def run_step(name: str, date_str: str) -> StepResult:
|
||||
"""执行单个 pipeline 步骤。
|
||||
|
||||
@@ -158,19 +216,45 @@ def run_step(name: str, date_str: str) -> StepResult:
|
||||
tail_msg=str(e)[:200], started_at=started)
|
||||
|
||||
|
||||
def run_pipeline(date_str: str, *, steps: list[str] | None = None) -> PipelineResult:
|
||||
def run_pipeline(
|
||||
date_str: str,
|
||||
*,
|
||||
steps: list[str] | None = None,
|
||||
resume: bool = False,
|
||||
state_path: Path = DEFAULT_STATE_PATH,
|
||||
) -> PipelineResult:
|
||||
"""串联执行全链路(M1→M6)。
|
||||
|
||||
参数:
|
||||
date_str: YYYYMMDD。
|
||||
steps: 可选步骤列表,默认全部 6 步。
|
||||
resume: True 时断点续跑——读取 data/pipeline/state.json 中该日期的
|
||||
记录,跳过连续成功的步骤,从第一个失败/未执行步骤继续。
|
||||
state_path: 断点状态文件路径(测试可注入)。
|
||||
"""
|
||||
names = steps or [k for k in STEP_COMMANDS if k not in ("report", "cninfo_crawl", "cninfo_extract", "cninfo_pdf")]
|
||||
result = PipelineResult(started_at=datetime.now())
|
||||
|
||||
for name in names:
|
||||
state = _load_pipeline_state(state_path)
|
||||
start_idx = 0
|
||||
if resume:
|
||||
start_idx = _resume_start_index(names, date_str, state)
|
||||
if start_idx >= len(names):
|
||||
logger.info("resume: {} 的所有步骤均已完成,无需续跑", date_str)
|
||||
result.finished_at = datetime.now()
|
||||
return result
|
||||
logger.info(
|
||||
"resume: 从步骤 {} 继续{}",
|
||||
names[start_idx],
|
||||
f" (跳过已成功 {names[:start_idx]})" if start_idx > 0 else "",
|
||||
)
|
||||
|
||||
for name in names[start_idx:]:
|
||||
sr = run_step(name, date_str)
|
||||
result.steps.append(sr)
|
||||
# 记录断点状态(无论成败,便于下次 resume)
|
||||
_update_step_state(state, date_str, sr)
|
||||
_save_pipeline_state(state, state_path)
|
||||
if not sr.success:
|
||||
logger.warning("步骤 {} 失败,后续步骤继续(可能降级)", name)
|
||||
# 步间留一点缓冲
|
||||
|
||||
Reference in New Issue
Block a user