feat: 增量处理与 pipeline 断点续跑

- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
  增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
  run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
2026-08-12 10:19:27 +08:00
parent c1a803968a
commit 2b4efea219
8 changed files with 518 additions and 35 deletions
+51 -16
View File
@@ -14,12 +14,14 @@
from __future__ import annotations
import argparse
import contextlib
import json
import sys
from datetime import date, datetime
from pathlib import Path
from loguru import logger
from pydantic import ValidationError
from extractor import Article, ExtractError, extract_article
from extractor.parser import _url_hash
@@ -238,16 +240,21 @@ def _process_cninfo_v2(rec: dict, json_path: Path, out_dir: Path) -> Article | N
return _save_article(article, out_dir)
def _append_index(article: Article, out_dir: Path) -> None:
"""把 Article 的扁平摘要追加到 index.jsonl。"""
flat = article.model_dump(exclude={"content", "images"}, mode="json")
flat["article_file"] = f"{article.url_hash}.json"
flat["content_preview"] = article.content[:80]
with (out_dir / "index.jsonl").open("a", encoding="utf-8") as f:
f.write(json.dumps(flat, ensure_ascii=False) + "\n")
def _save_article(article: Article, out_dir: Path) -> Article:
"""保存 Article JSON 并追加 index。"""
out_dir.mkdir(parents=True, exist_ok=True)
article_path = out_dir / f"{article.url_hash}.json"
article_path.write_text(article.model_dump_json(indent=2), encoding="utf-8")
flat = article.model_dump(exclude={"content", "images"}, mode="json")
flat["article_file"] = article_path.name
flat["content_preview"] = article.content[:80]
with (out_dir / "index.jsonl").open("a", encoding="utf-8") as f:
f.write(json.dumps(flat, ensure_ascii=False) + "\n")
_append_index(article, out_dir)
return article
@@ -257,37 +264,57 @@ def _process_source_day(
raw_root: Path,
out_root: Path,
body_xpath_map: dict[str, str] | None = None,
) -> tuple[int, int]:
"""处理单个源单日。返回 (成功数, 总数)。"""
*,
force: bool = False,
) -> tuple[int, int, int]:
"""处理单个源单日。返回 (成功数, 总数, 跳过数)。
默认增量:已提取的文章(输出目录已有 {url_hash}.json)跳过提取,仅回补 index 行;
force=True 时全量重提取并重建 index。
"""
raw_dir = raw_root / source_id / day
out_dir = out_root / source_id / day
records = _iter_article_records(raw_dir)
if not records:
logger.info("源 {} 日期 {} 无可处理记录", source_id, day)
return 0, 0
return 0, 0, 0
# 清理同日旧的 index.jsonl,避免重复追加
# 全量模式:重建 index;增量模式:保留旧 index 追加新条目
old_index = out_dir / "index.jsonl"
if old_index.exists():
if force and old_index.exists():
old_index.unlink()
succ = 0
skipped = 0
for rec in records:
url_hash = rec.get("url_hash") or _url_hash(rec.get("url") or "")
existing = out_dir / f"{url_hash}.json"
if not force and existing.is_file():
# 增量:跳过已提取,回补 index 行保持摘要完整
skipped += 1
with contextlib.suppress(json.JSONDecodeError, ValidationError, OSError):
_append_index(
Article.model_validate(json.loads(existing.read_text(encoding="utf-8"))),
out_dir,
)
continue
article = _process_one(rec, raw_dir, out_dir, body_xpath_map)
if article is not None:
succ += 1
total = len(records)
rate = succ / max(total, 1)
# 增量模式下「跳过已提取」视为已成功处理,避免全跳过时误报成功率 0%
rate = (succ + skipped) / max(total, 1)
logger.info(
"源 {} 日期 {} 提取完成: {}/{} 成功率 {:.0%}",
"源 {} 日期 {} 提取完成: {}/{} 成功率 {:.0%} (跳过已提取 {})",
source_id,
day,
succ,
total,
rate,
skipped,
)
return succ, total
return succ, total, skipped
def _list_source_dirs(raw_root: Path) -> list[str]:
@@ -307,6 +334,8 @@ def main() -> int:
default=date.today().strftime("%Y%m%d"),
help="处理日期 YYYYMMDD,默认今日",
)
parser.add_argument("--force", action="store_true",
help="强制全量重提取(默认跳过已提取文章)")
parser.add_argument("--log-level", default="INFO")
args = parser.parse_args()
@@ -335,18 +364,24 @@ def main() -> int:
started = datetime.now()
total_succ = 0
total_all = 0
total_skipped = 0
for src in sources:
succ, total = _process_source_day(src, args.date, raw_root, out_root, body_xpath_map)
succ, total, skipped = _process_source_day(
src, args.date, raw_root, out_root, body_xpath_map, force=args.force
)
total_succ += succ
total_all += total
total_skipped += skipped
elapsed = (datetime.now() - started).total_seconds()
rate = total_succ / max(total_all, 1)
# 增量模式下跳过已提取视为成功
rate = (total_succ + total_skipped) / max(total_all, 1)
logger.info(
"全部完成: {}/{} 成功率 {:.0%} 用时 {:.1f}s",
"全部完成: {}/{} 成功率 {:.0%} (跳过已提取 {}) 用时 {:.1f}s",
total_succ,
total_all,
rate,
total_skipped,
elapsed,
)
return 0 if rate >= 0.9 or total_all == 0 else 1