feat: 增量处理与 pipeline 断点续跑

- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
  增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
  run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
2026-08-12 10:19:27 +08:00
parent c1a803968a
commit 2b4efea219
8 changed files with 518 additions and 35 deletions
+47 -7
View File
@@ -6,12 +6,16 @@
data/events/{YYYYMMDD}/index.jsonl (扁平摘要)
data/events/{YYYYMMDD}/failed.jsonl (失败列表)
增量: 默认跳过已抽取的文章(输出目录已有 {url_hash}.json 视为已处理),
断点续跑/失败重试不会重复调用 LLM API;--force 强制全量重抽。
用法:
uv run python -m scripts.run_event_extraction
uv run python -m scripts.run_event_extraction --date 20260616
uv run python -m scripts.run_event_extraction --provider qwen --model qwen-plus
uv run python -m scripts.run_event_extraction --concurrency 5 --limit 10
uv run python -m scripts.run_event_extraction --input-root data/processed --no-deduped
uv run python -m scripts.run_event_extraction --force # 全量重抽
"""
from __future__ import annotations
@@ -79,6 +83,24 @@ def _collect_inputs(
return files
def _filter_existing(files: list[Path], out_dir: Path) -> tuple[list[Path], int]:
"""过滤掉已有产物(输出目录存在同名 {url_hash}.json)的输入。
输入文件名即 url_hash(如 {url_hash}.json),与 M4 产物命名一致。
返回 (待处理文件, 跳过数);断点续跑/失败重试借此避免重复调用 LLM API。
"""
pending: list[Path] = []
skipped = 0
for fp in files:
if (out_dir / f"{fp.stem}.json").exists():
skipped += 1
else:
pending.append(fp)
if skipped:
logger.info("跳过已处理 {} 篇(产物已存在),待处理 {}", skipped, len(pending))
return pending, skipped
def _load_article(p: Path) -> Article | None:
try:
return Article.model_validate(json.loads(p.read_text(encoding="utf-8")))
@@ -101,24 +123,40 @@ async def _run(args: argparse.Namespace) -> int:
input_root = Path(args.input_root)
use_deduped = not args.no_deduped
files = _collect_inputs(input_root, args.date, use_deduped, args.source)
if args.limit:
files = files[: args.limit]
if not files:
logger.error(
"{} 下未发现 {} 的文章(use_deduped={})",
input_root, args.date, use_deduped,
)
return 2
logger.info("待处理文章数: {}", len(files))
out_dir = Path(args.out_root) / args.date
out_dir.mkdir(parents=True, exist_ok=True)
# 增量:跳过已有产物(断点续跑/失败重试不重复调用 LLM API),--force 全量
skipped = 0
if not args.force:
files, skipped = _filter_existing(files, out_dir)
if args.limit:
files = files[: args.limit]
if not files:
logger.info(
"无待处理文章(全部已抽取,跳过 {} 篇),如需重抽请加 --force", skipped
)
return 0
logger.info("待处理文章数: {} (跳过已处理 {})", len(files), skipped)
index_path = out_dir / "index.jsonl"
failed_path = out_dir / "failed.jsonl"
# 重跑时清掉旧的 jsonl,避免重复追加
for p in (index_path, failed_path):
if p.exists():
p.unlink()
if args.force:
# 全量模式:重建 index / failed
for p in (index_path, failed_path):
if p.exists():
p.unlink()
else:
# 增量模式:index 累积追加;failed 只保留本次运行失败的
if failed_path.exists():
failed_path.unlink()
template = PromptTemplate(args.prompt)
semaphore = asyncio.Semaphore(args.concurrency)
@@ -209,6 +247,8 @@ def main() -> int:
help="LLM 异步并发上限")
parser.add_argument("--max-attempts", type=int, default=3,
help="单篇文章最大重试次数")
parser.add_argument("--force", action="store_true",
help="强制全量重抽(默认跳过已抽取文章)")
parser.add_argument("--limit", type=int, default=0,
help="最多处理 N 篇,0=不限制(用于联调)")
parser.add_argument("--prompt", default="prompts/event_extraction.md",