feat: 增量处理与 pipeline 断点续跑
- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
@@ -6,12 +6,16 @@
|
||||
data/events/{YYYYMMDD}/index.jsonl (扁平摘要)
|
||||
data/events/{YYYYMMDD}/failed.jsonl (失败列表)
|
||||
|
||||
增量: 默认跳过已抽取的文章(输出目录已有 {url_hash}.json 视为已处理),
|
||||
断点续跑/失败重试不会重复调用 LLM API;--force 强制全量重抽。
|
||||
|
||||
用法:
|
||||
uv run python -m scripts.run_event_extraction
|
||||
uv run python -m scripts.run_event_extraction --date 20260616
|
||||
uv run python -m scripts.run_event_extraction --provider qwen --model qwen-plus
|
||||
uv run python -m scripts.run_event_extraction --concurrency 5 --limit 10
|
||||
uv run python -m scripts.run_event_extraction --input-root data/processed --no-deduped
|
||||
uv run python -m scripts.run_event_extraction --force # 全量重抽
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -79,6 +83,24 @@ def _collect_inputs(
|
||||
return files
|
||||
|
||||
|
||||
def _filter_existing(files: list[Path], out_dir: Path) -> tuple[list[Path], int]:
|
||||
"""过滤掉已有产物(输出目录存在同名 {url_hash}.json)的输入。
|
||||
|
||||
输入文件名即 url_hash(如 {url_hash}.json),与 M4 产物命名一致。
|
||||
返回 (待处理文件, 跳过数);断点续跑/失败重试借此避免重复调用 LLM API。
|
||||
"""
|
||||
pending: list[Path] = []
|
||||
skipped = 0
|
||||
for fp in files:
|
||||
if (out_dir / f"{fp.stem}.json").exists():
|
||||
skipped += 1
|
||||
else:
|
||||
pending.append(fp)
|
||||
if skipped:
|
||||
logger.info("跳过已处理 {} 篇(产物已存在),待处理 {}", skipped, len(pending))
|
||||
return pending, skipped
|
||||
|
||||
|
||||
def _load_article(p: Path) -> Article | None:
|
||||
try:
|
||||
return Article.model_validate(json.loads(p.read_text(encoding="utf-8")))
|
||||
@@ -101,24 +123,40 @@ async def _run(args: argparse.Namespace) -> int:
|
||||
input_root = Path(args.input_root)
|
||||
use_deduped = not args.no_deduped
|
||||
files = _collect_inputs(input_root, args.date, use_deduped, args.source)
|
||||
if args.limit:
|
||||
files = files[: args.limit]
|
||||
if not files:
|
||||
logger.error(
|
||||
"{} 下未发现 {} 的文章(use_deduped={})",
|
||||
input_root, args.date, use_deduped,
|
||||
)
|
||||
return 2
|
||||
logger.info("待处理文章数: {}", len(files))
|
||||
|
||||
out_dir = Path(args.out_root) / args.date
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
# 增量:跳过已有产物(断点续跑/失败重试不重复调用 LLM API),--force 全量
|
||||
skipped = 0
|
||||
if not args.force:
|
||||
files, skipped = _filter_existing(files, out_dir)
|
||||
if args.limit:
|
||||
files = files[: args.limit]
|
||||
if not files:
|
||||
logger.info(
|
||||
"无待处理文章(全部已抽取,跳过 {} 篇),如需重抽请加 --force", skipped
|
||||
)
|
||||
return 0
|
||||
|
||||
logger.info("待处理文章数: {} (跳过已处理 {})", len(files), skipped)
|
||||
|
||||
index_path = out_dir / "index.jsonl"
|
||||
failed_path = out_dir / "failed.jsonl"
|
||||
# 重跑时清掉旧的 jsonl,避免重复追加
|
||||
for p in (index_path, failed_path):
|
||||
if p.exists():
|
||||
p.unlink()
|
||||
if args.force:
|
||||
# 全量模式:重建 index / failed
|
||||
for p in (index_path, failed_path):
|
||||
if p.exists():
|
||||
p.unlink()
|
||||
else:
|
||||
# 增量模式:index 累积追加;failed 只保留本次运行失败的
|
||||
if failed_path.exists():
|
||||
failed_path.unlink()
|
||||
|
||||
template = PromptTemplate(args.prompt)
|
||||
semaphore = asyncio.Semaphore(args.concurrency)
|
||||
@@ -209,6 +247,8 @@ def main() -> int:
|
||||
help="LLM 异步并发上限")
|
||||
parser.add_argument("--max-attempts", type=int, default=3,
|
||||
help="单篇文章最大重试次数")
|
||||
parser.add_argument("--force", action="store_true",
|
||||
help="强制全量重抽(默认跳过已抽取文章)")
|
||||
parser.add_argument("--limit", type=int, default=0,
|
||||
help="最多处理 N 篇,0=不限制(用于联调)")
|
||||
parser.add_argument("--prompt", default="prompts/event_extraction.md",
|
||||
|
||||
Reference in New Issue
Block a user