feat: 增量处理与 pipeline 断点续跑

- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
  增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
  不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
  run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
2026-08-12 10:19:27 +08:00
parent c1a803968a
commit 2b4efea219
8 changed files with 518 additions and 35 deletions
+49 -8
View File
@@ -13,6 +13,9 @@
data/embeddings/{day}/{url_hash}.json (含 vector 完整内容)
data/embeddings/{day}/index.jsonl (扁平摘要,不含向量,便于检索/调试)
data/embeddings/{day}/failed.jsonl (失败列表)
增量: 默认跳过已嵌入的文章(输出目录已有 {url_hash}.json 视为已处理),
断点续跑/失败重试不会重复调用 embed API;--force 强制全量重嵌入。
"""
from __future__ import annotations
@@ -142,6 +145,26 @@ def _collect_inputs(args: argparse.Namespace) -> list[tuple[Path, str]]:
return files
def _filter_existing(
files: list[tuple[Path, str]], out_dir: Path
) -> tuple[list[tuple[Path, str]], int]:
"""过滤掉已有产物(输出目录存在同名 {url_hash}.json)的输入。
输入与输出文件名均为 {url_hash}.json,直接比对 stem。
返回 (待处理, 跳过数);断点续跑/失败重试借此避免重复调用 embed API。
"""
pending: list[tuple[Path, str]] = []
skipped = 0
for fp, kind in files:
if (out_dir / f"{fp.stem}.json").exists():
skipped += 1
else:
pending.append((fp, kind))
if skipped:
logger.info("跳过已嵌入 {} 篇(产物已存在),待处理 {}", skipped, len(pending))
return pending, skipped
# --------------------------------------------------------------------------- #
# 主流程
# --------------------------------------------------------------------------- #
@@ -150,12 +173,24 @@ async def _run(args: argparse.Namespace) -> int:
load_dotenv()
files = _collect_inputs(args)
if args.limit:
files = files[: args.limit]
if not files:
logger.error("未发现任何输入文件: {} ({})", args.input, args.date)
return 2
logger.info("待嵌入文章数: {} (input={})", len(files), args.input)
out_dir = Path(args.out_root) / args.date
out_dir.mkdir(parents=True, exist_ok=True)
# 增量:跳过已有产物(断点续跑/失败重试不重复调用 embed API),--force 全量
skipped = 0
if not args.force:
files, skipped = _filter_existing(files, out_dir)
if args.limit:
files = files[: args.limit]
if not files:
logger.info(
"无待嵌入文章(全部已处理,跳过 {} 篇),如需重新嵌入请加 --force", skipped
)
return 0
logger.info("待嵌入文章数: {} (跳过已处理 {}; input={})", len(files), skipped, args.input)
# 准备每篇文本
prepared: list[tuple[str, Article, str | None]] = []
@@ -170,13 +205,17 @@ async def _run(args: argparse.Namespace) -> int:
logger.error("所有输入文件均无法解析")
return 2
out_dir = Path(args.out_root) / args.date
out_dir.mkdir(parents=True, exist_ok=True)
index_path = out_dir / "index.jsonl"
failed_path = out_dir / "failed.jsonl"
for p in (index_path, failed_path):
if p.exists():
p.unlink()
if args.force:
# 全量模式:重建 index / failed
for p in (index_path, failed_path):
if p.exists():
p.unlink()
else:
# 增量模式:index 累积追加;failed 只保留本次运行失败的
if failed_path.exists():
failed_path.unlink()
started = time.time()
succ_cnt = 0
@@ -282,6 +321,8 @@ def main() -> int:
help="每批送 embed 的条数(DashScope 上限 10)")
parser.add_argument("--limit", type=int, default=0,
help="最多处理 N 篇,0=不限")
parser.add_argument("--force", action="store_true",
help="强制全量重嵌入(默认跳过已嵌入文章)")
parser.add_argument("--log-level", default="INFO")
args = parser.parse_args()