feat: 增量处理与 pipeline 断点续跑
- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
+51
-16
@@ -14,12 +14,14 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import contextlib
|
||||
import json
|
||||
import sys
|
||||
from datetime import date, datetime
|
||||
from pathlib import Path
|
||||
|
||||
from loguru import logger
|
||||
from pydantic import ValidationError
|
||||
|
||||
from extractor import Article, ExtractError, extract_article
|
||||
from extractor.parser import _url_hash
|
||||
@@ -238,16 +240,21 @@ def _process_cninfo_v2(rec: dict, json_path: Path, out_dir: Path) -> Article | N
|
||||
return _save_article(article, out_dir)
|
||||
|
||||
|
||||
def _append_index(article: Article, out_dir: Path) -> None:
|
||||
"""把 Article 的扁平摘要追加到 index.jsonl。"""
|
||||
flat = article.model_dump(exclude={"content", "images"}, mode="json")
|
||||
flat["article_file"] = f"{article.url_hash}.json"
|
||||
flat["content_preview"] = article.content[:80]
|
||||
with (out_dir / "index.jsonl").open("a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(flat, ensure_ascii=False) + "\n")
|
||||
|
||||
|
||||
def _save_article(article: Article, out_dir: Path) -> Article:
|
||||
"""保存 Article JSON 并追加 index。"""
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
article_path = out_dir / f"{article.url_hash}.json"
|
||||
article_path.write_text(article.model_dump_json(indent=2), encoding="utf-8")
|
||||
flat = article.model_dump(exclude={"content", "images"}, mode="json")
|
||||
flat["article_file"] = article_path.name
|
||||
flat["content_preview"] = article.content[:80]
|
||||
with (out_dir / "index.jsonl").open("a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(flat, ensure_ascii=False) + "\n")
|
||||
_append_index(article, out_dir)
|
||||
return article
|
||||
|
||||
|
||||
@@ -257,37 +264,57 @@ def _process_source_day(
|
||||
raw_root: Path,
|
||||
out_root: Path,
|
||||
body_xpath_map: dict[str, str] | None = None,
|
||||
) -> tuple[int, int]:
|
||||
"""处理单个源单日。返回 (成功数, 总数)。"""
|
||||
*,
|
||||
force: bool = False,
|
||||
) -> tuple[int, int, int]:
|
||||
"""处理单个源单日。返回 (成功数, 总数, 跳过数)。
|
||||
|
||||
默认增量:已提取的文章(输出目录已有 {url_hash}.json)跳过提取,仅回补 index 行;
|
||||
force=True 时全量重提取并重建 index。
|
||||
"""
|
||||
raw_dir = raw_root / source_id / day
|
||||
out_dir = out_root / source_id / day
|
||||
|
||||
records = _iter_article_records(raw_dir)
|
||||
if not records:
|
||||
logger.info("源 {} 日期 {} 无可处理记录", source_id, day)
|
||||
return 0, 0
|
||||
return 0, 0, 0
|
||||
|
||||
# 清理同日旧的 index.jsonl,避免重复追加
|
||||
# 全量模式:重建 index;增量模式:保留旧 index 追加新条目
|
||||
old_index = out_dir / "index.jsonl"
|
||||
if old_index.exists():
|
||||
if force and old_index.exists():
|
||||
old_index.unlink()
|
||||
|
||||
succ = 0
|
||||
skipped = 0
|
||||
for rec in records:
|
||||
url_hash = rec.get("url_hash") or _url_hash(rec.get("url") or "")
|
||||
existing = out_dir / f"{url_hash}.json"
|
||||
if not force and existing.is_file():
|
||||
# 增量:跳过已提取,回补 index 行保持摘要完整
|
||||
skipped += 1
|
||||
with contextlib.suppress(json.JSONDecodeError, ValidationError, OSError):
|
||||
_append_index(
|
||||
Article.model_validate(json.loads(existing.read_text(encoding="utf-8"))),
|
||||
out_dir,
|
||||
)
|
||||
continue
|
||||
article = _process_one(rec, raw_dir, out_dir, body_xpath_map)
|
||||
if article is not None:
|
||||
succ += 1
|
||||
total = len(records)
|
||||
rate = succ / max(total, 1)
|
||||
# 增量模式下「跳过已提取」视为已成功处理,避免全跳过时误报成功率 0%
|
||||
rate = (succ + skipped) / max(total, 1)
|
||||
logger.info(
|
||||
"源 {} 日期 {} 提取完成: {}/{} 成功率 {:.0%}",
|
||||
"源 {} 日期 {} 提取完成: {}/{} 成功率 {:.0%} (跳过已提取 {})",
|
||||
source_id,
|
||||
day,
|
||||
succ,
|
||||
total,
|
||||
rate,
|
||||
skipped,
|
||||
)
|
||||
return succ, total
|
||||
return succ, total, skipped
|
||||
|
||||
|
||||
def _list_source_dirs(raw_root: Path) -> list[str]:
|
||||
@@ -307,6 +334,8 @@ def main() -> int:
|
||||
default=date.today().strftime("%Y%m%d"),
|
||||
help="处理日期 YYYYMMDD,默认今日",
|
||||
)
|
||||
parser.add_argument("--force", action="store_true",
|
||||
help="强制全量重提取(默认跳过已提取文章)")
|
||||
parser.add_argument("--log-level", default="INFO")
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -335,18 +364,24 @@ def main() -> int:
|
||||
started = datetime.now()
|
||||
total_succ = 0
|
||||
total_all = 0
|
||||
total_skipped = 0
|
||||
for src in sources:
|
||||
succ, total = _process_source_day(src, args.date, raw_root, out_root, body_xpath_map)
|
||||
succ, total, skipped = _process_source_day(
|
||||
src, args.date, raw_root, out_root, body_xpath_map, force=args.force
|
||||
)
|
||||
total_succ += succ
|
||||
total_all += total
|
||||
total_skipped += skipped
|
||||
|
||||
elapsed = (datetime.now() - started).total_seconds()
|
||||
rate = total_succ / max(total_all, 1)
|
||||
# 增量模式下跳过已提取视为成功
|
||||
rate = (total_succ + total_skipped) / max(total_all, 1)
|
||||
logger.info(
|
||||
"全部完成: {}/{} 成功率 {:.0%} 用时 {:.1f}s",
|
||||
"全部完成: {}/{} 成功率 {:.0%} (跳过已提取 {}) 用时 {:.1f}s",
|
||||
total_succ,
|
||||
total_all,
|
||||
rate,
|
||||
total_skipped,
|
||||
elapsed,
|
||||
)
|
||||
return 0 if rate >= 0.9 or total_all == 0 else 1
|
||||
|
||||
Reference in New Issue
Block a user