feat: 增量处理与 pipeline 断点续跑
- M2 run_extractor: 产物存在即跳过提取(仅回补 index),--force 全量;
增量成功率统计含跳过项,修复全跳过时误报失败
- M4 run_event_extraction: data/events/{day}/{url_hash}.json 已存在即跳过,
不重复调用 LLM API;--force 全量;failed 只留本次失败
- M5 run_embedding: data/embeddings/{day}/{url_hash}.json 已存在即跳过,
不重复调用 embed API;--force 全量
- scheduler/pipeline: 步骤结果按日期写入 data/pipeline/state.json(原子写),
run_pipeline(resume=True) 从首个失败/未执行步骤续跑
- run_scheduler + a-share CLI: --once --resume 断点续跑(--steps 互斥)
- 新增 tests/test_incremental.py 10 个测试(跳过逻辑 + resume)
- .gitignore: 忽略 data/pipeline/ 运行状态
This commit is contained in:
@@ -13,6 +13,9 @@
|
||||
data/embeddings/{day}/{url_hash}.json (含 vector 完整内容)
|
||||
data/embeddings/{day}/index.jsonl (扁平摘要,不含向量,便于检索/调试)
|
||||
data/embeddings/{day}/failed.jsonl (失败列表)
|
||||
|
||||
增量: 默认跳过已嵌入的文章(输出目录已有 {url_hash}.json 视为已处理),
|
||||
断点续跑/失败重试不会重复调用 embed API;--force 强制全量重嵌入。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -142,6 +145,26 @@ def _collect_inputs(args: argparse.Namespace) -> list[tuple[Path, str]]:
|
||||
return files
|
||||
|
||||
|
||||
def _filter_existing(
|
||||
files: list[tuple[Path, str]], out_dir: Path
|
||||
) -> tuple[list[tuple[Path, str]], int]:
|
||||
"""过滤掉已有产物(输出目录存在同名 {url_hash}.json)的输入。
|
||||
|
||||
输入与输出文件名均为 {url_hash}.json,直接比对 stem。
|
||||
返回 (待处理, 跳过数);断点续跑/失败重试借此避免重复调用 embed API。
|
||||
"""
|
||||
pending: list[tuple[Path, str]] = []
|
||||
skipped = 0
|
||||
for fp, kind in files:
|
||||
if (out_dir / f"{fp.stem}.json").exists():
|
||||
skipped += 1
|
||||
else:
|
||||
pending.append((fp, kind))
|
||||
if skipped:
|
||||
logger.info("跳过已嵌入 {} 篇(产物已存在),待处理 {}", skipped, len(pending))
|
||||
return pending, skipped
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# 主流程
|
||||
# --------------------------------------------------------------------------- #
|
||||
@@ -150,12 +173,24 @@ async def _run(args: argparse.Namespace) -> int:
|
||||
load_dotenv()
|
||||
|
||||
files = _collect_inputs(args)
|
||||
if args.limit:
|
||||
files = files[: args.limit]
|
||||
if not files:
|
||||
logger.error("未发现任何输入文件: {} ({})", args.input, args.date)
|
||||
return 2
|
||||
logger.info("待嵌入文章数: {} (input={})", len(files), args.input)
|
||||
|
||||
out_dir = Path(args.out_root) / args.date
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
# 增量:跳过已有产物(断点续跑/失败重试不重复调用 embed API),--force 全量
|
||||
skipped = 0
|
||||
if not args.force:
|
||||
files, skipped = _filter_existing(files, out_dir)
|
||||
if args.limit:
|
||||
files = files[: args.limit]
|
||||
if not files:
|
||||
logger.info(
|
||||
"无待嵌入文章(全部已处理,跳过 {} 篇),如需重新嵌入请加 --force", skipped
|
||||
)
|
||||
return 0
|
||||
logger.info("待嵌入文章数: {} (跳过已处理 {}; input={})", len(files), skipped, args.input)
|
||||
|
||||
# 准备每篇文本
|
||||
prepared: list[tuple[str, Article, str | None]] = []
|
||||
@@ -170,13 +205,17 @@ async def _run(args: argparse.Namespace) -> int:
|
||||
logger.error("所有输入文件均无法解析")
|
||||
return 2
|
||||
|
||||
out_dir = Path(args.out_root) / args.date
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
index_path = out_dir / "index.jsonl"
|
||||
failed_path = out_dir / "failed.jsonl"
|
||||
for p in (index_path, failed_path):
|
||||
if p.exists():
|
||||
p.unlink()
|
||||
if args.force:
|
||||
# 全量模式:重建 index / failed
|
||||
for p in (index_path, failed_path):
|
||||
if p.exists():
|
||||
p.unlink()
|
||||
else:
|
||||
# 增量模式:index 累积追加;failed 只保留本次运行失败的
|
||||
if failed_path.exists():
|
||||
failed_path.unlink()
|
||||
|
||||
started = time.time()
|
||||
succ_cnt = 0
|
||||
@@ -282,6 +321,8 @@ def main() -> int:
|
||||
help="每批送 embed 的条数(DashScope 上限 10)")
|
||||
parser.add_argument("--limit", type=int, default=0,
|
||||
help="最多处理 N 篇,0=不限")
|
||||
parser.add_argument("--force", action="store_true",
|
||||
help="强制全量重嵌入(默认跳过已嵌入文章)")
|
||||
parser.add_argument("--log-level", default="INFO")
|
||||
args = parser.parse_args()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user