refactor: 清理历史 AI Agent 文档残留 + 重构 docs/ + Pipeline 健壮性修复
- docs: 删除 CLAUDE.md / continuation.md / english-news-plan.md 及旧版 intlnews_usage.*,
统一迁移到 docs/{README,architecture,quickstart,usage,pipeline,configuration,deployment,development,faq}.md
- README: 精简为仓库入口,指向 docs/
- configs/sources.yaml: 更新注释指向新文档
- .env.example: 修正 DashScope Embedding 端点说明
Pipeline 修复:
- dedup/llm/embedding/vectorstore/reporter: 过滤 M2 no_content / 空正文,避免污染下游与 Qdrant
- dedup/pipeline: 改为先写唯一文件再写指纹,避免崩溃导致文章永久丢失
- crawler/orchestrator: sources_crawled 改为“尝试数”,成功数 = crawled - failed
- crawler/storage: write_index_jsonl 从文章路径推断日期,修复跨天/测试路径问题
- scheduler/pipeline: STEP_TIMEOUTS 实际生效(SIGALRM)
- scheduler/reporter: emb_count 排除 index.json;日报跳过无原文事件
- vectorstore/pipeline: payload 增加 source_ids;--recreate --all 时空日期也重建 collection
- app/cli: extract/dedup/translate/embed/index/pipeline 支持 --date;embed/index 支持 --all;crawl 全源失败返回非零
- scripts: domestic_full/crawl_8g/crawl_2g/pipeline 安全加载 .env;M1 全失败不标记且最终退出码=1
This commit is contained in:
+113
-9
@@ -56,6 +56,9 @@ def crawl(
|
||||
typer.echo(f"\n✅ 完成: {stats.sources_crawled} 源, {stats.total_articles} 篇文章")
|
||||
if stats.sources_failed:
|
||||
typer.echo(f"⚠️ {stats.sources_failed} 个源有错误")
|
||||
if stats.sources_crawled > 0 and stats.sources_failed >= stats.sources_crawled:
|
||||
typer.echo("❌ 所有源抓取失败", err=True)
|
||||
raise typer.Exit(code=1)
|
||||
except FileNotFoundError as e:
|
||||
typer.echo(f"❌ {e}", err=True)
|
||||
raise typer.Exit(code=1)
|
||||
@@ -71,12 +74,16 @@ def extract(
|
||||
None, "--source", "-s",
|
||||
help="只处理指定 source_id(不传则全部)",
|
||||
),
|
||||
date: str | None = typer.Option(
|
||||
None, "--date", "-d",
|
||||
help="YYYYMMDD,默认当前新闻日",
|
||||
),
|
||||
):
|
||||
"""M2: 英文正文提取(trafilatura)"""
|
||||
from extractor.pipeline import process_all_sources
|
||||
|
||||
try:
|
||||
stats = process_all_sources(source_filter=source)
|
||||
stats = process_all_sources(source_filter=source, date_str=date)
|
||||
typer.echo(
|
||||
f"\n✅ 提取: {stats['sources_processed']} 源, "
|
||||
f"{stats['total_articles']} 篇, {stats['elapsed_sec']:.1f}s"
|
||||
@@ -88,12 +95,17 @@ def extract(
|
||||
|
||||
|
||||
@app.command()
|
||||
def dedup():
|
||||
def dedup(
|
||||
date: str | None = typer.Option(
|
||||
None, "--date", "-d",
|
||||
help="YYYYMMDD,默认当前新闻日",
|
||||
),
|
||||
):
|
||||
"""M3: 三层去重"""
|
||||
from dedup.pipeline import dedup_all_sources
|
||||
|
||||
try:
|
||||
stats = dedup_all_sources()
|
||||
stats = dedup_all_sources(date_str=date)
|
||||
typer.echo(
|
||||
f"\n✅ 去重: {stats['sources_processed']} 源, "
|
||||
f"唯一 {stats['unique']} / 重复 {stats['duplicate']} / "
|
||||
@@ -106,12 +118,17 @@ def dedup():
|
||||
|
||||
|
||||
@app.command()
|
||||
def translate():
|
||||
def translate(
|
||||
date: str | None = typer.Option(
|
||||
None, "--date", "-d",
|
||||
help="YYYYMMDD,默认当前新闻日",
|
||||
),
|
||||
):
|
||||
"""M4: 全文翻译 + 投资事件抽取(LLM)"""
|
||||
from llm.pipeline import translate_all_deduped
|
||||
|
||||
try:
|
||||
stats = translate_all_deduped()
|
||||
stats = translate_all_deduped(date_str=date)
|
||||
typer.echo(
|
||||
f"\n✅ 翻译+事件抽取: {stats['success']}/{stats['total']} 篇, "
|
||||
f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})"
|
||||
@@ -123,12 +140,50 @@ def translate():
|
||||
|
||||
|
||||
@app.command()
|
||||
def embed():
|
||||
def embed(
|
||||
date: str | None = typer.Option(
|
||||
None, "--date", "-d",
|
||||
help="YYYYMMDD,默认当前新闻日",
|
||||
),
|
||||
all_dates: bool = typer.Option(
|
||||
False, "--all",
|
||||
help="处理 data/events 下所有日期(可用于历史回灌)",
|
||||
),
|
||||
):
|
||||
"""M5: 向量生成"""
|
||||
from pathlib import Path
|
||||
|
||||
from embedding.pipeline import embed_all_events
|
||||
|
||||
if all_dates and date:
|
||||
typer.echo("❌ --date 与 --all 不能同时使用", err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
try:
|
||||
stats = embed_all_events()
|
||||
if all_dates:
|
||||
dates = sorted(
|
||||
p.name for p in Path("data/events").iterdir()
|
||||
if p.is_dir() and p.name.isdigit()
|
||||
)
|
||||
if not dates:
|
||||
typer.echo("⚠️ data/events/ 下没有可处理日期")
|
||||
return
|
||||
total_success = 0
|
||||
total_count = 0
|
||||
for day in dates:
|
||||
stats = embed_all_events(date_str=day)
|
||||
total_success += stats["success"]
|
||||
total_count += stats["total"]
|
||||
typer.echo(
|
||||
f" [{day}] 向量生成: {stats['success']}/{stats['total']} 篇, "
|
||||
f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})"
|
||||
)
|
||||
typer.echo(
|
||||
f"\n✅ 全部日期向量生成: {total_success}/{total_count} 篇"
|
||||
)
|
||||
return
|
||||
|
||||
stats = embed_all_events(date_str=date)
|
||||
typer.echo(
|
||||
f"\n✅ 向量生成: {stats['success']}/{stats['total']} 篇, "
|
||||
f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})"
|
||||
@@ -145,12 +200,57 @@ def index(
|
||||
False, "--recreate",
|
||||
help="重建 collection(会删除已有数据)",
|
||||
),
|
||||
date: str | None = typer.Option(
|
||||
None, "--date", "-d",
|
||||
help="YYYYMMDD,默认当前新闻日",
|
||||
),
|
||||
all_dates: bool = typer.Option(
|
||||
False, "--all",
|
||||
help="处理 data/embeddings 下所有日期(可用于历史回灌)",
|
||||
),
|
||||
):
|
||||
"""M6: Qdrant 入库"""
|
||||
from pathlib import Path
|
||||
|
||||
from vectorstore.pipeline import get_collection_info, ingest_all_embeddings
|
||||
|
||||
if all_dates and date:
|
||||
typer.echo("❌ --date 与 --all 不能同时使用", err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
try:
|
||||
stats = ingest_all_embeddings(recreate=recreate)
|
||||
if all_dates:
|
||||
dates = sorted(
|
||||
p.name for p in Path("data/embeddings").iterdir()
|
||||
if p.is_dir() and p.name.isdigit()
|
||||
)
|
||||
if not dates:
|
||||
typer.echo("⚠️ data/embeddings/ 下没有可处理日期")
|
||||
return
|
||||
total_ingested = 0
|
||||
total_count = 0
|
||||
first = True
|
||||
for day in dates:
|
||||
# 全量回灌时只在第一次真正 recreate,避免后续清空已写数据
|
||||
stats = ingest_all_embeddings(
|
||||
date_str=day,
|
||||
recreate=recreate and first,
|
||||
)
|
||||
first = False
|
||||
total_ingested += stats["ingested"]
|
||||
total_count += stats["total"]
|
||||
typer.echo(
|
||||
f" [{day}] 入库: {stats['ingested']}/{stats['total']} 条, "
|
||||
f"{stats['elapsed_sec']:.1f}s"
|
||||
)
|
||||
info = get_collection_info()
|
||||
typer.echo(
|
||||
f"\n✅ 全部日期入库: {total_ingested}/{total_count} 条"
|
||||
)
|
||||
typer.echo(f"📊 Collection: {info['name']} — {info['vectors_count']} 条向量")
|
||||
return
|
||||
|
||||
stats = ingest_all_embeddings(date_str=date, recreate=recreate)
|
||||
typer.echo(
|
||||
f"\n✅ 入库: {stats['ingested']}/{stats['total']} 条, "
|
||||
f"{stats['elapsed_sec']:.1f}s"
|
||||
@@ -218,12 +318,16 @@ def pipeline(
|
||||
False, "--skip-report",
|
||||
help="跳过日报生成",
|
||||
),
|
||||
date: str | None = typer.Option(
|
||||
None, "--date", "-d",
|
||||
help="YYYYMMDD,默认当前新闻日",
|
||||
),
|
||||
):
|
||||
"""M7: 一键运行完整管道 M2→M6(+ 可选日报)"""
|
||||
from crawler.utils import get_news_day
|
||||
from scheduler.pipeline import run_pipeline
|
||||
|
||||
date_str = get_news_day()
|
||||
date_str = date or get_news_day()
|
||||
typer.echo(f"🚀 开始全链路管道,日期: {date_str}\n")
|
||||
|
||||
try:
|
||||
|
||||
Reference in New Issue
Block a user