Files
intl_news/app/cli.py
T
simon 9e4f5b4c75 fix: 去重合并多来源后同步到 events/Qdrant/MySQL,新增 sync-sources 回填命令
- dedup/pipeline: 合并 source_ids 后自动同步已生成的 events JSON 和 Qdrant payload
- 新增 sync_sources_from_deduped / uv run en-news sync-sources,用于历史数据回填
- sync-sources 会扫描 deduped 多来源唯一篇,更新 events、Qdrant、MySQL news_event.sources
- docs: 补充 sync-sources 使用说明
2026-08-23 10:55:23 +08:00

381 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""English Financial News — CLI 入口"""
import logging
import sys
import typer
from dotenv import load_dotenv
# 加载 .env 环境变量(必须在所有模块导入之前)
load_dotenv()
app = typer.Typer(
name="en-news",
help="国际财经新闻抓取与深度研究平台",
no_args_is_help=True,
)
# ── 日志配置 ──────────────────────────────────────────
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
stream=sys.stderr,
)
# 抑制第三方库噪音日志
for _lib in ("httpx", "httpcore", "openai", "urllib3"):
logging.getLogger(_lib).setLevel(logging.WARNING)
logger = logging.getLogger("en-news")
@app.command()
def crawl(
source: str | None = typer.Option(
None, "--source", "-s",
help="只抓取指定 source_id(不传则全部)",
),
profile: str | None = typer.Option(
None, "--profile", "-p",
help="Profile 名称(2g_headless / 8g_headful)",
),
):
"""M1: 抓取英文财经新闻(部署海外)"""
import os
from crawler.orchestrator import run_crawl_sync
if profile:
os.environ["EN_NEWS_PROFILE"] = profile
logger.info("📋 使用 Profile: %s", profile)
try:
stats = run_crawl_sync(source_filter=source)
typer.echo(f"\n✅ 完成: {stats.sources_crawled} 源, {stats.total_articles} 篇文章")
if stats.sources_failed:
typer.echo(f"⚠️ {stats.sources_failed} 个源有错误")
if stats.sources_crawled > 0 and stats.sources_failed >= stats.sources_crawled:
typer.echo("❌ 所有源抓取失败", err=True)
raise typer.Exit(code=1)
except FileNotFoundError as e:
typer.echo(f"❌ {e}", err=True)
raise typer.Exit(code=1)
except Exception as e:
logger.exception("抓取失败")
typer.echo(f"❌ 抓取出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def extract(
source: str | None = typer.Option(
None, "--source", "-s",
help="只处理指定 source_id(不传则全部)",
),
date: str | None = typer.Option(
None, "--date", "-d",
help="YYYYMMDD,默认当前新闻日",
),
):
"""M2: 英文正文提取(trafilatura)"""
from extractor.pipeline import process_all_sources
try:
stats = process_all_sources(source_filter=source, date_str=date)
typer.echo(
f"\n✅ 提取: {stats['sources_processed']} 源, "
f"{stats['total_articles']} 篇, {stats['elapsed_sec']:.1f}s"
)
except Exception as e:
logger.exception("正文提取失败")
typer.echo(f"❌ 提取出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def dedup(
date: str | None = typer.Option(
None, "--date", "-d",
help="YYYYMMDD,默认当前新闻日",
),
):
"""M3: 三层去重"""
from dedup.pipeline import dedup_all_sources
try:
stats = dedup_all_sources(date_str=date)
typer.echo(
f"\n✅ 去重: {stats['sources_processed']} 源, "
f"唯一 {stats['unique']} / 重复 {stats['duplicate']} / "
f"总计 {stats['total_articles']}, {stats['elapsed_sec']:.1f}s"
)
except Exception as e:
logger.exception("去重失败")
typer.echo(f"❌ 去重出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def translate(
date: str | None = typer.Option(
None, "--date", "-d",
help="YYYYMMDD,默认当前新闻日",
),
):
"""M4: 全文翻译 + 投资事件抽取(LLM)"""
from llm.pipeline import translate_all_deduped
try:
stats = translate_all_deduped(date_str=date)
typer.echo(
f"\n✅ 翻译+事件抽取: {stats['success']}/{stats['total']} 篇, "
f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})"
)
except Exception as e:
logger.exception("翻译失败")
typer.echo(f"❌ 翻译出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def embed(
date: str | None = typer.Option(
None, "--date", "-d",
help="YYYYMMDD,默认当前新闻日",
),
all_dates: bool = typer.Option(
False, "--all",
help="处理 data/events 下所有日期(可用于历史回灌)",
),
):
"""M5: 向量生成"""
from pathlib import Path
from embedding.pipeline import embed_all_events
if all_dates and date:
typer.echo("❌ --date 与 --all 不能同时使用", err=True)
raise typer.Exit(code=1)
try:
if all_dates:
dates = sorted(
p.name for p in Path("data/events").iterdir()
if p.is_dir() and p.name.isdigit()
)
if not dates:
typer.echo("⚠️ data/events/ 下没有可处理日期")
return
total_success = 0
total_count = 0
for day in dates:
stats = embed_all_events(date_str=day)
total_success += stats["success"]
total_count += stats["total"]
typer.echo(
f" [{day}] 向量生成: {stats['success']}/{stats['total']} 篇, "
f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})"
)
typer.echo(
f"\n✅ 全部日期向量生成: {total_success}/{total_count} 篇"
)
return
stats = embed_all_events(date_str=date)
typer.echo(
f"\n✅ 向量生成: {stats['success']}/{stats['total']} 篇, "
f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})"
)
except Exception as e:
logger.exception("向量生成失败")
typer.echo(f"❌ 向量生成出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def index(
recreate: bool = typer.Option(
False, "--recreate",
help="重建 collection(会删除已有数据)",
),
date: str | None = typer.Option(
None, "--date", "-d",
help="YYYYMMDD,默认当前新闻日",
),
all_dates: bool = typer.Option(
False, "--all",
help="处理 data/embeddings 下所有日期(可用于历史回灌)",
),
):
"""M6: Qdrant 入库"""
from pathlib import Path
from vectorstore.pipeline import get_collection_info, ingest_all_embeddings
if all_dates and date:
typer.echo("❌ --date 与 --all 不能同时使用", err=True)
raise typer.Exit(code=1)
try:
if all_dates:
dates = sorted(
p.name for p in Path("data/embeddings").iterdir()
if p.is_dir() and p.name.isdigit()
)
if not dates:
typer.echo("⚠️ data/embeddings/ 下没有可处理日期")
return
total_ingested = 0
total_count = 0
first = True
for day in dates:
# 全量回灌时只在第一次真正 recreate,避免后续清空已写数据
stats = ingest_all_embeddings(
date_str=day,
recreate=recreate and first,
)
first = False
total_ingested += stats["ingested"]
total_count += stats["total"]
typer.echo(
f" [{day}] 入库: {stats['ingested']}/{stats['total']} 条, "
f"{stats['elapsed_sec']:.1f}s"
)
info = get_collection_info()
typer.echo(
f"\n✅ 全部日期入库: {total_ingested}/{total_count} 条"
)
typer.echo(f"📊 Collection: {info['name']} — {info['vectors_count']} 条向量")
return
stats = ingest_all_embeddings(date_str=date, recreate=recreate)
typer.echo(
f"\n✅ 入库: {stats['ingested']}/{stats['total']} 条, "
f"{stats['elapsed_sec']:.1f}s"
)
info = get_collection_info()
typer.echo(f"📊 Collection: {info['name']} — {info['vectors_count']} 条向量")
except Exception as e:
logger.exception("入库失败")
typer.echo(f"❌ 入库出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def search(
query: str = typer.Argument(..., help="中文检索查询"),
top_k: int = typer.Option(10, help="返回结果数"),
):
"""M6: Qdrant 语义检索"""
from vectorstore.pipeline import search_news
try:
results = search_news(query, top_k=top_k)
if not results:
typer.echo("🔍 未找到相关结果")
return
typer.echo(f"\n🔍 搜索: {query}\n")
for i, r in enumerate(results, 1):
typer.echo(f"{i}. [{r.source_id}] score={r.score:.4f}")
typer.echo(f" 📰 {r.title_zh or r.title}")
if r.events:
events_summary = ", ".join(
f"{ev.get('event_type','')}({ev.get('sentiment','')})"
for ev in r.events[:3]
)
typer.echo(f" 📌 事件: {events_summary}")
typer.echo(f" 🔗 {r.url}")
typer.echo()
except Exception as e:
logger.exception("搜索失败")
typer.echo(f"❌ 搜索出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def report():
"""M7: 日报生成(结构化入库,不再产出 HTML)"""
from scheduler.reporter import generate_report
try:
report_id = generate_report()
if report_id is not None:
typer.echo(f"\n✅ 日报已入库: report_id={report_id}")
else:
typer.echo("⚠️ 无数据或入库失败,跳过日报生成")
except Exception as e:
logger.exception("日报生成失败")
typer.echo(f"❌ 日报生成出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def sync_sources():
"""M3/M9 辅助:把 deduped 已合并的多来源同步到 events/Qdrant/MySQL"""
from dedup.pipeline import sync_sources_from_deduped
try:
stats = sync_sources_from_deduped()
typer.echo(
f"\n✅ 来源同步完成: 扫描 {stats['deduped_scanned']} 个唯一篇, "
f"多来源 {stats['multi_source']} 个, "
f"events 更新 {stats['events_updated']}, "
f"Qdrant 更新 {stats['qdrant_updated']}, "
f"MySQL 更新 {stats['db_updated']}"
)
except Exception as e:
logger.exception("来源同步失败")
typer.echo(f"❌ 来源同步出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def pipeline(
skip_report: bool = typer.Option(
False, "--skip-report",
help="跳过日报生成",
),
date: str | None = typer.Option(
None, "--date", "-d",
help="YYYYMMDD,默认当前新闻日",
),
):
"""M7: 一键运行完整管道 M2→M6(+ 可选日报)"""
from crawler.utils import get_news_day
from scheduler.pipeline import run_pipeline
date_str = date or get_news_day()
typer.echo(f"🚀 开始全链路管道,日期: {date_str}\n")
try:
result = run_pipeline(date_str, skip_report=skip_report)
typer.echo(f"\n{'='*50}")
typer.echo(f"Pipeline 完成: {result.success_count}/{len(result.steps)} 成功")
for s in result.steps:
flag = "✅" if s.success else "❌"
typer.echo(f" {flag} {s.name}: {s.message} ({s.elapsed_sec:.0f}s)")
if result.finished_at and result.started_at:
total = (result.finished_at - result.started_at).total_seconds()
typer.echo(f"\n⏱ 总耗时: {total:.0f}s")
except Exception as e:
logger.exception("管道执行失败")
typer.echo(f"❌ 管道执行出错: {e}", err=True)
raise typer.Exit(code=1)
@app.command()
def mcp_server():
"""M8: 启动 MCP 服务(供 Claude Code / Cherry Studio 调用)"""
from mcp_server.server import mcp
typer.echo("🚀 启动国际财经 Deep Research MCP 服务...")
mcp.run()
if __name__ == "__main__":
app()