diff --git a/.env.example b/.env.example index ecd125f..b88e5ec 100644 --- a/.env.example +++ b/.env.example @@ -13,8 +13,10 @@ QWEN_API_KEY=sk-your-qwen-key QWEN_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 # --- DashScope / 阿里百炼 --- +# M5 Embedding 使用 OpenAI 兼容端点,默认复用 QWEN_BASE_URL; +# 如需独立端点,可设置 DASHSCOPE_EMBEDDING_BASE_URL。 DASHSCOPE_API_KEY=sk-your-dashscope-key -DASHSCOPE_BASE_URL=https://dashscope.aliyuncs.com/api/v1/services/embeddings/text-embedding/text-embedding +# DASHSCOPE_EMBEDDING_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 # --- Qdrant --- QDRANT_URL=http://localhost:6333 diff --git a/CLAUDE.md b/CLAUDE.md deleted file mode 100644 index ade49d5..0000000 --- a/CLAUDE.md +++ /dev/null @@ -1,740 +0,0 @@ -# CLAUDE.md - -# Claude Code 开发约束(English Financial News 项目) - -版本:v1.1 - -最后更新:2026-07-23 - ---- - -# 〇、命令速查(2026-07-23 实测) - -开发环境: - -* Python 3.11 + uv + `.venv`(详见第三节) -* 安装依赖:`uv sync`;新增依赖:`uv add 包名`;开发依赖:`uv add --dev 包名` - -常用命令: - -| 命令 | 用途 | -|------|------| -| `uv run pytest` | 跑全部测试(当前 162 passed / 2 failed,见下方已知问题) | -| `uv run ruff check .` | 代码规范检查(当前 13 错误,9 可自动修复) | -| `uv run en-news pipeline` | M2→M6 完整管道 | -| `uv run en-news report` | 仅生成日报 | -| `uv run en-news search "查询"` | Qdrant 语义检索 | -| `uv run en-news mcp-server` | 启动 M8 MCP 服务 | -| `bash scripts/domestic_full.sh` | 全流程(M1→M6→日报),Pi 上由 crontab 06/12/18/22 调度 | -| `bash scripts/domestic_crawl_8g.sh [源ID]` | 仅 M1 抓取(可单源测试) | -| `bash scripts/pipeline.sh` | M2→M6 管道 | - -已知问题: - -* `tests/test_crawler.py::test_write_and_load_index_jsonl`、`test_write_index_jsonl_dedup` 失败 — `crawler/storage.py:load_index()` 硬编码 `data/raw/...` 路径,未使用测试临时目录; -* ruff 未清零(多为 import 排序 I001,可 `--fix`); -* 日报/摘要 Prompt 内嵌在 `scheduler/reporter.py`,尚未拆分到 `prompts/`。 - ---- - -# 一、项目定位 - -本项目目标: - -构建面向国际财经新闻的私有化 Deep Research 平台。 - -核心能力包括: - -* 英文财经新闻抓取(Crawl4AI,12 个活跃源,国内 Pi 独立运行); -* 英文正文提取(trafilatura); -* 全文英译中(LLM DeepSeek); -* 投资事件抽取(LLM); -* 双语向量知识库(Qdrant); -* MCP 服务 + Cherry Studio / Claude Code Agent 深度研究; -* 每日 AI 摘要日报。 - -本项目不是: - -* 自动交易系统; -* 股票预测系统; -* 投资顾问系统。 - -Claude Code 必须始终围绕"研究辅助平台"进行设计。 - ---- - -# 二、Claude Code 总体行为准则 - -Claude Code 必须遵守以下原则: - -## 2.1 分阶段开发 - -禁止一次性完成整个项目。 - -必须: - -* 每次只完成一个 Milestone; -* 等待人工验收; -* 验收通过后再进入下一阶段。 - -禁止: - -* 擅自推进后续阶段; -* 推翻已完成模块。 - -Milestone 顺序严格按照 english-news-plan.md 第九节执行: -M0 → M1 → 同步机制 → M2 → M3 → M4 → M5+M6 → M7+日报 → M8 - ---- - -## 2.2 最小改动原则 - -修改代码时: - -优先局部修改。 - -禁止: - -为了优化而重写整个模块。 - -除非明确要求: - -"允许重构"。 - -否则: - -保持向后兼容。 - ---- - -## 2.3 先理解,再编码 - -开始编码前必须: - -明确: - -* 当前目标; -* 输入; -* 输出; -* 验收标准。 - -如果需求冲突: - -必须先提问。 - -不得自行猜测。 - ---- - -## 2.4 部署拓扑意识 - -本项目当前为单服务器架构(海外服务器已于 2026-07-14 停用,详见 README): - -| 服务器 | SSH | 部署路径 | 职责 | -|--------|-----|---------|------| -| 国内 | `ssh pi@192.168.1.160` | `/home/pi/intlnews` | M1→M8 全链路 + 日报 | - -历史:海外 `ecs-user@8.217.19.253:/opt/intlgrab` 曾负责 M1 抓取 → rsync 推送,已停用;`scripts/overseas_*.sh`、`scripts/domestic_sync.sh` 为遗留脚本(未清理)。 - -编码时注意: - -* M1 抓取在国内 Pi(8GB)以 headful Playwright + HTTP 代理(Shadowsocks + Privoxy)运行,需 Xvfb 虚拟显示器(见 README 前置条件); -* 强反爬源(Reuters / Investing.com / FT)走 Google News RSS 或 RSS 摘要,全文抓取成功率取决于代理线路质量; -* 抓取配置按运行环境拆分为 `configs/profiles/2g_headless.yaml` 与 `configs/profiles/8g_headful.yaml`,通过环境变量 `EN_NEWS_PROFILE` 选择; -* `data/raw/{source_id}/{yyyymmdd}/` 存放抓取原始 HTML + Markdown + index.jsonl,为全管道数据源头。 - ---- - -# 三、开发环境规范 - -## 3.1 Python 版本 - -统一使用: - -Python 3.11 - -禁止: - -* Python 3.13; -* Python 3.10 以下版本。 - ---- - -## 3.2 依赖管理 - -统一使用: - -uv - -禁止: - -requirements.txt 手工维护。 - -依赖定义: - -pyproject.toml - -安装命令: - -uv sync - -新增依赖: - -uv add 包名 - -开发依赖: - -uv add --dev 包名 - ---- - -## 3.3 虚拟环境 - -统一使用: - -.venv - -禁止: - -使用 Conda。 - -禁止: - -多个虚拟环境混用。 - ---- - -# 四、代码规范 - -## 4.1 类型注解 - -所有新增代码必须包含类型注解。 - -示例: - -def search_news( -keyword: str, -top_k: int -) -> list[dict]: -... - -禁止: - -省略类型。 - ---- - -## 4.2 中文说明 - -要求: - -代码使用英文命名。 - -注释与文档使用中文。 - -例如: - -# 新闻去重处理 - -def deduplicate_articles(): -... - ---- - -## 4.3 函数长度 - -单个函数: - -建议 ≤ 50 行。 - -超过: - -必须拆分。 - ---- - -## 4.4 文件长度 - -单文件: - -建议 ≤ 500 行。 - -超过: - -必须拆分模块。 - ---- - -## 4.5 禁止魔法数字 - -禁止: - -importance = 5 - -应写成: - -MAX_IMPORTANCE = 5 - ---- - -## 4.6 数据模型 - -统一使用 Pydantic 定义数据结构。 - -核心模型(定义见 english-news-plan.md 第四节): - -* `EnArticle` — 新闻文章(含中英文标题、正文、词数) -* `EnExtractedEvent` — 投资事件(事件类型、股票代码、情绪、重要度) - -禁止: - -大量裸 dict 传递。 - ---- - -# 五、日志规范 - -统一使用: - -logging - -禁止: - -print() - -日志级别: - -DEBUG - -INFO - -WARNING - -ERROR - -CRITICAL - -日志格式: - -时间 - 模块 - 级别 -消息 - -示例: - -2026-06-21 06:30:00 - translator - INFO -开始翻译 reuters 源文章 15 篇 - ---- - -# 六、异常处理规范 - -禁止: - -except: -pass - -必须: - -记录日志。 - -抛出明确异常。 - -示例: - -except Exception as e: - logger.exception(e) - raise - ---- - -# 七、测试规范 - -新增功能必须提供测试。 - -优先使用: - -pytest - -测试目录: - -tests/ - -命名: - -test_xxx.py - -测试覆盖: - -核心模块必须覆盖。 - -包括: - -* crawler(M1 抓取) -* extractor(M2 正文提取) -* dedup(M3 去重) -* llm(M4a 翻译 + M4b 事件抽取,单次 LLM 调用合并输出;`translator/` 为空壳目录,翻译实现在 `llm/`) -* embedding(M5 向量生成) -* vectorstore(M6 Qdrant 入库/检索) -* scheduler(M7 定时调度) -* mcp_server(M8 MCP 服务) - ---- - -# 八、配置规范 - -## 8.1 禁止硬编码 - -所有配置项(API Key、URL、路径、阈值、超时、并发数等)一律禁止写在代码中。 - -常量设置必须按实际情况分配到以下位置: - -| 配置类型 | 存放位置 | 示例 | -|---------|---------|------| -| 密钥/Token | `.env` | `DEEPSEEK_API_KEY`、`DASHSCOPE_API_KEY` | -| API 地址 | `.env` | `DEEPSEEK_BASE_URL`、`QWEN_BASE_URL` | -| Qdrant 连接 | `.env` | `QDRANT_URL`、`QDRANT_API_KEY` | -| 所有功能配置 | `configs/system.yaml` | 模型名、部署路径、阈值、超时、调度、日志 | -| 新闻源定义 | `configs/sources.yaml` | 源 ID、URL 模板、JS 渲染开关 | - -**原则**:`.env` 仅存 SSH/API 的密钥和地址。其余全部在 `system.yaml`。 - -代码中只允许引用配置,不允许定义配置值。 - -正确示例: - -```python -import os -LLM_API_KEY = os.environ["DEEPSEEK_API_KEY"] -``` - -```python -from yaml import safe_load -config = safe_load(open("configs/system.yaml")) -max_articles = config["crawler"]["max_articles_per_run"] -``` - -错误示例: - -```python -API_KEY = "sk-xxx" # ❌ 硬编码密钥 -LLM_TIMEOUT = 60 # ❌ 魔法数字硬编码 -SOURCES = [{"id": "reuters", ...}] # ❌ 源配置硬编码 -``` - -## 8.2 配置文件清单 - -* `configs/sources.yaml` — 英文财经源定义(id / name / url_pattern / js_render 等) -* `configs/system.yaml` — 系统级业务参数(去重阈值、LLM 超时、日报限制等) -* `.env` — 密钥、服务地址(不入 Git) -* `.env.example` — 脱敏示例(入 Git) - -## 8.3 环境变量读取规范 - -* 始终使用 `os.environ["KEY"]`(失败时抛出明确异常),不使用默认值硬编码; -* 如需默认值,必须在 `configs/system.yaml` 中定义,代码从配置文件读取; -* 禁止在 `os.getenv("KEY", "hardcoded_default")` 中写死默认值。 - ---- - -# 九、Prompt 管理规范 - -Prompt 必须独立维护。 - -目录: - -prompts/ - -禁止: - -在代码中直接拼接长 Prompt。 - -本项目 Prompt 文件: - -* `translation_and_extraction.md` — 翻译 + 事件抽取(单次 LLM 调用合并输出) - -注意:`daily_report.md`、`search_agent.md` 尚未创建;日报/摘要 Prompt 目前内嵌在 `scheduler/reporter.py`(技术债,待拆分到 prompts/)。 - ---- - -# 十、数据库规范 - -Qdrant 为唯一向量数据库。 - -Collection 名称:`en_finance_news` - -SQLite 用于: - -任务状态; -缓存; -调试。 - -禁止: - -引入多种向量数据库。 - -除非用户明确要求。 - ---- - -# 十一、Docker 规范 - -所有服务必须支持 Docker。 - -必须提供: - -docker-compose.yml - -必须支持: - -docker compose up -d - -启动。 - -注意: - -* 预期形态:docker-compose 包含 Qdrant + 调度器 + MCP 服务(尚未落地)。 - -不得依赖: - -手工安装。 - -当前状态(2026-07-23 核查):仓库中尚无 Dockerfile / docker-compose.yml,此要求尚未落实,属待办事项。 - ---- - -# 十二、Git 提交规范 - -完成每个 Milestone 后: - -更新: - -README.md - -continuation.md - -docs/ - -提交信息格式: - -feat: -新增功能 - -fix: -问题修复 - -refactor: -重构 - -docs: -文档更新 - -test: -测试 - -chore: -杂项 - -示例: - -feat: 完成 M1 英文财经新闻抓取模块(Crawl4AI) - ---- - -# 十三、README 更新规范 - -每次功能完成后: - -README 必须更新: - -包括: - -功能说明; - -部署方法(区分海外/国内); - -配置说明; - -示例命令; - -常见问题。 - -禁止: - -README 长期不维护。 - ---- - -# 十四、continuation.md 维护规范 - -Claude Code 每次结束工作前: - -必须更新 continuation.md。 - -内容包括: - -当前 Milestone; - -完成内容; - -待办事项; - -已知问题; - -技术债务; - -下次建议。 - -用于恢复上下文。 - ---- - -# 十五、性能要求 - -目标: - -初始支持 12 个英文财经新闻源(见 english-news-plan.md)。 - -逐步扩展至 ≥ 50 个源。 - -处理吞吐: - -≥ 100 篇 / 分钟(端到端:抓取 → 翻译 → 入库)。 - -Qdrant 检索: - -Top-K 响应时间 ≤ 2 秒。 - -LLM 翻译: - -单篇平均 ≤ 3 秒(DeepSeek flash 模型)。 - ---- - -# 十六、翻译质量规范 - -LLM 翻译必须: - -* 使用 DeepSeek 为默认 Provider(Qwen 为备选); -* System prompt 约束财经翻译风格:准确、简洁、专业术语一致; -* 保留英文原标题(`title`)和中文翻译标题(`title_zh`); -* 保留英文原文(`content_en`)和中文翻译(`content_zh`); -* 中文译文字数控制在原文 1.0×–1.5× 范围。 - -事件抽取必须: - -* 正确识别涉及的美股代码(如 AAPL、TSLA); -* 情绪判断有据可查(利好/利空/中性); -* 重要度 1-5 分,≥ 4 分进入日报"重要事件"板块。 - ---- - -# 十七、安全规范 - -禁止: - -提交: - -.env - -API Key(DeepSeek / DashScope / Qwen) - -Cookie - -Token - -个人隐私数据 - -SSH 私钥 - -必须: - -提供: - -.env.example - -示例配置(脱敏)。 - ---- - -# 十八、禁止事项 - -Claude Code 禁止: - -1. 未经允许重构已验收模块; -2. 一次性生成整个项目; -3. 擅自修改数据库结构(包括 Qdrant Collection Schema); -4. 删除已有测试; -5. 使用 print 调试; -6. 忽略异常; -7. 跳过验收直接进入下一阶段; -8. 引入未经说明的新技术栈; -9. 将 Prompt 写死在代码中; -10. 编写无法运行的伪代码冒充完成; -11. 将海外侧依赖(LLM/Embedding)引入 M1 抓取模块; -12. 擅自新增英文新闻源而不更新 configs/sources.yaml 和 plan。 - ---- - -# 十九、输出格式要求 - -Claude Code 完成任务时必须输出: - -【任务目标】 - -【完成内容】 - -【修改文件】 - -【运行方法】 - -【测试结果】 - -【存在问题】 - -【下一步建议】 - -不得只输出代码。 - -必须提供可验证说明。 - ---- - -# 二十、最高优先级原则 - -当多个原则冲突时,优先级如下: - -第一优先级: - -代码正确、可运行。 - -第二优先级: - -稳定性与可维护性。 - -第三优先级: - -向后兼容。 - -第四优先级: - -性能优化。 - -第五优先级: - -代码优雅。 - -宁可代码普通,也不要复杂炫技。 - -本项目追求: - -"小步迭代、稳定演进、长期维护"。 - ---- - -# 二十一、参考文档 - -* `english-news-plan.md` — 项目总体计划(权威来源) -* `docs/` — 设计文档 -* 对标项目:`news/`(A 股 Deep Research),架构模式可复用 - -—— CLAUDE.md 结束 —— diff --git a/README.md b/README.md index 7e427a6..9f5ca15 100644 --- a/README.md +++ b/README.md @@ -1,267 +1,54 @@ # English Financial News -国际财经新闻私有化 Deep Research 平台。 +国际财经新闻私有化 Deep Research 平台:抓取、去重、翻译、事件抽取、向量知识库、日报生成。 + +> 完整文档已整理到 [docs/](docs/README.md)。本文件仅作为仓库入口。 ## 核心能力 -- **M1** 英文财经新闻抓取(Crawl4AI,headful Playwright + HTTP 代理,13 个源) -- **M2** 英文正文提取(trafilatura) -- **M3** 三层去重(URL Hash / 内容 Hash / SimHash 模糊) -- **M4** 全文英译中 + 投资事件抽取(LLM DeepSeek v4-flash) -- **M5** 向量生成(DashScope text-embedding-v3,1024 维) -- **M6** 双语向量知识库(Qdrant) -- **M7** 定时调度(06/12/18/22) + 每日 AI 摘要日报(M9 起结构化写入 MySQL,不再产出 HTML) -- **M8** MCP 服务(Cherry Studio / Claude Code Agent 深度研究) - -## 部署架构 - -``` -国内服务器(Pi 4, 8GB) -┌──────────────────────────────────────┐ -│ M1 抓取(headful Playwright) │ -│ HTTP 代理 → Privoxy → SOCKS5 │ -│ ↓ │ -│ M2 正文提取 → M3 去重 │ -│ ↓ │ -│ M4 翻译+事件抽取(DeepSeek) │ -│ ↓ │ -│ M5 向量生成 → M6 Qdrant 入库 │ -│ ↓ │ -│ 日报生成 → 写入 MySQL(news_report) │ -│ ↓ │ -│ M8 MCP 服务(研究 Agent) │ -└──────────────────────────────────────┘ -``` - -> 海外服务器已于 2026-07-14 停用,全链路在 Pi 独立运行。 - ---- - -## 新闻源状态(2026-07-23) - -| 源 | ID | 策略 | 日产量 | 状态 | -|----|----|------|--------|------| -| Reuters | `reuters` | Google News RSS | ~30 篇摘要 | ⚠️ DataDome | -| CNBC | `cnbc` | CNBC RSS | ~25 篇 | ✅ | -| MarketWatch | `marketwatch` | RSS + stealth | ~25 篇 | ✅ | -| Financial Times | `ft` | FT RSS | ~20 篇摘要 | ⚠️ 付费墙 | -| Yahoo Finance | `yahoo_finance` | Yahoo RSS | ~30 篇 | ✅ | -| **Investing.com** | `investing` | **Google News RSS** + stealth | **25 篇摘要** | **✅ 2026-07-23 修复** | -| Seeking Alpha | `seekingalpha` | RSS + stealth | ~20 篇 | ✅ | -| Barron's | `barrons` | Dow Jones RSS + stealth | ~20 篇 | ✅ | -| WSJ | `wsj` | Dow Jones RSS + stealth | ~20 篇 | ✅ | -| Economist | `economist` | RSS + stealth | ~15 篇 | ✅ | -| **InvestingLive** | `investinglive` | **RSS(全文)** | **25 篇** | **✅ 2026-07-23 新增** | -| ZeroHedge | `zerohedge` | FeedBurner RSS | ~15 篇全文 | ✅ | -| ~~ForexLive~~ | ~~forexlive~~ | ~~已迁移~~ | ~~0~~ | ~~❌ 301→investinglive~~ | - -> **注:** RSS 摘要源的全文通过 headful Playwright + HTTP 代理回退抓取。当前国内服务器使用 Shadowsocks + Privoxy 代理访问海外。 - ---- +- M1 英文财经新闻抓取(RSS 优先 + Crawl4AI/Playwright Web 回退) +- M2 正文提取(trafilatura) +- M3 三层去重(URL / 内容 / SimHash) +- M4 全文英译中 + 投资事件抽取(DeepSeek) +- M5 向量生成(DashScope text-embedding-v3) +- M6 Qdrant 语义知识库 +- M7 全链路调度与每日 AI 摘要日报 +- M8 MCP 服务(Cherry Studio / Claude Code 接入) +- M9 日报结构化写入 MySQL ## 快速开始 -### 环境要求 - -- Python 3.11 -- uv -- Xvfb(Pi 服务器 headful 模式需要) -- Shadowsocks 客户端 + Privoxy(国内服务器海外代理) - -### 安装 - ```bash -# 创建虚拟环境并安装依赖 +# 安装 uv sync +cp .env.example .env # 编辑 API Key/数据库凭据 -# 配置环境变量 -cp .env.example .env -# 编辑 .env 填入真实 API Key(DeepSeek / DashScope / Qdrant) -``` - -### Pi 服务器前置条件 - -```bash -# 1. Xvfb 虚拟显示器(headful Playwright 需要) -sudo apt install xvfb -Xvfb :99 -screen 0 1280x1024x24 -ac +extension RANDR & - -# 2. Shadowsocks + Privoxy(HTTP 代理访问海外) -sudo apt install shadowsocks-libev privoxy -# 配置 /etc/shadowsocks-libev/config.json(服务端信息) -# 配置 /etc/privoxy/config(forward-socks5t 127.0.0.1:1088 .) -sudo systemctl enable --now shadowsocks-libev-local -sudo systemctl enable --now privoxy -``` - ---- - -## 操作命令 - -### 全流程自动化 - -```bash -# 全流程:M1 抓取 → M2→M6 管道 → 日报 -# 每天 06:00 / 12:00 / 18:00 / 22:00 自动执行 +# 全流程 bash scripts/domestic_full.sh -# 中断恢复:跳过当天已完成步骤(状态存 data/run_state/{date}.state) -# 支持步骤级断点:M1_crawl / M2_extract / M3_dedup / M4_translate / M5_embed / M6_index / report +# 断点续跑 bash scripts/domestic_full.sh --resume -# 仅 M2→M6 管道(同样支持 --resume) -bash scripts/pipeline.sh --resume -``` -### 单步执行 - -```bash -# M1 抓取全部源(headful + HTTP 代理) -bash scripts/domestic_crawl_8g.sh - -# M1 单源测试 -bash scripts/domestic_crawl_8g.sh investing -bash scripts/domestic_crawl_8g.sh investinglive - -# M2→M6 管道(需先有 raw 数据) +# M2→M6+日报管道 bash scripts/pipeline.sh -# 仅生成日报(写入 MySQL news_report/news_event,report_type="intl") -uv run en-news report +# CLI 单步 +uv run en-news --help ``` -### 调度查看 +## 文档导航 -```bash -# 查看最近日志 -tail -50 logs/full.log - -# 查看日报 -ls -lt data/reports/ -``` - ---- - -## 配置说明 - -| 文件 | 用途 | +| 文档 | 说明 | |------|------| -| `configs/sources.yaml` | 英文财经新闻源定义(13 个源) | -| `configs/system.yaml` | 模型按场景(`llm_scenes`)/重试/阈值等系统配置 | -| `configs/profiles/8g_headful.yaml` | Pi 服务器 headful 抓取配置(代理/超时) | -| `configs/profiles/8g_headful.yaml` | Pi 服务器 headful 抓取配置(代理/超时) | -| `.env` | 密钥 / 服务地址(不入 Git) | -| `prompts/` | LLM Prompt 模板(翻译/日报/搜索 Agent) | - -### AI 模型按场景配置(llm_scenes) - -大模型按场景独立配置,见 `configs/system.yaml` 的 `llm_scenes` 段: - -| 场景 | 用途 | 模型(当前) | 参数 | -|------|------|-------------|------| -| `translation` | M4 全文英译中 + 投资事件抽取 | deepseek-v4-flash | temperature=0.1, max_tokens=8192 | -| `daily_report` | M7 日报 AI 摘要(分批生成) | deepseek-v4-flash | temperature=0.3, max_tokens=1500 | - -场景未声明的字段回退 `llm` 默认段;Embedding 为单一场景(`en_finance_news` 库入库/检索向量必须同模型,不支持拆分)。 - -### 去重多来源(M3) - -去重时跨源重复的新闻,会把所有来源记录到保留的唯一篇 `source_ids` 字段(首个来源为 `source_id`),经翻译透传后: - -- `news_event.source`:拼接展示名(如 "Barron's, CNBC, Reuters",最多前 3 个,向后兼容) -- `news_event.sources`(TEXT,JSON 数组):**全部来源展示名,主源居首**,如 `["Barron's","CNBC","Reuters","Financial Times"]`,前端直接渲染完整来源列表 - -表结构变更(`news_event` 新增 `sources` 列)由 `report_db.init_schema()` 幂等迁移(`ALTER TABLE ... ADD COLUMN IF NOT EXISTS`,MariaDB 10.0.2+)。 - -### 中断恢复(--resume) - -全流程脚本支持断点续跑: - -- `scripts/domestic_full.sh --resume` / `scripts/pipeline.sh --resume`:跳过当天已完成步骤(步骤状态存 `data/run_state/{YYYYMMDD}.state`,按天换新,跨天自动失效) -- 步骤粒度:`M1_crawl` / `M2_extract` / `M3_dedup` / `M4_translate` / `M5_embed` / `M6_index` / `report`;某步失败不标记,`--resume` 从失败处重试 -- **文件级增量(各步骤内自动跳过已处理文件)**:M2 按 `data/processed/.../{url_hash}.json` 存在性、M3 按指纹库判重、M4 按 `data/events/.../{url_hash}.json` 存在性、M5 按 `data/embeddings/.../index.json`、M6 按 Qdrant upsert 幂等(点 id=url_hash)、日报按 MySQL 唯一键覆盖 - -### 终端进度与日志 - -执行全流程时,终端**实时显示完整进度**,且**显性标注当前阶段与 AI 模型**: - -- 每阶段标题:`━━━ M2 正文提取 ━━━`、`━━━ M4 翻译+事件抽取(AI 大模型: deepseek / deepseek-v4-flash)━━━`(AI 阶段自动从 `system.yaml` 读取供应商/模型) -- 每步进度:`▶ 步骤 开始` → 步骤内逐源/逐篇输出 → `✔ 完成(耗时 Ns)`;失败显示 `✗` 与退出码 -- AI 调用点在 Python 层同样显性打印(`AI 大模型(场景 translation/daily_report): provider=... model=...`、`初始化 Embedding 客户端: provider=dashscope model=...`) -- 日志:`logs/domestic_full_{ts}.log`(全流程)、`logs/pipeline_{ts}.log`(管道),已入 `.gitignore` - -### 日报入库(M9) - -日报内容结构化写入与 [news 项目](https://github.com/) 共用的 MySQL `myquant` 库(表 `news_report` / `news_event`,`report_type="intl"`,同一天重复生成幂等覆盖)。表结构与数据契约见 news 项目 `docs/db_schema.md`。 - -`.env` 需配置(与 news 项目 `.env` 的 `NEWS_DB_PASSWORD` 相同): - -```env -NEWS_DB_HOST=192.168.1.10 # pi5 经内网直连 pi 上的 autossh 隧道(0.0.0.0:13306 → doorcome.cn:3306) -NEWS_DB_PORT=13306 -NEWS_DB_USER=myquant -NEWS_DB_PASSWORD=xxx -NEWS_DB_NAME=myquant -``` - ---- - -## 数据存储 - -``` -data/ -├── raw/{source_id}/{yyyymmdd}/ M1 原始抓取(HTML + Markdown) -├── processed/{source_id}/ M2 正文提取结果 -├── dedup/ M3 去重指纹库(SQLite) -├── translations/{yyyymmdd}/ M4 翻译+事件抽取结果(JSONL) -├── embeddings/{yyyymmdd}/ M5 向量文件(JSONL) -├── reports/ M7 日报(M9 起不再产出 HTML,改存 MySQL) -└── vectorstore/ M6 Qdrant 数据 -``` - -Qdrant collection 名称:`en_finance_news` - ---- - -## MCP 服务 - -M8 模块提供 MCP(Model Context Protocol)服务,供 Cherry Studio / Claude Code 等客户端接入进行深度研究。 - -```bash -# 启动 MCP 服务 -uv run en-news mcp-server -``` - -支持的工具: - -| Tool | 功能 | -|------|------| -| `en_news_search` | 语义搜索知识库 | -| `en_news_filter` | 按源/日期/情绪筛选 | -| `en_news_trending` | 获取热点事件 | -| `en_news_report` | 获取最新日报 | -| `en_news_daily_brief` | 一键生成简报 | - ---- - -## 常见问题 - -### 为什么有些源只拿到摘要? - -部分网站有强反爬(Cloudflare/DataDome/付费墙),RSS 只返回摘要。全文可尝试 headful 浏览器回退,但成功率取决于代理线路质量。 - -### 日报存在哪里? - -M9 起日报结构化写入 MySQL `myquant` 库:主表 `news_report`(`report_type="intl"`)+ 明细表 `news_event`(`section="intl"`),数据契约与 [news 项目](https://github.com/) 一致(见其 `docs/db_schema.md`)。历史 HTML 日报(2026-06-16 ~ 2026-08-03)已由 news 项目解析导入。API / 前端读取同一批表。 - -### 如何添加新源? - -编辑 `configs/sources.yaml`,添加源配置(id/name/homepage/rss_url/article_url_pattern 等)。参考现有源格式。 - ---- - -## 项目计划 - -详见 [english-news-plan.md](english-news-plan.md) +| [docs/README.md](docs/README.md) | 文档中心 | +| [docs/architecture.md](docs/architecture.md) | 项目架构与数据流 | +| [docs/quickstart.md](docs/quickstart.md) | 快速开始 | +| [docs/usage.md](docs/usage.md) | 使用手册 | +| [docs/pipeline.md](docs/pipeline.md) | 流水线详解 | +| [docs/configuration.md](docs/configuration.md) | 配置说明 | +| [docs/deployment.md](docs/deployment.md) | 部署与运维 | +| [docs/development.md](docs/development.md) | 开发指南 | +| [docs/faq.md](docs/faq.md) | FAQ | ## 许可证 diff --git a/app/cli.py b/app/cli.py index 9a65a4e..53361be 100644 --- a/app/cli.py +++ b/app/cli.py @@ -56,6 +56,9 @@ def crawl( typer.echo(f"\n✅ 完成: {stats.sources_crawled} 源, {stats.total_articles} 篇文章") if stats.sources_failed: typer.echo(f"⚠️ {stats.sources_failed} 个源有错误") + if stats.sources_crawled > 0 and stats.sources_failed >= stats.sources_crawled: + typer.echo("❌ 所有源抓取失败", err=True) + raise typer.Exit(code=1) except FileNotFoundError as e: typer.echo(f"❌ {e}", err=True) raise typer.Exit(code=1) @@ -71,12 +74,16 @@ def extract( None, "--source", "-s", help="只处理指定 source_id(不传则全部)", ), + date: str | None = typer.Option( + None, "--date", "-d", + help="YYYYMMDD,默认当前新闻日", + ), ): """M2: 英文正文提取(trafilatura)""" from extractor.pipeline import process_all_sources try: - stats = process_all_sources(source_filter=source) + stats = process_all_sources(source_filter=source, date_str=date) typer.echo( f"\n✅ 提取: {stats['sources_processed']} 源, " f"{stats['total_articles']} 篇, {stats['elapsed_sec']:.1f}s" @@ -88,12 +95,17 @@ def extract( @app.command() -def dedup(): +def dedup( + date: str | None = typer.Option( + None, "--date", "-d", + help="YYYYMMDD,默认当前新闻日", + ), +): """M3: 三层去重""" from dedup.pipeline import dedup_all_sources try: - stats = dedup_all_sources() + stats = dedup_all_sources(date_str=date) typer.echo( f"\n✅ 去重: {stats['sources_processed']} 源, " f"唯一 {stats['unique']} / 重复 {stats['duplicate']} / " @@ -106,12 +118,17 @@ def dedup(): @app.command() -def translate(): +def translate( + date: str | None = typer.Option( + None, "--date", "-d", + help="YYYYMMDD,默认当前新闻日", + ), +): """M4: 全文翻译 + 投资事件抽取(LLM)""" from llm.pipeline import translate_all_deduped try: - stats = translate_all_deduped() + stats = translate_all_deduped(date_str=date) typer.echo( f"\n✅ 翻译+事件抽取: {stats['success']}/{stats['total']} 篇, " f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})" @@ -123,12 +140,50 @@ def translate(): @app.command() -def embed(): +def embed( + date: str | None = typer.Option( + None, "--date", "-d", + help="YYYYMMDD,默认当前新闻日", + ), + all_dates: bool = typer.Option( + False, "--all", + help="处理 data/events 下所有日期(可用于历史回灌)", + ), +): """M5: 向量生成""" + from pathlib import Path + from embedding.pipeline import embed_all_events + if all_dates and date: + typer.echo("❌ --date 与 --all 不能同时使用", err=True) + raise typer.Exit(code=1) + try: - stats = embed_all_events() + if all_dates: + dates = sorted( + p.name for p in Path("data/events").iterdir() + if p.is_dir() and p.name.isdigit() + ) + if not dates: + typer.echo("⚠️ data/events/ 下没有可处理日期") + return + total_success = 0 + total_count = 0 + for day in dates: + stats = embed_all_events(date_str=day) + total_success += stats["success"] + total_count += stats["total"] + typer.echo( + f" [{day}] 向量生成: {stats['success']}/{stats['total']} 篇, " + f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})" + ) + typer.echo( + f"\n✅ 全部日期向量生成: {total_success}/{total_count} 篇" + ) + return + + stats = embed_all_events(date_str=date) typer.echo( f"\n✅ 向量生成: {stats['success']}/{stats['total']} 篇, " f"{stats['elapsed_sec']:.1f}s ({stats['provider']}/{stats['model']})" @@ -145,12 +200,57 @@ def index( False, "--recreate", help="重建 collection(会删除已有数据)", ), + date: str | None = typer.Option( + None, "--date", "-d", + help="YYYYMMDD,默认当前新闻日", + ), + all_dates: bool = typer.Option( + False, "--all", + help="处理 data/embeddings 下所有日期(可用于历史回灌)", + ), ): """M6: Qdrant 入库""" + from pathlib import Path + from vectorstore.pipeline import get_collection_info, ingest_all_embeddings + if all_dates and date: + typer.echo("❌ --date 与 --all 不能同时使用", err=True) + raise typer.Exit(code=1) + try: - stats = ingest_all_embeddings(recreate=recreate) + if all_dates: + dates = sorted( + p.name for p in Path("data/embeddings").iterdir() + if p.is_dir() and p.name.isdigit() + ) + if not dates: + typer.echo("⚠️ data/embeddings/ 下没有可处理日期") + return + total_ingested = 0 + total_count = 0 + first = True + for day in dates: + # 全量回灌时只在第一次真正 recreate,避免后续清空已写数据 + stats = ingest_all_embeddings( + date_str=day, + recreate=recreate and first, + ) + first = False + total_ingested += stats["ingested"] + total_count += stats["total"] + typer.echo( + f" [{day}] 入库: {stats['ingested']}/{stats['total']} 条, " + f"{stats['elapsed_sec']:.1f}s" + ) + info = get_collection_info() + typer.echo( + f"\n✅ 全部日期入库: {total_ingested}/{total_count} 条" + ) + typer.echo(f"📊 Collection: {info['name']} — {info['vectors_count']} 条向量") + return + + stats = ingest_all_embeddings(date_str=date, recreate=recreate) typer.echo( f"\n✅ 入库: {stats['ingested']}/{stats['total']} 条, " f"{stats['elapsed_sec']:.1f}s" @@ -218,12 +318,16 @@ def pipeline( False, "--skip-report", help="跳过日报生成", ), + date: str | None = typer.Option( + None, "--date", "-d", + help="YYYYMMDD,默认当前新闻日", + ), ): """M7: 一键运行完整管道 M2→M6(+ 可选日报)""" from crawler.utils import get_news_day from scheduler.pipeline import run_pipeline - date_str = get_news_day() + date_str = date or get_news_day() typer.echo(f"🚀 开始全链路管道,日期: {date_str}\n") try: diff --git a/configs/sources.yaml b/configs/sources.yaml index d9cb8d8..2a0ac6c 100644 --- a/configs/sources.yaml +++ b/configs/sources.yaml @@ -1,5 +1,5 @@ # 英文财经新闻源配置 -# 每个源的字段说明见 english-news-plan.md M1 章节 +# 每个源的字段说明见 docs/architecture.md 与 docs/configuration.md settings: concurrency: 5 diff --git a/continuation.md b/continuation.md deleted file mode 100644 index d440862..0000000 --- a/continuation.md +++ /dev/null @@ -1,247 +0,0 @@ -# continuation.md — English Financial News 项目状态 - -> 最后更新:2026-08-12 - ---- - -## 2026-08-12 会话成果(五):日报多来源入库(news_event.sources 列) - -**背景:** 前端需要展示一条新闻的多个来源;`source` 拼接字段(≤3 个)不够。 - -**改动(用户已授权改表,列已由用户添加):** - -| 文件 | 改动内容 | -|------|---------| -| `report_db/models.py` | `EventRow` 新增 `sources: list[str]`(全部来源展示名,主源居首) | -| `report_db/schema.py` | DDL 加 `sources TEXT NULL` 列;`init_schema` 增加幂等 `ALTER TABLE ... ADD COLUMN IF NOT EXISTS sources`(旧表迁移) | -| `report_db/db.py` | `save_report` INSERT 写入 sources(JSON 数组) | -| `scheduler/reporter.py` | 新增 `_article_sources_full()`(完整来源列表,主源居首);`_article_source_label` 重构为基于它(source 保持 ≤3 拼接兼容);`_build_report_data` 事件写入 `sources` | -| `tests/test_report_db.py` | 多源完整列表断言(2 源 / 4 源截断对比 / 单源回退) | - -**DB 迁移:** `news_event.sources` TEXT 列(用户已加);`init_schema()` 幂等补列(MariaDB `ADD COLUMN IF NOT EXISTS`),pi5 实测幂等 OK。 - -**部署中踩坑记录:** `rsync -a report_db/ scheduler/reporter.py pi5:/home/pi/intlnews/` 的混合参数把文件散落到目标根目录(report_db/ 带斜杠 = 内容进根、reporter.py 也进了根),导致 pi5 上模块未更新、sources 写入 None。**教训:rsync 多源参数混用目录与文件时目标路径易错,部署后必须 grep 关键标识(如 source_list)验证。** 已清理根目录残留并正确同步。 - -**验证:** 单元测试 3 用例(多源完整列表);pi5 实盘:`init_schema` 幂等、日报 report_id=224 sources 列全部写入(`["InvestingLive"]` 单源数组)。当前真实数据无多源新闻入日报(跨源合并发生在历史日期文件,未重新翻译),多源数组待后续新数据自然出现(链路已通,前端可直接读 `sources` JSON 数组)。 - ---- - -## 2026-08-12 会话成果(四):脚本阶段/模型显性输出 - -**背景:** 执行全流程时终端未显性标注"当前阶段",AI 调用的供应商/模型只在客户端初始化时输出一次。 - -**改动:** - -| 文件 | 改动内容 | -|------|---------| -| `scripts/pipeline.sh` | 每阶段 banner(`━━━ M4 翻译+事件抽取(AI 大模型: deepseek / deepseek-v4-flash)━━━`);`ai_model_info()` 从 system.yaml 读取场景模型(translation/daily_report/embedding) | -| `scripts/domestic_full.sh` | M1 / 管道阶段标题统一 banner 风格 | -| `llm/pipeline.py` | 创建客户端后显性输出 `AI 大模型(场景 translation): provider/model` | -| `embedding/pipeline.py` | 向量化日志补充 provider | -| `scheduler/reporter.py` | daily_report 场景客户端初始化时输出 provider/model | - -**测试:** 本地模拟 banner 全部正确(M4/M5/日报 3 处 AI 标注);pi5 实测客户端初始化日志含 provider/model(deepseek/deepseek-v4-flash、dashscope/text-embedding-v3);全量 184 passed。 - ---- - -## 2026-08-12 会话成果(三):全流程终端进度输出 - -**背景:** `domestic_full.sh` / `pipeline.sh` 原用 `tail -5/-10` 截断输出,终端看不到中间进度。 - -**改动:** - -| 文件 | 改动内容 | -|------|---------| -| `scripts/_step_state.sh` | `step_run` 增加耗时统计与进度日志(`▶ 开始` / `✔ 完成(耗时 Ns)` / `✗ 失败(exit=N,耗时)`);支持 `STEP_LOG` 环境变量——步骤全部输出实时显示终端并同时写入日志(用 `PIPESTATUS[0]` 取原命令退出码,规避 pipefail/tee 吞退出码) | -| `scripts/pipeline.sh` | 步骤输出透传终端(去掉 `tail -10`);日志写 `logs/pipeline_{ts}.log` | -| `scripts/domestic_full.sh` | M1 与管道输出透传终端(去掉 `tail -5/-10`);日志写 `logs/domestic_full_{ts}.log` | - -**测试:** 无 pipefail / pipefail+条件 / pipefail+非条件 三环境下退出码捕获均正确(失败不 mark);pi5 实机 `--resume` 跑通(M3 31s、M6 20s 均显示耗时,日志文件生成)。 - -**注意:** 所有含 `${var}` 后跟全角标点的输出已用花括号包裹(bash 会把全角字符字节并入变量名,导致 unbound variable)。 - ---- - -## 2026-08-12 会话成果(二):全流程 --resume 断点续跑 - -**背景:** `scripts/domestic_full.sh`(M1→M6→日报)中断(ssh 断线/断电)后只能从头重跑。 - -**改动:** - -| 文件 | 改动内容 | -|------|---------| -| `scripts/_step_state.sh`(新增) | 步骤状态库:`step_run` / `step_should_run` / `step_mark` / `step_done`;状态文件 `data/run_state/{YYYYMMDD}.state`(按天换新);`step_run` 显式检查退出码(规避 bash set -e 条件上下文陷阱) | -| `scripts/pipeline.sh` | 支持 `--resume`;M2→M6→report 每步 `step_run` 包裹 | -| `scripts/domestic_full.sh` | 支持 `--resume`;M1 用 `step_should_run`/`step_mark`(部分源失败仍标记,原语义);透传 `--resume` 给 pipeline.sh | -| `.gitignore` | 忽略 `data/run_state/` | - -**文件级增量确认(各步骤已内置,无需新写):** M2 processed JSON 存在跳过、M3 指纹库判重、M4 events JSON 存在跳过、M5 embedding index、M6 Qdrant upsert 幂等(id=url_hash)、日报 MySQL 唯一键覆盖。 - -**测试:** 本地模拟 3 场景全过(失败步骤不 mark / set -e 退出不 mark / resume 跳过已完成);`--badarg` exit=1;pi5 同步后验证 step_run 机制与参数校验正常。 - -**注意:** crontab 07/12/18 跑的是不带 --resume 的全量(步骤级不跳过,靠文件级增量),--resume 供手动中断恢复。 - ---- - -## 2026-08-12 会话成果 - -### M9.2:AI 模型按场景配置 + 去重多来源 - -**背景:** ① 本项目所有 AI 大模型调用点(M4 翻译+事件抽取、M7 日报 AI 摘要、M5 向量化)此前共用同一套 `llm`/`embedding` 配置;② M3 去重跨源重复时丢弃重复篇来源信息。 - -**改动:** - -| 文件 | 改动内容 | -|------|---------| -| `configs/system.yaml` | 新增 `llm_scenes` 段(translation / daily_report,含用途/使用方法/模型要求说明,字段回退 `llm` 默认段);`embedding` 段注释说明单一场景原因 | -| `llm/client.py` | `_load_system_config(scene)` 场景合并;`load_llm_config(..., scene)` 支持按场景覆盖 provider/model/参数;场景统一用 `model` 键 | -| `llm/pipeline.py` | `load_llm_config(scene="translation")` | -| `scheduler/reporter.py` | `_call_llm_simple` 用 `scene="daily_report"`;`temperature` 硬编码 0.3 → `config.temperature`(技术债清除);新增 `_article_source_label()` 多来源拼接展示 | -| `extractor/models.py` / `llm/models.py` | `ProcessedArticle` / `EnTranslatedArticle` 新增 `source_ids` 字段 | -| `dedup/pipeline.py` | unique 初始化 `source_ids=[source_id]`;dup 时 `_merge_duplicate_source()` 跨日期目录合并来源进唯一篇 | -| `llm/extractor.py` | 同步/异步构造 `EnTranslatedArticle` 透传 `source_ids` | -| `tests/` | `test_llm.py` 场景配置 4 用例;`test_dedup.py::TestMergeSources` 3 用例;`test_report_db.py` 多来源拼接 2 用例 | - -**测试结果:** 本地全量 184 passed / 2 failed(原有 test_crawler 路径问题);ruff 无新增。 - -**部署验证(pi5 实盘):** -- scene 配置生效:translation=(v4-flash, 0.1)、daily_report=(v4-flash, 0.3, 1500) -- 去重合并实证:历史唯一篇 `049da7e0c87160dc`(barrons 主源)被 9 个跨源重复篇合并 → `source_ids` 10 个来源 -- 日报仍正常入库:report_id=224(2026-08-12, 15 事件);当天 DeepSeek 摘要 3 次返回空 → 规则兜底,日报未中断(失败处理按设计工作) -- 多来源展示:`_article_source_label` 实测 "Barron's, CNBC, Reuters";单源回退正常 - -**已知说明:** -- 历史 3368 个 deduped 旧文件无 `source_ids` 字段(不回填,向前生效);events 文件由下次 crontab 07:00 用新代码自然带出透传 -- 179 篇全重复(增量正常);138 个跨源重复的 merge 目标多为历史日期目录文件(指纹库 ±30 天窗口所致),非当天 uniques - ---- - -## 2026-08-04 会话成果 - -### M9:日报结构化入库(HTML → MySQL) - -**背景:** 与 news 项目对齐,日报内容不再产出 HTML / scp 上传 doorcome,改为写入共用 MySQL `myquant` 库(表 `news_report` / `news_event`,`report_type="intl"`)。历史日报(2026-06-16 ~ 2026-08-03,intl 128 份)已由 news 项目解析入库,本次仅做新日报按相同契约入库。 - -**改动:** - -| 文件 | 改动内容 | -|------|---------| -| `report_db/`(新增) | 复用 news 项目实现:`models.py`(EventRow/ReportData)、`schema.py`(news_report/news_event DDL)、`db.py`(load_db_config/connect/init_schema/save_report/fetch_report/exists_report,loguru 换 logging) | -| `scheduler/reporter.py` | 新增 `_build_report_data()`(intl 结构:事件全部 `section="intl"`,`file_name=""` 幂等覆盖,stats 含 pipeline/sentiment/importance/event_types/source_dist);`generate_report()` 返回 `int | None`,收集逻辑不变,末尾 `connect()+save_report()`;HTML 渲染/上传函数保留标 deprecated | -| `scheduler/pipeline.py` | `run_step_report` 适配 `report_id` 返回值 | -| `app/cli.py` | `report` 命令打印 `report_id` | -| `pyproject.toml` | `uv add pymysql`;`report_db/schema.py` 加 E501 per-file-ignore | -| `.env.example` / `README.md` | 新增 `NEWS_DB_*` 配置说明(pi5 经内网直连 `192.168.1.10:13306`,password 与 news 项目相同) | -| `tests/test_report_db.py`(新增) | 模型默认值 + `_build_report_data` 组装(板块/rank、字段映射、标题截断、stats 快照、`"?"` 归一化) | - -**测试结果:** `uv run pytest` → 174 passed / 2 failed(2 个失败为原有 `test_crawler.py` index.jsonl 路径问题,与本次无关);ruff 改动文件干净(仅剩原有 `_SENTENCE_END` N806)。 - -**部署:** 已 rsync 至 pi5(`/home/pi/intlnews`),`.env` 写入 NEWS_DB_*,`uv sync` 安装 pymysql,真实生成日报并回读 DB 验证(详见下方)。 - ---- - -## 2026-07-23 会话成果 - -### investing.com 反爬策略调整 - -**问题:** RSS 返回 403(Cloudflare),Web headful 间歇性失败(数据中心 IP 被标记) - -**改动:** - -| 文件 | 改动内容 | -|------|---------| -| `crawler/crawler.py` | `magic=True` + `enable_stealth=True` Crawl4AI/Playwright 全自动反检测;`--disable-webrtc` 防止 IP 泄露;`delay_before_return_html=3s` 等待 Cloudflare 挑战;`remove_consent_popups=True` 自动关弹窗;`wait_until="domcontentloaded"` 避免被 Cloudflare 阻塞;增加列表页 URL 过滤(`most-popular-news`/`trending` 等) | -| `crawler/rss_crawler.py` | 新增 `_make_rss_client()` 读取 system.yaml 代理配置创建 httpx 客户端;Google News RSS 跳过 `article_url_pattern` 域名过滤 | -| `configs/sources.yaml` | investing.com:`rss_url` 切换为 Google News RSS 代理(绕过 Cloudflare),增加 `anti_bot_mode: stealth`;forexlive → investinglive 迁移 | -| `scripts/domestic_crawl_8g.sh` | `HTTP_PROXY`/`HTTPS_PROXY` 移至 `.env` 加载之前,确保 httpx 继承代理 | - -**测试结果(investing.com):** - -| 策略 | 结果 | -|------|------| -| Google News RSS(代理) | ✅ 25/25 篇,423 words/篇平均 | -| Web headful 首页 | ✅ 可加载,提取 ~25 个文章链接(15s) | -| Web headful 文章页 | ❌ 所有文章页仍被 Cloudflare 拦截(403) | - -### forexlive → investinglive 迁移 - -**问题:** `forexlive.com` 域名 301 重定向到 `investinglive.com`,`_is_valid_article_url` 域名过滤拒绝所有链接 - -**改动:** `sources.yaml` — 移除 `forexlive` 源,新增 `investinglive` 源 - -| 字段 | 旧值 | 新值 | -|------|------|------| -| id | `forexlive` | `investinglive` | -| name | ForexLive | InvestingLive | -| homepage | `forexlive.com/` | `investinglive.com/` | -| rss_url | `forexlive.com/feed` | `investinglive.com/feed` | -| article_url_pattern | `/(news\|technical-analysis\|Education)/` | `/(news\|central-banks\|commodities\|stocks\|technical-analysis)/` | -| max_articles | 20 | 25 | - -**测试结果:** ✅ `found=25 success=25`,100%,全文 3-9KB/篇(RSS 内嵌完整正文) - ---- - -## 当前进度 - -``` -M0 ✅ 项目骨架 -M1 ✅ 新闻抓取(13/13 源,Pi headful Playwright + HTTP 代理) -M2 ✅ 正文提取(trafilatura + MD 回退,增量跳过已处理) -M3 ✅ 三层去重(L1 URL / L2 Content / L3 SimHash,SQLite 指纹库) -M4 ✅ 翻译+事件抽取(DeepSeek v4-flash,事件去重 Prompt 约束) -M5 ✅ 向量生成(DashScope text-embedding-v3,1024 维) -M6 ✅ Qdrant 入库(本地文件模式,语义搜索验证) -M7 ✅ 全链路管道 + 日报(事件级去重 + AI 摘要分批合并 + 防截断) -M8 ✅ MCP 服务(FastMCP 5 个 Tool) -``` - ---- - -## 各源抓取状态 - -| 源 | 策略 | 状态 | 说明 | -|----|------|------|------| -| Reuters | Google News RSS | ⚠️ 摘要,RSS 稳定 | DataDome 反爬无法突破 | -| CNBC | CNBC RSS | ✅ 稳定 | 原生 RSS 正常 | -| MarketWatch | RSS + stealth | ✅ 稳定 | RSS 优先,stealth 回退 | -| FT | FT RSS | ⚠️ 摘要,RSS 稳定 | 付费墙,全文不可用 | -| Yahoo Finance | Yahoo RSS | ✅ 稳定 | RSS 正常 | -| **Investing.com** | **Google News RSS** + stealth | **✅ RSS 25/篇,Web ❌** | **2026-07-23 修复:RSS 改用 Google News 代理** | -| Seeking Alpha | RSS + stealth | ✅ 稳定 | RSS 稳定 | -| Barron's | Dow Jones RSS + stealth | ✅ 稳定 | 无原生 RSS,Dow Jones 兜底 | -| WSJ | Dow Jones RSS + stealth | ✅ 稳定 | RSS 正常 | -| Economist | RSS + stealth | ✅ 稳定 | RSS 正常 | -| **InvestingLive** | **RSS** | **✅ 25/篇 100%** | **2026-07-23 新增(forexlive 迁移)** | -| ZeroHedge | FeedBurner RSS | ✅ 稳定 | RSS 含全文 | -| ~~Forexlive~~ | ~~RSS 已失效~~ | ~~❌ 301→investinglive~~ | **已替换为 InvestingLive** | - ---- - -## 待办事项 - -1. **Investing.com 全文抓取** — Google News RSS 仅摘要,Web headful 仍被 Cloudflare 拦截(需要更好代理线路或住宅 IP) -2. **Reuters 全文** — 同上,DataDome 需要更先进的代理 -3. **FT 付费墙** — 需要付费订阅或放弃全文 -4. **海外服务器 crontab 停用** — 确认不再参与流水线后清理 -5. **Pi crontab 确认** — 当前 `logs/full.log` 显示调度正常运行(06/12/18/22 四个时间点) - ---- - -## 服务器状态 - -| 角色 | 地址 | 路径 | 状态 | -|------|------|------|------| -| 海外 | `ecs-user@8.217.19.253` | `/opt/intlgrab` | ⏸️ 不再参与流水线 | -| 国内 | `pi@192.168.1.160` | `/home/pi/intlnews` | ✅ 全链路独立运行 | - ---- - -## CLI 命令速查 - -``` -bash scripts/domestic_full.sh 全流程(M1→M6→日报) -bash scripts/domestic_crawl_8g.sh 仅 M1 headful(可跟源 ID) -bash scripts/domestic_crawl_8g.sh investing 单源测试 -PYTHONPATH=. .venv/bin/python3 -c "..." 直接调用 orchestrator -``` diff --git a/crawler/orchestrator.py b/crawler/orchestrator.py index 3be568d..65845bf 100644 --- a/crawler/orchestrator.py +++ b/crawler/orchestrator.py @@ -95,13 +95,15 @@ async def crawl_all_sources( # 串行执行每个源 for source in sources: + # sources_crawled 表示“已尝试的源数量”,包含成功和失败的源; + # 因此成功数 = sources_crawled - sources_failed。 + stats.sources_crawled += 1 result = await _crawl_single_source_with_storage(source) if isinstance(result, Exception): logger.error("源抓取异常: %s", result) stats.sources_failed += 1 else: - stats.sources_crawled += 1 stats.total_articles += result.total_success stats.results.append(result) if result.error: diff --git a/crawler/storage.py b/crawler/storage.py index 9304dc9..c66f85e 100644 --- a/crawler/storage.py +++ b/crawler/storage.py @@ -2,6 +2,7 @@ import json import logging +import re from pathlib import Path from crawler.models import ArticleItem, CrawlResult @@ -10,6 +11,22 @@ from crawler.utils import get_news_day logger = logging.getLogger(__name__) +def _infer_news_day(result: CrawlResult) -> str | None: + """从本次抓取结果中的文件路径推断新闻日 YYYYMMDD。 + + 优先使用成功文章的 html_path / md_path 中的日期目录, + 避免 crawl 跨天边界时 write_index 与文件落盘日期不一致。 + """ + for article in result.articles: + if article.status != "success": + continue + path = article.html_path or article.md_path or "" + m = re.search(r"/(\d{8})/", path) + if m: + return m.group(1) + return None + + def write_index_jsonl(result: CrawlResult) -> Path: """将单源抓取结果写入 index.jsonl @@ -19,7 +36,7 @@ def write_index_jsonl(result: CrawlResult) -> Path: Returns: index 文件路径 """ - today = get_news_day() + today = _infer_news_day(result) or get_news_day() out_dir = Path(f"data/raw/{result.source_id}/{today}") out_dir.mkdir(parents=True, exist_ok=True) diff --git a/dedup/pipeline.py b/dedup/pipeline.py index 135c80f..4892b36 100644 --- a/dedup/pipeline.py +++ b/dedup/pipeline.py @@ -10,7 +10,7 @@ from datetime import datetime from pathlib import Path from crawler.utils import get_news_day -from dedup.deduper import Deduper +from dedup.deduper import Deduper, article_to_fingerprint from dedup.models import DedupResult from extractor.models import ProcessedArticle @@ -59,7 +59,12 @@ def _load_processed_articles( continue try: data = json.loads(json_file.read_text(encoding="utf-8")) - articles.append(ProcessedArticle(**data)) + article = ProcessedArticle(**data) + # M2 可能写入 no_content / failed:这些文章没有有效正文, + # 不应进入去重、翻译、向量化等下游环节。 + if article.status != "success" or not article.content.strip(): + continue + articles.append(article) except (json.JSONDecodeError, Exception) as e: logger.warning("解析 processed JSON 失败 %s: %s", json_file, e) @@ -99,7 +104,9 @@ def dedup_source( dup_count = 0 for article in articles: - result = deduper.ingest(article) + # 先只读判断,不写指纹;等唯一文件落盘成功后再写指纹, + # 避免“指纹已入库但唯一文件未生成”导致该文章后续被当重复丢弃。 + result = deduper.check(article) if result.is_duplicate: dup_count += 1 @@ -121,6 +128,8 @@ def dedup_source( article.model_dump_json(indent=2, ensure_ascii=False), encoding="utf-8", ) + # 唯一文件落盘成功后再写指纹 + deduper.store.upsert(article_to_fingerprint(article)) logger.debug("[%s] ✅ %s (%d words)", source_id, article.title[:40], article.word_count) diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..7307ae8 --- /dev/null +++ b/docs/README.md @@ -0,0 +1,27 @@ +# 文档中心 + +本项目 **English Financial News**(`en-news`)是一套面向国际财经新闻的私有化 Deep Research 平台:抓取、去重、翻译、事件抽取、向量化、语义检索、日报生成全部在本地/Pi 上独立运行。 + +> 本目录是项目文档的唯一入口。历史 AI Agent 文档(`CLAUDE.md`、`continuation.md`、`english-news-plan.md`、旧版 `docs/intlnews_usage.*`)已清理合并到以下文档中。 + +## 文档导航 + +| 文档 | 内容 | +|------|------| +| [架构说明](architecture.md) | 总体架构、模块职责、数据流、部署拓扑 | +| [快速开始](quickstart.md) | 环境准备、安装、配置、首次运行 | +| [使用手册](usage.md) | CLI 命令、Shell 脚本、MCP 服务、日报查看 | +| [流水线详解](pipeline.md) | M1→M9 各步骤输入/输出、增量逻辑、幂等机制 | +| [配置说明](configuration.md) | `system.yaml`、`sources.yaml`、Profile、`.env` | +| [部署与运维](deployment.md) | Pi 服务器部署、Xvfb/代理、crontab、日志清理 | +| [开发指南](development.md) | 源码结构、测试、已知问题、新增新闻源 | +| [FAQ](faq.md) | 常见问题与排查 | + +## 顶层入口 + +- 项目根目录的 [README.md](../README.md) 仅保留简短的仓库入口和文档链接。 +- 当前推荐的执行入口: + - 全流程:`bash scripts/domestic_full.sh` + - 仅 M2→M6+日报:`bash scripts/pipeline.sh` + - CLI 单步:`uv run en-news ` + - MCP:`uv run en-news mcp-server` diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..856cac5 --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,171 @@ +# 项目架构说明 + +## 1. 定位 + +`en-news` 是一个从英文财经新闻源采集数据,经过正文提取、去重、LLM 翻译/事件抽取、向量化、向量检索,最终生成中文日报的私有化工作流。 + +设计目标: + +- **私有化**:不依赖海外服务器,抓取与计算均可在国内 Raspberry Pi 上独立运行。 +- **增量友好**:每个阶段都尽量跳过已处理文件,断点续跑成本低。 +- **幂等**:重复运行不会产生重复数据或重复日报。 +- **可运维**:终端输出阶段/模型/耗时,日志统一写入 `logs/`。 + +## 2. 总体架构 + +```text +┌─────────────────────────────────────────────────────────────┐ +│ en-news 平台 │ +│ │ +│ configs/ · .env · prompts/ · data/ · logs/ │ +│ │ +│ app/cli.py ── Typer CLI 入口 │ +│ ├── crawl M1 抓取 │ +│ ├── extract M2 正文提取 │ +│ ├── dedup M3 三层去重 │ +│ ├── translate M4 翻译 + 事件抽取 │ +│ ├── embed M5 向量生成 │ +│ ├── index M6 Qdrant 入库 │ +│ ├── search M6 语义搜索 │ +│ ├── report M9 日报结构化入库 │ +│ ├── pipeline M2→M6(+日报)一键管道 │ +│ └── mcp-server M8 MCP 服务 │ +│ │ +│ scheduler/ ── 管道编排 / 日报生成 │ +│ mcp_server/ ── MCP 工具层(研究 Agent 接入) │ +└─────────────────────────────────────────────────────────────┘ +``` + +## 3. 模块职责 + +| 模块 | 对应阶段 | 核心职责 | +|------|----------|----------| +| `crawler/` | M1 | 从 RSS/Atom/网页抓取新闻,RSS 优先、Stealth/Headful Web 回退,写入 `data/raw/` | +| `extractor/` | M2 | 使用 trafilatura(含 Markdown 回退)提取正文,输出 `data/processed/` | +| `dedup/` | M3 | URL、内容、SimHash 三层去重,SQLite 指纹库,跨源来源合并,输出 `data/deduped/` | +| `llm/` | M4 | 调用 DeepSeek/Qwen 兼容接口,单次调用完成全文英译中+投资事件抽取,输出 `data/events/` | +| `embedding/` | M5 | 调用 DashScope text-embedding 批量向量化,输出 `data/embeddings/` | +| `vectorstore/` | M6 | 管理 Qdrant collection,幂等 upsert、语义搜索、过滤查询 | +| `scheduler/` | M7/M9 | 串联 M2→M6,生成每日 AI 摘要日报并写入 MySQL | +| `mcp_server/` | M8 | 暴露 MCP 工具供 Cherry Studio/Claude Code 等调用 | +| `report_db/` | M9 | MySQL `news_report` / `news_event` 建表与读写 | +| `app/cli.py` | 入口 | Typer CLI,各模块的同步调用入口 | +| `scripts/` | 运维 | Bash 全流程、断点续跑、抓取 Profile、日志清理 | + +## 4. 数据流 + +```text +M1 抓取 +data/raw/{source_id}/{YYYYMMDD}/ + ├── index.jsonl # 抓取索引(每条含 url_hash、标题、URL) + ├── {url_hash}.html # 原始 HTML + └── {url_hash}.md # Crawl4AI 可选的 Markdown + +M2 正文提取 +data/processed/{source_id}/{YYYYMMDD}/{url_hash}.json + # ProcessedArticle:标题、正文、词数、提取器、时间等 + +M3 三层去重 +data/deduped/{YYYYMMDD}/uniques/{url_hash}.json + # 跨源重复时在已保留唯一篇的 source_ids 中追加来源 + +M4 翻译 + 事件抽取 +data/events/{YYYYMMDD}/{url_hash}.json + # EnTranslatedArticle:中英文标题/正文、事件列表、source_ids + +M5 向量生成 +data/embeddings/{YYYYMMDD}/{url_hash}.json + # 向量文件(1024 维 text-embedding-v3)+ 每日期 index.json + +M6 Qdrant 入库 +vectorstore client → collection: en_finance_news + # payload 含标题、双语标题、事件、来源、时间、正文预览 + +M7/M9 日报 +scheduler/reporter.py → MySQL: + news_report(主表:日期、type=intl、AI 摘要、stats) + news_event(明细:Top 事件、来源、情绪、重要度、URL) +``` + +## 5. 关键技术点 + +### 5.1 抓取策略 + +- **RSS 优先**:大部分源先用 RSS/Atom/Google News RSS,减少对浏览器的依赖。 +- **Web 回退**:RSS 失败/为空时,根据 `anti_bot_mode` 使用 stealth 或 headful Playwright。 +- **Profile 覆盖**:`EN_NEWS_PROFILE` 环境变量可切换 `8g_headful` / `2g_headless` 等抓取配置。 + +### 5.2 三层去重 + +| 层级 | 维度 | 说明 | +|------|------|------| +| L1 | URL | 相同 URL 直接判重 | +| L2 | 内容 | 正文 hash 相同判重 | +| L3 | SimHash | 汉明距离 ≤ 阈值判为模糊重复 | + +指纹库为 `data/dedup/fingerprints.sqlite3`,SimHash 窗口默认 30 天。 + +### 5.3 LLM 场景配置 + +`configs/system.yaml` 中通过 `llm_scenes` 为不同任务配置独立模型: + +| 场景 | 用途 | 当前默认 | +|------|------|----------| +| `translation` | M4 翻译+事件抽取 | deepseek-v4-flash,temperature=0.1,max_tokens=8192 | +| `daily_report` | M7 日报 AI 摘要 | deepseek-v4-flash,temperature=0.3,max_tokens=1500 | + +未在场景中声明的字段自动回退到 `llm` 默认段。Embedding 是单一场景,不能按任务拆分(查询/入库向量必须同模型)。 + +### 5.4 增量与幂等 + +| 阶段 | 增量机制 | +|------|----------| +| M2 | `processed/{source}/{date}/{url_hash}.json` 存在则跳过 | +| M3 | 指纹库判重;重复篇不再写入,只合并来源 | +| M4 | `events/{date}/{url_hash}.json` 存在则跳过 | +| M5 | `embeddings/{date}/{url_hash}.json` 存在则跳过 | +| M6 | Qdrant point ID = url_hash 的 UUID,重复 upsert 覆盖 | +| M9 | MySQL 唯一键 `(report_date, report_type='intl', file_name='')` 覆盖 | + +Shell 层还提供步骤级 `--resume`:状态写在 `data/run_state/{YYYYMMDD}.state`。 + +### 5.5 日报生成模式 + +- 时间窗口:过去 25 小时(跨天自动聚合)。 +- 高重要度事件:`importance >= 4`,不足 3 条时逐级降阈。 +- AI 摘要:事件超过 10 条自动分批生成,再合并;LLM 失败走规则兜底,不阻断日报入库。 +- 日报不再生成 HTML/上传,而是结构化写入 MySQL。 + +## 6. 部署拓扑(2026-07 起) + +```text +海外服务器(已停用) + ↓ 历史:抓取后同步回国内 +国内服务器 Pi(当前唯一全链路节点) + /home/pi/intlnews + ├── Xvfb :99(headful 浏览器虚拟显示) + ├── Privoxy → SOCKS5(HTTP 代理访问海外) + ├── crontab: 06:00 / 12:00 / 18:00 / 22:00 执行 domestic_full.sh + ├── 本地 Qdrant 文件模式: data/qdrant_storage + └── MySQL: 通过内网隧道访问 news 项目共用 myquant 库 +``` + +## 7. 关键目录 + +```text +app/ CLI 入口 +crawler/ M1 抓取 +extractor/ M2 正文提取 +dedup/ M3 去重 +llm/ M4 翻译/事件抽取 +embedding/ M5 向量化 +vectorstore/ M6 Qdrant +scheduler/ M7/M9 编排与日报 +report_db/ M9 MySQL +mcp_server/ M8 MCP +prompts/ LLM Prompt 模板 +configs/ 系统/源/Profile 配置 +scripts/ Bash 自动化脚本 +tests/ 单元测试 +docs/ 本文档目录 +``` diff --git a/docs/configuration.md b/docs/configuration.md new file mode 100644 index 0000000..889142d --- /dev/null +++ b/docs/configuration.md @@ -0,0 +1,238 @@ +# 配置说明 + +## 1. 配置文件总览 + +| 文件 | 用途 | +|------|------| +| `configs/system.yaml` | 系统功能配置:代理、抓取、提取、去重、LLM、Embedding、Qdrant、日报、调度、日志 | +| `configs/sources.yaml` | 新闻源列表与抓取规则 | +| `configs/profiles/8g_headful.yaml` | Pi 8G 有头浏览器 + HTTP 代理配置 | +| `configs/profiles/2g_headless.yaml` | 2G 轻量 headless stealth 配置 | +| `.env` | 敏感配置:API Key、数据库密码、远程 Qdrant 地址 | +| `.env.example` | 环境变量模板 | +| `prompts/translation_and_extraction.md` | M4 翻译+事件抽取 Prompt 模板 | + +## 2. `.env` 环境变量 + +```env +# DeepSeek(M4 / M7 默认 LLM) +DEEPSEEK_API_KEY=sk-your-deepseek-key +DEEPSEEK_BASE_URL=https://api.deepseek.com + +# Qwen(可选,切换 LLM provider 时使用) +QWEN_API_KEY=sk-your-qwen-key +QWEN_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 + +# DashScope(M5 Embedding) +# 使用 OpenAI 兼容端点,默认复用 QWEN_BASE_URL;如需独立端点可设置 DASHSCOPE_EMBEDDING_BASE_URL +DASHSCOPE_API_KEY=sk-your-dashscope-key +# DASHSCOPE_EMBEDDING_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 + +# Qdrant(可选;默认本地文件模式) +QDRANT_URL=http://localhost:6333 +QDRANT_API_KEY= + +# MySQL(M9 日报入库) +NEWS_DB_HOST=192.168.1.10 +NEWS_DB_PORT=13306 +NEWS_DB_USER=myquant +NEWS_DB_PASSWORD= +NEWS_DB_NAME=myquant +``` + +## 3. `configs/system.yaml` + +### 3.1 servers + +```yaml +servers: + overseas_host: "ecs-user@8.217.19.253" + overseas_path: "/opt/intlgrab" + domestic_host: "pi@192.168.1.160" + domestic_path: "/home/pi/intlnews" +``` + +海外服务器已停用,字段保留为兼容参考。 + +### 3.2 proxy + +```yaml +proxy: + enabled: false + url: "http://127.0.0.1:3128" + bypass_domains: [] +``` + +- 默认关闭。 +- `8g_headful` Profile 会开启并通过 `http://127.0.0.1:3128`(Privoxy → SOCKS5)访问海外。 + +### 3.3 crawler + +| 字段 | 默认 | 说明 | +|------|------|------| +| `max_memory_mb` | 1800 | 内存上限,超限触发 GC/保护逻辑 | +| `source_timeout_sec` | 7200 | 单源超时 | +| `article_delay_sec` | 3.0 | 文章间冷却 | +| `page_timeout_sec` | 45 | 单页加载超时 | +| `viewport_width/height` | 1024/768 | 浏览器视口 | +| `headful` | false | true=有头浏览器 | +| `xvfb_display` | ":99" | Xvfb 虚拟显示器 | + +### 3.4 extractor / dedup + +```yaml +extractor: + min_content_words: 50 + trafilatura_fallback: true + +dedup: + hamming_distance_threshold: 3 + simhash_window_days: 30 + min_content_length: 100 +``` + +### 3.5 llm 与 llm_scenes + +```yaml +llm: + provider: "deepseek" + deepseek_model: "deepseek-v4-flash" + qwen_model: "qwen-plus" + timeout_sec: 60 + max_attempts: 3 + max_tokens: 8192 + temperature: 0.1 + concurrency: 3 + +llm_scenes: + translation: + provider: "deepseek" + model: "deepseek-v4-flash" + temperature: 0.1 + max_tokens: 8192 + daily_report: + provider: "deepseek" + model: "deepseek-v4-flash" + temperature: 0.3 + max_tokens: 1500 +``` + +- `translation`:M4 使用。 +- `daily_report`:M7 日报摘要使用。 +- 场景未声明的字段会回退 `llm`。 + +### 3.6 embedding + +```yaml +embedding: + provider: "dashscope" + dashscope_model: "text-embedding-v3" + dimension: 1024 + batch_size: 10 + max_attempts: 3 + timeout_sec: 30 +``` + +注意:Embedding 不支持按场景拆分配置,因为入库向量与查询向量必须同模型。 + +### 3.7 qdrant / report / schedule / logging + +```yaml +qdrant: + collection: "en_finance_news" + +report: + max_events: 20 + summary_max_chars: 500 + importance_threshold: 4 + upload_host: "simon@doorcome.cn" + upload_path: "/var/www/html/echart/research" + +schedule: + day_cutoff_hour: 6 + times: ["06:00", "12:00", "18:00", "22:00"] + +logging: + level: "INFO" + dir: "logs" +``` + +`report.upload_host/upload_path` 是 M9 之前 HTML 日报上传的旧配置,当前仅作保留。 + +## 4. `configs/sources.yaml` + +### 4.1 顶层 settings + +```yaml +settings: + concurrency: 5 + request_delay_sec: 2 + user_agent: "..." +``` + +### 4.2 单源字段 + +| 字段 | 说明 | +|------|------| +| `id` | 唯一标识,如 `reuters` | +| `name` | 展示名,如 `Reuters` | +| `enabled` | 是否启用 | +| `homepage` | 主页,用于回退/域名校验 | +| `article_url_pattern` | URL 正则,过滤非文章链接 | +| `js_render` | 是否需要 JS 渲染 | +| `max_articles_per_run` | 单次最多抓取篇数 | +| `rss_url` | RSS/Atom/Google News RSS 地址 | +| `anti_bot_mode` | `stealth` / `headful` 等 Web 回退模式 | + +### 4.3 当前源列表 + +| ID | 名称 | 抓取方式 | 状态 | +|----|------|----------|------| +| `reuters` | Reuters | Google News RSS(可能只有摘要) | ⚠️ DataDome | +| `cnbc` | CNBC | CNBC RSS | ✅ | +| `marketwatch` | MarketWatch | RSS + stealth 回退 | ✅ | +| `ft` | Financial Times | FT RSS(摘要) | ⚠️ 付费墙 | +| `yahoo_finance` | Yahoo Finance | Yahoo RSS | ✅ | +| `investing` | Investing.com | Google News RSS + stealth | ✅(摘要) | +| `seekingalpha` | Seeking Alpha | RSS + stealth | ✅ | +| `barrons` | Barron's | Dow Jones RSS + stealth | ✅ | +| `wsj` | WSJ | Dow Jones RSS + stealth | ✅ | +| `economist` | The Economist | RSS + stealth | ✅ | +| `investinglive` | InvestingLive | RSS(全文) | ✅ | +| `zerohedge` | ZeroHedge | FeedBurner RSS(全文) | ✅ | + +> 历史 `forexlive` 已迁移为 `investinglive`。 + +## 5. Profiles + +### 5.1 `8g_headful.yaml` + +适合国内 Pi(内存充裕): +- 开启 HTTP 代理 +- `headful: true` +- 更大视口/超时 +- 通过 Xvfb `:99` 运行 + +### 5.2 `2g_headless.yaml` + +适合低配/海外轻量: +- headless stealth +- 内存上限 1800MB +- 默认不开启代理 + +通过环境变量切换: + +```bash +EN_NEWS_PROFILE=8g_headful uv run en-news crawl +# 或直接使用对应脚本 +bash scripts/domestic_crawl_8g.sh +``` + +## 6. Prompt 模板 + +`prompts/translation_and_extraction.md` 包含: + +- `## System Prompt`:LLM System 指令 +- `## User Input`:带 `{title}`、`{source_name}`、`{publish_time}`、`{content}` 占位符 + +如需调整事件类型、输出格式或翻译风格,应修改此文件并同步测试。 diff --git a/docs/deployment.md b/docs/deployment.md new file mode 100644 index 0000000..e429131 --- /dev/null +++ b/docs/deployment.md @@ -0,0 +1,145 @@ +# 部署与运维 + +## 1. 目标环境 + +当前生产环境为国内单节点(Raspberry Pi 4, 8GB),路径 `/home/pi/intlnews`。海外服务器已停用。 + +```text +Pi /home/pi/intlnews +├── Python 3.11 + uv + .venv +├── Playwright(Chromium) +├── Xvfb :99 +├── Privoxy 127.0.0.1:3128 → ss-local SOCKS5 1088 +├── 本地 Qdrant 文件模式 data/qdrant_storage +└── crontab 定时 full pipeline +``` + +## 2. 系统依赖 + +```bash +sudo apt update +sudo apt install -y python3.11 python3.11-venv uv xvfb \ + shadowsocks-libev privoxy chromium # chromium 可根据 Playwright 安装方式调整 +``` + +> 本项目使用 `uv` 管理 Python 环境,不依赖系统 pip 安装依赖。Playwright 浏览器建议按 `crawl4ai` 文档安装。 + +## 3. 首次部署步骤 + +```bash +# 1. 拉取/同步代码到 /home/pi/intlnews +cd /home/pi/intlnews + +# 2. 安装 Python 依赖 +uv sync + +# 3. 准备 .env +cp .env.example .env +# 编辑 .env:DeepSeek / DashScope / MySQL 等 + +# 4. 启动 Xvfb(headful 抓取需要) +Xvfb :99 -screen 0 1280x1024x24 -ac +extension RANDR & + +# 5. 启动代理(如果国内访问海外需要) +sudo systemctl enable --now shadowsocks-libev-local +sudo systemctl enable --now privoxy +# 验证代理 +curl -x http://127.0.0.1:3128 -I https://www.google.com +``` + +## 4. 代理配置参考 + +- Shadowsocks 配置文件:`/etc/shadowsocks-libev/config.json` +- Privoxy 配置追加: + +``` +forward-socks5t 127.0.0.1:1088 . +listen-address 127.0.0.1:3128 +``` + +- `domestic_crawl_8g.sh` 会自动 `export HTTP_PROXY/HTTPS_PROXY` 到 `127.0.0.1:3128`,并使用 `8g_headful` Profile。 + +## 5. Xvfb 管理 + +```bash +# 查看是否运行 +pgrep -x Xvfb + +# 手动启动 +Xvfb :99 -screen 0 1280x1024x24 -ac +extension RANDR & + +# 开机自启(示例 systemd) +# 也可由 domestic_crawl_8g.sh 自动检测并启动 +``` + +## 6. 定时任务(crontab) + +推荐 crontab: + +```cron +0 6 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/cron_06.log 2>&1 +0 12 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/cron_12.log 2>&1 +0 18 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/cron_18.log 2>&1 +0 22 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/cron_22.log 2>&1 +``` + +如果希望中断恢复语义,可以在每次调度前追加 `--resume`?注意:`--resume` 通常用于手动恢复;日常定时全量不带 `--resume` 也能靠文件级增量避免重复处理。按需选择。 + +## 7. MySQL / 日报依赖 + +M9 日报写入共用 MySQL `myquant` 库,需要以下网络配置: + +- Pi 通过内网访问 `192.168.1.10:13306`(该端口是到远程 MySQL 的 autossh 隧道) +- 用户名/库名通常为 `myquant` / `myquant` +- `.env` 中必须配置 `NEWS_DB_PASSWORD` + +验证: + +```bash +uv run en-news report +``` + +成功会输出 `report_id=...`。 + +## 8. 日志与清理 + +- 日志目录:`logs/` + - `domestic_full_*.log` + - `pipeline_*.log` +- 保留策略:默认保留 14 天,执行 `scripts/cleanup_logs.sh`。 +- 可加入 crontab: + +```cron +30 3 * * * cd /home/pi/intlnews && bash scripts/cleanup_logs.sh >> /dev/null 2>&1 +``` + +## 9. 升级/同步代码 + +```bash +cd /home/pi/intlnews +# 拉取新代码 +git pull # 如果使用 git +# 同步依赖 +uv sync +# 执行一次冒烟 +uv run en-news --help +# 查看关键日志 +tail -100 logs/pipeline_*.log | tail -100 +``` + +## 10. 故障排查指引 + +| 现象 | 可能原因 | 处理 | +|------|----------|------| +| 抓取全部失败 | 代理未启动 / 网络不通 | `curl -x http://127.0.0.1:3128 ...` | +| headful 起不来 | Xvfb 未运行 / DISPLAY 不对 | 启动 Xvfb,`DISPLAY=:99` | +| M4 翻译失败 | DeepSeek Key 未配置 / 余额不足 | 检查 `.env`,单独跑 `uv run en-news translate` | +| M5 失败 | DashScope Key 未配置 | 检查 `.env` | +| report 失败 | `NEWS_DB_PASSWORD` 缺失 / 隧道不通 | 检查 `.env` 和 MySQL 端口 | +| Qdrant 入库失败 | 本地文件损坏 / collection 异常 | 备份后重建 `data/qdrant_storage`,`uv run en-news index --recreate` | + +## 11. 容量建议 + +- `data/` 会随时间增长,建议定期归档旧数据。 +- 日志 `logs/` 用 `cleanup_logs.sh` 清理。 +- Qdrant 本地文件模式对 Pi 友好,但大规模检索建议迁移到独立服务器/远程 Qdrant。 diff --git a/docs/development.md b/docs/development.md new file mode 100644 index 0000000..79cef8e --- /dev/null +++ b/docs/development.md @@ -0,0 +1,101 @@ +# 开发指南 + +## 1. 代码结构 + +```text +app/cli.py Typer CLI 入口 +crawler/ M1 抓取 + config.py 系统配置 + Profile 深度合并 + orchestrator.py 串行抓取编排、RSS 优先 Web 回退 + rss_crawler.py RSS/Atom/Google News RSS 抓取 + crawler.py Crawl4AI Web 抓取 + loader.py 读取 sources.yaml + storage.py index.jsonl 存储 +extractor/ M2 正文提取 +dedup/ M3 三层去重 +llm/ M4 LLM 翻译/事件抽取 +embedding/ M5 向量化 +vectorstore/ M6 Qdrant +scheduler/ M7/M9 管道编排与日报 +report_db/ M9 MySQL +mcp_server/ M8 MCP +prompts/ LLM Prompt 模板 +scripts/ Bash 自动化与运维 +tests/ 单元测试 +docs/ 项目文档 +``` + +## 2. 常用开发命令 + +```bash +# 安装开发依赖 +uv sync --extra dev + +# 运行全部测试 +uv run pytest + +# 运行单个测试文件 +uv run pytest tests/test_llm.py + +# 代码检查 +uv run ruff check . + +# 自动修复 +uv run ruff check . --fix + +# 运行 CLI +uv run en-news --help +``` + +## 3. 测试现状 + +- 当前测试覆盖:crawler、extractor、dedup、llm、embedding、vectorstore、scheduler、report_db、mcp_server。 +- 已知问题: + - `tests/test_crawler.py::test_write_and_load_index_jsonl` + - `tests/test_crawler.py::test_write_index_jsonl_dedup` + - 原因是 `crawler/storage.py:load_index()` 硬编码 `data/raw/...`,未使用测试临时目录。 +- `scheduler/reporter.py` 与 `report_db/schema.py` 因内含长模板/DDL,已在 `pyproject.toml` 中忽略 E501。 + +## 4. 新增新闻源 + +1. 在 `configs/sources.yaml` 的 `sources:` 列表新增条目: + ```yaml + - id: "example" + name: "Example News" + enabled: true + homepage: "https://www.example.com/" + article_url_pattern: "/news/" + js_render: false + max_articles_per_run: 20 + rss_url: "https://www.example.com/rss" + anti_bot_mode: "stealth" + ``` +2. 本地测试: + ```bash + uv run en-news crawl --source example + uv run en-news extract --source example + ``` +3. 确认 `data/raw/example/...` 和 `data/processed/example/...` 正常。 +4. 如果使用 Google News RSS,需确认 `rss_crawler.py` 能正确清理标题和识别跳转链接。 + +## 5. LLM Prompt 开发 + +M4 Prompt 位于 `prompts/translation_and_extraction.md`。 + +- 不要随意改变 JSON 输出结构,否则 `llm/models.py::LLMTranslationOutput` 可能校验失败。 +- 事件类型定义在 `llm/models.py::INTERNATIONAL_EVENT_TYPES`,Prompt 中应保持一致。 +- 修改后建议跑 `uv run pytest tests/test_llm.py`。 + +## 6. 数据模型核心字段 + +- `ProcessedArticle`(`extractor/models.py`):包含 `source_id`、`source_ids`、标题、URL、正文、词数等。 +- `EnTranslatedArticle`(`llm/models.py`):包含中英文标题/正文、`events`、`source_ids`。 +- `EventExtraction`(`llm/models.py`):事件类型、代码、情绪、重要度、摘要。 +- `ReportData` / `EventRow`(`report_db/models.py`):日报入库结构。 + +## 7. 文档维护规范 + +- 所有面向使用/部署/开发的 Markdown 文档统一放在 `docs/`。 +- 根目录 `README.md` 仅保留仓库入口和文档链接。 +- 新增功能或修改流程时,同步更新对应 `docs/*.md`。 +- 不再在仓库根目录维护 `CLAUDE.md` / `continuation.md` / `english-news-plan.md` 这类 AI Agent 会话残留。 diff --git a/docs/faq.md b/docs/faq.md new file mode 100644 index 0000000..8a5102d --- /dev/null +++ b/docs/faq.md @@ -0,0 +1,104 @@ +# FAQ 常见问题 + +## 1. 为什么有些源只拿到摘要? + +部分网站存在 Cloudflare / DataDome / 付费墙等反爬限制: + +- RSS 源只返回标题+摘要(如 Reuters、Investing.com、FT) +- Web 全文抓取可能被拦截 +- 当前策略是 RSS 优先 + stealth/headful Web 回退,但无法保证 100% 全文 + +解决方法: +- 使用质量更好的代理/住宅 IP +- 订阅相应媒体付费内容 +- 对 Pipeline 而言,摘要也能进入翻译/抽取,只是信息量较低 + +## 2. 日报存在哪里? + +M9 以后日报不再生成 HTML,而是结构化写入 MySQL: + +- `news_report`:日报主表(`report_type='intl'`) +- `news_event`:事件明细(`section='intl'`) + +历史 HTML 日报(2026-06-16 ~ 2026-08-03)仍保留在 `data/reports/` 或已由 news 项目导入。 + +## 3. 如何断点续跑? + +```bash +bash scripts/domestic_full.sh --resume +bash scripts/pipeline.sh --resume +``` + +- 步骤状态:`data/run_state/{YYYYMMDD}.state` +- 失败步骤不会标记,`--resume` 会重跑失败步骤 +- 各步骤还有文件级增量,重复运行不会重复处理 + +## 4. 如何添加新源? + +编辑 `configs/sources.yaml`,参考现有源增加一条配置,然后: + +```bash +uv run en-news crawl --source +uv run en-news extract --source +``` + +详细字段见 [配置说明](configuration.md)。 + +## 5. 为什么 translate 或 report 报 API Key/MySQL 错误? + +检查: + +```bash +# 是否已加载 .env +grep -E 'DEEPSEEK|DASHSCOPE|NEWS_DB' .env | sed 's/=.*/=***/' + +# DeepSeek +uv run en-news translate + +# MySQL +uv run en-news report +``` + +CLI 入口会 `load_dotenv()`,Shell 脚本也会加载 `.env`。 + +## 6. Qdrant 本地文件模式和远程模式如何选择? + +- 默认本地文件模式:`data/qdrant_storage`,零运维,适合 Pi 单机。 +- 远程模式:`.env` 配置非 localhost 的 `QDRANT_URL` 和 `QDRANT_API_KEY`,适合多端共享。 + +切换后需要确保 collection 中向量由同一 embedding 模型生成。 + +## 7. 是否可以只跑管道不生成日报? + +可以: + +```bash +uv run en-news pipeline --skip-report +``` + +注意 `scripts/pipeline.sh` 目前固定包含日报;如果 MySQL 未配置,可用 CLI `--skip-report` 或改用 `uv run en-news translate`、`embed`、`index` 手动执行。 + +## 8. 为什么日志里出现“AI 大模型(场景 ...)”? + +这是项目刻意保留的可观测性输出: + +- M4 会打印 `AI 大模型(场景 translation)` +- 日报会打印 `AI 大模型(场景 daily_report)` +- M5 会打印 Embedding 初始化信息 + +方便在日志中确认实际调用的供应商/模型。 + +## 9. 历史 AI Agent 文档去哪了? + +已清理: + +- 根目录 `CLAUDE.md` +- 根目录 `continuation.md` +- 根目录 `english-news-plan.md` +- 旧版 `docs/intlnews_usage.html` / `docs/intlnews_usage.md` + +所有有效内容已整理进当前 `docs/` 文档树。 + +## 10. 测试有失败是否影响生产? + +当前已知 2 个 `test_crawler.py` 失败,属于测试路径问题,不影响生产管道。建议后续修复 `crawler/storage.py:load_index()` 对测试临时目录的适配。 diff --git a/docs/intlnews_usage.html b/docs/intlnews_usage.html deleted file mode 100644 index 5720f14..0000000 --- a/docs/intlnews_usage.html +++ /dev/null @@ -1,390 +0,0 @@ - - - - - -IntlNews 操作手册 - - - -
- -

📰 IntlNews — 国际财经新闻 Deep Research

-

最后更新:2026-07-23 | 运行服务器:pi@192.168.1.160

- -
-项目定位:面向国际财经新闻的私有化 Deep Research 平台。抓取英文财经新闻 → 英译中 → 事件抽取 → 向量入库 → MCP Agent 深度研究。 -
- - -

一、系统架构

-Arch - -
-┌──────────────────────────────────────────────────────┐
-│                   Pi 4 (8GB)                          │
-│  ┌──────────┐   ┌──────────┐   ┌──────────────────┐  │
-│  │ M1 抓取   │──▶│ M2 正文  │──▶│ M3 三层去重      │  │
-│  │ headful   │   │ trafilat.│   │ URL/Content/SimH │  │
-│  │ Playwright│   │          │   │                  │  │
-│  │ + HTTP 代 │   │          │   │                  │  │
-│  │ 理(海外)  │   │          │   │                  │  │
-│  └──────────┘   └──────────┘   └────────┬─────────┘  │
-│                                         ▼            │
-│  ┌──────────┐   ┌──────────┐   ┌──────────────────┐  │
-│  │ M6 Qdrant│◀──│ M5 向量  │◀──│ M4 翻译+事件抽取  │  │
-│  │ 入库     │   │ DashScope│   │ DeepSeek v4      │  │
-│  └────┬─────┘   └──────────┘   └──────────────────┘  │
-│       │                                               │
-│       ▼                                               │
-│  ┌──────────┐   ┌──────────┐                          │
-│  │ M7 日报  │──▶│ doorcome │                          │
-│  │ 生成     │   │ 上传     │                          │
-│  └──────────┘   └──────────┘                          │
-│       │                                               │
-│       ▼                                               │
-│  ┌────────────────────────────────────────────────┐   │
-│  │ M8 MCP 服务(Cherry Studio / Claude Code)      │   │
-│  └────────────────────────────────────────────────┘   │
-└──────────────────────────────────────────────────────┘
-
- -
-关键变更:海外服务器已于 2026-07-14 停用。全链路在 Pi 独立运行,通过 Privoxy → SOCKS5 (ss-local) 代理访问海外新闻网站。
-反爬策略:2026-07-23 更新 — 启用 Crawl4AI magic 全自动反检测 + Playwright stealth 模式。investing.com 改用 Google News RSS 代理。 -
- - -

二、部署指南

- -

2.1 环境要求

-
    -
  • Python 3.11 + uv
  • -
  • Xvfb(headful Playwright 虚拟显示器)
  • -
  • Shadowsocks-libev + v2ray-plugin(SOCKS5 代理)
  • -
  • Privoxy(HTTP → SOCKS5 转换)
  • -
- -

2.2 初始化安装

-
-# 1. 克隆仓库
-git clone ssh://gitea:2222/simon/intl_news /home/pi/intlnews
-cd /home/pi/intlnews
-
-# 2. Python 环境
-uv sync
-
-# 3. 环境变量
-cp .env.example .env
-# 填入: DEEPSEEK_API_KEY, DASHSCOPE_API_KEY, QDRANT_URL
-
-# 4. Xvfb 虚拟显示器
-sudo apt install xvfb
-Xvfb :99 -screen 0 1280x1024x24 -ac +extension RANDR &
-
-# 5. Shadowsocks + Privoxy(海外代理)
-sudo apt install shadowsocks-libev privoxy
-# 配置 /etc/shadowsocks-libev/config.json
-# 配置 /etc/privoxy/config: forward-socks5t / 127.0.0.1:1088 .
-sudo systemctl enable --now shadowsocks-libev-local
-sudo systemctl enable --now privoxy
-
- -

2.3 定时调度(crontab)

-
-# crontab -e 添加:
-0 6  * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/full.log 2>&1
-0 12 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/full.log 2>&1
-0 18 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/full.log 2>&1
-0 22 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh >> logs/full.log 2>&1
-@reboot Xvfb :99 -screen 0 1280x1024x24 -ac +extension RANDR &
-
- - -

三、操作命令

- -

3.1 全流程运行

-
-# 一键全流程(M1→M6→日报)
-bash scripts/domestic_full.sh
-
- -

3.2 M1 抓取

-
-# 抓取全部 13 个源
-bash scripts/domestic_crawl_8g.sh
-
-# 单源调试
-bash scripts/domestic_crawl_8g.sh investing
-bash scripts/domestic_crawl_8g.sh investinglive
-bash scripts/domestic_crawl_8g.sh reuters
-
- -

3.3 M2→M6 管道

-
-# 完整管道(需要先有 raw 数据)
-bash scripts/pipeline.sh
-
-# 或逐步执行:
-PYTHONPATH=. .venv/bin/python3 -c "from scheduler.pipeline import run_pipeline; run_pipeline('$(date +%Y%m%d)')"
-
- -

3.4 日报

-
-# 生成当日日报
-PYTHONPATH=. .venv/bin/python3 -c "from scheduler.reporter import generate_report; print(generate_report())"
-
-# 查看日报列表
-ls -lt data/reports/
-
- -

3.5 日志查看

-
-# 全量日志
-tail -100 logs/full.log
-
-# 筛选指定源
-grep 'investing\|investinglive' logs/full.log | tail -20
-
- - -

四、新闻源状态

- - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
源ID策略日产量状态
ReutersreutersGoogle News RSS~30 篇摘要 DataDome
CNBCcnbcCNBC RSS~25 篇正常
MarketWatchmarketwatchRSS + stealth~25 篇正常
Financial TimesftFT RSS~20 篇摘要 付费墙
Yahoo Financeyahoo_financeYahoo RSS~30 篇正常
Investing.cominvestingGoogle News RSS + stealth25 篇已修复 RSS 代理
Seeking AlphaseekingalphaRSS + stealth~20 篇正常
Barron'sbarronsDow Jones RSS + stealth~20 篇正常
WSJwsjDow Jones RSS + stealth~20 篇正常
EconomisteconomistRSS + stealth~15 篇正常
InvestingLiveinvestingliveRSS(全文)25 篇2026-07-23 新增
ZeroHedgezerohedgeFeedBurner RSS~15 篇正常
ForexLiveforexlive——已移除 301→investinglive
- -
-⚠️ 注意:RSS 摘要源的全文依赖 headful Playwright 回退抓取,成功率取决于代理线路质量。investing.com 因 Cloudflare 反爬,全文抓取间歇性失败。 -
- - -

五、数据存储

- -
-/home/pi/intlnews/
-├── data/
-│   ├── raw/{source_id}/{yyyymmdd}/     M1 原始抓取(.html + .md + index.jsonl)
-│   ├── processed/{source_id}/           M2 正文提取结果
-│   ├── dedup/                            M3 去重指纹库(SQLite)
-│   ├── translations/{yyyymmdd}/         M4 翻译+事件抽取(JSONL)
-│   ├── embeddings/{yyyymmdd}/           M5 向量文件(JSONL)
-│   ├── reports/                          M7 日报(HTML)
-│   └── vectorstore/                      M6 Qdrant 本地数据
-├── configs/
-│   ├── sources.yaml                     新闻源定义(13 个源)
-│   ├── system.yaml                      系统参数
-│   └── profiles/8g_headful.yaml         Pi 抓取配置
-├── logs/full.log                        运行日志
-└── prompts/                             LLM Prompt 模板
-
- - -

六、MCP 服务

- -

M8 模块提供 MCP(Model Context Protocol)服务,供 Cherry Studio / Claude Code / Cursor 等客户端接入。

- -

6.1 启动

-
-cd /home/pi/intlnews
-PYTHONPATH=. .venv/bin/python3 -m mcp_server.server
-
- -

6.2 可用工具

- - - - - - - -
Tool参数功能
en_news_searchquery, top_k语义搜索知识库
en_news_filtersource_id, date, sentiment按条件筛选事件
en_news_trendingdays获取近期热点
en_news_report—获取最新日报
en_news_daily_brief—一键生成简报
- - -

七、配置参考

- -

7.1 环境变量(.env)

- - - - - -
变量用途
DEEPSEEK_API_KEYDeepSeek LLM(翻译+事件抽取)
DASHSCOPE_API_KEYDashScope Embedding(向量生成)
QDRANT_URLQdrant 服务地址
- -

7.2 新闻源配置(sources.yaml)

-

每个源包含:id, name, homepage, rss_url, article_url_pattern, anti_bot_mode 等。

-

新增源时参考现有格式,启用/禁用通过 enabled: true/false。

- -

7.3 Profile 配置

-

Pi 服务器使用 8g_headful profile(环境变量 EN_NEWS_PROFILE),通过 Privoxy HTTP 代理访问海外网站。

-
    -
  • proxy.enabled: true — 启用 HTTP 代理
  • -
  • crawler.headful: true — 有头浏览器 + Xvfb
  • -
  • crawler.magic: true — Crawl4AI 全自动反检测
  • -
  • crawler.enable_stealth: true — Playwright stealth
  • -
- - -

八、常见排障

- - - - - - - - - - - - - - - - - - - - - - - -
现象排查步骤
抓取 0 篇 - 1. 检查 logs/full.log 看 RSS/Web 错误
- 2. bash scripts/domestic_crawl_8g.sh <source_id> 单源测试
- 3. 检查代理是否正常:curl -I --proxy 127.0.0.1:3128 https://example.com -
RSS 超时 - 1. 确认 Shadowsocks 和 Privoxy 在运行
- 2. systemctl status shadowsocks-libev-local privoxy
- 3. 环境变量 HTTP_PROXY 是否设置 -
Cloudflare 拦截 - 1. 已启用 magic + stealth 自动反检测
- 2. 仍失败则需更换代理 IP(数据中心 IP 被标记)
- 3. 改用 Google News RSS 代理(已为 investing/reuters 启用) -
日报无数据 - 1. 检查 data/translations/{date}/ 是否有 JSONL
- 2. 检查 data/embeddings/{date}/ 是否有向量 -
MCP 连接失败 - 1. 确认 MCP 服务进程在运行
- 2. 检查端口监听:ss -tlnp | grep 8080 -
- - -

九、变更记录

- - - - - - - - - - - -
日期变更
2026-07-23 - 反爬策略增强:
- - crawler.py: magic/stealth/--disable-webrtc/remove_consent_popups
- - rss_crawler.py: httpx 代理支持
- - investing.com: Google News RSS 代理(绕过 Cloudflare)
- - forexlive → investinglive 迁移(域名已 301 重定向) -
2026-07-14 - - 海外服务器停用,Pi 全链路独立运行
- - 8g_headful profile 新增(headful + HTTP 代理)
- - Shadowsocks + Privoxy 部署 -
- - - -
- - diff --git a/docs/intlnews_usage.md b/docs/intlnews_usage.md deleted file mode 100644 index c01d582..0000000 --- a/docs/intlnews_usage.md +++ /dev/null @@ -1,672 +0,0 @@ -# 国际财经 Deep Research 平台 — 使用手册 - -> 版本:v1.0 -> 最后更新:2026-06-21 -> 项目路径:国内 `/home/pi/intlnews` / 海外 `/opt/intlgrab` - ---- - -## 目录 - -1. [项目概述](#1-项目概述) -2. [部署拓扑](#2-部署拓扑) -3. [环境配置](#3-环境配置) -4. [CLI 命令参考](#4-cli-命令参考) -5. [M1 — 新闻抓取](#5-m1--新闻抓取) -6. [M2 — 正文提取](#6-m2--正文提取) -7. [M3 — 三层去重](#7-m3--三层去重) -8. [M4 — 翻译 + 事件抽取](#8-m4--翻译--事件抽取) -9. [M5 — 向量生成](#9-m5--向量生成) -10. [M6 — Qdrant 入库与检索](#10-m6--qdrant-入库与检索) -11. [M7 — 全链路管道 + 日报](#11-m7--全链路管道--日报) -12. [M8 — MCP 服务](#12-m8--mcp-服务) -13. [数据目录结构](#13-数据目录结构) -14. [定时任务](#14-定时任务) -15. [常见问题](#15-常见问题) - ---- - -## 1. 项目概述 - -本项目构建面向**国际英文财经新闻**的私有化 Deep Research 平台。 - -核心能力: - -- 英文财经新闻抓取(Crawl4AI,12 个源) -- 正文提取(trafilatura) -- 全文英译中(DeepSeek LLM) -- 投资事件抽取(美股代码识别 + 情绪判断 + 重要度评分) -- 双语向量知识库(Qdrant,1024 维) -- 语义检索(中文自然语言) -- MCP 服务(Claude Code / Cherry Studio Agent 深度研究) -- 每日 AI 摘要日报(HTML) - -**本项目不是**:交易系统 / 股票预测系统 / 投资顾问系统。 - ---- - -## 2. 部署拓扑 - -``` -┌──────────────────────────────────────────────────┐ -│ Overseas Server (海外) │ -│ M1 Crawl4AI 抓取 → data/raw/ │ -│ 每天 4 次打包 → rsync 推送 │ -│ SSH: <海外服务器> │ -│ 路径: /opt/intlgrab │ -└────────────────────┬─────────────────────────────┘ - │ rsync - ▼ -┌──────────────────────────────────────────────────┐ -│ Domestic Server (国内) │ -│ M2 正文提取 → M3 去重 → M4 翻译+事件 │ -│ → M5 向量生成 → M6 Qdrant 入库 │ -│ → M7 调度 + 日报 → M8 MCP 服务 │ -│ SSH: <国内服务器> │ -│ 路径: /home/pi/intlnews │ -└──────────────────────────────────────────────────┘ -``` - ---- - -## 3. 环境配置 - -### 3.1 依赖安装 - -```bash -cd /home/pi/intlnews -uv sync -``` - -### 3.2 配置文件 - -| 文件 | 用途 | 示例 | -|------|------|------| -| `.env` | API Key / URL(不入 Git) | `DEEPSEEK_API_KEY=sk-xxx` | -| `configs/system.yaml` | 功能参数(模型、阈值、超时) | `llm.provider: deepseek` | -| `configs/sources.yaml` | 新闻源定义 | 12 个英文财经源 | - -### 3.3 必需环境变量(`.env`) - -```bash -# DeepSeek(M4 翻译+事件抽取) -DEEPSEEK_API_KEY=sk-your-key -DEEPSEEK_BASE_URL=https://api.deepseek.com - -# Qwen(备选 LLM) -QWEN_API_KEY=sk-your-key -QWEN_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 - -# DashScope(M5 向量生成) -DASHSCOPE_API_KEY=sk-your-key - -# Qdrant(留空使用本地文件模式) -QDRANT_URL=http://localhost:6333 -QDRANT_API_KEY= -``` - -### 3.4 关键配置项(`configs/system.yaml`) - -```yaml -crawler: - max_memory_mb: 1800 # 串行抓取内存上限 - -dedup: - hamming_distance_threshold: 3 # SimHash 汉明距离阈值 - simhash_window_days: 30 # 时间窗口 - -llm: - provider: "deepseek" - deepseek_model: "deepseek-v4-flash" - concurrency: 3 # LLM 并发数 - -embedding: - provider: "dashscope" - dashscope_model: "text-embedding-v3" - dimension: 1024 - -qdrant: - collection: "en_finance_news" - -schedule: - day_cutoff_hour: 6 # 新闻日切分点(06:00) -``` - ---- - -## 4. CLI 命令参考 - -所有命令通过 `uv run en-news` 执行: - -| 命令 | Milestone | 功能 | -|------|-----------|------| -| `crawl` | M1 | 抓取英文财经新闻 | -| `extract` | M2 | 正文提取 | -| `dedup` | M3 | 三层去重 | -| `translate` | M4 | 翻译 + 事件抽取 | -| `embed` | M5 | 向量生成 | -| `index` | M6 | Qdrant 入库 | -| `search ` | M6 | 语义检索 | -| `pipeline` | M7 | 一键全链路 M2→M6 | -| `report` | M7 | 日报生成 | -| `mcp-server` | M8 | 启动 MCP 服务 | - -常用选项: - -```bash -# 指定源 -uv run en-news crawl --source forexlive -uv run en-news extract --source forexlive - -# 指定 top_k -uv run en-news search "美联储利率决议" --top-k 5 - -# 重建 Qdrant collection -uv run en-news index --recreate - -# 全链路跳过日报 -uv run en-news pipeline --skip-report -``` - ---- - -## 5. M1 — 新闻抓取 - -### 5.1 手动抓取 - -```bash -# 抓取所有启用的源 -uv run en-news crawl - -# 只抓取指定源 -uv run en-news crawl -s forexlive -``` - -### 5.2 新闻源列表 - -| 源 ID | 名称 | 类型 | -|-------|------|------| -| reuters | Reuters | 综合财经 | -| cnbc | CNBC | 市场新闻 | -| marketwatch | MarketWatch | 市场数据 | -| ft | Financial Times | 财经深度 | -| yahoo_finance | Yahoo Finance | 综合 | -| investing | Investing.com | 全球市场 | -| seekingalpha | Seeking Alpha | 投资分析 | -| barrons | Barrons | 市场评论 | -| wsj | WSJ | 综合财经 | -| economist | The Economist | 经济分析 | -| forexlive | ForexLive | 外汇新闻 | -| zerohedge | ZeroHedge | 另类财经 | - -### 5.3 反爬策略 - -部分新闻源设有反爬保护(如 DataDome)。系统支持三种策略,按优先级自动选择: - -| 优先级 | 策略 | 配置字段 | 说明 | 适用源 | -|--------|------|---------|------|--------| -| 1 | **RSS 抓取** | `rss_url` | 通过 RSS/Atom Feed 获取文章列表,完全绕过反爬 | MarketWatch ✅ | -| 2 | **Stealth 模式** | `anti_bot_mode: "stealth"` | 隐藏 webdriver 特征(`--disable-blink-features=AutomationControlled`) | Reuters(海外内存不足待验证) | -| 3 | **Headful 模式** | `anti_bot_mode: "headful"` | 非 headless 浏览器,最像真人 | 重度反爬源回退 | - -配置示例(`configs/sources.yaml`): - -```yaml -- id: "marketwatch" - rss_url: "https://feeds.marketwatch.com/marketwatch/topstories" # RSS 优先 - anti_bot_mode: "headful" # RSS 失败时回退 - -- id: "reuters" - anti_bot_mode: "stealth" # 无 RSS,直接 stealth -``` - -**已验证**: - -| 源 | 方式 | 结果 | -|----|------|------| -| MarketWatch | RSS | ✅ 10/10 | -| WSJ | RSS | ✅ 20/20 | -| ForexLive | 标准 headless | ✅ 17/17 | -| ZeroHedge | 标准 + URL 过滤 | ✅ 15/15 | -| Barron's | RSS (Dow Jones) | ✅ 10/10 | -| Reuters | stealth | ⚠️ DataDome | -| SeekingAlpha | stealth | ⚠️ PerimeterX | - -### 5.4 海外定时抓取 - -```cron -# crontab(<海外服务器>)— 每天 4 次 -# 时序: 抓取(60min) → 打包(5min) → 10min后国内拉取 -0 6 * * * cd /opt/intlgrab && bash scripts/overseas_crawl.sh -5 7 * * * cd /opt/intlgrab && bash scripts/overseas_pack.sh -0 12 * * * cd /opt/intlgrab && bash scripts/overseas_crawl.sh -5 13 * * * cd /opt/intlgrab && bash scripts/overseas_pack.sh -0 18 * * * cd /opt/intlgrab && bash scripts/overseas_crawl.sh -5 19 * * * cd /opt/intlgrab && bash scripts/overseas_pack.sh -0 22 * * * cd /opt/intlgrab && bash scripts/overseas_crawl.sh -5 23 * * * cd /opt/intlgrab && bash scripts/overseas_pack.sh -``` - -### 5.4 增量抓取机制 - -抓取采用**两层增量**确保不重复下载和存储: - -| 层级 | 位置 | 机制 | -|------|------|------| -| 抓取层 | `_extract_article_urls` | 读取当日 `index.jsonl` 中已抓取的 url_hash,跳过已存在的 URL,不重复下载 | -| 存储层 | `write_index_jsonl` | 追加写入前再次检查 url_hash,已存在则跳过 | - -同一天内多次执行 `crawl`,只有新文章才会被下载和存储。 - -### 5.5 产物 - -``` -data/raw/{source_id}/{YYYYMMDD}/ -├── {url_hash}.html # 原始 HTML -├── {url_hash}.md # Crawl4AI Markdown -└── index.jsonl # 文章索引 -``` - ---- - -## 6. M2 — 正文提取 - -### 6.1 执行 - -```bash -# 提取所有源 -uv run en-news extract - -# 提取指定源 -uv run en-news extract -s forexlive -``` - -### 6.2 技术方案 - -- 优先使用 Crawl4AI 输出的 Markdown -- 回退 `trafilatura` 英文正文提取 -- 最小正文字数阈值:50 词 - -### 6.3 产物 - -``` -data/processed/{source_id}/{YYYYMMDD}/ -├── {url_hash}.json # ProcessedArticle -└── index.jsonl # 处理索引 -``` - ---- - -## 7. M3 — 三层去重 - -### 7.1 执行 - -```bash -uv run en-news dedup -``` - -### 7.2 去重逻辑 - -| 层级 | 方法 | 说明 | -|------|------|------| -| L1 | URL Hash | 完全相同 URL 直接命中 | -| L2 | 内容 Hash | 标准化后 SHA1[:16] 匹配(去标点/空白) | -| L3 | SimHash | 字符 3-gram,汉明距离 ≤ 3,30 天窗口 | - -### 7.3 产物 - -``` -data/deduped/{YYYYMMDD}/ -├── uniques/{url_hash}.json # 唯一文章 -└── index.json # 去重索引 -data/dedup/fingerprints.sqlite3 # 指纹库 -``` - ---- - -## 8. M4 — 翻译 + 事件抽取 - -### 8.1 执行 - -```bash -uv run en-news translate -``` - -### 8.2 技术方案 - -- **Provider**: DeepSeek v4-flash(默认)/ Qwen 备选 -- **单次调用**:翻译 + 事件抽取合并,节省 token -- **并发**:3 线程(`system.yaml` → `llm.concurrency`) -- **重试**:3 次指数退避(1s → 2s → 4s) - -### 8.3 输出格式 - -```json -{ - "title": "Fed Holds Rates Steady as Markets Rally", - "title_zh": "美联储维持利率不变,市场上涨", - "content_en": "The Federal Reserve held...", - "content_zh": "美联储周三维持利率不变...", - "events": [ - { - "event_type": "央行决议", - "stock_codes": [], - "sentiment": "neutral", - "importance": 5, - "summary_zh": "美联储维持利率不变,市场反弹" - } - ] -} -``` - -### 8.4 14 种事件类型 - -`财报披露` `并购收购` `产品发布` `监管政策` `宏观经济` -`央行决议` `行业动态` `技术突破` `高管变动` `诉讼法律` -`市场异动` `地缘政治` `大宗商品` `外汇波动` `其他` - -### 8.5 产物 - -``` -data/events/{YYYYMMDD}/ -├── {url_hash}.json # EnTranslatedArticle -└── index.json # 事件索引 -``` - ---- - -## 9. M5 — 向量生成 - -### 9.1 执行 - -```bash -uv run en-news embed -``` - -### 9.2 技术方案 - -- **Provider**: DashScope `text-embedding-v3` -- **维度**: 1024 -- **嵌入文本**: `标题: {title_zh}` + `事件: [{sentiment}] {event_type} 重要度{n} {summary_zh}` + `正文: {content_zh[:3000]}` -- **截断**: 4000 字符上限 - -### 9.3 产物 - -``` -data/embeddings/{YYYYMMDD}/ -├── {url_hash}.json # EmbeddingResult(1024 维) -└── index.json # 向量索引 -``` - ---- - -## 10. M6 — Qdrant 入库与检索 - -### 10.1 入库 - -```bash -# 增量入库 -uv run en-news index - -# 重建 collection(清空旧数据) -uv run en-news index --recreate -``` - -### 10.2 语义搜索 - -```bash -# 基本搜索 -uv run en-news search "美联储利率决议" - -# 指定返回条数 -uv run en-news search "伊朗霍尔木兹海峡" --top-k 5 -``` - -### 10.3 技术方案 - -- **模式**: 本地文件(`data/qdrant_storage/`),无需 Docker -- **Collection**: `en_finance_news` -- **距离**: Cosine -- **Payload**: title / title_zh / url / source_id / events / content_zh_preview - -### 10.4 产物 - -``` -data/qdrant_storage/ # Qdrant 本地文件存储 -``` - ---- - -## 11. M7 — 全链路管道 + 日报 - -### 11.1 手动全链路运行(完整流程) - -当需要手工执行完整的数据处理流程时,按顺序执行以下命令: - -```bash -# 步骤 1(海外): 抓取英文财经新闻 -ssh <海外服务器> "cd /opt/intlgrab && uv run en-news crawl" - -# 步骤 2(海外): 打包 raw 数据 -ssh <海外服务器> "cd /opt/intlgrab && bash scripts/overseas_pack.sh" - -# 步骤 3(国内): 拉取海外数据 -cd /home/pi/intlnews && bash scripts/domestic_sync.sh - -# 步骤 4: 一键全链路 M2→M6 + 日报 -cd /home/pi/intlnews && uv run en-news pipeline -``` - -也可以分步执行(适合调试): - -```bash -# 分步模式 -uv run en-news extract # M2: 正文提取 -uv run en-news dedup # M3: 三层去重 -uv run en-news translate # M4: 翻译 + 事件抽取 -uv run en-news embed # M5: 向量生成 -uv run en-news index # M6: Qdrant 入库 -uv run en-news report # 日报生成 -``` - -### 11.2 一键管道(自动) - -```bash -# 全链路 M2→M6 + 日报 -uv run en-news pipeline - -# 跳过日报生成 -uv run en-news pipeline --skip-report - -# 只生成日报 -uv run en-news report -``` - -### 11.2 管道步骤 - -``` -M2 extract → M3 dedup → M4 translate → M5 embed → M6 index → 日报 -``` - -每步失败记录日志但不阻断后续步骤(降级继续)。 - -### 11.3 HTML 日报 - -日报包含五个板块: - -1. 🤖 **AI 摘要** — LLM 根据当日 important≥4 事件生成要点总结 -2. 🔥 **重要事件** — 高重要度事件表格(标题/情绪/重要度/摘要/链接) -3. 📊 **数据总览** — M1→M6 管道统计数据 -4. 📈 **情绪分布** — 利好/利空/中性比例条 + 重要度分布 -5. 📋 **事件类型 TOP 10** - -日报输出:`data/reports/intl_news_daily_{YYYYMMDD}.html`(约 9KB),同时自动上传到 -`https://echart.doorcome.cn/research/{YYYYMMDD}/intl_news_daily_{YYYYMMDD}.html` - -### 11.4 国内定时调度 - -```cron -# crontab(<国内服务器>)— 每天 3 次 -# 全流程:SSH 触发海外打包 → 下载 → 管道串行 M2→M6 → 日报 -0 7 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh -0 12 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh -0 18 * * * cd /home/pi/intlnews && bash scripts/domestic_full.sh -``` - -`domestic_full.sh` 统一完成:远程打包 → 同步 → 管道 → 日报。 - ---- - -## 12. M8 — MCP 服务 - -### 12.1 启动 - -```bash -uv run en-news mcp-server -``` - -### 12.2 可用 Tool - -| Tool | 功能 | 示例 | -|------|------|------| -| `search_news` | 语义检索新闻 | `search_news("美联储利率决议")` | -| `search_by_stock` | 美股代码检索 | `search_by_stock("AAPL")` | -| `search_by_sentiment` | 按情绪检索 | `search_by_sentiment("加息", sentiment="negative")` | -| `get_today_events` | 当日重要事件 | `get_today_events(importance_min=4)` | -| `get_stats` | 系统统计 | `get_stats()` | - -### 12.3 Claude Code 配置 - -在 Claude Code 的 MCP 配置中添加: - -```json -{ - "mcpServers": { - "intl-news": { - "command": "uv", - "args": ["run", "en-news", "mcp-server"], - "cwd": "/home/pi/intlnews" - } - } -} -``` - ---- - -## 13. 数据目录结构 - -``` -data/ -├── raw/ # M1: 原始抓取 -│ └── {source_id}/{YYYYMMDD}/ -│ ├── {url_hash}.html -│ ├── {url_hash}.md -│ └── index.jsonl -├── processed/ # M2: 正文提取 -│ └── {source_id}/{YYYYMMDD}/ -│ └── {url_hash}.json -├── dedup/ # M3: 指纹库 -│ ├── fingerprints.sqlite3 -│ └── {YYYYMMDD}/ -│ ├── uniques/{url_hash}.json -│ └── index.json -├── events/ # M4: 翻译+事件 -│ └── {YYYYMMDD}/ -│ └── {url_hash}.json -├── embeddings/ # M5: 向量 -│ └── {YYYYMMDD}/ -│ └── {url_hash}.json -├── qdrant_storage/ # M6: Qdrant 本地存储 -└── reports/ # M7: 日报 - └── intl_news_daily_{YYYYMMDD}.html -``` - ---- - -## 14. 定时任务 - -### 14.1 时间线 - -``` -海外 国内 -───────────────────────────── ───────────────────────── -06:00 crawl (≈60min) 07:00 全流程(打包→下载→管道→日报) -12:00 crawl (≈60min) 12:00 全流程 -18:00 crawl (≈60min) 18:00 全流程 -22:00 crawl (≈60min) (夜间 crawl 结果次日 07:00 处理) -``` - -国内 `domestic_full.sh` 流程:`SSH触发海外打包 → 下载 → M2→M3→M4→M5→M6 → 日报`(串行)。 - -### 14.2 新闻日定义 - -- 切分点:`day_cutoff_hour: 6`(凌晨 06:00) -- 当天 06:00 至次日 05:59 属于同一个新闻日 -- 例如:2026-06-19 04:00 → 新闻日 "20260618" - ---- - -## 15. 常见问题 - -### Q: 如何新增新闻源? - -编辑 `configs/sources.yaml`,添加源配置: - -```yaml -- id: "new_source" - name: "New Source Name" - enabled: true - homepage: "https://example.com/finance/" - article_url_pattern: "/news/[^/]+/" - js_render: false - max_articles_per_run: 30 -``` - -### Q: 翻译质量不好怎么办? - -1. 调整 `configs/system.yaml` 中 `llm.temperature`(降低更保守) -2. 编辑 `prompts/translation_and_extraction.md` 优化 Prompt -3. 切换 Provider:`llm.provider: "qwen"` - -### Q: Qdrant 检索太慢? - -- 本地文件模式已足够快(17 条 < 0.01s) -- 数据量 > 10 万条时建议切换到 Docker 模式 -- 设置 `QDRANT_URL=http://your-server:6333` - -### Q: 如何查看日志? - -```bash -tail -f logs/sync.log # 同步日志(国内) -tail -f logs/pipeline.log # 管道日志(国内) -tail -f logs/crawl.log # 抓取日志(海外) -tail -f logs/pack.log # 打包日志(海外) -``` - -日志自动清理:每周日凌晨 3 点删除 14 天前的 `.log` 文件(`scripts/cleanup_logs.sh`)。 - -### Q: 数据如何备份? - -```bash -# 备份整个 data 目录 -tar czf intlnews_backup_$(date +%Y%m%d).tar.gz data/ -``` - ---- - -## 附录:技术栈 - -| 组件 | 技术 | -|------|------| -| 语言 | Python 3.11 | -| 包管理 | uv + pyproject.toml | -| 抓取 | Crawl4AI + Playwright | -| 正文提取 | trafilatura | -| 去重 | SimHash + SQLite | -| LLM | DeepSeek v4-flash(OpenAI SDK) | -| Embedding | DashScope text-embedding-v3 | -| 向量库 | Qdrant(本地文件模式) | -| MCP | FastMCP | -| CLI | Typer | -| 配置 | YAML + .env | -| 数据模型 | Pydantic v2 | diff --git a/docs/pipeline.md b/docs/pipeline.md new file mode 100644 index 0000000..b79fd8f --- /dev/null +++ b/docs/pipeline.md @@ -0,0 +1,162 @@ +# 流水线详解 + +本文档描述 M1→M9 每一步的输入、输出、配置与增量机制。 + +## M1 新闻抓取 + +- 模块:`crawler/` +- 输入:`configs/sources.yaml` 中的新闻源配置 +- 输出:`data/raw/{source_id}/{YYYYMMDD}/` + - `index.jsonl`:本次/当日抓取索引,包含 `url_hash`、标题、URL、摘要、时间 + - `{url_hash}.html` / `{url_hash}.md`:原始网页或 Markdown +- 策略: + 1. 有 `rss_url` 的源先走 RSS/Atom/Google News RSS; + 2. RSS 失败或返回空时,根据 `anti_bot_mode` 走 stealth/headful Web; + 3. 单源串行,`index.jsonl` 按 `url_hash` 追加去重。 +- 常用命令: + - `bash scripts/domestic_crawl_8g.sh [source_id]` + - `uv run en-news crawl --source cnbc` + +## M2 正文提取 + +- 模块:`extractor/` +- 输入:`data/raw/{source_id}/{date}/index.jsonl` +- 输出:`data/processed/{source_id}/{date}/{url_hash}.json` + - `ProcessedArticle` 包含标题、正文 `content`、词数、提取器名、发布时间等。 +- 提取器:`trafilatura`,失败时回退 Markdown。 +- 增量:已存在同名 `{url_hash}.json` 则跳过。 + +```bash +uv run en-news extract +# 或只处理一个源 +uv run en-news extract --source cnbc +``` + +## M3 三层去重 + +- 模块:`dedup/` +- 输入:`data/processed/**/{date}/*.json` +- 输出: + - `data/deduped/{date}/uniques/{url_hash}.json`(唯一篇) + - `data/dedup/fingerprints.sqlite3`(指纹库) + - `data/deduped/{date}/index.json`(去重统计) +- 逻辑: + 1. URL 完全一致 → 重复 + 2. 正文 hash 一致 → 重复 + 3. SimHash 汉明距离 ≤ 阈值 → 模糊重复 +- 跨源合并:重复篇的来源 ID 会追加到保留唯一篇的 `source_ids`,日报可展示多来源。 + +```bash +uv run en-news dedup +``` + +## M4 翻译 + 投资事件抽取 + +- 模块:`llm/` +- 输入:`data/deduped/{date}/uniques/*.json` +- 输出:`data/events/{date}/{url_hash}.json` + - `EnTranslatedArticle`:英文原字段 + `title_zh` / `content_zh` / `events` / `source_ids` + - 事件结构:`event_type` / `stock_codes` / `sentiment` / `importance` / `summary_zh` +- LLM:默认 DeepSeek `deepseek-v4-flash`,场景 `translation` +- 并发:默认 3 线程,来自 `system.yaml llm.concurrency` +- 增量:`events/{date}/{url_hash}.json` 存在则跳过。 + +```bash +uv run en-news translate +``` + +## M5 向量生成 + +- 模块:`embedding/` +- 输入:`data/events/{date}/*.json` +- 输出:`data/embeddings/{date}/{url_hash}.json` + - 包含 `url_hash`、`source_id`、`vector` 等 + - `data/embeddings/{date}/index.json` 记录该批统计 +- 向量:DashScope `text-embedding-v3`,默认 1024 维、batch_size=10 +- 增量:已存在向量文件则跳过。 + +```bash +uv run en-news embed +``` + +## M6 Qdrant 入库与检索 + +- 模块:`vectorstore/` +- 输入:`data/embeddings/{date}/` 与 `data/events/{date}/` +- 输出:Qdrant collection `en_finance_news` + - collection 大小:1024 维,余弦距离 + - point ID:`url_hash` 通过 UUIDv5 转换,确定性幂等 + - payload:标题、双语标题、事件、来源、时间、正文预览等 +- 支持模式: + - 本地文件模式(默认):`data/qdrant_storage` + - 远程 HTTP 模式:`.env` 中配置非本地的 `QDRANT_URL` +- 命令: + +```bash +# 入库 +uv run en-news index + +# 重建 collection(会清空) +uv run en-news index --recreate + +# 重建并全量回灌所有历史向量 +uv run en-news index --recreate --all + +# 指定日期入库 +uv run en-news index --date 20260801 + +# 检索 +uv run en-news search "苹果 财报" +``` + +## M7 全链路管道编排 + +- 模块:`scheduler/pipeline.py` +- 串行执行:`extract → dedup → translate → embed → index → report` +- 每步失败不阻断后续,最终返回各步骤成功/失败统计。 +- CLI:`uv run en-news pipeline` / `--skip-report` +- Shell:`bash scripts/pipeline.sh [--resume]` + +## M8 MCP 服务 + +- 模块:`mcp_server/server.py` +- 启动:`uv run en-news mcp-server` +- 工具: + - `search_news`:语义搜索 + - `search_by_stock`:按美股代码搜索 + - `search_by_sentiment`:按情绪过滤搜索 + - `get_today_events`:当日重要事件 + - `get_stats`:系统统计 +- 用途:供 Claude Code / Cherry Studio 等 MCP 客户端做深度研究。 + +## M9 日报结构化入库 + +- 模块:`scheduler/reporter.py` + `report_db/` +- 时间窗口:最近 25 小时 +- 输出:MySQL `myquant` 库 + - `news_report`:`report_type='intl'`、`report_date`、`generated_at`、`ai_summary`、`stats` + - `news_event`:`section='intl'`、Top 事件明细 +- 幂等:`(report_date, report_type='intl', file_name='')` 唯一键覆盖 +- 命令:`uv run en-news report` + +## 数据目录速查 + +```text +data/ +├── raw/{source_id}/{YYYYMMDD}/ M1 原始 +├── processed/{source_id}/{YYYYMMDD}/ M2 正文 +├── dedup/fingerprints.sqlite3 M3 指纹库 +├── deduped/{YYYYMMDD}/uniques/ M3 去重后唯一篇 +├── events/{YYYYMMDD}/ M4 翻译+事件 +├── embeddings/{YYYYMMDD}/ M5 向量 +├── qdrant_storage/ M6 本地 Qdrant +├── run_state/ 步骤级 --resume 状态 +└── reports/ 历史 HTML 日报(M9 已不再产出) +``` + +## 失败恢复建议 + +1. 如果某个 Python 步骤失败,直接重跑同一命令即可,已完成的文件会自动跳过。 +2. 如果 Shell 脚本中断,使用 `bash scripts/domestic_full.sh --resume` 或 `bash scripts/pipeline.sh --resume`。 +3. 日报失败通常与 MySQL 连接/凭据有关,可先单独执行 `uv run en-news report` 查看日志。 +4. 若 Qdrant 本地文件损坏,可考虑备份后删除 `data/qdrant_storage` 并重新 `uv run en-news index`(需重新入库全部向量)。 diff --git a/docs/quickstart.md b/docs/quickstart.md new file mode 100644 index 0000000..7a66244 --- /dev/null +++ b/docs/quickstart.md @@ -0,0 +1,93 @@ +# 快速开始 + +## 1. 环境要求 + +- Python 3.11(项目支持 `>=3.11,<3.13`) +- [uv](https://docs.astral.sh/uv/) 包管理器 +- Linux(推荐 Pi 4 / 8GB 内存或更高) +- 可选但推荐:Playwright 浏览器依赖、Xvfb、Privoxy/Shadowsocks(用于海外访问) + +## 2. 安装依赖 + +```bash +# 在项目根目录执行 +uv sync + +# 如需开发依赖(pytest、ruff) +uv sync --extra dev +``` + +## 3. 配置环境变量 + +```bash +cp .env.example .env +``` + +然后编辑 `.env`,至少配置以下密钥: + +| 变量 | 用途 | +|------|------| +| `DEEPSEEK_API_KEY` | M4 翻译+事件抽取、M7 日报摘要(DeepSeek) | +| `QWEN_API_KEY` 或 `DASHSCOPE_API_KEY` | Qwen LLM、M5 向量化(Embedding 复用兼容端点) | +| `DASHSCOPE_API_KEY` | M5 `text-embedding-v3` 向量化 | +| `QDRANT_URL` / `QDRANT_API_KEY` | 远程 Qdrant(可选;默认本地文件模式可留空) | +| `NEWS_DB_*` | M9 日报入库 MySQL(`NEWS_DB_PASSWORD` 必填) | + +完整变量说明见 [配置说明](configuration.md)。 + +## 4. 验证安装 + +```bash +uv run en-news --help +``` + +能看到 `crawl / extract / dedup / translate / embed / index / search / report / pipeline / mcp-server` 即安装成功。 + +## 5. 首次运行 + +### 5.1 抓取(M1) + +```bash +# 抓取全部源(8G headful + HTTP 代理,适合国内 Pi) +bash scripts/domestic_crawl_8g.sh + +# 只抓取某个源 +bash scripts/domestic_crawl_8g.sh reuters +``` + +> 如果不需要真实抓取,可先用测试数据或已有 `data/raw/` 数据跳过此步。 + +### 5.2 全链路管道(M2→M6 + 日报) + +```bash +bash scripts/pipeline.sh +``` + +或使用命令行入口: + +```bash +uv run en-news pipeline --skip-report # 只跑 M2→M6,不生成日报 +uv run en-news pipeline # M2→M6 + 日报 +``` + +### 5.3 语义搜索 + +```bash +uv run en-news search "美联储 利率 决议" +``` + +## 6. 日常全流程 + +```bash +# 全流程:M1 抓取 + 管道 + 日报 +bash scripts/domestic_full.sh + +# 中断后继续(跳过当天已完成步骤) +bash scripts/domestic_full.sh --resume +``` + +## 7. 常见首个坑 + +- 如果没有配置 `NEWS_DB_PASSWORD`,日报(`report`)步骤会失败。`scripts/pipeline.sh` 会包含日报;若暂时不配 MySQL,请使用 `uv run en-news pipeline --skip-report` 跳过日报。 +- 抓取海外站点需要代理和 Xvfb,详见 [部署与运维](deployment.md)。 +- 运行前请确保已在项目根目录(所有路径都相对项目根目录)。 diff --git a/docs/usage.md b/docs/usage.md new file mode 100644 index 0000000..12b6b68 --- /dev/null +++ b/docs/usage.md @@ -0,0 +1,141 @@ +# 使用手册 + +## 1. CLI 命令(`uv run en-news`) + +| 命令 | 功能 | 常用参数 | +|------|------|----------| +| `crawl` | M1 抓取新闻 | `--source/-s ` 只抓单源;`--profile/-p` 指定 Profile | +| `extract` | M2 正文提取 | `--source/-s ` 只处理单源;`--date/-d` 指定日期 | +| `dedup` | M3 三层去重 | `--date/-d` 指定日期 | +| `translate` | M4 翻译+事件抽取 | `--date/-d` 指定日期 | +| `embed` | M5 向量生成 | `--date/-d` 指定日期;`--all` 处理全部日期 | +| `index` | M6 写入 Qdrant | `--recreate` 重建;`--date/-d` 指定日期;`--all` 全量回灌 | +| `search` | M6 语义检索 | 必需 query;`--top-k` 返回条数 | +| `report` | M9 生成日报并入库 | 无 | +| `pipeline` | M2→M6(+日报) | `--skip-report` 跳过日报;`--date/-d` 指定日期 | +| `mcp-server` | M8 启动 MCP 服务 | 无 | + +示例: + +```bash +# 只抓取 CNBC +uv run en-news crawl --source cnbc + +# 单步正文提取 +uv run en-news extract + +# 搜索 +uv run en-news search "美联储" --top-k 5 + +# 重建 Qdrant collection(慎用!会清空现有向量) +uv run en-news index --recreate + +# 指定历史日期处理 +uv run en-news translate --date 20260801 +uv run en-news embed --date 20260801 +uv run en-news index --date 20260801 + +# 全量回灌历史向量(--recreate 搭配 --all 时只会在首个日期重建 collection) +uv run en-news index --all +uv run en-news embed --all +``` + +## 2. Shell 脚本 + +| 脚本 | 用途 | +|------|------| +| `scripts/domestic_full.sh` | 国内服务器全流程:M1 抓取 + M2→M6 管道 + 日报 | +| `scripts/domestic_full.sh --resume` | 同上,但跳过当天已完成步骤 | +| `scripts/pipeline.sh` | 仅 M2→M6 管道 + 日报 | +| `scripts/pipeline.sh --resume` | 同上,支持步骤级断点续跑 | +| `scripts/domestic_crawl_8g.sh [source]` | 8G headful + HTTP 代理抓取 M1 | +| `scripts/domestic_crawl_2g.sh [source]` | 2G headless stealth 轻量抓取 M1 | +| `scripts/cleanup_logs.sh` | 清理 14 天前日志 | +| `scripts/domestic_sync.sh [date]` | 从海外服务器同步 raw 数据(海外已停用,保留兼容) | + +### 2.1 全流程示例 + +```bash +cd /home/pi/intlnews + +# 每天定时任务可执行 +bash scripts/domestic_full.sh + +# 手动中断后继续 +bash scripts/domestic_full.sh --resume +``` + +### 2.2 步骤级断点说明 + +步骤状态记录在 `data/run_state/{YYYYMMDD}.state`: + +- 每成功一个步骤追加一行,如 `M1_crawl`、`M2_extract`、`M3_dedup`、`M4_translate`、`M5_embed`、`M6_index`、`report`。 +- `--resume` 只跳过已标记步骤;失败步骤不会标记,所以会从失败处重试。 +- 跨天自动失效,状态文件按天隔离。 + +## 3. MCP 服务(M8) + +MCP 是给 Cherry Studio / Claude Code 等 Agent 客户端接入的研究工具层。 + +启动: + +```bash +uv run en-news mcp-server +``` + +提供工具: + +| 工具 | 功能 | +|------|------| +| `search_news` | 语义检索新闻 | +| `search_by_stock` | 按美股代码检索相关事件 | +| `search_by_sentiment` | 按情绪倾向(利好/利空/中性)过滤检索 | +| `get_today_events` | 获取当日重要投资事件 | +| `get_stats` | 获取系统统计概览 | + +具体接入方式由客户端决定,通常是在 MCP 配置中指向 `uv run --directory /path/to/intlnews en-news mcp-server`。 + +## 4. 日报查看 + +日报已结构化写入 MySQL: + +- 主表:`news_report` + - `report_type = 'intl'` + - `ai_summary` 为 AI 日报摘要 + - `stats` 为 JSON 统计快照 +- 明细表:`news_event` + - `section = 'intl'` + - 包含标题、摘要、来源、情绪、重要度、URL + +可通过 SQL 查看最新日报: + +```sql +SELECT id, report_date, generated_at, ai_summary +FROM news_report +WHERE report_type = 'intl' +ORDER BY report_date DESC, generated_at DESC +LIMIT 5; +``` + +## 5. 日志 + +- 全流程日志:`logs/domestic_full_{YYYYMMDD_HHMMSS}.log` +- 管道日志:`logs/pipeline_{YYYYMMDD_HHMMSS}.log` +- 运行时也实时输出到终端。 + +快速查看最近日志: + +```bash +tail -100 logs/pipeline_$(ls -t logs/pipeline_*.log | head -1 | sed 's#logs/##') +``` + +## 6. 常见操作组合 + +| 目标 | 命令 | +|------|------| +| 只抓取新闻 | `bash scripts/domestic_crawl_8g.sh` | +| 从已有 raw 跑全管道 | `uv run en-news pipeline --skip-report` | +| 增量补齐翻译 | `uv run en-news translate`(自动跳过已有) | +| 只生成日报 | `uv run en-news report` | +| 搜索知识库 | `uv run en-news search "关键词"` | +| 启动 MCP | `uv run en-news mcp-server` | diff --git a/embedding/pipeline.py b/embedding/pipeline.py index fee89b2..cb27554 100644 --- a/embedding/pipeline.py +++ b/embedding/pipeline.py @@ -41,7 +41,12 @@ def _load_event_articles(date_str: str) -> list[EnTranslatedArticle]: continue try: data = json.loads(json_file.read_text(encoding="utf-8")) - articles.append(EnTranslatedArticle(**data)) + article = EnTranslatedArticle(**data) + # 防御性过滤:没有英文原文的事件文件通常是 M2 no_content + # 残留,不应继续向量化。 + if not (article.content_en or "").strip(): + continue + articles.append(article) except (json.JSONDecodeError, Exception) as e: logger.warning("解析事件文章失败 %s: %s", json_file, e) diff --git a/english-news-plan.md b/english-news-plan.md deleted file mode 100644 index 750fe5b..0000000 --- a/english-news-plan.md +++ /dev/null @@ -1,477 +0,0 @@ -# English Financial News — 项目开发计划 - -> 国际财经新闻抓取与深度研究平台 -> -> 对标 `news/` (A 股 Deep Research),复用其架构模式,聚焦英文财经源。 - ---- - -## 一、项目定位 - -构建面向国际财经新闻的私有化 Deep Research 平台。 - -核心能力: -- 英文财经新闻抓取(Crawl4AI) -- 新闻联播数据接入(api.doorcome.cn) -- 全文英译中(LLM) -- 投资事件抽取(LLM) -- 双语向量知识库(Qdrant) -- MCP 服务 + Cherry Studio / Claude Code Agent 深度研究 -- 每日 AI 摘要日报(含新闻联播投资相关解读) - -本项目不是: -- 交易系统 -- 预测系统 -- 投资顾问 - ---- - -## 二、部署拓扑 - -``` -┌──────────────────────────────────────────────────────────────┐ -│ Overseas Server (国外) │ -│ M1 Crawl4AI 抓取 → data/raw/{source}/{YYYYMMDD}/ │ -│ ↓ │ -│ rsync 增量推送 (定时) │ -└──────────────────────────┬───────────────────────────────────┘ - │ SSH / rsync - ▼ -┌──────────────────────────────────────────────────────────────┐ -│ Domestic Server (国内) │ -│ M2 正文提取 → M3 去重 → M4 翻译+事件抽取(LLM) │ -│ → M5 向量生成 → M6 Qdrant 入库 │ -│ → M7 定时调度 → M8 MCP 服务 │ -│ → 日报生成 (AI 摘要 + 重要事件 + 新闻联播投资解读) │ -│ │ -│ ┌── 新闻联播数据源 (独立 API) ──┐ │ -│ │ GET api.doorcome.cn/api/xwlbFine/ │ -│ │ → LLM 筛选投资相关 → 重要性评分 → AI 解读 │ -│ └──────────────────────────────┘ │ -└──────────────────────────────────────────────────────────────┘ -``` - -同步机制: -- 海外服务器每天定时 `rsync -avz` 推送 `data/raw/` 到国内 -- 国内定时任务先拉取同步,再执行 M2→M8 管道 -- rsync 天然增量(只传新文件),带宽高效 - ---- - -## 三、Milestone 分解 - -### M0 — 项目骨架 - -- `pyproject.toml`(Python 3.11 + uv) -- `.env.example` -- `configs/sources.yaml`(英文财经源配置) -- `CLAUDE.md` -- 目录结构初始化 -- CLI 入口:`en-news` - -### M1 — 新闻抓取(海外服务器) - -**技术选型**:Crawl4AI(异步 + Playwright JS 渲染),与 A 股项目一致。 - -**英文财经源**(初始 12 个): - -| 源 ID | 名称 | 类型 | -|-------|------|------| -| reuters | Reuters | 综合财经 | -| cnbc | CNBC | 市场新闻 | -| marketwatch | MarketWatch | 市场数据 | -| ft | Financial Times | 财经深度 | -| yahoo_finance | Yahoo Finance | 综合 | -| investing | Investing.com | 全球市场 | -| seekingalpha | Seeking Alpha | 投资分析 | -| barrons | Barrons | 市场评论 | -| wsj | WSJ | 综合财经 | -| economist | The Economist | 经济分析 | -| forexlive | ForexLive | 外汇新闻 | -| zerohedge | ZeroHedge | 另类财经 | - -每个源配置字段: -```yaml -- id: "reuters" - name: "Reuters" - enabled: true - homepage: "https://www.reuters.com/business/" - article_url_pattern: "/[^/]+/[^/]+/" - js_render: true - max_articles_per_run: 30 -``` - -**产物**:`data/raw/{source_id}/{YYYYMMDD}/{url_hash}.html` + `index.jsonl` - -**同步守护进程**: -```bash -# 海外 crontab:每天 05:00 UTC 推送 -0 5 * * * rsync -avz --ignore-existing /app/data/raw/ user@domestic:/app/data/raw/ -``` - -### M2 — 正文提取(国内服务器) - -**输入**:`data/raw/{source_id}/{date}/`(rsync 同步后) - -**提取方案**: -- 优先使用 Crawl4AI 输出的 Markdown(`{url_hash}.md`) -- 对未生成 Markdown 的源,用 `trafilatura`(英文正文提取,类 GNE 但针对英文优化) -- 输出标准化 `Article` 模型(`title` / `content` / `publish_time` / `author` / `source_id`) - -**产物**:`data/processed/{source_id}/{YYYYMMDD}/{url_hash}.json` - -### M3 — 去重 - -与 A 股项目相同的三层去重: -1. URL Hash(精确) -2. 内容 Hash(SHA256 前 64 字符) -3. SimHash 模糊(汉明距离 ≤ 3,30 天窗口) - -**产物**:`data/deduped/{YYYYMMDD}/uniques/{url_hash}.json` - -### M4 — 翻译 + 投资事件抽取(LLM) - -**单次 LLM 调用完成两件事**(节省 token): - -Prompt 设计: -``` -输入:英文财经新闻全文 -输出 JSON: -{ - "translation_zh": "中文全文翻译", - "events": [ - { - "event_type": "并购/财报/政策/行业/...", - "stock_codes": ["AAPL", "TSLA"], - "sentiment": "positive/negative/neutral", - "importance": 1-5, - "summary_zh": "事件中文摘要" - } - ] -} -``` - -**LLM Provider**:DeepSeek(默认)/ Qwen 备选 - -**产物**:`data/events/{YYYYMMDD}/{url_hash}.json` -```json -{ - "title": "Apple Reports Record Q2 Earnings", - "title_zh": "苹果公布创纪录第二季度财报", - "content_en": "...", - "content_zh": "...", - "events": [...] -} -``` - -### M5 — 向量生成 - -- Embedding Provider:DashScope `text-embedding-v3`(1024 维) -- 对中文翻译内容向量化(`content_zh` + `title_zh` + 事件摘要拼接) -- 也可选择英文原文向量化,支持双语检索 - -**产物**:`data/embeddings/{YYYYMMDD}/{url_hash}.json` - -### M6 — Qdrant 入库与检索 - -与 A 股项目相同: -- 本地文件模式(默认)或 Docker Server 模式 -- Collection:`en_finance_news` -- Payload:`source_id` / `title` / `title_zh` / `url` / `publish_time` / `events` / `sentiment` / `importance` -- Vector:从 M5 产物加载 - -```python -uv run en-news search "Fed interest rate decision" -``` - -### M7 — 新闻联播数据源(独立模块) - -**数据获取**: - -``` -GET https://api.doorcome.cn/api/xwlbFine/?start_date=YYYY-MM-DD&end_date=YYYY-MM-DD -``` - -**响应结构**: -```json -{ - "status": "success", - "data": { - "news": [ - { - "news_days": "2026-06-20", - "daily_sub_id": 1, - "news_title": "牢记总书记嘱托 各地在文脉赓序中推动城市高质量发展", - "news_improve": "历史文化是城市的灵魂...(全文约3000字)" - } - ] - } -} -``` - -每天约 12-15 条新闻,每条含完整文字稿。 - -**处理流程**: - -1. **获取** — 日报生成前拉取昨日新闻联播数据。例如 6/21 早上执行日报 → 拉取 6/20 的新闻联播(前一日 19:00 播出,次日凌晨 API 已就绪) -2. **投资相关性筛选(LLM)** — 用 DeepSeek 逐条判断是否与投资相关 - - 经济数据、产业政策、贸易谈判 → **高相关**(importance 4-5) - - 科技创新、区域发展、基建项目 → **中相关**(importance 2-3) - - 文化、民生、体育、天气 → **非相关**(跳过) -3. **AI 解读** — 对投资相关条目生成 2-3 句投资视角解读: - - 对哪些行业/板块有影响 - - 政策信号含义 - - 市场可能的反应 -4. **纳入日报** — 单独一节展示,区别于英文财经新闻 - -**产物**:`data/xwlb/{YYYYMMDD}/index.jsonl` -```json -{ - "news_days": "2026-06-20", - "daily_sub_id": 5, - "news_title": "中欧班列统一品牌十周年", - "is_investment_related": true, - "importance": 4, - "category": "贸易/物流", - "ai_interpretation": "中欧班列十年数据将直接反映一带一路贸易活跃度,利好物流、港口、跨境贸易板块..." -} -``` - -### M8 — 定时调度 - -- 国内 crontab 或 APScheduler 守护进程 -- 日期规则:T 日早上生成的日报覆盖 T-1 日数据(新闻 + 新闻联播) -- 流程: - ``` - 06:00 等待海外 rsync 完成(检查哨兵文件) - 06:30 M2→M3→M4→M5→M6 全链路(处理 T-1 日英文新闻) - 07:00 拉取 T-1 日新闻联播 → LLM 筛选解读 → 日报生成 - 12:00 增量(只处理当日新增) - 18:00 增量 - 22:00 增量 + 日报 - ``` - -### M9 — MCP 服务 - -暴露 5 个 MCP 工具给 Cherry Studio / Claude Code: - -| 工具 | 功能 | -|------|------| -| `search_en_news` | 通用语义检索(中文查询 → 中文翻译向量匹配) | -| `search_by_company` | 按公司名/股票代码检索 | -| `search_by_sentiment` | 按情绪检索+统计 | -| `search_by_event_type` | 按事件类型检索 | -| `get_daily_report` | 获取最新日报 | - ---- - -## 四、数据模型 - -```python -class EnArticle(BaseModel): - source_id: str - url: str - url_hash: str - title: str # 英文原标题 - title_zh: str # 中文翻译标题 - content_en: str # 英文原文 - content_zh: str # 中文全文翻译 - author: str - publish_time: datetime - source_name: str - word_count: int - word_count_zh: int - -class EnExtractedEvent(BaseModel): - article_url_hash: str - event_type: str # 并购/财报/政策/行业/市场/... - stock_codes: list[str] # 涉及美股代码 - sentiment: str # positive/negative/neutral - importance: int # 1-5 - summary_zh: str # 中文摘要 - summary_en: str # 英文摘要 - -class XwlbItem(BaseModel): - """新闻联播条目(经 LLM 筛选和解读)。""" - news_days: str # 播出日期 "2026-06-20" - daily_sub_id: int # 当日序号 (1-15) - news_title: str # 标题 - news_improve: str # 完整文字稿 - is_investment_related: bool # 是否投资相关 - importance: int # 投资重要性 1-5 (非投资为 0) - category: str # 投资类别 (贸易/产业政策/科技/基建/金融/...) - ai_interpretation: str # AI 投资解读 2-3 句 -``` - ---- - -## 五、日报设计 - -命令:`uv run en-news pipeline --once --report` - -执行顺序:新闻联播拉取 → LLM 筛选解读 → 日报 HTML 生成 - -六段式结构: - -1. **AI 摘要**(24h 国际财经 + 昨日新闻联播投资相关,≤500 字,最后一条不截断) - -2. **重要事件:国际新闻**(24h 英文源,importance ≥ 4 逐级回退,最多 20 篇) - -3. **📺 新闻联播投资解读**(昨日播出,投资相关条目 + AI 解读) - - 展示格式:标题 | 投资相关性类别 | 重要度 | AI 解读(2-3 句) - - 非投资相关条目不显示 - - 示例: - ``` - 📺 中欧班列统一品牌十周年 [贸易/物流] ⭐⭐⭐⭐ - AI解读: 中欧班列十年数据将直接反映一带一路贸易活跃度。 - 关注物流、港口、跨境贸易板块。政策利好中国外运、中远海控等。 - ``` - -4. **市场情绪分布**(利好/利空/中性,24h 国际新闻 + 新闻联播合并统计) - -5. **数据总览**(M1→M6 管道统计 + 各源抓取量 6 列网格 + 重要度分布 + 事件类型分布 + 新闻联播条目数) - -6. **完整新闻联播条目列表**(当天的全部条目,标题+简要,含非投资类,供快速浏览) - ---- - -## 六、配置规范 - -### configs/sources.yaml - -```yaml -settings: - concurrency: 5 - user_agent: "Mozilla/5.0 ..." - output_root: "data/raw" - -sources: - - id: "reuters" - name: "Reuters" - enabled: true - homepage: "https://www.reuters.com/business/" - article_url_pattern: "/[^/]+/[^/]+/" - js_render: false - max_articles_per_run: 30 -``` - -### .env 关键配置 - -```bash -# LLM -LLM_PROVIDER=deepseek -LLM_MODEL=deepseek-v4-flash -DEEPSEEK_API_KEY=sk-xxx - -# Embedding -EMBEDDING_PROVIDER=dashscope -DASHSCOPE_API_KEY=sk-xxx - -# 新闻联播 -XWLB_API_BASE=https://api.doorcome.cn - -# Schedule -SCHEDULE_TIMES=06:30,12:00,18:00,22:00 -SYNC_WAIT_SEC=300 - -# Qdrant -QDRANT_COLLECTION=en_finance_news - -# Overseas → Domestic -SYNC_HOST=user@domestic-server -SYNC_PORT=22 -SYNC_SENTINEL=/tmp/en_news_sync_done -``` - ---- - -## 七、目录结构 - -``` -english-news/ -├── crawler/ # M1 新闻抓取(Crawl4AI, 部署海外) -├── extractor/ # M2 英文正文提取(trafilatura) -├── dedup/ # M3 三层去重 -├── translator/ # M4a 全文翻译(LLM) -├── llm/ # M4b 投资事件抽取 -├── embedding/ # M5 向量生成 -├── vectorstore/ # M6 Qdrant 客户端 -├── xwlb/ # M7 新闻联播数据源 -├── scheduler/ # M8 定时任务 -├── mcp_server/ # M9 MCP 服务 -├── app/ # CLI 入口(en-news) -├── configs/ # sources.yaml / system config -├── prompts/ # LLM Prompt 模板 -├── scripts/ # 部署 & 运维脚本 -│ ├── sync_daemon.sh # 海外 rsync 推送脚本 -│ └── wait_for_sync.py # 国内等待同步完成 -├── tests/ # pytest -├── docs/ # 设计文档 -├── data/ # 数据目录(git ignored) -│ ├── raw/ # M1 产物(海外与国内共享) -│ ├── processed/ # M2 产物 -│ ├── deduped/ # M3 产物 -│ ├── events/ # M4 产物 -│ ├── embeddings/ # M5 产物 -│ ├── xwlb/ # 新闻联播数据(经 LLM 筛选与解读) -│ ├── qdrant_storage/ # M6 Qdrant 本地存储 -│ └── reports/ # 日报 HTML -├── logs/ # 运行日志 -├── pyproject.toml -├── CLAUDE.md -└── README.md -``` - ---- - -## 八、技术选型对比 - -| 模块 | A 股项目 | 英文财经项目 | 差异 | -|------|---------|-------------|------| -| 抓取 | Crawl4AI | Crawl4AI | 一致 | -| 正文提取 | GNE | trafilatura | 英文优化 | -| 翻译 | — | LLM (DeepSeek) | **新增** | -| 去重 | 三层 | 三层 | 一致 | -| LLM 事件 | DeepSeek/Qwen | DeepSeek/Qwen | 一致 | -| Embedding | DashScope | DashScope | 一致 | -| 向量库 | Qdrant | Qdrant | 一致 | -| 新闻联播 | — | api.doorcome.cn + LLM 筛选 | **新增** | -| 调度 | APScheduler | APScheduler | 一致 | -| MCP | FastMCP | FastMCP | 一致 | -| CLI | a-share | en-news | 命名区分 | - ---- - -## 九、开发优先级(建议顺序) - -| 序号 | Milestone | 预估工期 | 依赖 | -|------|-----------|---------|------| -| 1 | M0 项目骨架 | 0.5 天 | — | -| 2 | M1 抓取(海外) | 2 天 | M0 | -| 3 | rsync 同步机制 | 0.5 天 | M1 | -| 4 | M2 正文提取 | 1 天 | 同步就绪 | -| 5 | M3 去重 | 1 天 | M2 | -| 6 | M4 翻译+事件抽取 | 1.5 天 | M3 | -| 7 | M5 向量 + M6 Qdrant | 1 天 | M4 | -| 8 | M7 新闻联播数据源 | 1 天 | — (独立模块) | -| 9 | M8 调度 + 日报 | 1.5 天 | M6, M7 | -| 10 | M9 MCP 服务 | 1 天 | M6 | -| **总计** | | **~11 天** | | - ---- - -## 十、关键风险 - -| 风险 | 缓解措施 | -|------|---------| -| 英文源 JS 渲染复杂 | Crawl4AI 已有 Playwright 支持,优先使用轻量级源 | -| LLM 翻译质量不稳定 | System prompt 约束翻译风格,保留原文备查 | -| rsync 网络不稳定 | 添加重试机制 + 哨兵文件完整性检查 | -| 海外 IP 被封 | 轮换 User-Agent,控制请求频率(≥ 2s/req) | -| DeepSeek API 限流 | 并发控制 + Qwen 备选 provider | - ---- - -> 本计划文件:`~/Downloads/cc-projects/english-news-plan.md` -> -> 最后更新:2026-06-20 diff --git a/llm/pipeline.py b/llm/pipeline.py index 20231bc..dd1c1ea 100644 --- a/llm/pipeline.py +++ b/llm/pipeline.py @@ -56,7 +56,12 @@ def _load_deduped_articles( for json_file in sorted(base_dir.glob("*.json")): try: data = json.loads(json_file.read_text(encoding="utf-8")) - articles.append(ProcessedArticle(**data)) + article = ProcessedArticle(**data) + # 防御性过滤:旧版数据里可能残留 M2 no_content 唯一篇, + # 避免用空正文调用 LLM。 + if article.status != "success" or not article.content.strip(): + continue + articles.append(article) except (json.JSONDecodeError, Exception) as e: logger.warning("解析去重文章失败 %s: %s", json_file, e) diff --git a/scheduler/pipeline.py b/scheduler/pipeline.py index ec1d9f3..16583fb 100644 --- a/scheduler/pipeline.py +++ b/scheduler/pipeline.py @@ -4,6 +4,7 @@ """ import logging +import signal import time from dataclasses import dataclass, field from datetime import datetime @@ -30,6 +31,47 @@ STEP_TIMEOUTS: dict[str, int] = { } +class _StepTimeout(Exception): + """步骤超时专用异常,避免与业务 TimeoutError 混淆。""" + + +def _run_step_with_timeout(func, name: str, date_str: str, timeout: int | None) -> "StepResult": + """在支持 SIGALRM 的主线程中为单步执行添加超时保护。""" + if timeout is None: + return func(date_str) + + started = datetime.now() + + if not hasattr(signal, "SIGALRM"): + return func(date_str) + + def _handler(signum, frame): # noqa: ARG001 + raise _StepTimeout(f"step {name} timed out after {timeout}s") + + try: + old_handler = signal.getsignal(signal.SIGALRM) + except (ValueError, OSError): + # 非主线程无法设置信号处理器,直接不启用超时 + return func(date_str) + + signal.signal(signal.SIGALRM, _handler) + signal.setitimer(signal.ITIMER_REAL, timeout) + try: + return func(date_str) + except _StepTimeout: + elapsed = (datetime.now() - started).total_seconds() + return StepResult( + name=name, + success=False, + elapsed_sec=elapsed, + message=f"超时(>{timeout}s)", + started_at=started, + ) + finally: + signal.setitimer(signal.ITIMER_REAL, 0) + signal.signal(signal.SIGALRM, old_handler) + + @dataclass class StepResult: """单步执行结果。""" @@ -210,7 +252,9 @@ def run_pipeline( continue logger.info("── 步骤 %s 开始 ──", name) - sr = func(date_str) + sr = _run_step_with_timeout( + func, name, date_str, STEP_TIMEOUTS.get(name) + ) result.steps.append(sr) flag = "✅" if sr.success else "❌" diff --git a/scheduler/reporter.py b/scheduler/reporter.py index 762d79b..48c0885 100644 --- a/scheduler/reporter.py +++ b/scheduler/reporter.py @@ -210,6 +210,9 @@ def _load_events_window(now: datetime) -> list[dict]: continue try: data = json.loads(fp.read_text(encoding="utf-8")) + # 防御性过滤:M2 no_content 残留不应进入日报统计/摘要。 + if not (data.get("content_en") or "").strip(): + continue # 时间过滤:publish_time 在 25 小时内 pt_str = data.get("publish_time", "") pt = _try_parse_time(pt_str) @@ -267,7 +270,11 @@ def _collect_stats_window(now: datetime) -> dict: deduped += len(list(dedup_dir.glob("*.json"))) emb_dir = Path(f"data/embeddings/{day_str}") if emb_dir.is_dir(): - emb_count += len(list(emb_dir.glob("*.json"))) + # 不把 index.json 计入实际向量文章数 + emb_count += len([ + f for f in emb_dir.glob("*.json") + if f.name != "index.json" + ]) for idx in Path("data/raw").glob(f"*/{day_str}/index.jsonl"): src = idx.parent.parent.name n = sum(1 for _ in open(idx, encoding="utf-8")) diff --git a/scripts/domestic_crawl_2g.sh b/scripts/domestic_crawl_2g.sh index 88a9e9f..d40c19c 100755 --- a/scripts/domestic_crawl_2g.sh +++ b/scripts/domestic_crawl_2g.sh @@ -25,7 +25,11 @@ LOG "内存上限: 1800 MB" LOG "浏览器模式: stealth headless" # ── 加载 .env ── -export $(grep -v '^#' .env | grep -v '^$' | xargs 2>/dev/null || true) +if [ -f .env ]; then + set -a + . ./.env + set +a +fi # ── 设置 Profile ── export EN_NEWS_PROFILE="2g_headless" @@ -39,12 +43,18 @@ else LOG "目标源: 全部启用源" fi -PYTHONPATH=. .venv/bin/python3 -c " +if PYTHONPATH=. .venv/bin/python3 -c " from crawler.orchestrator import run_crawl_sync import sys source_filter = sys.argv[1] if len(sys.argv) > 1 else None stats = run_crawl_sync(source_filter=source_filter) print(f'抓取完成: {stats.sources_crawled} 源, {stats.total_articles} 篇') -" "$1" - -LOG "══════ 2G 抓取完成 ✅ ══════" +if stats.sources_crawled > 0 and stats.sources_failed >= stats.sources_crawled: + print(f'❌ 所有源抓取失败(sources_failed={stats.sources_failed})', file=sys.stderr) + sys.exit(1) +" "$1"; then + LOG "══════ 2G 抓取完成 ✅ ══════" +else + LOG "❌ 2G 抓取失败:本次未标记完成" + exit 1 +fi diff --git a/scripts/domestic_crawl_8g.sh b/scripts/domestic_crawl_8g.sh index 4b6587e..a8aae9b 100755 --- a/scripts/domestic_crawl_8g.sh +++ b/scripts/domestic_crawl_8g.sh @@ -33,7 +33,11 @@ LOG "HTTP 代理: 127.0.0.1:3128 (Privoxy → SOCKS5)" # ── 加载 .env ── export HTTP_PROXY=http://127.0.0.1:3128 export HTTPS_PROXY=http://127.0.0.1:3128 -export $(grep -v '^#' .env | grep -v '^$' | xargs 2>/dev/null || true) +if [ -f .env ]; then + set -a + . ./.env + set +a +fi # ── 启动 Xvfb(如果尚未运行)── if ! pgrep -x "Xvfb" > /dev/null; then @@ -59,12 +63,18 @@ else LOG "目标源: 全部启用源" fi -PYTHONPATH=. .venv/bin/python3 -c " +if PYTHONPATH=. .venv/bin/python3 -c " from crawler.orchestrator import run_crawl_sync import sys source_filter = sys.argv[1] if len(sys.argv) > 1 else None stats = run_crawl_sync(source_filter=source_filter) print(f'抓取完成: {stats.sources_crawled} 源, {stats.total_articles} 篇') -" "${1:-}" - -LOG "══════ 8G 抓取完成 ✅ ══════" +if stats.sources_crawled > 0 and stats.sources_failed >= stats.sources_crawled: + print(f'❌ 所有源抓取失败(sources_failed={stats.sources_failed})', file=sys.stderr) + sys.exit(1) +" "${1:-}"; then + LOG "══════ 8G 抓取完成 ✅ ══════" +else + LOG "❌ 8G 抓取失败:本次未标记完成" + exit 1 +fi diff --git a/scripts/domestic_full.sh b/scripts/domestic_full.sh index 5b05271..b820cdd 100755 --- a/scripts/domestic_full.sh +++ b/scripts/domestic_full.sh @@ -40,16 +40,26 @@ LOG "══════ 国内全流程开始(resume=${RESUME})════ LOG "全流程日志: ${FULL_LOG}(终端实时显示完整进度)" # ── 加载 .env ── -export $(grep -v '^#' .env | grep -v '^$' | xargs 2>/dev/null || true) +if [ -f .env ]; then + set -a + . ./.env + set +a +fi # ── 1. M1 Pi 抓取(headful Playwright + HTTP 代理)── # 部分源抓取失败不阻塞管道(原语义);抓取本身按 URL 去重(index.jsonl), # 已抓取过的 URL 不会重复写入;输出实时显示每源进度 +# 只有 M1 脚本成功退出(即不是“全部源失败”)才标记完成; +# 若全部源失败,不标记,--resume 可重新抓取。 +M1_FAILED=0 if step_should_run M1_crawl; then LOG "━━━ [1/2] M1 抓取(headful Playwright + HTTP 代理)━━━" - bash "$SCRIPT_DIR/domestic_crawl_8g.sh" 2>&1 | tee -a "$FULL_LOG" \ - || LOG "WARNING: 部分源抓取失败,继续管道" - step_mark M1_crawl + if bash "$SCRIPT_DIR/domestic_crawl_8g.sh" 2>&1 | tee -a "$FULL_LOG"; then + step_mark M1_crawl + else + LOG "WARNING: M1 抓取未完全成功,本次不标记 M1_crawl,--resume 可重试" + M1_FAILED=1 + fi fi # ── 2. M2→M6 管道(含日报)── @@ -60,4 +70,9 @@ else bash "$SCRIPT_DIR/pipeline.sh" 2>&1 | tee -a "$FULL_LOG" fi +if [ "$M1_FAILED" -eq 1 ]; then + LOG "⚠️ 全流程结束,但 M1 抓取未成功,退出码=1" + exit 1 +fi + LOG "══════ 国内全流程完成 ✅ ══════" diff --git a/scripts/pipeline.sh b/scripts/pipeline.sh index 81e45f5..d58cb73 100755 --- a/scripts/pipeline.sh +++ b/scripts/pipeline.sh @@ -66,7 +66,11 @@ print(f'{p} / {m}') } # 加载 .env -export $(grep -v '^#' .env | grep -v '^$' | xargs 2>/dev/null || true) +if [ -f .env ]; then + set -a + . ./.env + set +a +fi # ── M2: 正文提取 ── LOG "━━━ M2 正文提取 ━━━" diff --git a/vectorstore/pipeline.py b/vectorstore/pipeline.py index d8d3cfd..7b9d9cb 100644 --- a/vectorstore/pipeline.py +++ b/vectorstore/pipeline.py @@ -50,6 +50,11 @@ def _load_embedding_files(date_str: str) -> list[dict]: if event_file.exists(): article_data = json.loads(event_file.read_text(encoding="utf-8")) + # 防御性过滤:没有英文原文的事件/向量不应进入知识库。 + # 历史 no_content 残留满足此条件,重建/全量回灌时可被自动剔除。 + if not (article_data.get("content_en") or "").strip(): + continue + items.append({ "url_hash": emb_data["url_hash"], "source_id": emb_data.get("source_id", ""), @@ -77,6 +82,7 @@ def _build_payload(article_data: dict) -> dict: "title_zh": article_data.get("title_zh", ""), "url": article_data.get("url", ""), "source_id": article_data.get("source_id", ""), + "source_ids": article_data.get("source_ids", []), "source_name": article_data.get("source_name", ""), "publish_time": article_data.get("publish_time", ""), "events": article_data.get("events", []), @@ -107,6 +113,15 @@ def ingest_all_embeddings( items = _load_embedding_files(date_str) if not items: + # --recreate 时即使当天没有有效向量,也要先把 collection 重建/清空, + # 避免 --all --recreate 遇到首个空日期时跳过重建。 + if recreate: + client = make_qdrant_client() + store = VectorStore(client) + try: + store.init_collection(recreate=True) + finally: + store.close() logger.warning("嵌入目录无数据: data/embeddings/%s/", date_str) return {"date": date_str, "total": 0, "ingested": 0, "failed": 0, "elapsed_sec": 0}