From 65ead54b4fb494640b91a6dfda0b5d047b4bcd05 Mon Sep 17 00:00:00 2001 From: simon Date: Sat, 22 Aug 2026 17:10:39 +0800 Subject: [PATCH] =?UTF-8?q?docs:=20=E6=96=87=E6=A1=A3=E6=B8=85=E7=90=86?= =?UTF-8?q?=E4=B8=8E=E9=87=8D=E6=9E=84=20=E2=80=94=20=E7=BB=9F=E4=B8=80?= =?UTF-8?q?=E4=B8=BA=203=20=E4=B8=AA=E6=A0=B8=E5=BF=83=E6=96=87=E6=A1=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 删除 5 个过时/残留文档(project_plan/agent_prompt/optimization_plan/report_db_design/deploy/README) - 新建 docs/architecture.md(项目架构:11 包职责+数据模型+配置+产物) - 重写 docs/user-guide.md(CLI 全量+增量/断点续跑+MCP+FAQ) - 重写 README.md(精简入口+文档索引) - 更新 continuation.md(追加本次记录) - 更新 .gitignore(排除 data/* 运行产物) --- .env.example | 84 + .gitignore | 78 + .mcp.json | 16 + .serena/.gitignore | 2 + .serena/project.yml | 118 + CLAUDE.md | 662 ++++ README.md | 115 + a_share_cli/__init__.py | 1 + a_share_cli/main.py | 985 ++++++ api/__init__.py | 0 app/__init__.py | 0 configs/.gitkeep | 0 configs/__init__.py | 1 + configs/llm_models.yaml | 133 + configs/llm_models.yaml.bak | 133 + configs/loader.py | 69 + configs/sources.yaml | 218 ++ configs/watchlist.yaml | 80 + continuation.md | 386 +++ crawler/__init__.py | 32 + crawler/cninfo.py | 499 +++ crawler/config.py | 52 + crawler/engine.py | 445 +++ crawler/models.py | 161 + crawler/storage.py | 98 + dedup/__init__.py | 44 + dedup/deduper.py | 183 ++ dedup/hasher.py | 93 + dedup/models.py | 90 + dedup/store.py | 207 ++ deploy/a-share-research.service | 24 + deploy/crontab.example | 41 + docker-compose.yml | 40 + docs/architecture.md | 636 ++++ docs/db_schema.md | 125 + docs/user-guide.md | 752 +++++ embedding/__init__.py | 52 + embedding/base.py | 154 + embedding/factory.py | 71 + embedding/local.py | 113 + embedding/models.py | 49 + embedding/remote.py | 223 ++ extractor/__init__.py | 22 + extractor/models.py | 58 + extractor/parser.py | 367 +++ llm/__init__.py | 64 + llm/client.py | 208 ++ llm/extractor.py | 307 ++ llm/models.py | 188 ++ mcp_server/__init__.py | 4 + mcp_server/tools.py | 250 ++ prompts/company_analysis.md | 64 + prompts/event_extraction.md | 110 + prompts/industry_analysis.md | 65 + prompts/risk_analysis.md | 62 + pyproject.toml | 101 + report_db/__init__.py | 19 + report_db/db.py | 166 + report_db/models.py | 35 + report_db/schema.py | 50 + report_import/__init__.py | 21 + report_import/importer.py | 89 + report_import/parser.py | 298 ++ scheduler/__init__.py | 25 + scheduler/pipeline.py | 369 +++ scheduler/reporter.py | 1119 +++++++ scheduler/stock_reporter.py | 534 +++ scripts/__init__.py | 0 scripts/a-share-research.service | 25 + scripts/run_cninfo.py | 67 + scripts/run_crawler.py | 81 + scripts/run_dedup.py | 275 ++ scripts/run_embedding.py | 334 ++ scripts/run_event_extraction.py | 281 ++ scripts/run_extractor.py | 391 +++ scripts/run_mcp_server.py | 88 + scripts/run_qdrant_ingest.py | 210 ++ scripts/run_scheduler.py | 210 ++ scripts/run_xwlb.py | 107 + tests/__init__.py | 1 + tests/conftest.py | 34 + .../finance_news_daily_20260710_0720.html | 157 + .../intl_news_daily_20260711_070304.html | 99 + tests/test_cli.py | 93 + tests/test_crawler.py | 333 ++ tests/test_dedup.py | 502 +++ tests/test_embedding.py | 397 +++ tests/test_extractor.py | 481 +++ tests/test_incremental.py | 268 ++ tests/test_llm.py | 500 +++ tests/test_mcp.py | 129 + tests/test_report_builder.py | 285 ++ tests/test_report_db.py | 58 + tests/test_report_import.py | 32 + tests/test_report_parser.py | 98 + tests/test_scheduler.py | 132 + tests/test_vectorstore.py | 243 ++ uv.lock | 2917 +++++++++++++++++ vectorstore/__init__.py | 25 + vectorstore/client.py | 326 ++ vectorstore/models.py | 54 + 101 files changed, 21093 insertions(+) create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 .mcp.json create mode 100644 .serena/.gitignore create mode 100644 .serena/project.yml create mode 100644 CLAUDE.md create mode 100644 README.md create mode 100644 a_share_cli/__init__.py create mode 100644 a_share_cli/main.py create mode 100644 api/__init__.py create mode 100644 app/__init__.py create mode 100644 configs/.gitkeep create mode 100644 configs/__init__.py create mode 100644 configs/llm_models.yaml create mode 100644 configs/llm_models.yaml.bak create mode 100644 configs/loader.py create mode 100644 configs/sources.yaml create mode 100644 configs/watchlist.yaml create mode 100644 continuation.md create mode 100644 crawler/__init__.py create mode 100644 crawler/cninfo.py create mode 100644 crawler/config.py create mode 100644 crawler/engine.py create mode 100644 crawler/models.py create mode 100644 crawler/storage.py create mode 100644 dedup/__init__.py create mode 100644 dedup/deduper.py create mode 100644 dedup/hasher.py create mode 100644 dedup/models.py create mode 100644 dedup/store.py create mode 100644 deploy/a-share-research.service create mode 100644 deploy/crontab.example create mode 100644 docker-compose.yml create mode 100644 docs/architecture.md create mode 100644 docs/db_schema.md create mode 100644 docs/user-guide.md create mode 100644 embedding/__init__.py create mode 100644 embedding/base.py create mode 100644 embedding/factory.py create mode 100644 embedding/local.py create mode 100644 embedding/models.py create mode 100644 embedding/remote.py create mode 100644 extractor/__init__.py create mode 100644 extractor/models.py create mode 100644 extractor/parser.py create mode 100644 llm/__init__.py create mode 100644 llm/client.py create mode 100644 llm/extractor.py create mode 100644 llm/models.py create mode 100644 mcp_server/__init__.py create mode 100644 mcp_server/tools.py create mode 100644 prompts/company_analysis.md create mode 100644 prompts/event_extraction.md create mode 100644 prompts/industry_analysis.md create mode 100644 prompts/risk_analysis.md create mode 100644 pyproject.toml create mode 100644 report_db/__init__.py create mode 100644 report_db/db.py create mode 100644 report_db/models.py create mode 100644 report_db/schema.py create mode 100644 report_import/__init__.py create mode 100644 report_import/importer.py create mode 100644 report_import/parser.py create mode 100644 scheduler/__init__.py create mode 100644 scheduler/pipeline.py create mode 100644 scheduler/reporter.py create mode 100644 scheduler/stock_reporter.py create mode 100644 scripts/__init__.py create mode 100644 scripts/a-share-research.service create mode 100644 scripts/run_cninfo.py create mode 100644 scripts/run_crawler.py create mode 100644 scripts/run_dedup.py create mode 100644 scripts/run_embedding.py create mode 100644 scripts/run_event_extraction.py create mode 100644 scripts/run_extractor.py create mode 100644 scripts/run_mcp_server.py create mode 100644 scripts/run_qdrant_ingest.py create mode 100644 scripts/run_scheduler.py create mode 100644 scripts/run_xwlb.py create mode 100644 tests/__init__.py create mode 100644 tests/conftest.py create mode 100644 tests/fixtures/finance_news_daily_20260710_0720.html create mode 100644 tests/fixtures/intl_news_daily_20260711_070304.html create mode 100644 tests/test_cli.py create mode 100644 tests/test_crawler.py create mode 100644 tests/test_dedup.py create mode 100644 tests/test_embedding.py create mode 100644 tests/test_extractor.py create mode 100644 tests/test_incremental.py create mode 100644 tests/test_llm.py create mode 100644 tests/test_mcp.py create mode 100644 tests/test_report_builder.py create mode 100644 tests/test_report_db.py create mode 100644 tests/test_report_import.py create mode 100644 tests/test_report_parser.py create mode 100644 tests/test_scheduler.py create mode 100644 tests/test_vectorstore.py create mode 100644 uv.lock create mode 100644 vectorstore/__init__.py create mode 100644 vectorstore/client.py create mode 100644 vectorstore/models.py diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..2371367 --- /dev/null +++ b/.env.example @@ -0,0 +1,84 @@ +# ============================ +# A 股 Deep Research 平台 .env 示例 +# 复制为 .env 后填写真实值 +# 注意: 值后面不能跟 # 注释, python-dotenv 会当成值的一部分 +# ============================ + +# ---- LLM Provider ---- +# 推荐: 各场景按需独立配置见 configs/llm_models.yaml(优先级高于以下环境变量) +LLM_PROVIDER=deepseek + +# ---- DeepSeek ---- +DEEPSEEK_API_KEY= +DEEPSEEK_BASE_URL=https://api.deepseek.com +DEEPSEEK_MODEL=deepseek-chat + +# ---- Qwen / 百炼 ---- +# QWEN_API_KEY 未设时用 DASHSCOPE_API_KEY +QWEN_API_KEY= +QWEN_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 +QWEN_MODEL=qwen-plus + +# ---- 百炼通用 API Key (LLM Qwen + Embedding DashScope 兜底) ---- +DASHSCOPE_API_KEY= + +# ---- LLM 通用参数 ---- +LLM_MODEL= # 兜底模型名 (未设 DEEPSEEK_MODEL/QWEN_MODEL 时使用) +LLM_TEMPERATURE=0.1 +LLM_TIMEOUT_SEC=60 + +# ---- Embedding ---- +# provider: local (本地 BGE-M3) | dashscope (远程百炼) +EMBEDDING_PROVIDER=dashscope + +# 远程嵌入 (EMBEDDING_PROVIDER=dashscope 时生效) +# DASHSCOPE_EMBEDDING_API_KEY 未设时用 DASHSCOPE_API_KEY +DASHSCOPE_EMBEDDING_API_KEY= +DASHSCOPE_EMBEDDING_BASE_URL=https://dashscope.aliyuncs.com/compatible-mode/v1 +DASHSCOPE_EMBEDDING_MODEL=text-embedding-v3 + +# 本地嵌入 (EMBEDDING_PROVIDER=local 时生效, 需 uv sync --extra local-embedding) +# LOCAL_EMBEDDING_MODEL=BAAI/bge-m3 + +# ---- Qdrant ---- +QDRANT_HOST=localhost +QDRANT_PORT=6333 +QDRANT_API_KEY= +QDRANT_COLLECTION=a_share_news + +# ---- 调度 ---- +SCHEDULE_TIMES=07:00,12:00,18:00,22:00 +CNINFO_SCHEDULE_TIME=06:30 +STOCK_REPORT_TIME=07:30 +STOCK_REPORT_DAYS=15 + +# ---- Pipeline 步骤超时 ---- +# 优先级: TIMEOUT_{NAME} > PIPELINE_STEP_TIMEOUT > 代码硬编码默认值 +# PIPELINE_STEP_TIMEOUT 为全局兜底, 对所有未单独配置的步骤生效 +PIPELINE_STEP_TIMEOUT=1800 +# 以下为各步骤独立超时(秒), 值不设时取上方的全局值 +# TIMEOUT_CRAWLER=900 +# TIMEOUT_XWLB=60 +# TIMEOUT_EXTRACTOR=300 +# TIMEOUT_DEDUP=300 +# TIMEOUT_LLM=900 +# TIMEOUT_EMBEDDING=300 +# TIMEOUT_QDRANT=600 +# TIMEOUT_CNINFO_CRAWL=900 +# TIMEOUT_CNINFO_EXTRACT=300 +# TIMEOUT_CNINFO_PDF=300 + +# ---- cninfo 公告抓取 ---- +CNINFO_PDF_BASE=http://static.cninfo.com.cn + +# ---- 日报结构化入库 (M10) ---- +NEWS_DB_HOST=127.0.0.1 # 开发走 ssh 隧道: ssh -L 13306:127.0.0.1:13306 pi +NEWS_DB_PORT=13306 +NEWS_DB_USER=myquant +NEWS_DB_PASSWORD= # 填真实值,禁止写入源码/文档 +NEWS_DB_NAME=myquant +REPORT_HISTORY_DIR=data/reports_history + +# ---- LLM 日报摘要重试 ---- +LLM_RETRY_TIMES=3 # AI 摘要调用失败重试次数(默认 3) +LLM_RETRY_BACKOFF_SEC=2.0 # 指数退避基数,秒(默认 2.0: 2s,4s,8s...) diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..68e1c08 --- /dev/null +++ b/.gitignore @@ -0,0 +1,78 @@ +# Python +__pycache__/ +*.py[cod] +*$py.class +*.so +.Python +.venv/ +venv/ +env/ +build/ +dist/ +*.egg-info/ +*.egg +.eggs/ + +# uv +# uv.lock 应保留以锁定生产环境依赖版本(本项目为应用而非库) + +# 测试 / 覆盖率 +.pytest_cache/ +.coverage +htmlcov/ +.tox/ +.mypy_cache/ +.ruff_cache/ + +# IDE +.vscode/ +.idea/ +*.swp +*.swo + +# OS +.DS_Store +Thumbs.db + +# 工具本地配置(非项目文件) +reasonix.toml + +# 项目敏感配置 +.env +.env.local +.env.*.local +*.key +*.pem + +# 运行产物 +logs/*.log +logs/*.log.* +data/raw/ +data/processed/ +data/cache/ +data/pipeline/ +data/dedup/ +data/deduped/ +data/events/ +data/embeddings/ +data/qdrant_storage/ +data/reports_history/ +data/reports/ +data/ +data/reports/ +*.sqlite +*.sqlite3 +*.db + +# 日志目录(非 logs/ 下的) +configs/logs/ + +# Docker 卷 +qdrant_storage/ + +# 调试输出 +debug/ +tmp/ + +# M10: 历史日报源文件副本(可从 doorcome 重新拉取) +data/reports_history/ diff --git a/.mcp.json b/.mcp.json new file mode 100644 index 0000000..b04d386 --- /dev/null +++ b/.mcp.json @@ -0,0 +1,16 @@ +{ + "mcpServers": { + "serena-djapi": { + "command": "uv", + "args": [ + "run", + "--directory", + "/Users/summer/Downloads/cc-cursor/mcp-servers/serena", + "serena", + "start-mcp-server", + "--project", + "/Users/summer/Downloads/cc-projects/news" + ] + } + } +} diff --git a/.serena/.gitignore b/.serena/.gitignore new file mode 100644 index 0000000..2e510af --- /dev/null +++ b/.serena/.gitignore @@ -0,0 +1,2 @@ +/cache +/project.local.yml diff --git a/.serena/project.yml b/.serena/project.yml new file mode 100644 index 0000000..1d79927 --- /dev/null +++ b/.serena/project.yml @@ -0,0 +1,118 @@ +# the name by which the project can be referenced within Serena +project_name: "news" + + +# list of languages for which language servers are started; choose from: +# al ansible bash clojure cpp +# cpp_ccls crystal csharp csharp_omnisharp dart +# elixir elm erlang fortran fsharp +# go groovy haskell haxe hlsl +# java json julia kotlin lean4 +# lua luau markdown matlab msl +# nix ocaml pascal perl php +# php_phpactor powershell python python_jedi python_ty +# r rego ruby ruby_solargraph rust +# scala solidity swift systemverilog terraform +# toml typescript typescript_vts vue yaml +# zig +# (This list may be outdated. For the current list, see values of Language enum here: +# https://github.com/oraios/serena/blob/main/src/solidlsp/ls_config.py +# For some languages, there are alternative language servers, e.g. csharp_omnisharp, ruby_solargraph.) +# Note: +# - For C, use cpp +# - For JavaScript, use typescript +# - For Free Pascal/Lazarus, use pascal +# Special requirements: +# Some languages require additional setup/installations. +# See here for details: https://oraios.github.io/serena/01-about/020_programming-languages.html#language-servers +# When using multiple languages, the first language server that supports a given file will be used for that file. +# The first language is the default language and the respective language server will be used as a fallback. +# Note that when using the JetBrains backend, language servers are not used and this list is correspondingly ignored. +languages: [] + +# the encoding used by text files in the project +# For a list of possible encodings, see https://docs.python.org/3.11/library/codecs.html#standard-encodings +encoding: "utf-8" + +# line ending convention to use when writing source files. +# Possible values: unset (use global setting), "lf", "crlf", or "native" (platform default) +# This does not affect Serena's own files (e.g. memories and configuration files), which always use native line endings. +line_ending: + +# The language backend to use for this project. +# If not set, the global setting from serena_config.yml is used. +# Valid values: LSP, JetBrains +# Note: the backend is fixed at startup. If a project with a different backend +# is activated post-init, an error will be returned. +language_backend: + +# whether to use project's .gitignore files to ignore files +ignore_all_files_in_gitignore: true + +# advanced configuration option allowing to configure language server-specific options. +# Maps the language key to the options. +# Have a look at the docstring of the constructors of the LS implementations within solidlsp (e.g., for C# or PHP) to see which options are available. +# No documentation on options means no options are available. +ls_specific_settings: {} + +# list of additional paths to ignore in this project. +# Same syntax as gitignore, so you can use * and **. +# Note: global ignored_paths from serena_config.yml are also applied additively. +ignored_paths: [] + +# whether the project is in read-only mode +# If set to true, all editing tools will be disabled and attempts to use them will result in an error +# Added on 2025-04-18 +read_only: false + +# list of tool names to exclude. +# This extends the existing exclusions (e.g. from the global configuration) +# Find the list of tools here: https://oraios.github.io/serena/01-about/035_tools.html +excluded_tools: [] + +# list of tools to include that would otherwise be disabled (particularly optional tools that are disabled by default). +# This extends the existing inclusions (e.g. from the global configuration). +# Find the list of tools here: https://oraios.github.io/serena/01-about/035_tools.html +included_optional_tools: [] + +# fixed set of tools to use as the base tool set (if non-empty), replacing Serena's default set of tools. +# This cannot be combined with non-empty excluded_tools or included_optional_tools. +# Find the list of tools here: https://oraios.github.io/serena/01-about/035_tools.html +fixed_tools: [] + +# list of mode names that are to be activated by default, overriding the setting in the global configuration. +# The full set of modes to be activated is base_modes (from global config) + default_modes + added_modes. +# If the setting is undefined/empty, the default_modes from the global configuration (serena_config.yml) apply. +# Otherwise, this overrides the setting from the global configuration (serena_config.yml). +# Therefore, you can set this to [] if you do not want the default modes defined in the global config to apply +# for this project. +# This setting can, in turn, be overridden by CLI parameters (--mode). +# See https://oraios.github.io/serena/02-usage/050_configuration.html#modes +default_modes: + +# list of mode names to be activated additionally for this project, e.g. ["query-projects"] +# The full set of modes to be activated is base_modes (from global config) + default_modes + added_modes. +# See https://oraios.github.io/serena/02-usage/050_configuration.html#modes +added_modes: + +# initial prompt for the project. It will always be given to the LLM upon activating the project +# (contrary to the memories, which are loaded on demand). +initial_prompt: "" + +# time budget (seconds) per tool call for the retrieval of additional symbol information +# such as docstrings or parameter information. +# This overrides the corresponding setting in the global configuration; see the documentation there. +# If null or missing, use the setting from the global configuration. +symbol_info_budget: + +# list of regex patterns which, when matched, mark a memory entry as read‑only. +# Extends the list from the global configuration, merging the two lists. +read_only_memory_patterns: [] + +# list of regex patterns for memories to completely ignore. +# Matching memories will not appear in list_memories or activate_project output +# and cannot be accessed via read_memory or write_memory. +# To access ignored memory files, use the read_file tool on the raw file path. +# Extends the list from the global configuration, merging the two lists. +# Example: ["_archive/.*", "_episodes/.*"] +ignored_memory_patterns: [] diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..a257767 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,662 @@ +# CLAUDE.md + +# Claude Code 开发约束(A股 Deep Research 项目) + +版本:v1.0 + +最后更新:2026-06-16 + +--- + +# 一、项目定位 + +本项目目标: + +构建一个面向 A 股投资研究的私有化 Deep Research 平台。 + +核心能力包括: + +* 财经新闻抓取; +* 中文新闻提取; +* 投资事件抽取; +* 向量知识库; +* MCP 服务; +* Cherry Studio Agent 深度研究。 + +本项目不是: + +* 自动交易系统; +* 股票预测系统; +* 投资顾问系统。 + +Claude Code 必须始终围绕“研究辅助平台”进行设计。 + +--- + +# 二、Claude Code 总体行为准则 + +Claude Code 必须遵守以下原则: + +## 2.1 分阶段开发 + +禁止一次性完成整个项目。 + +必须: + +* 每次只完成一个 Milestone; +* 等待人工验收; +* 验收通过后再进入下一阶段。 + +禁止: + +* 擅自推进后续阶段; +* 推翻已完成模块。 + +--- + +## 2.2 最小改动原则 + +修改代码时: + +优先局部修改。 + +禁止: + +为了优化而重写整个模块。 + +除非明确要求: + +“允许重构”。 + +否则: + +保持向后兼容。 + +--- + +## 2.3 先理解,再编码 + +开始编码前必须: + +明确: + +* 当前目标; +* 输入; +* 输出; +* 验收标准。 + +如果需求冲突: + +必须先提问。 + +不得自行猜测。 + +--- + +# 三、开发环境规范 + +## 3.1 Python 版本 + +统一使用: + +Python 3.11 + +禁止: + +* Python 3.13; +* Python 3.10 以下版本。 + +--- + +## 3.2 依赖管理 + +统一使用: + +uv + +禁止: + +requirements.txt 手工维护。 + +依赖定义: + +pyproject.toml + +安装命令: + +uv sync + +新增依赖: + +uv add 包名 + +开发依赖: + +uv add --dev 包名 + +--- + +## 3.3 虚拟环境 + +统一使用: + +.venv + +禁止: + +使用 Conda。 + +禁止: + +多个虚拟环境混用。 + +--- + +# 四、代码规范 + +## 4.1 类型注解 + +所有新增代码必须包含类型注解。 + +示例: + +def search_news( +keyword: str, +top_k: int +) -> list[dict]: +... + +禁止: + +省略类型。 + +--- + +## 4.2 中文说明 + +要求: + +代码使用英文命名。 + +注释与文档使用中文。 + +例如: + +# 新闻去重处理 + +def deduplicate_articles(): +... + +--- + +## 4.3 函数长度 + +单个函数: + +建议 ≤ 50 行。 + +超过: + +必须拆分。 + +--- + +## 4.4 文件长度 + +单文件: + +建议 ≤ 500 行。 + +超过: + +必须拆分模块。 + +--- + +## 4.5 禁止魔法数字 + +禁止: + +importance = 5 + +应写成: + +MAX_IMPORTANCE = 5 + +--- + +# 五、日志规范 + +统一使用: + +loguru(现有代码全部使用 loguru,禁止混用 stdlib logging) + +禁止: + +print() + +日志级别: + +DEBUG + +INFO + +WARNING + +ERROR + +CRITICAL + +日志格式: + +时间 - 模块 - 级别 +消息 + +示例: + +2026-06-16 09:00:00 - crawler - INFO +开始抓取新浪财经 + +--- + +# 六、异常处理规范 + +禁止: + +except: +pass + +必须: + +记录日志。 + +抛出明确异常。 + +示例: + +except Exception as e: + logger.exception(e) + raise + +--- + +# 七、测试规范 + +新增功能必须提供测试。 + +优先使用: + +pytest + +测试目录: + +tests/ + +命名: + +test_xxx.py + +测试覆盖: + +核心模块必须覆盖。 + +包括: + +* crawler +* extractor +* dedup +* qdrant +* llm + +--- + +# 八、配置规范 + +禁止硬编码。 + +配置统一放置: + +configs/ + +支持: + +YAML + +环境变量 + +.env + +禁止: + +API Key 写入源码。 + +--- + +# 九、Prompt 管理规范 + +Prompt 必须独立维护。 + +目录: + +prompts/ + +禁止: + +在代码中直接拼接长 Prompt。 + +Prompt 文件命名: + +event_extraction.md + +company_analysis.md + +industry_analysis.md + +risk_analysis.md + +--- + +# 十、数据模型规范 + +统一使用: + +Pydantic + +禁止: + +大量裸 dict。 + +核心对象必须定义模型。 + +例如: + +Article + +Event + +EmbeddingResult + +SearchResult + +--- + +# 十一、数据库规范 + +Qdrant 为唯一向量数据库。 + +SQLite 用于: + +任务状态; +缓存; +调试。 + +禁止: + +引入多种向量数据库。 + +除非用户明确要求。 + +--- + +# 十二、Docker 规范 + +所有服务必须支持 Docker。 + +必须提供: + +docker-compose.yml + +必须支持: + +docker compose up -d + +启动。 + +不得依赖: + +手工安装。 + +--- + +# 十三、Git 提交规范 + +完成每个 Milestone 后: + +更新: + +README.md + +continuation.md + +docs/ + +提交信息格式: + +feat: +新增功能 + +fix: +问题修复 + +refactor: +重构 + +docs: +文档更新 + +test: +测试 + +chore: +杂项 + +示例: + +feat: 完成 Crawl4AI 新闻抓取模块 + +--- + +# 十四、README 更新规范 + +每次功能完成后: + +README 必须更新: + +包括: + +功能说明; + +部署方法; + +配置说明; + +示例命令; + +常见问题。 + +禁止: + +README 长期不维护。 + +--- + +# 十五、continuation.md 维护规范 + +Claude Code 每次结束工作前: + +必须更新 continuation.md。 + +内容包括: + +当前 Milestone; + +完成内容; + +待办事项; + +已知问题; + +技术债务; + +下次建议。 + +用于恢复上下文。 + +--- + +# 十六、性能要求 + +目标: + +支持至少: + +50 个财经新闻源。 + +支持: + +并发抓取。 + +新闻处理吞吐: + +≥100篇/分钟。 + +Qdrant 检索: + +Top-K 响应时间 ≤ 2 秒。 + +--- + +# 十七、安全规范 + +禁止: + +提交: + +.env + +API Key + +Cookie + +Token + +个人隐私数据 + +必须: + +提供: + +.env.example + +示例配置。 + +--- + +# 十八、禁止事项 + +Claude Code 禁止: + +1. 未经允许重构已验收模块; +2. 一次性生成整个项目; +3. 擅自修改数据库结构; +4. 删除已有测试; +5. 使用 print 调试; +6. 忽略异常; +7. 跳过验收直接进入下一阶段; +8. 引入未经说明的新技术栈; +9. 将 Prompt 写死在代码中; +10. 编写无法运行的伪代码冒充完成。 + +--- + +# 十九、输出格式要求 + +Claude Code 完成任务时必须输出: + +【任务目标】 + +【完成内容】 + +【修改文件】 + +【运行方法】 + +【测试结果】 + +【存在问题】 + +【下一步建议】 + +不得只输出代码。 + +必须提供可验证说明。 + +--- + +# 二十、最高优先级原则 + +当多个原则冲突时,优先级如下: + +第一优先级: + +代码正确、可运行。 + +第二优先级: + +稳定性与可维护性。 + +第三优先级: + +向后兼容。 + +第四优先级: + +性能优化。 + +第五优先级: + +代码优雅。 + +宁可代码普通,也不要复杂炫技。 + +本项目追求: + +“小步迭代、稳定演进、长期维护”。 + +# 二十一、项目现状速查(2026-07 校验) + +## 入口与常用命令 + +- 统一 CLI:`uv run a-share <子命令>`,入口 `a_share_cli/main.py` +- 子命令:`crawl` / `extract` / `dedup` / `events` / `embed` / `ingest` / `pipeline --once` / `search` / `report` / `status` / `discover` +- 测试:`uv run pytest`(tests/,asyncio_mode=auto;integration 标记默认跳过) +- 静态检查:`uv run ruff check .`、`uv run mypy`(pyproject.toml 已配置) +- 环境:`uv sync`;本地 BGE-M3 需 `uv sync --extra local-embedding` + +## 架构(9 个包) + +| 包 | 职责 | +| --- | --- | +| crawler | Crawl4AI 抓取;js_render=false 走 httpx 静态直连,否则 Playwright;含 cninfo 公告 | +| extractor | GNE 中文正文提取 | +| dedup | 新闻去重(SimHash) | +| llm | DeepSeek/Qwen 投资事件抽取 | +| embedding | 远程 DashScope / 本地 BGE-M3 | +| vectorstore | Qdrant;默认本地文件模式 data/qdrant_storage/ | +| scheduler | APScheduler 调度 + pipeline + 日报 reporter | +| mcp_server | MCP 工具(供 Cherry Studio) | +| api | 占位(空包) | + +scripts/ 为独立运行脚本 + systemd 服务文件 `scripts/a-share-research.service`。 + +## 关键约束与事实 + +- LLM 模型 `deepseek-v4-flash` 绝不允许修改(记忆:never-change-llm-model) +- 生产部署:Pi `pi@192.168.1.160:/home/pi/news/`,systemd 服务 `a-share-research`;改动后需 scp 同步 +- 改 sources.yaml 前先从 Pi 拉取、改完推回(记忆:sync-sources-yaml) +- 配置在 configs/sources.yaml、configs/watchlist.yaml、.env;API Key 只放 .env +- Prompt 在 prompts/*.md,禁止写死在代码中 +- 会话恢复上下文以 continuation.md 为准 + +—— CLAUDE.md 结束 —— + diff --git a/README.md b/README.md new file mode 100644 index 0000000..01ac57b --- /dev/null +++ b/README.md @@ -0,0 +1,115 @@ +# A 股 Deep Research 私有投研平台 + +> 面向 A 股投资研究的私有化 Deep Research 平台。 +> +> 自动抓取财经新闻 → 正文提取 → LLM 投资事件抽取 → 向量化 → Qdrant 知识库 → MCP 语义检索 → 日报生成。 + +**定位**:A 股研究辅助与知识管理平台,**非**自动交易、**非**预测、**非**投资顾问。 + +--- + +## 技术栈 + +| 模块 | 选型 | +|------|------| +| 抓取 | Crawl4AI (httpx 静态直连 + Playwright JS 渲染) | +| 正文提取 | GNE 中文新闻正文识别 | +| 事件抽取 | DeepSeek / Qwen (百炼) | +| 向量化 | DashScope text-embedding-v3 / 本地 BGE-M3 | +| 向量库 | Qdrant 本地文件模式 (零依赖,ARM64 兼容) | +| 调度 | APScheduler + systemd | +| 服务化 | MCP (FastMCP) | +| 日报入库 | MySQL/MariaDB (pymysql) | +| 运行时 | Python 3.11 + uv | + +--- + +## 快速开始 + +```bash +# 1. 安装 +cd /home/pi/news +uv sync +cp .env.example .env # 编辑填写 API Key + +# 2. 跑通全链路 +uv run a-share pipeline --once + +# 3. 语义检索 +uv run a-share search "宁德时代固态电池" + +# 4. 查看数据总览 +uv run a-share status +``` + +--- + +## 文档索引 + +| 文档 | 说明 | +|------|------| +| [docs/user-guide.md](docs/user-guide.md) | 用户手册 — CLI 全量参考、全链路、增量/断点续跑、MCP 服务、环境变量、FAQ | +| [docs/architecture.md](docs/architecture.md) | 项目架构 — 11 个包职责、数据模型、配置体系、产物目录 | +| [docs/db-schema.md](docs/db-schema.md) | 日报数据库表结构 — 供 API/前端对接 | +| [CLAUDE.md](CLAUDE.md) | AI 编码约束 | +| [continuation.md](continuation.md) | 会话恢复上下文 | + +--- + +## 当前状态 + +| 已完成 | M0~M10 | +|--------|--------| +| 新闻源 | 14 个 (13 Web + 1 API: 新闻联播) | +| 数据库 | `news_report` / `news_event` (myquant 库) | +| 生产部署 | pi5 `pi@192.168.1.160`, systemd 常驻 | +| 测试 | 225+ passed (pytest) | + +--- + +## 常用命令速查 + +```bash +uv run a-share crawl # M1 抓取 +uv run a-share extract --date 20260822 # M2 提取 +uv run a-share dedup --date 20260822 # M3 去重 +uv run a-share events --date 20260822 # M4 LLM 抽取 +uv run a-share embed --date 20260822 # M5 向量化 +uv run a-share ingest --date 20260822 # M6 入库 +uv run a-share pipeline --once --report # 全链路+日报 +uv run a-share pipeline --once --resume # 断点续跑 +uv run a-share search "关键词" # 语义检索 +uv run a-share status # 数据总览 +uv run a-share report --date 20260822 # 日报生成 +uv run a-share cninfo # 公告抓取 +uv run a-share discover --add # 新增新闻源 +``` + +--- + +## 项目结构 + +``` +news/ +├── crawler/ # M1 新闻抓取 +├── extractor/ # M2 正文提取 +├── dedup/ # M3 三层去重 +├── llm/ # M4 投资事件抽取 +├── embedding/ # M5 向量化 +├── vectorstore/ # M6 Qdrant 知识库 +├── scheduler/ # M7 定时任务 + pipeline +├── mcp_server/ # M8 MCP 服务 +├── a_share_cli/ # 统一 CLI +├── report_db/ # M10 日报 DB 层 +├── report_import/ # M10 历史日报导入 +├── configs/ # 配置文件 +├── prompts/ # LLM Prompt 模板 +├── scripts/ # 独立运行脚本 +├── tests/ # pytest 测试 +├── docs/ # 文档 +└── data/ # 数据产物 +``` + +--- + +—— README 结束 —— \ No newline at end of file diff --git a/a_share_cli/__init__.py b/a_share_cli/__init__.py new file mode 100644 index 0000000..46a63e1 --- /dev/null +++ b/a_share_cli/__init__.py @@ -0,0 +1 @@ +"""A 股 Deep Research 统一 CLI 入口。""" diff --git a/a_share_cli/main.py b/a_share_cli/main.py new file mode 100644 index 0000000..c00c98b --- /dev/null +++ b/a_share_cli/main.py @@ -0,0 +1,985 @@ +"""A 股 Deep Research 统一 CLI。 + +用法: + uv run a-share crawl # M1 抓取 + uv run a-share extract --date 20260616 # M2 提取 + uv run a-share dedup --date 20260616 # M3 去重 + uv run a-share events --date 20260616 # M4 LLM 抽取 + uv run a-share embed --date 20260616 # M5 向量化 + uv run a-share ingest --date 20260616 # M6 入库 + uv run a-share pipeline --once # 全链路 + uv run a-share discover https://xxx.com # 分析站点,推荐配置 + uv run a-share report # 生成日报 + uv run a-share search "宁德时代" # 检索知识库 + uv run a-share status # 状态总览 +""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +import sys +from collections import Counter +from datetime import date, datetime +from pathlib import Path +from typing import Any + +from dotenv import load_dotenv +from loguru import logger + +# --------------------------------------------------------------------------- # +# 工具 +# --------------------------------------------------------------------------- # + +# --------------------------------------------------------------------------- # +# 操作日志 +# --------------------------------------------------------------------------- # + +_OP_LOG_PATH = Path("logs") / "operations.log" + +def _log_operation_start(command: str, detail: str = "") -> None: + """记录操作开始。""" + _OP_LOG_PATH.parent.mkdir(parents=True, exist_ok=True) + ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + line = f"[{ts}] START | {command:20} | {detail}".rstrip() + with _OP_LOG_PATH.open("a", encoding="utf-8") as f: + f.write(line + "\n") + +def _log_operation_end(command: str, success: bool, elapsed: float, detail: str = "") -> None: + """记录操作结束。""" + ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + status = "OK" if success else "FAIL" + detail_str = f" | {detail}" if detail else "" + line = f"[{ts}] END | {command:20} | {status} | {elapsed:.1f}s{detail_str}" + with _OP_LOG_PATH.open("a", encoding="utf-8") as f: + f.write(line + "\n") + +def _wrap_cmd(name: str, func, *cmd_args: object) -> int: + """包装命令函数,自动记录开始/结束日志。""" + from time import perf_counter + detail_parts = [] + for a in cmd_args: + if a is not None and a is not False and a != "" and a != 0: + detail_parts.append(str(a)[:80]) + detail = " ".join(detail_parts) if detail_parts else "" + _log_operation_start(name, detail) + started = perf_counter() + try: + rc = func() + elapsed = perf_counter() - started + _log_operation_end(name, rc == 0, elapsed) + return rc + except Exception as e: + elapsed = perf_counter() - started + _log_operation_end(name, False, elapsed, f"{type(e).__name__}: {e}") + raise + + +def _setup_logger(level: str = "WARNING") -> None: + logger.remove() + logger.add(sys.stderr, level=level, format="{level} | {message}") + + +def _today() -> str: + return date.today().strftime("%Y%m%d") + + +def _resolve_timeout(step_name: str, hardcoded_default: int) -> int: + """解析步骤超时(秒),与 scheduler.pipeline 保持一致的优先级。 + + 优先级: + 1. TIMEOUT_{STEP_NAME} 环境变量 + 2. PIPELINE_STEP_TIMEOUT 环境变量 (全局兜底) + 3. 硬编码默认值 (本函数参数) + """ + import os + specific_key = f"TIMEOUT_{step_name.upper()}" + if specific_key in os.environ: + return int(os.environ[specific_key]) + if "PIPELINE_STEP_TIMEOUT" in os.environ: + return int(os.environ["PIPELINE_STEP_TIMEOUT"]) + return hardcoded_default + + +def _run_module(module: str, extra_args: list[str], timeout: int = 600) -> int: + """调用现有 scripts/run_*.py 模块。""" + cmd = ["uv", "run", "python", "-m", module, *extra_args] + logger.info("执行: {}", " ".join(cmd)) + try: + result = subprocess.run(cmd, timeout=timeout) + return result.returncode + except subprocess.TimeoutExpired: + logger.error("超时 ({}s): {}", timeout, " ".join(cmd)) + return 1 + + +# --------------------------------------------------------------------------- # +# 子命令 +# --------------------------------------------------------------------------- # + +def cmd_crawl(args: argparse.Namespace) -> int: + extra = [] + if args.source: + extra.extend(["--source", args.source]) + if args.no_save: + extra.append("--no-save") + return _run_module("scripts.run_crawler", extra, timeout=_resolve_timeout("crawler", 600)) + + +def cmd_extract(args: argparse.Namespace) -> int: + extra = ["--date", args.date or _today()] + if args.source: + extra.extend(["--source", args.source]) + return _run_module("scripts.run_extractor", extra, timeout=_resolve_timeout("extractor", 300)) + + +def cmd_dedup(args: argparse.Namespace) -> int: + extra = ["--date", args.date or _today()] + if args.reset: + extra.append("--reset") + return _run_module("scripts.run_dedup", extra, timeout=_resolve_timeout("dedup", 120)) + + +def cmd_events(args: argparse.Namespace) -> int: + extra = ["--date", args.date or _today()] + if args.provider: + extra.extend(["--provider", args.provider]) + if args.model: + extra.extend(["--model", args.model]) + if args.limit: + extra.extend(["--limit", str(args.limit)]) + if args.concurrency: + extra.extend(["--concurrency", str(args.concurrency)]) + return _run_module("scripts.run_event_extraction", extra, timeout=_resolve_timeout("llm", 900)) + + +def cmd_embed(args: argparse.Namespace) -> int: + extra = ["--date", args.date or _today()] + if args.provider: + extra.extend(["--provider", args.provider]) + if args.model: + extra.extend(["--model", args.model]) + return _run_module("scripts.run_embedding", extra, timeout=_resolve_timeout("embedding", 300)) + + +def cmd_ingest(args: argparse.Namespace) -> int: + extra = ["--date", args.date or _today()] + if args.recreate: + extra.append("--recreate") + return _run_module("scripts.run_qdrant_ingest", extra, timeout=_resolve_timeout("qdrant", 120)) + + +def cmd_pipeline(args: argparse.Namespace) -> int: + pipeline_timeout = _resolve_timeout("pipeline", 3600) + if args.cninfo_once: + extra = ["--once", "--date", args.date or _today(), + "--steps", "cninfo_crawl,cninfo_extract,cninfo_pdf,dedup,llm,embedding,qdrant"] + return _run_module("scripts.run_scheduler", extra, timeout=pipeline_timeout) + if args.once: + extra = ["--once", "--date", args.date or _today()] + if args.resume: + extra.append("--resume") + steps = args.steps + if args.report: + steps = (steps + ",report") if steps else "report" + if steps: + extra.extend(["--steps", steps]) + return _run_module("scripts.run_scheduler", extra, timeout=pipeline_timeout) + # 守护进程模式:不应通过 CLI 子进程启动(subprocess.run 无法正确管理后台进程)。 + # 直接在当前进程启动 APScheduler。 + print("🔁 启动守护进程模式 (APScheduler) …") + print(" 提示: 生产环境请使用 systemd 管理,见 docs/systemd.md") + from scripts.run_scheduler import _daemon as _run_daemon + # 构造一个简易的 namespace 给 _daemon + daemon_args = argparse.Namespace() + return _run_daemon(daemon_args) + + +# --------------------------------------------------------------------------- # +# search — 直接从终端检索知识库 +# --------------------------------------------------------------------------- # + +def cmd_search(args: argparse.Namespace) -> int: + """嵌入 query → Qdrant 检索 → 格式化输出。""" + load_dotenv() + _setup_logger("WARNING") + + from embedding import make_sync_provider + from vectorstore import SearchFilter, VectorStore, make_qdrant_client + + # 嵌入查询 + print(f"🔍 检索: {args.query}") + emb = make_sync_provider() # 读取 EMBEDDING_PROVIDER 环境变量 + vec = emb.embed_one(args.query) + + # 构建过滤 + filt = None + has_filter = any([ + args.source, args.sentiment, args.min_importance, + args.stock, args.industry, + ]) + if has_filter: + kwargs: dict[str, Any] = {} + if args.source: + kwargs["source_id"] = args.source + if args.sentiment: + kwargs["sentiment"] = args.sentiment + if args.min_importance: + kwargs["importance_min"] = args.min_importance + if args.stock: + # 股票代码可能带或不带后缀,两者都匹配 + code = args.stock.strip().upper() + stock_list = [code] if "." in code else [f"{code}.SZ", f"{code}.SH", f"{code}.BJ"] + kwargs["stock_codes"] = stock_list + if args.industry: + kwargs["industries"] = [args.industry] + filt = SearchFilter(**kwargs) + + # 检索 + client = make_qdrant_client() + store = VectorStore(client) + hits = store.query(query_vector=vec, top_k=args.top, filter=filt, score_threshold=0.2) + store.close() + emb.close() + + if not hits: + print(f"\n未找到与「{args.query}」相关的结果。") + return 0 + + print(f"\n共 {len(hits)} 条结果:\n") + for i, h in enumerate(hits, 1): + ev = h.event or {} + sentiment_icon = {"positive": "🟢", "neutral": "⚪", "negative": "🔴"}.get( + ev.get("sentiment"), "" + ) + print(f"{i}. {sentiment_icon} {h.title}") + print(f" 来源: {h.source_id} | 相似度: {h.score:.4f} | 时间: {h.publish_time}") + if ev.get("stock_codes"): + print(f" 代码: {','.join(ev['stock_codes'])}") + if ev.get("event_type"): + print(f" 事件: {ev.get('event_type')} (重要度 {ev.get('importance', '-')})") + if ev.get("summary"): + print(f" 摘要: {ev['summary']}") + print(f" {h.url}") + print() + + return 0 + + +# --------------------------------------------------------------------------- # +# status — 数据总览 +# --------------------------------------------------------------------------- # + +def cmd_status(args: argparse.Namespace) -> int: # noqa: ARG001 + """输出各层数据统计。""" + load_dotenv() + today_str = _today() + + def _count_jsonl(path: Path) -> int: + if not path.is_file(): + return 0 + return sum(1 for _ in open(path, encoding="utf-8")) + + def _count_dir(pattern: str) -> int: + return len(list(Path().glob(pattern))) + + def _count_json(pattern: str) -> int: + return len(list(Path().glob(pattern))) + + print(f"📊 A 股 Deep Research 状态 — {today_str}") + print("─" * 50) + + # M1 raw + raw_art = 0 + for idx in Path("data/raw").glob(f"*/{today_str}/index.jsonl"): + raw_art += _count_jsonl(idx) + print(f" M1 抓取: {raw_art} 篇原始文章") + + # M2 processed + proc = _count_json(f"data/processed/*/{today_str}/*.json") + print(f" M2 提取: {proc} 篇正文") + + # M3 deduped + deduped = _count_json(f"data/deduped/{today_str}/uniques/*.json") + dup_path = Path(f"data/deduped/{today_str}/duplicates.jsonl") + dups = _count_jsonl(dup_path) if dup_path.is_file() else 0 + print(f" M3 去重: {deduped} 篇唯一, {dups} 篇重复") + + # M4 events + events = _count_json(f"data/events/{today_str}/*.json") + if events > 0: + sentiments: Counter = Counter() + importances: Counter = Counter() + for fp in Path(f"data/events/{today_str}").glob("*.json"): + try: + obj = json.loads(fp.read_text(encoding="utf-8")) + ev = obj.get("event", {}) + sentiments[ev.get("sentiment", "?")] += 1 + importances[ev.get("importance", 0)] += 1 + except (json.JSONDecodeError, OSError): + pass + high = sum(v for k, v in importances.items() if k >= 4) + print(f" M4 事件: {events} 篇 " + f"(🟢{sentiments.get('positive', 0)} " + f"🔴{sentiments.get('negative', 0)} " + f"⚪{sentiments.get('neutral', 0)}, " + f"重要≥4: {high})") + + # M5 embeddings + emb_count = _count_json(f"data/embeddings/{today_str}/*.json") + print(f" M5 向量: {emb_count} 条") + + # cninfo 公告 + cninfo_raw = 0 + for idx in Path("data/raw/cninfo").glob("*/index.jsonl"): + cninfo_raw += _count_jsonl(idx) + if cninfo_raw > 0: + cninfo_proc = _count_json("data/processed/cninfo/*/*.json") + print(f" cninfo: {cninfo_raw} 条公告 (已提取 {cninfo_proc})") + else: + print(" cninfo: 暂无数据") + + # M6 Qdrant + try: + from vectorstore import VectorStore, make_qdrant_client + c = make_qdrant_client() + s = VectorStore(c) + total = s.count() + s.close() + print(f" M6 Qdrant: {total} 条向量") + except Exception: + print(" M6 Qdrant: 未连接") + + # M7 systemd + try: + r = subprocess.run( + ["systemctl", "is-active", "a-share-research"], + capture_output=True, text=True, timeout=5, + ) + svc = r.stdout.strip() + icon = "✅" if svc == "active" else "❌" + print(f" M7 调度: {icon} {svc}") + except Exception: + pass + + print("─" * 50) + + # 高重要度事件 + if events > 0: + print("\n🔥 今日重要度 ≥4 的事件:") + high_events: list[dict] = [] + for fp in Path(f"data/events/{today_str}").glob("*.json"): + try: + obj = json.loads(fp.read_text(encoding="utf-8")) + ev = obj.get("event", {}) + if ev.get("importance", 0) >= 4: + high_events.append({ + "title": obj.get("title", ""), + "source": obj.get("source_id", ""), + "sentiment": ev.get("sentiment", ""), + "importance": ev.get("importance", 0), + "summary": ev.get("summary", ""), + }) + except (json.JSONDecodeError, OSError): + pass + high_events.sort(key=lambda e: -e["importance"]) + for e in high_events[:10]: + icon = {"positive": "🟢", "negative": "🔴", "neutral": "⚪"}.get(e["sentiment"], "") + print(f" {icon} [{e['source']}] {e['title'][:50]} ({e['summary'][:40]})") + + return 0 + + +# --------------------------------------------------------------------------- # +# report — 生成每日摘要 HTML 报告并上传 +# --------------------------------------------------------------------------- # + +def cmd_report(args: argparse.Namespace) -> int: + """生成日报并结构化入库(M10:不再生成 HTML/上传)。""" + load_dotenv() + from scheduler.reporter import generate_report + + day_str = args.date or _today() + print(f"📊 生成日报: {day_str}") + report_id = generate_report(day_str, upload=not args.no_upload) + if report_id is None: + print("⚠️ 无数据或生成失败") + return 1 + print(f"✅ 日报已入库: report_id={report_id}") + return 0 + + +def cmd_report_import(args: argparse.Namespace) -> int: + """历史日报 HTML 解析入库(M10)。""" + load_dotenv() + from report_import.importer import import_history + + report_dir = args.dir or os.environ.get("REPORT_HISTORY_DIR", "data/reports_history") + print(f"📥 导入历史日报: {report_dir} (date={args.date or '全部'} type={args.type or '全部'})") + stats = import_history(report_dir, date_str=args.date, report_type=args.type, force=args.force) + print(f"✅ 扫描 {stats.scanned} | 新导入 {stats.imported} | 跳过 {stats.skipped} | 失败 {stats.failed}") + for err in stats.errors[:20]: + print(f" ❌ {err}") + return 0 if stats.failed == 0 else 1 + + +# --------------------------------------------------------------------------- # +# discover — 自动分析站点,生成 sources.yaml 配置建议 +# --------------------------------------------------------------------------- # + +def cmd_discover(args: argparse.Namespace) -> int: + """抓取首页 → 提取链接 → 按 URL 模式聚类 → 输出 yaml 配置。""" + import re + import urllib.parse + from collections import defaultdict + + print(f"🔍 分析站点: {args.url}") + print("⏳ 正在抓取首页(含 JS 渲染)...\n") + + # 1. 抓取首页 + try: + from crawl4ai import AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig + + async def _fetch(): + bconf = BrowserConfig(headless=True, verbose=False) + rconf = CrawlerRunConfig(cache_mode=CacheMode.BYPASS, page_timeout=30000) + async with AsyncWebCrawler(config=bconf) as crawler: + return await crawler.arun(url=args.url, config=rconf) + + import asyncio + result = asyncio.run(_fetch()) + except ImportError: + print("❌ 需要 crawl4ai 依赖,请确保已安装") + return 1 + except Exception as e: + print(f"❌ 抓取失败: {e}") + return 1 + + if not result or not getattr(result, "success", False): + print(f"❌ 页面抓取失败: {getattr(result, 'error_message', 'unknown')}") + return 1 + + html = getattr(result, "html", "") or "" + if not html: + print("❌ 页面无 HTML 内容") + return 1 + + # 2. 提取所有链接 + from bs4 import BeautifulSoup + soup = BeautifulSoup(html, "html.parser") + parsed_base = urllib.parse.urlparse(args.url) + base_domain = f"{parsed_base.scheme}://{parsed_base.netloc}" + + raw_links: list[tuple[str, str]] = [] + for a in soup.find_all("a", href=True): + href = a["href"].strip() + if not href or href.startswith(("javascript:", "#", "mailto:", "tel:")): + continue + absolute = urllib.parse.urljoin(args.url, href).split("#")[0] + anchor = (a.get_text() or "").strip()[:40] + raw_links.append((absolute, anchor)) + + # 3. 过滤出站内链接,排除导航/静态页 + site_links: list[tuple[str, str]] = [] + for url, anchor in raw_links: + if base_domain not in url: + continue + # 排除明显的导航/静态页 + path = urllib.parse.urlparse(url).path.lower() + if path in ("/", "") or any(path.endswith(ext) for ext in (".css", ".js", ".png", ".jpg", ".ico", ".svg", ".xml", ".pdf")): + continue + site_links.append((url, anchor)) + + if not site_links: + print("❌ 未发现站内链接(可能需要 JS 渲染或页面结构特殊)") + print(" 可尝试手动打开浏览器开发者工具查看网络请求") + return 1 + + unique_links = list(dict.fromkeys(site_links)) # 去重保序 + + # 4. 过滤导航/功能页 + _nav_words = ( + "about", "download", "feedback", "member", "login", "register", "app", + "calendar", "help", "contact", "privacy", "terms", "service", "buy", + "markets", "codes", "real", "lives", + ) + article_links: list[tuple[str, str]] = [] + for url, anchor in unique_links: + path_lower = urllib.parse.urlparse(url).path.lower() + parts_lower = [p for p in path_lower.split("/") if p] + # 跳过明显的导航/功能路径(子串匹配) + if any(nav in p for nav in _nav_words for p in parts_lower if len(p) > 2): + continue + article_links.append((url, anchor)) + + if not article_links: + # 降级:不过滤 + article_links = unique_links + + # 5. 按 URL 路径模式聚类 + clusters: dict[str, list[tuple[str, str]]] = defaultdict(list) + for url, anchor in article_links: + path = urllib.parse.urlparse(url).path.strip("/") + # 把数字/日期/hash 替换为占位符进行聚类 + pattern = re.sub(r"/\d{4,}", "/{id}", path) + pattern = re.sub(r"/[a-f0-9]{32,}", "/{hash}", pattern) + pattern = re.sub(r"/\d{4}-\d{2}-\d{2}", "/{date}", pattern) + # 恢复目录结构作为聚类键 + parts = pattern.split("/") + # 聚类键:用路径前缀(前两级目录) + 最后一段的模式 + key = "/".join(parts[:2]) + "/{...}" if len(parts) >= 2 else (parts[0] if parts else "/") + clusters[key].append((url, anchor)) + + # 过滤只有 1 条的聚类 + clusters = {k: v for k, v in clusters.items() if len(v) >= 2} + if not clusters: + print("⚠️ 未发现明显的文章链接模式(每个 URL 路径都不同)") + print(" 可手动检查页面结构") + return 1 + + # 6. 按链接数排序,但优先数字 ID 模式(文章特征) + def _score(kv: tuple) -> tuple: + _key, _links = kv + has_num_id = bool(re.search(r"/\d{4,}", _links[0][0])) # URL 含长数字 + article_word = any(w in _key.lower() for w in ("article", "news", "detail", "story")) + return (has_num_id or article_word, len(_links)) + + ranked = sorted(clusters.items(), key=_score, reverse=True) + + # 6. 对每个聚类生成正则 + site_host = urllib.parse.urlparse(args.url).netloc.replace(".", r"\.") + results: list[dict] = [] + for cluster_key, links in ranked: + # 从实际 URL 中提取路径前缀来构建更精确的正则 + sample_paths = [urllib.parse.urlparse(u).path for u, _ in links[:5]] + # 找路径公共前缀 + common_prefix = _common_path_prefix(sample_paths) + # 推测数字部分 + if re.search(r"/\d+", sample_paths[0]): + pattern_re = f"^https?://{site_host}{re.escape(common_prefix)}\\d+" + else: + pattern_re = f"^https?://{site_host}{re.escape(common_prefix)}.*" + # 去掉多余的转义 + pattern_re = pattern_re.replace(r"\d+", r"\d+") + results.append({ + "key": cluster_key, + "count": len(links), + "regex": pattern_re, + "samples": links[:3], + }) + + # 7. 输出结果 + for i, r in enumerate(results, 1): + stars = "★★★" if i == 1 else ("★★" if i == 2 else "★") + print(f"模式 {chr(64+i)} ({r['count']} 条, 推荐 {stars}):") + print(f" 正则: {r['regex']}") + print(" 样本:") + for url, anchor in r["samples"]: + print(f" {url}" + (f" [{anchor}]" if anchor else "")) + + # 8. 输出 yaml 建议 + best = results[0] + source_id = parsed_base.netloc.split(".")[0].replace("-", "_") + print() + print("─" * 60) + print("建议 yaml 配置(复制到 configs/sources.yaml):\n") + yaml_snippet = f""" - id: {source_id} + name: {source_id} + enabled: true + homepage: {args.url} + article_url_pattern: '{best['regex']}' + js_render: {str(not getattr(result, 'markdown', '')).lower() if hasattr(result, 'markdown') else 'true'} + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20""" + source_name = args.name or source_id + yaml_snippet = yaml_snippet.replace(f"name: {source_id}", f"name: {source_name}") + + # 有额外入口时,追加 extra_homepages 字段 + if args.extra_urls: + extra_lines = "\n".join(f" - {u}" for u in args.extra_urls) + yaml_snippet += f"\n extra_homepages:\n{extra_lines}" + + print(yaml_snippet) + print("─" * 60) + + if args.add: + yaml_path = Path("configs/sources.yaml") + if yaml_path.is_file(): + content = yaml_path.read_text(encoding="utf-8") + if f"id: {source_id}" in content: + print(f"\n⚠️ sources.yaml 中已存在 id={source_id},跳过添加") + else: + seq = content.count("\n - id:") + 1 + header = f"\n # ---- {seq}. {source_name} ----" + yaml_path.write_text(content.rstrip() + "\n" + header + "\n" + yaml_snippet + "\n", encoding="utf-8") + print(f"\n✅ 已追加到 configs/sources.yaml (第 {seq} 个源)") + else: + print("\n⚠️ configs/sources.yaml 不存在,无法自动添加") + + print(f"\n验证: uv run a-share crawl --source {source_id}") + return 0 + + +# --------------------------------------------------------------------------- # +# add-entry — 为已有源追加 extra_homepages +# --------------------------------------------------------------------------- # + +def cmd_add_entry(args: argparse.Namespace) -> int: + """给已有源追加 extra_homepages 入口(文本模式,保留 yaml 原有格式)。""" + import re + + yaml_path = Path("configs/sources.yaml") + if not yaml_path.is_file(): + print("❌ configs/sources.yaml 不存在") + return 1 + + content = yaml_path.read_text(encoding="utf-8") + + if f"id: {args.source}" not in content: + print(f"❌ 未找到源 id={args.source}") + return 1 + + # 提取该 source 块中已有的 extra_homepages URL + lines = content.splitlines() + in_block = False + existing: set[str] = set() + for line in lines: + if f"id: {args.source}" in line and line.strip().startswith("- id:"): + in_block = True + continue + if in_block and (line.strip().startswith("- id:") or line.strip().startswith("settings:")): + break + m = re.match(r"\s+- (https?://\S+)", line) + if m: + existing.add(m.group(1)) + + added = [] + seen_this_run: set[str] = set() + for url in args.urls: + if url in existing or url in seen_this_run: + print(f"⏭ 跳过(已存在): {url}") + else: + seen_this_run.add(url) + added.append(url) + print(f"✅ 已添加: {url}") + + if not added: + print("无新增入口") + return 0 + + # 在 source 块内插入/追加 + has_extra_header = any("extra_homepages:" in lines[i] for i in range(len(lines)) if in_block_from(lines, i, args.source)) + + if has_extra_header: + # 找到块内最后一个 extra URL 行,在其后追加 + insert_at = -1 + for i in range(len(lines)): + if in_block_from(lines, i, args.source) and re.match(r"\s+- https?://", lines[i]) and "extra_homepages:" in "\n".join(lines[max(0, i - 2):i + 1]): + insert_at = i + if insert_at > 0: + indent = " " + for u in added: + lines.insert(insert_at + 1, f"{indent}- {u}") + insert_at += 1 + else: + # 在 max_articles_per_run 行后插入 extra_homepages + for i in range(len(lines)): + if in_block_from(lines, i, args.source) and "max_articles_per_run:" in lines[i]: + indent = " " + new_block = [f"{indent}extra_homepages:"] + for u in added: + new_block.append(f"{indent} - {u}") + for j, nl in enumerate(new_block): + lines.insert(i + 1 + j, nl) + break + + yaml_path.write_text("\n".join(lines) + "\n", encoding="utf-8") + print(f"\n✅ 已写入 configs/sources.yaml ({len(added)} 个新入口)") + return 0 + + +def in_block_from(lines: list[str], idx: int, source_id: str) -> bool: + """检查行 idx 是否在 source_id 的配置块内。""" + for i in range(idx, -1, -1): + if lines[i].strip().startswith(f"- id: {source_id}"): + return True + if lines[i].strip().startswith("- id:") or lines[i].strip().startswith("settings:"): + return False + return False + + +# --------------------------------------------------------------------------- # +# watchlist 管理 +# --------------------------------------------------------------------------- # + +_WATCHLIST_PATH = Path("configs/watchlist.yaml") + + +def _load_watchlist() -> list[dict]: + import yaml + try: + with _WATCHLIST_PATH.open(encoding="utf-8") as f: + data = yaml.safe_load(f) or {} + return list(data.get("watchlist") or []) + except Exception: + return [] + + +def _save_watchlist(items: list[dict]) -> None: + import yaml + _WATCHLIST_PATH.parent.mkdir(parents=True, exist_ok=True) + with _WATCHLIST_PATH.open("w", encoding="utf-8") as f: + yaml.dump({"watchlist": items}, f, allow_unicode=True, default_flow_style=False, sort_keys=False) + + +def cmd_watchlist_add(args: argparse.Namespace) -> int: + items = _load_watchlist() + code = args.code.strip() + for it in items: + if it.get("code") == code: + it["name"] = args.name + if args.note: + it["note"] = args.note + _save_watchlist(items) + print(f"✅ 已更新: {code} {args.name}") + return 0 + items.append({"code": code, "name": args.name, "note": args.note or ""}) + _save_watchlist(items) + print(f"✅ 已添加: {code} {args.name}") + return 0 + + +def cmd_watchlist_remove(args: argparse.Namespace) -> int: + items = _load_watchlist() + code = args.code.strip() + new_items = [it for it in items if it.get("code") != code] + if len(new_items) == len(items): + print(f"⚠️ 未找到: {code}") + return 1 + _save_watchlist(new_items) + print(f"✅ 已移除: {code}") + return 0 + + +def cmd_watchlist_list(args: argparse.Namespace) -> int: # noqa: ARG001 + items = _load_watchlist() + if not items: + print("📋 关注列表为空") + print("添加: uv run a-share watchlist add 000001 平安银行") + return 0 + print(f"📋 关注列表 ({len(items)} 家):") + for it in items: + note = f" — {it.get('note','')}" if it.get("note") else "" + print(f" {it['code']} {it['name']}{note}") + return 0 + + +def cmd_stock_report(args: argparse.Namespace) -> int: + """生成所有关注公司个股日报。""" + from scheduler.stock_reporter import generate_all_stock_reports + n = generate_all_stock_reports(upload=not args.no_upload) + print(f"✅ 个股报告完成: {n} 家") + return 0 if n > 0 else 1 + + +def cmd_cninfo(args: argparse.Namespace) -> int: + """cninfo watchlist 抓取(公告+调研+互动易,API 优先)。""" + from crawler.cninfo import crawl_watchlist, enrich_articles_with_pdf + + if args.enrich_pdf: + n = enrich_articles_with_pdf(day_str=None, limit=args.pdf_limit) + print(f"✅ PDF 正文提取完成: {n} 篇") + return 0 if n > 0 else 1 + + # 默认: watchlist 全类型抓取(公告+调研+互动易) + results = crawl_watchlist(save=not args.no_save) + + if not results: + print("⚠️ cninfo 抓取: 无数据") + return 1 + + # 按类型统计 + ann = sum(1 for it in results if getattr(it, "item_type", "") == "announcement") + res = sum(1 for it in results if getattr(it, "item_type", "") == "research") + irm = sum(1 for it in results if getattr(it, "item_type", "") == "irm") + print(f"✅ cninfo 抓取完成: 公告 {ann} / 调研 {res} / 互动易 {irm} (共 {len(results)} 条)") + return 0 + + +def _common_path_prefix(paths: list[str]) -> str: + """找路径列表的公共前缀(逐字符)。""" + if not paths: + return "/" + shortest = min(paths, key=len) + for i, ch in enumerate(shortest): + if any(p[i:i+1] != ch for p in paths): + prefix = shortest[:i] + # 截断到最后一个 / + last_slash = prefix.rfind("/") + return prefix[:last_slash + 1] if last_slash >= 0 else "/" + return shortest + + +# --------------------------------------------------------------------------- # +# 主入口 +# --------------------------------------------------------------------------- # + +def main() -> int: + load_dotenv() + parser = argparse.ArgumentParser( + prog="a-share", + description="A 股 Deep Research 统一 CLI", + ) + sub = parser.add_subparsers(dest="command", help="子命令") + + # ---- crawl ---- + p = sub.add_parser("crawl", help="M1 新闻抓取") + p.add_argument("--source", default=None) + p.add_argument("--no-save", action="store_true") + p.set_defaults(func=cmd_crawl) + + # ---- extract ---- + p = sub.add_parser("extract", help="M2 正文提取") + p.add_argument("--date", default=None) + p.add_argument("--source", default=None) + p.set_defaults(func=cmd_extract) + + # ---- dedup ---- + p = sub.add_parser("dedup", help="M3 三层去重") + p.add_argument("--date", default=None) + p.add_argument("--reset", action="store_true") + p.set_defaults(func=cmd_dedup) + + # ---- events ---- + p = sub.add_parser("events", help="M4 LLM 事件抽取") + p.add_argument("--date", default=None) + p.add_argument("--provider", default=None) + p.add_argument("--model", default=None) + p.add_argument("--limit", type=int, default=0) + p.add_argument("--concurrency", type=int, default=0) + p.set_defaults(func=cmd_events) + + # ---- embed ---- + p = sub.add_parser("embed", help="M5 向量化") + p.add_argument("--date", default=None) + p.add_argument("--provider", default=None) + p.add_argument("--model", default=None) + p.set_defaults(func=cmd_embed) + + # ---- ingest ---- + p = sub.add_parser("ingest", help="M6 Qdrant 入库") + p.add_argument("--date", default=None) + p.add_argument("--recreate", action="store_true") + p.set_defaults(func=cmd_ingest) + + # ---- pipeline ---- + p = sub.add_parser("pipeline", help="M7 全链路 / 定时守护") + p.add_argument("--once", action="store_true", help="立即执行一次") + p.add_argument("--date", default=None) + p.add_argument("--steps", default=None, help="指定步骤(crawler,extractor,...,report)") + p.add_argument("--resume", action="store_true", + help="断点续跑(仅 --once):跳过连续成功步骤,从上次失败/未执行步骤继续") + p.add_argument("--report", action="store_true", help="全链路末尾生成日报") + p.add_argument("--cninfo-once", action="store_true", help="cninfo watchlist 全链路(公告+调研+IRM)") + p.set_defaults(func=cmd_pipeline) + + # ---- search ---- + p = sub.add_parser("search", help="检索知识库") + p.add_argument("query", help="自然语言查询") + p.add_argument("--top", type=int, default=10) + p.add_argument("--source", default=None) + p.add_argument("--sentiment", default=None, choices=["positive", "negative", "neutral"]) + p.add_argument("--min-importance", type=int, default=None, dest="min_importance") + p.add_argument("--stock", default=None, help="6 位股票代码") + p.add_argument("--industry", default=None, help="行业名") + p.set_defaults(func=cmd_search) + + # ---- add-entry ---- + p = sub.add_parser("add-entry", help="为已有源追加 extra_homepages 入口") + p.add_argument("source", help="已有源 id(如 wallstreetcn)") + p.add_argument("urls", nargs="+", help="额外入口 URL(可多个)") + p.set_defaults(func=cmd_add_entry) + + # ---- discover ---- + p = sub.add_parser("discover", help="分析站点首页,推荐 sources.yaml 配置") + p.add_argument("url", help="站点首页 URL(主入口)") + p.add_argument("--extra", action="append", default=[], dest="extra_urls", + help="额外入口 URL,可重复使用(同站多频道)") + p.add_argument("--add", action="store_true", help="自动追加到 configs/sources.yaml") + p.add_argument("--name", default=None, help="站点中文名(如 华尔街见闻)") + p.set_defaults(func=cmd_discover) + + # ---- report ---- + p = sub.add_parser("report", help="生成每日 HTML 报告并上传") + p.add_argument("--date", default=None) + p.add_argument("--no-upload", action="store_true", help="仅生成不上传") + p.set_defaults(func=cmd_report) + + # ---- watchlist ---- + p = sub.add_parser("watchlist", help="管理 cninfo 公告关注列表") + sp = p.add_subparsers(dest="wl_cmd") + pa = sp.add_parser("add", help="添加关注公司") + pa.add_argument("code", help="6 位股票代码") + pa.add_argument("name", help="公司简称") + pa.add_argument("--note", default="", help="备注") + pa.set_defaults(func=cmd_watchlist_add) + pr = sp.add_parser("remove", help="移除关注公司") + pr.add_argument("code", help="6 位股票代码") + pr.set_defaults(func=cmd_watchlist_remove) + pl = sp.add_parser("list", help="查看关注列表") + pl.set_defaults(func=cmd_watchlist_list) + + # ---- report-import ---- + p = sub.add_parser("report-import", help="历史日报 HTML 解析入库 (M10)") + p.add_argument("--dir", default=None, help="日报目录(默认 REPORT_HISTORY_DIR)") + p.add_argument("--date", default=None, help="仅导入指定日期 YYYYMMDD") + p.add_argument("--type", default=None, choices=["finance", "intl"], help="仅导入指定类型") + p.add_argument("--force", action="store_true", help="已存在也覆盖重导") + p.set_defaults(func=cmd_report_import) + + # ---- stock-report ---- + p = sub.add_parser("stock-report", help="生成关注列表个股日报") + p.add_argument("--no-upload", action="store_true", help="仅生成不上传") + p.set_defaults(func=cmd_stock_report) + + # ---- cninfo ---- + p = sub.add_parser("cninfo", help="cninfo watchlist 抓取(公告+调研+互动易, API 优先)") + p.add_argument("--no-save", action="store_true") + p.add_argument("--enrich-pdf", action="store_true", help="下载 PDF 补充正文") + p.add_argument("--pdf-limit", type=int, default=50, help="PDF 最多处理 N 篇") + p.set_defaults(func=cmd_cninfo) + + # ---- status ---- + p = sub.add_parser("status", help="数据总览") + p.set_defaults(func=cmd_status) + + args = parser.parse_args() + if args.command is None: + parser.print_help() + return 0 + + # 记录操作日志 + cmd_name = args.command + # 提取关键参数作为日志详情 + detail_parts = [] + for attr in ["source", "date", "query", "days", "stock", "industry", + "sentiment", "once", "cninfo_once", "report"]: + val = getattr(args, attr, None) + if val is not None and val is not False and val != "" and val != 0: + detail_parts.append(f"{attr}={val}") + detail = " ".join(detail_parts) if detail_parts else "" + + from time import perf_counter + _log_operation_start(cmd_name, detail) + started = perf_counter() + rc = args.func(args) + elapsed = perf_counter() - started + _log_operation_end(cmd_name, rc == 0, elapsed) + return rc + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/api/__init__.py b/api/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/app/__init__.py b/app/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/configs/.gitkeep b/configs/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/configs/__init__.py b/configs/__init__.py new file mode 100644 index 0000000..a9b0e6d --- /dev/null +++ b/configs/__init__.py @@ -0,0 +1 @@ +"""configs 配置包:提供 configs/*.yaml 的加载能力。""" diff --git a/configs/llm_models.yaml b/configs/llm_models.yaml new file mode 100644 index 0000000..0957755 --- /dev/null +++ b/configs/llm_models.yaml @@ -0,0 +1,133 @@ +# ============================================================================= +# configs/llm_models.yaml —— 大模型使用场景配置 +# +# 本文件集中配置本项目所有调用 AI 大模型的地方,每个场景可独立指定 +# provider(厂商/服务类型)与 model(模型名),互不影响、单独切换。 +# +# ── 配置优先级(从高到低)────────────────────────────────────────────────────── +# 1. 代码 / 命令行显式参数(如 --provider qwen --model qwen-plus) +# 2. 本文件 scenes.<场景>.xxx +# 3. 环境变量 / .env(LLM_PROVIDER、DEEPSEEK_MODEL 等,向后兼容) +# 4. 代码内置默认值 +# +# ── 规则 ───────────────────────────────────────────────────────────────────── +# · provider 取值: +# 对话大模型: deepseek | qwen(OpenAI 兼容 chat 接口) +# 向量嵌入模型: dashscope | local-bge(仅 embedding 场景) +# · model 留空 = 该场景不覆盖模型 → 回退 .env(如 DEEPSEEK_MODEL / QWEN_MODEL / +# LLM_MODEL);若全部缺失则直接报错,绝不静默使用内置默认模型。 +# · api_key_env / base_url_env 为可选字段,填写存放 API Key / 服务地址的 +# 环境变量名;API Key 一律放 .env,禁止写入本文件(安全规范)。 +# · 修改后无需重启常驻服务即可生效(每次调用重新读取;如需热更新缓存可重启)。 +# ============================================================================= + +# ---- 全局默认参数(各场景可覆盖;低于 .env,高于代码内置默认)---- +defaults: + timeout_sec: 60 + temperature: 0.1 + +scenes: + # ------------------------------------------------------------------------- # + # 场景 1: 投资事件抽取(M4 核心) + # ------------------------------------------------------------------------- # + # 用途:对每条唯一新闻调用大模型,抽取结构化投资事件 + # (stock_codes / company_names / industries / sentiment / importance / + # event_type / summary),输出严格 JSON,经 Pydantic 校验后落盘。 + # 调用方: llm/extractor.py、scripts/run_event_extraction.py、 + # scheduler/pipeline.py 的 llm 步骤。 + # 频率:每日数百~数千篇;建议异步并发(--concurrency,默认 3)。 + # 使用方式: + # uv run a-share events # 读本场景配置 + # uv run a-share events --provider qwen # CLI 临时覆盖 provider + # uv run a-share events --model qwen-max # CLI 临时覆盖模型 + # 对模型的要求: + # · 必须支持 OpenAI 兼容 chat/completions 接口; + # · 必须支持 JSON 结构化输出(response_format=json_object,硬性要求); + # · 上下文窗口 ≥ 16K tokens(单篇正文最多截断到 8000 字符); + # · 中文理解能力强,能区分 A 股事件类型(业绩预告/投资并购/宏观政策等 23 类); + # · 低 temperature(0.1)保证抽取稳定,避免字段抖动。 + # 建议模型: deepseek-v4-flash(生产实测) / deepseek-chat / qwen-plus / qwen-max + event_extraction: + provider: qwen # 建议 deepseek | qwen;留空则回退 .env 的 LLM_PROVIDER + model: qwen3.7-flash # 留空则回退 .env(DEEPSEEK_MODEL → LLM_MODEL) + api_key_env: DASHSCOPE_API_KEY # 例如: DEEPSEEK_API_KEY / QWEN_API_KEY / DASHSCOPE_API_KEY + base_url_env: QWEN_BASE_URL # 例如: DEEPSEEK_BASE_URL / QWEN_BASE_URL + temperature: 0.1 + timeout_sec: 60 + max_attempts: 3 # 单篇解析失败的最大重试次数 + + # ------------------------------------------------------------------------- # + # 场景 2: 日报 AI 摘要 + # ------------------------------------------------------------------------- # + # 用途:每天汇总「新闻联播 + 过去 24h 高重要度新闻 + 近 N 日重要公告/调研」, + # 生成 500 字以内的日报要点摘要,渲染进 HTML 日报。 + # 调用方: scheduler/reporter.py 的 _generate_ai_summary → _llm_summarize。 + # 频率:每天 1 次(07:00 定时任务,与 pipeline 同批)。 + # 使用方式:无需手动触发,定时任务自动执行;失败自动降级(日报留空,不影响入库)。 + # 对模型的要求: + # · OpenAI 兼容 chat 接口(不需要 JSON 输出); + # · 输出长度 ≥ 1500 tokens(max_tokens=1500,输出超长会被截断并记 WARNING); + # · 中文摘要能力强、要点化输出稳定(每条一行,以 "- " 开头); + # · 上下文窗口 ≥ 8K tokens(素材按 3000 字符/块分块,多块先分段再合并); + # · temperature 0.3 左右,兼顾稳定与表达;网络失败按指数退避重试 3 次。 + # · 输出长度需求:分段摘要约 800 tokens、合并摘要约 1500 tokens(代码内置, + # 不在本文件配置),模型应能稳定输出 1500+ tokens 的中文要点。 + # 建议模型: deepseek-v4-flash(生产实测) / deepseek-chat / qwen-plus + daily_report: + provider: # 建议 deepseek | qwen;留空则回退 .env 的 LLM_PROVIDER + model: + api_key_env: + base_url_env: + temperature: 0.3 + timeout_sec: 60 + + # ------------------------------------------------------------------------- # + # 场景 3: 个股 AI 要点分析 + # ------------------------------------------------------------------------- # + # 用途:针对观察清单中的单只股票,汇总其近期公告/调研/新闻/互动问答, + # 生成 5-8 条要点分析,渲染进个股日报。 + # 调用方: scheduler/stock_reporter.py 的 _generate_ai_summary。 + # 频率:每交易日 1 次(07:30),对 watchlist 内每只股票各调用一次。 + # 使用方式:定时任务自动执行;单次失败不影响其他股票(该股显示"AI 摘要暂不可用")。 + # 对模型的要求: + # · OpenAI 兼容 chat 接口(不需要 JSON 输出); + # · 输出长度 ≥ 500 tokens(max_tokens=500); + # · 中文要点分析能力,输入素材最多 3500 字符(公告+调研+新闻+互动问答); + # · temperature 0.3 左右。 + # · 输出长度需求:约 500 tokens(代码内置,不在本文件配置)。 + # 建议模型: deepseek-v4-flash(生产实测) / deepseek-chat / qwen-plus + stock_report: + provider: # 建议 deepseek | qwen;留空则回退 .env 的 LLM_PROVIDER + model: + api_key_env: + base_url_env: + temperature: 0.3 + timeout_sec: 60 + + # ------------------------------------------------------------------------- # + # 场景 4: 文本向量化 Embedding(新闻/事件入库 + 语义检索) + # ------------------------------------------------------------------------- # + # 用途:把新闻正文/事件文本转为向量,写入 Qdrant 知识库;检索时对查询文本 + # 同样向量化后做相似度搜索。M5 入库、MCP 检索、pipeline 均复用本场景。 + # 调用方: embedding/remote.py(远程)、embedding/local.py(本地)、 + # embedding/factory.py、scripts/run_embedding.py、mcp_server/tools.py。 + # 使用方式: + # uv run a-share embed # 读本场景配置 + # uv run a-share embed --provider local-bge # 临时切换本地模型 + # 注意:这是「嵌入模型」而非对话大模型,二选一: + # · provider: dashscope → 阿里百炼 text-embedding-v3(远程,需 API key); + # · provider: local-bge → 本地 BGE-M3(离线,需 uv sync --extra + # local-embedding,首次加载约 2.3GB)。 + # 对模型的要求: + # · 输出固定维度向量(本项目默认 1024 维,DashScope 与本地 BGE-M3 兼容); + # · 中文语义匹配效果好,支持 batch 输入(单批上限 batch_limit 条); + # · 远程需 OpenAI 兼容 embeddings 接口。 + # 建议模型: text-embedding-v3(远程) / BAAI/bge-m3(本地) + embedding: + provider: # 建议 dashscope | local-bge;留空则回退 .env 的 EMBEDDING_PROVIDER + model: # 留空则回退 .env(DASHSCOPE_EMBEDDING_MODEL / LOCAL_EMBEDDING_MODEL) + api_key_env: # 例如: DASHSCOPE_EMBEDDING_API_KEY / DASHSCOPE_API_KEY + base_url_env: # 例如: DASHSCOPE_EMBEDDING_BASE_URL + timeout_sec: 60 + max_attempts: 3 # 单批请求失败重试次数 + batch_limit: 10 # 单批最大条数(百炼实测上限 10,勿调大) diff --git a/configs/llm_models.yaml.bak b/configs/llm_models.yaml.bak new file mode 100644 index 0000000..f2800fe --- /dev/null +++ b/configs/llm_models.yaml.bak @@ -0,0 +1,133 @@ +# ============================================================================= +# configs/llm_models.yaml —— 大模型使用场景配置 +# +# 本文件集中配置本项目所有调用 AI 大模型的地方,每个场景可独立指定 +# provider(厂商/服务类型)与 model(模型名),互不影响、单独切换。 +# +# ── 配置优先级(从高到低)────────────────────────────────────────────────────── +# 1. 代码 / 命令行显式参数(如 --provider qwen --model qwen-plus) +# 2. 本文件 scenes.<场景>.xxx +# 3. 环境变量 / .env(LLM_PROVIDER、DEEPSEEK_MODEL 等,向后兼容) +# 4. 代码内置默认值 +# +# ── 规则 ───────────────────────────────────────────────────────────────────── +# · provider 取值: +# 对话大模型: deepseek | qwen(OpenAI 兼容 chat 接口) +# 向量嵌入模型: dashscope | local-bge(仅 embedding 场景) +# · model 留空 = 该场景不覆盖模型 → 回退 .env(如 DEEPSEEK_MODEL / QWEN_MODEL / +# LLM_MODEL);若全部缺失则直接报错,绝不静默使用内置默认模型。 +# · api_key_env / base_url_env 为可选字段,填写存放 API Key / 服务地址的 +# 环境变量名;API Key 一律放 .env,禁止写入本文件(安全规范)。 +# · 修改后无需重启常驻服务即可生效(每次调用重新读取;如需热更新缓存可重启)。 +# ============================================================================= + +# ---- 全局默认参数(各场景可覆盖;低于 .env,高于代码内置默认)---- +defaults: + timeout_sec: 60 + temperature: 0.1 + +scenes: + # ------------------------------------------------------------------------- # + # 场景 1: 投资事件抽取(M4 核心) + # ------------------------------------------------------------------------- # + # 用途:对每条唯一新闻调用大模型,抽取结构化投资事件 + # (stock_codes / company_names / industries / sentiment / importance / + # event_type / summary),输出严格 JSON,经 Pydantic 校验后落盘。 + # 调用方: llm/extractor.py、scripts/run_event_extraction.py、 + # scheduler/pipeline.py 的 llm 步骤。 + # 频率:每日数百~数千篇;建议异步并发(--concurrency,默认 3)。 + # 使用方式: + # uv run a-share events # 读本场景配置 + # uv run a-share events --provider qwen # CLI 临时覆盖 provider + # uv run a-share events --model qwen-max # CLI 临时覆盖模型 + # 对模型的要求: + # · 必须支持 OpenAI 兼容 chat/completions 接口; + # · 必须支持 JSON 结构化输出(response_format=json_object,硬性要求); + # · 上下文窗口 ≥ 16K tokens(单篇正文最多截断到 8000 字符); + # · 中文理解能力强,能区分 A 股事件类型(业绩预告/投资并购/宏观政策等 23 类); + # · 低 temperature(0.1)保证抽取稳定,避免字段抖动。 + # 建议模型: deepseek-v4-flash(生产实测) / deepseek-chat / qwen-plus / qwen-max + event_extraction: + provider: qwen # 建议 deepseek | qwen;留空则回退 .env 的 LLM_PROVIDER + model: qwen3.7-flash # 留空则回退 .env(DEEPSEEK_MODEL → LLM_MODEL) + api_key_env: DASHSCOPE_API_KEY # 例如: DEEPSEEK_API_KEY / QWEN_API_KEY / DASHSCOPE_API_KEY + base_url_env: QWEN_BASE_URL # 例如: DEEPSEEK_BASE_URL / QWEN_BASE_URL + temperature: 0.1 + timeout_sec: 60 + max_attempts: 3 # 单篇解析失败的最大重试次数 + + # ------------------------------------------------------------------------- # + # 场景 2: 日报 AI 摘要 + # ------------------------------------------------------------------------- # + # 用途:每天汇总「新闻联播 + 过去 24h 高重要度新闻 + 近 N 日重要公告/调研」, + # 生成 500 字以内的日报要点摘要,渲染进 HTML 日报。 + # 调用方: scheduler/reporter.py 的 _generate_ai_summary → _llm_summarize。 + # 频率:每天 1 次(07:00 定时任务,与 pipeline 同批)。 + # 使用方式:无需手动触发,定时任务自动执行;失败自动降级(日报留空,不影响入库)。 + # 对模型的要求: + # · OpenAI 兼容 chat 接口(不需要 JSON 输出); + # · 输出长度 ≥ 1500 tokens(max_tokens=1500,输出超长会被截断并记 WARNING); + # · 中文摘要能力强、要点化输出稳定(每条一行,以 "- " 开头); + # · 上下文窗口 ≥ 8K tokens(素材按 3000 字符/块分块,多块先分段再合并); + # · temperature 0.3 左右,兼顾稳定与表达;网络失败按指数退避重试 3 次。 + # · 输出长度需求:分段摘要约 800 tokens、合并摘要约 1500 tokens(代码内置, + # 不在本文件配置),模型应能稳定输出 1500+ tokens 的中文要点。 + # 建议模型: deepseek-v4-flash(生产实测) / deepseek-chat / qwen-plus + daily_report: + provider: deepseek # 建议 deepseek | qwen;留空则回退 .env 的 LLM_PROVIDER + model: deepseek-v4-flash + api_key_env: DEEPSEEK_API_KEY + base_url_env: DEEPSEEK_BASE_URL + temperature: 0.3 + timeout_sec: 60 + + # ------------------------------------------------------------------------- # + # 场景 3: 个股 AI 要点分析 + # ------------------------------------------------------------------------- # + # 用途:针对观察清单中的单只股票,汇总其近期公告/调研/新闻/互动问答, + # 生成 5-8 条要点分析,渲染进个股日报。 + # 调用方: scheduler/stock_reporter.py 的 _generate_ai_summary。 + # 频率:每交易日 1 次(07:30),对 watchlist 内每只股票各调用一次。 + # 使用方式:定时任务自动执行;单次失败不影响其他股票(该股显示"AI 摘要暂不可用")。 + # 对模型的要求: + # · OpenAI 兼容 chat 接口(不需要 JSON 输出); + # · 输出长度 ≥ 500 tokens(max_tokens=500); + # · 中文要点分析能力,输入素材最多 3500 字符(公告+调研+新闻+互动问答); + # · temperature 0.3 左右。 + # · 输出长度需求:约 500 tokens(代码内置,不在本文件配置)。 + # 建议模型: deepseek-v4-flash(生产实测) / deepseek-chat / qwen-plus + stock_report: + provider: # 建议 deepseek | qwen;留空则回退 .env 的 LLM_PROVIDER + model: + api_key_env: + base_url_env: + temperature: 0.3 + timeout_sec: 60 + + # ------------------------------------------------------------------------- # + # 场景 4: 文本向量化 Embedding(新闻/事件入库 + 语义检索) + # ------------------------------------------------------------------------- # + # 用途:把新闻正文/事件文本转为向量,写入 Qdrant 知识库;检索时对查询文本 + # 同样向量化后做相似度搜索。M5 入库、MCP 检索、pipeline 均复用本场景。 + # 调用方: embedding/remote.py(远程)、embedding/local.py(本地)、 + # embedding/factory.py、scripts/run_embedding.py、mcp_server/tools.py。 + # 使用方式: + # uv run a-share embed # 读本场景配置 + # uv run a-share embed --provider local-bge # 临时切换本地模型 + # 注意:这是「嵌入模型」而非对话大模型,二选一: + # · provider: dashscope → 阿里百炼 text-embedding-v3(远程,需 API key); + # · provider: local-bge → 本地 BGE-M3(离线,需 uv sync --extra + # local-embedding,首次加载约 2.3GB)。 + # 对模型的要求: + # · 输出固定维度向量(本项目默认 1024 维,DashScope 与本地 BGE-M3 兼容); + # · 中文语义匹配效果好,支持 batch 输入(单批上限 batch_limit 条); + # · 远程需 OpenAI 兼容 embeddings 接口。 + # 建议模型: text-embedding-v3(远程) / BAAI/bge-m3(本地) + embedding: + provider: # 建议 dashscope | local-bge;留空则回退 .env 的 EMBEDDING_PROVIDER + model: # 留空则回退 .env(DASHSCOPE_EMBEDDING_MODEL / LOCAL_EMBEDDING_MODEL) + api_key_env: # 例如: DASHSCOPE_EMBEDDING_API_KEY / DASHSCOPE_API_KEY + base_url_env: # 例如: DASHSCOPE_EMBEDDING_BASE_URL + timeout_sec: 60 + max_attempts: 3 # 单批请求失败重试次数 + batch_limit: 10 # 单批最大条数(百炼实测上限 10,勿调大) diff --git a/configs/loader.py b/configs/loader.py new file mode 100644 index 0000000..3179869 --- /dev/null +++ b/configs/loader.py @@ -0,0 +1,69 @@ +"""configs/ 目录下 YAML 配置加载器。 + +目前支持加载 configs/llm_models.yaml 的场景配置(scenes.)。 + +配置优先级(从高到低): + 1. 代码 / 命令行显式参数(如 --provider qwen --model qwen-plus) + 2. 本文件 YAML 场景配置(scenes.) + 3. 环境变量 / .env(LLM_PROVIDER、DEEPSEEK_MODEL 等,向后兼容) + 4. 代码内置默认值 + +说明:API Key 一律放 .env,本文件只保存环境变量名(api_key_env),禁止写密钥。 +""" + +from __future__ import annotations + +from functools import lru_cache +from pathlib import Path + +from loguru import logger + +DEFAULT_CONFIG_PATH = Path("configs/llm_models.yaml") + + +@lru_cache(maxsize=8) +def _load_yaml(path: Path) -> dict: + """读取 YAML 文件为 dict;文件缺失或解析失败返回空 dict(走兜底配置)。""" + if not path.is_file(): + logger.debug("配置文件不存在,使用内置/环境变量兜底: {}", path) + return {} + try: + import yaml + + data = yaml.safe_load(path.read_text(encoding="utf-8")) or {} + except Exception as e: # noqa: BLE001 - YAML 语法错误等 + logger.error("解析 {} 失败: {}", path, e) + return {} + return data if isinstance(data, dict) else {} + + +def load_scene_config(scene: str) -> dict: + """读取 llm_models.yaml 中 scenes. 的配置 dict。 + + 场景不存在或未配置时返回空 dict(调用方回退 .env / 内置默认)。 + scene 为空字符串同样返回空 dict。 + """ + if not scene: + return {} + data = _load_yaml(DEFAULT_CONFIG_PATH) + scenes = data.get("scenes") or {} + cfg = scenes.get(scene) + if cfg is None: + logger.debug("llm_models.yaml 未配置场景 {!r},使用 .env 兜底", scene) + return {} + if not isinstance(cfg, dict): + logger.warning("llm_models.yaml 场景 {!r} 应为 map,已忽略", scene) + return {} + return cfg + + +def load_defaults() -> dict: + """读取 llm_models.yaml 顶层 defaults(全局默认参数)。""" + data = _load_yaml(DEFAULT_CONFIG_PATH) + d = data.get("defaults") or {} + return d if isinstance(d, dict) else {} + + +def clear_cache() -> None: + """清空 YAML 缓存(测试或热更新配置时使用)。""" + _load_yaml.cache_clear() diff --git a/configs/sources.yaml b/configs/sources.yaml new file mode 100644 index 0000000..ddc860b --- /dev/null +++ b/configs/sources.yaml @@ -0,0 +1,218 @@ +# A 股新闻源配置(M1) +# ============================================================ +# 文档: +# - id: 短标识,用作目录名(小写字母/数字/下划线) +# - name: 中文名,用于日志与展示 +# - homepage: 列表/首页 URL,抓取入口 +# - article_url_pattern: 正则,从首页所有链接中筛选文章 URL +# - article_link_selector: 可选 CSS,优先在该选择器内查链接(留空=全页) +# - js_render: 是否需要 JS 渲染(动态站点必须 true) +# - wait_for: 可选 CSS,等待元素出现(JS 渲染时建议设置) +# - max_articles_per_run: 单次抓取上限,M1 阶段建议 ≤ 30 +# ============================================================ + +settings: + concurrency: 3 # 并发抓取上限(用户决策:3) + retry_max_attempts: 3 + retry_min_wait_sec: 1.0 + retry_max_wait_sec: 10.0 + headless: true + output_root: data/raw + +sources: + # ---- 1. 财联社(深度频道)---- + - id: cls + name: 财联社 + enabled: true + homepage: https://www.cls.cn/depth?id=1000 + article_url_pattern: '^https?://www\.cls\.cn/detail/\d+$' + article_link_selector: null + js_render: true + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.cls.cn/depth?id=1003 + - https://www.cls.cn/depth?id=1007 + - https://www.cls.cn/depth?id=1035 + - https://www.cls.cn/depth?id=1032 + - https://www.cls.cn/depth?id=1005 + + # ---- 2. 东方财富(财经频道)---- + - id: eastmoney + name: 东方财富 + enabled: true + homepage: https://finance.eastmoney.com/ + article_url_pattern: '^https?://finance\.eastmoney\.com/a/\d+\.html$' + article_link_selector: null + js_render: true + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + + # ---- 3. 新浪财经 ---- + - id: sina + name: 新浪财经 + enabled: true + homepage: https://finance.sina.com.cn/ + # 新浪文章 URL 形如 https://finance.sina.com.cn/stock/.../doc-xxx.shtml + article_url_pattern: '^https?://(finance|cj)\.sina\.com\.cn/.+/doc-[a-z0-9]+\.s?html$' + article_link_selector: null + js_render: false + wait_for: null + page_timeout_ms: 20000 + max_articles_per_run: 20 + extra_homepages: + - https://finance.sina.com.cn/stock/newstock/ + - https://finance.sina.com.cn/stock/ + - https://finance.sina.com.cn/forex/ + - https://finance.sina.com.cn/stock/hkstock/ + - https://finance.sina.com.cn/stock/usstock/ + + # ---- 4. 证券时报 ---- + - id: stcn + name: 证券时报 + enabled: true + homepage: https://www.stcn.com/article/list/yw.html + article_url_pattern: '^https?://www\.stcn\.com/article/detail/\d+\.html$' + article_link_selector: null + js_render: true + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.stcn.com/article/list/xw.html + - https://www.stcn.com/article/list/finance.html + - https://www.stcn.com/article/list/gd.html + + # ---- 5. 第一财经 ---- + - id: yicai + name: 第一财经 + enabled: true + homepage: https://www.yicai.com/news/ + article_url_pattern: '^https?://www\.yicai\.com/(news|brief)/\d+\.html$' + article_link_selector: null + js_render: true + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.yicai.com/brief/ + + # ---- 6. 华尔街见闻 ---- + - id: wallstreetcn + name: 华尔街见闻 + enabled: true + # 注意: 该站为 React SPA,js_render 触发反爬拦截(Crawl4AI 内部错误), + # 非 JS 模式下 Crawl4AI 可抓取首页链接但文章正文提取为空。 + # TODO: 研究通过 wallstreetcn API 或 RSS 获取全文 + homepage: https://wallstreetcn.com/news/global + article_url_pattern: '^https?://wallstreetcn\.com/(articles|livenews)/\d+' + js_render: false + wait_for: null + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://wallstreetcn.com/live/global + + # ---- 7. 新华网 ---- + - id: newscn + name: 新华网 + enabled: true + homepage: https://www.news.cn/finance/ + article_url_pattern: '^https?://www\.news\.cn/finance/\d+' + js_render: false + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.news.cn/money/yw/index.html + + # ---- 8. 中国经济网 ---- + - id: cecn + name: 中国经济网 + enabled: true + homepage: http://finance.ce.cn/ + article_url_pattern: '^https?://[a-z0-9]+\.ce\.cn/.+/t\d{8}_\d+\.shtml$' + js_render: false + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - http://finance.ce.cn/rolling/index.shtml + + # ---- 9. 中新经纬 ---- + - id: jwview + name: 中新经纬 + enabled: true + homepage: https://www.jwview.com/jr.html + article_url_pattern: '^https?://www\.jwview\.com/jingwei/html/\d{2}-\d{2}/\d+' + js_render: false + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.jwview.com/kb.html + + # ---- 10. 中国证券网 ---- + - id: cnstock + name: 中国证券网 + enabled: true + homepage: https://www.cnstock.com/channel/10011 + article_url_pattern: '^https?://www\.cnstock\.com/commonDetail/\d+' + js_render: false + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.cnstock.com/fastNews/10004 + + # ---- 11. 中证券网 ---- + - id: cscn + name: 中证券网 + enabled: true + homepage: https://www.cs.com.cn/yaowen.html + article_url_pattern: '^https?://www\.cs\.com\.cn/xwzx/\d{2}/\d{4}/\d{2}/\d+' + js_render: false + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - https://www.cs.com.cn/gongsi.html + + # ---- 12. 证券日报 ---- + - id: zqrb + name: 证券日报 + enabled: true + homepage: http://www.zqrb.cn/finance/index.html + # zqrb URL 包含日期: /gscy/gongsi/2026-06-17/A1781692801973.html + # 限制只匹配 2025 年后的文章,避免抓取到 2022 年旧闻 + article_url_pattern: '^https?://www\.zqrb\.cn/(?:gscy/gongsi|finance)/(?:202[5-9]|203\d)-\d{2}-\d{2}/' + js_render: false + wait_for: "css:body" + page_timeout_ms: 30000 + max_articles_per_run: 20 + extra_homepages: + - http://www.zqrb.cn/finance/hongguanjingji/index.html + - http://www.zqrb.cn/finance/guojijingji/index.html + + # ---- 13. 经济观察网 ---- + - id: eeo + name: 经济观察网 + enabled: true + homepage: https://www.eeo.com.cn/ + article_url_pattern: '^https?://www\.eeo\.com\.cn/\d{4}/\d{4}/\d+\.shtml$' + js_render: false + page_timeout_ms: 60000 + max_articles_per_run: 20 + extra_homepages: + - https://www.eeo.com.cn/jg/kuaixun/ + + # ---- 14. 新闻联播 (API, 不走 Playwright) ---- + - id: xwlb + name: 新闻联播 + enabled: true + homepage: https://api.doorcome.cn/api/xwlbFine/ # dummy, 满足模型校验 + article_url_pattern: '^xwlb://' # dummy, 不会被 crawler 处理 + js_render: false + max_articles_per_run: 25 diff --git a/configs/watchlist.yaml b/configs/watchlist.yaml new file mode 100644 index 0000000..1a9eeb0 --- /dev/null +++ b/configs/watchlist.yaml @@ -0,0 +1,80 @@ +# cninfo 公告关注列表 +# code: 6 位股票代码 +# name: 公司简称 +# note: 可选备注 +# orgId: cninfo 机构 ID(从 cninfo 搜索框输入代码后 URL 中获取) +watchlist: + - code: "000792" + name: "盐湖股份" + note: "盐湖提锂" + orgId: "gssz0000792" + + - code: "300316" + name: "晶盛机电" + note: "" + orgId: "9900022122" + + - code: "688556" + name: "高测股份" + note: "" + orgId: "gfbj0834278" + + - code: "688005" + name: "容百科技" + note: "" + orgId: "9900038937" + + - code: "300124" + name: "汇川技术" + note: "" + orgId: "9900012527" + + - code: "002241" + name: "歌尔股份" + note: "" + orgId: "9900004688" + + - code: "600900" + name: "长江电力" + note: "" + orgId: "gssh0600900" + + - code: "601126" + name: "四方股份" + note: "" + orgId: "9900016847" + + - code: "600089" + name: "特变电工" + note: "" + orgId: "gssh0600089" + + - code: "601179" + name: "中国西电" + note: "" + orgId: "9900010350" + + - code: "688676" + name: "金盘科技" + note: "" + orgId: "9900044215" + + - code: "688469" + name: "芯联集成" + note: "" + orgId: "9900053045" + + - code: "002129" + name: "TCL中环" + note: "" + orgId: "9900002703" + + - code: "600460" + name: "士兰微" + note: "" + orgId: "gssh0600460" + + - code: "601668" + name: "中国建筑" + note: "" + orgId: "9900005970" diff --git a/continuation.md b/continuation.md new file mode 100644 index 0000000..1e4d951 --- /dev/null +++ b/continuation.md @@ -0,0 +1,386 @@ +# continuation.md + +> `checkpoint` @ 2026-08-22 + +--- + +## 本次完成 (2026-08-22) — 文档清理与重构 + +**目标:** 清理历史上多个 AI agent 文档残留,重构项目文档为 4 个核心文件。 + +**删除的过时/残留文件 (4 个):** + +1. `project_plan.md` (700 行) — 9 个 Milestone 全部完成,计划文档已无价值 +2. `docs/agent_prompt.md` (80 行) — Cherry Studio Agent 使用说明,非本项目核心文档 +3. `docs/optimization_plan.md` (144 行) — 8 项优化中 5 项已完成,剩余不再跟踪 +4. `docs/report_db_design.md` (330 行) — M10 中间设计稿,逻辑已下沉到代码 + +**新建/重写的文档 (5 个):** + +| 文件 | 行数 | 说明 | +|------|------|------| +| `docs/architecture.md` | ~500 | 项目架构:11 包职责与导出 API、核心数据模型、配置体系、产物目录 | +| `docs/user-guide.md` | ~500 | 用户手册:15 个 CLI 子命令全量 + 增量/断点续跑 + MCP + systemd + FAQ | +| `README.md` | ~120 | 精简为项目入口:简介 + 文档索引 + 命令速查 | +| `docs/db-schema.md` | 保留 | 日报表结构(API/前端对接,无需改动) | +| `deploy/README.md` | 删除 | 已在服务器上直接修改,无需部署文档 | + +**docs/ 最终目录:** + +``` +docs/ +├── architecture.md # 项目架构(新增) +├── user-guide.md # 用户手册(重写) +└── db-schema.md # 日报表结构(保留) +``` + +**验证:** + +- 所有文档中文撰写,Markdown 格式,层次清晰 +- architecture.md 准确反映 11 个包的实际公开 API(基于源码 `__init__.py` 导出) +- user-guide.md 覆盖全部 15 个 CLI 子命令(crawl/extract/dedup/events/embed/ingest/pipeline/search/status/report/report-import/cninfo/stock-report/watchlist/discover/add-entry) +- 无过时信息残留(不再提 M9 Agent/HTML 日报/旧版 Mac 部署等) + +**待办:** + +- git 提交本次改动(包括 2026-08-12 增量处理改动,仍未提交) +- 同步 pi5(代码 + 文档),重启 a-share-research + +--- + +--- + +## 当前状态 + +| 项目 | 值 | +| --- | --- | +| 新闻源 | 14 个(13 Web + 1 API: xwlb 新闻联播) | +| Qdrant | 本地文件模式 `data/qdrant_storage/` | +| 日报 | **M10 完成并已部署 pi5: 结构化入库 MySQL;日报按当天日期生成(新闻 30h 回溯 / xwlb 取前一日 / 公告调研近 15 日)** | +| DB 连接 | pi 上 systemd 服务 `a-share-db-tunnel` 常驻(0.0.0.0:13306 → doorcome.cn:3306);**pi5 直连 192.168.1.10:13306** | +| 调度器 | APScheduler,systemd `a-share-research.service`(pi5);每天 07:00 首次任务生成日报(12/18/22 点不生成) | +| LLM | 场景化配置 `configs/llm_models.yaml`(4 场景: event_extraction/daily_report/stock_report/embedding);YAML 优先、`.env` 兜底;模型必须显式配置,无内置兜底 | +| 去重 | 多源记录:指纹库 `source_ids` 列 + uniques JSON `sources` 字段 + `data/deduped/{day}/sources.json` | +| 增量/断点 | M2/M4/M5 产物存在即跳过(`--force` 全量);`pipeline --once --resume` 断点续跑(状态 `data/pipeline/state.json`) | +| 服务器 | `pi@192.168.1.160`(生产)/ `pi@192.168.1.10`(DB 隧道宿主) | +| 抓取方式 | js_render=false → httpx 直连;js_render=true → Playwright | + +--- + +## 本次完成 (2026-08-12) — 增量处理与 pipeline 断点续跑 + +**目标:** ① 全链路中断后可从断点恢复;② 各子任务排除已处理文件,避免全量重跑与重复 API 计费。 + +**1. 各步骤增量处理(产物存在即跳过,`--force` 全量):** +- M2 `run_extractor.py`:输出目录已有 `{url_hash}.json` 即跳过提取,仅回补 index 行;`--force` 重建;成功率统计含跳过项(修复全跳过时误报 rc=1) +- M4 `run_event_extraction.py`:`data/events/{day}/{url_hash}.json` 已存在即跳过(**不重复调用 LLM API**);`--force` 全量;failed.jsonl 只保留本次失败、index 累积追加 +- M5 `run_embedding.py`:`data/embeddings/{day}/{url_hash}.json` 已存在即跳过(**不重复调用 embed API**);`--force` 全量 +- M1(seen_urls 增量)/ M3(指纹库判重)/ M6(upsert 幂等)为既有能力,README 汇总成表 + +**2. pipeline 断点续跑(scheduler/pipeline.py + run_scheduler.py):** +- 新增 `data/pipeline/state.json`(按日期隔离,记录每步骤 ok/failed + 退出码 + 耗时),原子写 +- `run_pipeline(resume=True)` 跳过连续成功前缀,从首个失败/未执行步骤继续执行到结尾 +- `pipeline --once --resume`(默认全量不变;`--resume` 与 `--steps` 互斥报错);定时守护模式不受影响 + +**验证(Mac 本地,20260616 数据 100 篇):** +- pytest **225 passed**(新增 tests/test_incremental.py 10 个:M2/M4/M5 跳过、状态记录、resume 续跑、resume 全完成 noop、--resume+--steps 互斥);crawler 3 个基线失败仍与本次无关 +- ruff 零新增(9 个基线错误不变) +- 端到端:M2 增量重跑 140 条跳过 126 条,2.5s 完成、rc=0(修复前误报失败);state.json 正确记录 extractor ok + +**效果评估(100 篇规模中断重跑场景):** +- M4 减少约 100 次 LLM API 调用、M5 减少约 100 次 embedding API 调用 → 中断恢复不再重复计费,耗时从分钟级降至秒级 +- M2 重跑从全量 GNE 提取(分钟级)降至约 2.5s +- 断点恢复操作:中断后直接重跑同一条 `pipeline --once --resume` 命令即可 + +**待办:** +- 同步 pi5(代码 + 文档),重启 `a-share-research`;首次同步后 pi5 的 `data/pipeline/state.json` 不存在 → resume 按全量处理,行为安全 +- git 提交(本次改动尚未提交) + +--- + +## 本次完成 (2026-08-05) — 日报可靠性修复与取数逻辑优化 + +**目标:** 解决日报 AI 摘要偶发失败;修正日报日期与 xwlb/新闻取数语义。 + +**1. AI 摘要可靠性(llm/client.py + scheduler/reporter.py):** +- 去掉内置默认模型兜底(`deepseek-chat`/`qwen-plus`),模型必须显式配置(`DEEPSEEK_MODEL`/`QWEN_MODEL` → `LLM_MODEL`),缺失即报错,避免静默用错模型 +- `_llm_call` 增加指数退避重试:`_LLM_RETRY_TIMES`(默认 3 次)、`_LLM_RETRY_BACKOFF_SEC`(默认 2.0s,可 .env 覆盖),全部失败才抛异常 +- 确认 AI 摘要模型:`deepseek` + `deepseek-v4-flash`(生产实测) + +**2. 日报取数逻辑(scheduler/pipeline.py + reporter.py):** +- pipeline report 步骤:`report_date = date.today()`(原为昨天+回溯 3 天) +- `_collect_news_events`:读当天+前一天事件目录,按 `publish_time` 过滤最近 30 小时(`_NEWS_LOOKBACK_HOURS=30`);时区统一(naive 假定本地时区);无时间戳事件保留 +- `_collect_xwlb`:固定查 `day_str` 前一日(《新闻联播》19:00 播出,早间日报只能取昨晚已播出的一期);`source_date` 同步为实际来源日 +- **公告/调研/互动保持原设置:近 15 日(`STOCK_REPORT_DAYS=15`),irm 互动仍跳过**——未受 30h 改动影响 + +**验证(pi5):** +- 单测 9 个(30h 回溯/带时区时间戳/重试/模型缺失报错/xwlb 前一日)全部通过;全量 212 passed + 3 crawler 预存在失败 +- 生产端到端 report_id=191(2026-08-05):news 20 + cninfo 20 + xwlb 34(08-04 联播),AI 摘要 2076 字 +- 生产服务已重启生效 + +**本次代码尚未 git 提交(见待办)。** + +--- + +## 本次完成 (2026-08-03 22:00) — pi5 部署与生产测试 + +**操作:** M10 代码全量同步 pi5 + 生产环境验收(所有测试在 pi5 执行,Mac 不再作为测试环境)。 + +**文件同步:** rsync 本地 → `pi5:/home/pi/news/`(排除 .venv/data/logs/.git/.env/configs/*.yaml);清除 pi5 根目录 6 月 17 日旧版散文件(已 tar 备份 /tmp/news_backup_20260803.tar.gz);pi5 `uv sync` 装 pymysql。 + +**DB 隧道修复:** pi 原 autossh 参数 `-L 13306:0.0.0.0:3306` 未生效(0.0.0.0 被当远端目标),改为 `-L 0.0.0.0:13306:127.0.0.1:3306` 并持久化为 systemd 服务 `a-share-db-tunnel`(enabled + active)。 + +**测试结果(pi5):** +- 全量 pytest:**208 passed, 3 failed**(3 个失败均为 crawler retry mock 预先存在问题,与 M10 无关) +- `report-import` 幂等:已存在文件正确 skipped +- 生产端到端:`a-share report --date 20260802/20260803` → report_id=181/182 入库成功(AI 摘要正常,57/58 事件) +- DB 总量:finance 52 + intl 128 = 180 行(历史 177 + 端到端测试 2 + 本机 1) +- 生产服务 `a-share-research` 已重启 active,22:00 定时任务起用新代码 + +**已知问题:** +- `tests/test_crawler.py` 3 个 retry 测试失败(预先存在,crawl4ai mock 行为) +- Mac 本机 anaconda/uv python 出站到 192.168.1.10 被拦截(EHOSTUNREACH;nc/bash/系统 python 正常)——仅影响本机,pi5 不受影响;Mac 本地用 `127.0.0.1` + ssh 隧道绕过 +- 生产 pi5 的 `.env` 里 `NEWS_DB_HOST=192.168.1.10`(直连);Mac 本地 `.env` 为 `127.0.0.1`(隧道)——**两处 .env 不同,勿互相覆盖** + +--- + +## 本次完成 (2026-08-03) — M10 日报结构化入库 + +**目标:** 日报前后端分离的数据层——日报内容结构化存入 MySQL(`news_` 前缀表),历史 178 份日报 HTML 解析入库;本项目不做 API/前端(用户另行实现)。 + +**表结构(myquant 库,见 docs/db_schema.md):** +- `news_report`:主表,唯一键 (report_date, report_type, file_name);`file_name=''` 表示新生成日报(每天每类型一行,重复生成覆盖) +- `news_event`:事件明细,section ∈ xwlb/news/cninfo/intl + +**新增/修改文件:** +- `report_db/`(models/schema/db):连接、建表、幂等 save_report +- `report_import/`(parser/importer):历史 HTML 解析(表头驱动列映射)+ 批量导入 +- `scheduler/reporter.py`:`generate_report()` 完全切换为结构化入库;`_render_html/_upload` 标记废弃保留 +- `a_share_cli/main.py`:新增 `report-import` 子命令;`report` 命令改为入库提示 +- `pyproject.toml`:`uv add pymysql`(唯一新增依赖) +- `.env`:新增 `NEWS_DB_*`(连接 127.0.0.1:13306,需 ssh 隧道)、`REPORT_HISTORY_DIR` +- `docs/report_db_design.md`(实现逻辑)、`docs/db_schema.md`(表结构,供 API/前端) + +**验证结果:** +- 历史 178 份全部入库:scanned=178 imported=7(首轮)+170 skipped=171 failed=0;DB 177 行(finance 49 + intl 128,1 个跨目录同名文件被幂等合并)+ 事件 4222 条 +- 端到端:`a-share report --date 20260616` → report_id=180 入库成功,无 HTML 产出 +- 单测 24 个通过(parser 19 + builder 4 + models/import 补充) + +**命令:** +```bash +ssh -L 13306:127.0.0.1:13306 pi # Mac 开发连库隧道 +uv run a-share report --date YYYYMMDD # 生成日报入库 +uv run a-share report-import # 历史导入(幂等) +``` + +**注意:** 本机 .env 无 LLM API key,AI 摘要会降级 WARNING(不影响入库);生产 pi5 需配置 NEWS_DB_PASSWORD 且解决 13306 隧道可达性(见待确认项)。 + +--- + +## 历史 (2026-07-17) — 禁用个股日报 + +**操作:** `STOCK_REPORT_TIME=` 设为空,`run_scheduler.py` 加空值守卫。 + +**文件:** +- `scripts/run_scheduler.py` L127-141:`STOCK_REPORT_TIME` 为空时跳过注册并输出 `已禁用个股日报` +- Pi `.env`:`STOCK_REPORT_TIME=08:00` → `STOCK_REPORT_TIME=` + +**恢复方法:** 改回 `STOCK_REPORT_TIME=08:00`,重启 `a-share-research`。 + +--- + +## 本次完成 (2026-07-13) — eeo 反爬修复 + 静态直连 + 超时保护 + +### 问题 + +eeo(经济观察网)在 Pi 上用 Playwright 连续抓取后触发反爬,服务器故意不响应导致 `page.goto()` 超时(30s → 60s 均超时)。而在本机 Mac 上一切正常——反爬是 **IP 级别**的,Pi 的 IP 已被限流。 + +### 根因 + +eeo 是纯静态页面(`js_render: false`),但 Crawl4AI 始终走 Playwright。eeo 检测到 headless Chrome 的连续请求模式后,对后续请求挂起 TCP 连接直至超时。 + +### 修复 + +**`crawler/engine.py` — 三个改动:** + +1. **新增 `_fetch_static()`**:`js_render=False` 且无 `wait_for` 的源用 `httpx` 直连,绕过 Playwright 反爬。HTML → Markdown 用 `markdownify` + BeautifulSoup 清洗。 + +2. **新增 `_TimeoutGuard`**:同源连续超时 2 次后自动跳过剩余请求,防止整个 pipeline 被单一故障源拖死。 + +3. **`_crawl_once` 分流逻辑**: + ``` + js_render=False 且无 wait_for → _fetch_static() (httpx) + 其他 → Playwright + ``` + +**`configs/sources.yaml` — eeo 源两处调整:** +- 删除 `wait_for: "css:body"`(静态页面无需等待 CSS 选择器) +- `page_timeout_ms: 30000 → 60000`(恢复 Crawl4AI 默认值) + +### 测试结果 + +| 环境 | Playwright (修复前) | httpx (修复后) | +|------|---------------------|-----------------| +| Mac 本地 | 全部通过 | 36 篇 / 7 秒 | +| Pi 服务器 | **前 3 个请求全部 15s 超时** | 36 篇 / 9 秒,0 失败 | + +### 验证方法 + +Pi 上本地复现了反爬(Playwright 前 3 个 URL 全部超时),同批 URL 用 httpx 全部瞬间返回——证明方案有效。 + +--- + +## 历史 (2026-06-24) + +### 1. 环境变量全面审计与修复 + +**触发:** 服务器 `dedup` 步骤超时 120s。 + +**修复 6 个文件:** + +| # | 文件 | 修改 | +|---|------|------| +| 1 | `scheduler/pipeline.py` | 超时解析重写:`TIMEOUT_{NAME}` > `PIPELINE_STEP_TIMEOUT` > 硬编码 > 1800s | +| 2 | `a_share_cli/main.py` | 新增 `_resolve_timeout()`,CLI 命令超时改为读环境变量 | +| 3 | `crawler/cninfo.py` | `CNINFO_PDF_BASE` 从硬编码改为 `os.environ.get()` | +| 4 | `scheduler/reporter.py` | `STOCK_REPORT_DAYS` 默认值 7→15;`max_tokens` 600→1500 | +| 5 | `.env` | 清除 10 个死配置;`EMBEDDING_MODEL`→`LOCAL_EMBEDDING_MODEL`;补全遗漏参数 | +| 6 | `.env.example` | 重写为仅包含实际生效配置 | + +**审计结果:** + +| 类别 | 数量 | 处理 | +|------|------|------| +| ✅ 正常工作 | 23 | — | +| ❌ 死配置 | 10 | 已清除 | +| ⚠️ 默认值不一致 | 1 | 已统一 | +| 🔧 硬编码 | 1 | 已修复 | + +### 2. AI 摘要截断修复 + +**触发:** 日报 AI 摘要内容极少(137 字),末尾被截断。 + +**根因:** `_llm_call()` `max_tokens=600` 太低,deepseek-v4-flash 输出被硬截断。 + +**修复:** `max_tokens` 三处提升 + 截断检测日志。 + +| 位置 | 修复前 | 修复后 | +|------|--------|--------| +| `_llm_call()` 默认 | 600 | 1500 | +| 分块摘要 | 400 | 800 | +| 合并摘要 | 600 | 1500 | + +**验证:** 手动生成日报,摘要从 137 字 → 486 字,结构完整无截断。 + +--- + +## 超时优先级(最终设计) + +``` +TIMEOUT_{NAME} 环境变量 ← 单步精确控制 + ↓ 未设 +PIPELINE_STEP_TIMEOUT 环境变量 ← 全局兜底 (覆盖所有硬编码) + ↓ 未设 +STEP_TIMEOUTS 硬编码字典 ← 代码内默认值 + ↓ 步骤不在字典中 +1800s ← 最终兜底 +``` + +--- + +## 服务器 .env 当前配置 + +```env +# ---- 调度 ---- +SCHEDULE_TIMES=07:00,12:00,18:00,22:00 +CNINFO_SCHEDULE_TIME=06:00 +STOCK_REPORT_TIME=08:00 +STOCK_REPORT_DAYS=15 + +# ---- Pipeline 超时 ---- +PIPELINE_STEP_TIMEOUT=3600 +TIMEOUT_CRAWLER=900 +TIMEOUT_XWLB=60 +TIMEOUT_EXTRACTOR=300 +TIMEOUT_DEDUP=900 +TIMEOUT_LLM=900 +TIMEOUT_EMBEDDING=900 +TIMEOUT_QDRANT=1800 +TIMEOUT_CNINFO_CRAWL=900 +TIMEOUT_CNINFO_EXTRACT=900 +TIMEOUT_CNINFO_PDF=900 +``` + +--- + +## 调度器 + +| 时间 | 步骤 | +|------|------| +| 06:00 | cninfo 公告管道 | +| 07:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant→**日报** | +| 08:00 | 个股日报 | +| 12:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | +| 18:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | +| 22:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | + +--- + +## 常用命令 + +```bash +# 全链路 + 日报 +uv run a-share pipeline --once --report + +# 仅日报 (指定日期) +/home/pi/.local/bin/uv run python -m scripts.run_scheduler --once --date 20260623 --steps report + +# 检索 +uv run a-share search "关键词" --top 5 + +# 同步到服务器 +scp pi@192.168.1.160:/home/pi/news// + +# 重启服务 +ssh pi@192.168.1.160 "sudo systemctl restart a-share-research" +``` + +--- + +## 已知技术债务 + +| 问题 | 文件 | 状态 | +|------|------|------| +| QdrantClient HTTP 超时硬编码 10s | `vectorstore/client.py:93` | 无可配环境变量 | +| DashScope HTTP 超时硬编码 60s | `embedding/remote.py:81` | 无可配环境变量 | +| 超时解析逻辑在 pipeline.py 和 main.py 各一份 | 两处 | 可抽取公共函数 | +| `deepseek-v4-flash` 概率性返回空 | — | 分块处理 + 失败跳过 | +| wallstreetcn JS 渲染拦截 | — | 正文提取率 0% | +| eeo 列表页仍走 Playwright(仅文章页走 httpx) | `crawler/engine.py` | 列表页目前正常,暂不改动 | + +--- + +## 架构备忘:静态/动态抓取分流 + +``` +SourceConfig.js_render == False 且 wait_for == None + → _fetch_static() → httpx 直连 (无浏览器, 快, 不被反爬) + → markdownify + BeautifulSoup (清洗 script/style 后转 Markdown) + +SourceConfig.js_render == True 或 wait_for 有值 + → _crawl_once() → Playwright (现有逻辑) +``` + +`_TimeoutGuard`: +- 同源连续超时 2 次 → `skip=True` +- 后续请求在 `semaphore` 内检查 `guard.skip`,立即返回 `Skipped` +- 一次成功即重置计数器 + +--- + +## 记忆 + +- `never-change-llm-model.md` — 绝不允许修改大模型配置 +- `deploy-html.md` — user_guide.html 同步到 Pi 和 Web 服务器 +- `sync-sources-yaml.md` — 修改 sources.yaml 前从 Pi 拉取,改完推回 +- `update-user-guide.md` — 更新 md+html+部署 diff --git a/crawler/__init__.py b/crawler/__init__.py new file mode 100644 index 0000000..1b082bc --- /dev/null +++ b/crawler/__init__.py @@ -0,0 +1,32 @@ +"""新闻抓取模块 (M1)。 + +公共 API: + - load_crawler_config: 加载 sources.yaml + - crawl_all: 一键抓取所有启用的源 + - crawl_source: 抓取单个源(供测试/调试) + - CrawlerConfig / SourceConfig / CrawlResult: 数据模型 +""" + +from .config import load_crawler_config +from .engine import crawl_all, crawl_source, extract_article_links +from .models import ( + ArticleLink, + CrawlerConfig, + CrawlerSettings, + CrawlResult, + CrawlStage, + SourceConfig, +) + +__all__ = [ + "ArticleLink", + "CrawlResult", + "CrawlStage", + "CrawlerConfig", + "CrawlerSettings", + "SourceConfig", + "crawl_all", + "crawl_source", + "extract_article_links", + "load_crawler_config", +] diff --git a/crawler/cninfo.py b/crawler/cninfo.py new file mode 100644 index 0000000..b4f01b4 --- /dev/null +++ b/crawler/cninfo.py @@ -0,0 +1,499 @@ +"""cninfo 巨潮资讯网爬虫 v2.0。 + +从 watchlist.yaml 的 code + orgId 拼接 URL,通过 Crawl4AI(Playwright)渲染 SPA 页面提取数据。 +三种数据类型: + 1. 公司最新公告 → https://www.cninfo.com.cn/new/disclosure/stock?stockCode={code}&orgId={orgId}#latestAnnouncement + 2. 投资者调研 → 同上, #research + 3. 互动易问答 → https://irm.cninfo.com.cn/ircs/search?keyword={code} + +策略: + - 公告/调研: Playwright SPA 渲染 → DOM 提取(A 方案,REST API 的 stock 参数不可靠) + - 互动易: requests 优先,失败回退 Playwright + - 增量: ann_id 去重,首次从 2026-01-01 全量,后续增量 7 天 +""" + +from __future__ import annotations + +import asyncio +import hashlib +import json +import os +import re +import time +from datetime import date, datetime, timedelta +from pathlib import Path +from typing import Any + +import requests +from bs4 import BeautifulSoup +from loguru import logger + +from .models import CninfoItem + +# --------------------------------------------------------------------------- # +# 常量 +# --------------------------------------------------------------------------- # + +SOURCE_ID = "cninfo" +CNINFO_PDF_BASE = os.environ.get("CNINFO_PDF_BASE", "http://static.cninfo.com.cn") + +# 日期范围: 从 2026-01-01 开始 +DATE_START = "2026-01-01" + +# 翻页限制 +MAX_ANNOUNCE_PAGES = 5 +MAX_RESEARCH_PAGES = 1 +MAX_IRM_ITEMS = 20 + +# 请求间隔(秒) +REQUEST_DELAY = 0.5 + + +# --------------------------------------------------------------------------- # +# 工具函数 +# --------------------------------------------------------------------------- # + +def _url_hash(url: str) -> str: + return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16] + + +def _today_str() -> str: + return date.today().strftime("%Y%m%d") + + +def _load_watchlist() -> list[dict]: + import yaml + try: + with open("configs/watchlist.yaml", encoding="utf-8") as f: + data = yaml.safe_load(f) or {} + return list(data.get("watchlist") or []) + except Exception: + logger.exception("加载 watchlist.yaml 失败") + return [] + + +def _make_api_headers() -> dict[str, str]: + return { + "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " + "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36", + "Accept": "text/html,application/xhtml+xml", + } + + +def _build_stock_url(code: str, org_id: str) -> str: + """拼接 cninfo 个股公告页 URL。""" + return f"https://www.cninfo.com.cn/new/disclosure/stock?stockCode={code}&orgId={org_id}" + + +# --------------------------------------------------------------------------- # +# Crawl4AI SPA 渲染 +# --------------------------------------------------------------------------- # + +async def _render_page(url: str, timeout_ms: int = 60000, + delay_ms: int = 20) -> str: + """用 Crawl4AI 渲染 SPA 页面,返回 HTML 字符串。""" + from crawl4ai import AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig + + bconf = BrowserConfig(headless=True, verbose=False) + rconf = CrawlerRunConfig( + cache_mode=CacheMode.BYPASS, + page_timeout=timeout_ms, + delay_before_return_html=delay_ms, + ) + async with AsyncWebCrawler(config=bconf) as c: + result = await c.arun(url=url, config=rconf) + return getattr(result, "html", "") or "" + + +# --------------------------------------------------------------------------- # +# 公告 + 调研: 从渲染后的 SPA 页面 DOM 提取 +# --------------------------------------------------------------------------- # + +def _parse_announcement_list(html: str, code: str, + item_type: str = "announcement") -> list[CninfoItem]: + """从 cninfo 个股页面的渲染 HTML 中提取公告/调研列表。 + + 页面结构: 公告列表以 标签呈现,关键属性: + - data-seccode: 股票代码 + - data-id: 公告 ID + - href: 包含 announcementTime / announcementType 等参数 + """ + soup = BeautifulSoup(html, "html.parser") + items: list[CninfoItem] = [] + seen: set[str] = set() + + # 查找所有带 data-seccode=code 的公告链接 + for el in soup.find_all(attrs={"data-seccode": code}): + ann_id = (el.get("data-id") or "").strip() + if not ann_id or ann_id in seen: + continue + seen.add(ann_id) + + title = el.get_text(strip=True) + if len(title) < 5: + continue + + href = el.get("href", "") + + # 从 href 提取发布时间 + pub_time = "" + date_match = re.search(r"announcementTime=(\d{4}-\d{2}-\d{2})", href) + if date_match: + pub_time = date_match.group(1) + + # 公告类型 + ann_type = "" + type_match = re.search(r"announcementType=(\w+)", href) + if type_match: + ann_type = type_match.group(1) + + # 板块代码 + plate = "" + plate_match = re.search(r"plate=(\w+)", href) + if plate_match: + plate = plate_match.group(1) + + # PDF URL + pdf_url = "" + if pub_time and ann_id: + pdf_url = f"{CNINFO_PDF_BASE}/finalpage/{pub_time}/{ann_id}.PDF" + + items.append(CninfoItem( + stock_code=code, + stock_name="", + title=title, + content="", + publish_time=pub_time, + item_type=item_type, + url=pdf_url, + ann_id=ann_id, + extra={"announcement_type": ann_type, "plate": plate}, + )) + + return items + + +def _parse_irm_text(text: str, code: str) -> list[CninfoItem]: + """从互动易页面文本中提取问答。 + + 注意: cninfo 互动易搜索页是 Vue SPA,问答数据通过需认证的 API 加载。 + 公开访问时页面显示"暂无数据",此时应返回空列表。 + """ + items: list[CninfoItem] = [] + url = f"https://irm.cninfo.com.cn/ircs/search?keyword={code}" + + # 检测是否为无数据页面 + if "暂无数据" in text: + logger.debug(" {} 互动易页面显示'暂无数据'(需登录认证)", code) + return items + + # 检测是否为 SPA 空壳(无 JS 渲染时只有导航文本) + text_stripped = text.strip() + if len(text_stripped) < 200 and code in text_stripped: + logger.debug(" {} 互动易页面内容过短(SPA 空壳)", code) + return items + + # 按常见分隔模式拆分问答块 + blocks = re.split(r"\n(?=\d+\.|\b[问答][::])", text) + for i, block in enumerate(blocks[:MAX_IRM_ITEMS]): + block = block.strip() + if len(block) > 30 and code in block: + items.append(CninfoItem( + stock_code=code, + stock_name="", + title=f"{code} 互动问答 #{i + 1}", + content=block[:5000], + publish_time="", + item_type="irm", + url=url, + ann_id=_url_hash(f"{url}#{i}"), + )) + return items + + +# --------------------------------------------------------------------------- # +# 互动易 +# --------------------------------------------------------------------------- # + +async def _fetch_irm_playwright(code: str) -> list[CninfoItem]: + """Playwright 渲染互动易搜索页。""" + url = f"https://irm.cninfo.com.cn/ircs/search?keyword={code}" + try: + html = await _render_page(url, timeout_ms=30000, delay_ms=8) + except Exception as e: + logger.warning(" {} 互动易 Playwright 失败: {}", code, e) + return [] + + soup = BeautifulSoup(html, "html.parser") + body_text = soup.get_text() + return _parse_irm_text(body_text, code) + + +def _fetch_irm_requests(code: str) -> list[CninfoItem]: + """requests 获取互动易搜索页。""" + url = f"https://irm.cninfo.com.cn/ircs/search?keyword={code}" + headers = _make_api_headers() + headers["Referer"] = "https://irm.cninfo.com.cn/" + + try: + r = requests.get(url, headers=headers, timeout=15) + r.raise_for_status() + except Exception as e: + logger.debug(" {} 互动易 requests 失败: {}", code, e) + return [] + + soup = BeautifulSoup(r.text, "html.parser") + body_text = soup.get_text() + return _parse_irm_text(body_text, code) + + +# --------------------------------------------------------------------------- # +# 单股票抓取 +# --------------------------------------------------------------------------- # + +async def _crawl_one_stock_async(stock: dict, + start_date: str = DATE_START, + end_date: str | None = None) -> list[CninfoItem]: + """异步抓取单个公司的公告 + 调研 + 互动易。""" + code = stock["code"] + name = stock["name"] + org_id = stock.get("orgId", "").strip() + + if not org_id: + logger.warning("{} ({}) 未配置 orgId,跳过", code, name) + return [] + + end_date = end_date or date.today().strftime("%Y-%m-%d") + base_url = _build_stock_url(code, org_id) + results: list[CninfoItem] = [] + + # --- 1) 公告 (#latestAnnouncement) --- + announce_url = f"{base_url}#latestAnnouncement" + logger.info("抓取 {} ({}) 公告: {}", code, name, announce_url[:80]) + try: + html = await _render_page(announce_url, timeout_ms=60000, delay_ms=20) + # 日期过滤: 只保留 start_date 之后的 + items = _parse_announcement_list(html, code, item_type="announcement") + filtered = [it for it in items if it.publish_time >= start_date] + # 限制页数: 每个页面约 30 条, 5 页 ≈ 150 条(SPA 一次加载可能超过一页) + filtered = filtered[:MAX_ANNOUNCE_PAGES * 30] + for it in filtered: + it.stock_name = name + results.extend(filtered) + logger.info(" {} 公告: {} 条 (过滤后)", code, len(filtered)) + except Exception as e: + logger.error(" {} 公告抓取失败: {}", code, e) + + # --- 2) 调研 (#research) --- + research_url = f"{base_url}#research" + logger.info("抓取 {} ({}) 调研: {}", code, name, research_url[:80]) + try: + html = await _render_page(research_url, timeout_ms=60000, delay_ms=20) + items = _parse_announcement_list(html, code, item_type="research") + filtered = [it for it in items if it.publish_time >= start_date] + filtered = filtered[:MAX_RESEARCH_PAGES * 30] + for it in filtered: + it.stock_name = name + results.extend(filtered) + logger.info(" {} 调研: {} 条 (过滤后)", code, len(filtered)) + except Exception as e: + logger.warning(" {} 调研抓取失败(可能无调研页面): {}", code, e) + + # --- 3) 互动易 --- + logger.info("抓取 {} ({}) 互动易", code, name) + try: + irm_items = _fetch_irm_requests(code) + if not irm_items: + logger.info(" {} 互动易 requests 无结果,回退 Playwright...", code) + irm_items = await _fetch_irm_playwright(code) + for it in irm_items: + it.stock_name = name + results.extend(irm_items) + logger.info(" {} 互动易: {} 条", code, len(irm_items)) + except Exception as e: + logger.error(" {} 互动易抓取失败: {}", code, e) + + return results + + +# --------------------------------------------------------------------------- # +# 增量保存 +# --------------------------------------------------------------------------- # + +def _load_seen_ids(out_dir: Path) -> set[str]: + """从 index.jsonl 加载已保存的公告 ID 集合。""" + seen: set[str] = set() + index_path = out_dir / "index.jsonl" + if index_path.is_file(): + with index_path.open("r", encoding="utf-8") as f: + for line in f: + try: + rec = json.loads(line.strip()) + aid = rec.get("ann_id", "") + if aid: + seen.add(aid) + except (json.JSONDecodeError, KeyError): + continue + return seen + + +def _save_items(items: list[CninfoItem], out_dir: Path) -> int: + """增量保存 CninfoItem 列表,跳过已存在的 ann_id。返回新增条数。""" + out_dir.mkdir(parents=True, exist_ok=True) + seen = _load_seen_ids(out_dir) + logger.info("cninfo 增量模式: 已有 {} 条历史记录", len(seen)) + + index_path = out_dir / "index.jsonl" + new_count = 0 + + with index_path.open("a", encoding="utf-8") as index_f: + for item in items: + if item.ann_id and item.ann_id in seen: + continue + seen.add(item.ann_id) + new_count += 1 + + fname = _url_hash(item.ann_id or item.title + item.stock_code) + json_path = out_dir / f"{fname}.json" + json_path.write_text( + item.model_dump_json(indent=2, ensure_ascii=False), + encoding="utf-8", + ) + + meta = item.model_dump(mode="json") + meta["source_id"] = SOURCE_ID + meta["json_file"] = f"{fname}.json" + index_f.write(json.dumps(meta, ensure_ascii=False) + "\n") + + logger.info("cninfo 已保存: 新增 {} 条 (总计 {} 条) -> {}", new_count, len(seen), out_dir) + return new_count + + +# --------------------------------------------------------------------------- # +# 主入口 +# --------------------------------------------------------------------------- # + +def crawl_watchlist(*, save: bool = True) -> list[CninfoItem]: + """从 watchlist 抓取所有公司的公告+调研+互动易。 + + 首次运行从 2026-01-01 开始,后续增量运行(最近 7 天)。 + 通过判定 data/raw/cninfo/ 下是否有历史 index.jsonl 来区分首次/增量。 + """ + watchlist = _load_watchlist() + if not watchlist: + logger.warning("关注列表为空") + return [] + + out_dir = Path("data/raw") / SOURCE_ID / _today_str() + + # 判断首次还是增量 + has_history = False + raw_root = Path("data/raw") / SOURCE_ID + if raw_root.is_dir(): + for d in sorted(raw_root.glob("*"), reverse=True): + idx = d / "index.jsonl" + if idx.is_file() and idx.stat().st_size > 0: + has_history = True + break + + if has_history: + start = (date.today() - timedelta(days=7)).strftime("%Y-%m-%d") + logger.info("cninfo 增量模式: 日期范围 {} ~ 今天", start) + else: + start = DATE_START + logger.info("cninfo 首次全量: 日期范围 {} ~ 今天", start) + + end = date.today().strftime("%Y-%m-%d") + + # 并发抓取所有股票 + async def _run_all(): + tasks = [_crawl_one_stock_async(s, start_date=start, end_date=end) + for s in watchlist] + all_items: list[CninfoItem] = [] + for coro in asyncio.as_completed(tasks): + try: + items = await coro + all_items.extend(items) + except Exception as e: + logger.error("某股票抓取异常: {}", e) + return all_items + + all_items = asyncio.run(_run_all()) + + # 统计 + ann_count = sum(1 for it in all_items if it.item_type == "announcement") + res_count = sum(1 for it in all_items if it.item_type == "research") + irm_count = sum(1 for it in all_items if it.item_type == "irm") + logger.info( + "cninfo 抓取完成: 公告 {} / 调研 {} / 互动易 {} ({} 家公司,共 {} 条)", + ann_count, res_count, irm_count, len(watchlist), len(all_items), + ) + + if save and all_items: + new_count = _save_items(all_items, out_dir) + logger.info("cninfo 本轮新增 {} 条", new_count) + + return all_items + + +# --------------------------------------------------------------------------- # +# PDF 正文提取 +# --------------------------------------------------------------------------- # + +def enrich_articles_with_pdf(day_str: str | None = None, *, limit: int = 0) -> int: + """下载 PDF 并用 MarkItDown 提取正文补充到 Article.content。""" + day_str = day_str or _today_str() + proc_dir = Path("data/processed") / SOURCE_ID / day_str + if not proc_dir.is_dir(): + return 0 + + from markitdown import MarkItDown + + enriched = 0 + for fp in sorted(proc_dir.glob("*.json")): + if limit and enriched >= limit: + break + try: + article_data = json.loads(fp.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError): + continue + + content = article_data.get("content", "") + pdf_url = "" + + # 多种方式提取 PDF URL + pdf_match = re.search(r"原文链接:\s*(https?://\S+\.pdf)", content, re.IGNORECASE) + if pdf_match: + pdf_url = pdf_match.group(1) + else: + pdf_match = re.search(r"PDF链接:\s*(https?://\S+\.pdf)", content, re.IGNORECASE) + if pdf_match: + pdf_url = pdf_match.group(1) + else: + url = article_data.get("url", "") + if url.lower().endswith(".pdf"): + pdf_url = url + + if not pdf_url: + continue + + try: + r = requests.get(pdf_url, headers=_make_api_headers(), timeout=30) + r.raise_for_status() + tmp_path = Path("/tmp") / f"cninfo_pdf_{_url_hash(pdf_url)}.pdf" + tmp_path.write_bytes(r.content) + md = MarkItDown() + result = md.convert(str(tmp_path)) + text = (result.text_content or "").strip()[:20000] + if tmp_path.exists(): + tmp_path.unlink() + if text and len(text) >= 50: + article_data["content"] = text + article_data["word_count"] = len(text) + fp.write_text(json.dumps(article_data, ensure_ascii=False, indent=2), + encoding="utf-8") + enriched += 1 + except Exception as e: + logger.warning("PDF 提取失败 {}: {}", pdf_url, e) + + return enriched diff --git a/crawler/config.py b/crawler/config.py new file mode 100644 index 0000000..02a8782 --- /dev/null +++ b/crawler/config.py @@ -0,0 +1,52 @@ +"""抓取配置加载器。 + +从 YAML 文件读取 sources.yaml,并校验为 CrawlerConfig。 +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any + +import yaml +from loguru import logger + +from .models import CrawlerConfig + +DEFAULT_CONFIG_PATH = Path("configs/sources.yaml") + + +def load_crawler_config(path: str | Path | None = None) -> CrawlerConfig: + """加载并校验抓取配置。 + + 参数: + path: YAML 配置路径,默认 configs/sources.yaml。 + + 返回: + CrawlerConfig 实例。 + + 异常: + FileNotFoundError: 配置文件不存在。 + ValueError: YAML 顶层不是 mapping 或 sources 缺失。 + pydantic.ValidationError: schema 校验失败。 + """ + config_path = Path(path) if path else DEFAULT_CONFIG_PATH + if not config_path.is_file(): + raise FileNotFoundError(f"抓取配置文件不存在: {config_path}") + + with config_path.open("r", encoding="utf-8") as f: + raw: Any = yaml.safe_load(f) + + if not isinstance(raw, dict): + raise ValueError(f"配置文件顶层必须是 mapping,实际类型: {type(raw).__name__}") + if "sources" not in raw: + raise ValueError("配置文件缺少 sources 字段") + + config = CrawlerConfig.model_validate(raw) + logger.info( + "已加载抓取配置: 总 {} 个源, 启用 {} 个, 并发上限 {}", + len(config.sources), + len(config.enabled_sources()), + config.settings.concurrency, + ) + return config diff --git a/crawler/engine.py b/crawler/engine.py new file mode 100644 index 0000000..4eb31e0 --- /dev/null +++ b/crawler/engine.py @@ -0,0 +1,445 @@ +"""Crawl4AI 异步抓取引擎。 + +设计: + 1. 一个 AsyncWebCrawler 实例服务所有源(共用浏览器,降低开销); + 2. 每个 URL 独立的 CrawlerRunConfig(wait_for/timeout 来自源配置); + 3. 全局 Semaphore 控制并发(默认 3); + 4. 每个 URL 用 tenacity 重试(默认 3 次,指数退避); + 5. 两阶段: 先抓首页 -> 解析文章链接 -> 并发抓文章详情。 +""" + +from __future__ import annotations + +import asyncio +import re +from typing import Any +from urllib.parse import urljoin, urlparse + +import httpx +from bs4 import BeautifulSoup +from crawl4ai import AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig +from loguru import logger +from markdownify import markdownify as _md_convert + +from .models import ( + ArticleLink, + CrawlerConfig, + CrawlerSettings, + CrawlResult, + CrawlStage, + SourceConfig, +) +from .storage import mark_url_seen, save_result + +# --------------------------------------------------------------------------- # +# 工具函数 +# --------------------------------------------------------------------------- # + +def _markdown_text(crawl4ai_markdown: Any) -> str: + """从 Crawl4AI 的 markdown 字段中安全取出字符串。 + + 新版 Crawl4AI(0.4+) 返回 MarkdownGenerationResult 对象; + 旧版可能直接是 str。这里做兼容。 + """ + if crawl4ai_markdown is None: + return "" + if isinstance(crawl4ai_markdown, str): + return crawl4ai_markdown + raw = getattr(crawl4ai_markdown, "raw_markdown", None) + if isinstance(raw, str): + return raw + fit = getattr(crawl4ai_markdown, "fit_markdown", None) + if isinstance(fit, str): + return fit + return str(crawl4ai_markdown) + + +def extract_article_links( + html: str, + base_url: str, + source: SourceConfig, +) -> list[ArticleLink]: + """从列表页 HTML 中抽取文章链接,使用源配置的正则筛选。 + + 步骤: + 1. 若设置了 article_link_selector,先 narrow 到选择器范围; + 2. 找出所有 ; + 3. urljoin 转绝对 URL; + 4. 用 article_url_pattern 正则筛选; + 5. 去重保持顺序; + 6. 截断到 max_articles_per_run。 + """ + soup = BeautifulSoup(html, "html.parser") + scope = ( + soup.select_one(source.article_link_selector) + if source.article_link_selector + else soup + ) + if scope is None: + logger.warning( + "源 {} 设置了 article_link_selector={!r},但页面中未匹配到", source.id, source.article_link_selector + ) + return [] + + pattern = re.compile(source.article_url_pattern) + seen: set[str] = set() + links: list[ArticleLink] = [] + + for a in scope.find_all("a", href=True): + href = a["href"].strip() + if not href or href.startswith(("javascript:", "#", "mailto:")): + continue + absolute = urljoin(base_url, href) + # 去掉 fragment + absolute = absolute.split("#", 1)[0] + if not pattern.match(absolute): + continue + if absolute in seen: + continue + seen.add(absolute) + anchor = (a.get_text() or "").strip() or None + links.append(ArticleLink(source_id=source.id, url=absolute, anchor_text=anchor)) + if len(links) >= source.max_articles_per_run: + break + + logger.info( + "源 {} 列表页抽取链接 {} 条 (上限 {})", + source.id, + len(links), + source.max_articles_per_run, + ) + return links + + +# --------------------------------------------------------------------------- # +# 单 URL 抓取 +# --------------------------------------------------------------------------- # + +def _build_run_config(source: SourceConfig) -> CrawlerRunConfig: + """根据源配置构造 CrawlerRunConfig。""" + return CrawlerRunConfig( + cache_mode=CacheMode.BYPASS, + wait_for=source.wait_for, + page_timeout=source.page_timeout_ms, + excluded_tags=["script", "style", "noscript"], + ) + + +class _TimeoutGuard: + """单源连续超时计数器,达到阈值后标记跳过,防止整个进程被拖死。""" + + def __init__(self, source_id: str, max_consecutive: int = 2) -> None: + self.source_id = source_id + self.max_consecutive = max_consecutive + self._count = 0 + self.skip = False + + def record(self, error: str | None, success: bool) -> None: + """根据抓取结果更新计数器。""" + if self.skip: + return + if success: + self._count = 0 + return + if error and "timeout" in error.lower(): + self._count += 1 + if self._count >= self.max_consecutive: + self.skip = True + logger.warning( + "源 {} 连续超时 {} 次,跳过剩余请求", + self.source_id, + self._count, + ) + + def skipped_result(self, url: str, stage: CrawlStage) -> CrawlResult: + """生成"已跳过"结果。""" + return CrawlResult( + source_id=self.source_id, + stage=stage, + url=url, + success=False, + error="Skipped: 源连续超时已跳过", + attempts=0, + ) + + +async def _fetch_static( + url: str, + source: SourceConfig, + stage: CrawlStage, + attempt: int, +) -> CrawlResult: + """使用 httpx 直连抓取静态页面(js_render=False),绕过 Playwright 反爬检测。""" + headers = { + "User-Agent": ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " + "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36" + ), + } + timeout_sec = source.page_timeout_ms / 1000.0 + try: + async with httpx.AsyncClient( + timeout=timeout_sec, + follow_redirects=True, + headers=headers, + ) as client: + r = await client.get(url) + html = r.text + # HTML → Markdown:先清洗再转换 + markdown = "" + if html: + try: + soup = BeautifulSoup(html, "html.parser") + for tag in soup(["script", "style", "noscript", "meta", "link"]): + tag.decompose() + markdown = _md_convert(str(soup)) or "" + except Exception: + markdown = "" + return CrawlResult( + source_id=source.id, + stage=stage, + url=url, + success=True, + status_code=r.status_code, + html=html, + markdown=markdown, + attempts=attempt, + ) + except Exception as e: + logger.warning("静态抓取异常 {} {}: {}", source.id, url, e) + return CrawlResult( + source_id=source.id, + stage=stage, + url=url, + success=False, + error=f"{type(e).__name__}: {e}", + attempts=attempt, + ) + + +async def _crawl_once( + crawler: AsyncWebCrawler, + url: str, + source: SourceConfig, + stage: CrawlStage, + attempt: int, +) -> CrawlResult: + """单次抓取(无重试),用于被 retry 循环包裹。 + + - js_render=False 且无 wait_for → httpx 直连,绕过浏览器反爬; + - 其他情况 → Playwright。 + """ + # 静态页面走 httpx 直连 + if not source.js_render and not source.wait_for: + return await _fetch_static(url, source, stage, attempt) + + # 动态页面走 Playwright + run_config = _build_run_config(source) + try: + c4_result = await crawler.arun(url=url, config=run_config) + except Exception as e: + logger.warning("抓取异常 {} {}: {}", source.id, url, e) + return CrawlResult( + source_id=source.id, + stage=stage, + url=url, + success=False, + error=f"{type(e).__name__}: {e}", + attempts=attempt, + ) + + success = bool(getattr(c4_result, "success", False)) + return CrawlResult( + source_id=source.id, + stage=stage, + url=url, + success=success, + status_code=getattr(c4_result, "status_code", None), + html=getattr(c4_result, "html", "") or "", + markdown=_markdown_text(getattr(c4_result, "markdown", None)), + error=None if success else (getattr(c4_result, "error_message", None) or "unknown"), + attempts=attempt, + ) + + +async def crawl_url_with_retry( + crawler: AsyncWebCrawler, + url: str, + source: SourceConfig, + stage: CrawlStage, + settings: CrawlerSettings, + semaphore: asyncio.Semaphore, + guard: _TimeoutGuard | None = None, +) -> CrawlResult: + """带并发限制与指数退避重试的单 URL 抓取。 + + 重试策略: + 最多 settings.retry_max_attempts 次; + 失败 (success=False) 触发重试,等待时间 min(min_wait * 2^(n-1), max_wait)。 + + 若传入 guard,每次重试前检查是否已被跳过。 + """ + async with semaphore: + # 检查是否已被跳过 + if guard and guard.skip: + return guard.skipped_result(url, stage) + + max_attempts = settings.retry_max_attempts + result: CrawlResult | None = None + + for attempt_no in range(1, max_attempts + 1): + result = await _crawl_once(crawler, url, source, stage, attempt_no) + if result.success: + if guard: + guard.record(None, True) + return result + if attempt_no >= max_attempts: + break + # 重试前检查是否已被其他并发请求触发跳过 + if guard and guard.skip: + return guard.skipped_result(url, stage) + wait = min( + settings.retry_min_wait_sec * (2 ** (attempt_no - 1)), + settings.retry_max_wait_sec, + ) + logger.info( + "源 {} {} 第 {}/{} 次失败,{} 秒后重试: {}", + source.id, + url, + attempt_no, + max_attempts, + wait, + result.error, + ) + await asyncio.sleep(wait) + + # 此处 result 必非 None(循环至少跑一次) + assert result is not None + if guard: + guard.record(result.error, False) + return result + + +# --------------------------------------------------------------------------- # +# 抓取流程 +# --------------------------------------------------------------------------- # + +async def crawl_source( + crawler: AsyncWebCrawler, + source: SourceConfig, + settings: CrawlerSettings, + semaphore: asyncio.Semaphore, + save: bool = True, +) -> list[CrawlResult]: + """抓取单个新闻源:遍历所有入口 → 抽取链接(去重) → 并发抓文章。""" + homepages = source.all_homepages() + logger.info( + "===== 开始抓取源: {} ({}) {} 个入口 =====", + source.id, source.name, len(homepages), + ) + + all_list_results: list[CrawlResult] = [] + all_links: list[ArticleLink] = [] + + # 增量:加载历史 URL hash + from .storage import load_seen_urls + from .storage import url_hash as storage_url_hash + seen_hashes = load_seen_urls(source.id) + if seen_hashes: + logger.info("源 {} 增量模式: 已记录 {} 条历史 URL", source.id, len(seen_hashes)) + + # 遍历每个入口 + for hp_url in homepages: + list_result = await crawl_url_with_retry( + crawler=crawler, url=hp_url, source=source, + stage=CrawlStage.LIST, settings=settings, semaphore=semaphore, + ) + if save: + save_result(list_result, settings.output_root) + all_list_results.append(list_result) + + if not list_result.success: + logger.warning("源 {} 入口 {} 抓取失败,跳过: {}", source.id, hp_url, list_result.error) + continue + + parsed_base = urlparse(hp_url) + base = f"{parsed_base.scheme}://{parsed_base.netloc}" + links = extract_article_links(list_result.html, base, source) + + # 去重 + 增量:按 hash 去重(当前运行跨入口 + 历史记录) + new_links = [ln for ln in links if storage_url_hash(ln.url) not in seen_hashes] + for ln in new_links: + seen_hashes.add(storage_url_hash(ln.url)) + all_links.extend(new_links) + # 旧链接数(跨入口去重前) + old_count = len(links) - len(new_links) + logger.info("源 {} 入口 {} 抽取链接 {} 条(去重后新增 {}, 跳过旧 {} 条)", + source.id, hp_url, len(links), len(new_links), old_count) + + if not all_links: + logger.warning("源 {} 所有入口未发现任何文章链接", source.id) + return all_list_results + + # 连续超时保护:同源连续超时 2 次后跳过剩余请求 + guard = _TimeoutGuard(source.id) + + article_tasks = [ + crawl_url_with_retry( + crawler=crawler, url=link.url, source=source, + stage=CrawlStage.ARTICLE, settings=settings, semaphore=semaphore, + guard=guard, + ) + for link in all_links + ] + article_results = await asyncio.gather(*article_tasks, return_exceptions=False) + + if save: + for r in article_results: + save_result(r, settings.output_root) + + success_cnt = sum(1 for r in article_results if r.success) + # 记录成功抓取的 URL(增量去重) + new_saved = 0 + for r in article_results: + if r.success: + mark_url_seen(source.id, r.url) + new_saved += 1 + + logger.info( + "源 {} 完成: 文章 {}/{} 成功, 新增 {} 条已记录", + source.id, success_cnt, len(article_results), new_saved, + ) + return [*all_list_results, *article_results] + + +async def crawl_all( + config: CrawlerConfig, + save: bool = True, +) -> list[CrawlResult]: + """抓取所有启用的源,返回扁平的 CrawlResult 列表。""" + settings = config.settings + enabled = config.enabled_sources() + if not enabled: + logger.warning("没有启用的新闻源,直接返回") + return [] + + browser_config = BrowserConfig( + headless=settings.headless, + user_agent=settings.user_agent, + verbose=False, + ) + semaphore = asyncio.Semaphore(settings.concurrency) + all_results: list[CrawlResult] = [] + + async with AsyncWebCrawler(config=browser_config) as crawler: + for source in enabled: + try: + results = await crawl_source(crawler, source, settings, semaphore, save=save) + all_results.extend(results) + except Exception as e: # 单源异常不应中断整体抓取 + logger.exception("源 {} 抓取异常: {}", source.id, e) + + total = len(all_results) + succ = sum(1 for r in all_results if r.success) + logger.info("全部完成: {}/{} 成功 (成功率 {:.0%})", succ, total, succ / max(total, 1)) + return all_results diff --git a/crawler/models.py b/crawler/models.py new file mode 100644 index 0000000..b80ecdc --- /dev/null +++ b/crawler/models.py @@ -0,0 +1,161 @@ +"""新闻抓取模块的数据模型与配置模型。 + +所有核心对象统一使用 Pydantic 定义(CLAUDE.md 第十条要求)。 +""" + +from __future__ import annotations + +from datetime import datetime +from enum import StrEnum + +from pydantic import BaseModel, Field, HttpUrl, field_validator + + +class CrawlStage(StrEnum): + """抓取阶段枚举,区分列表页和文章页。""" + + LIST = "list" # 列表/首页(用于发现文章链接) + ARTICLE = "article" # 文章详情页 + + +class SourceConfig(BaseModel): + """单个新闻源的配置。 + + 映射 configs/sources.yaml 中 sources 列表的每一项。 + """ + + id: str = Field(..., description="新闻源短 ID,用作目录名,仅小写字母与下划线") + name: str = Field(..., description="中文名,用于日志展示") + enabled: bool = Field(default=True, description="是否启用") + homepage: HttpUrl = Field(..., description="新闻列表页/首页 URL") + extra_homepages: list[HttpUrl] = Field( + default_factory=list, + description="额外的列表页 URL(同站多频道时使用),共用同一个 article_url_pattern", + ) + + def all_homepages(self) -> list[str]: + """返回所有入口 URL(主入口 + 额外入口)。""" + urls = [str(self.homepage)] + urls.extend(str(u) for u in self.extra_homepages) + return urls + + # 链接发现规则 + article_url_pattern: str = Field( + ..., + description="正则表达式,从首页所有链接中筛选出文章 URL", + ) + article_link_selector: str | None = Field( + default=None, + description="可选 CSS 选择器,优先在选择器内查找文章链接", + ) + + # JS 渲染相关 + js_render: bool = Field( + default=True, + description="是否需要 JS 渲染(财联社/东方财富等动态站点必须为 true)", + ) + wait_for: str | None = Field( + default=None, + description="可选 CSS,等待该元素出现后再抓取(JS 渲染时使用)", + ) + page_timeout_ms: int = Field(default=30_000, ge=1_000, description="页面加载超时(毫秒)") + + # GNE 提取提示(可选) + body_xpath: str | None = Field( + default=None, + description="GNE body_xpath 参数,强制指定正文容器 XPath", + ) + + # 抓取数量限制 + max_articles_per_run: int = Field( + default=20, + ge=1, + le=200, + description="单次运行最多抓取的文章数", + ) + + @field_validator("id") + @classmethod + def _validate_id(cls, v: str) -> str: + if not v.replace("_", "").isalnum() or not v.islower(): + raise ValueError("source.id 必须为小写字母/数字/下划线") + return v + + +class CrawlerSettings(BaseModel): + """全局抓取设置。""" + + concurrency: int = Field(default=3, ge=1, le=20, description="并发抓取上限") + retry_max_attempts: int = Field(default=3, ge=1, le=10, description="单 URL 最大重试次数") + retry_min_wait_sec: float = Field(default=1.0, ge=0.0, description="重试最小等待秒") + retry_max_wait_sec: float = Field(default=10.0, ge=0.0, description="重试最大等待秒") + user_agent: str = Field( + default=( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/124.0 Safari/537.36" + ), + description="浏览器 User-Agent", + ) + headless: bool = Field(default=True, description="浏览器是否无头") + output_root: str = Field(default="data/raw", description="抓取结果根目录") + + +class CrawlerConfig(BaseModel): + """整个抓取系统的配置(对应 sources.yaml 顶层)。""" + + settings: CrawlerSettings = Field(default_factory=CrawlerSettings) + sources: list[SourceConfig] + + def enabled_sources(self) -> list[SourceConfig]: + """返回启用的源,按 id 排序。""" + return sorted([s for s in self.sources if s.enabled], key=lambda s: s.id) + + +class CninfoItem(BaseModel): + """cninfo 公告/调研/互动易数据条目。 + + 一条记录对应一条公告、一次调研活动或一组互动问答。 + """ + + stock_code: str = Field(..., description="股票代码,6 位数字") + stock_name: str = Field(..., description="公司简称") + title: str = Field(..., description="标题") + content: str = Field(default="", description="正文/摘要内容") + publish_time: str = Field(default="", description="发布时间,格式 YYYY-MM-DD") + item_type: str = Field( + default="announcement", + description="数据类型: announcement(公告)|research(调研)|irm(互动易)", + ) + url: str = Field(default="", description="详情页 URL 或 PDF 链接") + ann_id: str = Field(default="", description="公告唯一 ID,用于增量去重") + extra: dict = Field(default_factory=dict, description="附加字段(公告类型等)") + + +class ArticleLink(BaseModel): + """从列表页发现的文章链接。""" + + source_id: str + url: str + anchor_text: str | None = None + + +class CrawlResult(BaseModel): + """单次抓取结果(列表页或文章页通用)。""" + + source_id: str + stage: CrawlStage + url: str + success: bool + status_code: int | None = None + title: str | None = None + html: str = "" + markdown: str = "" + error: str | None = None + fetched_at: datetime = Field(default_factory=datetime.now) + attempts: int = 1 + + def short_summary(self) -> str: + """生成单行摘要,便于日志输出。""" + flag = "OK" if self.success else "FAIL" + size = len(self.html) + return f"[{flag}] {self.source_id} {self.stage.value} {self.url} html={size}B" diff --git a/crawler/storage.py b/crawler/storage.py new file mode 100644 index 0000000..e10ea7a --- /dev/null +++ b/crawler/storage.py @@ -0,0 +1,98 @@ +"""抓取结果本地保存。 + +目录结构: + {output_root}/{source_id}/{YYYYMMDD}/ + ├── {url_hash}.html # 原始 HTML + ├── {url_hash}.md # Crawl4AI 输出的 Markdown + └── index.jsonl # 元数据(每行一条 CrawlResult) +""" + +from __future__ import annotations + +import hashlib +import json +from datetime import date +from pathlib import Path + +from loguru import logger + +from .models import CrawlResult + + +def url_hash(url: str) -> str: + """对 URL 取 SHA1,取前 16 位作为文件名。""" + return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16] + + +def build_output_dir(output_root: str | Path, source_id: str, day: date | None = None) -> Path: + """构造保存目录: {output_root}/{source_id}/{YYYYMMDD}/。""" + target_day = day or date.today() + return Path(output_root) / source_id / target_day.strftime("%Y%m%d") + + +def save_result( + result: CrawlResult, + output_root: str | Path, + day: date | None = None, + *, + skip_existing: bool = True, +) -> Path | None: + """保存单条抓取结果到本地。 + + 返回 HTML 文件路径;失败结果只追加 index.jsonl,不写 HTML/MD,返回 None。 + + skip_existing=True 时:如果当日该 URL 已保存(HTML 文件已存在),跳过写入, + 仅记录日志,返回已存在路径。保证增量抓取不会重复保存。 + """ + out_dir = build_output_dir(output_root, result.source_id, day) + out_dir.mkdir(parents=True, exist_ok=True) + + h = url_hash(result.url) + html_path: Path | None = None + + # 增量去重:如果当日已抓取过同一 URL,跳过 + existing_html = out_dir / f"{h}.html" + if skip_existing and existing_html.exists(): + logger.debug("跳过重复 URL (今日已抓取): {} -> {}", result.url, h) + return existing_html + + if result.success and result.html: + html_path = out_dir / f"{h}.html" + html_path.write_text(result.html, encoding="utf-8") + + if result.markdown: + md_path = out_dir / f"{h}.md" + md_path.write_text(result.markdown, encoding="utf-8") + + # 元数据(不含 html/markdown 大字段,避免 jsonl 巨大) + meta = result.model_dump(exclude={"html", "markdown"}, mode="json") + meta["url_hash"] = h + meta["html_file"] = html_path.name if html_path else None + + index_path = out_dir / "index.jsonl" + with index_path.open("a", encoding="utf-8") as f: + f.write(json.dumps(meta, ensure_ascii=False) + "\n") + + logger.debug("已保存抓取结果: {}", result.short_summary()) + return html_path + + +def load_seen_urls(source_id: str, output_root: str | Path = "data/raw") -> set[str]: + """加载某个源已抓取的所有 URL hash(跨日期累积)。 + + 读取 {output_root}/{source_id}/seen_urls.txt,每行一个 url_hash。 + """ + seen_path = Path(output_root) / source_id / "seen_urls.txt" + if not seen_path.is_file(): + return set() + with seen_path.open("r", encoding="utf-8") as f: + return {line.strip() for line in f if line.strip()} + + +def mark_url_seen(source_id: str, url: str, output_root: str | Path = "data/raw") -> None: + """追加一个已抓取 URL 的 hash 到 seen_urls.txt。""" + seen_path = Path(output_root) / source_id / "seen_urls.txt" + seen_path.parent.mkdir(parents=True, exist_ok=True) + h = url_hash(url) + with seen_path.open("a", encoding="utf-8") as f: + f.write(h + "\n") diff --git a/dedup/__init__.py b/dedup/__init__.py new file mode 100644 index 0000000..824fade --- /dev/null +++ b/dedup/__init__.py @@ -0,0 +1,44 @@ +"""三层新闻去重模块 (M3)。 + +公共 API: + - Deduper: 主类(check / ingest / stats) + - FingerprintStore: SQLite 指纹库(底层,通常无需直接用) + - DedupResult / DedupLayer / DedupStats / Fingerprint: 数据模型 + - simhash64 / hamming / content_hash / normalize_content: 指纹算法 +""" + +from .deduper import ( + DEFAULT_TIME_WINDOW_DAYS, + Deduper, + article_to_fingerprint, +) +from .hasher import ( + DEFAULT_HAMMING_THRESHOLD, + NGRAM_SIZE, + SIMHASH_BITS, + content_hash, + hamming, + normalize_content, + simhash64, +) +from .models import DedupLayer, DedupResult, DedupStats, Fingerprint +from .store import DEFAULT_DB_PATH, FingerprintStore + +__all__ = [ + "DEFAULT_DB_PATH", + "DEFAULT_HAMMING_THRESHOLD", + "DEFAULT_TIME_WINDOW_DAYS", + "NGRAM_SIZE", + "SIMHASH_BITS", + "DedupLayer", + "DedupResult", + "DedupStats", + "Deduper", + "Fingerprint", + "FingerprintStore", + "article_to_fingerprint", + "content_hash", + "hamming", + "normalize_content", + "simhash64", +] diff --git a/dedup/deduper.py b/dedup/deduper.py new file mode 100644 index 0000000..1dfaf59 --- /dev/null +++ b/dedup/deduper.py @@ -0,0 +1,183 @@ +"""三层去重主流程。 + +调用顺序:check / ingest 内部按 L1 -> L2 -> L3 顺序判定,任意层命中即返回。 + +Deduper 不要求线程安全;批处理串行调用即可。 +""" + +from __future__ import annotations + +from datetime import datetime +from pathlib import Path + +from loguru import logger + +from extractor import Article + +from .hasher import ( + DEFAULT_HAMMING_THRESHOLD, + content_hash, + hamming, + simhash64, +) +from .models import DedupLayer, DedupResult, DedupStats, Fingerprint +from .store import DEFAULT_DB_PATH, FingerprintStore + +# 默认时间窗口(±N 天) +DEFAULT_TIME_WINDOW_DAYS = 30 + + +def _publish_date(article: Article) -> str | None: + """从 Article.publish_time 取 YYYY-MM-DD 字符串。""" + if article.publish_time is None: + return None + return article.publish_time.strftime("%Y-%m-%d") + + +def article_to_fingerprint(article: Article) -> Fingerprint: + """构造 Fingerprint(用于 ingest 写入或对外只读)。 + + source_ids 初始为 [article.source_id],后续重复文章命中时由 ingest 合并。 + """ + return Fingerprint( + url_hash=article.url_hash, + content_hash=content_hash(article.content), + simhash=simhash64(article.content), + source_id=article.source_id, + url=article.url, + title=article.title, + publish_date=_publish_date(article), + ingested_at=datetime.now(), + source_ids=[article.source_id], + ) + + +class Deduper: + """三层去重器。 + + 构造完毕后: + - check(article) 仅判断,不写入; + - ingest(article) 判断,不重复则写入指纹库,返回结果。 + """ + + def __init__( + self, + db_path: str | Path = DEFAULT_DB_PATH, + simhash_threshold: int = DEFAULT_HAMMING_THRESHOLD, + time_window_days: int = DEFAULT_TIME_WINDOW_DAYS, + ) -> None: + self.store = FingerprintStore(db_path) + self.simhash_threshold = simhash_threshold + self.time_window_days = time_window_days + + def close(self) -> None: + self.store.close() + + def __enter__(self) -> Deduper: + return self + + def __exit__(self, *_: object) -> None: + self.close() + + # ------------------------------------------------------------------ # + # 公共 API + # ------------------------------------------------------------------ # + + def check(self, article: Article) -> DedupResult: + """三层判重(只读)。命中时附带匹配指纹的多源信息(all_source_ids)。""" + fp = article_to_fingerprint(article) + + # L1: URL hash + existing = self.store.get_by_url_hash(fp.url_hash) + if existing is not None: + return DedupResult( + url_hash=fp.url_hash, + is_duplicate=True, + matched_layer=DedupLayer.URL, + matched_url_hash=existing.url_hash, + matched_url=existing.url, + matched_title=existing.title, + matched_source_id=existing.source_id, + all_source_ids=existing.source_ids, + ) + + # L2: 内容 hash + existing = self.store.find_by_content_hash(fp.content_hash) + if existing is not None: + return DedupResult( + url_hash=fp.url_hash, + is_duplicate=True, + matched_layer=DedupLayer.CONTENT, + matched_url_hash=existing.url_hash, + matched_url=existing.url, + matched_title=existing.title, + matched_source_id=existing.source_id, + all_source_ids=existing.source_ids, + ) + + # L3: SimHash 模糊 + candidates = self.store.candidates_for_simhash( + fp.publish_date, self.time_window_days + ) + best_dist: int | None = None + best_match: Fingerprint | None = None + for c in candidates: + d = hamming(fp.simhash, c.simhash) + if d <= self.simhash_threshold and (best_dist is None or d < best_dist): + best_dist = d + best_match = c + if d == 0: # 不可能更近,提前结束 + break + + if best_match is not None: + return DedupResult( + url_hash=fp.url_hash, + is_duplicate=True, + matched_layer=DedupLayer.SIMHASH, + matched_url_hash=best_match.url_hash, + matched_url=best_match.url, + matched_title=best_match.title, + matched_source_id=best_match.source_id, + all_source_ids=best_match.source_ids, + hamming_distance=best_dist, + ) + + return DedupResult(url_hash=fp.url_hash, is_duplicate=False) + + def ingest(self, article: Article) -> DedupResult: + """判重 + 不重复则入库。 + + 命中重复时,把当前文章的 source_id 合并进匹配指纹的 source_ids + (记录同一内容组的全部来源),并更新 all_source_ids 后返回。 + """ + result = self.check(article) + if not result.is_duplicate: + fp = article_to_fingerprint(article) + self.store.upsert(fp) + logger.debug("入库: {} {}", fp.url_hash, fp.title[:30]) + return result + + # 重复:合并来源到匹配指纹(主源保持首位,Fingerprint validator 负责去重) + if result.matched_url_hash and article.source_id not in result.all_source_ids: + matched = self.store.get_by_url_hash(result.matched_url_hash) + if matched is not None: + merged = [*matched.source_ids, article.source_id] + self.store.upsert(matched.model_copy(update={"source_ids": merged})) + result = result.model_copy( + update={"all_source_ids": merged} + ) + logger.debug( + "合并来源 {} -> {} ({} 个源)", + article.source_id, result.matched_url_hash, len(merged), + ) + logger.debug("命中重复: {}", result.short_summary()) + return result + + def stats(self) -> DedupStats: + lo, hi = self.store.date_range() + return DedupStats( + total=self.store.count(), + by_source=self.store.count_by_source(), + earliest=lo, + latest=hi, + ) diff --git a/dedup/hasher.py b/dedup/hasher.py new file mode 100644 index 0000000..24e1db2 --- /dev/null +++ b/dedup/hasher.py @@ -0,0 +1,93 @@ +"""三层去重的指纹算法。 + +核心: + - normalize_content:把 content 折叠成纯净文本,用于跨源比对; + - content_hash:normalize 后 SHA1[:16]; + - simhash64:字符 3-gram + md5 加权累加,产出 64 位无符号整数; + - hamming:两个 SimHash 的汉明距离。 + +设计取舍: + SimHash 的"分词"用字符 3-gram 而非 jieba。理由: + 1. 中文场景下字符 3-gram 与词级 SimHash 在重复识别上效果接近, + 而前者无外部依赖、ARM/嵌入式友好; + 2. M5 Embedding 后续不依赖 jieba,引入只为 M3 不划算; + 3. 重复率验收门槛 ≤ 5%(project_plan.md 第七章),3-gram 经验上 + 足以分辨。 +""" + +from __future__ import annotations + +import hashlib +import unicodedata + +# 64 位 SimHash 位宽 +SIMHASH_BITS = 64 +SIMHASH_MASK = (1 << SIMHASH_BITS) - 1 + +# 默认 SimHash 汉明距离阈值(<= 此值视为重复) +DEFAULT_HAMMING_THRESHOLD = 3 + +# 字符 n-gram 长度 +NGRAM_SIZE = 3 + + +def normalize_content(text: str) -> str: + """把 content 折叠成"无空白无标点"形式,用于 L2 内容 hash 与 SimHash 输入。 + + 使用 Unicode 类别判断: + - P* Punctuation(所有中英文标点) + - Z* Separator(空格 / 行 / 段分隔符) + - C* Control(NUL / 换行控制等) + 保留 L*(字母)、N*(数字)、S*(符号,如 +/-、% 等),以及 CJK 字符。 + """ + if not text: + return "" + return "".join( + ch for ch in text if unicodedata.category(ch)[0] not in ("P", "Z", "C") + ) + + +def content_hash(text: str) -> str: + """对 normalize_content(text) 做 SHA1,取前 16 hex 字符。""" + norm = normalize_content(text) + return hashlib.sha1(norm.encode("utf-8")).hexdigest()[:16] + + +def _ngrams(text: str, n: int = NGRAM_SIZE) -> list[str]: + """字符级 n-gram。文本短于 n 时,直接整体作为单个 token。""" + if len(text) < n: + return [text] if text else [] + return [text[i : i + n] for i in range(len(text) - n + 1)] + + +def simhash64(text: str) -> int: + """64 位 SimHash。返回无符号整数,空文本返回 0。""" + norm = normalize_content(text) + if not norm: + return 0 + + grams = _ngrams(norm) + if not grams: + return 0 + + v = [0] * SIMHASH_BITS + for gram in grams: + h = int(hashlib.md5(gram.encode("utf-8"), usedforsecurity=False).hexdigest(), 16) + # 取低 64 位 + h64 = h & SIMHASH_MASK + for i in range(SIMHASH_BITS): + if (h64 >> i) & 1: + v[i] += 1 + else: + v[i] -= 1 + + fp = 0 + for i in range(SIMHASH_BITS): + if v[i] > 0: + fp |= 1 << i + return fp + + +def hamming(a: int, b: int) -> int: + """两个 SimHash 的汉明距离。""" + return bin((a ^ b) & SIMHASH_MASK).count("1") diff --git a/dedup/models.py b/dedup/models.py new file mode 100644 index 0000000..99dc1b7 --- /dev/null +++ b/dedup/models.py @@ -0,0 +1,90 @@ +"""三层去重模块的数据模型。""" + +from __future__ import annotations + +from datetime import datetime +from enum import StrEnum +from typing import Literal, Self + +from pydantic import BaseModel, Field, model_validator + + +class DedupLayer(StrEnum): + """命中去重的层。""" + + URL = "url" # L1: 完全相同 URL + CONTENT = "content" # L2: 标准化后 content 完全一致 + SIMHASH = "simhash" # L3: SimHash 汉明距离 <= 阈值 + + +class Fingerprint(BaseModel): + """单篇文章的指纹记录,持久化到 SQLite。 + + source_ids: 同一内容组(去重后视为同一篇新闻)的全部来源列表, + 第一位是主源(即本指纹的 source_id);重复文章命中时由 + Deduper.ingest 自动合并,实现「一条唯一新闻记录多个源」。 + """ + + url_hash: str = Field(..., description="主键,与 Article.url_hash 一致") + content_hash: str = Field(..., description="标准化 content 的 SHA1[:16]") + simhash: int = Field(..., description="64 位 SimHash 整数(无符号)") + source_id: str + url: str + title: str + publish_date: str | None = Field(default=None, description="YYYY-MM-DD,用于时间窗口") + ingested_at: datetime = Field(default_factory=datetime.now) + source_ids: list[str] = Field( + default_factory=list, + description="同内容组全部来源(去重合并),始终包含 source_id 且其居首", + ) + + @model_validator(mode="after") + def _ensure_source_ids(self) -> Self: + """保证 source_ids 非空、去重且以主源 source_id 开头。""" + seen: list[str] = [] + for s in [self.source_id, *self.source_ids]: + if s and s not in seen: + seen.append(s) + self.source_ids = seen + return self + + +class DedupResult(BaseModel): + """对单篇文章的判重结果。""" + + url_hash: str + is_duplicate: bool + matched_layer: DedupLayer | None = None + matched_url_hash: str | None = None + matched_url: str | None = None + matched_title: str | None = None + matched_source_id: str | None = Field( + default=None, description="匹配指纹的主源 source_id" + ) + all_source_ids: list[str] = Field( + default_factory=list, + description="该内容组(唯一新闻)的全部来源;含匹配指纹自身的来源", + ) + hamming_distance: int | None = Field( + default=None, description="仅 SimHash 层有值" + ) + + def short_summary(self) -> str: + if not self.is_duplicate: + return f"[UNIQUE] {self.url_hash}" + layer = self.matched_layer.value if self.matched_layer else "?" + extra = f" hd={self.hamming_distance}" if self.hamming_distance is not None else "" + return f"[DUP/{layer}] {self.url_hash} ~ {self.matched_url_hash}{extra}" + + +class DedupStats(BaseModel): + """指纹库统计。""" + + total: int = 0 + by_source: dict[str, int] = Field(default_factory=dict) + earliest: str | None = None + latest: str | None = None + + +# 类型别名,便于在批处理日志中归类 +DedupVerdict = Literal["unique", "duplicate"] diff --git a/dedup/store.py b/dedup/store.py new file mode 100644 index 0000000..f6edd51 --- /dev/null +++ b/dedup/store.py @@ -0,0 +1,207 @@ +"""SQLite 指纹存储。 + +注意:SimHash 是 64 位无符号整数,SQLite INTEGER 是 64 位有符号 +(范围 [-2^63, 2^63-1])。直接存可能溢出/转负数,虽然 XOR 仍然 +正确但语义混乱。这里统一存为 16 位 hex TEXT,避免符号问题。 + +source_ids 列存 JSON 数组文本(同一内容组全部来源);旧库无此列时 +自动 ALTER TABLE 迁移,旧数据读取时回退为 [source_id]。 +""" + +from __future__ import annotations + +import json +import sqlite3 +from datetime import datetime, timedelta +from pathlib import Path +from typing import Any + +from loguru import logger + +from .models import Fingerprint + +DEFAULT_DB_PATH = Path("data/dedup/fingerprints.sqlite3") + +_SCHEMA_SQL = """ +CREATE TABLE IF NOT EXISTS fingerprints ( + url_hash TEXT PRIMARY KEY, + content_hash TEXT NOT NULL, + simhash_hex TEXT NOT NULL, + source_id TEXT NOT NULL, + url TEXT NOT NULL, + title TEXT NOT NULL, + publish_date TEXT, + ingested_at TEXT NOT NULL, + source_ids TEXT +); +CREATE INDEX IF NOT EXISTS idx_content_hash ON fingerprints(content_hash); +CREATE INDEX IF NOT EXISTS idx_publish_date ON fingerprints(publish_date); +CREATE INDEX IF NOT EXISTS idx_source_id ON fingerprints(source_id); +""" + +# 兼容旧库:为已存在但缺少 source_ids 列的表补列 +_ALTER_SQL = "ALTER TABLE fingerprints ADD COLUMN source_ids TEXT" + + +def _to_hex(simhash: int) -> str: + return f"{simhash:016x}" + + +def _from_hex(hex_str: str) -> int: + return int(hex_str, 16) + + +def _to_sources_json(source_ids: list[str]) -> str: + return json.dumps(source_ids, ensure_ascii=False) + + +def _from_sources_json(raw: str | None, fallback: str) -> list[str]: + """解析 source_ids 列;NULL/损坏时回退 [主源]。""" + if not raw: + return [fallback] + try: + val = json.loads(raw) + except (TypeError, ValueError): + return [fallback] + if isinstance(val, list) and val: + # 保证主源在首位(兼容手改/旧数据) + cleaned = [s for s in val if s and s != fallback] + return [fallback, *cleaned] + return [fallback] + + +def _row_to_fp(row: sqlite3.Row) -> Fingerprint: + return Fingerprint( + url_hash=row["url_hash"], + content_hash=row["content_hash"], + simhash=_from_hex(row["simhash_hex"]), + source_id=row["source_id"], + url=row["url"], + title=row["title"], + publish_date=row["publish_date"], + ingested_at=datetime.fromisoformat(row["ingested_at"]), + source_ids=_from_sources_json(row["source_ids"], row["source_id"]), + ) + + +class FingerprintStore: + """SQLite 包装。线程不安全(每个线程请新建实例)。""" + + def __init__(self, db_path: str | Path = DEFAULT_DB_PATH) -> None: + self.db_path = Path(db_path) + self.db_path.parent.mkdir(parents=True, exist_ok=True) + self._conn: sqlite3.Connection = sqlite3.connect( + self.db_path, isolation_level=None + ) + self._conn.row_factory = sqlite3.Row + self._conn.executescript(_SCHEMA_SQL) + self._migrate_source_ids() + logger.debug("打开指纹库: {}", self.db_path) + + def _migrate_source_ids(self) -> None: + """旧库兼容:为缺少 source_ids 列的表补列(新库无需执行)。""" + try: + self._conn.execute(_ALTER_SQL) + logger.info("指纹库迁移:为 fingerprints 表新增 source_ids 列") + except sqlite3.OperationalError: + logger.debug("source_ids 列已存在,跳过迁移") + + def close(self) -> None: + self._conn.close() + + def __enter__(self) -> FingerprintStore: + return self + + def __exit__(self, *_: Any) -> None: + self.close() + + # ------------------------------------------------------------------ # + # 查询 + # ------------------------------------------------------------------ # + + def get_by_url_hash(self, url_hash: str) -> Fingerprint | None: + row = self._conn.execute( + "SELECT * FROM fingerprints WHERE url_hash = ?", (url_hash,) + ).fetchone() + return _row_to_fp(row) if row else None + + def find_by_content_hash(self, content_hash: str) -> Fingerprint | None: + """返回任一匹配项。""" + row = self._conn.execute( + "SELECT * FROM fingerprints WHERE content_hash = ? LIMIT 1", + (content_hash,), + ).fetchone() + return _row_to_fp(row) if row else None + + def candidates_for_simhash( + self, + publish_date: str | None, + window_days: int, + ) -> list[Fingerprint]: + """返回 publish_date ± window_days 内的指纹候选。 + + publish_date 为 None 时,不限定窗口(返回全部,慎用)。 + """ + if publish_date is None or window_days < 0: + rows = self._conn.execute("SELECT * FROM fingerprints").fetchall() + return [_row_to_fp(r) for r in rows] + + try: + center = datetime.strptime(publish_date, "%Y-%m-%d") + except ValueError: + logger.debug("publish_date 不可解析: {!r},退化为全表扫描", publish_date) + rows = self._conn.execute("SELECT * FROM fingerprints").fetchall() + return [_row_to_fp(r) for r in rows] + + lo = (center - timedelta(days=window_days)).strftime("%Y-%m-%d") + hi = (center + timedelta(days=window_days)).strftime("%Y-%m-%d") + rows = self._conn.execute( + "SELECT * FROM fingerprints " + "WHERE publish_date IS NULL OR (publish_date >= ? AND publish_date <= ?)", + (lo, hi), + ).fetchall() + return [_row_to_fp(r) for r in rows] + + def count(self) -> int: + return self._conn.execute("SELECT COUNT(*) FROM fingerprints").fetchone()[0] + + def count_by_source(self) -> dict[str, int]: + rows = self._conn.execute( + "SELECT source_id, COUNT(*) AS n FROM fingerprints GROUP BY source_id" + ).fetchall() + return {r["source_id"]: r["n"] for r in rows} + + def date_range(self) -> tuple[str | None, str | None]: + row = self._conn.execute( + "SELECT MIN(publish_date) AS lo, MAX(publish_date) AS hi FROM fingerprints" + ).fetchone() + return (row["lo"], row["hi"]) if row else (None, None) + + # ------------------------------------------------------------------ # + # 写入 + # ------------------------------------------------------------------ # + + def upsert(self, fp: Fingerprint) -> None: + self._conn.execute( + "INSERT OR REPLACE INTO fingerprints " + "(url_hash, content_hash, simhash_hex, source_id, url, title, " + " publish_date, ingested_at, source_ids) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)", + ( + fp.url_hash, + fp.content_hash, + _to_hex(fp.simhash), + fp.source_id, + fp.url, + fp.title, + fp.publish_date, + fp.ingested_at.isoformat(), + _to_sources_json(fp.source_ids), + ), + ) + + def delete(self, url_hash: str) -> None: + self._conn.execute("DELETE FROM fingerprints WHERE url_hash = ?", (url_hash,)) + + def clear(self) -> None: + """清空指纹库,主要用于测试。""" + self._conn.execute("DELETE FROM fingerprints") diff --git a/deploy/a-share-research.service b/deploy/a-share-research.service new file mode 100644 index 0000000..cec1ec3 --- /dev/null +++ b/deploy/a-share-research.service @@ -0,0 +1,24 @@ +[Unit] +Description=A 股 Deep Research 定时调度 +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=pi +Group=pi +WorkingDirectory=/home/pi/cc-projects/news +ExecStart=/home/pi/cc-projects/news/.venv/bin/python -m scripts.run_scheduler +Restart=always +RestartSec=30 + +# 日志 +StandardOutput=append:/home/pi/cc-projects/news/logs/scheduler-stdout.log +StandardError=append:/home/pi/cc-projects/news/logs/scheduler-stderr.log + +# 环境变量 +Environment="PATH=/home/pi/cc-projects/news/.venv/bin:/usr/local/bin:/usr/bin:/bin" +Environment="TZ=Asia/Shanghai" + +[Install] +WantedBy=multi-user.target diff --git a/deploy/crontab.example b/deploy/crontab.example new file mode 100644 index 0000000..198c201 --- /dev/null +++ b/deploy/crontab.example @@ -0,0 +1,41 @@ +# A 股 Deep Research 定时调度 (crontab) +# 替代 APScheduler 守护进程,更简单可靠 +# +# 安装: crontab deploy/crontab.example +# 查看: crontab -l +# 编辑: crontab -e +# 清空: crontab -r +# +# 日志: logs/scheduler-cron.log + +SHELL=/bin/zsh +PATH=/usr/local/bin:/usr/bin:/bin +TZ=Asia/Shanghai + +# 注意: 所有命令的 --date 使用 "" (cron 运行时自动取当天) + +# ============================================================ +# 每日 4 次定时任务 +# ============================================================ + +# 07:00 — 首次抓取 + 日报 +0 7 * * * cd /Users/summer/Downloads/cc-projects/news && uv run a-share pipeline --once --report --steps crawler,extractor,dedup,llm,embedding,qdrant >> logs/scheduler-cron.log 2>&1 + +# 12:00 — 午间补充抓取(不含日报) +0 12 * * * cd /Users/summer/Downloads/cc-projects/news && uv run a-share pipeline --once --steps crawler,extractor,dedup,llm,embedding,qdrant >> logs/scheduler-cron.log 2>&1 + +# 18:00 — 收盘后抓取(不含日报) +0 18 * * * cd /Users/summer/Downloads/cc-projects/news && uv run a-share pipeline --once --steps crawler,extractor,dedup,llm,embedding,qdrant >> logs/scheduler-cron.log 2>&1 + +# 22:00 — 晚间抓取(不含日报) +0 22 * * * cd /Users/summer/Downloads/cc-projects/news && uv run a-share pipeline --once --steps crawler,extractor,dedup,llm,embedding,qdrant >> logs/scheduler-cron.log 2>&1 + +# ============================================================ +# cninfo 公告管道(每日 06:30,日报前) +# ============================================================ +# 30 6 * * * cd /Users/summer/Downloads/cc-projects/news && uv run a-share pipeline --cninfo-once >> logs/scheduler-cron.log 2>&1 + +# ============================================================ +# 个股日报(每日 07:30,日报之后) +# ============================================================ +# 30 7 * * * cd /Users/summer/Downloads/cc-projects/news && uv run a-share stock-report >> logs/scheduler-cron.log 2>&1 diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..632d1ed --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,40 @@ +# A 股 Deep Research 平台 docker-compose +# 当前仅声明 Qdrant 占位(Milestone 6 启用);其余服务在对应 Milestone 加入 + +services: + # ---- 向量数据库(Milestone 6 启用)---- + qdrant: + image: qdrant/qdrant:v1.13.5 + container_name: a_share_qdrant + restart: unless-stopped + profiles: ["m6", "full"] # 默认不启动;docker compose --profile m6 up -d 启用 + ports: + - "6333:6333" # REST + - "6334:6334" # gRPC + volumes: + - ./data/qdrant_storage:/qdrant/storage + environment: + - QDRANT__SERVICE__GRPC_PORT=6334 + healthcheck: + test: ["CMD-SHELL", "bash -c ' 版本:v2.0 | 最后更新:2026-08-22 + +--- + +## 目录 + +1. [整体架构](#1-整体架构) +2. [数据流全景](#2-数据流全景) +3. [包职责与导出 API](#3-包职责与导出-api) +4. [核心数据模型](#4-核心数据模型) +5. [配置体系](#5-配置体系) +6. [产物目录结构](#6-产物目录结构) +7. [运行环境](#7-运行环境) + +--- + +## 1. 整体架构 + +``` +┌─────────────────────────────────────────────────────────────────┐ +│ 统一 CLI (a_share_cli) │ +│ crawl | extract | dedup | events | embed | ingest | pipeline │ +│ search | status | report | cninfo | watchlist | discover │ +└──────────────────────────┬──────────────────────────────────────┘ + │ + ┌──────────────────────┼──────────────────────────────────┐ + │ │ │ + ▼ ▼ ▼ +┌─────────┐ ┌──────────┐ ┌────────┐ ┌──────────┐ ┌──────────┐ +│ crawler │──▶│extractor │──▶│ dedup │──▶│ llm │──▶│embedding │ +│ (M1) │ │ (M2) │ │ (M3) │ │ (M4) │ │ (M5) │ +└─────────┘ └──────────┘ └────────┘ └──────────┘ └─────┬────┘ + │ + ▼ + ┌──────────┐ + │vectorstore│ + │ (M6) │ + └─────┬────┘ + │ + ┌──────────────────────────────────────┤ + │ │ + ▼ ▼ + ┌──────────┐ ┌──────────┐ + │ mcp_server│ │scheduler │ + │ (M8) │ │ (M7) │ + └──────────┘ └─────┬────┘ + │ + ┌─────────────────┤ + ▼ ▼ + ┌──────────┐ ┌──────────────┐ + │ reporter │ │stock_reporter │ + │ (日报) │ │ (个股日报) │ + └─────┬────┘ └──────────────┘ + │ + ┌──────────────┼──────────────┐ + ▼ ▼ ▼ + ┌──────────┐ ┌──────────┐ ┌──────────┐ + │report_db │ │report_ │ │ MySQL │ + │ (M10) │ │ import │ │(myquant) │ + └──────────┘ └──────────┘ └──────────┘ +``` + +**11 个包**(按管道顺序): + +| 序号 | 包 | 模块 | 概述 | +|------|-----|------|------| +| 1 | `crawler/` | M1 | 新闻抓取:httpx 静态直连 + Playwright JS 渲染 + cninfo 公告 API | +| 2 | `extractor/` | M2 | GNE 中文新闻正文提取,输出 Article | +| 3 | `dedup/` | M3 | 三层去重:URL Hash → Content Hash → SimHash | +| 4 | `llm/` | M4 | DeepSeek/Qwen 投资事件抽取,输出 ExtractedEvent | +| 5 | `embedding/` | M5 | DashScope 远程 / 本地 BGE-M3 向量化 | +| 6 | `vectorstore/` | M6 | Qdrant 本地文件模式,语义检索 | +| 7 | `scheduler/` | M7 | APScheduler 定时任务 + pipeline 编排 + 日报生成 | +| 8 | `mcp_server/` | M8 | MCP 5 工具,供 Cherry Studio/Claude Code 调用 | +| 9 | `a_share_cli/` | CLI | 统一命令行入口(argparse),所有子命令 | +| 10 | `report_db/` | M10 | 日报 MySQL 结构化入库(pymysql) | +| 11 | `report_import/` | M10 | 历史日报 HTML 解析与批量导入 | + +辅助目录: + +| 目录 | 用途 | +|------|------| +| `configs/` | sources.yaml(新闻源)、watchlist.yaml(关注列表)、llm_models.yaml(LLM 场景配置)、loader.py | +| `prompts/` | 4 个 LLM Prompt 模板(event_extraction/company_analysis/industry_analysis/risk_analysis) | +| `scripts/` | 独立运行脚本(每个模块一个入口) + systemd 服务文件 | +| `tests/` | pytest 测试(225+ passed) | +| `api/` | 占位(空包,预留给未来 API 服务) | +| `app/` | 空目录(预留) | + +--- + +## 2. 数据流全景 + +``` +财经网站 / cninfo API + │ + ▼ + ┌───────────┐ + │ M1 抓取 │ → data/raw/{source}/{YYYYMMDD}/*.html + index.jsonl + └─────┬─────┘ + │ + ▼ + ┌───────────┐ + │ M2 提取 │ → data/processed/{source}/{YYYYMMDD}/{url_hash}.json (Article) + └─────┬─────┘ + │ + ▼ + ┌───────────┐ + │ M3 去重 │ → data/deduped/{YYYYMMDD}/uniques/{url_hash}.json + └─────┬─────┘ data/dedup/fingerprints.sqlite3 (指纹库) + │ + ▼ + ┌───────────┐ + │ M4 LLM │ → data/events/{YYYYMMDD}/{url_hash}.json (ExtractedEvent) + └─────┬─────┘ + │ + ▼ + ┌───────────┐ + │ M5 向量化 │ → data/embeddings/{YYYYMMDD}/{url_hash}.json (1024 维向量) + └─────┬─────┘ + │ + ▼ + ┌───────────┐ + │ M6 入库 │ → data/qdrant_storage/ (Qdrant 本地文件) + └─────┬─────┘ + │ + ┌────┴────┐ + ▼ ▼ +┌──────┐ ┌──────┐ +│ MCP │ │ 日报 │ +│ 检索 │ │ 生成 │ +└──────┘ └──┬───┘ + │ + ▼ + ┌────────┐ + │ MySQL │ → news_report / news_event (myquant 库) + └────────┘ +``` + +**关键设计**: + +- **增量处理**:M2/M4/M5 产物存在即跳过(2026-08-12 新增),`--force` 全量重建 +- **断点续跑**:`pipeline --once --resume` 从上次失败步骤继续,状态文件 `data/pipeline/state.json` +- **去重多源记录**:M3 指纹库 `source_ids` 列记录同一篇新闻的多个来源 + +--- + +## 3. 包职责与导出 API + +### 3.1 `crawler/` — M1 新闻抓取 + +| 文件 | 职责 | +|------|------| +| `engine.py` | Crawl4AI 异步引擎:httpx 静态直连(js_render=false) / Playwright(js_render=true) | +| `config.py` | 加载 `configs/sources.yaml` | +| `models.py` | SourceConfig / CrawlerConfig / CrawlResult / ArticleLink / CninfoItem / CrawlStage | +| `storage.py` | 保存 raw HTML + index.jsonl,URL 去重 | +| `cninfo.py` | cninfo 公告/调研/互动易 API 抓取 | + +**公开 API**: + +```python +from crawler import ( + crawl_all, crawl_source, extract_article_links, + load_crawler_config, + SourceConfig, CrawlerConfig, CrawlResult, CrawlStage, ArticleLink, CninfoItem, +) +``` + +**抓取分流规则**: + +| 条件 | 引擎 | 特点 | +|------|------|------| +| `js_render=False` 且无 `wait_for` | httpx 直连 | 快,不触发反爬 | +| `js_render=True` 或有 `wait_for` | Playwright | 支持 JS 渲染 | + +**新闻源**:14 个(13 Web + 1 API:新闻联播),见 `configs/sources.yaml` + +### 3.2 `extractor/` — M2 正文提取 + +| 文件 | 职责 | +|------|------| +| `parser.py` | GNE 中文正文提取,模板文本过滤 | +| `models.py` | Article / ExtractError | + +**公开 API**: + +```python +from extractor import ( + extract_article, Article, ExtractError, MIN_CONTENT_LENGTH, SOURCE_NAME_MAP, +) +``` + +**Article 模型**(M2 输出,M3+ 输入的唯一格式): + +| 字段 | 类型 | 说明 | +|------|------|------| +| `source_id` | str | M1 源 ID | +| `url` | str | 文章 URL | +| `url_hash` | str | SHA1 前 16 位,主键 | +| `title` | str | 标题 | +| `content` | str | 清理后纯文本正文 | +| `author` | str\|None | 作者 | +| `publish_time` | datetime\|None | 标准化发布时间 | +| `word_count` | int | 中文字符数 | +| `item_type` | str\|None | cninfo 类型:announcement/research/irm | + +### 3.3 `dedup/` — M3 三层去重 + +| 文件 | 职责 | +|------|------| +| `deduper.py` | Deduper 主类:check / ingest / stats | +| `hasher.py` | SimHash 64 位 + 汉明距离 + content_hash | +| `store.py` | FingerprintStore,SQLite 持久化 | +| `models.py` | DedupLayer / DedupResult / DedupStats / Fingerprint | + +**公开 API**: + +```python +from dedup import ( + Deduper, article_to_fingerprint, FingerprintStore, + DedupLayer, DedupResult, DedupStats, Fingerprint, + simhash64, hamming, content_hash, normalize_content, + DEFAULT_HAMMING_THRESHOLD, DEFAULT_TIME_WINDOW_DAYS, +) +``` + +**三层去重逻辑**: + +| 层 | 算法 | 命中条件 | +|----|------|----------| +| L1 URL | URL Hash (SHA1) | 完全相同 URL | +| L2 Content | 标准化正文 SHA1 | 正文完全一致 | +| L3 SimHash | 64 位 SimHash + 汉明距离 | 距离 ≤ 3 (默认) | + +### 3.4 `llm/` — M4 投资事件抽取 + +| 文件 | 职责 | +|------|------| +| `client.py` | OpenAI 兼容客户端,指数退避重试,LLMConfig 加载 | +| `extractor.py` | Prompt 模板加载,JSON 解析,extract_event / extract_event_async | +| `models.py` | EventExtraction / ExtractedEvent / Sentiment / LLMCallError / EVENT_TYPES | + +**公开 API**: + +```python +from llm import ( + extract_event, extract_event_async, parse_event_json, + PromptTemplate, load_llm_config, LLMConfig, + make_sync_client, make_async_client, + EventExtraction, ExtractedEvent, Sentiment, LLMCallError, + EVENT_TYPES, MAX_CONTENT_CHARS, MAX_IMPORTANCE, MIN_IMPORTANCE, + SCENE_EVENT_EXTRACTION, SCENE_DAILY_REPORT, SCENE_STOCK_REPORT, +) +``` + +**ExtractedEvent 模型**(M4 落盘格式): + +| 字段 | 类型 | 说明 | +|------|------|------| +| `source_id` | str | 主源 | +| `url` | str | 原文 URL | +| `url_hash` | str | SHA1 前 16 位 | +| `title` | str | 标题 | +| `publish_time` | datetime\|None | 发布时间 | +| `sources` | list[str] | 全部来源(去重合并) | +| `event` | EventExtraction | LLM 抽取结果 | +| `provider` | str | deepseek / qwen | +| `model` | str | 模型名 | +| `attempts` | int | LLM 实际调用次数(含重试) | + +**EventExtraction**(LLM 输出 JSON 结构): + +| 字段 | 类型 | 说明 | +|------|------|------| +| `stock_codes` | list[str] | 6 位代码,可带 .SH/.SZ/.BJ | +| `company_names` | list[str] | 公司中文简称 | +| `industries` | list[str] | 行业(申万二级) | +| `sentiment` | Sentiment | positive/neutral/negative | +| `importance` | int | 1-5 重要程度 | +| `event_type` | str | 23 种事件类型之一 | +| `summary` | str | 一句话摘要(≤200 字) | + +**23 种事件类型**:业绩预告/业绩快报/财报披露/合作签约/投资并购/重大合同/产品发布/技术突破/监管处罚/诉讼仲裁/股东减持/股东增持/回购/分红/高管变动/资产重组/停牌复牌/ST警示/退市风险/宏观政策/行业政策/国际局势/其他 + +**LLM 场景配置**(`configs/llm_models.yaml`): + +| 场景 | 用途 | 调用方 | +|------|------|--------| +| `event_extraction` | 投资事件抽取(M4) | llm/extractor.py | +| `daily_report` | 日报 AI 摘要 | scheduler/reporter.py | +| `stock_report` | 个股 AI 要点 | scheduler/stock_reporter.py | +| `embedding` | 文本向量化(M5) | embedding/factory.py | + +### 3.5 `embedding/` — M5 向量化 + +| 文件 | 职责 | +|------|------| +| `base.py` | EmbeddingProvider / AsyncEmbeddingProvider ABC + compose_text | +| `remote.py` | DashScope 远程 Embedding(OpenAI 兼容) | +| `local.py` | 本地 BGE-M3(可选,需 `uv sync --extra local-embedding`) | +| `factory.py` | resolve_provider_type / make_sync_provider / make_async_provider | +| `models.py` | EmbeddingProviderType / EmbeddingResult / EmbeddingError | + +**公开 API**: + +```python +from embedding import ( + make_sync_provider, make_async_provider, resolve_provider_type, + compose_text, EmbeddingProvider, AsyncEmbeddingProvider, + EmbeddingResult, EmbeddingError, EmbeddingProviderType, + DashScopeEmbeddingProvider, DashScopeAsyncEmbeddingProvider, + DASHSCOPE_DEFAULT_DIM, DASHSCOPE_DEFAULT_MODEL, DASHSCOPE_BATCH_LIMIT, + MAX_TEXT_CHARS, +) +``` + +**向量维度**:1024(DashScope `text-embedding-v3` / 本地 BGE-M3) + +**文本组装**:`compose_text()` 统一策略——标题 + 正文 + 事件摘要,最大 8000 字符 + +### 3.6 `vectorstore/` — M6 Qdrant 知识库 + +| 文件 | 职责 | +|------|------| +| `client.py` | VectorStore 封装(init/upsert/query/count/info) + make_qdrant_client 工厂 | +| `models.py` | SearchFilter / SearchResult / CollectionInfo | + +**公开 API**: + +```python +from vectorstore import ( + VectorStore, make_qdrant_client, + SearchFilter, SearchResult, CollectionInfo, + DEFAULT_COLLECTION, DEFAULT_VECTOR_DIM, +) +``` + +**Qdrant 模式**:默认本地文件模式(`data/qdrant_storage/`),零依赖,ARM64 兼容 + +**SearchFilter 支持**:source_id / stock_codes / company_names / industries / sentiment / importance_min / event_types / publish_date_from / publish_date_to + +### 3.7 `scheduler/` — M7 定时任务与 Pipeline + +| 文件 | 职责 | +|------|------| +| `pipeline.py` | run_pipeline / run_step,编排 M1→M6,断点续跑 | +| `reporter.py` | 日报生成(收集→AI 摘要→MySQL 入库) | +| `stock_reporter.py` | 个股日报生成(关注列表) | + +**公开 API**: + +```python +from scheduler import ( + run_pipeline, run_step, PipelineResult, StepResult, + STEP_COMMANDS, STEP_TIMEOUTS, +) +``` + +**Pipeline 步骤**: + +| 步骤名 | 中文 | 默认超时 | 说明 | +|--------|------|----------|------| +| `crawler` | M1 新闻抓取 | 900s | 13 源,含 Playwright | +| `xwlb` | M1 新闻联播 | 60s | 纯 HTTP API | +| `extractor` | M2 正文提取 | 300s | GNE 提取 | +| `dedup` | M3 新闻去重 | 120s | 三层去重 | +| `llm` | M4 LLM 抽取 | 900s | API 调用 | +| `embedding` | M5 向量化 | 300s | DashScope | +| `qdrant` | M6 Qdrant 入库 | 300s | 本地文件写入 | +| `report` | 日报生成 | 30s | 仅 07:00 执行 | +| `cninfo_crawl` | cninfo 公告抓取 | 900s | 06:30 执行 | +| `cninfo_extract` | cninfo 正文提取 | 300s | — | +| `cninfo_pdf` | cninfo PDF 补充 | — | — | + +**超时优先级**:`TIMEOUT_{NAME}` 环境变量 > `PIPELINE_STEP_TIMEOUT` > 硬编码默认值 > 1800s + +**调度时间表**: + +| 时间 | 步骤 | +|------|------| +| 06:30 | cninfo 公告管道 | +| 07:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant→**日报** | +| 07:30 | 个股日报 | +| 12:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | +| 18:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | +| 22:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | + +### 3.8 `mcp_server/` — M8 MCP 服务 + +| 文件 | 职责 | +|------|------| +| `tools.py` | FastMCP 服务器,5 个 MCP 工具 | + +**5 个 MCP 工具**: + +| 工具 | 参数 | 说明 | +|------|------|------| +| `search_news` | query, top_k | 通用语义检索 | +| `search_company_news` | query, company, top_k | 按公司过滤 | +| `search_industry_news` | query, industry, top_k | 按行业过滤 | +| `search_stock_events` | query, stock_code, top_k | 按股票代码过滤 | +| `search_sentiment_trend` | query, sentiment, top_k | 情绪趋势 + 统计 | + +**启动方式**:stdio 模式(Cherry Studio/Claude Code 自动管理进程)或 SSE 模式(调试:`--sse 8765`) + +### 3.9 `a_share_cli/` — 统一 CLI + +| 文件 | 职责 | +|------|------| +| `main.py` | argparse 子命令路由,所有命令行操作 | + +**全部子命令**: + +| 命令 | 功能 | 关键参数 | +|------|------|----------| +| `crawl` | M1 抓取 | `--source`, `--no-save` | +| `extract` | M2 提取 | `--date`, `--source` | +| `dedup` | M3 去重 | `--date`, `--reset` | +| `events` | M4 LLM 抽取 | `--date`, `--provider`, `--model`, `--limit`, `--concurrency` | +| `embed` | M5 向量化 | `--date`, `--provider`, `--model` | +| `ingest` | M6 入库 | `--date`, `--recreate` | +| `pipeline` | 全链路/守护 | `--once`, `--resume`, `--steps`, `--report`, `--cninfo-once` | +| `search` | 检索知识库 | `query`, `--top`, `--source`, `--sentiment`, `--stock`, `--industry`, `--min-importance` | +| `status` | 数据总览 | 无参数 | +| `report` | 生成日报 | `--date`, `--no-upload` | +| `report-import` | 历史日报导入 | `--dir`, `--date`, `--type`, `--force` | +| `cninfo` | 公告抓取 | `--no-save`, `--enrich-pdf`, `--pdf-limit` | +| `stock-report` | 个股日报 | `--no-upload` | +| `watchlist` | 关注列表管理 | `add`/`remove`/`list` | +| `discover` | 站点分析 | `url`, `--name`, `--extra`, `--add` | +| `add-entry` | 追加入口 | `source`, `urls...` | + +### 3.10 `report_db/` — M10 日报 DB 层 + +| 文件 | 职责 | +|------|------| +| `db.py` | 连接/事务/建表/保存/查询 | +| `models.py` | EventRow / ReportData (Pydantic) | +| `schema.py` | DDL(CREATE TABLE) | + +**公开 API**: + +```python +from report_db import ( + connect, init_schema, save_report, load_db_config, + exists_report, fetch_report, + EventRow, ReportData, +) +``` + +### 3.11 `report_import/` — M10 历史日报导入 + +| 文件 | 职责 | +|------|------| +| `parser.py` | BeautifulSoup 解析 finance/intl 历史 HTML | +| `importer.py` | 批量导入,幂等,ImportStats 统计 | + +**公开 API**: + +```python +from report_import import ( + import_history, ImportStats, + parse_report, parse_finance_report, parse_intl_report, ReportParseError, +) +``` + +--- + +## 4. 核心数据模型 + +### 4.1 管道数据模型关系 + +``` +CrawlResult (M1) ──→ Article (M2) ──→ DedupResult (M3) + │ + ▼ + ExtractedEvent (M4) + │ + ▼ + EmbeddingResult (M5) + │ + ▼ + SearchResult (M6) +``` + +### 4.2 日报数据模型 + +``` +ReportData (Pydantic) +├── report_date: date +├── report_type: str (finance | intl) +├── file_name: str +├── generated_at: datetime +├── ai_summary: str | None +├── stats: dict (JSON) +└── events: list[EventRow] + ├── section: str (xwlb | news | cninfo | intl) + ├── rank: int + ├── importance: int | None + ├── event_type: str | None + ├── title: str + ├── summary: str | None + ├── sentiment: str | None + └── source: str | None +``` + +### 4.3 全部 Pydantic 模型清单 + +| 包 | 模型 | 用途 | +|----|------|------| +| crawler | SourceConfig | 单个新闻源配置 | +| crawler | CrawlerConfig | 全局抓取配置 | +| crawler | CrawlerSettings | 抓取设置(并发/重试/UA) | +| crawler | CrawlResult | 单次抓取结果 | +| crawler | ArticleLink | 列表页发现的链接 | +| crawler | CninfoItem | 公告/调研/互动易条目 | +| extractor | Article | 标准化文章(M2 输出) | +| dedup | Fingerprint | 指纹记录(SQLite) | +| dedup | DedupResult | 判重结果 | +| dedup | DedupStats | 指纹库统计 | +| llm | EventExtraction | LLM 输出 JSON 结构 | +| llm | ExtractedEvent | M4 落盘格式(文章+事件) | +| embedding | EmbeddingResult | 向量化结果 | +| vectorstore | SearchFilter | 检索过滤条件 | +| vectorstore | SearchResult | 单条检索结果 | +| vectorstore | CollectionInfo | Qdrant Collection 信息 | +| scheduler | StepResult | 单步执行结果 | +| scheduler | PipelineResult | 全链路执行结果 | +| report_db | EventRow | 日报事件行 | +| report_db | ReportData | 完整日报数据 | + +--- + +## 5. 配置体系 + +### 5.1 配置文件 + +| 文件 | 格式 | 用途 | 热更新 | +|------|------|------|--------| +| `configs/sources.yaml` | YAML | 14 个新闻源配置 | 每次抓取重读 | +| `configs/llm_models.yaml` | YAML | 4 个 LLM 场景配置 | 每次调用重读 | +| `configs/watchlist.yaml` | YAML | cninfo 公告关注列表 | 每次操作重读 | +| `.env` | dotenv | API Key + 调度/超时/DB 配置 | 重启服务生效 | + +### 5.2 配置优先级(LLM 场景) + +``` +CLI 显式参数 (--provider / --model) + ↓ +configs/llm_models.yaml scenes.<场景>.xxx + ↓ +.env 环境变量 (LLM_PROVIDER / DEEPSEEK_MODEL 等) + ↓ +代码内置默认值 (仅超时/温度等参数) +``` + +> **模型名**无内置兜底:缺失即报错,绝不静默使用错误模型。 + +### 5.3 关键环境变量 + +| 变量 | 用途 | 生产值示例 | +|------|------|-----------| +| `DASHSCOPE_API_KEY` | 百炼 LLM + Embedding | sk-xxx | +| `DEEPSEEK_API_KEY` | DeepSeek LLM | sk-xxx | +| `SCHEDULE_TIMES` | 调度时间 | `07:00,12:00,18:00,22:00` | +| `PIPELINE_STEP_TIMEOUT` | 全局超时 | `1800` | +| `NEWS_DB_HOST` | 日报 DB 主机 | `192.168.1.10`(pi5) / `127.0.0.1`(Mac) | +| `NEWS_DB_PORT` | 日报 DB 端口 | `13306` | + +--- + +## 6. 产物目录结构 + +``` +data/ +├── raw/ # M1 原始抓取 +│ ├── cls/{YYYYMMDD}/ # 按源分目录 +│ │ ├── index.jsonl # 文章列表(CrawlResult 摘要) +│ │ └── *.html # 原始 HTML +│ ├── eastmoney/{YYYYMMDD}/ +│ ├── ... +│ └── cninfo/{YYYYMMDD}/ # 公告独立目录 +│ +├── processed/ # M2 正文提取 +│ ├── cls/{YYYYMMDD}/ +│ │ └── {url_hash}.json # Article +│ └── ... +│ +├── dedup/ # M3 指纹库(跨日) +│ └── fingerprints.sqlite3 +│ +├── deduped/ # M3 去重产物 +│ └── {YYYYMMDD}/ +│ ├── uniques/ +│ │ └── {url_hash}.json # 唯一文章 +│ ├── duplicates.jsonl # 重复记录 +│ └── sources.json # 多源映射 +│ +├── events/ # M4 事件抽取 +│ └── {YYYYMMDD}/ +│ └── {url_hash}.json # ExtractedEvent +│ +├── embeddings/ # M5 向量化 +│ └── {YYYYMMDD}/ +│ └── {url_hash}.json # 1024 维向量 + 事件 +│ +├── qdrant_storage/ # M6 Qdrant 本地文件 +│ +├── pipeline/ # 断点续跑状态 +│ └── state.json +│ +└── reports_history/ # 历史日报 HTML(可选) +``` + +--- + +## 7. 运行环境 + +| 项目 | 值 | +|------|-----| +| Python | 3.11 | +| 包管理器 | uv | +| 虚拟环境 | `.venv/` | +| 生产服务器 | 树莓派 5 (ARM64), `pi@192.168.1.160` | +| 数据库 | MySQL/MariaDB (myquant 库), 通过 `pi@192.168.1.10` autossh 隧道 | +| 系统服务 | systemd `a-share-research` + `a-share-db-tunnel` | +| 测试 | pytest (asyncio_mode=auto, integration 标记默认跳过) | +| 静态检查 | ruff + mypy | +| LLM 模型 | `deepseek-v4-flash`(锁定,不可修改) | + +--- + +—— 架构文档结束 —— \ No newline at end of file diff --git a/docs/db_schema.md b/docs/db_schema.md new file mode 100644 index 0000000..56aa701 --- /dev/null +++ b/docs/db_schema.md @@ -0,0 +1,125 @@ +# 日报结构化入库:数据库表结构与数据契约 + +> 版本:v1.0 | 2026-08-03 +> 用途:供 API / 前端对接读取日报数据。表位于 MySQL `myquant` 库,表前缀 `news_`。 +> 连接:`192.168.1.10:13306`(pi 上 autossh 隧道 → doorcome.cn:3306 MariaDB 10.11),用户 `myquant`(密码在服务器 `.env` 的 `NEWS_DB_PASSWORD`)。 + +--- + +## 1. 表结构 + +### 1.1 news_report(日报主表,一行 = 一份日报) + +| 字段 | 类型 | 说明 | +| --- | --- | --- | +| id | BIGINT UNSIGNED PK | 自增主键 | +| report_date | DATE | 日报日期 | +| report_type | VARCHAR(16) | `finance`=A 股日报 / `intl`=国际财经日报 | +| file_name | VARCHAR(160) | 历史文件源文件名;**新生成日报为空字符串 `""`** | +| generated_at | DATETIME | 生成时间 | +| ai_summary | TEXT | AI 摘要全文(含换行,按条目分行) | +| stats | JSON | 数据总览统计快照(见第 3 节),可为 NULL | +| created_at | DATETIME | 入库时间 | + +唯一键:`(report_date, report_type, file_name)` —— 历史同一天多次生成(intl 一日 3 次)保留多行;新生成日报 `file_name=''` 每天每类型仅一行,重复生成覆盖。 + +### 1.2 news_event(日报事件明细,一行 = 一条事件) + +| 字段 | 类型 | 说明 | +| --- | --- | --- | +| id | BIGINT UNSIGNED PK | 自增主键 | +| report_id | BIGINT UNSIGNED | FK → news_report.id | +| section | VARCHAR(16) | 板块:`xwlb`=新闻联播 / `news`=财经新闻 / `cninfo`=公告调研 / `intl`=国际重要事件 | +| rank | INT | 板块内序号(1 起) | +| importance | INT NULL | 重要度 1-5 | +| event_type | VARCHAR(64) NULL | 事件类型(如 宏观经济/地缘政治/新闻联播/公告) | +| title | VARCHAR(512) | 标题 | +| summary | TEXT NULL | 摘要/正文 | +| sentiment | VARCHAR(8) NULL | `positive` / `negative` / `neutral` | +| source | VARCHAR(64) NULL | 来源(如 `cls`、`investinglive.com`) | +| url | VARCHAR(512) NULL | 原文链接(新闻联播为空) | +| created_at | DATETIME | 入库时间 | + +索引:`idx_report_section (report_id, section)`。 + +--- + +## 2. 数据契约 + +- **幂等语义**:同一 `(report_date, report_type, file_name)` 重复写入会覆盖主表并全量替换事件(DELETE + INSERT),不会产生重复行。 +- **取最新**:同一天存在多份时(历史 intl 一日 3 次),前端按 `generated_at` 取最新;新日报 `file_name=''` 每天唯一。 +- **板块差异**:finance 日报含 `xwlb`+`news`+`cninfo` 三板块;intl 日报仅 `intl` 板块。前端按 `section` 过滤展示。 +- **历史覆盖范围**:2026-06-16 ~ 2026-08-03,共 177 行(finance 49 + intl 128;finance 少 1 因为两个目录存在同名文件被幂等合并)。事件总计 4222 条。 + +--- + +## 3. stats JSON 结构 + +`news_report.stats` 为数据总览快照,前端自行解析。finance 与 intl 的 key 集合不同: + +| key | finance | intl | 内容 | +| --- | --- | --- | --- | +| `pipeline` | ✅ | ✅ | M1→M6 管道各环节数量:`{label: 数量}` | +| `sources` | ✅ | — | 各新闻源文章数:`{源名: 数量}` | +| `news` | ✅ | — | 新闻统计:`{total, hi_threshold, sentiments, importances, event_types}` | +| `cninfo` | ✅ | — | 公告调研统计:`{total, hi_threshold, by_day, announcement, research, irm}` | +| `xwlb` | ✅ | — | 联播统计:`{total, date}`(有数据时才有) | +| `sentiment` | ✅ | ✅ | 情绪分布(历史文件为图例文本列表;新生成在 `news.sentiments`) | +| `importance` | ✅ | ✅ | 重要度分布:`[{重要度, 数量}, ...]` | +| `event_types` | ✅ | ✅ | 事件类型 TOP:`[{事件类型, 数量}, ...]` | +| `source_dist` | — | ✅ | 文章来源分布:`[{来源, 文章数}, ...]` | + +> 历史文件与新生成日报的 stats 结构存在差异(历史为 HTML 解析快照,新生成为结构化组装),前端建议按 key 防御性读取。 + +### 3.1 口径说明(重要,避免误解) + +`stats` 内各数字口径不同,请勿直接互相比较: + +| 字段 | 口径 | +| --- | --- | +| `pipeline.raw_total` | **日报日期当天**抓取的文章数(`data/raw/{src}/{date}/index.jsonl` 中 `stage=article 且 success` 的条目)。`raw_by_source` 是各源明细,**其和 = raw_total**;当天未抓取/无文章的源显示 0 | +| `pipeline.raw_total_24h` | 最近 24 小时内**抓取**(按 `fetched_at`)的文章数;`raw_by_source_24h` 为各源明细,和 = raw_total_24h。当天 07:00 抓取的数据其值 ≈ raw_total(并非"24h 内发布的新闻",raw 层无发布时间的可靠字段) | +| `pipeline.proc / deduped / dups / emb_count / qdrant_count` | 抽取 / 去重后 / 重复 / 向量化 / Qdrant 总量(`qdrant_count` 为全量累计,非当天) | +| `news.total` | **过去 30 小时窗口内**经 LLM 抽取的新闻事件数。**≠ raw_total**:raw 是抓取的文章数,news 是抽取后的事件数(会有过滤/合并),两者不可互相验证 | +| `news.importances` | `{重要度等级(1-5): 事件数}`,**各等级之和 = news.total** | +| `news.sentiments` | `{情绪: 事件数}`(positive/negative/neutral),和 = news.total | +| `news.event_types` | `{事件类型: 事件数}`(TOP 10) | + +--- + +## 4. 常用查询示例(API 实现参考) + +```sql +-- 某类型日报列表(取每天最新一份) +SELECT r.* FROM news_report r +JOIN ( + SELECT report_date, report_type, MAX(generated_at) AS g + FROM news_report GROUP BY report_date, report_type +) t ON r.report_date = t.report_date AND r.report_type = t.report_type + AND r.generated_at = t.g +WHERE r.report_type = 'finance' AND r.report_date >= '2026-07-01' +ORDER BY r.report_date DESC; + +-- 某日报的全部事件(按板块) +SELECT section, rank, importance, event_type, title, summary, sentiment, source, url +FROM news_event WHERE report_id = ? ORDER BY section, rank; + +-- 最近 N 天重要事件聚合(跨日报检索) +SELECT e.* FROM news_event e +JOIN news_report r ON r.id = e.report_id +WHERE r.report_date >= DATE_SUB(CURDATE(), INTERVAL 7 DAY) + AND e.importance >= 4 +ORDER BY e.importance DESC, r.report_date DESC; +``` + +--- + +## 5. 相关命令(数据生产侧) + +```bash +uv run a-share report --date YYYYMMDD # 生成当日日报并入库(finance) +uv run a-share report-import # 历史 HTML 全量解析入库(幂等) +uv run a-share report-import --date YYYYMMDD --type intl +``` + +代码:`report_db/`(连接/写入)、`report_import/`(历史解析/导入)、`scheduler/reporter.py`(日报生成)。 diff --git a/docs/user-guide.md b/docs/user-guide.md new file mode 100644 index 0000000..68d9259 --- /dev/null +++ b/docs/user-guide.md @@ -0,0 +1,752 @@ +# A 股 Deep Research 用户手册 + +> 版本:v2.0 | 最后更新:2026-08-22 + +--- + +## 目录 + +1. [项目概述](#1-项目概述) +2. [环境准备](#2-环境准备) +3. [快速开始](#3-快速开始) +4. [统一 CLI 参考](#4-统一-cli-参考) +5. [全链路与 Pipeline](#5-全链路与-pipeline) +6. [分模块使用](#6-分模块使用) +7. [定时任务与 systemd](#7-定时任务与-systemd) +8. [MCP 服务](#8-mcp-服务) +9. [日报系统](#9-日报系统) +10. [环境变量参考](#10-环境变量参考) +11. [常见问题](#11-常见问题) + +--- + +## 1. 项目概述 + +**A 股 Deep Research** 是一个私有化 A 股投研辅助平台,自动完成: + +``` +财经网站抓取 → 正文提取 → 去重 → LLM 投资事件抽取 → 向量化 → 知识库检索 → MCP 服务 +``` + +**核心能力**: + +- 从 14 个财经新闻源自动抓取(含 cninfo 公告) +- 23 种投资事件类型自动抽取(利好/利空/重要度) +- 1024 维语义向量检索(Qdrant 本地文件模式) +- MCP 协议接入 Cherry Studio / Claude Code +- 每日自动生成结构化日报并入库 MySQL + +**定位**:研究辅助与知识管理平台,**非**交易系统、**非**预测工具、**非**投资顾问。 + +**技术栈**:Python 3.11 / Crawl4AI / GNE / DeepSeek / Qwen / DashScope / Qdrant / APScheduler / MCP + +--- + +## 2. 环境准备 + +### 2.1 前置条件 + +- Python 3.11 +- uv 包管理器 + +```bash +# macOS +brew install uv + +# Linux +curl -LsSf https://astral.sh/uv/install.sh | sh +``` + +### 2.2 安装 + +```bash +cd /home/pi/news +uv sync + +# 可选:本地 BGE-M3 嵌入(离线,无需 API key) +uv sync --extra local-embedding + +# 可选:Playwright 浏览器(仅 JS 渲染源抓取需要) +uv run python -m playwright install chromium +``` + +### 2.3 配置 + +```bash +cp .env.example .env +# 编辑 .env,至少填写: +# DASHSCOPE_API_KEY=sk-xxx (百炼 LLM + Embedding) +# DEEPSEEK_API_KEY=sk-xxx (DeepSeek LLM,可选) +``` + +### 2.4 验证 + +```bash +uv run pytest -m "not integration" # 应显示 225+ passed +uv run a-share status # 查看数据状态 +``` + +--- + +## 3. 快速开始 + +### 3.1 三条命令入门 + +```bash +# 1. 抓取全网新闻,跑通全链路 +uv run a-share pipeline --once + +# 2. 语义检索 +uv run a-share search "宁德时代固态电池" + +# 3. 查看数据总览 +uv run a-share status +``` + +### 3.2 全链路 + 日报 + +```bash +# 跑全链路末尾生成日报 +uv run a-share pipeline --once --report + +# 或单独生成日报(已有数据时) +uv run a-share report --date 20260822 +``` + +### 3.3 cninfo 公告管道 + +```bash +# 全链路:公告抓取 → 提取 → PDF → 去重 → LLM → 入库 +uv run a-share pipeline --cninfo-once + +# 或分步 +uv run a-share cninfo # 只抓公告 +uv run a-share cninfo --enrich-pdf # 下载 PDF 补充正文 +``` + +--- + +## 4. 统一 CLI 参考 + +全部操作通过 `a-share` 命令完成。子命令速查: + +| 子命令 | 功能 | 常用参数 | +|--------|------|----------| +| `crawl` | M1 新闻抓取 | `--source cls` | +| `extract` | M2 正文提取 | `--date 20260822` | +| `dedup` | M3 三层去重 | `--date 20260822 --reset` | +| `events` | M4 LLM 事件抽取 | `--provider qwen --limit 5` | +| `embed` | M5 向量化 | `--provider local-bge` | +| `ingest` | M6 Qdrant 入库 | `--date 20260822 --recreate` | +| `pipeline` | 全链路/定时守护 | `--once --resume --report` | +| `search` | 语义检索 | `--source --stock --sentiment` | +| `status` | 数据总览 | 无参数 | +| `report` | 生成日报 | `--date 20260822` | +| `report-import` | 历史日报入库 | `--date 20260616 --type finance` | +| `cninfo` | 公告/调研/互动易 | `--enrich-pdf` | +| `stock-report` | 个股日报 | `--no-upload` | +| `watchlist` | 关注列表管理 | `add`/`remove`/`list` | +| `discover` | 站点分析 | `url --name --add` | +| `add-entry` | 追加多频道入口 | `source_id url...` | + +### 4.1 `a-share crawl` — M1 抓取 + +```bash +uv run a-share crawl # 抓取全部启用源 +uv run a-share crawl --source cls # 单源调试 +uv run a-share crawl --no-save # 试跑,不写文件 +``` + +### 4.2 `a-share extract` — M2 提取 + +```bash +uv run a-share extract --date 20260822 # 指定日期 +uv run a-share extract --source sina --date ... # 单源 +``` + +产物:`data/processed/{source}/{date}/{url_hash}.json` (Article) + +### 4.3 `a-share dedup` — M3 去重 + +```bash +uv run a-share dedup --date 20260822 # 增量去重 +uv run a-share dedup --date 20260822 --reset # 重建指纹库 +``` + +产物:`data/deduped/{date}/uniques/*.json` + `data/dedup/fingerprints.sqlite3` + +### 4.4 `a-share events` — M4 LLM 抽取 + +```bash +uv run a-share events --date 20260822 # 默认读 configs/llm_models.yaml +uv run a-share events --provider qwen --model qwen-plus # 临时覆盖 +uv run a-share events --limit 5 # 小批量调试 +uv run a-share events --concurrency 5 # 调并发(默认 3) +``` + +产物:`data/events/{date}/{url_hash}.json` (ExtractedEvent) + +### 4.5 `a-share embed` — M5 向量化 + +```bash +uv run a-share embed --date 20260822 +uv run a-share embed --provider local-bge # 本地 BGE-M3 +``` + +产物:`data/embeddings/{date}/{url_hash}.json` (1024 维向量) + +### 4.6 `a-share ingest` — M6 入库 + +```bash +uv run a-share ingest --date 20260822 +uv run a-share ingest --recreate # 重建 collection +``` + +### 4.7 `a-share search` — 语义检索 + +```bash +# 基础检索 +uv run a-share search "宁德时代固态电池" + +# 结构化过滤 +uv run a-share search "政策" --source cls --sentiment positive +uv run a-share search "减持" --stock 300750 --min-importance 3 +uv run a-share search "芯片" --industry 半导体 --top 5 +``` + +### 4.8 `a-share status` — 数据总览 + +```bash +uv run a-share status +``` + +输出各层统计(M1→M6 数据量) + 今日高重要度事件 TOP 10 + systemd 服务状态。 + +### 4.9 `a-share discover` — 站点分析 + +```bash +# 自动分析首页,输出 yaml 建议 +uv run a-share discover https://wallstreetcn.com/news/global + +# 自动写入 sources.yaml +uv run a-share discover https://example.com/news --name 某某财经 --add + +# 多频道入口 +uv run a-share discover https://example.com/news \ + --extra https://example.com/tech \ + --extra https://example.com/market \ + --name 某某财经 --add +``` + +### 4.10 `a-share add-entry` — 追加入口 + +```bash +uv run a-share add-entry wallstreetcn https://xxx.com/news/新频道 +``` + +### 4.11 `a-share watchlist` — 关注列表 + +```bash +uv run a-share watchlist add 300750 宁德时代 --note "动力电池龙头" +uv run a-share watchlist list +uv run a-share watchlist remove 000001 +``` + +关注列表存储在 `configs/watchlist.yaml`,用于 cninfo 公告过滤和个股日报。 + +### 4.12 `a-share cninfo` — 公告管道 + +```bash +uv run a-share cninfo # 抓取关注公司公告+调研+互动易 +uv run a-share cninfo --enrich-pdf # 下载 PDF 补充正文 +uv run a-share cninfo --pdf-limit 50 # 限制 PDF 处理数量 +``` + +### 4.13 `a-share stock-report` — 个股日报 + +```bash +uv run a-share stock-report # 为关注列表中每家公司生成日报 +uv run a-share stock-report --no-upload # 仅生成,不上传 +``` + +### 4.14 `a-share report` — 日报生成 + +```bash +uv run a-share report --date 20260822 # 生成指定日期日报并入库 MySQL +``` + +日报内容:新闻联播摘要 + 高重要度新闻 + 公告调研 + AI 要点分析。 + +### 4.15 `a-share report-import` — 历史日报导入 + +```bash +uv run a-share report-import # 全量导入(幂等) +uv run a-share report-import --date 20260616 # 只导入指定日期 +uv run a-share report-import --type intl # 只导入 intl 日报 +uv run a-share report-import --force # 覆盖已存在 +``` + +--- + +## 5. 全链路与 Pipeline + +### 5.1 一次性执行 + +```bash +# 全链路 M1→M6 +uv run a-share pipeline --once + +# 全链路 + 日报 +uv run a-share pipeline --once --report + +# 指定步骤 +uv run a-share pipeline --once --steps crawler,extractor,dedup + +# cninfo 全链路 +uv run a-share pipeline --cninfo-once +``` + +### 5.2 增量处理与断点续跑 + +**增量跳过**:M2/M4/M5 产物已存在时自动跳过,不重复调用 API/计费。使用 `--force` 参数可强制全量重建。 + +```bash +# pipeline 断点续跑:从上次失败步骤继续 +uv run a-share pipeline --once --resume + +# 断点续跑 + 日报(中断后直接重跑同一条命令即可) +uv run a-share pipeline --once --resume --report +``` + +断点状态文件:`data/pipeline/state.json`(按日期隔离,记录每步骤 ok/failed + 耗时) + +### 5.3 全链路耗时参考(100 篇文章) + +| 步骤 | 耗时 | 说明 | +|------|------|------| +| crawler | ~3 min | 含 Playwright 浏览器渲染 | +| xwlb | ~1 s | 新闻联播 API | +| extractor | ~24 s | GNE 正文提取 | +| dedup | ~2 s | 三层去重 | +| llm | ~45 s | DeepSeek 事件抽取 | +| embedding | ~10 s | DashScope 向量化 | +| qdrant | ~4 s | 写入知识库 | + +### 5.4 步骤超时配置 + +优先级:`TIMEOUT_{NAME}` 环境变量 > `PIPELINE_STEP_TIMEOUT` > 硬编码默认值 + +```bash +# .env 中设置 +PIPELINE_STEP_TIMEOUT=1800 # 全局兜底 +TIMEOUT_CRAWLER=900 # 单步精确控制 +TIMEOUT_LLM=900 +TIMEOUT_DEDUP=300 +``` + +--- + +## 6. 分模块使用 + +### 6.1 M1 — 新闻抓取 + +**新闻源**:14 个(13 Web + 1 API:新闻联播),配置文件 `configs/sources.yaml` + +**抓取分流**: + +| 条件 | 引擎 | 特点 | +|------|------|------| +| `js_render=false` 且无 `wait_for` | httpx 直连 | 快,不受反爬影响 | +| `js_render=true` 或有 `wait_for` | Playwright | 支持 JS 渲染 | + +**新增新闻源**: + +```bash +# 自动分析 → 输出 yaml 建议 +uv run a-share discover https://example.com/news + +# 自动分析 + 写入 sources.yaml +uv run a-share discover https://example.com/news --name 某某财经 --add + +# 验证新源 +uv run a-share crawl --source example +``` + +**为已有源追加多频道入口**: + +```bash +uv run a-share add-entry wallstreetcn https://xxx.com/news/china https://xxx.com/news/tech +``` + +### 6.2 M2 — 正文提取 + +输入:`data/raw/{source}/{date}/*.html` +输出:`data/processed/{source}/{date}/{url_hash}.json` (Article) + +```bash +uv run a-share extract --date 20260822 +uv run a-share extract --source cls --date 20260822 +``` + +### 6.3 M3 — 三层去重 + +| 层 | 算法 | 说明 | +|----|------|------| +| L1 | URL Hash | 完全相同 URL | +| L2 | Content Hash | 标准化正文完全一致 | +| L3 | SimHash | 汉明距离 ≤ 3(默认) | + +```bash +uv run a-share dedup --date 20260822 +uv run a-share dedup --reset # 重建指纹库 +``` + +### 6.4 M4 — 投资事件抽取 + +**23 种事件类型**:业绩预告/业绩快报/财报披露/合作签约/投资并购/重大合同/产品发布/技术突破/监管处罚/诉讼仲裁/股东减持/股东增持/回购/分红/高管变动/资产重组/停牌复牌/ST警示/退市风险/宏观政策/行业政策/国际局势/其他 + +**LLM 配置**:`configs/llm_models.yaml`(4 场景独立配置),支持 DeepSeek 和 Qwen + +```bash +uv run a-share events --date 20260822 +uv run a-share events --provider qwen --model qwen-plus +uv run a-share events --limit 5 --concurrency 3 +``` + +### 6.5 M5 — 向量化 + +**两种模式**: + +| 模式 | 命令 | 要求 | +|------|------|------| +| DashScope 远程 | 默认 | `DASHSCOPE_API_KEY` | +| 本地 BGE-M3 | `--provider local-bge` | `uv sync --extra local-embedding` | + +```bash +uv run a-share embed --date 20260822 +uv run a-share embed --provider local-bge --date 20260822 +``` + +### 6.6 M6 — Qdrant 知识库 + +**默认本地文件模式**(零依赖,ARM64 兼容),数据在 `data/qdrant_storage/` + +```bash +uv run a-share ingest --date 20260822 +uv run a-share ingest --recreate # 重建 collection +``` + +**Python 检索示例**: + +```python +from vectorstore import VectorStore, SearchFilter, make_qdrant_client + +c = make_qdrant_client() +store = VectorStore(c) + +hits = store.query( + query_vector=my_vector, + top_k=10, + filter=SearchFilter( + source_id="cls", + importance_min=3, + sentiment="positive", + stock_codes=["300750"], + ), +) +store.close() +``` + +### 6.7 cninfo 公告管道 + +cninfo(巨潮资讯网)是独立的 A 股公告/调研/互动易抓取管道,与新闻抓取分开调度。 + +```bash +# 全链路一条命令 +uv run a-share pipeline --cninfo-once + +# 分步操作 +uv run a-share cninfo # 只抓公告(关注公司) +uv run a-share cninfo --enrich-pdf # PDF 正文补充 +uv run a-share extract --source cninfo --date 20260822 # 正文提取 +``` + +--- + +## 7. 定时任务与 systemd + +### 7.1 安装 systemd 服务 + +```bash +sudo cp scripts/a-share-research.service /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable a-share-research +``` + +### 7.2 日常操作 + +```bash +sudo systemctl start a-share-research # 启动 +sudo systemctl stop a-share-research # 停止 +sudo systemctl restart a-share-research # 重启 +sudo systemctl status a-share-research # 查看状态 + +# 查看日志 +journalctl -u a-share-research -f # 实时系统日志 +tail -f logs/scheduler.log # 文件日志 +``` + +### 7.3 调度时间表 + +| 时间 | 步骤 | +|------|------| +| 06:30 | cninfo 公告管道 | +| 07:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant→**日报** | +| 07:30 | 个股日报 | +| 12:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | +| 18:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | +| 22:00 | crawler→xwlb→extractor→dedup→llm→embedding→qdrant | + +### 7.4 修改调度时间 + +```bash +# 编辑 .env 中的 SCHEDULE_TIMES,格式: HH:MM,HH:MM,... +nano /home/pi/news/.env +# 例: SCHEDULE_TIMES=08:00,14:00,20:00 + +# 重启生效 +sudo systemctl restart a-share-research +``` + +### 7.5 前台守护模式(调试) + +```bash +uv run a-share pipeline +# Ctrl+C 退出 +``` + +--- + +## 8. MCP 服务 + +### 8.1 连接 Cherry Studio + +Cherry Studio → 设置 → MCP 服务器 → 添加: + +```json +{ + "mcpServers": { + "a-share-research": { + "command": "uv", + "args": ["run", "python", "-m", "scripts.run_mcp_server"], + "cwd": "/home/pi/news" + } + } +} +``` + +### 8.2 连接 Claude Code + +编辑项目根目录 `.mcp.json`: + +```json +{ + "mcpServers": { + "a-share-research": { + "command": "uv", + "args": ["run", "python", "-m", "scripts.run_mcp_server"], + "cwd": "/home/pi/news" + } + } +} +``` + +### 8.3 5 个 MCP 工具 + +| 工具 | 参数 | 示例 | +|------|------|------| +| `search_news` | query, top_k | `search_news("宁德时代固态电池")` | +| `search_company_news` | query, company, top_k | `search_company_news("价格", company="贵州茅台")` | +| `search_industry_news` | query, industry, top_k | `search_industry_news("政策", industry="半导体")` | +| `search_stock_events` | query, stock_code, top_k | `search_stock_events("重大事件", stock_code="300750")` | +| `search_sentiment_trend` | query, sentiment, top_k | `search_sentiment_trend("AI算力", sentiment="all")` | + +### 8.4 调试(SSE 模式) + +```bash +uv run python -m scripts.run_mcp_server --sse 8765 +# 浏览器访问 http://:8765/sse +``` + +### 8.5 工作原理 + +``` +用户输入自然语言查询 + → DashScope 嵌入(1024 维) + → Qdrant 余弦相似度检索 + → 按过滤条件筛选 + → Markdown 格式化返回 +``` + +--- + +## 9. 日报系统 + +### 9.1 日报生成 + +```bash +uv run a-share report --date 20260822 +``` + +日报内容:新闻联播摘要 + 近 30 小时高重要度新闻 + 近 15 日公告/调研 + AI 要点分析。 + +**入库目标**:MySQL `myquant` 库,表 `news_report`(主表) + `news_event`(事件明细)。 + +### 9.2 数据库表结构 + +详见 `docs/db-schema.md`(供 API/前端对接)。 + +### 9.3 历史日报导入 + +```bash +# 一次性导入 178 份历史日报 HTML +uv run a-share report-import + +# 或按日期/类型筛选 +uv run a-share report-import --date 20260616 --type finance +uv run a-share report-import --force # 覆盖已存在 +``` + +### 9.4 个股日报 + +```bash +uv run a-share stock-report +``` + +为 `configs/watchlist.yaml` 中每家公司生成综合日报(AI 要点 + 公告 + 新闻 + 互动问答)。 + +--- + +## 10. 环境变量参考 + +### 10.1 LLM 与 Embedding + +| 变量 | 默认值 | 说明 | +|------|--------|------| +| `LLM_PROVIDER` | `deepseek` | 默认 LLM 提供商 | +| `DEEPSEEK_API_KEY` | — | DeepSeek API Key | +| `DEEPSEEK_BASE_URL` | `https://api.deepseek.com` | DeepSeek API 地址 | +| `DASHSCOPE_API_KEY` | — | 百炼 API Key(LLM Qwen + Embedding 共用) | +| `QWEN_BASE_URL` | `https://dashscope.aliyuncs.com/compatible-mode/v1` | Qwen API 地址 | +| `EMBEDDING_PROVIDER` | `dashscope` | 嵌入提供商 | +| `LLM_RETRY_TIMES` | `3` | AI 摘要调用失败重试次数 | +| `LLM_RETRY_BACKOFF_SEC` | `2.0` | 指数退避基数(秒) | + +> **LLM 场景配置**:`configs/llm_models.yaml` 优先级高于环境变量,每个场景(事件抽取/日报摘要/个股分析/嵌入)可独立指定 provider 和 model。 + +### 10.2 调度与超时 + +| 变量 | 默认值 | 说明 | +|------|--------|------| +| `SCHEDULE_TIMES` | `07:00,12:00,18:00,22:00` | 定时任务时间 | +| `CNINFO_SCHEDULE_TIME` | `06:30` | cninfo 公告时间 | +| `STOCK_REPORT_TIME` | `07:30` | 个股日报时间 | +| `STOCK_REPORT_DAYS` | `15` | 个股日报回溯天数 | +| `PIPELINE_STEP_TIMEOUT` | — | 全局步骤超时(秒) | +| `TIMEOUT_CRAWLER` | — | 抓取步骤超时 | +| `TIMEOUT_LLM` | — | LLM 步骤超时 | +| `TIMEOUT_DEDUP` | — | 去重步骤超时 | +| `TIMEOUT_EMBEDDING` | — | 向量化步骤超时 | +| `TIMEOUT_QDRANT` | — | Qdrant 步骤超时 | +| `TIMEOUT_XWLB` | — | 新闻联播步骤超时 | + +### 10.3 Qdrant + +| 变量 | 默认值 | 说明 | +|------|--------|------| +| `QDRANT_HOST` | `localhost` | Qdrant 服务地址 | +| `QDRANT_PORT` | `6333` | Qdrant 端口 | +| `QDRANT_COLLECTION` | `a_share_news` | Collection 名 | + +### 10.4 日报入库 + +| 变量 | 默认值 | 说明 | +|------|--------|------| +| `NEWS_DB_HOST` | `127.0.0.1` | 日报 MySQL 主机 | +| `NEWS_DB_PORT` | `13306` | 日报 MySQL 端口 | +| `NEWS_DB_USER` | `myquant` | 数据库用户 | +| `NEWS_DB_PASSWORD` | — | 数据库密码 | +| `NEWS_DB_NAME` | `myquant` | 数据库名 | +| `REPORT_HISTORY_DIR` | `data/reports_history` | 历史日报目录 | + +### 10.5 cninfo + +| 变量 | 默认值 | 说明 | +|------|--------|------| +| `CNINFO_PDF_BASE` | `http://static.cninfo.com.cn` | PDF 基础 URL | + +--- + +## 11. 常见问题 + +**Q: 某源抓取 0 篇文章?** + +检查 `configs/sources.yaml` 中该源的 `article_url_pattern` 正则是否匹配真实链接格式。先用 `uv run a-share crawl --source ` 单源调试。 + +**Q: M2 提取的正文是模板文本(如"郑重声明")?** + +提取器已内置关键词黑名单自动过滤。若遇到新模板,可在 `extractor/parser.py` 的 `_BOILERPLATE_PATTERNS` 追加。 + +**Q: M4 调用 LLM 报错?** + +检查 `.env` 中 `DASHSCOPE_API_KEY` 是否填写。可用 `--provider qwen` 切换到百炼测试。模型名缺失时直接报错,检查 `configs/llm_models.yaml` 中 `event_extraction` 场景的 `model` 字段。 + +**Q: Qdrant 搜索不到结果?** + +```bash +uv run python -c " +from vectorstore import VectorStore, make_qdrant_client +c=make_qdrant_client(); s=VectorStore(c) +print(s.count()); s.close() +" +``` + +若 count=0,运行 `uv run a-share ingest --date ` 入库。 + +**Q: 如何从零重建知识库?** + +```bash +rm -rf data/raw data/processed data/dedup data/deduped data/events data/embeddings data/qdrant_storage +uv run a-share pipeline --once +``` + +**Q: 如何避免重复调用 LLM/Embedding API?** + +M2/M4/M5 默认增量处理:产物已存在即跳过(2026-08-12 新增)。中断后重跑: + +```bash +uv run a-share pipeline --once --resume --report +``` + +从断点继续,已成功的步骤不会重复执行。 + +**Q: 树莓派 Qdrant Docker 启动崩溃?** + +树莓派 ARM64 内核使用 16KB 内存页,Qdrant 官方 Docker 镜像内置的 jemalloc 仅支持 4KB 页。使用本地文件模式(默认)即可,无需 Docker。 + +**Q: 如何新增新闻源?** + +```bash +uv run a-share discover https://example.com/news --name 某某财经 --add +uv run a-share crawl --source 某某财经 # 验证 +``` + +**Q: 服务器 .env 和本地 .env 有何不同?** + +- **生产 pi5**:`NEWS_DB_HOST=192.168.1.10`(直连 DB 隧道) +- **Mac 本地**:`NEWS_DB_HOST=127.0.0.1`(通过 ssh 隧道) +- 两处 `.env` 不同,**勿互相覆盖** + +--- + +—— 用户手册结束 —— \ No newline at end of file diff --git a/embedding/__init__.py b/embedding/__init__.py new file mode 100644 index 0000000..a906c0b --- /dev/null +++ b/embedding/__init__.py @@ -0,0 +1,52 @@ +"""Embedding 向量化模块 (M5)。 + +公共 API: + - EmbeddingResult / EmbeddingError / EmbeddingProviderType + - EmbeddingProvider / AsyncEmbeddingProvider (ABC) + - compose_text (统一文本组装策略) + - DashScopeEmbeddingProvider / DashScopeAsyncEmbeddingProvider + - LocalBGEEmbeddingProvider / LocalBGEAsyncEmbeddingProvider (可选) + - resolve_provider_type / make_sync_provider / make_async_provider +""" + +from .base import ( + MAX_TEXT_CHARS, + AsyncEmbeddingProvider, + EmbeddingProvider, + compose_text, +) +from .factory import ( + make_async_provider, + make_sync_provider, + resolve_provider_type, +) +from .models import ( + EmbeddingError, + EmbeddingProviderType, + EmbeddingResult, +) +from .remote import ( + DASHSCOPE_BATCH_LIMIT, + DASHSCOPE_DEFAULT_DIM, + DASHSCOPE_DEFAULT_MODEL, + DashScopeAsyncEmbeddingProvider, + DashScopeEmbeddingProvider, +) + +__all__ = [ + "DASHSCOPE_BATCH_LIMIT", + "DASHSCOPE_DEFAULT_DIM", + "DASHSCOPE_DEFAULT_MODEL", + "MAX_TEXT_CHARS", + "AsyncEmbeddingProvider", + "DashScopeAsyncEmbeddingProvider", + "DashScopeEmbeddingProvider", + "EmbeddingError", + "EmbeddingProvider", + "EmbeddingProviderType", + "EmbeddingResult", + "compose_text", + "make_async_provider", + "make_sync_provider", + "resolve_provider_type", +] diff --git a/embedding/base.py b/embedding/base.py new file mode 100644 index 0000000..75b2c23 --- /dev/null +++ b/embedding/base.py @@ -0,0 +1,154 @@ +"""Embedding provider 抽象接口与文本组装工具。 + +文本组装策略 (compose_text): + 优先组装 ExtractedEvent 时,把"语义浓缩"信息前置: + title | sentiment+importance+event_type | summary | content[截断] + 退化为 Article 时: + title | content[截断] + 超长截断保护:默认 4000 字符(BGE-M3 max_seq=8192,远程也保守取值)。 +""" + +from __future__ import annotations + +from abc import ABC, abstractmethod +from datetime import datetime +from typing import Any + +from extractor import Article + +# 拼接后送入 embedder 的字符数上限 +MAX_TEXT_CHARS = 4000 + + +def _from_event_dict(event_obj: dict[str, Any]) -> tuple[Article, str, str | None]: + """从 ExtractedEvent JSON dict 中,取出 Article 元数据 + 加权 head 段。 + + 返回 (article, head, summary): + head 为"事件标签摘要",会作为前缀拼到嵌入文本前; + summary 为 event.summary。 + """ + title = event_obj.get("title") or "" + url = event_obj.get("url") or "" + url_hash = event_obj.get("url_hash") or "" + source_id = event_obj.get("source_id") or "" + publish_time_raw = event_obj.get("publish_time") + publish_time = ( + datetime.fromisoformat(publish_time_raw) + if isinstance(publish_time_raw, str) and publish_time_raw + else None + ) + + ev = event_obj.get("event") or {} + sentiment = ev.get("sentiment") or "neutral" + importance = ev.get("importance") or 0 + event_type = ev.get("event_type") or "其他" + stock_codes = ev.get("stock_codes") or [] + company_names = ev.get("company_names") or [] + industries = ev.get("industries") or [] + summary = ev.get("summary") or "" + + head_parts = [ + f"sentiment={sentiment}", + f"importance={importance}", + f"event_type={event_type}", + ] + if company_names: + head_parts.append("公司=" + ",".join(company_names[:5])) + if industries: + head_parts.append("行业=" + ",".join(industries[:5])) + if stock_codes: + head_parts.append("代码=" + ",".join(stock_codes[:5])) + head = "[" + " ".join(head_parts) + "]" + + # 用 ExtractedEvent 中存在的字段构造一个最小 Article 让下游兼容 + article = Article( + source_id=source_id, + url=url, + url_hash=url_hash, + title=title, + content=ev.get("summary") or title, # 占位,真正的正文从原 article 文件读 + publish_time=publish_time, + word_count=0, + ) + return article, head, summary + + +def compose_text( + article: Article, + *, + head: str | None = None, + summary: str | None = None, + max_chars: int = MAX_TEXT_CHARS, +) -> str: + """把 Article 组装成单段嵌入文本。 + + 参数: + article: 输入文章(用于 title + content) + head: 可选事件标签摘要(由 ExtractedEvent 提取),前置可提高检索信号 + summary: 可选 LLM 生成的一句话摘要,前置 head 之后 + max_chars: 整段最大字符数,超出截断 content + """ + parts: list[str] = [f"标题:{article.title}"] + if head: + parts.append(head) + if summary: + parts.append(f"摘要:{summary}") + body = article.content or "" + parts.append("正文:" + body) + text = "\n".join(parts) + if len(text) > max_chars: + text = text[:max_chars] + return text + + +# --------------------------------------------------------------------------- # +# Provider 抽象接口 +# --------------------------------------------------------------------------- # + +class EmbeddingProvider(ABC): + """同步嵌入 provider 抽象。""" + + name: str + model: str + dim: int + + @abstractmethod + def embed_batch(self, texts: list[str]) -> list[list[float]]: + """批量嵌入,返回与输入等长的向量列表。""" + + def embed_one(self, text: str) -> list[float]: + """单条嵌入,默认走 batch=1。""" + return self.embed_batch([text])[0] + + def close(self) -> None: # noqa: B027 - 默认空实现,子类按需覆盖 + """释放资源(如 HTTP client / 模型)。""" + + def __enter__(self) -> EmbeddingProvider: + return self + + def __exit__(self, *_: object) -> None: + self.close() + + +class AsyncEmbeddingProvider(ABC): + """异步嵌入 provider(用于批处理高并发)。""" + + name: str + model: str + dim: int + + @abstractmethod + async def embed_batch(self, texts: list[str]) -> list[list[float]]: + ... + + async def embed_one(self, text: str) -> list[float]: + return (await self.embed_batch([text]))[0] + + async def close(self) -> None: # noqa: B027 - 默认空实现 + ... + + async def __aenter__(self) -> AsyncEmbeddingProvider: + return self + + async def __aexit__(self, *_: object) -> None: + await self.close() diff --git a/embedding/factory.py b/embedding/factory.py new file mode 100644 index 0000000..10f7c8f --- /dev/null +++ b/embedding/factory.py @@ -0,0 +1,71 @@ +"""Embedding provider 工厂:根据配置构造合适后端。 + +配置优先级: 显式参数 > configs/llm_models.yaml scenes.embedding > .env > 默认。 +""" + +from __future__ import annotations + +import os + +from configs.loader import load_scene_config + +from .base import AsyncEmbeddingProvider, EmbeddingProvider +from .models import EmbeddingError, EmbeddingProviderType +from .remote import ( + DashScopeAsyncEmbeddingProvider, + DashScopeEmbeddingProvider, +) + + +def _read_env(key: str, default: str | None = None) -> str | None: + val = os.environ.get(key) + if val is None or val.strip() == "": + return default + return val.strip() + + +def resolve_provider_type(provider: str | None = None) -> EmbeddingProviderType: + """根据 provider 参数 / YAML 场景 / env 解析出 EmbeddingProviderType。 + + 映射: + dashscope / qwen / remote -> DASHSCOPE + local / local-bge / bge / bge-m3 -> LOCAL_BGE + 默认 dashscope。 + """ + scene_provider = load_scene_config("embedding").get("provider") + p = ( + provider + or scene_provider + or _read_env("EMBEDDING_PROVIDER", "dashscope") + or "dashscope" + ).lower() + if p in ("dashscope", "qwen", "remote"): + return EmbeddingProviderType.DASHSCOPE + if p in ("local", "local-bge", "bge", "bge-m3"): + return EmbeddingProviderType.LOCAL_BGE + raise EmbeddingError(f"未知 embedding provider: {provider!r}") + + +def make_sync_provider( + provider: str | None = None, + **kwargs: object, +) -> EmbeddingProvider: + """构造同步 provider。""" + pt = resolve_provider_type(provider) + if pt == EmbeddingProviderType.DASHSCOPE: + return DashScopeEmbeddingProvider(**kwargs) # type: ignore[arg-type] + # 本地后端 + from .local import LocalBGEEmbeddingProvider + return LocalBGEEmbeddingProvider(**kwargs) # type: ignore[arg-type] + + +def make_async_provider( + provider: str | None = None, + **kwargs: object, +) -> AsyncEmbeddingProvider: + """构造异步 provider。""" + pt = resolve_provider_type(provider) + if pt == EmbeddingProviderType.DASHSCOPE: + return DashScopeAsyncEmbeddingProvider(**kwargs) # type: ignore[arg-type] + from .local import LocalBGEAsyncEmbeddingProvider + return LocalBGEAsyncEmbeddingProvider(**kwargs) # type: ignore[arg-type] diff --git a/embedding/local.py b/embedding/local.py new file mode 100644 index 0000000..4aff44f --- /dev/null +++ b/embedding/local.py @@ -0,0 +1,113 @@ +"""本地 BGE-M3 嵌入实现(可选)。 + +需要额外安装本地依赖: + uv sync --extra local-embedding + +模型 BAAI/bge-m3 首次加载约 2.3 GB(从 HuggingFace 自动下载)。 +默认 1024 维,与 DashScope 兼容。 +""" + +from __future__ import annotations + +import asyncio +import os +from typing import TYPE_CHECKING + +from loguru import logger + +from .base import AsyncEmbeddingProvider, EmbeddingProvider +from .models import EmbeddingError + +if TYPE_CHECKING: + from sentence_transformers import SentenceTransformer + +from configs.loader import load_scene_config + +LOCAL_DEFAULT_MODEL = "BAAI/bge-m3" +LOCAL_DEFAULT_DIM = 1024 + + +def _read_env(key: str, default: str | None = None) -> str | None: + val = os.environ.get(key) + if val is None or val.strip() == "": + return default + return val.strip() + + +def _try_import_st() -> type[SentenceTransformer]: + try: + from sentence_transformers import SentenceTransformer # noqa: F811 + except ImportError as e: + raise EmbeddingError( + "本地 BGE-M3 后端需要 sentence-transformers,请执行: " + "uv sync --extra local-embedding" + ) from e + return SentenceTransformer + + +class LocalBGEEmbeddingProvider(EmbeddingProvider): + """本地 BGE-M3 同步实现。""" + + name = "local-bge" + + def __init__( + self, + *, + model: str | None = None, + device: str | None = None, + normalize: bool = True, + ) -> None: + st_cls = _try_import_st() + # 模型优先级:CLI/参数 > YAML scenes.embedding > LOCAL_EMBEDDING_MODEL > 默认值 + # 不读全局 EMBEDDING_MODEL,避免与 DashScope 冲突 + scene_model = load_scene_config("embedding").get("model") + self.model = ( + model + or scene_model + or _read_env("LOCAL_EMBEDDING_MODEL", LOCAL_DEFAULT_MODEL) + or LOCAL_DEFAULT_MODEL + ) + self.dim = LOCAL_DEFAULT_DIM + self._device = device # None -> 让 ST 自选 cpu/cuda + self._normalize = normalize + logger.info("加载本地嵌入模型 {} (device={}),首次会下载...", self.model, device or "auto") + self._st = st_cls(self.model, device=device) + # 实际维度自检 + actual_dim = self._st.get_sentence_embedding_dimension() + if actual_dim and actual_dim != self.dim: + logger.warning( + "BGE 模型实际维度 {} 与默认 {} 不一致,以实际为准", actual_dim, self.dim + ) + self.dim = actual_dim + + def embed_batch(self, texts: list[str]) -> list[list[float]]: + if not texts: + return [] + try: + arr = self._st.encode( + texts, + normalize_embeddings=self._normalize, + convert_to_numpy=True, + show_progress_bar=False, + ) + except Exception as e: # noqa: BLE001 + raise EmbeddingError(f"BGE-M3 推理失败: {type(e).__name__}: {e}") from e + return [v.tolist() for v in arr] + + +class LocalBGEAsyncEmbeddingProvider(AsyncEmbeddingProvider): + """本地 BGE-M3 异步包装。 + + sentence-transformers 本身是同步的,这里用 asyncio.to_thread 包一层, + 主要为了让批处理脚本能用统一的 async 接口。 + """ + + name = "local-bge" + + def __init__(self, **kwargs: object) -> None: + self._sync = LocalBGEEmbeddingProvider(**kwargs) # type: ignore[arg-type] + self.model = self._sync.model + self.dim = self._sync.dim + + async def embed_batch(self, texts: list[str]) -> list[list[float]]: + return await asyncio.to_thread(self._sync.embed_batch, texts) diff --git a/embedding/models.py b/embedding/models.py new file mode 100644 index 0000000..cb60b3b --- /dev/null +++ b/embedding/models.py @@ -0,0 +1,49 @@ +"""Embedding 模块的数据模型 (M5)。""" + +from __future__ import annotations + +from datetime import datetime +from enum import StrEnum + +from pydantic import BaseModel, Field + + +class EmbeddingProviderType(StrEnum): + """支持的 embedding provider 标识。""" + + DASHSCOPE = "dashscope" # 远程 Qwen / 百炼 + LOCAL_BGE = "local-bge" # 本地 BGE-M3 + + +class EmbeddingResult(BaseModel): + """单篇文章的嵌入结果(落盘格式)。""" + + url_hash: str = Field(..., description="主键,与 Article.url_hash 一致") + source_id: str = Field(..., description="来源源 id") + title: str = Field(..., description="原文标题(便于人工检索)") + text: str = Field( + ..., description="实际送入 embedder 的文本(已截断/拼接)" + ) + vector: list[float] = Field(..., description="嵌入向量") + dim: int = Field(..., gt=0, description="向量维度") + + provider: str = Field(..., description="dashscope / local-bge") + model: str = Field(..., description="嵌入模型名") + embedded_at: datetime = Field(default_factory=datetime.now) + char_count: int = Field(default=0, ge=0, description="text 字符数,便于排查") + publish_time: datetime | None = None + + def short_summary(self) -> str: + return ( + f"[{self.source_id}] {self.title[:30]} " + f"dim={self.dim} provider={self.provider}" + ) + + +class EmbeddingError(Exception): + """嵌入调用失败。""" + + def __init__(self, reason: str, *, attempts: int = 0) -> None: + super().__init__(reason) + self.reason = reason + self.attempts = attempts diff --git a/embedding/remote.py b/embedding/remote.py new file mode 100644 index 0000000..36d1b6e --- /dev/null +++ b/embedding/remote.py @@ -0,0 +1,223 @@ +"""DashScope / Qwen 远程嵌入实现。 + +通过 OpenAI 兼容接口调用阿里百炼的 text-embedding-v3: + base_url: https://dashscope.aliyuncs.com/compatible-mode/v1 + model: text-embedding-v3 (1024 维) + 限制: 单次请求 input ≤ 25 条 + +配置来源(优先级从高到低): + 1. 构造参数(model / api_key / base_url / max_attempts) + 2. configs/llm_models.yaml 的 scenes.embedding + 3. 环境变量 / .env: + DASHSCOPE_EMBEDDING_API_KEY / DASHSCOPE_API_KEY + DASHSCOPE_EMBEDDING_BASE_URL / QWEN_BASE_URL + DASHSCOPE_EMBEDDING_MODEL + 4. 代码内置默认值(text-embedding-v3 / 1024 维) +""" + +from __future__ import annotations + +import asyncio +import os + +from loguru import logger +from openai import AsyncOpenAI, OpenAI + +from configs.loader import load_scene_config + +from .base import AsyncEmbeddingProvider, EmbeddingProvider +from .models import EmbeddingError + +DASHSCOPE_DEFAULT_BASE = "https://dashscope.aliyuncs.com/compatible-mode/v1" +DASHSCOPE_DEFAULT_MODEL = "text-embedding-v3" +DASHSCOPE_DEFAULT_DIM = 1024 +DASHSCOPE_BATCH_LIMIT = 10 # 百炼实测单批上限(2026-06,文档曾标 25 但 API 报 400) + +# 重试策略 +DEFAULT_MAX_ATTEMPTS = 3 +RETRY_BASE_WAIT_SEC = 1.0 +RETRY_MAX_WAIT_SEC = 8.0 + +# embedding 场景名(对应 configs/llm_models.yaml scenes.embedding) +SCENE_EMBEDDING = "embedding" + + +def _read_env(key: str, default: str | None = None) -> str | None: + val = os.environ.get(key) + if val is None or val.strip() == "": + return default + return val.strip() + + +def _scene() -> dict: + """读取 YAML embedding 场景配置(不存在时为空 dict)。""" + return load_scene_config(SCENE_EMBEDDING) + + +def _scene_int(key: str, default: int) -> int: + try: + return int(_scene().get(key) or default) + except (TypeError, ValueError): + return default + + +def _scene_float(key: str, default: float) -> float: + try: + return float(_scene().get(key) or default) + except (TypeError, ValueError): + return default + + +def _resolve_config() -> tuple[str, str, str]: + """读取 API key / base_url / model,返回 (api_key, base_url, model)。 + + 优先级: YAML 场景 > 环境变量 > 内置默认。 + 模型名优先级:DASHSCOPE_EMBEDDING_MODEL > 默认值。 + 不再读全局 EMBEDDING_MODEL,避免与 LOCAL provider 冲突。 + """ + sc = _scene() + api_key_env = sc.get("api_key_env") or "DASHSCOPE_EMBEDDING_API_KEY" + api_key = _read_env(api_key_env) or _read_env("DASHSCOPE_API_KEY") or "" + if not api_key: + raise EmbeddingError( + f"{api_key_env} 或 DASHSCOPE_API_KEY 未配置" + ) + # YAML base_url_env -> DASHSCOPE_EMBEDDING_BASE_URL -> QWEN_BASE_URL(兜底) -> 默认 + base_url = ( + _read_env(sc.get("base_url_env") or "DASHSCOPE_EMBEDDING_BASE_URL") + or _read_env("QWEN_BASE_URL") + or DASHSCOPE_DEFAULT_BASE + ) + model = ( + sc.get("model") + or _read_env("DASHSCOPE_EMBEDDING_MODEL", DASHSCOPE_DEFAULT_MODEL) + or DASHSCOPE_DEFAULT_MODEL + ) + return api_key, base_url, model + + +def _chunked(items: list[str], size: int) -> list[list[str]]: + """把列表按 size 分块。""" + return [items[i : i + size] for i in range(0, len(items), size)] + + +class DashScopeEmbeddingProvider(EmbeddingProvider): + """同步实现,主要用于测试/单条调用。""" + + name = "dashscope" + + def __init__( + self, + *, + model: str | None = None, + api_key: str | None = None, + base_url: str | None = None, + timeout_sec: float | None = None, + max_attempts: int | None = None, + batch_limit: int | None = None, + ) -> None: + env_key, env_base, env_model = _resolve_config() + self.model = model or env_model + self.dim = DASHSCOPE_DEFAULT_DIM + self.max_attempts = max_attempts or _scene_int("max_attempts", DEFAULT_MAX_ATTEMPTS) + self.batch_limit = batch_limit or _scene_int("batch_limit", DASHSCOPE_BATCH_LIMIT) + timeout = timeout_sec or _scene_float("timeout_sec", 60.0) + self._client = OpenAI( + api_key=api_key or env_key, + base_url=base_url or env_base, + timeout=timeout, + ) + + def embed_batch(self, texts: list[str]) -> list[list[float]]: + if not texts: + return [] + results: list[list[float]] = [] + for chunk in _chunked(texts, self.batch_limit): + results.extend(self._call_with_retry(chunk)) + return results + + def _call_with_retry(self, batch: list[str]) -> list[list[float]]: + import time + + last_err: Exception | None = None + for attempt in range(1, self.max_attempts + 1): + try: + resp = self._client.embeddings.create(model=self.model, input=batch) + return [d.embedding for d in resp.data] + except Exception as e: # noqa: BLE001 + last_err = e + logger.warning( + "DashScope embed 失败 尝试 {}/{}: {}: {}", + attempt, self.max_attempts, type(e).__name__, e, + ) + if attempt < self.max_attempts: + wait = min(RETRY_BASE_WAIT_SEC * (2 ** (attempt - 1)), RETRY_MAX_WAIT_SEC) + time.sleep(wait) + raise EmbeddingError( + f"DashScope embed 放弃 {self.max_attempts} 次: {last_err}", + attempts=self.max_attempts, + ) + + def close(self) -> None: + self._client.close() + + +class DashScopeAsyncEmbeddingProvider(AsyncEmbeddingProvider): + """异步实现,用于批处理。""" + + name = "dashscope" + + def __init__( + self, + *, + model: str | None = None, + api_key: str | None = None, + base_url: str | None = None, + timeout_sec: float | None = None, + max_attempts: int | None = None, + batch_limit: int | None = None, + ) -> None: + env_key, env_base, env_model = _resolve_config() + self.model = model or env_model + self.dim = DASHSCOPE_DEFAULT_DIM + self.max_attempts = max_attempts or _scene_int("max_attempts", DEFAULT_MAX_ATTEMPTS) + self.batch_limit = batch_limit or _scene_int("batch_limit", DASHSCOPE_BATCH_LIMIT) + timeout = timeout_sec or _scene_float("timeout_sec", 60.0) + self._client = AsyncOpenAI( + api_key=api_key or env_key, + base_url=base_url or env_base, + timeout=timeout, + ) + + async def embed_batch(self, texts: list[str]) -> list[list[float]]: + if not texts: + return [] + results: list[list[float]] = [] + for chunk in _chunked(texts, self.batch_limit): + results.extend(await self._call_with_retry(chunk)) + return results + + async def _call_with_retry(self, batch: list[str]) -> list[list[float]]: + last_err: Exception | None = None + for attempt in range(1, self.max_attempts + 1): + try: + resp = await self._client.embeddings.create( + model=self.model, input=batch + ) + return [d.embedding for d in resp.data] + except Exception as e: # noqa: BLE001 + last_err = e + logger.warning( + "DashScope embed 失败 尝试 {}/{}: {}: {}", + attempt, self.max_attempts, type(e).__name__, e, + ) + if attempt < self.max_attempts: + wait = min(RETRY_BASE_WAIT_SEC * (2 ** (attempt - 1)), RETRY_MAX_WAIT_SEC) + await asyncio.sleep(wait) + raise EmbeddingError( + f"DashScope embed 放弃 {self.max_attempts} 次: {last_err}", + attempts=self.max_attempts, + ) + + async def close(self) -> None: + await self._client.close() diff --git a/extractor/__init__.py b/extractor/__init__.py new file mode 100644 index 0000000..6b4410b --- /dev/null +++ b/extractor/__init__.py @@ -0,0 +1,22 @@ +"""中文新闻正文提取模块 (M2)。 + +公共 API: + - extract_article: 从 HTML 提取 Article + - Article: 统一文章模型 + - ExtractError: 提取失败异常 +""" + +from .models import Article, ExtractError +from .parser import ( + MIN_CONTENT_LENGTH, + SOURCE_NAME_MAP, + extract_article, +) + +__all__ = [ + "MIN_CONTENT_LENGTH", + "SOURCE_NAME_MAP", + "Article", + "ExtractError", + "extract_article", +] diff --git a/extractor/models.py b/extractor/models.py new file mode 100644 index 0000000..458df47 --- /dev/null +++ b/extractor/models.py @@ -0,0 +1,58 @@ +"""正文提取模块的数据模型。 + +核心:Article 是 M2 与下游(去重 / LLM / Embedding)的唯一交换格式。 +""" + +from __future__ import annotations + +from datetime import datetime + +from pydantic import BaseModel, Field + + +class Article(BaseModel): + """提取后的标准化文章。 + + 本模型是 M2 输出 / M3+ 输入的唯一格式。 + """ + + # ---- 标识 ---- + source_id: str = Field(..., description="M1 sources.yaml 中的 source id") + url: str = Field(..., description="文章 URL") + url_hash: str = Field(..., description="URL 的 SHA1 前 16 位,用作主键") + + # ---- 正文字段 ---- + title: str = Field(..., min_length=1, description="文章标题") + content: str = Field(..., min_length=1, description="清理后的正文纯文本") + author: str | None = Field(default=None, description="作者(可选)") + source_name: str | None = Field(default=None, description="网站中文名") + + # ---- 时间 ---- + publish_time: datetime | None = Field( + default=None, description="标准化后的发布时间(本地时区或 naive)" + ) + publish_time_raw: str | None = Field( + default=None, description="原始时间字符串,便于人工核对" + ) + extracted_at: datetime = Field(default_factory=datetime.now, description="提取时刻") + + # ---- 辅助 ---- + images: list[str] = Field(default_factory=list, description="正文中的图片 URL") + word_count: int = Field(default=0, ge=0, description="正文中文字符数") + item_type: str | None = Field( + default=None, description="cninfo 数据类型: announcement|research|irm, 新闻源为 None" + ) + + def short_summary(self) -> str: + """单行摘要,用于日志。""" + t = self.publish_time.strftime("%Y-%m-%d") if self.publish_time else "??" + return f"[{self.source_id}] {t} 《{self.title[:40]}》 {self.word_count}字" + + +class ExtractError(Exception): + """提取失败时抛出。""" + + def __init__(self, reason: str, url: str | None = None) -> None: + super().__init__(reason) + self.reason = reason + self.url = url diff --git a/extractor/parser.py b/extractor/parser.py new file mode 100644 index 0000000..426b88a --- /dev/null +++ b/extractor/parser.py @@ -0,0 +1,367 @@ +"""中文新闻正文提取(M2)。 + +核心流程: + 1. GNE 主提取(title / content / publish_time / author / images); + 2. title 校正:若 GNE 抓到 标签内容明显短于 <h1>,优先 <h1>; + 3. content 清理:剥离 GNE 习惯性附加在正文头部的 title/time/author 行, + 去除连续空行,去重前导空格; + 4. publish_time 标准化为 datetime,加合理性检查与 HTML 中文日期兜底; + 5. 中文字符数统计; + 6. 校验最小长度,过短抛 ExtractError。 +""" + +from __future__ import annotations + +import hashlib +import re +from datetime import datetime, timedelta +from typing import Any + +from bs4 import BeautifulSoup +from dateutil import parser as date_parser +from gne import GeneralNewsExtractor +from loguru import logger + +from .models import Article, ExtractError + +# 全局复用单例,GeneralNewsExtractor 内部加载规则文件,避免重复初始化 +_EXTRACTOR = GeneralNewsExtractor() + +# 验收门槛:正文太短视为提取失败 +MIN_CONTENT_LENGTH = 50 + +# 模板兜底检测:关键词组(必须全部出现) + 长度上限(超过则视为合法长文)。 +# 命中说明 GNE 提取失败,落到了网站固定模板/广告/版权声明文本,实际文章正文未抓到。 +# 长度上限避免真实长篇文章中偶尔提到这些词被误判。 +_BOILERPLATE_PATTERNS: list[tuple[tuple[str, ...], int]] = [ + # eastmoney:页面底部"郑重声明...证券法...东方财富社区管理规定" + (("郑重声明", "证券法"), 800), + # yicai:"第一财经广告合作 ... 著作权归第一财经所有" + (("第一财经广告合作", "著作权"), 800), + # yicai 变体:"未经第一财经书面授权 不得以任何方式加以使用" + (("第一财经", "未经", "书面授权"), 800), + # sina:嵌入式个人专栏推送(同一篇被多个新闻页复用) + (("北京红竹", "跷跷板"), 800), + # 通用版权页兜底 + (("未经", "授权", "禁止转载"), 600), +] + + +def _is_boilerplate(content: str) -> tuple[bool, str | None]: + """检查内容是否为已知模板兜底文本。 + + 返回 (is_boilerplate, matched_pattern_summary)。 + """ + if not content: + return False, None + for keywords, max_len in _BOILERPLATE_PATTERNS: + if len(content) > max_len: + continue + if all(kw in content for kw in keywords): + return True, "+".join(keywords) + return False, None + + +# 站点中文名映射(便于在 Article.source_name 标注) +SOURCE_NAME_MAP = { + "cls": "财联社", + "eastmoney": "东方财富", + "sina": "新浪财经", + "stcn": "证券时报", + "yicai": "第一财经", +} + + +# --------------------------------------------------------------------------- # +# 工具 +# --------------------------------------------------------------------------- # + +def _url_hash(url: str) -> str: + return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16] + + +_CHINESE_CHAR_RE = re.compile(r"[一-鿿]") + + +def _count_chinese(text: str) -> int: + return len(_CHINESE_CHAR_RE.findall(text)) + + +# --------------------------------------------------------------------------- # +# title 校正 +# --------------------------------------------------------------------------- # + +def _refine_title(html: str, gne_title: str) -> str: + """优先使用 <h1>;若无 h1 或 h1 比 gne_title 短,则保留 gne_title。 + + 经验:GNE 偶尔会抓 <title> 标签,而 <title> 常包含站名后缀 + (如 "宁德时代 - 新华网"),H1 通常更纯净。 + """ + soup = BeautifulSoup(html, "html.parser") + h1 = soup.find("h1") + h1_text = (h1.get_text(strip=True) if h1 else "").strip() + + g = (gne_title or "").strip() + + # 两个都为空 -> 失败由调用方处理 + if not h1_text and not g: + return "" + + # 只有一个非空 + if not h1_text: + return g + if not g: + return h1_text + + # H1 是 gne_title 的子串(说明 gne 带了网站后缀),用 H1 + if h1_text in g and len(h1_text) < len(g): + return h1_text + + # gne_title 是 H1 子串,用 H1 + if g in h1_text: + return h1_text + + # 两者差异大,gne_title 更短(可能就是页面 <title> 简称),用 H1 + if len(h1_text) > len(g) * 1.2: + return h1_text + + return g + + +# --------------------------------------------------------------------------- # +# content 清理 +# --------------------------------------------------------------------------- # + +_MULTI_BLANK_RE = re.compile(r"\n{3,}") +_TRAILING_SPACE_RE = re.compile(r"[ \t]+\n") + + +def _clean_content(content: str, title: str, author: str | None, time_raw: str | None) -> str: + """去除 GNE 输出 content 头部混入的 title/time/author 行,以及多余空行。""" + if not content: + return "" + + lines = [ln.rstrip() for ln in content.splitlines()] + # 跳过头部若干行,只要它们与 title/author/time_raw 有明显重合 + drop_targets: list[str] = [s for s in (title, author, time_raw) if s] + drop_targets_norm = {t.strip() for t in drop_targets if t and t.strip()} + + cleaned: list[str] = [] + head_skipping = True + for ln in lines: + s = ln.strip() + if head_skipping: + if not s: + # 头部空行直接跳 + continue + if s in drop_targets_norm: + continue + # 仅由 author 字符串前缀(如 "记者: 张三" vs "张三") + if any(s.startswith(t) or t.startswith(s) for t in drop_targets_norm if len(t) >= 4): + continue + head_skipping = False + cleaned.append(ln) + + # 去尾部空行 + while cleaned and not cleaned[-1].strip(): + cleaned.pop() + + text = "\n".join(cleaned) + text = _TRAILING_SPACE_RE.sub("\n", text) + text = _MULTI_BLANK_RE.sub("\n\n", text) + return text.strip() + + +# --------------------------------------------------------------------------- # +# 时间标准化 +# --------------------------------------------------------------------------- # + +# 中文时间常见模式预清理(GNE 可能给出 "2026年6月15日 14:30") +_CN_DATE_RE = re.compile(r"(\d{4})年(\d{1,2})月(\d{1,2})日") +_CN_TIME_RE = re.compile(r"(\d{1,2})时(\d{1,2})分(?:(\d{1,2})秒)?") + +# 时间合理性边界:超过当前 +1 天为未来时间;早于 -365 天视为页脚等噪声 +_TIME_FUTURE_TOLERANCE = timedelta(days=1) +_TIME_PAST_TOLERANCE = timedelta(days=365) + +# HTML 中文日期兜底正则(在 HTML 全文中找首个看似发布时间的字符串) +# 命中形如 "2026年06月16日 17:54" / "2026年6月16日 17:54:30" / "2026-06-16 17:54" +_HTML_DATE_FALLBACK_RE = re.compile( + r"(\d{4}[-年/]\s*\d{1,2}[-月/]\s*\d{1,2}日?" # 日期 + r"(?:\s+\d{1,2}[:时]\d{1,2}(?:[:分]\d{1,2}秒?)?)?)" # 可选时分秒 +) + + +def _normalize_time(raw: str | None) -> datetime | None: + """把抽到的时间字符串解析为 datetime。失败返回 None。""" + if not raw: + return None + text = raw.strip() + if not text: + return None + + # 拒绝明显不是具体日期的模式(日期范围/部分日期) + if re.search(r"\d{4}年\d{1,2}\s*[-~至到]", text): + return None + + text = _CN_DATE_RE.sub(r"\1-\2-\3", text) + text = _CN_TIME_RE.sub( + lambda m: f"{m.group(1)}:{m.group(2)}" + (f":{m.group(3)}" if m.group(3) else ""), + text, + ) + + # 必须有完整的年月日才解析(过滤只有年月的片段) + if not re.search(r"\d{4}-\d{1,2}-\d{1,2}", text): + return None + + try: + return date_parser.parse(text, fuzzy=True) + except (ValueError, OverflowError) as e: + logger.debug("时间解析失败 raw={!r} err={}", raw, e) + return None + + +def _is_reasonable_time(dt: datetime | None, ref: datetime | None = None) -> bool: + """判断 dt 是否在 ref 附近合理范围内。 + + - 未来超过 1 天 -> 不合理; + - 早于 365 天 -> 不合理(GNE 偶尔抓到页脚备案/版权时间)。 + """ + if dt is None: + return False + ref = ref or datetime.now() + # 同时移除时区信息以便比较(GNE 给出的时间多为 naive) + if dt.tzinfo is not None: + dt = dt.replace(tzinfo=None) + if dt > ref + _TIME_FUTURE_TOLERANCE: + return False + return dt >= ref - _TIME_PAST_TOLERANCE + + +def _fallback_time_from_html(html: str) -> tuple[datetime | None, str | None]: + """从 HTML 全文中正则搜索中文日期模式,返回 (datetime, raw_str)。 + + 搜索顺序对所有匹配做合理性过滤,选择第一个合理时间。这是站点无关的 + 通用兜底,适用于 GNE 误抓页脚备案时间(如 eastmoney 的 2019-01-16)的场景。 + """ + if not html: + return None, None + ref = datetime.now() + for match in _HTML_DATE_FALLBACK_RE.finditer(html): + raw = match.group(1).strip() + dt = _normalize_time(raw) + if dt is not None and _is_reasonable_time(dt, ref): + logger.debug("HTML 兜底时间命中: {!r} -> {}", raw, dt) + return dt, raw + return None, None + + +def _resolve_publish_time( + html: str, gne_time_raw: str | None +) -> tuple[datetime | None, str | None]: + """两阶段时间解析:先 GNE,合理性失败则 HTML 兜底。 + + 返回 (publish_time, publish_time_raw)。 + """ + primary = _normalize_time(gne_time_raw) + if _is_reasonable_time(primary): + return primary, gne_time_raw + + # GNE 时间不可用或不合理,尝试从 HTML 兜底 + fallback_dt, fallback_raw = _fallback_time_from_html(html) + if fallback_dt is not None: + if gne_time_raw and primary is not None: + logger.info( + "GNE 时间 {!r} 与当前差距过大,改用 HTML 兜底 {!r}", gne_time_raw, fallback_raw + ) + return fallback_dt, fallback_raw + + # 都失败,保留 GNE 原始字符串供人工核对 + return None, gne_time_raw + + +# --------------------------------------------------------------------------- # +# 主入口 +# --------------------------------------------------------------------------- # + +def extract_article( + html: str, + source_id: str, + url: str, + *, + extra_noise_xpath: list[str] | None = None, + extra_config: dict[str, str] | None = None, +) -> Article: + """从 HTML 中提取 Article。 + + 参数: + html: 原始 HTML 字符串。 + source_id: M1 sources.yaml 中的 source id。 + url: 文章 URL。 + extra_noise_xpath: 额外的噪声节点 XPath(如评论区/相关阅读容器)。 + + 返回: + Article。 + + 异常: + ExtractError: GNE 提取失败 / 正文过短。 + """ + if not html or not html.strip(): + raise ExtractError("空 HTML", url=url) + + try: + # 从 url 推 host(GNE 用它解析图片相对路径) + from urllib.parse import urlparse + parsed = urlparse(url) + host = f"{parsed.scheme}://{parsed.netloc}" if parsed.scheme else "" + + result: dict[str, Any] = _EXTRACTOR.extract( + html, + host=host, + body_xpath=(extra_config or {}).get("body_xpath", ""), + noise_node_list=extra_noise_xpath or [], + ) + except Exception as e: + raise ExtractError(f"GNE 提取异常: {type(e).__name__}: {e}", url=url) from e + + gne_title = result.get("title", "") or "" + gne_content = result.get("content", "") or "" + gne_time = result.get("publish_time", "") or None + gne_author = (result.get("author", "") or "").strip() or None + images = list(result.get("images", []) or []) + + title = _refine_title(html, gne_title) + if not title: + raise ExtractError("未能提取到标题", url=url) + + content = _clean_content(gne_content, title, gne_author, gne_time) + if len(content) < MIN_CONTENT_LENGTH: + raise ExtractError( + f"正文过短(长度 {len(content)} < {MIN_CONTENT_LENGTH})", + url=url, + ) + + is_bp, bp_reason = _is_boilerplate(content) + if is_bp: + raise ExtractError( + f"内容疑似模板兜底({bp_reason}),长度 {len(content)}", + url=url, + ) + + publish_time, publish_time_raw = _resolve_publish_time(html, gne_time) + + article = Article( + source_id=source_id, + url=url, + url_hash=_url_hash(url), + title=title, + content=content, + author=gne_author, + source_name=SOURCE_NAME_MAP.get(source_id), + publish_time=publish_time, + publish_time_raw=publish_time_raw, + images=images, + word_count=_count_chinese(content), + ) + logger.debug("提取成功: {}", article.short_summary()) + return article diff --git a/llm/__init__.py b/llm/__init__.py new file mode 100644 index 0000000..8ac4347 --- /dev/null +++ b/llm/__init__.py @@ -0,0 +1,64 @@ +"""LLM 投资事件抽取模块 (M4)。 + +公共 API: + - load_llm_config / make_sync_client / make_async_client + - extract_event / extract_event_async + - PromptTemplate / parse_event_json + - EventExtraction / ExtractedEvent / Sentiment / EVENT_TYPES / LLMCallError +""" + +from .client import ( + DEFAULT_MAX_ATTEMPTS, + DEFAULT_TEMPERATURE, + DEFAULT_TIMEOUT_SEC, + SCENE_DAILY_REPORT, + SCENE_EVENT_EXTRACTION, + SCENE_STOCK_REPORT, + LLMConfig, + load_llm_config, + make_async_client, + make_sync_client, +) +from .extractor import ( + DEFAULT_PROMPT_PATH, + MAX_CONTENT_CHARS, + PromptTemplate, + extract_event, + extract_event_async, + parse_event_json, +) +from .models import ( + EVENT_TYPES, + MAX_IMPORTANCE, + MIN_IMPORTANCE, + EventExtraction, + ExtractedEvent, + LLMCallError, + Sentiment, +) + +__all__ = [ + "DEFAULT_MAX_ATTEMPTS", + "DEFAULT_PROMPT_PATH", + "DEFAULT_TEMPERATURE", + "DEFAULT_TIMEOUT_SEC", + "EVENT_TYPES", + "MAX_CONTENT_CHARS", + "MAX_IMPORTANCE", + "MIN_IMPORTANCE", + "SCENE_DAILY_REPORT", + "SCENE_EVENT_EXTRACTION", + "SCENE_STOCK_REPORT", + "EventExtraction", + "ExtractedEvent", + "LLMCallError", + "LLMConfig", + "PromptTemplate", + "Sentiment", + "extract_event", + "extract_event_async", + "load_llm_config", + "make_async_client", + "make_sync_client", + "parse_event_json", +] diff --git a/llm/client.py b/llm/client.py new file mode 100644 index 0000000..7f60ec5 --- /dev/null +++ b/llm/client.py @@ -0,0 +1,208 @@ +"""LLM 客户端抽象与工厂。 + +支持 DeepSeek 和 Qwen(百炼),两者均为 OpenAI 兼容接口,共用 openai SDK。 + +配置来源(优先级从高到低): + 1. 代码 / CLI 显式参数(provider / model) + 2. configs/llm_models.yaml 场景配置(scene 参数,见 configs/loader.py) + 3. 环境变量 / .env(LLM_PROVIDER、DEEPSEEK_MODEL 等,向后兼容) + 4. 代码内置默认值 + +环境变量(兜底): + LLM_PROVIDER = deepseek | qwen (默认 deepseek) + DeepSeek: DEEPSEEK_API_KEY / DEEPSEEK_BASE_URL / DEEPSEEK_MODEL + Qwen: QWEN_API_KEY / QWEN_BASE_URL / QWEN_MODEL + (QWEN_API_KEY -> DASHSCOPE_API_KEY 兜底) + 模型必须显式配置(provider 对应的 *_MODEL 或 LLM_MODEL),不再提供内置默认模型。 + LLM_TEMPERATURE / LLM_TIMEOUT_SEC +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass + +from loguru import logger +from openai import AsyncOpenAI, OpenAI + +from configs.loader import load_defaults, load_scene_config + +# 默认基址 +_DEEPSEEK_DEFAULT_BASE = "https://api.deepseek.com" +_QWEN_DEFAULT_BASE = "https://dashscope.aliyuncs.com/compatible-mode/v1" + +# 抽取任务默认参数 +DEFAULT_TIMEOUT_SEC = 60.0 +DEFAULT_TEMPERATURE = 0.1 +DEFAULT_MAX_ATTEMPTS = 3 + +# 场景名 -> configs/llm_models.yaml 中 scenes 的 key +SCENE_EVENT_EXTRACTION = "event_extraction" +SCENE_DAILY_REPORT = "daily_report" +SCENE_STOCK_REPORT = "stock_report" + + +@dataclass +class LLMConfig: + """LLM 调用配置(provider / model / api_key / base_url / 参数)。""" + + provider: str # "deepseek" / "qwen" + model: str + api_key: str + base_url: str + timeout_sec: float = DEFAULT_TIMEOUT_SEC + temperature: float = DEFAULT_TEMPERATURE + max_attempts: int = DEFAULT_MAX_ATTEMPTS # 单次任务失败重试次数 + + def __post_init__(self) -> None: + if not self.api_key: + raise ValueError(f"LLM provider={self.provider} 的 API key 为空") + + +def _read_env(key: str, default: str | None = None) -> str | None: + val = os.environ.get(key) + if val is None or val.strip() == "": + return default + return val.strip() + + +def _first_env(keys: list[str | None]) -> str | None: + """按顺序返回第一个非空的环境变量值。""" + for k in keys: + if not k: + continue + v = _read_env(k) + if v: + return v + return None + + +def _num(value: object) -> float | None: + """把 YAML 数字/字符串安全转 float;非法或为空返回 None。""" + if value is None or value == "": + return None + try: + return float(value) + except (TypeError, ValueError): + return None + + +def load_llm_config( + provider: str | None = None, + *, + model: str | None = None, + scene: str | None = None, +) -> LLMConfig: + """按优先级构造 LLMConfig:显式参数 > YAML 场景 > 环境变量 > 内置默认。 + + scene 对应 configs/llm_models.yaml 中 scenes 的 key + (event_extraction / daily_report / stock_report),该场景未配置的字段 + 回退到环境变量,保持向后兼容。 + """ + sc = load_scene_config(scene or "") + dflt = load_defaults() + + p = ( + provider + or sc.get("provider") + or _read_env("LLM_PROVIDER", "deepseek") + or "deepseek" + ).lower() + + # 各 provider 的 api_key / base_url / model 环境变量链 + provider_envs: dict[str, tuple[list[str | None], list[str | None], list[str | None]]] = { + "deepseek": ( + [sc.get("api_key_env"), "DEEPSEEK_API_KEY"], + [sc.get("base_url_env"), "DEEPSEEK_BASE_URL"], + ["DEEPSEEK_MODEL", "LLM_MODEL"], + ), + "qwen": ( + [sc.get("api_key_env"), "QWEN_API_KEY", "DASHSCOPE_API_KEY"], + [sc.get("base_url_env"), "QWEN_BASE_URL"], + ["QWEN_MODEL", "LLM_MODEL"], + ), + } + + if p == "deepseek": + key_envs, base_envs, model_envs = provider_envs["deepseek"] + default_base = _DEEPSEEK_DEFAULT_BASE + elif p in ("qwen", "dashscope"): + key_envs, base_envs, model_envs = provider_envs["qwen"] + default_base = _QWEN_DEFAULT_BASE + p = "qwen" # 内部统一用 qwen + else: + raise ValueError(f"未知 LLM provider: {p!r},仅支持 deepseek / qwen") + + api_key = _first_env(key_envs) or "" + base_url = _first_env(base_envs) or default_base + # 模型优先级:显式参数 > YAML 场景 > 环境变量;模型必须显式配置,无内置兜底 + m = model or sc.get("model") or _first_env(model_envs) + if not m: + env_hint = "/".join(v for v in model_envs if v) + raise ValueError( + f"未配置 LLM 模型(场景 {scene or 'default'}): " + f"请在 configs/llm_models.yaml 的 model 或 .env 设置 {env_hint}" + ) + + timeout = _pick_float(sc, dflt, "timeout_sec", "LLM_TIMEOUT_SEC", DEFAULT_TIMEOUT_SEC) + temperature = _pick_float(sc, dflt, "temperature", "LLM_TEMPERATURE", DEFAULT_TEMPERATURE) + max_attempts = _pick_int(sc, "max_attempts", DEFAULT_MAX_ATTEMPTS) + + return LLMConfig( + provider=p, + model=m, + api_key=api_key, + base_url=base_url, + timeout_sec=timeout, + temperature=temperature, + max_attempts=max_attempts, + ) + + +def _pick_float( + sc: dict, + dflt: dict, + sc_key: str, + env_key: str, + default: float, +) -> float: + """数值参数选择:YAML 场景 > 环境变量 > YAML defaults > 内置默认(零值合法)。""" + v = _num(sc.get(sc_key)) + if v is not None: + return v + v = _num(_read_env(env_key)) + if v is not None: + return v + v = _num(dflt.get(sc_key)) + return v if v is not None else default + + +def _pick_int(sc: dict, sc_key: str, default: int) -> int: + v = _num(sc.get(sc_key)) + return int(v) if v is not None else default + + +def make_sync_client(config: LLMConfig) -> OpenAI: + """构造同步 OpenAI 客户端(指向 DeepSeek/Qwen 兼容端点)。""" + logger.debug( + "初始化同步 LLM 客户端: provider={} model={} base_url={}", + config.provider, config.model, config.base_url, + ) + return OpenAI( + api_key=config.api_key, + base_url=config.base_url, + timeout=config.timeout_sec, + ) + + +def make_async_client(config: LLMConfig) -> AsyncOpenAI: + """构造异步 OpenAI 客户端(用于批处理高并发)。""" + logger.debug( + "初始化异步 LLM 客户端: provider={} model={} base_url={}", + config.provider, config.model, config.base_url, + ) + return AsyncOpenAI( + api_key=config.api_key, + base_url=config.base_url, + timeout=config.timeout_sec, + ) diff --git a/llm/extractor.py b/llm/extractor.py new file mode 100644 index 0000000..adc2d12 --- /dev/null +++ b/llm/extractor.py @@ -0,0 +1,307 @@ +"""LLM 投资事件抽取主流程。 + +输入:Article(M2/M3 输出) +输出:ExtractedEvent(含 Pydantic 校验过的 EventExtraction) + +设计: + 1. 加载 prompts/event_extraction.md,字符串替换填入文章字段; + 2. 调用 LLM JSON mode (response_format={"type":"json_object"}); + 3. 解析 JSON -> Pydantic EventExtraction(强校验)+ 重试; + 4. 限制正文长度避免触顶 context window。 +""" + +from __future__ import annotations + +import asyncio +import json +from pathlib import Path +from typing import Any, Protocol + +from loguru import logger +from openai import AsyncOpenAI, OpenAI + +from extractor import Article + +from .client import LLMConfig +from .models import EventExtraction, ExtractedEvent, LLMCallError + +# Prompt 模板默认路径 +DEFAULT_PROMPT_PATH = Path("prompts/event_extraction.md") + +# 文章正文截断长度(防止超出上下文窗口,DeepSeek/Qwen 都支持 32K+,这里保守取 8K 字符) +MAX_CONTENT_CHARS = 8000 + +# 重试设置 +DEFAULT_MAX_ATTEMPTS = 3 +RETRY_BASE_WAIT_SEC = 1.0 +RETRY_MAX_WAIT_SEC = 8.0 + + +# --------------------------------------------------------------------------- # +# Prompt 渲染 +# --------------------------------------------------------------------------- # + +class PromptTemplate: + """Prompt 模板加载器,支持 {placeholder} 字符串替换。""" + + def __init__(self, template_path: str | Path = DEFAULT_PROMPT_PATH) -> None: + self._path = Path(template_path) + self._template = self._path.read_text(encoding="utf-8") + + def render(self, article: Article) -> str: + content = article.content + if len(content) > MAX_CONTENT_CHARS: + logger.debug( + "文章 {} 超长截断: {} -> {}", + article.url_hash, len(content), MAX_CONTENT_CHARS, + ) + content = content[:MAX_CONTENT_CHARS] + "\n\n[正文过长已截断]" + + publish_time_str = ( + article.publish_time.strftime("%Y-%m-%d %H:%M") + if article.publish_time + else "未知" + ) + return ( + self._template + .replace("{title}", article.title) + .replace("{publish_time}", publish_time_str) + .replace("{source_name}", article.source_name or article.source_id) + .replace("{content}", content) + ) + + +# --------------------------------------------------------------------------- # +# JSON 提取(LLM 偶尔会包 ```json 围栏) +# --------------------------------------------------------------------------- # + +def _extract_json_object(text: str) -> str: + """从 LLM 输出中提取首个 JSON 对象字符串(去围栏 / 取首个 {...})。""" + s = text.strip() + if s.startswith("```"): + # 去除 ```json ... ``` 围栏 + s = s.strip("`") + # 可能形如 "json\n{...}" + if s.lower().startswith("json"): + s = s[4:].lstrip("\n").lstrip() + # 末尾可能还有 ``` + if s.endswith("```"): + s = s[:-3] + # 取首个 { 到对应 } + start = s.find("{") + if start < 0: + return s + depth = 0 + for i in range(start, len(s)): + if s[i] == "{": + depth += 1 + elif s[i] == "}": + depth -= 1 + if depth == 0: + return s[start : i + 1] + return s[start:] + + +def parse_event_json(raw: str) -> EventExtraction: + """把 LLM 输出文本解析为 EventExtraction(可能抛 LLMCallError)。""" + payload = _extract_json_object(raw) + try: + obj = json.loads(payload) + except json.JSONDecodeError as e: + raise LLMCallError(f"JSON 解析失败: {e}") from e + if not isinstance(obj, dict): + raise LLMCallError(f"JSON 顶层非对象: {type(obj).__name__}") + try: + return EventExtraction.model_validate(obj) + except Exception as e: # noqa: BLE001 - pydantic ValidationError 等多类型 + raise LLMCallError(f"事件 schema 校验失败: {e}") from e + + +# --------------------------------------------------------------------------- # +# Article -> ExtractedEvent +# --------------------------------------------------------------------------- # + +class _SyncChat(Protocol): + def chat(self, *args: Any, **kwargs: Any) -> Any: ... + + +def _call_llm_sync( + client: OpenAI, + config: LLMConfig, + prompt: str, +) -> tuple[str, dict[str, int | None]]: + """同步单次 LLM 调用,返回 (raw_text, usage)。usage 含 prompt_tokens / completion_tokens。""" + resp = client.chat.completions.create( + model=config.model, + messages=[ + { + "role": "system", + "content": "你是 A 股投资研究助手,严格按用户指定的 JSON 格式输出。", + }, + {"role": "user", "content": prompt}, + ], + temperature=config.temperature, + response_format={"type": "json_object"}, + ) + text = resp.choices[0].message.content or "" + usage = { + "prompt_tokens": getattr(resp.usage, "prompt_tokens", None) if resp.usage else None, + "completion_tokens": ( + getattr(resp.usage, "completion_tokens", None) if resp.usage else None + ), + } + return text, usage + + +async def _call_llm_async( + client: AsyncOpenAI, + config: LLMConfig, + prompt: str, +) -> tuple[str, dict[str, int | None]]: + """异步单次 LLM 调用。""" + resp = await client.chat.completions.create( + model=config.model, + messages=[ + { + "role": "system", + "content": "你是 A 股投资研究助手,严格按用户指定的 JSON 格式输出。", + }, + {"role": "user", "content": prompt}, + ], + temperature=config.temperature, + response_format={"type": "json_object"}, + ) + text = resp.choices[0].message.content or "" + usage = { + "prompt_tokens": getattr(resp.usage, "prompt_tokens", None) if resp.usage else None, + "completion_tokens": ( + getattr(resp.usage, "completion_tokens", None) if resp.usage else None + ), + } + return text, usage + + +def extract_event( + client: OpenAI, + config: LLMConfig, + article: Article, + *, + template: PromptTemplate | None = None, + max_attempts: int | None = None, + sources: list[str] | None = None, +) -> ExtractedEvent: + """同步抽取单篇文章的事件(带重试)。 + + max_attempts 为 None 时使用 config.max_attempts(来自 YAML/环境变量配置)。 + sources 为该新闻全部来源(来自去重层多源记录);None 时兜底 [article.source_id]。 + """ + tpl = template or PromptTemplate() + prompt = tpl.render(article) + max_attempts = max_attempts or config.max_attempts + + last_err: Exception | None = None + for attempt in range(1, max_attempts + 1): + try: + raw, usage = _call_llm_sync(client, config, prompt) + event = parse_event_json(raw) + return ExtractedEvent( + source_id=article.source_id, + url=article.url, + url_hash=article.url_hash, + title=article.title, + publish_time=article.publish_time, + sources=sources or [article.source_id], + event=event, + provider=config.provider, + model=config.model, + attempts=attempt, + prompt_tokens=usage.get("prompt_tokens"), + completion_tokens=usage.get("completion_tokens"), + ) + except LLMCallError as e: + last_err = e + logger.warning( + "LLM 抽取失败 url={} 尝试 {}/{}: {}", + article.url, attempt, max_attempts, e.reason, + ) + except Exception as e: # noqa: BLE001 - 网络/限流等 + last_err = e + logger.warning( + "LLM 调用异常 url={} 尝试 {}/{}: {}: {}", + article.url, attempt, max_attempts, type(e).__name__, e, + ) + if attempt < max_attempts: + wait = min(RETRY_BASE_WAIT_SEC * (2 ** (attempt - 1)), RETRY_MAX_WAIT_SEC) + import time + + time.sleep(wait) + + raise LLMCallError( + f"LLM 抽取放弃,共 {max_attempts} 次尝试: {last_err}", + attempts=max_attempts, + ) + + +async def extract_event_async( + client: AsyncOpenAI, + config: LLMConfig, + article: Article, + *, + template: PromptTemplate | None = None, + max_attempts: int | None = None, + sources: list[str] | None = None, + semaphore: asyncio.Semaphore | None = None, +) -> ExtractedEvent: + """异步抽取(批处理用),与同步版逻辑等价。 + + max_attempts 为 None 时使用 config.max_attempts。 + sources 为该新闻全部来源;None 时兜底 [article.source_id]。 + """ + tpl = template or PromptTemplate() + prompt = tpl.render(article) + max_attempts = max_attempts or config.max_attempts + + async def _run() -> ExtractedEvent: + last_err: Exception | None = None + for attempt in range(1, max_attempts + 1): + try: + raw, usage = await _call_llm_async(client, config, prompt) + event = parse_event_json(raw) + return ExtractedEvent( + source_id=article.source_id, + url=article.url, + url_hash=article.url_hash, + title=article.title, + publish_time=article.publish_time, + sources=sources or [article.source_id], + event=event, + provider=config.provider, + model=config.model, + attempts=attempt, + prompt_tokens=usage.get("prompt_tokens"), + completion_tokens=usage.get("completion_tokens"), + ) + except LLMCallError as e: + last_err = e + logger.warning( + "LLM 抽取失败 url={} 尝试 {}/{}: {}", + article.url, attempt, max_attempts, e.reason, + ) + except Exception as e: # noqa: BLE001 + last_err = e + logger.warning( + "LLM 调用异常 url={} 尝试 {}/{}: {}: {}", + article.url, attempt, max_attempts, type(e).__name__, e, + ) + if attempt < max_attempts: + wait = min(RETRY_BASE_WAIT_SEC * (2 ** (attempt - 1)), RETRY_MAX_WAIT_SEC) + await asyncio.sleep(wait) + raise LLMCallError( + f"LLM 抽取放弃,共 {max_attempts} 次尝试: {last_err}", + attempts=max_attempts, + ) + + if semaphore is None: + return await _run() + async with semaphore: + return await _run() diff --git a/llm/models.py b/llm/models.py new file mode 100644 index 0000000..0518287 --- /dev/null +++ b/llm/models.py @@ -0,0 +1,188 @@ +"""LLM 投资事件抽取的数据模型 (M4)。 + +EventExtraction 是 LLM 严格输出 schema(JSON mode 解析后用 Pydantic 校验)。 +ExtractedEvent 把 EventExtraction 与原文章元数据合并,作为 M4 最终落盘格式。 +""" + +from __future__ import annotations + +import re +from datetime import datetime +from enum import StrEnum +from typing import Self + +from pydantic import BaseModel, Field, field_validator, model_validator + + +class Sentiment(StrEnum): + """事件情绪倾向。""" + + POSITIVE = "positive" # 利好 + NEUTRAL = "neutral" # 中性 + NEGATIVE = "negative" # 利空 + + +# 事件类型枚举(Prompt 中也会展示给 LLM) +EVENT_TYPES: tuple[str, ...] = ( + "业绩预告", + "业绩快报", + "财报披露", + "合作签约", + "投资并购", + "重大合同", + "产品发布", + "技术突破", + "监管处罚", + "诉讼仲裁", + "股东减持", + "股东增持", + "回购", + "分红", + "高管变动", + "资产重组", + "停牌复牌", + "ST警示", + "退市风险", + "宏观政策", + "行业政策", + "国际局势", + "其他", +) + +# A 股股票代码:6 位数字(000xxx/300xxx/600xxx 等),也可带 .SH/.SZ/.BJ 后缀 +_STOCK_CODE_RE = re.compile(r"^\d{6}(\.(SH|SZ|BJ))?$") + +# 重要程度合理区间(LLM 偶尔会给 0/6/10,这里夹紧) +MIN_IMPORTANCE = 1 +MAX_IMPORTANCE = 5 + + +class EventExtraction(BaseModel): + """LLM 输出的 JSON 直接映射到此模型。""" + + stock_codes: list[str] = Field( + default_factory=list, + description="A 股 6 位代码,允许带 .SH/.SZ/.BJ 后缀;无相关股票时为空", + ) + company_names: list[str] = Field( + default_factory=list, description="涉及公司中文简称,无关时为空" + ) + industries: list[str] = Field( + default_factory=list, description="所属行业(申万二级粒度优先);无关时为空" + ) + sentiment: Sentiment = Field(..., description="positive/neutral/negative") + importance: int = Field( + ..., ge=MIN_IMPORTANCE, le=MAX_IMPORTANCE, description="1-5 重要程度" + ) + event_type: str = Field(..., description="事件类型,见 EVENT_TYPES") + summary: str = Field( + default="", + max_length=200, + description="一句话事件摘要(≤ 100 字),便于人工浏览", + ) + + @field_validator("stock_codes") + @classmethod + def _strip_and_validate_stock_codes(cls, v: list[str]) -> list[str]: + """剔除空字符串、统一大写、过滤明显非法格式。""" + cleaned: list[str] = [] + for code in v: + s = (code or "").strip().upper().replace(" ", "") + if not s: + continue + if _STOCK_CODE_RE.match(s): + cleaned.append(s) + # 去重保持顺序 + seen: set[str] = set() + out: list[str] = [] + for c in cleaned: + if c not in seen: + seen.add(c) + out.append(c) + return out + + @field_validator("company_names", "industries") + @classmethod + def _strip_text_lists(cls, v: list[str]) -> list[str]: + cleaned = [(s or "").strip() for s in v] + cleaned = [s for s in cleaned if s] + seen: set[str] = set() + out: list[str] = [] + for c in cleaned: + if c not in seen: + seen.add(c) + out.append(c) + return out + + @field_validator("event_type") + @classmethod + def _normalize_event_type(cls, v: str) -> str: + s = (v or "").strip() + if not s: + return "其他" + return s + + @model_validator(mode="after") + def _post_check(self) -> Self: + """中性情绪时 importance 应较低(1-3),纠正常见误判。""" + # 不强制纠正,留作后续校准。占位,便于以后扩展。 + return self + + +class ExtractedEvent(BaseModel): + """落盘格式:文章元数据 + LLM 抽取结果 + 调用元信息。 + + sources: 该唯一新闻的全部来源(主源 source_id 居首)。 + 来自去重层多源记录(M3 sources.json / uniques JSON 的 sources 字段), + 旧产物无此字段时兜底为 [source_id]。 + """ + + # ---- 来源标识 ---- + source_id: str + url: str + url_hash: str + title: str + publish_time: datetime | None = None + sources: list[str] = Field( + default_factory=list, + description="全部来源(主源居首,去重保序);旧产物无字段时兜底为 [source_id]", + ) + + # ---- 抽取结果 ---- + event: EventExtraction + + # ---- 调用元信息 ---- + provider: str = Field(..., description="deepseek / qwen 等") + model: str + extracted_at: datetime = Field(default_factory=datetime.now) + attempts: int = Field(default=1, ge=1, description="LLM 实际调用次数(含重试)") + prompt_tokens: int | None = None + completion_tokens: int | None = None + + @model_validator(mode="after") + def _ensure_sources(self) -> Self: + """保证 sources 非空、去重且以主源 source_id 开头。""" + seen: list[str] = [] + for s in [self.source_id, *self.sources]: + if s and s not in seen: + seen.append(s) + self.sources = seen + return self + + def short_summary(self) -> str: + ev = self.event + codes = ",".join(ev.stock_codes) or "-" + return ( + f"[{self.source_id}] {self.title[:30]} " + f"-> {ev.sentiment.value}/{ev.importance}/{ev.event_type} " + f"({codes})" + ) + + +class LLMCallError(Exception): + """LLM 调用失败(网络 / 解析 / 校验)。""" + + def __init__(self, reason: str, *, attempts: int = 0) -> None: + super().__init__(reason) + self.reason = reason + self.attempts = attempts diff --git a/mcp_server/__init__.py b/mcp_server/__init__.py new file mode 100644 index 0000000..cb879da --- /dev/null +++ b/mcp_server/__init__.py @@ -0,0 +1,4 @@ +"""MCP 服务模块 (M8)。 + +暴露 5 个 MCP 工具给 Cherry Studio / Claude Code 调用。 +""" diff --git a/mcp_server/tools.py b/mcp_server/tools.py new file mode 100644 index 0000000..b61579e --- /dev/null +++ b/mcp_server/tools.py @@ -0,0 +1,250 @@ +"""MCP 工具实现。 + +每个工具:接收自然语言查询 → DashScope 嵌入 → Qdrant 检索 → 格式化返回。 +嵌入 provider 复用 M5,检索复用 M6。 +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any + +from dotenv import load_dotenv +from loguru import logger +from mcp.server.fastmcp import FastMCP + +from embedding import make_sync_provider +from vectorstore import SearchFilter, VectorStore, make_qdrant_client + +# 加载 .env(API key 等) +load_dotenv() + +# --------------------------------------------------------------------------- # +# 单例(模块加载时初始化,所有工具共用) +# --------------------------------------------------------------------------- # + +@dataclass +class _Backend: + embedder: Any # EmbeddingProvider + vector_store: VectorStore + +_backend: _Backend | None = None + + +def _get_backend() -> _Backend: + global _backend + if _backend is None: + emb = make_sync_provider() # 读取 EMBEDDING_PROVIDER 环境变量 + logger.info("MCP embedder 就绪: dim={}", emb.dim) + client = make_qdrant_client() + store = VectorStore(client) + logger.info("MCP vector_store 就绪: count={}", store.count()) + _backend = _Backend(embedder=emb, vector_store=store) + return _backend + + +# --------------------------------------------------------------------------- # +# 嵌入 + 检索 +# --------------------------------------------------------------------------- # + +def _search( + query: str, + top_k: int = 10, + filter: SearchFilter | None = None, +) -> list[dict[str, Any]]: + """嵌入查询文本 → Qdrant 检索 → 返回 dict 列表。""" + be = _get_backend() + vec = be.embedder.embed_one(query) + results = be.vector_store.query( + query_vector=vec, top_k=top_k, filter=filter, score_threshold=0.3, + ) + return [ + { + "title": r.title, + "url": r.url, + "source": r.source_id, + "score": round(r.score, 4), + "publish_time": r.publish_time.isoformat() if r.publish_time else None, + "event": { + "sentiment": (r.event or {}).get("sentiment"), + "importance": (r.event or {}).get("importance"), + "event_type": (r.event or {}).get("event_type"), + "stock_codes": (r.event or {}).get("stock_codes", []), + "company_names": (r.event or {}).get("company_names", []), + "industries": (r.event or {}).get("industries", []), + "summary": (r.event or {}).get("summary"), + }, + } + for r in results + ] + + +# --------------------------------------------------------------------------- # +# MCP 服务器 & 工具 +# --------------------------------------------------------------------------- # + +mcp = FastMCP( + name="A股DeepResearch", + instructions="A 股 Deep Research 知识库——语义检索财经新闻与投资事件。", +) + + +def _fmt_results(hits: list[dict[str, Any]], query: str) -> str: + """把检索结果格式化为 Markdown 文本。""" + if not hits: + return f"未找到与「{query}」相关的结果。" + lines = [f"# 检索结果: {query}", "", f"共 {len(hits)} 条:", ""] + for i, h in enumerate(hits, 1): + ev = h["event"] + sentiment = {"positive": "🟢利好", "neutral": "⚪中性", "negative": "🔴利空"}.get( + ev.get("sentiment"), "" + ) + lines.append(f"### {i}. {h['title']}") + lines.append(f"- 来源: {h['source']} | 相似度: {h['score']} | {sentiment}") + lines.append(f"- 时间: {h['publish_time'] or '未知'}") + if ev.get("company_names"): + lines.append(f"- 公司: {', '.join(ev['company_names'][:5])}") + if ev.get("stock_codes"): + lines.append(f"- 代码: {', '.join(ev['stock_codes'][:5])}") + if ev.get("industries"): + lines.append(f"- 行业: {', '.join(ev['industries'][:3])}") + if ev.get("summary"): + lines.append(f"- 摘要: {ev['summary']}") + lines.append(f"- URL: {h['url']}") + lines.append("") + return "\n".join(lines) + + +# --------------------------------------------------------------------------- # +# Tool 1: search_news +# --------------------------------------------------------------------------- # + +@mcp.tool() +def search_news(query: str, top_k: int = 10) -> str: + """语义检索 A 股财经新闻知识库。 + + 参数: + query: 自然语言查询(如 "宁德时代最新动态" "AI 行业政策") + top_k: 返回条数(默认 10) + + 返回: Markdown 格式的检索结果,含标题、来源、相似度、URL。 + """ + logger.info("search_news query={!r} top_k={}", query, top_k) + hits = _search(query, top_k=top_k) + return _fmt_results(hits, query) + + +# --------------------------------------------------------------------------- # +# Tool 2: search_company_news +# --------------------------------------------------------------------------- # + +@mcp.tool() +def search_company_news(query: str, company: str, top_k: int = 10) -> str: + """检索指定公司的相关新闻。 + + 参数: + query: 自然语言查询 + company: 公司名称(如 "宁德时代" "贵州茅台") + top_k: 返回条数(默认 10) + + 返回: Markdown 格式的检索结果。 + """ + logger.info("search_company_news query={!r} company={!r}", query, company) + hits = _search( + query, top_k=top_k, + filter=SearchFilter(company_names=[company]), + ) + if not hits: + # 降级:无精确命中时做纯语义搜索 + logger.info("company 精确命中 0 条,降级为纯语义搜索") + hits = _search(query, top_k=top_k) + return _fmt_results(hits, f"{query} (公司:{company})") + + +# --------------------------------------------------------------------------- # +# Tool 3: search_industry_news +# --------------------------------------------------------------------------- # + +@mcp.tool() +def search_industry_news(query: str, industry: str, top_k: int = 10) -> str: + """检索指定行业的相关新闻。 + + 参数: + query: 自然语言查询 + industry: 行业名(如 "动力电池" "白酒" "半导体") + top_k: 返回条数(默认 10) + + 返回: Markdown 格式的检索结果。 + """ + logger.info("search_industry_news query={!r} industry={!r}", query, industry) + hits = _search( + query, top_k=top_k, + filter=SearchFilter(industries=[industry]), + ) + if not hits: + logger.info("industry 精确命中 0 条,降级为纯语义搜索") + hits = _search(query, top_k=top_k) + return _fmt_results(hits, f"{query} (行业:{industry})") + + +# --------------------------------------------------------------------------- # +# Tool 4: search_stock_events +# --------------------------------------------------------------------------- # + +@mcp.tool() +def search_stock_events(query: str, stock_code: str, top_k: int = 10) -> str: + """检索指定股票代码相关的投资事件。 + + 参数: + query: 自然语言查询 + stock_code: 6 位 A 股代码(如 "300750" "000001",可带 .SH/.SZ 后缀) + top_k: 返回条数(默认 10) + + 返回: Markdown 格式的检索结果。 + """ + # 标准化:去掉后缀,补一致性 + code = stock_code.strip().split(".")[0].upper() + logger.info("search_stock_events query={!r} code={!r}", query, code) + hits = _search( + query, top_k=top_k, + filter=SearchFilter(stock_codes=[code]), + ) + if not hits: + logger.info("stock_code 精确命中 0 条,降级为纯语义搜索") + hits = _search(query, top_k=top_k) + return _fmt_results(hits, f"{query} (代码:{code})") + + +# --------------------------------------------------------------------------- # +# Tool 5: search_sentiment_trend +# --------------------------------------------------------------------------- # + +@mcp.tool() +def search_sentiment_trend(query: str, sentiment: str = "all", top_k: int = 20) -> str: + """检索并统计特定情绪倾向的新闻。 + + 参数: + query: 自然语言查询 + sentiment: positive(利好) / negative(利空) / neutral(中性) / all(全部,默认) + top_k: 返回条数(默认 20) + + 返回: Markdown 格式的检索结果 + 情绪分布统计。 + """ + logger.info("search_sentiment_trend query={!r} sentiment={!r}", query, sentiment) + filt = None + if sentiment in ("positive", "negative", "neutral"): + filt = SearchFilter(sentiment=sentiment) + + hits = _search(query, top_k=top_k, filter=filt) + + # 统计情绪分布 + pos = sum(1 for h in hits if h["event"].get("sentiment") == "positive") + neg = sum(1 for h in hits if h["event"].get("sentiment") == "negative") + neu = sum(1 for h in hits if h["event"].get("sentiment") == "neutral") + + header = ( + f"# 情绪趋势: {query}\n\n" + f"共 {len(hits)} 条 | " + f"🟢利好 {pos} | 🔴利空 {neg} | ⚪中性 {neu}\n" + ) + return header + "\n" + _fmt_results(hits, query) diff --git a/prompts/company_analysis.md b/prompts/company_analysis.md new file mode 100644 index 0000000..a51b7f2 --- /dev/null +++ b/prompts/company_analysis.md @@ -0,0 +1,64 @@ +# 公司深度分析 Prompt + +你是 A 股投研分析师。基于知识库中的新闻事件,对指定公司进行全面深度分析。 + +--- + +## 分析流程 + +1. 使用 `search_company_news` 检索该公司所有相关新闻和事件; +2. 使用 `search_stock_events` 检索该公司股票代码相关事件; +3. 使用 `search_sentiment_trend` 分析情绪变化趋势; +4. 综合以上信息,输出结构化研究报告。 + +--- + +## 报告结构 + +### 一、公司概况 +- 行业定位与核心业务 +- 近期(近 30 天)重大事件概览 + +### 二、利好因素梳理 +- 按重要性排序(从 event.importance 取值) +- 逐条说明事件内容、影响逻辑、市场反应 +- 标注是否为持续性利好(如行业政策)还是一次性事件(如单笔合同) + +### 三、利空因素梳理 +- 按重要性排序 +- 逐条说明事件内容、风险逻辑 +- 评估利空是否已充分消化 + +### 四、情绪趋势分析 +- 近 30 天情绪变化曲线(positive / neutral / negative 比例) +- 是否存在"情绪拐点"(如由正转负、长期低迷后首次转正) +- 情绪与股价/事件的对应关系 + +### 五、关键时间线 +- 按时间顺序列出重大事件 +- 标注事件类型(event_type)、影响程度(importance) + +### 六、投资逻辑总结 +- 核心理由(bull case) +- 主要风险(bear case) +- 当前观点倾向(偏多/中性/偏空)及理由 + +--- + +## 约束 + +1. 所有事实必须来自检索到的新闻,不得编造; +2. 股票代码以 A 股 6 位数字呈现(如 300750); +3. 涉及具体股价/涨跌幅时注明来源; +4. 不确定的推断标注"待验证"; +5. 语言简洁,避免空洞套话; +6. 最终输出为 Markdown 格式,分节清晰。 + +--- + +## 示例分析对象 + +向 Agent 提问时指定: +- 公司名(如"宁德时代") +- 股票代码(如"300750") +- 分析时间范围(默认近 30 天) diff --git a/prompts/event_extraction.md b/prompts/event_extraction.md new file mode 100644 index 0000000..b78d018 --- /dev/null +++ b/prompts/event_extraction.md @@ -0,0 +1,110 @@ +# 投资事件抽取 Prompt + +你是 A 股投资研究助手。基于以下中文财经新闻,抽取**结构化投资事件**信息,严格按 JSON Schema 返回。 + +--- + +## 任务 + +阅读「文章原文」,抽取并填写下列字段;**不允许编造文章未提及的内容**。无法判断的字段按"约束"中的默认值处理。 + +--- + +## JSON Schema + +返回**纯 JSON 对象**(不要包裹 ```json 围栏,不要附加任何前后说明): + +``` +{ + "stock_codes": list[str], // A 股 6 位代码(可带 .SH/.SZ/.BJ 后缀);无明确个股则空 [] + "company_names": list[str], // 涉及公司中文简称(去除"股份有限公司"等后缀) + "industries": list[str], // 所属行业(申万二级粒度优先,如"动力电池"/"光伏设备") + "sentiment": "positive" | "neutral" | "negative", // 利好 / 中性 / 利空 + "importance": 1..5, // 1=琐碎,3=普通,5=重大 + "event_type": str, // 见下方"事件类型"枚举,不在内则用"其他" + "summary": str // ≤ 100 字的中文事件摘要 +} +``` + +--- + +## 事件类型枚举 + +业绩预告 / 业绩快报 / 财报披露 / 合作签约 / 投资并购 / 重大合同 / 产品发布 / 技术突破 / +监管处罚 / 诉讼仲裁 / 股东减持 / 股东增持 / 回购 / 分红 / 高管变动 / 资产重组 / +停牌复牌 / ST警示 / 退市风险 / 宏观政策 / 行业政策 / 国际局势 / 其他 + +--- + +## 约束 + +1. **stock_codes**:必须是 6 位数字;若文章只点名公司未给代码,**不要凭记忆补全**,留空; +2. **company_names**:仅写明确出现在原文的公司;不输出"宁德时代股份有限公司"这种全称,用"宁德时代"; +3. **industries**:1~3 个最相关的行业;若文章是宏观/政策类,可用"宏观"/"行业政策"等; +4. **sentiment**:严格按文章对涉及个股/行业的影响判断;**纯客观信息描述用 neutral**; +5. **importance**:5 仅用于会显著移动股价的重大事件(政策、并购、ST、重大业绩超预期);常规公告 2~3;琐碎信息 1; +6. **summary**:中文,主谓宾完整,不引用原文长句; +7. **不输出文章未涉及的字段**;**不附加 markdown 围栏 / 注释 / 解释**; +8. 如果文章与 A 股无关(如纯美股、宏观国际),`stock_codes` / `company_names` / `industries` 可空,`sentiment=neutral` `importance=1` `event_type="国际局势"` 或合适的类别。 + +--- + +## 示例 + +### 示例 1 - 合作签约,利好 + +输入: + +> 标题:宁德时代与某车企签署 100GWh 长期供货协议 +> 正文:6 月 15 日,宁德时代(300750.SZ)发布公告,与某新能源车企签署了为期五年的电池供货协议, +> 累计供货约 100GWh,预计带来超过 1500 亿元营收。分析师认为这将进一步巩固公司全球龙头地位。 + +输出: + +``` +{"stock_codes":["300750.SZ"],"company_names":["宁德时代"],"industries":["动力电池","新能源汽车"],"sentiment":"positive","importance":5,"event_type":"重大合同","summary":"宁德时代签 5 年 100GWh 供货协议,涉及金额超 1500 亿元,巩固龙头地位。"} +``` + +### 示例 2 - 监管处罚,利空 + +输入: + +> 标题:皇台酒业实际控制人涉嫌信息披露违法被立案 +> 正文:皇台酒业公告,公司实际控制人赵满堂因涉嫌信息披露违法违规被证监会立案调查。 +> 公司回应称此事项与生产经营无直接关联。 + +输出: + +``` +{"stock_codes":[],"company_names":["皇台酒业"],"industries":["白酒"],"sentiment":"negative","importance":3,"event_type":"监管处罚","summary":"皇台酒业实控人因涉嫌信披违规被证监会立案,公司称与经营无关。"} +``` + +### 示例 3 - 国际,中性 + +输入: + +> 标题:特朗普:美国将把重心转向俄乌问题 +> 正文:特朗普在记者会上表示,美方将把外交工作重心转回俄乌冲突调解……(无 A 股具体涉及) + +输出: + +``` +{"stock_codes":[],"company_names":[],"industries":["宏观"],"sentiment":"neutral","importance":2,"event_type":"国际局势","summary":"特朗普称美方外交重心将转向俄乌调解,无 A 股直接关联。"} +``` + +--- + +## 文章原文 + +标题:{title} + +发布时间:{publish_time} + +来源:{source_name} + +正文: +{content} + +--- + +仅输出 JSON 对象,不要附加任何其它文字: diff --git a/prompts/industry_analysis.md b/prompts/industry_analysis.md new file mode 100644 index 0000000..8b7198e --- /dev/null +++ b/prompts/industry_analysis.md @@ -0,0 +1,65 @@ +# 行业分析 Prompt + +你是 A 股行业研究员。基于知识库中的新闻事件,对指定行业进行全面分析。 + +--- + +## 分析流程 + +1. 使用 `search_industry_news` 检索该行业相关新闻; +2. 使用 `search_news` 以行业关键词做广义语义搜索(如"动力电池 产能""光伏 政策"); +3. 使用 `search_sentiment_trend` 分析行业整体情绪; +4. 综合以上信息,输出结构化行业研究报告。 + +--- + +## 报告结构 + +### 一、行业概览 +- 行业定义与细分赛道(申万二级优先) +- 近期(近 30 天)关键事件数量与类型分布 +- 行业整体情绪倾向(偏乐观/中性/偏悲观) + +### 二、政策环境 +- 国家/地方层面相关新政策 +- 政策方向判断(鼓励/规范/限制) +- 政策影响范围与时间窗口 + +### 三、供需与技术动态 +- 产能、出货量、价格等基本面变化 +- 重大技术突破或产品发布 +- 上游原材料/设备动态(若有) + +### 四、重要公司动态 +- 行业龙头近况(检索前 3-5 家相关公司) +- 竞争格局变化(新进入者/并购/退出) + +### 五、资金面与市场表现 +- 板块资金流向(若新闻涉及) +- 券商观点汇总(研报/评级调整) +- 机构关注度变化 + +### 六、行业展望 +- 未来 1-3 个月关键催化剂 +- 主要风险因素 +- 行业配置建议(超配/标配/低配)及理由 + +--- + +## 约束 + +1. 所有事实必须来自检索到的新闻,不得编造; +2. 行业分类优先使用申万二级名称; +3. 涉及具体股票时列出代码; +4. 不确定的推断标注"待验证"; +5. 最终输出为 Markdown 格式,分节清晰。 + +--- + +## 示例分析对象 + +- "动力电池" +- "光伏设备" +- "半导体" +- "白酒" +- "人工智能" diff --git a/prompts/risk_analysis.md b/prompts/risk_analysis.md new file mode 100644 index 0000000..99f76f9 --- /dev/null +++ b/prompts/risk_analysis.md @@ -0,0 +1,62 @@ +# 风险识别与分析 Prompt + +你是 A 股风控分析师。基于知识库中的新闻事件,识别和评估相关风险。 + +--- + +## 分析流程 + +1. 根据分析对象(公司/行业/主题),使用相应的 MCP 工具检索近期新闻; +2. 重点筛选 `sentiment=negative` 和 `event_type` 为"监管处罚/诉讼仲裁/ST警示/退市风险/股东减持"等类型的事件; +3. 对每条风险事件评估影响程度和概率; +4. 输出结构化风险报告。 + +--- + +## 报告结构 + +### 一、风险总览 +- 分析对象及范围 +- 风险事件总数及类型分布 +- 综合风险等级(高/中/低)及判定依据 + +### 二、监管与合规风险 +- 证监会/交易所处罚、问询函、立案调查 +- 信息披露违规 +- ST/退市风险警示 + +### 三、经营与财务风险 +- 业绩大幅变脸(预减/首亏) +- 大客户/大订单流失 +- 原材料成本剧烈波动 +- 产能过剩/库存积压 + +### 四、管理层与治理风险 +- 高管变动/减持 +- 股权质押比例过高 +- 关联交易/资金占用 + +### 五、市场与竞争风险 +- 行业政策不利变化 +- 竞争格局恶化(价格战/替代品) +- 国际市场风险(贸易摩擦/制裁) + +### 六、风险矩阵 +| 风险事件 | 类型 | 影响程度(1-5) | 发生概率 | 应对建议 | +| --- | --- | --- | --- | --- | +| ... | ... | ... | 高/中/低 | ... | + +### 七、总结与建议 +- 最需要关注的 Top 3 风险 +- 风险是否已被市场定价(股价是否已反映) +- 后续监控要点 + +--- + +## 约束 + +1. 风险判定须有新闻事实支撑; +2. 区分"已发生"与"潜在"风险; +3. 影响程度 1-5:1=影响有限,5=可能致命; +4. 最终输出为 Markdown 格式; +5. 不确定的推断标注"待验证"。 diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..04b933d --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,101 @@ +[project] +name = "a-share-research" +version = "0.1.0" +description = "A 股 Deep Research 私有投研平台" +readme = "README.md" +requires-python = ">=3.11,<3.12" +authors = [ + { name = "summer" } +] +license = { text = "Proprietary" } + +# 注意:依赖按 Milestone 分阶段加入,避免一次性引入。 +# 当前已启用:M1 抓取层 +dependencies = [ + "pydantic>=2.7", + "pyyaml>=6.0", + "python-dotenv>=1.0", + "loguru>=0.7", + # ---- M1 抓取层 ---- + "crawl4ai>=0.6,<0.8", + "tenacity>=9.0", + "beautifulsoup4>=4.12", + # ---- M2 正文提取 ---- + "gne>=0.3", + "python-dateutil>=2.9", + "lxml>=5.0", + # ---- M4 LLM 投资事件抽取 ---- + "openai>=1.40", + "httpx>=0.27", + # ---- M5 Embedding(远程后端必装,本地后端走 optional)---- + "numpy>=1.26", + # ---- M6 Qdrant 向量库 ---- + "qdrant-client>=1.10", + # ---- M7 定时任务 ---- + "apscheduler>=3.10", + # ---- M8 MCP 服务 ---- + "mcp>=1.0", + "pypdf>=6.13.3", + "markitdown[all]>=0.1.5", + "pymysql>=1.2.0", +] + +[project.optional-dependencies] +# 本地 BGE-M3 后端:首次下载约 2.3 GB,按需安装 +# uv sync --extra local-embedding +local-embedding = [ + "sentence-transformers>=3.0", + "torch>=2.2", +] +# 占位,后续 Milestone 按需启用 +# mcp = ["mcp>=1.0"] + +[dependency-groups] +dev = [ + "pytest>=8.0", + "pytest-asyncio>=0.23", + "pytest-cov>=5.0", + "ruff>=0.5", + "mypy>=1.10", +] + +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[tool.hatch.build.targets.wheel] +packages = [ + "crawler", + "extractor", + "dedup", + "llm", + "embedding", + "vectorstore", + "scheduler", + "mcp_server", + "api", +] + +[project.scripts] +a-share = "a_share_cli.main:main" + +[tool.ruff] +line-length = 100 +target-version = "py311" + +[tool.ruff.lint] +select = ["E", "F", "W", "I", "N", "UP", "B", "C4", "SIM"] +ignore = ["E501"] + +[tool.pytest.ini_options] +testpaths = ["tests"] +asyncio_mode = "auto" +addopts = "-ra -q" +markers = [ + "integration: 需要真实网络/浏览器的集成测试,默认跳过", +] + +[tool.mypy] +python_version = "3.11" +strict = false +ignore_missing_imports = true diff --git a/report_db/__init__.py b/report_db/__init__.py new file mode 100644 index 0000000..a4084d5 --- /dev/null +++ b/report_db/__init__.py @@ -0,0 +1,19 @@ +"""日报结构化入库(Milestone 10)。 + +职责:日报内容(finance/intl)结构化后写入 MySQL(news_report / news_event), +供用户另行实现的 API/前端读取。本项目不提供 API 与前端。 +""" + +from .db import connect, exists_report, fetch_report, init_schema, load_db_config, save_report +from .models import EventRow, ReportData + +__all__ = [ + "EventRow", + "ReportData", + "connect", + "exists_report", + "fetch_report", + "init_schema", + "load_db_config", + "save_report", +] diff --git a/report_db/db.py b/report_db/db.py new file mode 100644 index 0000000..1787fd3 --- /dev/null +++ b/report_db/db.py @@ -0,0 +1,166 @@ +"""日报结构化入库:DB 连接、建表、写入。""" + +from __future__ import annotations + +import json +import os +from dataclasses import dataclass +from typing import Any + +import pymysql +from loguru import logger + +from .models import ReportData +from .schema import _ALTER_ADD_SOURCES, DDL_STATEMENTS + + +@dataclass(frozen=True) +class DbConfig: + """MySQL 连接配置(来自环境变量 NEWS_DB_*)。""" + + host: str + port: int + user: str + password: str + name: str + + +def load_db_config() -> DbConfig: + """从环境变量读取 NEWS_DB_*,缺失密码时抛异常(禁止默认密码)。""" + host = os.environ.get("NEWS_DB_HOST", "127.0.0.1") + port = int(os.environ.get("NEWS_DB_PORT", "13306")) + user = os.environ.get("NEWS_DB_USER", "myquant") + password = os.environ.get("NEWS_DB_PASSWORD", "") + name = os.environ.get("NEWS_DB_NAME", "myquant") + if not password: + logger.error("NEWS_DB_PASSWORD 未配置,请在 .env 中设置") + raise ValueError("NEWS_DB_PASSWORD 未配置") + return DbConfig(host=host, port=port, user=user, password=password, name=name) + + +def connect(cfg: DbConfig | None = None) -> pymysql.Connection: + """建立短连接(autocommit=False)。失败时记录日志并抛出。""" + cfg = cfg or load_db_config() + try: + conn = pymysql.connect( + host=cfg.host, + port=cfg.port, + user=cfg.user, + password=cfg.password, + database=cfg.name, + charset="utf8mb4", + autocommit=False, + cursorclass=pymysql.cursors.DictCursor, + ) + except Exception: + logger.exception("连接 MySQL 失败: host={} port={} user={}", cfg.host, cfg.port, cfg.user) + raise + logger.debug("MySQL 已连接: {}/{}", cfg.host, cfg.name) + return conn + + +def init_schema(conn: pymysql.Connection) -> None: + """建表(CREATE TABLE IF NOT EXISTS ×2)+ 旧表 sources 列迁移,幂等。""" + with conn.cursor() as cur: + for ddl in DDL_STATEMENTS: + cur.execute(ddl) + # 旧库兼容:news_event 已存在但缺 sources 列时补列 + try: + cur.execute(_ALTER_ADD_SOURCES) + logger.info("news_event 迁移:新增 sources 多源列") + except pymysql.err.OperationalError as e: + if e.args and "Duplicate column" in str(e.args[0]): + logger.debug("news_event.sources 列已存在,跳过迁移") + else: + raise + conn.commit() + logger.info("news_report / news_event 建表完成") + + +def save_report(conn: pymysql.Connection, report: ReportData) -> int: + """事务内写入一份日报。 + + 幂等策略: + - 主表按 (report_date, report_type, file_name) 唯一键 upsert; + - 事件表 DELETE 该 report 旧行后全量 INSERT(整份覆盖一致)。 + 返回 report_id。 + """ + with conn.cursor() as cur: + stats_json = json.dumps(report.stats, ensure_ascii=False) if report.stats else None + cur.execute( + """ + INSERT INTO news_report + (report_date, report_type, file_name, generated_at, ai_summary, stats) + VALUES (%s, %s, %s, %s, %s, %s) + ON DUPLICATE KEY UPDATE + generated_at = VALUES(generated_at), + ai_summary = VALUES(ai_summary), + stats = VALUES(stats) + """, + ( + report.report_date, + report.report_type, + report.file_name, + report.generated_at, + report.ai_summary, + stats_json, + ), + ) + cur.execute( + "SELECT id FROM news_report WHERE report_date=%s AND report_type=%s AND file_name=%s", + (report.report_date, report.report_type, report.file_name), + ) + row = cur.fetchone() + if row is None: # pragma: no cover - 理论不可达 + raise RuntimeError("写入 news_report 后查询不到 report_id") + report_id: int = row["id"] + + cur.execute("DELETE FROM news_event WHERE report_id=%s", (report_id,)) + for ev in report.events: + cur.execute( + """ + INSERT INTO news_event + (report_id, section, rank, importance, event_type, title, + summary, sentiment, source, sources, url) + VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) + """, + ( + report_id, + ev.section, + ev.rank, + ev.importance, + ev.event_type, + ev.title, + ev.summary, + ev.sentiment, + ev.source, + json.dumps(ev.sources, ensure_ascii=False) if ev.sources else None, + ev.url, + ), + ) + conn.commit() + logger.info("日报已入库: report_id={} date={} type={} events={}", + report_id, report.report_date, report.report_type, len(report.events)) + return report_id + + +def fetch_report(conn: pymysql.Connection, report_id: int) -> dict[str, Any] | None: + """读侧辅助(联调/测试用),返回主表行。""" + with conn.cursor() as cur: + cur.execute("SELECT * FROM news_report WHERE id=%s", (report_id,)) + return cur.fetchone() + + +def exists_report( + conn: pymysql.Connection, + report_date: Any, + report_type: str, + file_name: str, +) -> bool: + """判断主表是否已存在该唯一键记录(历史导入幂等用)。""" + with conn.cursor() as cur: + cur.execute( + "SELECT 1 FROM news_report WHERE report_date=%s AND report_type=%s AND file_name=%s", + (report_date, report_type, file_name), + ) + return cur.fetchone() is not None diff --git a/report_db/models.py b/report_db/models.py new file mode 100644 index 0000000..efea1a4 --- /dev/null +++ b/report_db/models.py @@ -0,0 +1,35 @@ +"""日报结构化入库:数据模型。""" + +from __future__ import annotations + +from datetime import date, datetime +from typing import Any + +from pydantic import BaseModel, Field + + +class EventRow(BaseModel): + """一条事件记录,对应 news_event 一行。""" + + section: str # xwlb | news | cninfo | intl + rank: int # 板块内序号(从 1 开始) + importance: int | None = None + event_type: str | None = None + title: str + summary: str | None = None + sentiment: str | None = None # positive | negative | neutral | '' + source: str | None = None # 主源 + sources: list[str] | None = None # 全部来源(主源居首),多源新闻记录 + url: str | None = None + + +class ReportData(BaseModel): + """一份完整日报:news_report 一行 + news_event 多行。""" + + report_date: date + report_type: str # finance | intl + file_name: str = "" # 源文件名(新生成日报可为空) + generated_at: datetime + ai_summary: str | None = None + stats: dict[str, Any] = Field(default_factory=dict) + events: list[EventRow] = Field(default_factory=list) diff --git a/report_db/schema.py b/report_db/schema.py new file mode 100644 index 0000000..6c5fb59 --- /dev/null +++ b/report_db/schema.py @@ -0,0 +1,50 @@ +"""MySQL DDL:日报结构化入库(表前缀 news_,目标 MariaDB 10.11)。 + +与 project_plan.md「十八、Milestone 10」及 docs/report_db_design.md 保持一致。 +""" + +from __future__ import annotations + +DDL_STATEMENTS: list[str] = [ + # 日报主表 + """ + CREATE TABLE IF NOT EXISTS news_report ( + id BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY, + report_date DATE NOT NULL COMMENT '日报日期', + report_type VARCHAR(16) NOT NULL COMMENT 'finance=A股日报 / intl=国际财经日报', + file_name VARCHAR(160) NOT NULL DEFAULT '' COMMENT '源文件名(历史解析);新生成可为空', + generated_at DATETIME NOT NULL COMMENT '生成时间', + ai_summary TEXT NULL COMMENT 'AI 摘要全文', + stats JSON NULL COMMENT '数据总览统计快照(管道/情绪/重要度/事件类型/来源分布)', + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + UNIQUE KEY uk_report_file (report_date, report_type, file_name), + KEY idx_report_date (report_date) + ) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_unicode_ci COMMENT='每日日报主表' + """, + # 日报事件明细 + """ + CREATE TABLE IF NOT EXISTS news_event ( + id BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY, + report_id BIGINT UNSIGNED NOT NULL COMMENT 'FK → news_report.id', + section VARCHAR(16) NOT NULL COMMENT '板块: xwlb=新闻联播 / news=财经新闻 / cninfo=公告调研 / intl=国际重要事件', + rank INT NOT NULL DEFAULT 0 COMMENT '板块内序号', + importance INT NULL COMMENT '重要度 1-5', + event_type VARCHAR(64) NULL COMMENT '事件类型', + title VARCHAR(512) NOT NULL COMMENT '标题', + summary TEXT NULL COMMENT '摘要/正文', + sentiment VARCHAR(8) NULL COMMENT 'positive/negative/neutral', + source VARCHAR(64) NULL COMMENT '主来源', + sources TEXT NULL COMMENT '全部来源 JSON 数组(主源居首),多源新闻记录', + url VARCHAR(512) NULL COMMENT '原文链接', + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + KEY idx_report_section (report_id, section), + KEY idx_title (title(255)) + ) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_unicode_ci COMMENT='日报事件明细' + """, +] + +# 旧库迁移:为已存在的 news_event 表补充 sources 列(MySQL 无 ADD COLUMN IF NOT EXISTS) +_ALTER_ADD_SOURCES = ( + "ALTER TABLE news_event ADD COLUMN sources TEXT NULL " + "COMMENT '全部来源 JSON 数组(主源居首),多源新闻记录' AFTER source" +) diff --git a/report_import/__init__.py b/report_import/__init__.py new file mode 100644 index 0000000..824b7d2 --- /dev/null +++ b/report_import/__init__.py @@ -0,0 +1,21 @@ +"""历史日报解析与批量导入(Milestone 10)。 + +将 doorcome 历史日报 HTML(finance/intl)解析为结构化数据并写入 MySQL。 +""" + +from .importer import ImportStats, import_history +from .parser import ( + ReportParseError, + parse_finance_report, + parse_intl_report, + parse_report, +) + +__all__ = [ + "ImportStats", + "ReportParseError", + "import_history", + "parse_finance_report", + "parse_intl_report", + "parse_report", +] diff --git a/report_import/importer.py b/report_import/importer.py new file mode 100644 index 0000000..f851609 --- /dev/null +++ b/report_import/importer.py @@ -0,0 +1,89 @@ +"""历史日报批量导入(解析 → 入库,幂等)。""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from pathlib import Path + +from loguru import logger + +from report_db import connect, exists_report, save_report +from report_import.parser import ReportParseError, parse_report + +_REPORT_FILE_RE = re.compile(r"([a-z]+)_news_daily_\d{8}(?:_\d{4,6})?\.html$") + + +@dataclass +class ImportStats: + """一次导入的统计结果。""" + + scanned: int = 0 # 扫描到的日报文件数 + imported: int = 0 # 新入库 + skipped: int = 0 # 已存在(幂等跳过) + failed: int = 0 # 解析失败 + errors: list[str] = field(default_factory=list) + + +def _match_file(path: Path, date_str: str | None, report_type: str | None) -> bool: + """按文件名判断是否属于本次导入范围。""" + m = _REPORT_FILE_RE.search(path.name) + if not m: + return False + if report_type is not None and m.group(1) != report_type: + return False + return date_str is None or date_str in path.name + + +def import_history( + report_dir: str | Path, + date_str: str | None = None, + report_type: str | None = None, + *, + force: bool = False, +) -> ImportStats: + """扫描 {report_dir}/{YYYYMMDD}/ 下全部 `*_news_daily_*.html` 并入库。 + + - 幂等:主表唯一键 (report_date, report_type, file_name) 已存在则跳过; + - `force=True` 时跳过存在性检查,直接覆盖重导; + - 单文件解析失败不影响其他文件。 + """ + stats = ImportStats() + root = Path(report_dir) + if not root.is_dir(): + logger.error("日报目录不存在: {}", root) + raise FileNotFoundError(f"日报目录不存在: {root}") + + files = sorted(p for p in root.glob("*/[a-z]*_news_daily_*.html") if _match_file(p, date_str, report_type)) + stats.scanned = len(files) + logger.info("扫描到日报文件 {} 份: {}", stats.scanned, root) + + conn = connect() + try: + for path in files: + try: + html = path.read_text(encoding="utf-8") + report = parse_report(html, path.name) + except ReportParseError as e: + stats.failed += 1 + stats.errors.append(f"{path.name}: {e}") + logger.warning("解析失败: {} ({})", path.name, e) + continue + except Exception as e: # 防御未知异常,不中断批量 + stats.failed += 1 + stats.errors.append(f"{path.name}: {type(e).__name__}: {e}") + logger.exception("读取/解析异常: {}", path.name) + continue + + if not force and exists_report(conn, report.report_date, report.report_type, report.file_name): + stats.skipped += 1 + logger.debug("已存在, 跳过: {}", path.name) + continue + save_report(conn, report) + stats.imported += 1 + finally: + conn.close() + + logger.info("导入完成: scanned={} imported={} skipped={} failed={}", + stats.scanned, stats.imported, stats.skipped, stats.failed) + return stats diff --git a/report_import/parser.py b/report_import/parser.py new file mode 100644 index 0000000..a50bb48 --- /dev/null +++ b/report_import/parser.py @@ -0,0 +1,298 @@ +"""历史日报 HTML 解析器(finance / intl)。 + +策略:表头驱动列映射,不依赖列位置;板块按 h2 标题识别; +数据总览按 h3 标题归类为 stats JSON 快照(前端自行解析)。 +""" + +from __future__ import annotations + +import re +from datetime import date, datetime + +from bs4 import BeautifulSoup, Tag + +from report_db.models import EventRow, ReportData + +# 情绪图标 → sentiment 值 +_SENTIMENT_ICON: dict[str, str] = {"⚪": "neutral", "🔴": "negative", "🟢": "positive"} + +# 数据总览 h3 标题关键词 → stats key +_STATS_SECTION_KEYS: list[tuple[str, str]] = [ + ("管道", "pipeline"), + ("各源", "sources"), + ("情绪", "sentiment"), + ("重要度", "importance"), + ("事件类型", "event_types"), + ("来源", "source_dist"), +] + + +class ReportParseError(Exception): + """整份文件解析失败。""" + + +# --------------------------------------------------------------------------- # +# 文件名 / 时间解析 +# --------------------------------------------------------------------------- # + +def _parse_datetime_from_filename(file_name: str) -> tuple[date, datetime] | None: + """从文件名解析日报日期与生成时间。 + + 支持 `{type}_news_daily_{YYYYMMDD}.html` 与带时间戳的 + `{type}_news_daily_{YYYYMMDD}_{HHMMSS}.html` / `..._{HHMM}.html`。 + """ + m = re.search(r"_daily_(\d{8})(?:_(\d{4})(\d{2})?)?", file_name) + if not m: + return None + day = date(int(m.group(1)[:4]), int(m.group(1)[4:6]), int(m.group(1)[6:8])) + hh = mm = ss = 0 + if m.group(2): + hh, mm = int(m.group(2)[:2]), int(m.group(2)[2:4]) + ss = int(m.group(3) or 0) + return day, datetime(day.year, day.month, day.day, hh, mm, ss) + + +def _parse_header_generated_at(soup: BeautifulSoup, fallback: datetime) -> datetime: + """从 <header> 中"生成于 YYYY-MM-DD HH:MM:SS"解析生成时间。""" + p = soup.select_one("header p") + if p: + m = re.search(r"生成于 (\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2})", p.get_text()) + if m: + return datetime.strptime(m.group(1), "%Y-%m-%d %H:%M:%S") + return fallback + + +# --------------------------------------------------------------------------- # +# AI 摘要 +# --------------------------------------------------------------------------- # + +def _parse_ai_summary(soup: BeautifulSoup) -> str | None: + """AI 摘要:<div class="ai-summary">,li 逐行输出。""" + div = soup.select_one("div.ai-summary") + if div is None: + return None + items = [li.get_text(strip=True) for li in div.find_all("li") if li.get_text(strip=True)] + if items: + return "\n".join(items) + text = re.sub(r"\s+", " ", div.get_text(strip=True)) + return text or None + + +# --------------------------------------------------------------------------- # +# 事件表解析 +# --------------------------------------------------------------------------- # + +def _to_int(text: str) -> int | None: + m = re.search(r"\d+", text or "") + return int(m.group(0)) if m else None + + +def _parse_summary(cell: Tag) -> tuple[str | None, str | None]: + """摘要列:剥除 <small>[来源]</small>,返回 (摘要, 来源)。""" + source = None + small = cell.find("small") + if small: + m = re.search(r"\[([^\]]+)\]", small.get_text()) + if m: + source = m.group(1) + small.decompose() + text = re.sub(r"\s+", " ", cell.get_text(strip=True)) + return (text or None, source) + + +def _clean_title(cell: Tag) -> str: + """标题列:剥除 <small> 股票代码标注等,返回纯标题。""" + for small in cell.find_all("small"): + small.decompose() + return re.sub(r"\s+", " ", cell.get_text(strip=True)) + + +def _parse_event_table(table: Tag, section: str) -> list[EventRow]: + """事件表 → EventRow 列表。表头驱动列映射,兼容 xwlb/news/cninfo/intl 四类表。""" + rows = table.find_all("tr") + if len(rows) < 2: + return [] + header_cells = rows[0].find_all(["th", "td"]) + col_index: dict[str, int] = {} + icon_col: int | None = None + for i, cell in enumerate(header_cells): + text = cell.get_text(strip=True) + if text: + col_index[text] = i + elif icon_col is None: + icon_col = i + + out: list[EventRow] = [] + for row in rows[1:]: + tds = row.find_all("td") + if not tds: + continue + + def col(name: str, tds: list[Tag] = tds) -> Tag | None: # noqa: B008 - 绑定循环变量 + idx = col_index.get(name) + return tds[idx] if idx is not None and idx < len(tds) else None + + title_cell = col("标题") + if title_cell is None: + continue + a = title_cell.find("a") + url = a.get("href") if a else None + + summary_cell = col("摘要") + summary: str | None = None + source: str | None = None + if summary_cell is not None: + summary, source = _parse_summary(summary_cell) + if source is None: + src_cell = col("源") + if src_cell is not None and src_cell.get_text(strip=True): + source = src_cell.get_text(strip=True) + + sentiment = None + if icon_col is not None and icon_col < len(tds): + sentiment = _SENTIMENT_ICON.get(tds[icon_col].get_text(strip=True)) + + imp_cell = col("重要度") + type_cell = col("事件类型") + out.append( + EventRow( + section=section, + rank=_to_int(tds[0].get_text(strip=True)) or len(out) + 1, + importance=_to_int(imp_cell.get_text(strip=True)) if imp_cell else None, + event_type=type_cell.get_text(strip=True) if type_cell else None, + title=_clean_title(title_cell), + summary=summary, + sentiment=sentiment, + source=source, + url=url, + ) + ) + return out + + +def _collect_events(soup: BeautifulSoup, report_type: str) -> list[EventRow]: + """按 h2 板块标题收集各事件表。""" + events: list[EventRow] = [] + for h2 in soup.find_all("h2"): + title = h2.get_text() + table = h2.find_next_sibling("table") + if table is None: + continue + section: str | None = None + if "新闻联播" in title: + section = "xwlb" + elif "公告" in title or "调研" in title: + section = "cninfo" + elif "重要事件" in title: + section = "intl" if report_type == "intl" else "news" + if section is not None: + events.extend(_parse_event_table(table, section)) + return events + + +# --------------------------------------------------------------------------- # +# 数据总览 stats +# --------------------------------------------------------------------------- # + +def _table_to_rows(table: Tag) -> list[dict[str, str]]: + """表格 → [{表头: 值, ...}, ...](首行为表头)。""" + rows: list[list[str]] = [] + for tr in table.find_all("tr"): + cells = [re.sub(r"\s+", " ", c.get_text(strip=True)) for c in tr.find_all(["th", "td"])] + if cells: + rows.append(cells) + if not rows: + return [] + header = rows[0] + return [dict(zip(header, r, strict=False)) for r in rows[1:]] + + +def _parse_stats_block(h3: Tag) -> tuple[str, object] | None: + """h3 数据总览区块 → (stats_key, value)。缺失/未知板块返回 None。""" + title = h3.get_text() + key = next((k for kw, k in _STATS_SECTION_KEYS if kw in title), None) + if key is None: + return None + block = h3.find_next_sibling() + if block is None: + return key, {} + + if block.name == "table": + return key, _table_to_rows(block) + + classes = block.get("class", []) if isinstance(block.get("class"), list) else [] + if "stats-grid" in classes: + cards: dict[str, str | int] = {} + for card in block.find_all("div", class_="stat-card"): + label = card.select_one(".label") + num = card.select_one(".num") + if label is not None: + num_text = num.get_text(strip=True) if num else "" + cards[label.get_text(strip=True)] = _to_int(num_text) if _to_int(num_text) is not None else num_text + return key, cards + if "source-grid" in classes: + items: dict[str, str | int] = {} + for item in block.find_all("div", class_="source-item"): + name = item.select_one(".s-name") + count = item.select_one(".s-count") + if name is not None: + count_text = count.get_text(strip=True) if count else "" + items[name.get_text(strip=True)] = _to_int(count_text) or count_text + return key, items + if "sentiment-bar" in classes: + legend = block.find_next_sibling("div", class_="sentiment-legend") + spans = legend.find_all("span") if legend else [] + return key, [re.sub(r"\s+", " ", s.get_text(strip=True)) for s in spans] + + text = re.sub(r"\s+", " ", block.get_text(strip=True)) + return key, text[:500] + + +def _collect_stats(soup: BeautifulSoup) -> dict[str, object]: + stats: dict[str, object] = {} + for h3 in soup.find_all("h3"): + parsed = _parse_stats_block(h3) + if parsed is not None: + stats[parsed[0]] = parsed[1] + return stats + + +# --------------------------------------------------------------------------- # +# 主入口 +# --------------------------------------------------------------------------- # + +def _parse(html: str, file_name: str, report_type: str) -> ReportData: + parsed = _parse_datetime_from_filename(file_name) + if parsed is None: + raise ReportParseError(f"无法从文件名解析日报日期: {file_name}") + day, gen_from_file = parsed + + soup = BeautifulSoup(html, "html.parser") + generated_at = _parse_header_generated_at(soup, gen_from_file) + + return ReportData( + report_date=day, + report_type=report_type, + file_name=file_name, + generated_at=generated_at, + ai_summary=_parse_ai_summary(soup), + stats=_collect_stats(soup), + events=_collect_events(soup, report_type), + ) + + +def parse_finance_report(html: str, file_name: str) -> ReportData: + """解析 A 股日报 finance_news_daily_*.html。""" + return _parse(html, file_name, "finance") + + +def parse_intl_report(html: str, file_name: str) -> ReportData: + """解析国际财经日报 intl_news_daily_*.html。""" + return _parse(html, file_name, "intl") + + +def parse_report(html: str, file_name: str) -> ReportData: + """按文件名前缀自动分流 finance / intl。""" + if "intl_news_daily" in file_name: + return parse_intl_report(html, file_name) + return parse_finance_report(html, file_name) diff --git a/scheduler/__init__.py b/scheduler/__init__.py new file mode 100644 index 0000000..7fd674b --- /dev/null +++ b/scheduler/__init__.py @@ -0,0 +1,25 @@ +"""定时任务模块 (M7)。 + +公共 API: + - run_pipeline: 串联执行 M1→M6 全链路 + - run_step: 执行单个步骤 + - StepResult / PipelineResult: 结果模型 +""" + +from .pipeline import ( + STEP_COMMANDS, + STEP_TIMEOUTS, + PipelineResult, + StepResult, + run_pipeline, + run_step, +) + +__all__ = [ + "STEP_COMMANDS", + "STEP_TIMEOUTS", + "PipelineResult", + "StepResult", + "run_pipeline", + "run_step", +] diff --git a/scheduler/pipeline.py b/scheduler/pipeline.py new file mode 100644 index 0000000..993c54b --- /dev/null +++ b/scheduler/pipeline.py @@ -0,0 +1,369 @@ +"""定时任务主流程(M7)。 + +编排 M1→M6 全链路,每一步调用已有脚本。 +单步失败记录日志但不阻断后续(后续步骤可能使用旧缓存数据,降级继续)。 + +断点恢复: + 每次运行把各步骤结果写入 data/pipeline/state.json(按日期隔离); + run_pipeline(resume=True) 时跳过连续成功的步骤,从第一个失败/未执行 + 的步骤继续,实现 `pipeline --once --resume` 断点续跑。 +""" + +from __future__ import annotations + +import json +import os +import subprocess +import time +from dataclasses import dataclass, field +from datetime import date, datetime +from pathlib import Path + +from loguru import logger + +# 断点状态文件(按日期隔离,记录每步骤结果) +DEFAULT_STATE_PATH = Path("data/pipeline/state.json") + +# 步骤名 → 中文阶段名(显性输出用) +_STAGE_LABELS: dict[str, str] = { + "crawler": "M1 新闻抓取", + "xwlb": "M1 新闻联播抓取", + "extractor": "M2 正文提取", + "dedup": "M3 新闻去重", + "llm": "M4 LLM 事件抽取", + "embedding": "M5 向量化嵌入", + "qdrant": "M6 Qdrant 入库", + "report": "日报生成", + "cninfo_crawl": "cninfo 公告抓取", + "cninfo_extract": "cninfo 正文提取", + "cninfo_pdf": "cninfo PDF 补充", +} + +# 使用对话大模型的步骤 → 对应 configs/llm_models.yaml 场景名 +_AI_LLM_SCENES: dict[str, str] = { + "llm": "event_extraction", + "report": "daily_report", +} + +# 步骤超时(秒) +STEP_TIMEOUTS: dict[str, int] = { + "crawler": 900, # M1 抓取(含 Playwright 浏览器,13 源约 8-12 min) + "xwlb": 60, # M1 新闻联播 API(纯 HTTP,秒级) + "extractor": 300, # M2 正文提取 + "dedup": 120, # M3 去重 + "llm": 900, # M4 LLM 事件抽取(API 调用,100 篇约 30s 但加限流余量) + "embedding": 300, # M5 向量化 + "qdrant": 300, # M6 入库(数据量大时需较长时间) + "report": 30, # 日报生成+上传 + "cninfo_crawl": 900, # cninfo watchlist URL 驱动(SPA 渲染,每只约 25s) +} + +# 步骤对应的 uv run 命令(参数中 {date} 会被替换为实际日期) +# crawler 不支持 --date,固定写当天目录; dedup 不加 --reset 以保持增量 +STEP_COMMANDS: dict[str, list[str]] = { + "crawler": ["uv", "run", "python", "-m", "scripts.run_crawler"], # 无 --date + "xwlb": ["uv", "run", "python", "-m", "scripts.run_xwlb"], + "extractor": ["uv", "run", "python", "-m", "scripts.run_extractor", "--date", "{date}"], + "dedup": ["uv", "run", "python", "-m", "scripts.run_dedup", "--date", "{date}"], + "llm": ["uv", "run", "python", "-m", "scripts.run_event_extraction", "--date", "{date}"], + "embedding": ["uv", "run", "python", "-m", "scripts.run_embedding", "--date", "{date}"], + "qdrant": ["uv", "run", "python", "-m", "scripts.run_qdrant_ingest", "--date", "{date}"], + "report": [], # 特殊步骤:仅在每天首次定时任务时追加 + "cninfo_crawl": ["uv", "run", "a-share", "cninfo"], + "cninfo_extract": ["uv", "run", "a-share", "extract", "--source", "cninfo", "--date", "{date}"], + "cninfo_pdf": ["uv", "run", "a-share", "cninfo", "--enrich-pdf", "--pdf-limit", "100"], +} + + +@dataclass +class StepResult: + name: str + success: bool + elapsed_sec: float + exit_code: int | None = None + tail_msg: str = "" + started_at: datetime | None = None + + +@dataclass +class PipelineResult: + steps: list[StepResult] = field(default_factory=list) + started_at: datetime | None = None + finished_at: datetime | None = None + + @property + def all_success(self) -> bool: + return all(s.success for s in self.steps) + + +def _load_pipeline_state(path: Path = DEFAULT_STATE_PATH) -> dict: + """读取断点状态文件;不存在或损坏时返回空 dict。""" + if not path.is_file(): + return {} + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError) as e: + logger.warning("pipeline 状态文件损坏,忽略: {} ({})", path, e) + return {} + return data if isinstance(data, dict) else {} + + +def _save_pipeline_state(state: dict, path: Path = DEFAULT_STATE_PATH) -> None: + """原子写状态文件(tmp + rename,避免中断写坏)。""" + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_suffix(".json.tmp") + tmp.write_text(json.dumps(state, ensure_ascii=False, indent=2), encoding="utf-8") + tmp.replace(path) + + +def _update_step_state(state: dict, date_str: str, sr: StepResult) -> None: + """把单步结果写入状态(ok/failed,含退出码与耗时)。""" + day_state = state.setdefault(date_str, {}) + day_state[sr.name] = { + "status": "ok" if sr.success else "failed", + "exit_code": sr.exit_code, + "started_at": sr.started_at.isoformat() if sr.started_at else None, + "elapsed_sec": round(sr.elapsed_sec, 1), + } + + +def _resume_start_index( + names: list[str], + date_str: str, + state: dict, +) -> int: + """计算断点续跑起始下标:跳过连续 ok 前缀,从首个失败/未记录步骤开始。 + + 返回 0..len(names)-1;全部成功时返回 len(names)(表示无需续跑)。 + """ + day_state = state.get(date_str, {}) + for i, name in enumerate(names): + rec = day_state.get(name) + if rec is None or rec.get("status") != "ok": + return i + return len(names) + + +def _llm_scene_desc(scene: str) -> str | None: + """解析某 LLM 场景的 provider/model(仅展示,不校验 API key)。""" + try: + from configs.loader import load_scene_config + + def _env(key: str) -> str | None: + v = os.environ.get(key) + return v.strip() if v else None + + sc = load_scene_config(scene) + p = (sc.get("provider") or _env("LLM_PROVIDER") or "deepseek").lower() + if p in ("qwen", "dashscope"): + p = "qwen" + model = sc.get("model") + if not model: + if p == "qwen": + model = _env("QWEN_MODEL") or _env("LLM_MODEL") + else: + model = _env("DEEPSEEK_MODEL") or _env("LLM_MODEL") + if not model: + return None + return f"provider={p}, model={model}" + except Exception as e: # noqa: BLE001 - 配置缺失时降级展示 + logger.debug("AI 描述解析失败(场景 {}): {}", scene, e) + return None + + +def _embedding_desc() -> str | None: + """解析 embedding 场景的 provider/model(不实例化模型,避免加载本地权重)。""" + try: + from configs.loader import load_scene_config + from embedding import resolve_provider_type + + sc = load_scene_config("embedding") + pt = resolve_provider_type().value # dashscope | local-bge + model = sc.get("model") + if not model: + if pt == "dashscope": + model = os.environ.get("DASHSCOPE_EMBEDDING_MODEL") or "text-embedding-v3" + else: + model = os.environ.get("LOCAL_EMBEDDING_MODEL") or "BAAI/bge-m3" + return f"provider={pt}, model={model}" + except Exception as e: # noqa: BLE001 + logger.debug("embedding 描述解析失败: {}", e) + return None + + +def _log_stage_header(name: str, date_str: str, index: int, total: int) -> None: + """显性输出当前阶段(中文名 + 步骤名 + 序号 + 日期)。""" + label = _STAGE_LABELS.get(name, name) + logger.info("") + logger.info("═" * 56) + logger.info("阶段 {}/{}: {} [{}] 日期 {}", index, total, label, name, date_str) + logger.info("═" * 56) + + +def _log_ai_info(name: str) -> None: + """本阶段用到 AI 大模型时,显性告知供应商与模型名。""" + if name in _AI_LLM_SCENES: + scene = _AI_LLM_SCENES[name] + desc = _llm_scene_desc(scene) + logger.info("🤖 本阶段使用 AI 大模型: {}", desc or "未配置(可能跳过或降级)") + elif name == "embedding": + desc = _embedding_desc() + logger.info("🤖 本阶段使用 AI 嵌入模型: {}", desc or "未配置(可能降级)") + + +def run_step(name: str, date_str: str) -> StepResult: + """执行单个 pipeline 步骤。 + + 参数: + name: 步骤名(crawler/extractor/.../report) + date_str: YYYYMMDD 日期字符串 + + 返回: StepResult。 + """ + # report 步骤:内部函数,不走子进程 + # 日报按当天日期生成: 新闻由 _collect_news_events 回溯过去 30 小时, + # xwlb 由 _collect_xwlb 固定取前一日(已播出)联播。 + if name == "report": + started = datetime.now() + try: + from .reporter import generate_report # noqa: E402 + + report_date = date.today().strftime("%Y%m%d") + logger.info("日报: report_date={} (新闻 30h 回溯, xwlb 前一日)", report_date) + + path = generate_report(report_date, upload=True) + elapsed = (datetime.now() - started).total_seconds() + ok = path is not None + return StepResult( + name=name, success=ok, elapsed_sec=elapsed, + tail_msg=str(path) if path else f"无数据 (report_date={report_date})", + started_at=started, + ) + except Exception as e: # noqa: BLE001 + elapsed = (datetime.now() - started).total_seconds() + logger.exception("日报生成异常: {}", e) + return StepResult(name=name, success=False, elapsed_sec=elapsed, + tail_msg=str(e)[:200], started_at=started) + + cmd = STEP_COMMANDS.get(name) + if cmd is None: + return StepResult(name=name, success=False, elapsed_sec=0, + tail_msg=f"未知步骤: {name}") + + full_cmd = [arg.replace("{date}", date_str) for arg in cmd] + # 超时优先级: + # 1. TIMEOUT_{NAME} 环境变量 (单步精确控制) + # 2. PIPELINE_STEP_TIMEOUT 环境变量 (全局兜底, 覆盖硬编码) + # 3. STEP_TIMEOUTS 硬编码字典 (代码内默认值) + # 4. 1800s (最终兜底) + import os + specific_key = f"TIMEOUT_{name.upper()}" + if specific_key in os.environ: + timeout = int(os.environ[specific_key]) + elif "PIPELINE_STEP_TIMEOUT" in os.environ: + timeout = int(os.environ["PIPELINE_STEP_TIMEOUT"]) + else: + timeout = STEP_TIMEOUTS.get(name, 1800) + started = datetime.now() + logger.info("步骤 {} 开始: {}", name, " ".join(full_cmd)) + + try: + # 不捕获输出,子进程日志直接流到终端(用户能看到每个源的抓取进度) + proc = subprocess.run( + full_cmd, + timeout=timeout, + ) + elapsed = (datetime.now() - started).total_seconds() + ok = proc.returncode == 0 + + # dedup 返回 1 是"重复率过高"(无新文章的正常场景) + if name == "dedup" and proc.returncode == 1: + ok = True + logger.info("dedup 重复率超过阈值(无新数据场景,视为成功)") + + tail_msg = f"rc={proc.returncode}" if not ok else "" + + if ok: + logger.info("步骤 {} 完成 ({}s) ✅", name, elapsed) + else: + logger.error("步骤 {} 失败 rc={} ({}s)", name, proc.returncode, elapsed) + + return StepResult( + name=name, success=ok, elapsed_sec=elapsed, + exit_code=proc.returncode, tail_msg=tail_msg, + started_at=started, + ) + except subprocess.TimeoutExpired: + elapsed = (datetime.now() - started).total_seconds() + logger.error("步骤 {} 超时 (>{:.0f}s)", name, elapsed) + return StepResult(name=name, success=False, elapsed_sec=elapsed, + tail_msg="超时", started_at=started) + except Exception as e: # noqa: BLE001 + elapsed = (datetime.now() - started).total_seconds() + logger.exception("步骤 {} 异常: {}", name, e) + return StepResult(name=name, success=False, elapsed_sec=elapsed, + tail_msg=str(e)[:200], started_at=started) + + +def run_pipeline( + date_str: str, + *, + steps: list[str] | None = None, + resume: bool = False, + state_path: Path = DEFAULT_STATE_PATH, +) -> PipelineResult: + """串联执行全链路(M1→M6)。 + + 参数: + date_str: YYYYMMDD。 + steps: 可选步骤列表,默认全部 6 步。 + resume: True 时断点续跑——读取 data/pipeline/state.json 中该日期的 + 记录,跳过连续成功的步骤,从第一个失败/未执行步骤继续。 + state_path: 断点状态文件路径(测试可注入)。 + """ + names = steps or [k for k in STEP_COMMANDS if k not in ("report", "cninfo_crawl", "cninfo_extract", "cninfo_pdf")] + result = PipelineResult(started_at=datetime.now()) + + state = _load_pipeline_state(state_path) + start_idx = 0 + if resume: + start_idx = _resume_start_index(names, date_str, state) + if start_idx >= len(names): + logger.info("resume: {} 的所有步骤均已完成,无需续跑", date_str) + result.finished_at = datetime.now() + return result + logger.info( + "resume: 从步骤 {} 继续{}", + names[start_idx], + f" (跳过已成功 {names[:start_idx]})" if start_idx > 0 else "", + ) + + for i, name in enumerate(names[start_idx:], start=start_idx + 1): + # 显性输出当前阶段 + AI 大模型信息 + _log_stage_header(name, date_str, i, len(names)) + _log_ai_info(name) + sr = run_step(name, date_str) + result.steps.append(sr) + # 记录断点状态(无论成败,便于下次 resume) + _update_step_state(state, date_str, sr) + _save_pipeline_state(state, state_path) + if not sr.success: + logger.warning("步骤 {} 失败,后续步骤继续(可能降级)", name) + # 步间留一点缓冲 + time.sleep(0.5) + + result.finished_at = datetime.now() + total = (result.finished_at - result.started_at).total_seconds() if result.started_at else 0 + succ = sum(1 for s in result.steps if s.success) + rate = succ / max(len(result.steps), 1) + logger.info( + "Pipeline 完成: {}/{} 步骤成功 ({:.0%}) 总耗时 {:.0f}s", + succ, len(result.steps), rate, total, + ) + + # 输出摘要 + for s in result.steps: + flag = "✅" if s.success else "❌" + logger.info(" {} {} {}s", flag, s.name, s.elapsed_sec) + + return result diff --git a/scheduler/reporter.py b/scheduler/reporter.py new file mode 100644 index 0000000..e9f9498 --- /dev/null +++ b/scheduler/reporter.py @@ -0,0 +1,1119 @@ +"""每日摘要报告生成器 v2.0。 + +输出 HTML 日报,包含: + 一、AI 摘要 (最新新闻 + {cninfo_days}d 公告/调研/互动) + 二、重要事件: 新闻 (最新抓取, importance≥4, 最多20篇) + 三、重要事件: 公告/互动 ({cninfo_days}d, cninfo, 最多20篇) + 四、数据总览 (重要度/各源/M1-M6/情绪/事件类型分布) +""" + +from __future__ import annotations + +import json +import os as _os +import re as _re +import subprocess +import time +from collections import Counter +from datetime import date, datetime, timedelta +from pathlib import Path +from typing import TYPE_CHECKING, Any + +if TYPE_CHECKING: + from llm.client import LLMConfig + +from dotenv import load_dotenv +from loguru import logger + +from report_db.models import EventRow, ReportData # noqa: F401 - 供 _build_report_data 注解使用 + +# 确保 .env 已加载(模块级常量依赖环境变量) +load_dotenv() + +# --------------------------------------------------------------------------- # +# 配置 +# --------------------------------------------------------------------------- # + +UPLOAD_HOST = "simon@doorcome.cn" +UPLOAD_BASE = "/var/www/html/echart/research" + +CNINFO_DAYS_BACK = int(_os.environ.get("STOCK_REPORT_DAYS", "15")) # 与个股日报共用参数, 默认值保持一致 +NEWS_DAYS_BACK = 1 # 新闻回溯天数 + +_MAX_HIGH_EVENTS = 20 + +# LLM 摘要调用重试参数(环境变量可覆盖) +_LLM_RETRY_TIMES = int(_os.environ.get("LLM_RETRY_TIMES", "3")) +_LLM_RETRY_BACKOFF_SEC = float(_os.environ.get("LLM_RETRY_BACKOFF_SEC", "2.0")) + +# 日报新闻回溯窗口(小时):07:00 生成当日日报时覆盖昨日全天至今晨的新闻 +_NEWS_LOOKBACK_HOURS = 30 + + +def _load_source_names() -> dict[str, str]: + import yaml + try: + with open("configs/sources.yaml", encoding="utf-8") as f: + data = yaml.safe_load(f) + return {s["id"]: s["name"] for s in (data.get("sources") or []) if s.get("id")} + except Exception: + return {} + + +def _source_name(src_id: str) -> str: + return _load_source_names().get(src_id, src_id) + + +def _load_watchlist_codes() -> set[str]: + import yaml + try: + with open("configs/watchlist.yaml", encoding="utf-8") as f: + data = yaml.safe_load(f) or {} + return {it["code"] for it in (data.get("watchlist") or []) if it.get("code")} + except Exception: + return set() + + +# --------------------------------------------------------------------------- # +# 数据收集 +# --------------------------------------------------------------------------- # + +def _count_jsonl(path: Path) -> int: + if not path.is_file(): + return 0 + return sum(1 for _ in open(path, encoding="utf-8")) + + +def _count_json(pattern: str) -> int: + return len(list(Path().glob(pattern))) + + +def _load_events_from_dir(day_str: str) -> list[dict]: + """从 data/events/{day_str}/ 加载所有事件。""" + events: list[dict] = [] + ev_dir = Path(f"data/events/{day_str}") + if not ev_dir.is_dir(): + return events + for fp in sorted(ev_dir.glob("*.json")): + try: + obj = json.loads(fp.read_text(encoding="utf-8")) + ev = obj.get("event", {}) + events.append({ + "title": obj.get("title", ""), + "url": obj.get("url", ""), + "source_id": obj.get("source_id", ""), + "sources": obj.get("sources") or [obj.get("source_id", "")], + "publish_time": obj.get("publish_time"), + "event": ev, + }) + except (json.JSONDecodeError, OSError): + pass + return events + + +def _collect_news_events(day_str: str) -> dict[str, Any]: + """收集新闻事件(排除 cninfo)。 + + 读取 `day_str` 与前一天两个事件目录,按 publish_time 过滤最近 + `_NEWS_LOOKBACK_HOURS`(默认 30)小时内的新闻——07:00 生成当日日报时 + 可覆盖昨日全天至今晨的新闻。无 publish_time 的事件保留(容错)。 + """ + day = datetime.strptime(day_str, "%Y%m%d").date() + prev_day = (day - timedelta(days=1)).strftime("%Y%m%d") + all_ev = _load_events_from_dir(day_str) + _load_events_from_dir(prev_day) + + # publish_time 过滤: 最近 30 小时(时间缺失/格式异常的事件保留) + cutoff = (datetime.now() - timedelta(hours=_NEWS_LOOKBACK_HOURS)).astimezone() + news_ev: list[dict] = [] + for e in all_ev: + if e["source_id"] == "cninfo": + continue + pt = e.get("publish_time") + if pt: + try: + # naive 时间假定为本地时区, 与带时区(aware)的 cutoff 统一比较 + t = datetime.fromisoformat(pt) + if t.tzinfo is None: + t = t.astimezone() + if t < cutoff: + continue + except (ValueError, TypeError): + pass # 时间格式异常时保留 + news_ev.append(e) + + sentiments: Counter = Counter() + importances: Counter = Counter() + event_types: Counter = Counter() + for e in news_ev: + ev = e["event"] + sentiments[ev.get("sentiment", "?")] += 1 + importances[ev.get("importance", 0)] += 1 + event_types[ev.get("event_type", "?")] += 1 + + # 高重要度: 优先 ≥4, 不足时逐级回退(≥3 → ≥2 → 全部按 importance 排序) + def _get_high(evs, threshold): + return sorted( + [e for e in evs if e["event"].get("importance", 0) >= threshold], + key=lambda e: -e["event"].get("importance", 0), + ) + + min_show = 3 + high = _get_high(news_ev, 4) + hi_threshold = 4 + if len(high) < min_show: + high = _get_high(news_ev, 3) + hi_threshold = 3 + if len(high) < min_show: + high = _get_high(news_ev, 2) + hi_threshold = 2 + if len(high) < min_show: + high = sorted(news_ev, key=lambda e: -e["event"].get("importance", 0)) + hi_threshold = 0 + high = high[:_MAX_HIGH_EVENTS] + + return { + "total": len(news_ev), + "high": high, + "hi_threshold": hi_threshold, + "sentiments": dict(sentiments), + "importances": dict(sorted(importances.items())), + "event_types": dict(event_types.most_common(10)), + } + + +def _collect_cninfo_events(today_str: str, days_back: int = CNINFO_DAYS_BACK) -> dict[str, Any]: + """收集近 N 日 cninfo 公告/调研/互动(直接从 processed 数据读取,不依赖 M4 事件抽取)。 + + cninfo 公告/调研数据已结构化(stock_code/name/title/time/type), + 无需经过 LLM 事件抽取即可直接用于日报。 + """ + today = datetime.strptime(today_str, "%Y%m%d") + since_str = (today - timedelta(days=days_back)).strftime("%Y-%m-%d") + wl_codes = _load_watchlist_codes() + + items: list[dict] = [] + seen_urls: set[str] = set() + proc_root = Path("data/processed/cninfo") + if not proc_root.is_dir(): + return {"total": 0, "high": [], "hi_threshold": 4, "by_day": {}, + "announcement": 0, "research": 0, "irm": 0} + + for day_dir in sorted(proc_root.glob("*"), reverse=True): + if not day_dir.is_dir(): + continue + for fp in sorted(day_dir.glob("*.json"), reverse=True): + if fp.name == "index.jsonl": + continue + try: + obj = json.loads(fp.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError): + continue + + url = obj.get("url") or "" + if url in seen_urls: + continue + seen_urls.add(url) + + pt = (obj.get("publish_time") or "").strip() + # 按 publish_time 过滤 + if pt and pt[:10] < since_str: + continue + + item_type = obj.get("item_type") or "announcement" + # 互动易数据跳过(当前无法获取真实数据) + if item_type == "irm": + continue + # 过滤旧数据的 IRM 假阳性(标题为通用占位符或 URL 为 irm 搜索页) + if "互动问答" in (obj.get("title") or ""): + continue + if "irm.cninfo.com.cn" in (obj.get("url") or ""): + continue + + title = obj.get("title") or "" + url = obj.get("url") or "" + stock_name = obj.get("author") or "" + content = obj.get("content") or "" + + # 计算重要度(基于是否在 watchlist 中 + 内容长度) + code_in_title = "" + for c in wl_codes: + if c in title: + code_in_title = c + break + importance = 3 if code_in_title else 2 + if item_type == "research": + importance = 3 # 调研通常更重要 + + items.append({ + "title": title, + "url": url, + "source_id": "cninfo", + "publish_time": pt, + "event": { + "stock_codes": [code_in_title] if code_in_title else [], + "company_names": [stock_name] if stock_name else [], + "industries": [], + "sentiment": "neutral", + "importance": importance, + "event_type": { + "announcement": "公司公告", + "research": "投资者调研", + }.get(item_type, "公告"), + "summary": content[:120] if content else title[:120], + }, + }) + + # 按发布时间排序(最新在前) + items.sort(key=lambda e: e.get("publish_time") or "", reverse=True) + + # 按发布时间的日期分组统计 + by_day: Counter = Counter() + for e in items: + pt = (e.get("publish_time") or "")[:10] + if pt: + by_day[pt] += 1 + + # 高重要度: 优先 ≥4(调研),逐级回退 + def _get_high(evs, threshold): + return sorted( + [e for e in evs if e["event"].get("importance", 0) >= threshold], + key=lambda e: -e["event"].get("importance", 0), + ) + min_show = 3 + high = _get_high(items, 4) + hi_threshold = 4 + if len(high) < min_show: + high = _get_high(items, 2) + hi_threshold = 2 + if len(high) < min_show: + high = sorted(items, key=lambda e: -e["event"].get("importance", 0)) + hi_threshold = 0 + high = high[:_MAX_HIGH_EVENTS] + + return { + "total": len(items), + "high": high, + "hi_threshold": hi_threshold, + "by_day": dict(by_day.most_common(7)), + "announcement": sum(1 for e in items if "公告" in (e["event"].get("event_type", "") or "")), + "research": sum(1 for e in items if "调研" in (e["event"].get("event_type", "") or "")), + "irm": sum(1 for e in items if "互动" in (e["event"].get("event_type", "") or "")), + } + + +def _score_xwlb_importance(title: str, content: str = "") -> int: + """新闻联播条目启发式重要度评分 (1-5)。 + + 基于标题+正文关键词匹配,优先匹配高等级: + 5: 直接涉及股市/金融/货币政策 + 4: 重大经济/产业政策/能源 + 3: 领导人活动/外交/区域发展/外资 + 2: 一般国内要闻/农业/生态 + 1: 文化/体育/社会/国际简讯 + """ + text = title + content + L5 = ["降准", "降息", "印花税", "IPO", "注册制", "退市", + "并购重组", "增持", "回购", "证券", "股市", "上市"] + L4 = ["经济", "财政", "税收", "国债", "专项债", "碳", + "产业", "制造业", "新能源", "芯片", "半导体", "人工智能", + "算力", "平台经济", "房地产", "外贸", "消费", "投资", + "供应链", "能源", "电力"] + L3 = ["习近平", "李强", "总理", "主席", "会谈", "访问", + "自贸区", "长三角", "粤港澳", "一带一路", + "央企", "国企", "营商环境", "外资", "达沃斯"] + L2 = ["会议", "改革", "立法", "监管", "粮食", "农业", + "水利", "铁路", "公路", "港口", "生态", "救灾"] + + if any(kw in text for kw in L5): + return 5 + if any(kw in text for kw in L4): + return 4 + if any(kw in text for kw in L3): + return 3 + if any(kw in text for kw in L2): + return 2 + return 1 + + +def _collect_xwlb(day_str: str) -> dict[str, Any]: + """收集新闻联播要闻(从 doorcome API /api/xwlbFine/ 获取)。 + + 《新闻联播》每天 19:00 播出:日报在早上生成时当日联播尚未播出, + 因此固定取 `day_str` 前一日(最近一期已播出)的联播数据。 + + API 返回 AI 精编后的独立新闻条目(含标题+正文), + 跳过第 1 条"内容提要"(仅为节目开场白)。 + + 返回: {"items": [event_dict, ...], "date": "MM月DD日", "source_date": "前一日"} + """ + import urllib.request + + # 取前一晚(已播出)的联播:day_str 前一天 + prev_day = (datetime.strptime(day_str, "%Y%m%d") - timedelta(days=1)).strftime("%Y%m%d") + result: dict[str, Any] = {"items": [], "date": "", "source_date": prev_day} + + api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={prev_day}&end_date={prev_day}" + try: + req = urllib.request.Request(api_url) + with urllib.request.urlopen(req, timeout=15) as resp: + body = json.loads(resp.read().decode("utf-8")) + except Exception as e: + logger.warning("新闻联播 API 请求失败: {}", e) + return result + + raw_news = body.get("data", {}).get("news", []) + if not raw_news: + return result + + # 提取日期 + dates = {n.get("news_days", "") for n in raw_news if n.get("news_days")} + if dates: + d = min(dates) + result["date"] = f"{d[5:7]}月{d[8:10]}日" + result["source_date"] = prev_day + + # 转换为事件格式,跳过第 1 条(内容提要/开场白) + events: list[dict] = [] + for n in raw_news: + sid = n.get("daily_sub_id", 0) + if sid <= 1: # 跳过"内容提要" + continue + title = n.get("news_title", "") + content = n.get("news_improve", "") + importance = _score_xwlb_importance(title, content) + events.append({ + "title": title[:100], + "url": "", # 新闻联播无独立文章链接 + "source_id": "xwlb", + "publish_time": n.get("news_days", ""), + "event": { + "stock_codes": [], + "company_names": [], + "industries": [], + "sentiment": "neutral", + "importance": importance, + "event_type": "新闻联播", + "summary": content[:80] if content else title[:80], + }, + }) + + # 按重要度降序 + events.sort(key=lambda e: (-e["event"]["importance"], e["title"])) + result["items"] = events + return result + + +def _load_article_urls_from_index(index_path: Path) -> list[dict]: + """从 index.jsonl 中加载 stage=article 的条目(排除列表页)。""" + if not index_path.is_file(): + return [] + articles: list[dict] = [] + for line in open(index_path, encoding="utf-8"): + try: + obj = json.loads(line) + if obj.get("stage") == "article" and obj.get("success"): + articles.append(obj) + except (json.JSONDecodeError, KeyError): + pass + return articles + + +def _collect_pipeline_stats(day_str: str) -> dict[str, Any]: + """收集管道统计数据(仅计文章级条目 + 24h 抓取新鲜度, 用 fetched_at)。""" + now = datetime.now() + cutoff_24h = now - timedelta(hours=24) + + raw_by_source: dict[str, int] = {} + raw_by_source_24h: dict[str, int] = {} + raw_total = 0 + raw_total_24h = 0 + + for idx in Path("data/raw").glob(f"*/{day_str}/index.jsonl"): + src = idx.parent.parent.name + articles = _load_article_urls_from_index(idx) + n = len(articles) + name = _source_name(src) + raw_by_source[name] = n + raw_total += n + + # 统计 24h 内抓取的文章(用 fetched_at;原实现从 URL 猜日期, 对多数源失效导致全 0) + n_24h = 0 + for art in articles: + fa = art.get("fetched_at") + if fa: + try: + if datetime.fromisoformat(fa) >= cutoff_24h: + n_24h += 1 + except ValueError: + n_24h += 1 # 时间格式异常时保守计入 + else: + n_24h += 1 # 时间缺失时保守计入 + raw_by_source_24h[name] = n_24h + raw_total_24h += n_24h + + # cninfo raw(仅计文章级条目) + cninfo_raw = 0 + for idx in Path("data/raw/cninfo").glob("*/index.jsonl"): + cninfo_raw += len(_load_article_urls_from_index(idx)) + + proc = _count_json(f"data/processed/*/{day_str}/*.json") + deduped = _count_json(f"data/deduped/{day_str}/uniques/*.json") + dup_path = Path(f"data/deduped/{day_str}/duplicates.jsonl") + dups = _count_jsonl(dup_path) + emb_count = _count_json(f"data/embeddings/{day_str}/*.json") + + qdrant_count = 0 + try: + from vectorstore import VectorStore, make_qdrant_client + c = make_qdrant_client() + s = VectorStore(c) + qdrant_count = s.count() + s.close() + except Exception: + pass + + return { + "raw_total": raw_total, + "raw_total_24h": raw_total_24h, + "raw_by_source": raw_by_source, + "raw_by_source_24h": raw_by_source_24h, + "cninfo_raw": cninfo_raw, + "proc": proc, + "deduped": deduped, + "dups": dups, + "emb_count": emb_count, + "qdrant_count": qdrant_count, + } + + +# --------------------------------------------------------------------------- # +# AI 摘要 +# --------------------------------------------------------------------------- # + +def _generate_ai_summary(news: dict, cninfo: dict, day_str: str, + xwlb: dict | None = None) -> str: + """LLM 生成 500 字以内日报摘要,囊括全部新闻、公告及新闻联播。""" + lines: list[str] = [] + + # 新闻联播(如有) + if xwlb and xwlb.get("items"): + items = xwlb["items"] + lines.append(f"## 新闻联播要闻 ({xwlb.get('date', '')}, {len(items)} 条, 按重要度排序)") + for e in items[:10]: + ev = e["event"] + lines.append(f"- [重要度{ev.get('importance', 0)}] {e['title']}") + + # 新闻(已在 _collect_news_events 中过滤) + if news["high"]: + lines.append(f"## 过去 24 小时高重要度新闻 ({len(news['high'])} 条)") + for e in news["high"][:12]: + ev = e["event"] + sentiment = ev.get("sentiment", "") + s_icon = {"positive": "利好", "negative": "利空", "neutral": "中性"}.get(sentiment, "") + lines.append(f"- [{s_icon}][{ev.get('event_type', '')}] {e['title']}。{ev.get('summary', '')}") + + # 公告/调研 + if cninfo["high"]: + lines.append(f"## 近 {CNINFO_DAYS_BACK} 日重要公告/调研 ({len(cninfo['high'])} 条)") + for e in cninfo["high"][:8]: + ev = e.get("event", {}) + lines.append(f"- [{ev.get('event_type', '公司公告')}] {e['title']}") + + if not lines: + return "" + + try: + from llm.client import SCENE_DAILY_REPORT, load_llm_config, make_sync_client + config = load_llm_config(scene=SCENE_DAILY_REPORT) + client = make_sync_client(config) + return _llm_summarize(client, config, lines, day_str) + except Exception as e: + logger.warning("AI 摘要生成失败: {}", e) + return "" + + +def _split_lines_into_chunks(lines: list[str], max_chars: int = 3000) -> list[list[str]]: + """将 lines 按 max_chars 分块,保证每条新闻(line)不被截断。""" + chunks: list[list[str]] = [] + current: list[str] = [] + current_len = 0 + + for line in lines: + line_len = len(line) + 1 # +1 for newline + if current and current_len + line_len > max_chars: + chunks.append(current) + current = [] + current_len = 0 + current.append(line) + current_len += line_len + + if current: + chunks.append(current) + return chunks + + +def _llm_summarize(client, config: LLMConfig, lines: list[str], day_str: str) -> str: + """LLM 摘要:单块直接总结,多块先分段总结再合并。 + + config 为 llm.client.LLMConfig(daily_report 场景),提供 model / temperature。 + """ + chunks = _split_lines_into_chunks(lines) + + if len(chunks) == 1: + return _llm_call(client, config, _build_prompt(chunks[0], day_str)) + + # 多块:每块独立总结 + partials: list[str] = [] + for i, chunk in enumerate(chunks, 1): + prompt = f"""以下是今日日报素材的第 {i}/{len(chunks)} 部分 (共 {len(lines)} 条, 本批 {len(chunk)} 条),请用要点总结,每条一行,以 "- " 开头: + +{chr(10).join(chunk)} + +直接输出要点列表:""" + result = _llm_call(client, config, prompt, max_tokens=800) + if result: + partials.append(result) + logger.info("AI 摘要: 分块 {}/{} 完成 ({} 字)", i, len(chunks), len(result)) + + if not partials: + logger.warning("AI 摘要: 所有分块均返回空") + return "" + + if len(partials) < len(chunks): + logger.warning("AI 摘要: {}/{} 分块返回空, 仅合并成功部分", len(chunks) - len(partials), len(chunks)) + + # 合并:将各块摘要合成最终日报摘要 + merge_prompt = f"""以下是 {len(partials)} 组分段摘要,请合并为一份完整的日报摘要 ({day_str}): + +{chr(10).join(f'--- 第{i+1}组 ---{chr(10)}{p}' for i, p in enumerate(partials))} + +请合并为要点总结,每条一行以 "- " 开头,要求: +1. 前 3 条为影响最大的事件,说明为什么重要 +2. 汇总近 {CNINFO_DAYS_BACK} 日公司公告/调研核心信息 +3. 市场情绪基调(利好/利空/中性) +4. 值得持续关注的行业或主题 +5. 纯要点,不要开场白/结束语 +6. 总字数 500 字以内 + +直接输出要点列表:""" + return _llm_call(client, config, merge_prompt, max_tokens=1500) + + +def _build_prompt(lines: list[str], day_str: str) -> str: + """构建标准日报摘要 prompt。""" + return f"""以下是今日需要总结的全部内容(含新闻联播、财经新闻、公司公告),请据此生成日报摘要 ({day_str}): + +{chr(10).join(lines)} + +请用要点总结,每条一行,以 "- " 开头,要求: +1. 前 3 条为过去 24 小时影响最大的事件(优先参考新闻联播中的重大政策信号),说明为什么重要 +2. 汇总近 {CNINFO_DAYS_BACK} 日重要公司公告/调研的核心信息 +3. 市场情绪基调(利好/利空/中性) +4. 值得持续关注的行业或主题 +5. 纯要点,不要开场白/结束语/标题 +6. 总字数控制在 500 字以内 + +直接输出要点列表:""" + + +def _llm_call(client, config: LLMConfig, prompt: str, max_tokens: int = 1500) -> str: + """单次 LLM 调用(带重试),返回 strip 后的文本。 + + config 为 llm.client.LLMConfig(daily_report 场景),提供 model / temperature。 + 失败按指数退避重试 `_LLM_RETRY_TIMES` 次(默认 3),全部失败则抛出最后一次异常。 + 若 finish_reason 为 'length' 则说明达到 max_tokens 上限被截断。 + """ + last_exc: Exception | None = None + for attempt in range(_LLM_RETRY_TIMES): + try: + resp = client.chat.completions.create( + model=config.model, + messages=[ + {"role": "system", "content": "你是 A 股日报撰写助手,输出简洁、有洞察的新闻摘要。"}, + {"role": "user", "content": prompt}, + ], + temperature=config.temperature, + max_tokens=max_tokens, + ) + content = (resp.choices[0].message.content or "").strip() + finish = getattr(resp.choices[0], "finish_reason", None) + if finish == "length": + logger.warning( + "AI 摘要可能被截断: max_tokens={} finish_reason=length 实际输出 {} 字符", + max_tokens, len(content), + ) + return content + except Exception as e: + last_exc = e + if attempt < _LLM_RETRY_TIMES - 1: + wait = _LLM_RETRY_BACKOFF_SEC * (2 ** attempt) + logger.warning( + "AI 摘要 LLM 调用失败(第 {}/{} 次): {}; {} 秒后重试", + attempt + 1, _LLM_RETRY_TIMES, e, round(wait, 2), + ) + time.sleep(wait) + logger.error("AI 摘要 LLM 调用重试 {} 次仍失败: {}", _LLM_RETRY_TIMES, last_exc) + assert last_exc is not None + raise last_exc + + +# --------------------------------------------------------------------------- # +# HTML 渲染 +# --------------------------------------------------------------------------- # + +_HTML_TEMPLATE = """<!DOCTYPE html> +<html lang="zh-CN"> +<head> +<meta charset="UTF-8"> +<meta name="viewport" content="width=device-width, initial-scale=1.0"> +<title>A 股 Deep Research 日报 — {date}_{time} + + + +
+
+

📊 A 股 Deep Research 日报

+

{date} · 生成于 {generated_at}

+
+
+
+ + +

一、AI 摘要

+
{ai_summary}
+ + +{xwlb_section} + + +

三、🔥 重要事件:新闻 ({raw_total_24h}/{raw_total} 篇 24h 内, importance ≥ {news_threshold}, 共 {news_high_count} 篇)

+{news_table} + + +

四、📋 重要事件:公告 / 调研 / 互动 (近 {cninfo_days} 日, importance ≥ {cninfo_threshold}, 共 {cninfo_high_count} 篇)

+{cninfo_table} + + +

五、数据总览

+ +

5.1 M1 → M6 管道

+
+
{raw_total} ({raw_total_24h} 24h)
M1 原始文章
+
{cninfo_raw}
M1 cninfo
+
{proc}
M2 正文提取
+
{deduped}
M3 去重唯一
+
{emb_count}
M5 向量
+
{qdrant_count}
M6 Qdrant
+
+ +

5.2 各源数据 ({raw_total_24h}/{raw_total} 篇 24h 内)

+
+{source_cards} +
+ +

5.3 情绪分布 (最新抓取)

+{sentiment_section} + +

5.4 重要度分布

+ + {importance_headers} + {importance_counts} +
重要度
数量
+ +

5.5 事件类型分布

+ + + {event_type_rows} +
事件类型数量
+ +
+
+
+

A 股 Deep Research 私有投研平台 · 自动生成于 {generated_at}

+
+
+ +""" + + +def _render_event_table(events: list[dict], show_source: bool = True, + show_summary: bool = True) -> str: + """渲染事件表格。""" + if not events: + return "

暂无符合条件的数据

" + rows: list[str] = [] + wl_codes = _load_watchlist_codes() + for i, e in enumerate(events, 1): + ev = e["event"] + sentiment = ev.get("sentiment", "") + icon = {"positive": "🟢", "negative": "🔴", "neutral": "⚪"}.get(sentiment, "") + badge_cls = {"positive": "badge-pos", "negative": "badge-neg"}.get(sentiment, "badge-neu") + imp = ev.get("importance", 0) + imp_cls = f"imp-{imp}" if imp >= 4 else "" + title = e["title"][:70] + src = _source_name(e.get("source_id", "")) + codes_in_event = {c.strip().split(".")[0] for c in (ev.get("stock_codes") or [])} + star = "⭐ " if codes_in_event & wl_codes else "" + code_str = f" ({','.join(list(codes_in_event)[:3])})" if codes_in_event else "" + + url = e.get("url", "") + title_cell = f'{star}
{title}{code_str}' if url else f"{star}{title}{code_str}" + + cols = [ + f"{i}", + f'{icon}', + f"{title_cell}", + ] + if show_source: + cols.append(f"{src}") + cols.append(f'{imp}') + cols.append(f"{ev.get('event_type', '')}") + if show_summary: + cols.append(f"{(ev.get('summary', '') or '')[:60]}") + + rows.append(f'{"".join(cols)}') + + headers = ["#", "", "标题"] + if show_source: + headers.append("源") + headers += ["重要度", "事件类型"] + if show_summary: + headers.append("摘要") + header_row = "".join(f"{h}" for h in headers) + return f"{header_row}{''.join(rows)}
" + + +def _render_source_cards(raw_by_source: dict[str, int], + raw_by_source_24h: dict[str, int], + cninfo_raw: int) -> str: + """渲染源数据卡片,每行 6 个,显示总量和 24h 新鲜数。""" + cards: list[str] = [] + # 新闻源 + for name, count in sorted(raw_by_source.items()): + fresh = raw_by_source_24h.get(name, 0) + cards.append( + f'
' + f'
{count} ({fresh} 24h)
' + f'
{name}
' + f'
' + ) + # cninfo + cards.append( + f'
' + f'
{cninfo_raw}
' + f'
📋 cninfo
' + f'
' + ) + return "\n".join(cards) + + +def _render_xwlb_section(xwlb: dict | None) -> str: + """渲染新闻联播要闻 HTML 板块(重要事件格式,按重要度排序)。""" + if not xwlb or not xwlb.get("items"): + return "" + + items = xwlb["items"] + date_label = xwlb.get("date", "") + source_date = xwlb.get("source_date", "") + + table_html = _render_event_table(items, show_source=False, show_summary=False) + + return f"""

二、📺 新闻联播 ({date_label}, 共 {len(items)} 条, 按重要度排序)

+{table_html} +

+ 来源: 央视《新闻联播》· 数据取自 doorcome API (xwlbFine) · {source_date} +

""" + + +def _render_html(news: dict, cninfo: dict, pipeline: dict, + ai_summary: str, day_str: str, + xwlb: dict | None = None) -> str: + """组装完整 HTML(M10 起废弃:日报已改为结构化入库,此函数不再被调用,保留以便回退)。""" + + # AI 摘要 → HTML + summary_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", ai_summary) + summary_html = _re.sub(r"\*(.+?)\*", r"\1", summary_html) + summary_html = _re.sub(r"`(.+?)`", r"\1", summary_html) + if summary_html.strip(): + lines = summary_html.strip().splitlines() + if any(ln.strip().startswith("- ") for ln in lines): + items = [] + for ln in lines: + s = ln.strip() + if s.startswith("- "): + items.append(f"
  • {s[2:]}
  • ") + elif s: + items.append(f"
  • {s}
  • ") + summary_html = f"
      {''.join(items)}
    " + else: + summary_html = summary_html.replace("\n", "
    ") + else: + summary_html = "

    AI 摘要暂不可用

    " + + # 新闻表格 + news_table = _render_event_table(news["high"]) + + # cninfo 表格 + cninfo_table = _render_event_table(cninfo["high"], show_source=False, show_summary=False) + + # 源数据卡片 + source_cards = _render_source_cards( + pipeline["raw_by_source"], + pipeline.get("raw_by_source_24h", {}), + pipeline["cninfo_raw"], + ) + + # 情绪 + s = news["sentiments"] + pos = s.get("positive", 0) + neg = s.get("negative", 0) + neu = s.get("neutral", 0) + total_s = max(pos + neg + neu, 1) + sentiment_section = ( + f'
    ' + f'
    ' + f'
    ' + f'
    ' + f'
    ' + f'
    ' + f'🟢 利好 {pos} ({pos/total_s:.0%})' + f'🔴 利空 {neg} ({neg/total_s:.0%})' + f'⚪ 中性 {neu} ({neu/total_s:.0%})' + f'
    ' + ) + + # 重要度 + imps = news["importances"] + imp_keys = sorted(imps.keys()) + importance_headers = "".join(f"等级 {k}" for k in imp_keys) + importance_counts = "".join(f"{imps[k]}" for k in imp_keys) + + # 事件类型 + et = news["event_types"] + event_type_rows = "\n".join( + f"{k}{v}" for k, v in et.items() + ) + + return _HTML_TEMPLATE.format( + date=day_str, + time=datetime.now().strftime("%H%M"), + generated_at=datetime.now().strftime("%Y-%m-%d %H:%M:%S"), + ai_summary=summary_html, + xwlb_section=_render_xwlb_section(xwlb) if xwlb else "", + news_high_count=len(news["high"]), + news_threshold=news.get("hi_threshold", 4), + news_table=news_table, + cninfo_high_count=len(cninfo["high"]), + cninfo_threshold=cninfo.get("hi_threshold", 4), + cninfo_days=CNINFO_DAYS_BACK, + cninfo_table=cninfo_table, + raw_total=pipeline["raw_total"], + raw_total_24h=pipeline.get("raw_total_24h", pipeline["raw_total"]), + cninfo_raw=pipeline["cninfo_raw"], + proc=pipeline["proc"], + deduped=pipeline["deduped"], + emb_count=pipeline["emb_count"], + qdrant_count=pipeline["qdrant_count"], + source_cards=source_cards, + sentiment_section=sentiment_section, + importance_headers=importance_headers, + importance_counts=importance_counts, + event_type_rows=event_type_rows, + ) + + +# --------------------------------------------------------------------------- # +# 生成 + 上传 +# --------------------------------------------------------------------------- # + +def _build_report_data(news: dict, cninfo: dict, pipeline: dict, + ai_summary: str, day_str: str, + xwlb: dict | None = None) -> ReportData: + """组装结构化日报数据(M10:写入 MySQL 的前置步骤)。 + + 事件板块映射:news["high"]→news / cninfo["high"]→cninfo / xwlb["items"]→xwlb。 + 数据总览统计以 JSON 快照存入 stats(前端自行解析)。 + """ + events: list[EventRow] = [] + + def _rows(items: list[dict], section: str) -> None: + for i, e in enumerate(items, 1): + ev = e.get("event", {}) + events.append( + EventRow( + section=section, + rank=i, + importance=ev.get("importance"), + event_type=ev.get("event_type"), + title=str(e.get("title", ""))[:512], + summary=(ev.get("summary") or None), + sentiment=ev.get("sentiment") or None, + source=e.get("source_id") or None, + sources=e.get("sources") or None, + url=e.get("url") or None, + ) + ) + + _rows(news.get("high", []), "news") + _rows(cninfo.get("high", []), "cninfo") + if xwlb: + _rows(xwlb.get("items", []), "xwlb") + + stats: dict[str, Any] = { + "pipeline": pipeline, + "news": { + "total": news.get("total", 0), + "hi_threshold": news.get("hi_threshold"), + "sentiments": news.get("sentiments", {}), + "importances": news.get("importances", {}), + "event_types": news.get("event_types", {}), + }, + "cninfo": { + "total": cninfo.get("total", 0), + "hi_threshold": cninfo.get("hi_threshold"), + "by_day": cninfo.get("by_day", {}), + "announcement": cninfo.get("announcement", 0), + "research": cninfo.get("research", 0), + "irm": cninfo.get("irm", 0), + }, + } + if xwlb: + stats["xwlb"] = {"total": len(xwlb.get("items", [])), "date": xwlb.get("date", "")} + + return ReportData( + report_date=datetime.strptime(day_str, "%Y%m%d").date(), + report_type="finance", + file_name="", # 新生成日报唯一键退化为 (report_date, finance, "") + generated_at=datetime.now(), + ai_summary=ai_summary or None, + stats=stats, + events=events, + ) + + +def generate_report(day_str: str | None = None, *, upload: bool = True) -> int | None: + """生成每日日报并结构化入库(M10 完全切换,不再生成 HTML)。 + + `upload` 参数保留以兼容 scheduler/pipeline.py 调用,已无实际作用。 + 返回 report_id(成功)或 None(无数据/失败)。 + """ + day_str = day_str or date.today().strftime("%Y%m%d") + logger.info("生成日报: {}", day_str) + + # 收集数据 + try: + news = _collect_news_events(day_str) + cninfo = _collect_cninfo_events(day_str, days_back=CNINFO_DAYS_BACK) + pipeline = _collect_pipeline_stats(day_str) + xwlb = _collect_xwlb(day_str) + except Exception as e: + logger.exception("收集日报数据失败: {}", e) + return None + + if news["total"] == 0 and cninfo["total"] == 0 and not xwlb.get("items"): + logger.warning("{} 无数据,跳过日报生成", day_str) + return None + + # AI 摘要(新闻联播 + 新闻 + cninfo) + ai_summary = _generate_ai_summary(news, cninfo, day_str, xwlb=xwlb) + + # 结构化入库(替代原 HTML 渲染 + 上传) + report = _build_report_data(news, cninfo, pipeline, ai_summary, day_str, xwlb=xwlb) + try: + from report_db import connect, save_report + + conn = connect() + try: + report_id = save_report(conn, report) + finally: + conn.close() + except Exception as e: + logger.exception("日报入库失败: {}", e) + return None + + logger.info("日报已入库: report_id={}", report_id) + return report_id + + +def _upload(html_path: Path, file_tag: str) -> bool: + """上传 HTML 报告到 Web 服务器(M10 起废弃:不再被调用,保留以便回退)。""" + today_str = date.today().strftime("%Y%m%d") + remote_dir = f"{UPLOAD_BASE}/{today_str}/" + logger.info("上传日报到 {}:{}", UPLOAD_HOST, remote_dir) + + try: + r1 = subprocess.run( + ["ssh", UPLOAD_HOST, f"mkdir -p {remote_dir}"], + timeout=15, capture_output=True, text=True, + ) + if r1.returncode != 0: + logger.warning("创建远程目录失败: {}", r1.stderr.strip()) + return False + r2 = subprocess.run( + ["scp", str(html_path), f"{UPLOAD_HOST}:{remote_dir}finance_news_daily_{file_tag}.html"], + timeout=30, capture_output=True, text=True, + ) + if r2.returncode != 0: + logger.warning("上传日报失败: {}", r2.stderr.strip()) + return False + logger.info("上传完成: http://doorcome.cn/echart/research/{}/", today_str) + return True + except Exception as e: + logger.warning("上传日报异常(不阻塞): {}", e) + return False diff --git a/scheduler/stock_reporter.py b/scheduler/stock_reporter.py new file mode 100644 index 0000000..b6e57b4 --- /dev/null +++ b/scheduler/stock_reporter.py @@ -0,0 +1,534 @@ +"""个股日报生成器 v2.0。 + +根据 watchlist.yaml 配置,为每只关注股票生成日报: + - AI 要点分析(DeepSeek 生成) + - 公告 / 调研 / 互动问答(cninfo v2.0 CninfoItem 数据) + - 相关新闻(Qdrant 语义检索) + - HTML 报告 + 自动上传 + +数据来源: + - cninfo 数据: data/raw/cninfo/{YYYYMMDD}/*.json (CninfoItem v2.0 格式) + - 新闻: Qdrant 向量检索 +""" + +from __future__ import annotations + +import json +import os as _os +import re +import subprocess +import time as _time +from datetime import date, datetime, timedelta +from pathlib import Path +from typing import Any + +from loguru import logger + +UPLOAD_HOST = "simon@doorcome.cn" +UPLOAD_BASE = "/var/www/html/echart/research" + + +# --------------------------------------------------------------------------- # +# 配置 +# --------------------------------------------------------------------------- # + +def _load_source_names() -> dict[str, str]: + import yaml + try: + with open("configs/sources.yaml", encoding="utf-8") as f: + data = yaml.safe_load(f) + names = {s["id"]: s["name"] for s in (data.get("sources") or []) if s.get("id")} + except Exception: + names = {} + names["cninfo"] = "巨潮资讯网" + return names + + +_SOURCE_NAMES = _load_source_names() + + +def _load_watchlist() -> list[dict]: + import yaml + try: + with open("configs/watchlist.yaml", encoding="utf-8") as f: + data = yaml.safe_load(f) or {} + return list(data.get("watchlist") or []) + except Exception: + return [] + + +def _report_days() -> int: + """从 .env 读取报告天数,默认 15。""" + return int(_os.environ.get("STOCK_REPORT_DAYS", "15")) + + +# --------------------------------------------------------------------------- # +# cninfo 数据读取 (v2.0 CninfoItem 格式) +# --------------------------------------------------------------------------- # + +def _read_cninfo_items( + code: str, + days_back: int | None = None, + item_type: str | None = None, +) -> list[dict]: + """从 data/raw/cninfo/ 中读取指定股票的 CninfoItem JSON 数据。 + + Args: + code: 6 位股票代码 + days_back: 向前追溯天数,为 None 则使用 STOCK_REPORT_DAYS + item_type: 过滤类型 None=全部, announcement/research/irm + """ + if days_back is None: + days_back = _report_days() + + since = date.today() - timedelta(days=days_back) + since_str = since.strftime("%Y-%m-%d") + raw_root = Path("data/raw/cninfo") + if not raw_root.is_dir(): + logger.warning("cninfo raw 目录不存在: {}", raw_root) + return [] + + items: list[dict] = [] + + for day_dir in sorted(raw_root.glob("*"), reverse=True): + # 解析目录日期(抓取日期),用于早期跳出循环 + try: + day_str = day_dir.name + if len(day_str) != 8: + continue + # 目录日期仅用于性能优化:如果目录日期太旧(>days_back*2),跳过 + day_date = date(int(day_str[:4]), int(day_str[4:6]), int(day_str[6:])) + if day_date < since - timedelta(days=days_back): + continue + except ValueError: + continue + + # 从此日期的 index.jsonl 读取 + idx_path = day_dir / "index.jsonl" + if not idx_path.is_file(): + # 直接读取 JSON 文件 + for jf in sorted(day_dir.glob("*.json"), reverse=True): + item = _try_load_cninfo_item(jf, code, item_type) + if item: + pt = (item.get("publish_time") or "").strip() + if pt and pt >= since_str: + items.append(item) + elif not pt: + # publish_time 为空(如 irm),仍纳入但标记 + items.append(item) + else: + with idx_path.open("r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + + # 按股票和类型过滤 + if rec.get("stock_code") != code: + continue + if item_type and rec.get("item_type") != item_type: + continue + + pt = (rec.get("publish_time") or "").strip() + # 按 publish_time 过滤(而非抓取日期) + if pt and pt < since_str: + continue + + items.append({ + "title": (rec.get("title") or "").strip(), + "url": (rec.get("url") or "").strip(), + "source": "巨潮资讯网", + "score": 1.0, + "publish_time": pt, + "item_type": (rec.get("item_type") or "").strip(), + "event": rec.get("extra", {}), + }) + + # 限制同一天/同一类型最多取 50 条 + if len(items) >= 50: + break + + return items + + +def _try_load_cninfo_item(json_path: Path, code: str, + item_type: str | None) -> dict | None: + """从单个 CninfoItem JSON 文件加载(无 index.jsonl 时的回退)。""" + try: + data = json.loads(json_path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError): + return None + if data.get("stock_code") != code: + return None + if item_type and data.get("item_type") != item_type: + return None + return { + "title": (data.get("title") or "").strip(), + "url": (data.get("url") or "").strip(), + "source": "巨潮资讯网", + "score": 1.0, + "publish_time": (data.get("publish_time") or "").strip(), + "item_type": (data.get("item_type") or "").strip(), + "event": data.get("extra", {}), + } + + +# --------------------------------------------------------------------------- # +# Qdrant 新闻搜索 +# --------------------------------------------------------------------------- # + +def _search_news_from_qdrant( + emb: Any, store: Any, query_text: str, + stock_codes: list[str], top_k: int = 30, + days_back: int | None = None, company_name: str = "", +) -> list[dict]: + """多策略搜索 Qdrant: 股票代码 → 公司名 → 语义。""" + from vectorstore import SearchFilter + + if days_back is None: + days_back = _report_days() + + vec = emb.embed_one(query_text) + since = (date.today() - timedelta(days=days_back)).strftime("%Y-%m-%d") + + # 补全后缀 + codes_with_suffix = [] + for c in stock_codes: + codes_with_suffix.extend([f"{c}.SZ", f"{c}.SH", f"{c}.BJ", c]) + + hits: list = [] + hits = store.query(query_vector=vec, top_k=top_k, + filter=SearchFilter(stock_codes=codes_with_suffix, + publish_date_from=since)) + if not hits and company_name: + hits = store.query(query_vector=vec, top_k=top_k, + filter=SearchFilter(company_names=[company_name], + publish_date_from=since)) + if not hits: + hits = store.query(query_vector=vec, top_k=top_k, + filter=SearchFilter(publish_date_from=since)) + return [ + { + "title": h.title, "url": h.url, + "source": _SOURCE_NAMES.get(h.source_id, h.source_id), + "score": round(h.score, 4), + "publish_time": h.publish_time.isoformat() if h.publish_time else None, + "event": h.event or {}, + } + for h in hits + if _SOURCE_NAMES.get(h.source_id, h.source_id) != "巨潮资讯网" + ] + + +# --------------------------------------------------------------------------- # +# LLM AI 要点分析 +# --------------------------------------------------------------------------- # + +def _generate_ai_summary(company_name: str, announcements: list[dict], + news: list[dict], research: list[dict], + irm: list[dict]) -> str: + """LLM 生成个股要点分析。""" + from llm.client import SCENE_STOCK_REPORT, load_llm_config, make_sync_client + + lines = [] + + if announcements: + lines.append(f"## 公告 ({len(announcements)} 条)") + for a in announcements[:10]: + lines.append(f"- {a['title']} ({a.get('publish_time', '')})") + + if research: + lines.append(f"## 调研 ({len(research)} 条)") + for r in research[:5]: + lines.append(f"- {r['title']} ({r.get('publish_time', '')})") + + if news: + lines.append(f"## 新闻 ({len(news)} 条)") + for n in news[:10]: + ev = n.get("event", {}) + summary = ev.get("summary", "") + lines.append( + f"- [{n['source']}] {n['title']}" + + (f"。{summary}" if summary else "") + ) + + if irm: + lines.append(f"## 互动问答 ({len(irm)} 条)") + for q in irm[:5]: + lines.append(f"- {q['title']}") + + if not lines: + return "暂无足够数据生成 AI 摘要" + + prompt = f"""以下是 {company_name} 近期的公告、调研、新闻和互动问答: + +{chr(10).join(lines)[:3500]} + +请输出 5-8 条要点分析,每条以 "- " 开头: +1. 最重要的公告或事件是什么?影响如何? +2. 近期有哪些值得关注的动态? +3. 市场情绪倾向(利好/利空)? +4. 后续需要关注什么? + +直接输出要点列表:""" + + try: + config = load_llm_config(scene=SCENE_STOCK_REPORT) + client = make_sync_client(config) + resp = client.chat.completions.create( + model=config.model, + messages=[{"role": "user", "content": prompt}], + temperature=config.temperature, max_tokens=500, + ) + return (resp.choices[0].message.content or "").strip() + except Exception as e: + logger.warning("个股 AI 摘要失败: {}", e) + return "AI 摘要暂不可用" + + +# --------------------------------------------------------------------------- # +# HTML 渲染 +# --------------------------------------------------------------------------- # + +def _clean_markdown(text: str) -> str: + """LLM 输出的简单 Markdown 转 HTML 片段。""" + text = re.sub(r"\*\*(.+?)\*\*", r"\1", text) + text = re.sub(r"\*(.+?)\*", r"\1", text) + text = re.sub(r"`(.+?)`", r"\1", text) + return text + + +_HTML_TEMPLATE = """ + + + + +{company_name}({stock_code}) 个股日报 — {report_date} + + + +
    +

    {company_name} ({stock_code}) 个股日报

    +

    报告期间: {date_from} ~ {date_to} · 生成于 {generated_at}

    +
    +
    + +

    一、AI 要点分析

    +
    {ai_summary_html}
    + +

    二、公司公告 ({ann_count} 条)

    +

    近 {report_days} 日公告,来源 巨潮资讯网

    +{ann_table} + +

    三、调研活动 ({research_count} 条)

    +

    近 {report_days} 日投资者关系活动,来源 巨潮资讯网

    +{research_table} + +

    四、相关新闻 ({news_count} 条)

    +

    近 {report_days} 日财经新闻

    +{news_table} + +

    五、互动问答 ({irm_count} 条)

    +

    近 {report_days} 日互动易平台问答

    +{irm_table} + +
    +

    A 股 Deep Research · 个股日报 · {generated_at}

    + +""" + + +def _render_table(items: list[dict], max_rows: int = 10) -> str: + if not items: + return "

    暂无数据

    " + rows = [] + for i, item in enumerate(items[:max_rows], 1): + ev = item.get("event", {}) + sentiment = ev.get("sentiment", "") + badge = {"positive": "badge-pos", "negative": "badge-neg"}.get(sentiment, "badge-neu") + icon = {"positive": "🟢", "negative": "🔴", "neutral": "⚪"}.get(sentiment, "") + short_title = item["title"][:60] + if len(item["title"]) > 60: + short_title += "..." + date_str = (item.get("publish_time") or "")[:10] + url = item.get("url", "") + title_cell = ( + f'{short_title}' + if url else short_title + ) + rows.append( + f'{i}' + f'{icon}' + f'{title_cell}' + f'{item["source"]}' + f'{date_str}' + ) + return ( + f"" + f"{''.join(rows)}
    #标题来源日期
    " + ) + + +# --------------------------------------------------------------------------- # +# 单股报告生成 +# --------------------------------------------------------------------------- # + +def _generate_stock_report_with_backend(stock: dict, store: Any, emb: Any, *, + upload: bool = True) -> Path | None: + """为单个股票生成日报(使用共享 Qdrant backend)。""" + code = stock["code"] + name = stock["name"] + days = _report_days() + + logger.info("生成个股报告: {} ({}) 近{}日", code, name, days) + + # ---- 从 cninfo v2.0 数据读取 ---- + ann_items = _read_cninfo_items(code, days_back=days, item_type="announcement") + research_items = _read_cninfo_items(code, days_back=days, item_type="research") + irm_items = _read_cninfo_items(code, days_back=days, item_type="irm") + + # ---- 新闻: Qdrant 语义检索 ---- + news_items = _search_news_from_qdrant( + emb, store, f"{name} {code}", stock_codes=[code], + top_k=30, days_back=days, company_name=name, + ) + + # ---- AI 摘要 ---- + ai = _generate_ai_summary(name, ann_items, news_items, research_items, irm_items) + ai = _clean_markdown(ai) + ai_html = ( + "
      " + "".join( + f"
    • {ln[2:]}
    • " if ln.startswith("- ") else f"
    • {ln}
    • " + for ln in ai.strip().splitlines() if ln.strip() + ) + "
    " + if ai else "

    AI 摘要暂不可用

    " + ) + + # ---- 渲染 HTML ---- + today = date.today() + start_date = today - timedelta(days=days) + html = _HTML_TEMPLATE.format( + company_name=name, stock_code=code, + report_date=today.strftime("%Y-%m-%d"), + date_from=start_date.strftime("%Y-%m-%d"), + date_to=today.strftime("%Y-%m-%d"), + generated_at=datetime.now().strftime("%Y-%m-%d %H:%M:%S"), + report_days=days, + ai_summary_html=ai_html, + ann_count=len(ann_items), + ann_table=_render_table(ann_items), + research_count=len(research_items), + research_table=_render_table(research_items, max_rows=10), + news_count=len(news_items), + news_table=_render_table(news_items), + irm_count=len(irm_items), + irm_table=_render_table(irm_items, max_rows=10), + ) + + # ---- 保存 ---- + out_dir = Path("data/reports/stocks") + out_dir.mkdir(parents=True, exist_ok=True) + fname = f"{code}_{name}_个股日报_{today.strftime('%Y%m%d')}.html" + html_path = out_dir / fname + html_path.write_text(html, encoding="utf-8") + logger.info("个股报告已保存: {} ({} KB)", html_path, len(html) // 1024) + + # ---- 上传 ---- + if upload: + _upload_stock_report(html_path, today.strftime("%Y%m%d")) + + return html_path + + +# --------------------------------------------------------------------------- # +# 上传 +# --------------------------------------------------------------------------- # + +def _upload_stock_report(html_path: Path, day_str: str) -> bool: + """上传个股报告到 Web 服务器。""" + remote_dir = f"{UPLOAD_BASE}/{day_str}/" + try: + r1 = subprocess.run( + ["ssh", UPLOAD_HOST, f"mkdir -p {remote_dir}"], + timeout=15, capture_output=True, text=True, + ) + r2 = subprocess.run( + ["scp", str(html_path), f"{UPLOAD_HOST}:{remote_dir}{html_path.name}"], + timeout=30, capture_output=True, text=True, + ) + return r1.returncode == 0 and r2.returncode == 0 + except Exception: + return False + + +# --------------------------------------------------------------------------- # +# 批量生成主入口 +# --------------------------------------------------------------------------- # + +def generate_all_stock_reports(upload: bool = True) -> int: + """为关注列表中所有股票生成个股日报。返回生成的报告数。""" + watchlist = _load_watchlist() + if not watchlist: + logger.warning("关注列表为空,跳过个股报告") + return 0 + + # 共享 backend(Qdrant + Embedder, 加锁重试) + from dotenv import load_dotenv + load_dotenv() + from embedding import make_sync_provider + from vectorstore import VectorStore, make_qdrant_client + + emb = make_sync_provider() + for retry in range(5): + try: + client = make_qdrant_client() + store = VectorStore(client) + break + except RuntimeError: + if retry < 4: + logger.warning("Qdrant 被占用,{} 秒后重试...", (retry + 1) * 2) + _time.sleep((retry + 1) * 2) + else: + raise + + count = 0 + for stock in watchlist: + code = stock.get("code", "") + name = stock.get("name", "") + try: + path = _generate_stock_report_with_backend(stock, store, emb, upload=upload) + if path: + count += 1 + except Exception as e: + logger.error("个股报告生成失败 {} {}: {}", code, name, e) + + store.close() + emb.close() + logger.info("个股报告完成: {}/{} 家", count, len(watchlist)) + return count diff --git a/scripts/__init__.py b/scripts/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/scripts/a-share-research.service b/scripts/a-share-research.service new file mode 100644 index 0000000..d33c825 --- /dev/null +++ b/scripts/a-share-research.service @@ -0,0 +1,25 @@ +[Unit] +Description=A 股 Deep Research 定时任务调度器 +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=pi +WorkingDirectory=/home/pi/news +Environment="PATH=/home/pi/.local/bin:/usr/local/bin:/usr/bin:/bin" +Environment="LANG=en_US.UTF-8" +Environment="LC_ALL=en_US.UTF-8" +Environment="PYTHONIOENCODING=utf-8" +ExecStart=/home/pi/.local/bin/uv run python -m scripts.run_scheduler +Restart=on-failure +RestartSec=30 +StandardOutput=append:/home/pi/news/logs/scheduler.log +StandardError=append:/home/pi/news/logs/scheduler_error.log + +# 优雅退出(等当前 pipeline 跑完) +KillSignal=SIGINT +TimeoutStopSec=900 + +[Install] +WantedBy=multi-user.target diff --git a/scripts/run_cninfo.py b/scripts/run_cninfo.py new file mode 100644 index 0000000..f0d2d2c --- /dev/null +++ b/scripts/run_cninfo.py @@ -0,0 +1,67 @@ +"""cninfo 公告抓取入口。 + +用法: + uv run python -m scripts.run_cninfo # 抓取今天公告 + uv run python -m scripts.run_cninfo --days 3 # 抓取最近 3 天 + uv run python -m scripts.run_cninfo --max-pages 50 # 最多 50 页 + uv run python -m scripts.run_cninfo --no-save # 试跑不存盘 +""" + +from __future__ import annotations + +import argparse +import sys +from datetime import date, timedelta +from pathlib import Path + +from dotenv import load_dotenv +from loguru import logger + +from crawler.cninfo import crawl_announcements + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add(sys.stderr, level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}") + log_path = Path("logs") / "cninfo.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +def main() -> int: + load_dotenv() + parser = argparse.ArgumentParser(description="cninfo 巨潮资讯网公告抓取") + parser.add_argument("--days", type=int, default=0, + help="抓取最近 N 天的公告,默认读 CNINFO_DAYS_BACK") + parser.add_argument("--start", default=None, help="开始日期 YYYY-MM-DD") + parser.add_argument("--end", default=None, help="结束日期 YYYY-MM-DD") + parser.add_argument("--max-pages", type=int, default=0, help="最大页数") + parser.add_argument("--no-save", action="store_true", help="仅抓取不存盘") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + + start = args.start + end = args.end + if args.days > 0 and not start: + start = (date.today() - timedelta(days=args.days)).strftime("%Y-%m-%d") + if not end: + end = date.today().strftime("%Y-%m-%d") + + results = crawl_announcements( + start_date=start, + end_date=end, + max_pages=args.max_pages or None, + save=not args.no_save, + ) + + if not results: + return 1 + logger.info("cninfo 抓取完成: {} 条", len(results)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_crawler.py b/scripts/run_crawler.py new file mode 100644 index 0000000..f6523da --- /dev/null +++ b/scripts/run_crawler.py @@ -0,0 +1,81 @@ +"""M1 抓取入口脚本。 + +用法: + uv run python -m scripts.run_crawler + uv run python -m scripts.run_crawler --source cls + uv run python -m scripts.run_crawler --config configs/sources.yaml --no-save +""" + +from __future__ import annotations + +import argparse +import asyncio +import sys +from pathlib import Path + +from crawl4ai import AsyncWebCrawler, BrowserConfig +from loguru import logger + +from crawler import crawl_all, crawl_source, load_crawler_config + + +def _setup_logger(level: str) -> None: + """配置 loguru,输出到控制台与 logs/crawler.log。""" + logger.remove() + logger.add(sys.stderr, level=level, format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}") + log_path = Path("logs") / "crawler.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add( + log_path, + level="DEBUG", + rotation="10 MB", + retention=5, + encoding="utf-8", + enqueue=True, + ) + + +async def _run(args: argparse.Namespace) -> int: + config = load_crawler_config(args.config) + + if args.source: + source = next((s for s in config.sources if s.id == args.source), None) + if source is None: + logger.error("未找到源 id={}", args.source) + return 2 + if not source.enabled: + logger.warning("源 {} 在 yaml 中标记为 disabled,本次仍将抓取", source.id) + + browser_config = BrowserConfig( + headless=config.settings.headless, + user_agent=config.settings.user_agent, + verbose=False, + ) + sem = asyncio.Semaphore(config.settings.concurrency) + async with AsyncWebCrawler(config=browser_config) as crawler: + results = await crawl_source(crawler, source, config.settings, sem, save=not args.no_save) + else: + results = await crawl_all(config, save=not args.no_save) + + succ = sum(1 for r in results if r.success) + total = len(results) + rate = succ / max(total, 1) + logger.info("抓取结束 总 {} 成功 {} 成功率 {:.0%}", total, succ, rate) + # 验收标准: 成功率 >= 90% + return 0 if rate >= 0.9 else 1 + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股新闻抓取(M1)") + parser.add_argument("--config", default="configs/sources.yaml", help="sources.yaml 路径") + parser.add_argument("--source", default=None, help="只抓单个源 id,默认抓全部启用源") + parser.add_argument("--no-save", action="store_true", help="不写入本地文件,仅试跑") + parser.add_argument("--log-level", default="INFO", help="DEBUG/INFO/WARNING/ERROR") + args = parser.parse_args() + + _setup_logger(args.log_level) + return asyncio.run(_run(args)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_dedup.py b/scripts/run_dedup.py new file mode 100644 index 0000000..c84e93d --- /dev/null +++ b/scripts/run_dedup.py @@ -0,0 +1,275 @@ +"""M3 批量去重入口脚本。 + +输入: data/processed/{source}/{YYYYMMDD}/*.json (M2 产物) +输出: + - 指纹库:data/dedup/fingerprints.sqlite3 (source_ids 列记录多源) + - 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json (含 sources 多源字段) + - 多源记录:data/deduped/{YYYYMMDD}/sources.json + {url_hash: [source_id, ...]},一条唯一新闻的全部来源 + - 重复记录:data/deduped/{YYYYMMDD}/duplicates.jsonl + (含 matched_source_id / matched_source_ids) + +用法: + uv run python -m scripts.run_dedup # 处理今日全部源 + uv run python -m scripts.run_dedup --date 20260616 + uv run python -m scripts.run_dedup --source sina --date 20260616 + uv run python -m scripts.run_dedup --reset # 清空指纹库重新建立 +""" + +from __future__ import annotations + +import argparse +import json +import sys +from collections import Counter +from datetime import date +from pathlib import Path + +from loguru import logger +from pydantic import ValidationError + +from dedup import Deduper +from extractor import Article + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + log_path = Path("logs") / "dedup.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +def _load_article(json_path: Path) -> Article | None: + try: + data = json.loads(json_path.read_text(encoding="utf-8")) + return Article.model_validate(data) + except (json.JSONDecodeError, ValidationError) as e: + logger.warning("跳过无法解析的 article 文件 {}: {}", json_path, e) + return None + + +def _list_source_dirs(processed_root: Path) -> list[str]: + if not processed_root.is_dir(): + return [] + return sorted(p.name for p in processed_root.iterdir() if p.is_dir()) + + +def _write_unique( + url_hash: str, + article: Article, + uniques_dir: Path, + sources_map: dict[str, list[str]], +) -> None: + """写 uniques JSON,附加 sources 多源字段(向后兼容:下游 Pydantic 忽略多余字段)。""" + data = json.loads(article.model_dump_json()) + data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [article.source_id]))) + out_path = uniques_dir / f"{url_hash}.json" + out_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8") + + +def _update_unique_sources( + url_hash: str, + uniques_dir: Path, + sources_map: dict[str, list[str]], +) -> None: + """仅更新已存在 uniques 文件的 sources 字段(不覆盖原文内容)。 + + 跨日命中时对应 uniques 文件在历史日期目录,不在本次处理范围,以指纹库为准。 + """ + uniq_path = uniques_dir / f"{url_hash}.json" + if not uniq_path.is_file(): + return + try: + data = json.loads(uniq_path.read_text(encoding="utf-8")) + data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, []))) + uniq_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8") + except (json.JSONDecodeError, OSError) as e: + logger.warning("更新 uniques 多源失败 {}: {}", uniq_path, e) + + +def _merge_sources(url_hash: str, new_source: str, sources_map: dict[str, list[str]]) -> None: + """把新源并入 url_hash 的源列表(去重保序,主源居首)。""" + cur = sources_map.setdefault(url_hash, []) + if new_source not in cur: + cur.append(new_source) + + +def _process_source_day( + source_id: str, + day: str, + processed_root: Path, + out_root: Path, + deduper: Deduper, + sources_map: dict[str, list[str]], +) -> tuple[int, int, Counter]: + """处理单源单日。返回 (uniques, duplicates, layer_counter)。 + + sources_map: 本次去重涉及内容组的 url_hash -> 全部来源列表(跨源累积, + 最终写入 data/deduped/{day}/sources.json,供「显示新闻源」使用; + 跨日命中的历史内容组也会记录,权威多源以指纹库 source_ids 列为准)。 + """ + src_dir = processed_root / source_id / day + if not src_dir.is_dir(): + logger.info("源 {} 日期 {} 无 processed 目录,跳过", source_id, day) + return 0, 0, Counter() + + files = sorted(src_dir.glob("*.json")) + if not files: + logger.info("源 {} 日期 {} 无文章,跳过", source_id, day) + return 0, 0, Counter() + + uniques_dir = out_root / day / "uniques" + uniques_dir.mkdir(parents=True, exist_ok=True) + dup_log = out_root / day / "duplicates.jsonl" + + uniq_cnt = 0 + dup_cnt = 0 + layer_cnt: Counter = Counter() + + with dup_log.open("a", encoding="utf-8") as dup_f: + for fp in files: + article = _load_article(fp) + if article is None: + continue + result = deduper.ingest(article) + if result.is_duplicate: + dup_cnt += 1 + if result.matched_layer is not None: + layer_cnt[result.matched_layer.value] += 1 + # 记录多源:把被去重文章的源并入对应唯一新闻 + if result.matched_url_hash: + _merge_sources(result.matched_url_hash, article.source_id, sources_map) + # 若该唯一新闻文件在当天目录,同步更新其 sources 字段 + _update_unique_sources(result.matched_url_hash, uniques_dir, sources_map) + dup_f.write( + json.dumps( + { + "source_id": article.source_id, + "url": article.url, + "url_hash": article.url_hash, + "title": article.title, + "matched_layer": ( + result.matched_layer.value + if result.matched_layer + else None + ), + "matched_url": result.matched_url, + "matched_url_hash": result.matched_url_hash, + "matched_title": result.matched_title, + "matched_source_id": result.matched_source_id, + "matched_source_ids": result.all_source_ids, + "hamming_distance": result.hamming_distance, + }, + ensure_ascii=False, + ) + + "\n" + ) + else: + uniq_cnt += 1 + sources_map[article.url_hash] = [article.source_id] + _write_unique(article.url_hash, article, uniques_dir, sources_map) + + total = uniq_cnt + dup_cnt + rate = dup_cnt / max(total, 1) + logger.info( + "源 {} 日期 {}: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}", + source_id, + day, + uniq_cnt, + dup_cnt, + rate, + dict(layer_cnt), + ) + return uniq_cnt, dup_cnt, layer_cnt + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股新闻三层去重 (M3)") + parser.add_argument("--processed-root", default="data/processed") + parser.add_argument("--out-root", default="data/deduped") + parser.add_argument("--db", default="data/dedup/fingerprints.sqlite3") + parser.add_argument("--source", default=None, help="只处理单源") + parser.add_argument( + "--date", default=date.today().strftime("%Y%m%d"), help="日期 YYYYMMDD" + ) + parser.add_argument("--simhash-threshold", type=int, default=3) + parser.add_argument("--window-days", type=int, default=30) + parser.add_argument("--reset", action="store_true", help="处理前清空指纹库") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + processed_root = Path(args.processed_root) + out_root = Path(args.out_root) + + sources = [args.source] if args.source else _list_source_dirs(processed_root) + if not sources: + logger.error("{} 下无源目录", processed_root) + return 2 + + with Deduper( + db_path=args.db, + simhash_threshold=args.simhash_threshold, + time_window_days=args.window_days, + ) as deduper: + if args.reset: + logger.warning("--reset:清空指纹库 {}", args.db) + deduper.store.clear() + + # 清掉同日 duplicates.jsonl 避免重复追加(uniques 用 url_hash 文件名,会自然覆盖) + dup_log = out_root / args.date / "duplicates.jsonl" + if dup_log.exists(): + dup_log.unlink() + + total_uniq = 0 + total_dup = 0 + total_layers: Counter = Counter() + # 当天唯一新闻 url_hash -> 全部来源列表(跨源累积,多源记录) + sources_map: dict[str, list[str]] = {} + for src in sources: + u, d, lc = _process_source_day( + src, args.date, processed_root, out_root, deduper, sources_map + ) + total_uniq += u + total_dup += d + total_layers.update(lc) + + # 多源记录汇总:data/deduped/{day}/sources.json + # {url_hash: [source_id, ...]},配合 uniques/{url_hash}.json 的 sources 字段 + # 与指纹库 source_ids 列,提供「一条唯一新闻多个来源」的完整记录。 + sources_path = out_root / args.date / "sources.json" + sources_path.write_text( + json.dumps( + {k: v for k, v in sources_map.items() if v}, + ensure_ascii=False, + indent=2, + ), + encoding="utf-8", + ) + logger.info( + "多源记录已写入 {} ({} 条唯一新闻,{} 条含多源)", + sources_path, + len(sources_map), + sum(1 for v in sources_map.values() if len(v) > 1), + ) + + total = total_uniq + total_dup + rate = total_dup / max(total, 1) + logger.info( + "全部完成: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}", + total_uniq, + total_dup, + rate, + dict(total_layers), + ) + # 验收门槛: ≤ 5% + return 0 if rate <= 0.05 or total == 0 else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_embedding.py b/scripts/run_embedding.py new file mode 100644 index 0000000..5e5e088 --- /dev/null +++ b/scripts/run_embedding.py @@ -0,0 +1,334 @@ +"""M5 批量嵌入入口脚本。 + +输入策略: + A. 默认 events 优先 (--input events): + data/events/{day}/*.json (M4 ExtractedEvent),用 LLM 摘要 + 事件标签 + 正文组装文本; + 同时回查对应 article(M2 输出)以拿正文,如查不到则用 ExtractedEvent.event.summary 作为正文。 + B. articles 模式 (--input articles): + data/processed/{source}/{day}/*.json (M2 Article 直出),仅用 title + content。 + C. deduped 模式 (--input deduped): + data/deduped/{day}/uniques/*.json (M3 唯一文章),仅 title + content。 + +输出: + data/embeddings/{day}/{url_hash}.json (含 vector 完整内容) + data/embeddings/{day}/index.jsonl (扁平摘要,不含向量,便于检索/调试) + data/embeddings/{day}/failed.jsonl (失败列表) + +增量: 默认跳过已嵌入的文章(输出目录已有 {url_hash}.json 视为已处理), + 断点续跑/失败重试不会重复调用 embed API;--force 强制全量重嵌入。 +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import sys +import time + +# 修复树莓派等环境的 ASCII locale 问题(UnicodeEncodeError) +os.environ.setdefault("PYTHONUTF8", "1") +from datetime import date, datetime +from pathlib import Path +from typing import Any + +from dotenv import load_dotenv +from loguru import logger +from pydantic import ValidationError + +from embedding import ( + EmbeddingError, + EmbeddingResult, + compose_text, + make_async_provider, +) +from embedding.base import _from_event_dict +from extractor import Article + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + log_path = Path("logs") / "embedding.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +# --------------------------------------------------------------------------- # +# 输入收集与文本组装 +# --------------------------------------------------------------------------- # + +def _list_source_dirs(p: Path) -> list[str]: + if not p.is_dir(): + return [] + return sorted(d.name for d in p.iterdir() if d.is_dir()) + + +def _load_article_by_hash(processed_root: Path, day: str, url_hash: str) -> Article | None: + """根据 url_hash 在 data/processed/*/{day}/ 中查 Article。""" + for src_dir in processed_root.iterdir(): + if not src_dir.is_dir(): + continue + candidate = src_dir / day / f"{url_hash}.json" + if candidate.is_file(): + try: + return Article.model_validate( + json.loads(candidate.read_text(encoding="utf-8")) + ) + except (json.JSONDecodeError, ValidationError): + return None + return None + + +def _build_text_from_event( + event_path: Path, + processed_root: Path, + day: str, +) -> tuple[str, Article, str | None] | None: + """从 ExtractedEvent JSON 构造嵌入文本与文章元数据。""" + try: + obj: dict[str, Any] = json.loads(event_path.read_text(encoding="utf-8")) + except json.JSONDecodeError as e: + logger.warning("跳过损坏 event 文件 {}: {}", event_path, e) + return None + + article, head, summary = _from_event_dict(obj) + # 真正的正文要去 processed/ 找 + real_article = _load_article_by_hash(processed_root, day, article.url_hash) + if real_article is None: + # 退化:用 summary 作为正文(质量降级,但仍可嵌入) + real_article = article + logger.debug("未找到 Article(url_hash={}),用 summary 作为正文兜底", article.url_hash) + else: + # 用 processed 的真实数据,但保留 event 元数据(time/source 等) + real_article = real_article.model_copy( + update={"publish_time": article.publish_time or real_article.publish_time} + ) + text = compose_text(real_article, head=head, summary=summary) + return text, real_article, summary + + +def _build_text_from_article(article_path: Path) -> tuple[str, Article, str | None] | None: + try: + obj = json.loads(article_path.read_text(encoding="utf-8")) + article = Article.model_validate(obj) + except (json.JSONDecodeError, ValidationError) as e: + logger.warning("跳过损坏 article 文件 {}: {}", article_path, e) + return None + return compose_text(article), article, None + + +def _collect_inputs(args: argparse.Namespace) -> list[tuple[Path, str]]: + """返回 [(input_file, kind), ...],kind 为 'event'/'article'。""" + day = args.date + files: list[tuple[Path, str]] = [] + if args.input == "events": + d = Path(args.events_root) / day + files = [(p, "event") for p in sorted(d.glob("*.json"))] + elif args.input == "deduped": + d = Path(args.deduped_root) / day / "uniques" + files = [(p, "article") for p in sorted(d.glob("*.json"))] + elif args.input == "articles": + proc = Path(args.processed_root) + srcs = [args.source] if args.source else _list_source_dirs(proc) + for src in srcs: + sd = proc / src / day + if sd.is_dir(): + files.extend((p, "article") for p in sorted(sd.glob("*.json"))) + else: + raise ValueError(f"未知 --input 模式: {args.input}") + return files + + +def _filter_existing( + files: list[tuple[Path, str]], out_dir: Path +) -> tuple[list[tuple[Path, str]], int]: + """过滤掉已有产物(输出目录存在同名 {url_hash}.json)的输入。 + + 输入与输出文件名均为 {url_hash}.json,直接比对 stem。 + 返回 (待处理, 跳过数);断点续跑/失败重试借此避免重复调用 embed API。 + """ + pending: list[tuple[Path, str]] = [] + skipped = 0 + for fp, kind in files: + if (out_dir / f"{fp.stem}.json").exists(): + skipped += 1 + else: + pending.append((fp, kind)) + if skipped: + logger.info("跳过已嵌入 {} 篇(产物已存在),待处理 {}", skipped, len(pending)) + return pending, skipped + + +# --------------------------------------------------------------------------- # +# 主流程 +# --------------------------------------------------------------------------- # + +async def _run(args: argparse.Namespace) -> int: + load_dotenv() + + files = _collect_inputs(args) + if not files: + logger.error("未发现任何输入文件: {} ({})", args.input, args.date) + return 2 + + out_dir = Path(args.out_root) / args.date + out_dir.mkdir(parents=True, exist_ok=True) + # 增量:跳过已有产物(断点续跑/失败重试不重复调用 embed API),--force 全量 + skipped = 0 + if not args.force: + files, skipped = _filter_existing(files, out_dir) + if args.limit: + files = files[: args.limit] + if not files: + logger.info( + "无待嵌入文章(全部已处理,跳过 {} 篇),如需重新嵌入请加 --force", skipped + ) + return 0 + logger.info("待嵌入文章数: {} (跳过已处理 {}; input={})", len(files), skipped, args.input) + + # 准备每篇文本 + prepared: list[tuple[str, Article, str | None]] = [] + for fp, kind in files: + if kind == "event": + built = _build_text_from_event(fp, Path(args.processed_root), args.date) + else: + built = _build_text_from_article(fp) + if built is not None: + prepared.append(built) + if not prepared: + logger.error("所有输入文件均无法解析") + return 2 + + index_path = out_dir / "index.jsonl" + failed_path = out_dir / "failed.jsonl" + if args.force: + # 全量模式:重建 index / failed + for p in (index_path, failed_path): + if p.exists(): + p.unlink() + else: + # 增量模式:index 累积追加;failed 只保留本次运行失败的 + if failed_path.exists(): + failed_path.unlink() + + started = time.time() + succ_cnt = 0 + fail_cnt = 0 + + async with make_async_provider( + args.provider, + **({"model": args.model} if args.model else {}), + ) as provider: + logger.info( + "Embedding provider={} model={} dim={}", + provider.name, provider.model, provider.dim, + ) + + # 异步分批 + batch_size = args.batch_size + for i in range(0, len(prepared), batch_size): + batch = prepared[i : i + batch_size] + texts = [t for t, _, _ in batch] + try: + vectors = await provider.embed_batch(texts) + except EmbeddingError as e: + logger.warning("批 {} 嵌入失败: {}", i // batch_size, e) + for _, art, _ in batch: + fail_cnt += 1 + with failed_path.open("a", encoding="utf-8") as f: + f.write( + json.dumps( + {"url_hash": art.url_hash, "url": art.url, + "error": str(e)}, + ensure_ascii=False, + ) + + "\n" + ) + continue + + for (text, article, _summary), vec in zip(batch, vectors, strict=True): + if len(vec) != provider.dim: + logger.warning( + "维度不一致 url_hash={} 实际={} 预期={}", + article.url_hash, len(vec), provider.dim, + ) + result = EmbeddingResult( + url_hash=article.url_hash, + source_id=article.source_id, + title=article.title, + text=text, + vector=vec, + dim=len(vec), + provider=provider.name, + model=provider.model, + embedded_at=datetime.now(), + char_count=len(text), + publish_time=article.publish_time, + ) + + # 落盘:完整 + index + (out_dir / f"{result.url_hash}.json").write_text( + result.model_dump_json(indent=2), encoding="utf-8" + ) + meta = result.model_dump(exclude={"vector", "text"}, mode="json") + meta["text_preview"] = text[:60] + with index_path.open("a", encoding="utf-8") as f: + f.write(json.dumps(meta, ensure_ascii=False) + "\n") + succ_cnt += 1 + logger.info( + "已处理批 {}: 成功 +{}/{} (累计成功 {})", + i // batch_size + 1, len(batch), len(batch), succ_cnt, + ) + + elapsed = time.time() - started + total = succ_cnt + fail_cnt + rate = succ_cnt / max(total, 1) + logger.info( + "Embedding 完成: 成功 {}/{} 成功率 {:.1%} 用时 {:.1f}s", + succ_cnt, total, rate, elapsed, + ) + return 0 if rate >= 0.95 or total == 0 else 1 + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股新闻 Embedding 向量化 (M5)") + parser.add_argument( + "--input", default="events", + choices=["events", "deduped", "articles"], + help="输入来源:events (M4) / deduped (M3) / articles (M2)", + ) + parser.add_argument("--events-root", default="data/events") + parser.add_argument("--deduped-root", default="data/deduped") + parser.add_argument("--processed-root", default="data/processed") + parser.add_argument("--source", default=None, + help="--input articles 时按源过滤") + parser.add_argument("--out-root", default="data/embeddings") + parser.add_argument( + "--date", default=date.today().strftime("%Y%m%d"), + help="日期 YYYYMMDD,默认今日", + ) + parser.add_argument("--provider", default=None, + help="dashscope (默认) / local-bge,可被 .env 覆盖") + parser.add_argument("--model", default=None, + help="嵌入模型名,覆盖 .env 中的 *_EMBEDDING_MODEL") + parser.add_argument("--batch-size", type=int, default=10, + help="每批送 embed 的条数(DashScope 上限 10)") + parser.add_argument("--limit", type=int, default=0, + help="最多处理 N 篇,0=不限") + parser.add_argument("--force", action="store_true", + help="强制全量重嵌入(默认跳过已嵌入文章)") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + return asyncio.run(_run(args)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_event_extraction.py b/scripts/run_event_extraction.py new file mode 100644 index 0000000..c791ea7 --- /dev/null +++ b/scripts/run_event_extraction.py @@ -0,0 +1,281 @@ +"""M4 LLM 投资事件抽取批处理。 + +输入: data/deduped/{YYYYMMDD}/uniques/*.json (M3 唯一文章产物) + 或者 data/processed/{source}/{YYYYMMDD}/*.json (M2 直出,跳过去重时使用) +输出: data/events/{YYYYMMDD}/{url_hash}.json (ExtractedEvent) + data/events/{YYYYMMDD}/index.jsonl (扁平摘要) + data/events/{YYYYMMDD}/failed.jsonl (失败列表) + +增量: 默认跳过已抽取的文章(输出目录已有 {url_hash}.json 视为已处理), + 断点续跑/失败重试不会重复调用 LLM API;--force 强制全量重抽。 + +用法: + uv run python -m scripts.run_event_extraction + uv run python -m scripts.run_event_extraction --date 20260616 + uv run python -m scripts.run_event_extraction --provider qwen --model qwen-plus + uv run python -m scripts.run_event_extraction --concurrency 5 --limit 10 + uv run python -m scripts.run_event_extraction --input-root data/processed --no-deduped + uv run python -m scripts.run_event_extraction --force # 全量重抽 +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import sys +import time +from collections import Counter +from datetime import date +from pathlib import Path + +from dotenv import load_dotenv +from loguru import logger +from pydantic import ValidationError + +from extractor import Article +from llm import ( + SCENE_EVENT_EXTRACTION, + ExtractedEvent, + LLMCallError, + PromptTemplate, + extract_event_async, + load_llm_config, + make_async_client, +) + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + log_path = Path("logs") / "llm.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +def _collect_inputs( + input_root: Path, + day: str, + use_deduped: bool, + source_filter: str | None, +) -> list[Path]: + """根据是否走 dedup 选择输入文件清单。""" + if use_deduped: + # data/deduped/{day}/uniques/*.json + d = input_root / day / "uniques" + if not d.is_dir(): + return [] + return sorted(d.glob("*.json")) + # data/processed/{source}/{day}/*.json + files: list[Path] = [] + if source_filter: + srcs = [source_filter] + else: + srcs = sorted(p.name for p in input_root.iterdir() if p.is_dir()) + for src in srcs: + d = input_root / src / day + if d.is_dir(): + files.extend(sorted(d.glob("*.json"))) + return files + + +def _filter_existing(files: list[Path], out_dir: Path) -> tuple[list[Path], int]: + """过滤掉已有产物(输出目录存在同名 {url_hash}.json)的输入。 + + 输入文件名即 url_hash(如 {url_hash}.json),与 M4 产物命名一致。 + 返回 (待处理文件, 跳过数);断点续跑/失败重试借此避免重复调用 LLM API。 + """ + pending: list[Path] = [] + skipped = 0 + for fp in files: + if (out_dir / f"{fp.stem}.json").exists(): + skipped += 1 + else: + pending.append(fp) + if skipped: + logger.info("跳过已处理 {} 篇(产物已存在),待处理 {}", skipped, len(pending)) + return pending, skipped + + +def _load_article(p: Path) -> Article | None: + try: + return Article.model_validate(json.loads(p.read_text(encoding="utf-8"))) + except (json.JSONDecodeError, ValidationError) as e: + logger.warning("跳过无法解析的 article 文件 {}: {}", p, e) + return None + + +def _load_sources(p: Path) -> list[str] | None: + """从输入 JSON 读取 sources 多源字段(deduped uniques 才有);无则返回 None。 + + None 表示无多源记录,由 extract_event 兜底为 [source_id]。 + """ + try: + data = json.loads(p.read_text(encoding="utf-8")) + srcs = data.get("sources") + if isinstance(srcs, list) and srcs: + return [s for s in srcs if s] + except (json.JSONDecodeError, OSError): + pass + return None + + +async def _run(args: argparse.Namespace) -> int: + load_dotenv() # 读 .env 到 os.environ + # scene=event_extraction: 读取 configs/llm_models.yaml 场景 1 配置,未配置字段回退 .env + config = load_llm_config( + provider=args.provider, model=args.model, scene=SCENE_EVENT_EXTRACTION + ) + logger.info( + "LLM provider={} model={} base_url={}", + config.provider, config.model, config.base_url, + ) + + input_root = Path(args.input_root) + use_deduped = not args.no_deduped + files = _collect_inputs(input_root, args.date, use_deduped, args.source) + if not files: + logger.error( + "{} 下未发现 {} 的文章(use_deduped={})", + input_root, args.date, use_deduped, + ) + return 2 + + out_dir = Path(args.out_root) / args.date + out_dir.mkdir(parents=True, exist_ok=True) + # 增量:跳过已有产物(断点续跑/失败重试不重复调用 LLM API),--force 全量 + skipped = 0 + if not args.force: + files, skipped = _filter_existing(files, out_dir) + if args.limit: + files = files[: args.limit] + if not files: + logger.info( + "无待处理文章(全部已抽取,跳过 {} 篇),如需重抽请加 --force", skipped + ) + return 0 + + logger.info("待处理文章数: {} (跳过已处理 {})", len(files), skipped) + + index_path = out_dir / "index.jsonl" + failed_path = out_dir / "failed.jsonl" + if args.force: + # 全量模式:重建 index / failed + for p in (index_path, failed_path): + if p.exists(): + p.unlink() + else: + # 增量模式:index 累积追加;failed 只保留本次运行失败的 + if failed_path.exists(): + failed_path.unlink() + + template = PromptTemplate(args.prompt) + semaphore = asyncio.Semaphore(args.concurrency) + + succ_cnt = 0 + fail_cnt = 0 + sentiment_cnt: Counter = Counter() + layer_attempts: Counter = Counter() + started = time.time() + + async with make_async_client(config) as client: + + async def _process(fp: Path) -> tuple[Path, ExtractedEvent | None, str | None]: + article = _load_article(fp) + if article is None: + return fp, None, "无法解析输入" + sources = _load_sources(fp) # 多源记录(去重层),无则 None + try: + event = await extract_event_async( + client=client, + config=config, + article=article, + template=template, + max_attempts=args.max_attempts, + sources=sources, + semaphore=semaphore, + ) + return fp, event, None + except LLMCallError as e: + return fp, None, str(e) + + tasks = [_process(fp) for fp in files] + for coro in asyncio.as_completed(tasks): + fp, event, err = await coro + if event is None: + fail_cnt += 1 + with failed_path.open("a", encoding="utf-8") as f: + f.write( + json.dumps( + {"file": str(fp), "error": err}, + ensure_ascii=False, + ) + + "\n" + ) + continue + + succ_cnt += 1 + sentiment_cnt[event.event.sentiment.value] += 1 + layer_attempts[event.attempts] += 1 + + event_path = out_dir / f"{event.url_hash}.json" + event_path.write_text(event.model_dump_json(indent=2), encoding="utf-8") + + meta = event.model_dump(mode="json") + # 扁平 index 不含完整 content,但保留 event 字段 + with index_path.open("a", encoding="utf-8") as f: + f.write(json.dumps(meta, ensure_ascii=False) + "\n") + + logger.info(event.short_summary()) + + elapsed = time.time() - started + total = succ_cnt + fail_cnt + rate = succ_cnt / max(total, 1) + logger.info( + "完成: 成功 {}/{} 成功率 {:.1%} 用时 {:.1f}s 情绪={} 重试分布={}", + succ_cnt, total, rate, elapsed, dict(sentiment_cnt), dict(layer_attempts), + ) + # 验收门槛: ≥ 95% JSON 成功率 + return 0 if rate >= 0.95 or total == 0 else 1 + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股 LLM 投资事件抽取 (M4)") + parser.add_argument( + "--input-root", default="data/deduped", + help="输入根目录(默认 dedup 输出);配合 --no-deduped 时改为 data/processed", + ) + parser.add_argument("--no-deduped", action="store_true", + help="跳过 M3,直接读 M2 处理产物") + parser.add_argument("--source", default=None, help="--no-deduped 时按源过滤") + parser.add_argument("--out-root", default="data/events") + parser.add_argument( + "--date", default=date.today().strftime("%Y%m%d"), + help="日期 YYYYMMDD,默认今日", + ) + parser.add_argument("--provider", default=None, + help="LLM provider: deepseek / qwen,默认读 .env") + parser.add_argument("--model", default=None, help="模型名,默认 provider 默认值") + parser.add_argument("--concurrency", type=int, default=3, + help="LLM 异步并发上限") + parser.add_argument("--max-attempts", type=int, default=3, + help="单篇文章最大重试次数") + parser.add_argument("--force", action="store_true", + help="强制全量重抽(默认跳过已抽取文章)") + parser.add_argument("--limit", type=int, default=0, + help="最多处理 N 篇,0=不限制(用于联调)") + parser.add_argument("--prompt", default="prompts/event_extraction.md", + help="Prompt 模板路径") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + return asyncio.run(_run(args)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_extractor.py b/scripts/run_extractor.py new file mode 100644 index 0000000..135b21c --- /dev/null +++ b/scripts/run_extractor.py @@ -0,0 +1,391 @@ +"""M2 批量正文提取入口脚本。 + +输入: data/raw/{source}/{YYYYMMDD}/index.jsonl (M1 产物) +输出: data/processed/{source}/{YYYYMMDD}/{url_hash}.json (Article) + data/processed/{source}/{YYYYMMDD}/index.jsonl (扁平元数据,便于检索) + +用法: + uv run python -m scripts.run_extractor # 处理今日所有源 + uv run python -m scripts.run_extractor --date 20260616 + uv run python -m scripts.run_extractor --source sina --date 20260616 + uv run python -m scripts.run_extractor --raw-root data/raw --out-root data/processed +""" + +from __future__ import annotations + +import argparse +import contextlib +import json +import sys +from datetime import date, datetime +from pathlib import Path + +from loguru import logger +from pydantic import ValidationError + +from extractor import Article, ExtractError, extract_article +from extractor.parser import _url_hash + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + log_path = Path("logs") / "extractor.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +def _iter_article_records(raw_dir: Path) -> list[dict]: + """读取 M1 产物的 index.jsonl,返回所有可处理的记录。 + + 新闻源: stage=article + success + html_file + cninfo: 有 json_file 字段(CninfoItem 格式) + """ + index_path = raw_dir / "index.jsonl" + if not index_path.is_file(): + return [] + out: list[dict] = [] + with index_path.open("r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + except json.JSONDecodeError as e: + logger.warning("跳过非法 jsonl 行 in {}: {}", index_path, e) + continue + # cninfo: 新格式(CninfoItem → json_file) + if rec.get("source_id") == "cninfo" or rec.get("json_file"): + if rec.get("json_file"): + out.append(rec) + continue + # 新闻源: 旧格式(CrawlResult → html_file) + if rec.get("stage") == "article" and rec.get("success") and rec.get("html_file"): + out.append(rec) + return out + + +def _process_one(rec: dict, raw_dir: Path, out_dir: Path, + body_xpath_map: dict[str, str] | None = None) -> Article | None: + """处理单条记录,失败返回 None。cninfo 等结构化源跳过 GNE。""" + src_id = rec.get("source_id", "") + + # cninfo: 新格式(json_file)或旧格式(html_file+meta JSON)直接解析 + json_file = rec.get("json_file", "") + html_file = rec.get("html_file", "") + is_cninfo = (src_id == "cninfo" or bool(json_file)) + + if is_cninfo: + if json_file: + json_path = raw_dir / json_file + if json_path.is_file(): + return _process_cninfo_v2(rec, json_path, out_dir) + else: + logger.warning("cninfo JSON 文件丢失: {}", json_path) + return None + # 旧格式兼容: html_file + JSON + if html_file: + html_path = raw_dir / html_file + if html_path.is_file(): + return _process_cninfo(rec, html_path, out_dir) + return None + + html_path = raw_dir / html_file + if not html_path or not html_path.is_file(): + logger.warning("HTML 文件丢失: {}", html_path) + return None + + html = html_path.read_text(encoding="utf-8", errors="ignore") + extra_config: dict[str, str] = {} + if body_xpath_map and src_id in body_xpath_map: + extra_config["body_xpath"] = body_xpath_map[src_id] + try: + article = extract_article( + html=html, + source_id=src_id, + url=rec["url"], + extra_config=extra_config or None, + ) + except ExtractError as e: + logger.warning("提取失败 {} {}: {}", rec["source_id"], rec["url"], e.reason) + return None + + return _save_article(article, out_dir) + + +def _process_cninfo(rec: dict, html_path: Path, out_dir: Path) -> Article | None: + """处理 cninfo 公告记录:从 meta JSON 解析结构化数据。""" + import re + html = html_path.read_text(encoding="utf-8", errors="ignore") + # 提取 {...} 中的 JSON + m = re.search(r"(.+?)", html, re.DOTALL) + if not m: + logger.warning("cninfo HTML 不含 meta JSON: {}", html_path) + return None + try: + meta = json.loads(m.group(1)) + except json.JSONDecodeError: + logger.warning("cninfo meta JSON 解析失败: {}", html_path) + return None + + sec_code = (meta.get("secCode") or "").strip() + sec_name = (meta.get("secName") or "").strip() + ann_type = (meta.get("announcementType") or "").strip() + pdf_url = (meta.get("pdfUrl") or "").strip() + title = rec.get("title") or meta.get("title") or "" + + # 构建正文:结构化摘要 + 公告类别翻译 + content_parts = [f"公司: {sec_name}({sec_code})", f"公告标题: {title}"] + if ann_type: + content_parts.append(f"公告类别编码: {ann_type}") + if pdf_url: + content_parts.append(f"PDF: {pdf_url}") + content = "\n".join(content_parts) + + # 时间 + fetched = rec.get("fetched_at") + publish_time = None + if isinstance(fetched, str): + try: + from datetime import datetime as dt + publish_time = dt.fromisoformat(fetched) + except ValueError: + pass + + article = Article( + source_id="cninfo", + url=rec.get("url") or "", + url_hash=_url_hash(rec.get("url") or ""), + title=title, + content=content, + author=sec_name, + source_name="巨潮资讯网", + publish_time=publish_time, + publish_time_raw=fetched, + word_count=len(content), + ) + return _save_article(article, out_dir) + + +def _process_cninfo_v2(rec: dict, json_path: Path, out_dir: Path) -> Article | None: + """处理 cninfo v2 格式: 直接读取 CninfoItem JSON 并转换为 Article。""" + try: + item_data = json.loads(json_path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError) as e: + logger.warning("cninfo JSON 读取失败 {}: {}", json_path, e) + return None + + stock_code = item_data.get("stock_code", "") + stock_name = item_data.get("stock_name", "") + title = item_data.get("title", "") + content = item_data.get("content", "") + publish_time_str = item_data.get("publish_time", "") + item_type = item_data.get("item_type", "announcement") + url = item_data.get("url", "") + extra = item_data.get("extra", {}) + + # 类型中文映射 + type_map = { + "announcement": "公告", + "research": "投资者调研", + "irm": "互动问答", + } + type_cn = type_map.get(item_type, item_type) + + # 构建正文 + content_parts = [ + f"公司: {stock_name}({stock_code})", + f"类型: {type_cn}", + f"标题: {title}", + ] + if extra.get("announcement_type"): + content_parts.append(f"公告类别: {extra['announcement_type']}") + if url: + content_parts.append(f"原文链接: {url}") + if content: + content_parts.append(f"\n正文:\n{content}") + full_content = "\n".join(content_parts) + + # 发布时间解析 + publish_time = None + publish_time_raw = publish_time_str + if publish_time_str: + try: + from datetime import datetime as dt + publish_time = dt.strptime(publish_time_str, "%Y-%m-%d") + except ValueError: + try: + publish_time = dt.fromisoformat(publish_time_str) + except ValueError: + pass + + article = Article( + source_id="cninfo", + url=url, + url_hash=_url_hash(url or title), + title=title, + content=full_content, + author=stock_name, + source_name="巨潮资讯网", + publish_time=publish_time, + publish_time_raw=publish_time_raw, + word_count=len(full_content), + item_type=item_type, + ) + return _save_article(article, out_dir) + + +def _append_index(article: Article, out_dir: Path) -> None: + """把 Article 的扁平摘要追加到 index.jsonl。""" + flat = article.model_dump(exclude={"content", "images"}, mode="json") + flat["article_file"] = f"{article.url_hash}.json" + flat["content_preview"] = article.content[:80] + with (out_dir / "index.jsonl").open("a", encoding="utf-8") as f: + f.write(json.dumps(flat, ensure_ascii=False) + "\n") + + +def _save_article(article: Article, out_dir: Path) -> Article: + """保存 Article JSON 并追加 index。""" + out_dir.mkdir(parents=True, exist_ok=True) + article_path = out_dir / f"{article.url_hash}.json" + article_path.write_text(article.model_dump_json(indent=2), encoding="utf-8") + _append_index(article, out_dir) + return article + + +def _process_source_day( + source_id: str, + day: str, + raw_root: Path, + out_root: Path, + body_xpath_map: dict[str, str] | None = None, + *, + force: bool = False, +) -> tuple[int, int, int]: + """处理单个源单日。返回 (成功数, 总数, 跳过数)。 + + 默认增量:已提取的文章(输出目录已有 {url_hash}.json)跳过提取,仅回补 index 行; + force=True 时全量重提取并重建 index。 + """ + raw_dir = raw_root / source_id / day + out_dir = out_root / source_id / day + + records = _iter_article_records(raw_dir) + if not records: + logger.info("源 {} 日期 {} 无可处理记录", source_id, day) + return 0, 0, 0 + + # 全量模式:重建 index;增量模式:保留旧 index 追加新条目 + old_index = out_dir / "index.jsonl" + if force and old_index.exists(): + old_index.unlink() + + succ = 0 + skipped = 0 + for rec in records: + url_hash = rec.get("url_hash") or _url_hash(rec.get("url") or "") + existing = out_dir / f"{url_hash}.json" + if not force and existing.is_file(): + # 增量:跳过已提取,回补 index 行保持摘要完整 + skipped += 1 + with contextlib.suppress(json.JSONDecodeError, ValidationError, OSError): + _append_index( + Article.model_validate(json.loads(existing.read_text(encoding="utf-8"))), + out_dir, + ) + continue + article = _process_one(rec, raw_dir, out_dir, body_xpath_map) + if article is not None: + succ += 1 + total = len(records) + # 增量模式下「跳过已提取」视为已成功处理,避免全跳过时误报成功率 0% + rate = (succ + skipped) / max(total, 1) + logger.info( + "源 {} 日期 {} 提取完成: {}/{} 成功率 {:.0%} (跳过已提取 {})", + source_id, + day, + succ, + total, + rate, + skipped, + ) + return succ, total, skipped + + +def _list_source_dirs(raw_root: Path) -> list[str]: + """列出 raw_root 下所有源 id(子目录名)。""" + if not raw_root.is_dir(): + return [] + return sorted(p.name for p in raw_root.iterdir() if p.is_dir()) + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股新闻正文提取 (M2)") + parser.add_argument("--raw-root", default="data/raw", help="M1 抓取产物根目录") + parser.add_argument("--out-root", default="data/processed", help="M2 提取结果根目录") + parser.add_argument("--source", default=None, help="只处理单个源 id,默认全部") + parser.add_argument( + "--date", + default=date.today().strftime("%Y%m%d"), + help="处理日期 YYYYMMDD,默认今日", + ) + parser.add_argument("--force", action="store_true", + help="强制全量重提取(默认跳过已提取文章)") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + raw_root = Path(args.raw_root) + out_root = Path(args.out_root) + + sources = [args.source] if args.source else _list_source_dirs(raw_root) + if not sources: + logger.error("{} 下未发现任何源目录", raw_root) + return 2 + + # 加载源配置,构建 source_id → body_xpath 映射 + body_xpath_map: dict[str, str] = {} + try: + from crawler.config import load_crawler_config + cfg = load_crawler_config() + for s in cfg.sources: + if s.body_xpath: + body_xpath_map[s.id] = s.body_xpath + if body_xpath_map: + logger.info("已加载 body_xpath 配置: {}", dict(body_xpath_map)) + except Exception: + pass + + started = datetime.now() + total_succ = 0 + total_all = 0 + total_skipped = 0 + for src in sources: + succ, total, skipped = _process_source_day( + src, args.date, raw_root, out_root, body_xpath_map, force=args.force + ) + total_succ += succ + total_all += total + total_skipped += skipped + + elapsed = (datetime.now() - started).total_seconds() + # 增量模式下跳过已提取视为成功 + rate = (total_succ + total_skipped) / max(total_all, 1) + logger.info( + "全部完成: {}/{} 成功率 {:.0%} (跳过已提取 {}) 用时 {:.1f}s", + total_succ, + total_all, + rate, + total_skipped, + elapsed, + ) + return 0 if rate >= 0.9 or total_all == 0 else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_mcp_server.py b/scripts/run_mcp_server.py new file mode 100644 index 0000000..0048de4 --- /dev/null +++ b/scripts/run_mcp_server.py @@ -0,0 +1,88 @@ +"""M8 MCP 服务入口。 + +Cherry Studio / Claude Code 通过 stdio 协议调用。 + +用法: + uv run python -m scripts.run_mcp_server # stdio 模式(默认) + uv run python -m scripts.run_mcp_server --sse 8765 # HTTP SSE 模式(调试用) + +Cherry Studio 配置: + { + "mcpServers": { + "a-share-research": { + "command": "uv", + "args": ["run", "python", "-m", "scripts.run_mcp_server"], + "cwd": "/home/pi/news" + } + } + } +""" + +from __future__ import annotations + +import argparse +from pathlib import Path + +from loguru import logger + + +def _setup_logger() -> None: + logger.remove() + log_path = Path("logs") / "mcp_server.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + # MCP stdio 模式下 stderr 会被协议占用,只写文件日志 + logger.add( + log_path, + level="DEBUG", + rotation="10 MB", + retention=5, + encoding="utf-8", + enqueue=True, + ) + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股 Deep Research MCP 服务") + parser.add_argument("--sse", type=int, default=None, + help="启动 HTTP SSE 模式在指定端口(调试用)") + parser.add_argument("--host", default="0.0.0.0", help="SSE 监听地址") + args = parser.parse_args() + + _setup_logger() + logger.info("MCP 服务启动 mode={}", "sse" if args.sse else "stdio") + + if args.sse: + _run_sse(args.host, args.sse) + else: + _run_stdio() + + return 0 + + +def _run_stdio() -> None: + from mcp_server.tools import mcp # noqa: E402 + mcp.run() + + +def _run_sse(host: str, port: int) -> None: + from mcp.server.fastmcp import FastMCP # noqa: E402 + + from mcp_server.tools import ( # noqa: E402 + search_company_news, + search_industry_news, + search_news, + search_sentiment_trend, + search_stock_events, + ) + + sse_mcp = FastMCP(name="A股DeepResearch", host=host, port=port) + sse_mcp.tool()(search_news) + sse_mcp.tool()(search_company_news) + sse_mcp.tool()(search_industry_news) + sse_mcp.tool()(search_stock_events) + sse_mcp.tool()(search_sentiment_trend) + sse_mcp.run(transport="sse") + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_qdrant_ingest.py b/scripts/run_qdrant_ingest.py new file mode 100644 index 0000000..fabe77e --- /dev/null +++ b/scripts/run_qdrant_ingest.py @@ -0,0 +1,210 @@ +"""M6 Qdrant 批量入库脚本。 + +输入: data/embeddings/{date}/*.json (M5 产物,含 vector + payload) +目标: Qdrant Collection a_share_news +策略: 幂等 upsert(url_hash 做 point ID,同 ID 覆盖不重复计数) + +用法: + uv run python -m scripts.run_qdrant_ingest # 入库今日 + uv run python -m scripts.run_qdrant_ingest --date 20260616 + uv run python -m scripts.run_qdrant_ingest --recreate # 重建 collection + 全量入库 + uv run python -m scripts.run_qdrant_ingest --memory # 内存模式(测试) + uv run python -m scripts.run_qdrant_ingest --limit 10 # 试跑 N 条 +""" + +from __future__ import annotations + +import argparse +import json +import sys +from datetime import date +from pathlib import Path +from typing import Any + +from loguru import logger + +from vectorstore import SearchResult, VectorStore, make_qdrant_client + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + log_path = Path("logs") / "qdrant.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +def _load_embedding_result(path: Path) -> dict[str, Any] | None: + try: + return json.loads(path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError) as e: + logger.warning("跳过损坏文件 {}: {}", path, e) + return None + + +def _result_to_point( + obj: dict[str, Any], + events_dir: Path | None = None, +) -> dict[str, Any] | None: + """把 EmbeddingResult JSON dict 转换为 Qdrant Point 格式。 + + Payload 包含 title/url/source_id/publish_time/event/计数 等。 + 事件字段优先从 M4 ExtractedEvent 补充(EmbeddingResult 本身不含 event)。 + """ + vector = obj.get("vector") + if not vector: + return None + url_hash = obj["url_hash"] + payload = { + "url_hash": url_hash, + "title": obj.get("title") or "", + "url": obj.get("url") or "", + "source_id": obj.get("source_id") or "", + "publish_time": obj.get("publish_time"), + "char_count": obj.get("char_count"), + "word_count": obj.get("word_count"), + "event": { + "stock_codes": [], + "company_names": [], + "industries": [], + "sentiment": "neutral", + "importance": 1, + "event_type": "其他", + "summary": "", + }, + } + # 优先从 M4 ExtractedEvent JSON 补事件字段 + ev = _load_event_from_m4(events_dir, url_hash) if events_dir else None + if ev is not None: + payload["event"] = ev + elif "event" in obj and isinstance(obj["event"], dict): + ev_src = obj["event"] + payload["event"]["stock_codes"] = list(ev_src.get("stock_codes") or []) + payload["event"]["company_names"] = list(ev_src.get("company_names") or []) + payload["event"]["industries"] = list(ev_src.get("industries") or []) + payload["event"]["sentiment"] = ev_src.get("sentiment") or "neutral" + payload["event"]["importance"] = ev_src.get("importance") or 1 + payload["event"]["event_type"] = ev_src.get("event_type") or "其他" + payload["event"]["summary"] = ev_src.get("summary") or "" + return {"id": url_hash, "vector": vector, "payload": payload} + + +def _load_event_from_m4(events_dir: Path, url_hash: str) -> dict[str, Any] | None: + """从 M4 ExtractedEvent JSON 中提取事件 payload 子集。""" + event_file = events_dir / f"{url_hash}.json" + if not event_file.is_file(): + return None + try: + obj = json.loads(event_file.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError): + return None + ev_inner = obj.get("event") + if not isinstance(ev_inner, dict): + return None + return { + "stock_codes": list(ev_inner.get("stock_codes") or []), + "company_names": list(ev_inner.get("company_names") or []), + "industries": list(ev_inner.get("industries") or []), + "sentiment": ev_inner.get("sentiment") or "neutral", + "importance": ev_inner.get("importance") or 1, + "event_type": ev_inner.get("event_type") or "其他", + "summary": ev_inner.get("summary") or "", + } + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股新闻 Qdrant 入库 (M6)") + parser.add_argument("--embeddings-root", default="data/embeddings") + parser.add_argument( + "--date", default=date.today().strftime("%Y%m%d"), + help="日期 YYYYMMDD,默认今日", + ) + parser.add_argument("--limit", type=int, default=0, + help="最多处理 N 条,0=不限") + parser.add_argument("--recreate", action="store_true", + help="入库前删除并重建 collection") + parser.add_argument("--host", default=None, help="Qdrant host(HTTP 模式)") + parser.add_argument("--port", type=int, default=None, help="Qdrant HTTP port") + parser.add_argument("--path", default="data/qdrant_storage", + help="本地文件模式路径(默认,无需 Docker)") + parser.add_argument("--collection", default=None, help="Collection 名") + parser.add_argument("--memory", action="store_true", + help="内存模式(仅测试)") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + + # 收集文件 + emb_dir = Path(args.embeddings_root) / args.date + events_dir = Path("data/events") / args.date # M4 产物(补 event 字段) + if not emb_dir.is_dir(): + logger.error("嵌入目录不存在: {}", emb_dir) + return 2 + files = sorted(emb_dir.glob("*.json")) + if args.limit: + files = files[: args.limit] + if not files: + logger.error("{} 下无嵌入文件", emb_dir) + return 2 + logger.info("待入库文章数: {} (date={}), events补: {}", + len(files), args.date, "yes" if events_dir.is_dir() else "no") + + # 转换 + points: list[dict[str, Any]] = [] + for fp in files: + obj = _load_embedding_result(fp) + if obj is None: + continue + pt = _result_to_point(obj, events_dir=events_dir if events_dir.is_dir() else None) + if pt is not None: + points.append(pt) + if not points: + logger.error("所有嵌入文件均无法解析") + return 2 + + client = make_qdrant_client( + host=args.host, port=args.port, memory=args.memory, + path=args.path if not args.host and not args.memory else None, + ) + store = VectorStore(client, collection_name=args.collection) + + # 初始化 collection + try: + store.init_collection(recreate=args.recreate) + except Exception: # noqa: BLE001 - Qdrant 未启动时会爆连接错误 + logger.exception("无法连接 Qdrant,请先 docker compose --profile m6 up -d: {}") + return 3 + + # 入库 + before = store.count() + store.upsert(points) + after = store.count() + + logger.info("入库完成: 前 {} -> 后 {} (净增 {})", before, after, after - before) + + # 简单自检:用第一条向量做检索,验证可查回 + if points and not args.memory: + probe = points[0] + try: + results: list[SearchResult] = store.query( + query_vector=probe["vector"], top_k=1, + ) + if results: + r = results[0] + logger.info("自检 OK: top-1 title={!r} score={:.4f}", r.title[:30], r.score) + else: + logger.warning("自检:检索返回空") + except Exception as e: # noqa: BLE001 + logger.warning("自检失败(不阻塞): {}", e) + + store.close() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_scheduler.py b/scripts/run_scheduler.py new file mode 100644 index 0000000..5d11f4f --- /dev/null +++ b/scripts/run_scheduler.py @@ -0,0 +1,210 @@ +"""M7 定时任务调度入口。 + +支持两种模式: + --once 立刻执行一次全链路 + (默认) 启动 APScheduler,按 .env 中 SCHEDULE_TIMES 定时执行 + +用法: + uv run python -m scripts.run_scheduler # 启动定时服务 + uv run python -m scripts.run_scheduler --once # 立即执行一次 + uv run python -m scripts.run_scheduler --once --date 20260616 + uv run python -m scripts.run_scheduler --once --steps crawler,extractor +""" + +from __future__ import annotations + +import argparse +import signal +import sys +from datetime import date, datetime +from pathlib import Path +from typing import Any + +from dotenv import load_dotenv +from loguru import logger + +from scheduler import STEP_COMMANDS, run_pipeline +from scheduler.stock_reporter import generate_all_stock_reports + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + log_path = Path("logs") / "scheduler.log" + log_path.parent.mkdir(parents=True, exist_ok=True) + logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8") + + +def _parse_schedule_times(raw: str) -> list[tuple[int, int]]: + """解析 SCHEDULE_TIMES 环境变量。 + + 格式: "07:00,12:00,18:00,22:00" + 返回: [(7,0), (12,0), (18,0), (22,0)] + """ + out: list[tuple[int, int]] = [] + for part in raw.split(","): + part = part.strip() + if not part: + continue + try: + h, m = part.split(":") + out.append((int(h), int(m))) + except ValueError: + logger.warning("SCHEDULE_TIMES 格式错误: {!r},跳过", part) + return out + + +def _once(args: argparse.Namespace) -> int: + """单次执行模式。 + + 默认全量执行;--resume 时断点续跑(跳过连续成功步骤,从失败/未执行步骤继续)。 + """ + steps = None + if args.steps: + steps = [s.strip() for s in args.steps.split(",")] + if args.resume and args.steps: + logger.error("--resume 与 --steps 不能同时使用(断点续跑针对全链路)") + return 2 + run_pipeline(args.date, steps=steps, resume=args.resume) + return 0 + + +def _daemon(args: argparse.Namespace) -> int: + """守护进程模式(APScheduler)。""" + import os + + from apscheduler.schedulers.background import BackgroundScheduler # noqa: E402 + from apscheduler.triggers.cron import CronTrigger # noqa: E402 + + times_raw = os.environ.get("SCHEDULE_TIMES", "07:00,12:00,18:00,22:00") + times = _parse_schedule_times(times_raw) + if not times: + logger.error("SCHEDULE_TIMES 为空或全部非法,无法启动定时任务") + return 2 + + # 找出最早的时间(当天首次运行),仅该次追加日报步骤 + sorted_times = sorted(times) + first_hour, first_minute = sorted_times[0] if sorted_times else (0, 0) + + scheduler = BackgroundScheduler() + + # 包装函数:每次触发时重新计算日期,避免 date.today() 在注册时冻结。 + def _scheduled_pipeline(steps: list[str] | None = None) -> None: + run_pipeline(date.today().strftime("%Y%m%d"), steps=steps) + + for hour, minute in times: + trigger = CronTrigger(hour=hour, minute=minute, timezone="Asia/Shanghai") + is_first = (hour == first_hour and minute == first_minute) + job_kwargs: dict | None = None + if is_first: + job_kwargs = { + "steps": [k for k in STEP_COMMANDS if k not in ("report", "cninfo_crawl", "cninfo_extract", "cninfo_pdf")] + ["report"] + } + scheduler.add_job( + _scheduled_pipeline, + trigger=trigger, + kwargs=job_kwargs, + id=f"pipeline_{hour:02d}{minute:02d}", + name=f"全链路 {'+日报' if is_first else ''} {hour:02d}:{minute:02d}", + ) + logger.info("已注册定时任务: {}每天 {:02d}:{:02d}{}", trigger, hour, minute, + " (含日报)" if is_first else "") + + # cninfo 公告管道(可配置,默认 06:30) + cninfo_raw = os.environ.get("CNINFO_SCHEDULE_TIME", "06:30") + cninfo_parts = cninfo_raw.split(":") + cninfo_h, cninfo_m = int(cninfo_parts[0]), int(cninfo_parts[1]) if len(cninfo_parts) > 1 else 0 + cninfo_trigger = CronTrigger(hour=cninfo_h, minute=cninfo_m, timezone="Asia/Shanghai") + cninfo_steps = ["cninfo_crawl", "cninfo_extract", "cninfo_pdf", + "dedup", "llm", "embedding", "qdrant"] + scheduler.add_job( + _scheduled_pipeline, + trigger=cninfo_trigger, + kwargs={"steps": cninfo_steps}, + id="pipeline_cninfo", + name=f"cninfo 公告管道 {cninfo_h:02d}:{cninfo_m:02d}", + ) + logger.info("已注册定时任务: cninfo 公告管道 每天 {:02d}:{:02d}", cninfo_h, cninfo_m) + + # 个股日报(可配置,默认 07:30, 设为空可禁用) + stock_raw = os.environ.get("STOCK_REPORT_TIME", "07:30") + if stock_raw: + stock_parts = stock_raw.split(":") + stock_h, stock_m = int(stock_parts[0]), int(stock_parts[1]) if len(stock_parts) > 1 else 0 + stock_trigger = CronTrigger(hour=stock_h, minute=stock_m, timezone="Asia/Shanghai") + scheduler.add_job( + generate_all_stock_reports, + trigger=stock_trigger, + id="stock_report", + name=f"个股日报 {stock_h:02d}:{stock_m:02d}", + ) + logger.info("已注册定时任务: 个股日报 每天 {:02d}:{:02d}", stock_h, stock_m) + else: + logger.info("STOCK_REPORT_TIME 为空, 已禁用个股日报") + + # 优雅退出 + def _shutdown(signum: int, frame: Any) -> None: + logger.info("收到信号 {}, 关闭调度器...", signum) + scheduler.shutdown(wait=False) + raise SystemExit(0) + + signal.signal(signal.SIGINT, _shutdown) + signal.signal(signal.SIGTERM, _shutdown) + + scheduler.start() + logger.info("调度器已启动,等待触发... (按 Ctrl+C 退出)") + + # 启动时检查是否有因重启/宕机错过的定时任务,30 分钟内补跑 + now = datetime.now() + for hour, minute in times: + scheduled = now.replace(hour=hour, minute=minute, second=0, microsecond=0) + missed_minutes = (now - scheduled).total_seconds() / 60 + if 0 < missed_minutes < 30: + logger.warning( + "检测到错过的定时任务 {:02d}:{:02d} ({} 分钟前),立即补跑一次", + hour, minute, int(missed_minutes), + ) + steps = [k for k in STEP_COMMANDS if k != "report"] + if (hour, minute) == sorted_times[0]: + steps.append("report") + run_pipeline(date.today().strftime("%Y%m%d"), steps=steps) + + import contextlib + + with contextlib.suppress(SystemExit, KeyboardInterrupt): + # 保持主线程存活,直到收到退出信号 + signal.pause() + + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description="A 股新闻定时任务 (M7)") + parser.add_argument("--once", action="store_true", help="立即执行一次全链路") + parser.add_argument( + "--date", default=date.today().strftime("%Y%m%d"), + help="日期 YYYYMMDD (仅 --once 模式)", + ) + parser.add_argument("--steps", default=None, + help="仅执行指定步骤,逗号分隔 (如 crawler,extractor)") + parser.add_argument( + "--resume", action="store_true", + help="断点续跑(仅 --once):跳过连续成功步骤,从上次失败/未执行步骤继续", + ) + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + load_dotenv() + + if args.once: + return _once(args) + return _daemon(args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_xwlb.py b/scripts/run_xwlb.py new file mode 100644 index 0000000..1a657ad --- /dev/null +++ b/scripts/run_xwlb.py @@ -0,0 +1,107 @@ +"""M1 新闻联播 API 抓取脚本。 + +从 doorcome API /api/xwlbFine/ 获取 AI 精编的新闻联播条目, +转换为与 Web 抓取兼容的格式(data/raw/xwlb/{date}/), +供 M2-M6 管道统一处理。 + +新闻联播晚间播出,始终抓取前一天数据,不依赖 --date 参数。 +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +import urllib.request +from datetime import date, datetime, timedelta +from pathlib import Path + +from loguru import logger + + +def _setup_logger(level: str) -> None: + logger.remove() + logger.add( + sys.stderr, + level=level, + format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}", + ) + + +def _url_hash(s: str) -> str: + return hashlib.sha1(s.encode("utf-8")).hexdigest()[:16] + + +def main() -> int: + parser = argparse.ArgumentParser(description="新闻联播 API 抓取 (M1-xwlb)") + parser.add_argument("--output-root", default="data/raw") + parser.add_argument("--log-level", default="INFO") + args = parser.parse_args() + + _setup_logger(args.log_level) + + # 新闻联播晚间播出,始终抓取前一天 + day_str = (date.today() - timedelta(days=1)).strftime("%Y%m%d") + api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={day_str}&end_date={day_str}" + + logger.info("请求新闻联播 API: {}", api_url) + try: + req = urllib.request.Request(api_url) + with urllib.request.urlopen(req, timeout=15) as resp: + body = json.loads(resp.read().decode("utf-8")) + except Exception as e: + logger.error("API 请求失败: {}", e) + return 2 + + raw_news = body.get("data", {}).get("news", []) + if not raw_news: + logger.warning("{} 无新闻联播数据", day_str) + return 0 + + # 准备输出目录 + out_dir = Path(args.output_root) / "xwlb" / day_str + out_dir.mkdir(parents=True, exist_ok=True) + + index_path = out_dir / "index.jsonl" + saved = 0 + + for n in raw_news: + sid = n.get("daily_sub_id", 0) + title = n.get("news_title", "") + content = n.get("news_improve", "") + news_day = n.get("news_days", day_str) + + fake_url = f"xwlb://{news_day}/{sid}" + h = _url_hash(fake_url) + html_file = f"{h}.html" + + html_content = f""" +{title} +

    {title}

    {content}
    +""" + (out_dir / f"{h}.html").write_text(html_content, encoding="utf-8") + + meta = { + "source_id": "xwlb", + "stage": "article", + "url": fake_url, + "success": True, + "status_code": 200, + "title": title, + "error": None, + "fetched_at": datetime.now().isoformat(), + "attempts": 1, + "url_hash": h, + "html_file": html_file, + } + with index_path.open("a", encoding="utf-8") as f: + f.write(json.dumps(meta, ensure_ascii=False) + "\n") + saved += 1 + + logger.info("新闻联播 {} 抓取完成: {} 条 -> {}", day_str, saved, out_dir) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..40c1c1a --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1 @@ +"""tests 包标记。""" diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..4de763c --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,34 @@ +"""pytest 共享 fixture。""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + + +@pytest.fixture +def sample_sources_yaml(tmp_path: Path) -> Path: + """生成最小可用 sources.yaml 用于测试。""" + content = """ +settings: + concurrency: 2 + retry_max_attempts: 2 + retry_min_wait_sec: 0.01 + retry_max_wait_sec: 0.02 + headless: true + output_root: data/raw + +sources: + - id: testsrc + name: 测试源 + enabled: true + homepage: https://example.com/list + article_url_pattern: '^https://example\\.com/article/\\d+$' + js_render: false + page_timeout_ms: 5000 + max_articles_per_run: 5 +""" + p = tmp_path / "sources.yaml" + p.write_text(content, encoding="utf-8") + return p diff --git a/tests/fixtures/finance_news_daily_20260710_0720.html b/tests/fixtures/finance_news_daily_20260710_0720.html new file mode 100644 index 0000000..dabb912 --- /dev/null +++ b/tests/fixtures/finance_news_daily_20260710_0720.html @@ -0,0 +1,157 @@ + + + + + +A 股 Deep Research 日报 — 20260710_0720 + + + +
    +
    +

    📊 A 股 Deep Research 日报

    +

    20260710 · 生成于 2026-07-11 07:20:27

    +
    +
    +
    + + +

    一、AI 摘要

    +
    • 国务院印发《十五五碳达峰行动方案》,提出2030年新型储能装机3亿千瓦等目标,为新能源与储能行业确立长期增长路径,是当前最具影响力的政策信号。
    • 长鑫科技启动招股拟募资295亿元,成2026年A股最大IPO,带动存储产业链及半导体板块强势上涨,兆易创新业绩预增超1000%印证行业高景气。
    • 李强主持召开国务院常务会议部署防汛抗洪救灾工作,强调保障受灾群众生活,利好水利建设、应急物资等板块,体现政策托底效应。
    • 近15日重要公告/调研:晶盛机电、盐湖股份获机构调研;高测股份公告开展期货套期保值业务并推进限制性股票激励计划。
    • 市场情绪基调:利好。政策利好(碳达峰方案)、IPO巨
    + + +

    二、📺 新闻联播 (07月10日, 共 16 条, 按重要度排序)

    +
    #标题重要度事件类型
    1⚪张国清赴广西指导支持受灾群众生活保障和灾后恢复工作4新闻联播
    2⚪李强主持召开国务院常务会议 部署防汛抗洪救灾等工作4新闻联播
    3⚪习近平会见朝鲜内阁总理朴泰成3新闻联播
    4⚪人民日报将发表评论员文章:加快推进高水平科技自立自强3新闻联播
    5⚪赵乐际会见朝鲜内阁总理朴泰成3新闻联播
    6⚪赵乐际会见纳米比亚总统恩戴特瓦3新闻联播
    7⚪中央层面整治形式主义为基层减负专项工作机制会议在京召开2新闻联播
    8⚪全国用电负荷创历史新高 达15.18亿千瓦2新闻联播
    9⚪我国夏粮丰收 产量首次突破3000亿斤2新闻联播
    10⚪暑运启动以来全国铁路发送旅客超1.23亿人次2新闻联播
    11⚪2026年中国航海日上海主题活动启动1新闻联播
    12⚪上半年全国口岸出入境人员3.69亿人次 创历史新高1新闻联播
    13⚪伊朗媒体称布什尔核电站遭美军袭击 美伊仍进行核谈判1新闻联播
    14⚪俄称红利曼战斗进入收尾阶段 乌称打击俄海上目标1新闻联播
    15⚪特朗普称将向乌克兰发放爱国者生产许可 俄方回应1新闻联播
    16⚪长征十号乙运载火箭完成全球首次海上网系回收1新闻联播
    +

    + 来源: 央视《新闻联播》· 数据取自 doorcome API (xwlbFine) · 20260710 +

    + + +

    三、🔥 重要事件:新闻 (0/758 篇 24h 内, importance ≥ 4, 共 20 篇)

    +
    #标题源重要度事件类型摘要
    1🟢操盘必读:影响股市利好或利空消息_2026年7月10日_财经新闻新浪财经5宏观政策国务院印发《十五五碳达峰行动方案》,提出2030年新型储能装机3亿千瓦、新能源汽车占比30%,推动储能规模化发展。
    2🟢股海导航_2026年7月10日_沪深股市公告与交易提示新浪财经5业绩预告兆易创新预计2026上半年净利润约69亿元,同比增长1099%,存储芯片量价齐升。
    3🟢长鑫科技启动招股带火“朋友圈” 存储产业链股票强势上涨证券时报5其他长鑫科技启动招股拟募资295亿元,带动存储产业链股票大涨,兆易创新等业绩预增。
    4🟢7月10日重要资讯一览证券时报5重大合同东阳光控股子公司签署130亿元至150亿元算力服务合同,金额巨大,利好公司。
    5🟢长鑫科技启动招股,带火“朋友圈”!外资看多中国AI资产!新浪财经5其他长鑫科技启动招股,拟募资295亿元成2026年A股最大IPO,带动半导体板块大涨,兆易创新等业绩预增,外资看多中国AI资
    6🟢国务院印发《“十五五”碳达峰行动方案》 到2030年新型储能装机容量力争达到3亿千瓦东方财富5宏观政策国务院印发“十五五”碳达峰行动方案,提出到2030年新型储能装机力争达3亿千瓦等目标,利好新能源及储能行业。
    7🟢锂业务爆发式增长 紫金矿业上半年净利润预增约68%中国证券网5业绩预告紫金矿业预计2026年上半年归母净利润约391亿元,同比增约68%,锂业务产量大幅增长514%。
    8🟢长征十号乙运载火箭成功实现一子级可控回收东方财富5技术突破长征十号乙运载火箭成功实现一子级可控回收,为全球首次运载火箭网系回收,标志着重复使用火箭技术取得历史性突破。
    9🟢长鑫科技上市在即,有望纳入哪些指数?何时能借道ETF布局?财联社5其他长鑫科技即将上市,有望逐步纳入多个行业和宽基指数,为ETF布局提供机会,提升底层资产质量。
    10🟢最高预增2001.8%!A股业绩利好刷屏东方财富5业绩预告三维通信预计2026年半年度净利润同比增长1428.58%—2001.80%,源于互联网业务收入增长。
    11🟢碳达峰是约束更是机遇东方财富5宏观政策国务院印发《“十五五”碳达峰行动方案》,提出将绿色低碳导向融入国民经济循环,推动高质量发展,带来产业、科技、市场增量红利
    12🟢多路游资集体出动,浪潮信息、华天科技等获重金扫货 (000977,002185,600353)证券时报5业绩预告浪潮信息预计2026年上半年净利润26-31亿元,同比增长226%-288%,股价连续涨停创历史新高,多路游资重仓买入科
    13🟢长征十号乙首飞告捷 全球首创海上网系回收成功 商业航天下半年迎三重共振财联社5技术突破长征十号乙运载火箭首飞成功,全球首创海上网系回收,标志中国可回收火箭技术取得里程碑式突破。
    14🟢兆易创新业绩预增1099%+工业富联最高预增101% 电子ETF华宝(515260)盘中再涨2.8%!算力硬件迎业绩喜报潮!新浪财经5业绩预告兆易创新和工业富联发布业绩预增公告,分别预增1099%和最高101%。
    15🔴债市早参7月10日财联社5其他龙大转债发生实质性违约,ST龙大无力兑付到期本息,标志可转债市场刚性兑付时代终结。
    16🟢国务院印发“十五五”碳达峰行动方案 设立国家低碳转型基金;完善绿色金融标准和信息披露要求证券时报5宏观政策国务院印发十五五碳达峰行动方案,设定2030年碳减排目标,部署能源结构调整、产业绿色低碳转型等重点任务,并设立国家低碳转
    17🟢突然沸腾,今天是航天投资人!这么多ETF,到底选哪个?证券时报5技术突破长征十号乙火箭成功实施全球首次海上网系回收,大幅降低发射成本,推动商业航天从技术验证走向规模商用。
    18🟢中科曙光官宣!AI基础设施,迈入10万卡时代!浪潮信息创历史新高,大数据ETF华宝(516700)最高上探5.22%新浪财经5技术突破中科曙光官宣首个全国产10万卡AI超集群落成,浪潮信息预告上半年净利润同比增226%-288%,创历史新高。
    19🟢7月10日东方财富财经晚报(附新闻联播) (300475)东方财富5业绩预告香农芯创预计2026上半年净利润35-40亿元,同比增长2118%-2434%,Q2环比增长63%-101%,存储业务大
    20🟢AI芯片龙头燧原科技 科创板IPO注册获批东方财富5其他燧原科技科创板IPO注册获批,拟募资60亿元用于芯片研发,公司为AI芯片龙头,近年营收高速增长。
    + + +

    四、📋 重要事件:公告 / 调研 / 互动 (近 15 日, importance ≥ 2, 共 20 篇)

    +
    #标题重要度事件类型
    1⚪⭐ 晶盛机电:300316晶盛机电投资者关系管理信息20260709 (300316)3投资者调研
    2⚪⭐ 盐湖股份:000792盐湖股份投资者关系管理信息20260708 (000792)3投资者调研
    3⚪高测股份:关于开展期货套期保值业务的可行性分析报告2公司公告
    4⚪高测股份:关于召开2026年第二次临时股东会的通知2公司公告
    5⚪高测股份:关于向公司2025年限制性股票激励计划激励对象授予预留部分限制性股票的公告2公司公告
    6⚪高测股份:青岛高测科技股份有限公司2025年限制性股票激励计划预留授予事项的法律意见书2公司公告
    7⚪高测股份:第四届董事会第十九次会议决议公告2公司公告
    8⚪高测股份:董事会薪酬与考核委员会关于公司2025年限制性股票激励计划预留授予激励对象名单的核查意见(截止预留授予日)2公司公告
    9⚪高测股份:2025年限制性股票激励计划预留授予部分激励对象名单(截止预留授予日)2公司公告
    10⚪高测股份:关于续聘会计师事务所的公告2公司公告
    11⚪高测股份:关于开展期货套期保值业务的公告2公司公告
    12⚪芯联集成:芯联集成电路制造股份有限公司关于变更持续督导保荐代表人的公告2公司公告
    13⚪长江电力:长江电力2025年年度权益分派实施公告2公司公告
    14⚪歌尔股份:关于“家园6号”员工持股计划第三个锁定期届满的提示性公告2公司公告
    15⚪歌尔股份:关于为子公司提供担保的进展公告2公司公告
    16⚪中国建筑:中国建筑2025年年度权益分派实施公告2公司公告
    17⚪长江电力:长江电力第七届董事会第二次会议决议公告2公司公告
    18⚪长江电力:长江电力2026年半年度发电量完成情况公告2公司公告
    19⚪歌尔股份:关于2023年股票期权激励计划预留授予部分第二个行权期采用自主行权模式的提示性公告2公司公告
    20⚪容百科技:2026年半年度业绩预告的自愿性披露公告2公司公告
    + + +

    五、数据总览

    + +

    5.1 M1 → M6 管道

    +
    +
    758 (0 24h)
    M1 原始文章
    +
    0
    M1 cninfo
    +
    755
    M2 正文提取
    +
    597
    M3 去重唯一
    +
    583
    M5 向量
    +
    11415
    M6 Qdrant
    +
    + +

    5.2 各源数据 (0/758 篇 24h 内)

    +
    +
    0 (0 24h)
    cninfo
    +
    37 (0 24h)
    东方财富
    +
    19 (0 24h)
    中国经济网
    +
    107 (0 24h)
    中国证券网
    +
    26 (0 24h)
    中新经纬
    +
    24 (0 24h)
    中证券网
    +
    19 (0 24h)
    华尔街见闻
    +
    0 (0 24h)
    新华网
    +
    202 (0 24h)
    新浪财经
    +
    17 (0 24h)
    新闻联播
    +
    56 (0 24h)
    第一财经
    +
    0 (0 24h)
    经济观察网
    +
    0 (0 24h)
    证券日报
    +
    137 (0 24h)
    证券时报
    +
    114 (0 24h)
    财联社
    +
    0
    📋 cninfo
    +
    + +

    5.3 情绪分布 (最新抓取)

    +
    🟢 利好 161 (27%)🔴 利空 45 (8%)⚪ 中性 388 (65%)
    + +

    5.4 重要度分布

    + + + +
    重要度等级 1等级 2等级 3等级 4等级 5
    数量1161841679037
    + +

    5.5 事件类型分布

    + + + + + + + + + + + + +
    事件类型数量
    其他211
    国际局势107
    宏观政策80
    行业政策53
    业绩预告35
    技术突破29
    产品发布17
    监管处罚16
    投资并购14
    合作签约7
    + +
    +
    +
    +

    A 股 Deep Research 私有投研平台 · 自动生成于 2026-07-11 07:20:27

    +
    +
    + + \ No newline at end of file diff --git a/tests/fixtures/intl_news_daily_20260711_070304.html b/tests/fixtures/intl_news_daily_20260711_070304.html new file mode 100644 index 0000000..66c4840 --- /dev/null +++ b/tests/fixtures/intl_news_daily_20260711_070304.html @@ -0,0 +1,99 @@ + + + + + +国际财经 Deep Research 日报 — 20260711_070304 + + + +
    +
    +

    🌍 国际财经 Deep Research 日报

    +

    20260711_070304 · 生成于 2026-07-11 07:03:30

    +
    +
    +
    + + +

    一、🤖 AI 摘要

    +
      +
    • 美伊冲突升级与霍尔木兹海峡危机:美国空袭伊朗并遭报复,伊朗利用海峡控制权威胁全球石油供应,油价持续飙升,能源成本上升加剧通胀担忧,利空风险资产但利好能源股,是当日影响最大的地缘政治事件。
    • +
    • 日本央行政策转向:日本政府协调养老金增持国内资产并强化加息预期,推动日元显著升值,利好日本金融市场,反映全球货币政策分化加剧。
    • +
    • 欧洲央行加息预期与通胀缓解:9月加息已完全定价,德法通胀符合预期,欧元区通胀压力整体缓解,市场影响中性,但后续加息路径需关注。
    • +
    • 市场情绪基调:整体偏利空——地缘政治风险主导,油价压制股市情绪;分化明显(能源股利好,其他风险资产承压);日本市场因政策利好相对独立;加密行业受监管预期小幅提振。
    • +
    • 值得持续关注的行业与主题:AI行业成本压力与投资分化(Palo Alto要求降价90%,亚马逊、SK海力士加码基础设施);稳定币及加密监管进展(Circle获准运营信托银行);霍尔木兹海峡局势演变;全球通胀与央行政策路径。
    • +
    + + +

    二、🔥 重要事件 (importance ≥ 4, 19 条)

    +
    #标题重要度事件类型摘要
    1⚪美伊紧张局势加剧及供应担忧重燃,油价持续飙升4地缘政治美伊冲突升级,美国空袭伊朗,伊朗报复攻击,双方紧张局势加剧,推动油价大幅上涨。 [investinglive.com]
    2🔴市场能否迎来更平静的周末?4地缘政治美伊冲突持续但未升级,霍尔木兹海峡实际关闭,市场保持谨慎。 [investinglive.com]
    3⚪最新美伊争端中双方均无意升级局势4地缘政治美伊冲突出现缓和迹象,双方均无意升级,但卡塔尔LNG生产放缓,原油市场关注中国成品油出口政策变化。 [investinglive.com]
    4🔴日本要求GPIF增持国内资产,JGB抛售暴露央行独立性担忧4宏观经济日本政府推动GPIF增持国内资产,引发市场对央行独立性担忧,JGB收益率升至数十年高位。 [investinglive.com]
    5🟢伊朗传出新一轮爆炸声4地缘政治伊朗连续遭袭后油价反弹,但伊朗否认爆炸,地缘局势不明朗 [investinglive.com]
    6⚪欧洲央行会议纪要:所有委员认为通胀前景风险偏向上行4央行决议欧洲央行会议纪要显示所有委员认为通胀前景风险偏上行,但强调数据依赖,不给出未来利率路径指引。 [investinglive.com]
    7🔴伊朗革命卫队海军:美国在霍尔木兹海峡的“冒险和干预”只会招致“毁灭性回应”4地缘政治伊朗革命卫队海军警告美国在霍尔木兹海峡的干预将招致毁灭性回应,上周该海峡商业船舶流量下降19%,每日通行次数从120次降至25次。 [investinglive.com]
    8⚪黄金收复失地重归区间震荡,交易员静待美国CPI报告4宏观经济美国CPI数据即将公布,可能对美联储利率预期产生重大影响,进而驱动黄金价格波动。 [investinglive.com]
    9⚪日元走高 - 日本财务大臣寻求推动GPIF等养老金基金投资国内资产的措施4监管政策日本财务大臣推动GPIF等养老金基金增持国内资产,并扩大面向家庭的国债产品,以扩大国内需求、减少对外部买家的依赖,同时为渐进加息做准备。 [investinglive.com]
    10⚪美联储威廉姆斯补充:风险仍更多在通胀方面4央行决议美联储威廉姆斯表示通胀风险仍存,政策需保持数据依赖,劳动力市场稳定。 [investinglive.com]
    11🔴伊朗利用霍尔木兹海峡控制权的策略似乎奏效4地缘政治伊朗利用霍尔木兹海峡控制权施加地缘政治压力,威胁全球石油供应和经济前景。 [investinglive.com]
    12🔴investingLive欧洲时段总结:市场消化美伊头条,情绪转为审慎4地缘政治美伊紧张局势持续,特朗普称伊朗希望达成协议,但市场仍担忧供应中断,油价波动,美元承压,股市谨慎。 [investinglive.com]
    13⚪欧洲央行还会再加息一次吗?4央行决议市场预期欧洲央行今年再加息一次,9月加息已被完全定价,年底前约36个基点加息计入。 [investinglive.com]
    14🟢亚洲股市因芯片反弹和日本养老金资金流入预期而上涨,伊朗风险消退4监管政策日本财务大臣表示将鼓励GPIF等养老基金增加国内资产持有,对日元和日本金融市场构成长期利好。 [investinglive.com]
    15🔴代币反抗走向主流:Palo Alto CEO要求AI价格下降90% [AMZN,PANW]4行业动态Palo Alto Networks CEO要求AI代币价格下降90%,反映企业客户对AI成本的不满,可能引发AI供应商定价压力。 [ZeroHedge]
    16🟢开启7月10日:美元整体走低,美元兑日元领跌4地缘政治中东地缘紧张局势升级,霍尔木兹海峡航运大幅下降,推高原油价格。 [investinglive.com]
    17⚪美国财报季即将到来 [AAPL,AMZN,GOOGL,INTC,META]4财报披露大型科技股如Alphabet、特斯拉、微软、Meta、苹果、亚马逊等将在7月下旬集中披露财报,市场期待业绩表现。 [investinglive.com]
    18🟢稳定币发行商Circle获准开展银行业务,股价上涨5%4监管政策Circle获得美国货币监理署批准运营为信托银行,股价上涨5%。 [CNBC]
    19🔴欧洲市场收盘综述:美伊紧张局势持续,油价小幅上涨4宏观经济德国6月通胀率确认为2.3%,法国通胀下降,欧元区通胀压力缓解 [investinglive.com]
    + + +

    三、📊 数据总览

    + +

    3.1 M1→M6 管道

    +
    +
    622
    M1 原始文章
    +
    622
    M2 正文提取
    +
    123
    M3 去重唯一
    +
    125
    M5 向量
    +
    1569
    M6 Qdrant
    +
    + +

    3.2 情绪分布 (当日事件)

    +
    🟢 利好 13 (18%)🔴 利空 18 (24%)⚪ 中性 43 (58%)
    + +

    3.3 重要度分布

    +
    重要度数量
    等级 26
    等级 346
    等级 422
    + +

    3.4 事件类型 TOP 10

    +
    事件类型数量
    宏观经济17
    地缘政治14
    行业动态10
    市场异动7
    央行决议7
    大宗商品4
    外汇波动4
    监管政策4
    财报披露4
    技术突破1
    + +

    3.5 文章来源分布

    +
    来源文章数
    ForexLive44
    ZeroHedge3
    CNBC2
    Yahoo Finance2
    Seeking Alpha1
    + +
    +
    +
    +

    国际财经 Deep Research 私有投研平台 · 自动生成于 2026-07-11 07:03:30

    +
    +
    + + \ No newline at end of file diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..b4a244f --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,93 @@ +"""统一 CLI 测试。 + +验证子命令路由 + argparse 解析正确。 +""" + +from __future__ import annotations + +import sys +from unittest.mock import patch + +from a_share_cli.main import main + + +def _run(args: str) -> int: + with patch.object(sys, "argv", ["a-share", *args.split()]): + try: + return main() + except SystemExit as e: + return e.code if isinstance(e.code, int) else 1 + + +def test_no_args_shows_help() -> None: + rc = _run("") + assert rc == 0 + + +def test_crawl_default() -> None: + with patch("a_share_cli.main.cmd_crawl", return_value=0) as mock: + _run("crawl") + mock.assert_called_once() + + +def test_crawl_with_source() -> None: + with patch("a_share_cli.main.cmd_crawl", return_value=0) as mock: + _run("crawl --source cls") + args = mock.call_args[0][0] + assert args.source == "cls" + + +def test_extract_with_date() -> None: + with patch("a_share_cli.main.cmd_extract", return_value=0) as mock: + _run("extract --date 20260616 --source sina") + args = mock.call_args[0][0] + assert args.date == "20260616" + assert args.source == "sina" + + +def test_events_with_provider() -> None: + with patch("a_share_cli.main.cmd_events", return_value=0) as mock: + _run("events --provider qwen --limit 5") + args = mock.call_args[0][0] + assert args.provider == "qwen" + assert args.limit == 5 + + +def test_search_default() -> None: + with patch("a_share_cli.main.cmd_search", return_value=0) as mock: + _run("search 宁德时代") + args = mock.call_args[0][0] + assert args.query == "宁德时代" + assert args.top == 10 + + +def test_search_with_filters() -> None: + with patch("a_share_cli.main.cmd_search", return_value=0) as mock: + _run("search 芯片 --source cls --sentiment positive --min-importance 3 --top 5") + args = mock.call_args[0][0] + assert args.query == "芯片" + assert args.source == "cls" + assert args.sentiment == "positive" + assert args.min_importance == 3 + assert args.top == 5 + + +def test_search_with_stock() -> None: + with patch("a_share_cli.main.cmd_search", return_value=0) as mock: + _run("search 重大合同 --stock 300750.sz") + args = mock.call_args[0][0] + assert args.stock == "300750.sz" + + +def test_pipeline_once() -> None: + with patch("a_share_cli.main.cmd_pipeline", return_value=0) as mock: + _run("pipeline --once --steps crawler,extractor") + args = mock.call_args[0][0] + assert args.once is True + assert args.steps == "crawler,extractor" + + +def test_status() -> None: + with patch("a_share_cli.main.cmd_status", return_value=0) as mock: + _run("status") + mock.assert_called_once() diff --git a/tests/test_crawler.py b/tests/test_crawler.py new file mode 100644 index 0000000..9488fad --- /dev/null +++ b/tests/test_crawler.py @@ -0,0 +1,333 @@ +"""M1 抓取模块单元测试。 + +不依赖真实网络:用 mock 替换 Crawl4AI 的 arun。 +""" + +from __future__ import annotations + +import asyncio +import json +from datetime import date +from pathlib import Path +from typing import Any +from unittest.mock import AsyncMock + +import pytest + +from crawler import engine +from crawler.config import load_crawler_config +from crawler.engine import ( + crawl_url_with_retry, + extract_article_links, +) +from crawler.models import CrawlerSettings, CrawlResult, CrawlStage, SourceConfig +from crawler.storage import build_output_dir, save_result, url_hash + +# --------------------------------------------------------------------------- # +# 配置加载 +# --------------------------------------------------------------------------- # + +def test_load_crawler_config_ok(sample_sources_yaml: Path) -> None: + cfg = load_crawler_config(sample_sources_yaml) + assert cfg.settings.concurrency == 2 + assert len(cfg.sources) == 1 + assert cfg.sources[0].id == "testsrc" + assert cfg.enabled_sources()[0].id == "testsrc" + + +def test_load_crawler_config_real_sources_yaml() -> None: + """项目内置的 configs/sources.yaml 必须可解析(回归保护)。""" + real = Path("configs/sources.yaml") + if not real.is_file(): + pytest.skip("configs/sources.yaml 未生成,跳过") + cfg = load_crawler_config(real) + # 验收标准: 至少 5 个启用源 + assert len(cfg.enabled_sources()) >= 5, "启用源应至少 5 个(M1 验收标准)" + assert cfg.settings.concurrency == 3, "用户决策: 并发上限 3" + + +def test_load_crawler_config_missing_file(tmp_path: Path) -> None: + with pytest.raises(FileNotFoundError): + load_crawler_config(tmp_path / "nonexistent.yaml") + + +def test_source_id_validation() -> None: + with pytest.raises(ValueError): + SourceConfig( + id="Bad-ID", # 含大写与连字符 + name="x", + homepage="https://example.com", + article_url_pattern="^.*$", + ) + + +# --------------------------------------------------------------------------- # +# 链接抽取 +# --------------------------------------------------------------------------- # + +def _src(**kwargs: Any) -> SourceConfig: + base: dict[str, Any] = { + "id": "testsrc", + "name": "测试", + "homepage": "https://example.com/list", + "article_url_pattern": r"^https://example\.com/article/\d+$", + "js_render": False, + "max_articles_per_run": 10, + } + base.update(kwargs) + return SourceConfig(**base) + + +def test_extract_article_links_basic() -> None: + html = """ + + 文章一 + 文章二 + 外站 + 关于 + JS + 文章一(重复) + + """ + links = extract_article_links(html, "https://example.com/list", _src()) + urls = [link.url for link in links] + assert urls == [ + "https://example.com/article/123", + "https://example.com/article/456", + ] + assert links[0].anchor_text == "文章一" + + +def test_extract_article_links_respects_max() -> None: + html_parts = [ + f'a{i}' for i in range(20) + ] + html = "" + "".join(html_parts) + "" + src = _src(max_articles_per_run=3) + links = extract_article_links(html, "https://example.com/list", src) + assert len(links) == 3 + + +def test_extract_article_links_strips_fragment() -> None: + html = 'x' + links = extract_article_links(html, "https://example.com/list", _src()) + assert links[0].url == "https://example.com/article/1" + + +# --------------------------------------------------------------------------- # +# 重试机制 +# --------------------------------------------------------------------------- # + +class _FakeC4Result: + """模拟 Crawl4AI 的返回对象。""" + + def __init__( + self, + success: bool = True, + html: str = "ok", + markdown: str = "ok", + status_code: int = 200, + error_message: str | None = None, + ) -> None: + self.success = success + self.html = html + self.markdown = markdown + self.status_code = status_code + self.error_message = error_message + + +@pytest.mark.asyncio +async def test_retry_succeeds_on_third_attempt() -> None: + """前两次失败,第三次成功;返回的 attempts 应为 3。""" + fake_crawler = AsyncMock() + fake_crawler.arun = AsyncMock( + side_effect=[ + _FakeC4Result(success=False, html="", error_message="boom-1"), + _FakeC4Result(success=False, html="", error_message="boom-2"), + _FakeC4Result(success=True), + ] + ) + + settings = CrawlerSettings( + concurrency=1, + retry_max_attempts=3, + retry_min_wait_sec=0.0, + retry_max_wait_sec=0.0, + ) + sem = asyncio.Semaphore(1) + src = _src() + res = await crawl_url_with_retry( + crawler=fake_crawler, + url="https://example.com/article/1", + source=src, + stage=CrawlStage.ARTICLE, + settings=settings, + semaphore=sem, + ) + assert res.success is True + assert res.attempts == 3 + assert fake_crawler.arun.await_count == 3 + + +@pytest.mark.asyncio +async def test_retry_gives_up_after_max() -> None: + fake_crawler = AsyncMock() + fake_crawler.arun = AsyncMock( + return_value=_FakeC4Result(success=False, html="", error_message="nope") + ) + settings = CrawlerSettings( + concurrency=1, + retry_max_attempts=2, + retry_min_wait_sec=0.0, + retry_max_wait_sec=0.0, + ) + sem = asyncio.Semaphore(1) + res = await crawl_url_with_retry( + crawler=fake_crawler, + url="https://example.com/article/1", + source=_src(), + stage=CrawlStage.ARTICLE, + settings=settings, + semaphore=sem, + ) + assert res.success is False + assert res.attempts == 2 + assert fake_crawler.arun.await_count == 2 + assert res.error == "nope" + + +@pytest.mark.asyncio +async def test_exception_is_swallowed_and_retried() -> None: + """arun 抛异常应被捕获并触发重试。""" + fake_crawler = AsyncMock() + fake_crawler.arun = AsyncMock( + side_effect=[RuntimeError("net down"), _FakeC4Result(success=True)] + ) + settings = CrawlerSettings( + concurrency=1, retry_max_attempts=2, retry_min_wait_sec=0.0, retry_max_wait_sec=0.0 + ) + sem = asyncio.Semaphore(1) + res = await crawl_url_with_retry( + crawler=fake_crawler, + url="https://example.com/article/1", + source=_src(), + stage=CrawlStage.ARTICLE, + settings=settings, + semaphore=sem, + ) + assert res.success is True + assert res.attempts == 2 + + +# --------------------------------------------------------------------------- # +# 存储 +# --------------------------------------------------------------------------- # + +def test_url_hash_stable() -> None: + h1 = url_hash("https://example.com/a") + h2 = url_hash("https://example.com/a") + h3 = url_hash("https://example.com/b") + assert h1 == h2 + assert h1 != h3 + assert len(h1) == 16 + + +def test_build_output_dir(tmp_path: Path) -> None: + d = build_output_dir(tmp_path, "cls", date(2026, 6, 16)) + assert d == tmp_path / "cls" / "20260616" + + +def test_save_result_writes_html_md_and_index(tmp_path: Path) -> None: + result = CrawlResult( + source_id="cls", + stage=CrawlStage.ARTICLE, + url="https://example.com/article/1", + success=True, + status_code=200, + title="标题", + html="hi", + markdown="# hi", + ) + html_path = save_result(result, tmp_path, day=date(2026, 6, 16)) + assert html_path is not None and html_path.is_file() + assert html_path.read_text(encoding="utf-8") == "hi" + + md_path = html_path.with_suffix(".md") + assert md_path.is_file() + assert md_path.read_text(encoding="utf-8") == "# hi" + + index = html_path.parent / "index.jsonl" + assert index.is_file() + line = index.read_text(encoding="utf-8").strip() + obj = json.loads(line) + assert obj["url"] == "https://example.com/article/1" + assert obj["success"] is True + assert obj["html_file"] == html_path.name + assert "html" not in obj # 大字段不应进入元数据 + + +def test_save_result_failure_only_appends_index(tmp_path: Path) -> None: + result = CrawlResult( + source_id="cls", + stage=CrawlStage.ARTICLE, + url="https://example.com/article/2", + success=False, + error="timeout", + ) + html_path = save_result(result, tmp_path, day=date(2026, 6, 16)) + assert html_path is None + index = tmp_path / "cls" / "20260616" / "index.jsonl" + assert index.is_file() + obj = json.loads(index.read_text(encoding="utf-8").strip()) + assert obj["success"] is False + + +# --------------------------------------------------------------------------- # +# Markdown 兼容 +# --------------------------------------------------------------------------- # + +def test_markdown_text_handles_str() -> None: + assert engine._markdown_text("plain") == "plain" + + +def test_markdown_text_handles_object_raw_markdown() -> None: + class _Obj: + raw_markdown = "from raw" + + assert engine._markdown_text(_Obj()) == "from raw" + + +def test_markdown_text_handles_none() -> None: + assert engine._markdown_text(None) == "" + + +# --------------------------------------------------------------------------- # +# 集成测试(默认跳过,需要真实浏览器与网络) +# --------------------------------------------------------------------------- # + +@pytest.mark.integration +@pytest.mark.asyncio +async def test_real_homepage_crawl_smoke() -> None: + """真实抓取 example.com 烟测,验证端到端可运行。 + + 运行: uv run pytest -m integration + """ + from crawler import crawl_all + from crawler.models import CrawlerConfig + + cfg = CrawlerConfig( + settings=CrawlerSettings(concurrency=1, retry_max_attempts=1, output_root="data/raw_test"), + sources=[ + SourceConfig( + id="example", + name="example", + homepage="https://example.com/", + article_url_pattern=r"^https://www\.iana\.org/.*$", + js_render=False, + page_timeout_ms=15000, + max_articles_per_run=1, + ) + ], + ) + results = await crawl_all(cfg, save=False) + assert any(r.success for r in results), "example.com 烟测应至少一个成功" diff --git a/tests/test_dedup.py b/tests/test_dedup.py new file mode 100644 index 0000000..69c07e6 --- /dev/null +++ b/tests/test_dedup.py @@ -0,0 +1,502 @@ +"""M3 三层去重模块单元测试。""" + +from __future__ import annotations + +from datetime import datetime +from pathlib import Path + +import pytest + +from dedup import ( + DEFAULT_HAMMING_THRESHOLD, + Deduper, + DedupLayer, + Fingerprint, + FingerprintStore, + article_to_fingerprint, + content_hash, + hamming, + normalize_content, + simhash64, +) +from extractor import Article + +# --------------------------------------------------------------------------- # +# fixtures +# --------------------------------------------------------------------------- # + +def _article( + *, + url: str = "https://www.cls.cn/detail/1", + url_hash: str = "abc1234567890000", + source_id: str = "cls", + title: str = "宁德时代发布新一代麒麟电池", + content: str = ( + "宁德时代今日正式发布了新一代麒麟电池产品,能量密度达到 255 Wh/kg," + "显著优于上一代产品。该电池将于2026年第三季度量产。" + ), + publish_time: datetime | None = datetime(2026, 6, 16, 10, 0), +) -> Article: + return Article( + source_id=source_id, + url=url, + url_hash=url_hash, + title=title, + content=content, + publish_time=publish_time, + word_count=len(content), + ) + + +@pytest.fixture +def tmp_db(tmp_path: Path) -> Path: + return tmp_path / "fp.sqlite3" + + +# --------------------------------------------------------------------------- # +# hasher +# --------------------------------------------------------------------------- # + +def test_normalize_content_strips_punct_and_whitespace() -> None: + norm = normalize_content("你好, 世界!\n这是 中文。") + assert norm == "你好世界这是中文" + + +def test_normalize_content_handles_empty() -> None: + assert normalize_content("") == "" + assert normalize_content(" \n\t ") == "" + + +def test_content_hash_deterministic_and_punct_invariant() -> None: + a = "今天天气很好。" + b = "今天,天气,很好!!!" + assert content_hash(a) == content_hash(b) + + +def test_content_hash_differs_for_different_text() -> None: + assert content_hash("今天天气很好") != content_hash("今天天气不好") + + +def test_simhash_identical_text_same_value() -> None: + text = "宁德时代发布新一代麒麟电池产品 能量密度大幅提升" + assert simhash64(text) == simhash64(text) + + +def test_simhash_minor_changes_close_distance() -> None: + """长文本(贴近真实新闻)的轻度改写,hamming 距离应在阈值内。""" + base = ( + "宁德时代今日正式发布新一代麒麟电池产品,能量密度达到 255 瓦时每公斤," + "显著优于上一代产品。该电池将于 2026 年第三季度量产,首批应用于多款新能源汽车。" + "公司股价应声上涨 5.2%,分析师认为这将进一步巩固宁德时代在全球动力电池领域的领先地位。" + ) * 2 + rewritten = "财联社讯:" + base + "(完)" + d = hamming(simhash64(base), simhash64(rewritten)) + assert d <= DEFAULT_HAMMING_THRESHOLD, f"长文本前后加标识汉明距离 {d} 不应超过阈值" + + +def test_simhash_unrelated_text_far_distance() -> None: + """完全不相关的两段长文本汉明距离应远大于阈值。""" + a = "宁德时代发布新一代麒麟电池产品,能量密度达到255瓦时每公斤。" * 3 + b = "美联储宣布维持利率不变,市场普遍预期下次会议将开启降息周期。" * 3 + d = hamming(simhash64(a), simhash64(b)) + assert d > DEFAULT_HAMMING_THRESHOLD * 2 + + +def test_simhash_empty_returns_zero() -> None: + assert simhash64("") == 0 + assert simhash64(" ") == 0 + + +def test_hamming_basics() -> None: + assert hamming(0, 0) == 0 + assert hamming(0xFF, 0x00) == 8 + assert hamming(0xFF00FF00, 0x00FF00FF) == 32 + + +# --------------------------------------------------------------------------- # +# FingerprintStore +# --------------------------------------------------------------------------- # + +def test_store_upsert_and_get(tmp_db: Path) -> None: + fp = Fingerprint( + url_hash="hash1", + content_hash="ch1", + simhash=0xDEADBEEFCAFEBABE, + source_id="cls", + url="https://x/1", + title="A", + publish_date="2026-06-16", + ) + with FingerprintStore(tmp_db) as store: + store.upsert(fp) + got = store.get_by_url_hash("hash1") + assert got is not None + assert got.content_hash == "ch1" + assert got.simhash == 0xDEADBEEFCAFEBABE + assert got.publish_date == "2026-06-16" + + +def test_store_upsert_replaces_existing(tmp_db: Path) -> None: + base = Fingerprint( + url_hash="h", + content_hash="ch1", + simhash=1, + source_id="cls", + url="u", + title="t", + ) + updated = base.model_copy(update={"content_hash": "ch2", "simhash": 999}) + with FingerprintStore(tmp_db) as store: + store.upsert(base) + store.upsert(updated) + got = store.get_by_url_hash("h") + assert got is not None + assert got.content_hash == "ch2" + assert got.simhash == 999 + assert store.count() == 1 + + +def test_store_find_by_content_hash(tmp_db: Path) -> None: + with FingerprintStore(tmp_db) as store: + store.upsert(Fingerprint( + url_hash="h1", content_hash="ch", simhash=0, + source_id="cls", url="u1", title="t1" + )) + assert store.find_by_content_hash("ch") is not None + assert store.find_by_content_hash("nope") is None + + +def test_store_candidates_within_window(tmp_db: Path) -> None: + with FingerprintStore(tmp_db) as store: + for d, h in [("2026-05-01", "old"), ("2026-06-15", "near"), ("2026-07-30", "far")]: + store.upsert(Fingerprint( + url_hash=h, content_hash=h, simhash=0, + source_id="cls", url=f"u/{h}", title=h, publish_date=d, + )) + cands = store.candidates_for_simhash("2026-06-16", window_days=7) + url_hashes = sorted(c.url_hash for c in cands) + assert url_hashes == ["near"] + + +def test_store_candidates_no_date_returns_all(tmp_db: Path) -> None: + with FingerprintStore(tmp_db) as store: + store.upsert(Fingerprint( + url_hash="h1", content_hash="c1", simhash=0, + source_id="cls", url="u", title="t", publish_date=None + )) + cands = store.candidates_for_simhash(None, 30) + assert len(cands) == 1 + + +def test_store_simhash_handles_high_bit(tmp_db: Path) -> None: + """64 位 SimHash 高位为 1 时,hex 存取应保持无符号。""" + high = (1 << 63) | 0x1234 + with FingerprintStore(tmp_db) as store: + store.upsert(Fingerprint( + url_hash="h", content_hash="c", simhash=high, + source_id="cls", url="u", title="t", + )) + got = store.get_by_url_hash("h") + assert got is not None + assert got.simhash == high + + +def test_store_count_by_source(tmp_db: Path) -> None: + with FingerprintStore(tmp_db) as store: + for i, src in enumerate(["cls", "cls", "sina"]): + store.upsert(Fingerprint( + url_hash=f"h{i}", content_hash=f"c{i}", simhash=i, + source_id=src, url=f"u{i}", title=f"t{i}", + )) + counts = store.count_by_source() + assert counts == {"cls": 2, "sina": 1} + + +# --------------------------------------------------------------------------- # +# Deduper - 三层判重 +# --------------------------------------------------------------------------- # + +def test_dedup_first_article_is_unique(tmp_db: Path) -> None: + art = _article() + with Deduper(db_path=tmp_db) as d: + result = d.ingest(art) + assert not result.is_duplicate + assert result.matched_layer is None + assert d.stats().total == 1 + + +def test_dedup_layer1_url_hash(tmp_db: Path) -> None: + """同一 url_hash 直接命中 L1。""" + a1 = _article() + a2 = _article() # 同 url_hash 同 url + with Deduper(db_path=tmp_db) as d: + d.ingest(a1) + result = d.ingest(a2) + assert result.is_duplicate + assert result.matched_layer == DedupLayer.URL + assert d.stats().total == 1, "L1 命中应不写入新指纹" + + +def test_dedup_layer2_content_hash(tmp_db: Path) -> None: + """url 不同但 content 完全一致 -> L2。""" + a1 = _article(url="https://a.com/1", url_hash="hash1aaaaaaaaaaa") + a2 = _article(url="https://b.com/2", url_hash="hash2bbbbbbbbbbb") + with Deduper(db_path=tmp_db) as d: + d.ingest(a1) + result = d.ingest(a2) + assert result.is_duplicate + assert result.matched_layer == DedupLayer.CONTENT + assert result.matched_url_hash == "hash1aaaaaaaaaaa" + + +def test_dedup_layer2_punctuation_difference_still_caught(tmp_db: Path) -> None: + """标点/空白差异不应阻止 L2 命中(normalize_content 应剥离)。""" + base = "今天天气很好我们去公园散步" + a1 = _article( + url="https://a/1", url_hash="aaaa", content="今天天气很好。我们去公园散步!" + ) + a2 = _article( + url="https://b/2", url_hash="bbbb", content="今天天气,很好;我们去公园 散步!!" + ) + assert content_hash(a1.content) == content_hash(a2.content) + assert normalize_content(a1.content) == base + with Deduper(db_path=tmp_db) as d: + d.ingest(a1) + result = d.ingest(a2) + assert result.matched_layer == DedupLayer.CONTENT + + +def test_dedup_layer3_simhash_minor_rewrite(tmp_db: Path) -> None: + """长文本 + 转载前后缀,落入 SimHash 层(贴近真实跨源转载场景)。""" + long_body = ( + "宁德时代今日正式发布新一代麒麟电池产品,能量密度达到 255 瓦时每公斤," + "显著优于上一代产品。该电池将于 2026 年第三季度量产,首批应用于多款新能源汽车。" + "公司股价应声上涨 5.2%,分析师认为这将进一步巩固宁德时代在全球动力电池领域的领先地位。" + ) * 2 + rewritten = "财联社讯:" + long_body + "(完)" + a1 = _article(url="https://a/1", url_hash="aaaaa", content=long_body) + a2 = _article(url="https://b/2", url_hash="bbbbb", content=rewritten) + # 必要前提:content_hash 不同(否则会被 L2 截胡) + assert content_hash(a1.content) != content_hash(a2.content) + + with Deduper(db_path=tmp_db) as d: + d.ingest(a1) + result = d.ingest(a2) + assert result.is_duplicate + assert result.matched_layer == DedupLayer.SIMHASH + assert result.hamming_distance is not None + assert result.hamming_distance <= DEFAULT_HAMMING_THRESHOLD + + +def test_dedup_layer3_unrelated_articles_kept(tmp_db: Path) -> None: + a1 = _article(url="https://a/1", url_hash="aaaaa", + content="宁德时代发布新一代麒麟电池产品,能量密度达到 255 瓦时每公斤。" * 5) + a2 = _article(url="https://b/2", url_hash="bbbbb", + content="美联储宣布维持联邦基金利率不变,市场预期下次会议将开启降息。" * 5, + title="美联储利率决议") + with Deduper(db_path=tmp_db) as d: + d.ingest(a1) + result = d.ingest(a2) + assert not result.is_duplicate + assert d.stats().total == 2 + + +def test_dedup_layer3_outside_time_window_kept(tmp_db: Path) -> None: + """SimHash 相近,但 publish_date 距离过远(> 30 天)不去重。""" + body = ( + "宁德时代今日正式发布新一代麒麟电池产品,能量密度达到 255 瓦时每公斤," + "显著优于上一代产品。该电池将于第三季度量产,首批应用于多款新能源汽车。" * 2 + ) + a1 = _article( + url="https://a/1", url_hash="aaaa1", content=body, + publish_time=datetime(2026, 1, 1, 9, 0), + ) + a2 = _article( + url="https://b/2", url_hash="bbbb2", content=body[3:], # 微改 -> 走 L3 + publish_time=datetime(2026, 6, 16, 9, 0), + ) + assert content_hash(a1.content) != content_hash(a2.content) + with Deduper(db_path=tmp_db, time_window_days=30) as d: + d.ingest(a1) + result = d.ingest(a2) + assert not result.is_duplicate, "时间窗口外不应命中 SimHash" + + +def test_dedup_threshold_zero_only_exact_simhash(tmp_db: Path) -> None: + """阈值 0 -> 仅当 SimHash 完全相同才视为重复(且会先被 L2 拦截)。""" + a1 = _article(url="https://a/1", url_hash="aaaa1", + content="宁德时代发布新一代麒麟电池产品 能量密度大幅提升") + a2 = _article(url="https://b/2", url_hash="bbbb2", + content="财联社讯 宁德时代今天发布了新一代麒麟电池 能量密度提升明显") + with Deduper(db_path=tmp_db, simhash_threshold=0) as d: + d.ingest(a1) + result = d.ingest(a2) + # 两段相似但不同的文本,阈值 0 时不应判重 + assert not result.is_duplicate + + +# --------------------------------------------------------------------------- # +# Deduper - check 不写入 +# --------------------------------------------------------------------------- # + +def test_check_does_not_write(tmp_db: Path) -> None: + art = _article() + with Deduper(db_path=tmp_db) as d: + result = d.check(art) + assert not result.is_duplicate + assert d.stats().total == 0 # check 不应入库 + + +def test_article_to_fingerprint_fields() -> None: + art = _article() + fp = article_to_fingerprint(art) + assert fp.url_hash == art.url_hash + assert fp.simhash == simhash64(art.content) + assert fp.content_hash == content_hash(art.content) + assert fp.publish_date == "2026-06-16" + + +def test_article_to_fingerprint_handles_none_publish_time() -> None: + art = _article(publish_time=None) + fp = article_to_fingerprint(art) + assert fp.publish_date is None + + +# --------------------------------------------------------------------------- # +# stats +# --------------------------------------------------------------------------- # + +def test_stats_aggregates_by_source(tmp_db: Path) -> None: + with Deduper(db_path=tmp_db) as d: + d.ingest(_article(source_id="cls", url="https://cls/1", + url_hash="cls0000000000001")) + d.ingest(_article(source_id="cls", url="https://cls/2", + url_hash="cls0000000000002", + content="完全不同的另一篇文章" * 30)) + d.ingest(_article(source_id="sina", url="https://sina/1", + url_hash="sina000000000001", + content="第三篇 完全不同 主题 美联储 利率" * 20)) + stats = d.stats() + assert stats.total == 3 + assert stats.by_source == {"cls": 2, "sina": 1} + assert stats.earliest is not None + + +# --------------------------------------------------------------------------- # +# 多源记录(source_ids) +# --------------------------------------------------------------------------- # + +def test_fingerprint_source_ids_default_to_source() -> None: + """source_ids 未显式给定时,自动包含主源 source_id。""" + fp = Fingerprint( + url_hash="h", content_hash="c", simhash=0, + source_id="cls", url="u", title="t", + ) + assert fp.source_ids == ["cls"] + + +def test_fingerprint_source_ids_keeps_main_source_first() -> None: + """source_ids 无论怎么传,主源 source_id 始终居首且去重。""" + fp = Fingerprint( + url_hash="h", content_hash="c", simhash=0, + source_id="cls", url="u", title="t", + source_ids=["sina", "cls", "eastmoney", "sina"], + ) + assert fp.source_ids[0] == "cls" + assert len(fp.source_ids) == len(set(fp.source_ids)) # 无重复 + + +def test_store_persists_source_ids(tmp_db: Path) -> None: + fp = Fingerprint( + url_hash="h", content_hash="c", simhash=0, + source_id="cls", url="u", title="t", + source_ids=["cls", "sina", "eastmoney"], + ) + with FingerprintStore(tmp_db) as store: + store.upsert(fp) + got = store.get_by_url_hash("h") + assert got is not None + assert got.source_ids == ["cls", "sina", "eastmoney"] + + +def test_store_migrates_old_schema_without_source_ids(tmp_db: Path) -> None: + """旧库(无 source_ids 列)打开时应自动迁移,旧数据回退为 [source_id]。""" + import sqlite3 + + conn = sqlite3.connect(tmp_db) + conn.executescript( + "CREATE TABLE fingerprints (" + " url_hash TEXT PRIMARY KEY, content_hash TEXT NOT NULL, simhash_hex TEXT NOT NULL," + " source_id TEXT NOT NULL, url TEXT NOT NULL, title TEXT NOT NULL," + " publish_date TEXT, ingested_at TEXT NOT NULL);" + ) + conn.execute( + "INSERT INTO fingerprints VALUES (?, ?, ?, ?, ?, ?, ?, ?)", + ("old1", "ch1", "0000000000000000", "cls", "u1", "t1", "2026-06-01", "2026-06-01T00:00:00"), + ) + conn.commit() + conn.close() + + with FingerprintStore(tmp_db) as store: + got = store.get_by_url_hash("old1") + assert got is not None + assert got.source_ids == ["cls"] # 迁移后回退主源 + # 迁移后可正常写入多源 + store.upsert(Fingerprint( + url_hash="new1", content_hash="c2", simhash=1, + source_id="sina", url="u2", title="t2", + source_ids=["sina", "cls"], + )) + assert store.get_by_url_hash("new1") is not None # type: ignore[union-attr] + + +def test_ingest_merges_sources_on_duplicate(tmp_db: Path) -> None: + """同一内容被多个源发布时,重复文章的来源并入唯一新闻指纹。""" + body = "宁德时代今日发布新一代麒麟电池,能量密度 255Wh/kg。" * 4 + a1 = _article(source_id="cls", url="https://cls/a", url_hash="aaaa111111111111", + content=body) + a2 = _article(source_id="sina", url="https://sina/b", url_hash="bbbb222222222222", + content=body) + a3 = _article(source_id="eastmoney", url="https://em/c", url_hash="cccc333333333333", + content=body) + + with Deduper(db_path=tmp_db) as d: + r1 = d.ingest(a1) + assert not r1.is_duplicate + r2 = d.ingest(a2) + assert r2.is_duplicate + assert r2.matched_layer == DedupLayer.CONTENT + assert r2.matched_source_id == "cls" + # 命中后 all_source_ids 立即包含两个源 + assert r2.all_source_ids == ["cls", "sina"] + + r3 = d.ingest(a3) + assert r3.is_duplicate + assert r3.all_source_ids == ["cls", "sina", "eastmoney"] + + # 指纹库持久化多源 + matched = d.store.get_by_url_hash("aaaa111111111111") + assert matched is not None + assert matched.source_ids == ["cls", "sina", "eastmoney"] + assert d.stats().total == 1 # 内容组只算 1 条唯一 + + +def test_check_reports_all_sources_without_writing(tmp_db: Path) -> None: + """check(只读)命中重复时也能看到全部来源,且不写库。""" + body = "宁德时代发布新一代麒麟电池产品。" * 6 + a1 = _article(source_id="cls", url="https://cls/a", url_hash="aaaa111111111111", + content=body) + a2 = _article(source_id="sina", url="https://sina/b", url_hash="bbbb222222222222", + content=body) + + with Deduper(db_path=tmp_db) as d: + d.ingest(a1) + d.ingest(a2) + # 第三次来一篇同样内容的文章,仅 check + a3 = _article(source_id="eastmoney", url="https://em/c", url_hash="cccc333333333333", + content=body) + result = d.check(a3) + assert result.is_duplicate + assert result.matched_source_id == "cls" + assert result.all_source_ids == ["cls", "sina"] + assert d.stats().total == 1 # check 不写库 diff --git a/tests/test_embedding.py b/tests/test_embedding.py new file mode 100644 index 0000000..8385332 --- /dev/null +++ b/tests/test_embedding.py @@ -0,0 +1,397 @@ +"""M5 嵌入模块单元测试。 + +不依赖真实 LLM/HuggingFace,所有 provider 调用通过 mock 注入。 +""" + +from __future__ import annotations + +import asyncio +import json +from datetime import datetime +from pathlib import Path +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from embedding import ( + DASHSCOPE_BATCH_LIMIT, + AsyncEmbeddingProvider, + EmbeddingError, + EmbeddingProvider, + EmbeddingProviderType, + EmbeddingResult, + compose_text, + make_async_provider, + make_sync_provider, + resolve_provider_type, +) +from embedding.base import _from_event_dict +from embedding.remote import ( + DashScopeAsyncEmbeddingProvider, + DashScopeEmbeddingProvider, + _chunked, +) +from extractor import Article + +# --------------------------------------------------------------------------- # +# fixtures +# --------------------------------------------------------------------------- # + +def _article( + *, + url: str = "https://www.cls.cn/detail/1", + url_hash: str = "abc1234567890000", + title: str = "宁德时代签订 100GWh 长期供货协议", + content: str = "宁德时代与某车企签 5 年 100GWh 协议,涉及金额超 1500 亿。" * 3, + publish_time: datetime | None = datetime(2026, 6, 16, 10, 0), +) -> Article: + return Article( + source_id="cls", + url=url, + url_hash=url_hash, + title=title, + content=content, + publish_time=publish_time, + word_count=len(content), + ) + + +def _embedding_response(vectors: list[list[float]]) -> MagicMock: + """构造与 OpenAI SDK 一致的 embeddings.create 返回。""" + resp = MagicMock() + resp.data = [MagicMock(embedding=v) for v in vectors] + return resp + + +# --------------------------------------------------------------------------- # +# compose_text +# --------------------------------------------------------------------------- # + +def test_compose_text_basic() -> None: + art = _article() + text = compose_text(art) + assert text.startswith("标题:") + assert "正文:" in text + assert art.title in text + assert art.content[:30] in text + + +def test_compose_text_with_head_and_summary() -> None: + art = _article() + text = compose_text(art, head="[sentiment=positive]", summary="一句话摘要") + assert "[sentiment=positive]" in text + assert "摘要:一句话摘要" in text + + +def test_compose_text_truncates_overlong() -> None: + art = _article(content="字" * 10000) + text = compose_text(art, max_chars=500) + assert len(text) <= 500 + + +def test_from_event_dict_extracts_head_and_article() -> None: + event_obj = { + "source_id": "cls", + "url": "https://x/1", + "url_hash": "h1", + "title": "宁德合作", + "publish_time": "2026-06-16T10:00:00", + "event": { + "stock_codes": ["300750.SZ"], + "company_names": ["宁德时代"], + "industries": ["动力电池"], + "sentiment": "positive", + "importance": 5, + "event_type": "重大合同", + "summary": "签订 100GWh 协议", + }, + } + article, head, summary = _from_event_dict(event_obj) + assert article.url_hash == "h1" + assert article.publish_time == datetime(2026, 6, 16, 10, 0, 0) + assert "sentiment=positive" in head + assert "importance=5" in head + assert "300750.SZ" in head + assert "宁德时代" in head + assert "动力电池" in head + assert summary == "签订 100GWh 协议" + + +def test_from_event_dict_handles_missing_publish_time() -> None: + article, _, _ = _from_event_dict({"source_id": "x", "url": "u", "url_hash": "h", + "title": "t", "event": { + "sentiment": "neutral", + "importance": 1, + "event_type": "其他", + }}) + assert article.publish_time is None + + +# --------------------------------------------------------------------------- # +# remote 工具 +# --------------------------------------------------------------------------- # + +def test_chunked_splits_evenly() -> None: + assert _chunked(list(range(7)), 3) == [[0, 1, 2], [3, 4, 5], [6]] + assert _chunked([], 3) == [] + assert _chunked([1, 2, 3], 10) == [[1, 2, 3]] + + +def test_dashscope_batch_limit_is_10() -> None: + assert DASHSCOPE_BATCH_LIMIT == 10 + + +# --------------------------------------------------------------------------- # +# Provider type 解析 +# --------------------------------------------------------------------------- # + +def test_resolve_provider_type_dashscope_aliases(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("EMBEDDING_PROVIDER", raising=False) + assert resolve_provider_type("dashscope") == EmbeddingProviderType.DASHSCOPE + assert resolve_provider_type("qwen") == EmbeddingProviderType.DASHSCOPE + assert resolve_provider_type("remote") == EmbeddingProviderType.DASHSCOPE + + +def test_resolve_provider_type_local_aliases(monkeypatch: pytest.MonkeyPatch) -> None: + for name in ("local", "local-bge", "bge", "bge-m3"): + assert resolve_provider_type(name) == EmbeddingProviderType.LOCAL_BGE + + +def test_resolve_provider_type_default_is_dashscope(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("EMBEDDING_PROVIDER", raising=False) + assert resolve_provider_type() == EmbeddingProviderType.DASHSCOPE + + +def test_resolve_provider_type_env_override(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("EMBEDDING_PROVIDER", "local-bge") + assert resolve_provider_type() == EmbeddingProviderType.LOCAL_BGE + + +def test_resolve_provider_type_scene_override(monkeypatch: pytest.MonkeyPatch) -> None: + """configs/llm_models.yaml 的 scenes.embedding.provider 优先于 .env。""" + monkeypatch.delenv("EMBEDDING_PROVIDER", raising=False) + monkeypatch.setattr( + "embedding.factory.load_scene_config", + lambda scene: {"provider": "local-bge"} if scene == "embedding" else {}, + ) + assert resolve_provider_type() == EmbeddingProviderType.LOCAL_BGE + + +def test_resolve_provider_type_explicit_arg_beats_scene( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """显式参数优先级最高,覆盖 YAML 场景。""" + monkeypatch.setattr( + "embedding.factory.load_scene_config", + lambda scene: {"provider": "local-bge"} if scene == "embedding" else {}, + ) + assert resolve_provider_type("dashscope") == EmbeddingProviderType.DASHSCOPE + + +def test_resolve_provider_type_unknown_raises() -> None: + with pytest.raises(EmbeddingError): + resolve_provider_type("anthropic-emb") + + +# --------------------------------------------------------------------------- # +# DashScope 同步/异步(mock 网络) +# --------------------------------------------------------------------------- # + +def test_dashscope_sync_embed_batch(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeEmbeddingProvider(model="text-embedding-v3") + fake = MagicMock() + fake.embeddings.create = MagicMock( + side_effect=lambda model, input: _embedding_response([[0.1] * 1024] * len(input)) + ) + provider._client = fake + out = provider.embed_batch(["a", "b", "c"]) + assert len(out) == 3 + assert all(len(v) == 1024 for v in out) + + +def test_dashscope_sync_chunks_when_over_batch_limit(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeEmbeddingProvider() + fake = MagicMock() + fake.embeddings.create = MagicMock( + side_effect=lambda model, input: _embedding_response([[0.0] * 1024] * len(input)) + ) + provider._client = fake + texts = [f"t{i}" for i in range(25)] # > 10 -> 应分 3 批 (10+10+5) + provider.embed_batch(texts) + assert fake.embeddings.create.call_count == 3 + + +def test_dashscope_sync_retries_then_succeeds(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeEmbeddingProvider(max_attempts=3) + fake = MagicMock() + fake.embeddings.create = MagicMock( + side_effect=[ + RuntimeError("rate-limit"), + _embedding_response([[0.1] * 1024]), + ] + ) + provider._client = fake + out = provider.embed_batch(["x"]) + assert len(out) == 1 + assert fake.embeddings.create.call_count == 2 + + +def test_dashscope_sync_gives_up(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeEmbeddingProvider(max_attempts=2) + fake = MagicMock() + fake.embeddings.create = MagicMock(side_effect=RuntimeError("net")) + provider._client = fake + with pytest.raises(EmbeddingError) as exc: + provider.embed_batch(["x"]) + assert exc.value.attempts == 2 + + +def test_dashscope_missing_api_key_raises(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("DASHSCOPE_API_KEY", raising=False) + with pytest.raises(EmbeddingError): + DashScopeEmbeddingProvider() + + +@pytest.mark.asyncio +async def test_dashscope_async_embed_batch(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeAsyncEmbeddingProvider() + fake = MagicMock() + fake.embeddings.create = AsyncMock( + side_effect=lambda model, input: _embedding_response([[0.1] * 1024] * len(input)) + ) + provider._client = fake + out = await provider.embed_batch(["a", "b"]) + assert len(out) == 2 + + +@pytest.mark.asyncio +async def test_dashscope_async_chunks(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeAsyncEmbeddingProvider() + fake = MagicMock() + fake.embeddings.create = AsyncMock( + side_effect=lambda model, input: _embedding_response([[0.0] * 1024] * len(input)) + ) + provider._client = fake + texts = [f"t{i}" for i in range(15)] # 2 批 (10+5) + await provider.embed_batch(texts) + assert fake.embeddings.create.await_count == 2 + + +@pytest.mark.asyncio +async def test_dashscope_async_retries(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + provider = DashScopeAsyncEmbeddingProvider(max_attempts=2) + fake = MagicMock() + fake.embeddings.create = AsyncMock( + side_effect=[RuntimeError("transient"), _embedding_response([[0.0] * 1024])] + ) + provider._client = fake + out = await provider.embed_batch(["a"]) + assert len(out) == 1 + + +# --------------------------------------------------------------------------- # +# factory +# --------------------------------------------------------------------------- # + +def test_make_sync_provider_dashscope(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + p = make_sync_provider("dashscope") + assert isinstance(p, EmbeddingProvider) + assert p.name == "dashscope" + assert p.dim == 1024 + + +def test_make_async_provider_dashscope(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test") + p = make_async_provider("qwen") + assert isinstance(p, AsyncEmbeddingProvider) + assert p.name == "dashscope" + + +def test_make_sync_provider_local_without_st_raises(monkeypatch: pytest.MonkeyPatch) -> None: + """无 sentence-transformers 时,本地 provider 应给出友好错误。""" + import sys + # 模拟 sentence_transformers 缺失 + monkeypatch.setitem(sys.modules, "sentence_transformers", None) + with pytest.raises(EmbeddingError) as exc: + make_sync_provider("local") + assert "sentence-transformers" in str(exc.value) + + +# --------------------------------------------------------------------------- # +# EmbeddingResult 模型 + 序列化 +# --------------------------------------------------------------------------- # + +def test_embedding_result_serializes(tmp_path: Path) -> None: + r = EmbeddingResult( + url_hash="abc", + source_id="cls", + title="t", + text="text", + vector=[0.1, 0.2, 0.3], + dim=3, + provider="dashscope", + model="text-embedding-v3", + ) + p = tmp_path / "r.json" + p.write_text(r.model_dump_json(), encoding="utf-8") + obj = json.loads(p.read_text(encoding="utf-8")) + assert obj["dim"] == 3 + assert obj["vector"] == [0.1, 0.2, 0.3] + + +def test_embedding_result_short_summary() -> None: + r = EmbeddingResult( + url_hash="abc", + source_id="cls", + title="宁德时代签约", + text="x", + vector=[0.0] * 4, + dim=4, + provider="dashscope", + model="text-embedding-v3", + ) + s = r.short_summary() + assert "cls" in s and "dim=4" in s and "dashscope" in s + + +# --------------------------------------------------------------------------- # +# 集成式: compose_text + 假异步 provider +# --------------------------------------------------------------------------- # + +class _FakeAsyncProvider(AsyncEmbeddingProvider): + name = "fake" + model = "fake-1" + dim = 8 + + async def embed_batch(self, texts: list[str]) -> list[list[float]]: + return [[float(len(t))] * self.dim for t in texts] + + +@pytest.mark.asyncio +async def test_fake_async_provider_round_trip() -> None: + art = _article() + text = compose_text(art) + async with _FakeAsyncProvider() as p: + v = (await p.embed_batch([text]))[0] + assert len(v) == 8 + assert v[0] == float(len(text)) + + +@pytest.mark.asyncio +async def test_fake_async_provider_concurrent_batches() -> None: + p = _FakeAsyncProvider() + res = await asyncio.gather( + p.embed_batch(["a", "bb"]), + p.embed_batch(["ccc"]), + ) + assert res[0][0][0] == 1.0 + assert res[0][1][0] == 2.0 + assert res[1][0][0] == 3.0 diff --git a/tests/test_extractor.py b/tests/test_extractor.py new file mode 100644 index 0000000..598b270 --- /dev/null +++ b/tests/test_extractor.py @@ -0,0 +1,481 @@ +"""M2 正文提取模块单元测试。""" + +from __future__ import annotations + +import json +from datetime import datetime +from pathlib import Path + +import pytest + +from extractor import Article, ExtractError, extract_article +from extractor.parser import ( + _clean_content, + _count_chinese, + _fallback_time_from_html, + _is_boilerplate, + _is_reasonable_time, + _normalize_time, + _refine_title, + _resolve_publish_time, + _url_hash, +) + +# --------------------------------------------------------------------------- # +# 构造测试 HTML +# --------------------------------------------------------------------------- # + +SAMPLE_HTML = """ + +宁德时代发布麒麟电池 - 财联社 + +
    +
    广告:抢购618
    + +
    +

    宁德时代发布新一代麒麟电池 能量密度达 255Wh/kg

    +
    + 2026-06-15 14:30:00 + 记者 张三 +
    +

    财联社6月15日电,宁德时代今日正式发布了新一代麒麟电池产品,能量密度达到 255 Wh/kg, +显著优于上一代产品的 213 Wh/kg。该产品定位高端电动车市场。

    +

    据公司公告,该电池将于2026年第三季度量产,首批应用于多款新能源汽车。 +公司股价应声上涨5.2%,创年内新高。

    +

    分析师认为,这将进一步巩固宁德时代在动力电池领域的全球领先地位, +预计2026年公司动力电池出货量将同比增长30%以上。

    +

    同时,公司还披露了海外建厂计划,德国与匈牙利工厂将于2027年投产。

    +
    + + + +
    +

    评论 (88)

    +

    用户A: 太牛了!

    +

    用户B: 行业要变天

    +
    + +
    版权所有 财联社
    + + +""" + + +SHORT_HTML = """ +

    很短

    太短

    +""" + + +# --------------------------------------------------------------------------- # +# 主提取流程 +# --------------------------------------------------------------------------- # + +def test_extract_basic() -> None: + article = extract_article( + html=SAMPLE_HTML, + source_id="cls", + url="https://www.cls.cn/detail/123456", + ) + assert isinstance(article, Article) + assert article.source_id == "cls" + assert article.source_name == "财联社" + assert "宁德时代" in article.title + assert "麒麟电池" in article.title + assert article.url_hash == _url_hash("https://www.cls.cn/detail/123456") + + # 关键正文短语必须保留 + assert "255 Wh/kg" in article.content or "255Wh/kg" in article.content + assert "海外建厂" in article.content + + # 时间被解析 + assert article.publish_time is not None + assert article.publish_time.year == 2026 + assert article.publish_time.month == 6 + assert article.publish_time.day == 15 + + # 字数 + assert article.word_count > 30 + + +def test_extract_h1_preferred_over_short_title() -> None: + """ 仅有站名时应优先 H1。""" + html = """ + <html><head><title>财联社 +

    这是一个明显更长更详细的真实文章标题

    +
    +

    正文段落一,正文段落一,正文段落一,正文段落一,正文段落一。

    +

    正文段落二,正文段落二,正文段落二,正文段落二,正文段落二。

    +

    正文段落三,正文段落三,正文段落三,正文段落三,正文段落三。

    +
    + """ + article = extract_article(html, "cls", "https://www.cls.cn/detail/1") + assert article.title == "这是一个明显更长更详细的真实文章标题" + + +def test_extract_too_short_raises() -> None: + with pytest.raises(ExtractError): + extract_article(SHORT_HTML, "cls", "https://www.cls.cn/detail/2") + + +def test_extract_empty_html_raises() -> None: + with pytest.raises(ExtractError): + extract_article("", "cls", "https://www.cls.cn/detail/3") + with pytest.raises(ExtractError): + extract_article(" \n\n ", "cls", "https://www.cls.cn/detail/4") + + +def test_extract_unknown_source_has_no_source_name() -> None: + article = extract_article(SAMPLE_HTML, "unknown_src", "https://example.com/a/1") + assert article.source_name is None + assert article.source_id == "unknown_src" + + +# --------------------------------------------------------------------------- # +# 模板兜底检测(M2.2) +# --------------------------------------------------------------------------- # + +def test_is_boilerplate_eastmoney_legal_disclaimer() -> None: + text = ( + "郑重声明: 1.根据《证券法》规定,禁止编造、传播虚假信息或者误导性信息," + "扰乱证券市场;2.用户在本社区发表的所有资料、言论等仅代表个人观点。" + ) + is_bp, reason = _is_boilerplate(text) + assert is_bp + assert reason is not None and "郑重声明" in reason + + +def test_is_boilerplate_yicai_ad_copyright() -> None: + text = ( + "第一财经广告合作,请点击这里。此内容为第一财经原创,著作权归第一财经所有。" + "未经第一财经书面授权,不得以任何方式加以使用。" + ) + is_bp, reason = _is_boilerplate(text) + assert is_bp + + +def test_is_boilerplate_sina_embedded_post() -> None: + text = ( + "北京红竹 今天 16:02:49 本质上就是一句话:资金开始从核心抱团,进入扩散阶段。" + "银行这边今天已经明显走弱。银行和科技之间还是跷跷板关系。" * 2 + ) + is_bp, _ = _is_boilerplate(text) + assert is_bp + + +def test_is_boilerplate_long_real_article_passes() -> None: + """真实长篇文章中即使提到"郑重声明"等词,长度 > 800 不应误判。""" + legitimate = ( + "公司发布郑重声明,回应近期市场关注的多项议题。根据《证券法》及相关法律法规要求," + "公司将严格履行信息披露义务。" * 30 # ~1200 字 + ) + assert len(legitimate) > 800 + is_bp, _ = _is_boilerplate(legitimate) + assert not is_bp + + +def test_is_boilerplate_empty_returns_false() -> None: + assert _is_boilerplate("") == (False, None) + assert _is_boilerplate("普通正文,没有任何模板特征,长度也合理。" * 5) == (False, None) + + +def test_extract_article_raises_on_boilerplate() -> None: + """模拟一篇 GNE 兜底失败的页面,extract_article 应抛 ExtractError。""" + html = """ + 东方财富 +

    是否有中国船只通过霍尔木兹海峡?外交部回应

    +
    + 郑重声明: 1.根据《证券法》规定,禁止编造、传播虚假信息或者误导性信息, + 扰乱证券市场;2.用户在本社区发表的所有资料、言论等仅代表个人观点,与本网站立场无关。 + 《东方财富社区管理规定》 +
    + + """ + with pytest.raises(ExtractError) as exc: + extract_article( + html, + source_id="eastmoney", + url="https://finance.eastmoney.com/a/123.html", + ) + assert "模板" in str(exc.value) + + +# --------------------------------------------------------------------------- # +# title 校正 +# --------------------------------------------------------------------------- # + +def test_refine_title_uses_h1_when_gne_is_substring() -> None: + html = "

    完整真实标题

    " + assert _refine_title(html, "完整真实") == "完整真实标题" + + +def test_refine_title_uses_h1_when_gne_has_site_suffix() -> None: + html = "

    真实标题

    " + assert _refine_title(html, "真实标题 - 财联社") == "真实标题" + + +def test_refine_title_falls_back_to_gne_when_no_h1() -> None: + html = "

    x

    " + assert _refine_title(html, "GNE 标题") == "GNE 标题" + + +def test_refine_title_returns_empty_when_both_missing() -> None: + assert _refine_title("", "") == "" + + +# --------------------------------------------------------------------------- # +# content 清理 +# --------------------------------------------------------------------------- # + +def test_clean_content_strips_repeated_header() -> None: + raw = "宁德时代发布新一代麒麟电池\n2026-06-15 14:30:00\n记者 张三\n\n正文第一段。\n\n\n\n正文第二段。" + cleaned = _clean_content( + raw, + title="宁德时代发布新一代麒麟电池", + author="记者 张三", + time_raw="2026-06-15 14:30:00", + ) + assert cleaned.startswith("正文第一段") + assert "正文第二段" in cleaned + # 连续 4 个 \n 应被压缩为 \n\n + assert "\n\n\n" not in cleaned + + +def test_clean_content_keeps_body_when_no_header_match() -> None: + raw = "段落一\n段落二\n段落三" + cleaned = _clean_content(raw, title="完全不同的标题", author=None, time_raw=None) + assert cleaned == "段落一\n段落二\n段落三" + + +def test_clean_content_handles_empty() -> None: + assert _clean_content("", "", None, None) == "" + + +# --------------------------------------------------------------------------- # +# 时间标准化 +# --------------------------------------------------------------------------- # + +def test_normalize_time_iso() -> None: + dt = _normalize_time("2026-06-15 14:30:00") + assert dt is not None and dt == datetime(2026, 6, 15, 14, 30, 0) + + +def test_normalize_time_chinese_format() -> None: + dt = _normalize_time("2026年6月15日 14时30分") + assert dt is not None + assert dt.year == 2026 and dt.month == 6 and dt.day == 15 + assert dt.hour == 14 and dt.minute == 30 + + +def test_normalize_time_chinese_seconds() -> None: + dt = _normalize_time("2026年1月1日 9时5分3秒") + assert dt is not None and dt.second == 3 + + +def test_normalize_time_invalid_returns_none() -> None: + assert _normalize_time("不是时间") is None + assert _normalize_time("") is None + assert _normalize_time(None) is None + + +# --------------------------------------------------------------------------- # +# 时间合理性 + HTML 兜底 +# --------------------------------------------------------------------------- # + +def test_is_reasonable_time_within_range() -> None: + ref = datetime(2026, 6, 16, 12, 0, 0) + assert _is_reasonable_time(datetime(2026, 6, 15, 8, 0), ref) + assert _is_reasonable_time(datetime(2025, 12, 1, 0, 0), ref) + # 同一天即将到来的时间 + assert _is_reasonable_time(datetime(2026, 6, 16, 23, 0), ref) + + +def test_is_reasonable_time_rejects_far_future() -> None: + ref = datetime(2026, 6, 16, 12, 0) + assert not _is_reasonable_time(datetime(2026, 6, 18, 0, 0), ref) + + +def test_is_reasonable_time_rejects_far_past() -> None: + ref = datetime(2026, 6, 16, 12, 0) + # 距 ref 超过 365 天 -> 不合理(模拟 GNE 抓到的 2019 年页脚时间) + assert not _is_reasonable_time(datetime(2019, 1, 16, 10, 40), ref) + + +def test_is_reasonable_time_strips_tzinfo() -> None: + """带时区的时间也能与 naive ref 比较。""" + from datetime import UTC + + aware = datetime(2026, 6, 16, 12, 0, tzinfo=UTC) + assert _is_reasonable_time(aware, datetime(2026, 6, 16, 12, 0)) + + +def test_is_reasonable_time_handles_none() -> None: + assert _is_reasonable_time(None) is False + + +def test_fallback_time_from_html_finds_eastmoney_pattern() -> None: + """模拟 eastmoney 真实结构,兜底应能命中 .infos 内的中文日期。""" + html = """ +
    +
    2026年06月16日 17:54
    +
    来源:发改委网站
    +
    + """ + dt, raw = _fallback_time_from_html(html) + assert dt is not None + assert dt.year == 2026 and dt.month == 6 and dt.day == 16 + assert dt.hour == 17 and dt.minute == 54 + assert raw is not None and "2026" in raw + + +def test_fallback_time_skips_unreasonable_dates() -> None: + """页脚备案/版权时间(2019-01-16)出现在前,真实发布时间在后,应跳过前者。""" + html = """ + +
    +
    备案号 京ICP-XXX 备案日期: 2019-01-16
    +
    +
    +
    2026年06月16日 17:54
    +
    + + """ + dt, raw = _fallback_time_from_html(html) + assert dt is not None + assert dt.year == 2026 + assert "2026" in (raw or "") + + +def test_fallback_time_returns_none_when_no_match() -> None: + dt, raw = _fallback_time_from_html("没有日期") + assert dt is None and raw is None + + +def test_resolve_publish_time_uses_gne_when_reasonable() -> None: + today = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + dt, raw = _resolve_publish_time("", today) + assert dt is not None + assert raw == today + + +def test_resolve_publish_time_falls_back_when_gne_unreasonable() -> None: + """模拟 eastmoney 场景:GNE 给出 2019-01-16,HTML 中含真实时间。""" + html = '
    2026年06月16日 17:54
    ' + dt, raw = _resolve_publish_time(html, "2019-01-16 10:40:21") + assert dt is not None and dt.year == 2026 and dt.month == 6 and dt.day == 16 + # raw 应反映兜底来源,而非原始 GNE 字符串 + assert "2026" in (raw or "") + + +def test_resolve_publish_time_keeps_gne_raw_when_all_fail() -> None: + """GNE 不合理且 HTML 也无可用时间 -> publish_time None,raw 保留 GNE。""" + dt, raw = _resolve_publish_time("无", "2010-01-01 00:00:00") + assert dt is None + assert raw == "2010-01-01 00:00:00" + + +# --------------------------------------------------------------------------- # +# 端到端:真实 eastmoney 结构应解析出 2026 时间 +# --------------------------------------------------------------------------- # + +def test_extract_article_uses_html_fallback_for_eastmoney_like() -> None: + """模拟真实 eastmoney 文章页:GNE 拿到页脚错误时间,extractor 应兜底。""" + html = """ + 事关六张网建设 +
    备案信息发布时间 2019-01-16 10:40:21
    +
    +
    事关“六张网”建设 国家发展改革委召开重要座谈会
    +
    +
    +
    2026年06月16日 17:54
    +
    来源:发改委网站
    +
    +
    +
    +
    +

    事关“六张网”建设 国家发展改革委召开重要座谈会

    +

    近日,国家发展改革委召开座谈会,围绕加快推进“六张网”建设的具体举措进行专题研讨。

    +

    会议指出,“六张网”建设关系国家长远发展,要从体制机制、关键技术、重点项目三方面协同推进。

    +

    与会专家就资金保障、跨部门协调、技术标准统一等议题展开了深入交流。

    +
    + + """ + article = extract_article( + html, + source_id="eastmoney", + url="https://finance.eastmoney.com/a/123456789.html", + ) + assert article.publish_time is not None + assert article.publish_time.year == 2026 + assert article.publish_time.month == 6 + assert article.publish_time.day == 16 + assert "2026" in (article.publish_time_raw or "") + + +# --------------------------------------------------------------------------- # +# 工具 +# --------------------------------------------------------------------------- # + +def test_url_hash_stable_and_short() -> None: + h1 = _url_hash("https://example.com/a") + h2 = _url_hash("https://example.com/a") + h3 = _url_hash("https://example.com/b") + assert h1 == h2 != h3 + assert len(h1) == 16 + + +def test_count_chinese() -> None: + assert _count_chinese("hello 你好 world") == 2 + assert _count_chinese("ABC123") == 0 + assert _count_chinese("中文测试") == 4 + + +def test_article_short_summary() -> None: + a = Article( + source_id="cls", + url="https://x/1", + url_hash="abc123", + title="测试标题", + content="一段正文 " * 20, + publish_time=datetime(2026, 6, 15, 14, 30), + word_count=20, + ) + s = a.short_summary() + assert "cls" in s + assert "2026-06-15" in s + assert "测试标题" in s + + +# --------------------------------------------------------------------------- # +# Article 模型字段约束 +# --------------------------------------------------------------------------- # + +def test_article_requires_min_length_title_content() -> None: + with pytest.raises(Exception): # noqa: B017 - pydantic ValidationError + Article( + source_id="cls", + url="https://x/1", + url_hash="h", + title="", + content="", + ) + + +def test_article_serializes_to_json(tmp_path: Path) -> None: + a = Article( + source_id="cls", + url="https://x/1", + url_hash="abc", + title="标题", + content="一二三四五六七八九十一二三四五六七八九十一二三四五六七八九十", + publish_time=datetime(2026, 6, 15, 14, 30), + word_count=30, + ) + p = tmp_path / "a.json" + p.write_text(a.model_dump_json(), encoding="utf-8") + obj = json.loads(p.read_text(encoding="utf-8")) + assert obj["source_id"] == "cls" + assert obj["publish_time"].startswith("2026-06-15") diff --git a/tests/test_incremental.py b/tests/test_incremental.py new file mode 100644 index 0000000..38833cd --- /dev/null +++ b/tests/test_incremental.py @@ -0,0 +1,268 @@ +"""增量处理与 pipeline 断点续跑测试。 + +覆盖: + - M2/M4/M5 脚本的「产物存在即跳过」过滤逻辑 + - scheduler.pipeline 断点状态记录与 --resume 续跑逻辑 +""" + +from __future__ import annotations + +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +# --------------------------------------------------------------------------- # +# M4 / M5: 产物存在即跳过 +# --------------------------------------------------------------------------- # + +def test_llm_filter_existing_skips_done(tmp_path: Path) -> None: + """M4:输出目录已有 {url_hash}.json 的输入被过滤,不重复调用 LLM API。""" + from scripts.run_event_extraction import _filter_existing + + out_dir = tmp_path / "out" + out_dir.mkdir() + # 已处理 + (out_dir / "aaa.json").write_text("{}", encoding="utf-8") + (out_dir / "ccc.json").write_text("{}", encoding="utf-8") + files = [ + tmp_path / "in" / "aaa.json", # 已处理 → 跳过 + tmp_path / "in" / "bbb.json", # 未处理 → 待处理 + tmp_path / "in" / "ccc.json", # 已处理 → 跳过 + ] + pending, skipped = _filter_existing(files, out_dir) + assert skipped == 2 + assert [p.stem for p in pending] == ["bbb"] + + +def test_llm_filter_existing_force_keeps_all(tmp_path: Path) -> None: + """M4:--force 时不做过滤(全量重抽由调用方控制)。""" + from scripts.run_event_extraction import _filter_existing + + out_dir = tmp_path / "out" + out_dir.mkdir() + (out_dir / "aaa.json").write_text("{}", encoding="utf-8") + files = [tmp_path / "in" / "aaa.json"] + # _filter_existing 本身不含 force 逻辑,验证在 force 下不会被调用: + # 直接验证「已存在也被返回」需由上层跳过调用,这里仅确认过滤函数行为。 + pending, skipped = _filter_existing(files, out_dir) + assert skipped == 1 + assert pending == [] + + +def test_embedding_filter_existing_skips_done(tmp_path: Path) -> None: + """M5:输出目录已有 {url_hash}.json 的输入被过滤,不重复调用 embed API。""" + from scripts.run_embedding import _filter_existing + + out_dir = tmp_path / "emb" + out_dir.mkdir() + (out_dir / "h1.json").write_text("{}", encoding="utf-8") + files = [ + (tmp_path / "in" / "h1.json", "event"), # 已处理 → 跳过 + (tmp_path / "in" / "h2.json", "event"), # 未处理 → 待处理 + ] + pending, skipped = _filter_existing(files, out_dir) + assert skipped == 1 + assert [p.stem for p, _ in pending] == ["h2"] + + +# --------------------------------------------------------------------------- # +# M2: 已提取文章跳过 +# --------------------------------------------------------------------------- # + +def test_extractor_process_source_day_skips_existing(tmp_path: Path) -> None: + """M2:输出目录已有产物的记录被跳过提取,且 index 回补完整。""" + from scripts.run_extractor import _process_source_day + + raw_dir = tmp_path / "raw" / "cls" / "20260616" + raw_dir.mkdir(parents=True) + # 两条 raw 记录(url_hash 与产物文件名一致) + rec1 = {"source_id": "cls", "url": "https://a/1", "url_hash": "aaa1111111111111", + "stage": "article", "success": True, "html_file": "aaa1111111111111.html"} + rec2 = {"source_id": "cls", "url": "https://b/2", "url_hash": "bbb2222222222222", + "stage": "article", "success": True, "html_file": "bbb2222222222222.html"} + with (raw_dir / "index.jsonl").open("a", encoding="utf-8") as f: + f.write(json.dumps(rec1, ensure_ascii=False) + "\n") + f.write(json.dumps(rec2, ensure_ascii=False) + "\n") + + out_dir = tmp_path / "proc" / "cls" / "20260616" + out_dir.mkdir(parents=True) + # 预置一条已有产物(视为已提取) + article = { + "source_id": "cls", "url": "https://a/1", "url_hash": "aaa1111111111111", + "title": "已有", "content": "内容", "word_count": 2, + } + (out_dir / "aaa1111111111111.json").write_text( + json.dumps(article, ensure_ascii=False), encoding="utf-8" + ) + + # 另一条无 html 文件 → 提取失败(但不影响跳过逻辑断言) + succ, total, skipped = _process_source_day( + "cls", "20260616", tmp_path / "raw", tmp_path / "proc" + ) + assert total == 2 + assert skipped == 1 # 已有产物被跳过 + assert succ == 0 # 另一条因 html 缺失提取失败 + # index 回补了被跳过条目的行 + idx = (out_dir / "index.jsonl").read_text(encoding="utf-8").strip() + assert "aaa1111111111111" in idx + + +def test_extractor_process_source_day_force_rebuilds(tmp_path: Path) -> None: + """M2:--force 时不做跳过,并重建 index。""" + from scripts.run_extractor import _process_source_day + + raw_dir = tmp_path / "raw" / "cls" / "20260616" + raw_dir.mkdir(parents=True) + rec = {"source_id": "cls", "url": "https://a/1", "url_hash": "aaa1111111111111", + "stage": "article", "success": True, "html_file": "aaa1111111111111.html"} + with (raw_dir / "index.jsonl").open("a", encoding="utf-8") as f: + f.write(json.dumps(rec, ensure_ascii=False) + "\n") + + out_dir = tmp_path / "proc" / "cls" / "20260616" + out_dir.mkdir(parents=True) + (out_dir / "aaa1111111111111.json").write_text("{}", encoding="utf-8") + (out_dir / "index.jsonl").write_text("旧内容", encoding="utf-8") + + succ, total, skipped = _process_source_day( + "cls", "20260616", tmp_path / "raw", tmp_path / "proc", force=True + ) + assert skipped == 0 + # force 模式重建 index(旧内容被清掉;此处无 html 提取失败,index 为空或不存在) + idx = out_dir / "index.jsonl" + assert not idx.exists() or idx.read_text(encoding="utf-8") == "" + + +# --------------------------------------------------------------------------- # +# pipeline 断点状态与 --resume +# --------------------------------------------------------------------------- # + +@pytest.fixture +def fake_subprocess(monkeypatch: pytest.MonkeyPatch): + """mock subprocess.run,按步骤名返回 returncode,并记录调用顺序。""" + from scheduler import pipeline + + calls: list[str] = [] + + def _step_name(cmd: list[str]) -> str: + """从命令中提取脚本名,如 scripts.run_extractor → run_extractor。""" + return next(c.split(".")[-1] for c in cmd if "scripts.run_" in c) + + def _fake_run(cmd, timeout=None): # noqa: ARG001 + calls.append(_step_name(cmd)) + return SimpleNamespace(returncode=0) + + monkeypatch.setattr(pipeline.subprocess, "run", _fake_run) + return calls + + +def _run_with_steps(monkeypatch: pytest.MonkeyPatch, failures: set[str]): + """构造 run_step:指定步骤(如 'extractor')返回失败。""" + from scheduler import pipeline + + def _fake_run(cmd, timeout=None): # noqa: ARG001 + name = next(c.split(".")[-1] for c in cmd if "scripts.run_" in c) + name = name.replace("run_", "") # run_extractor → extractor + return SimpleNamespace(returncode=1 if name in failures else 0) + + monkeypatch.setattr(pipeline.subprocess, "run", _fake_run) + + +def test_pipeline_records_state(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """全量运行后状态文件按日期记录每个步骤的 ok/failed。""" + from scheduler import pipeline + + _run_with_steps(monkeypatch, failures={"extractor"}) + state_path = tmp_path / "state.json" + steps = ["extractor", "dedup", "llm"] + pipeline.run_pipeline("20260616", steps=steps, state_path=state_path) + + state = pipeline._load_pipeline_state(state_path) + day = state["20260616"] + assert day["extractor"]["status"] == "failed" + assert day["dedup"]["status"] == "ok" + assert day["llm"]["status"] == "ok" + # dedup 返回 1 被特判为成功,故用 extractor 制造失败 + assert day["extractor"]["exit_code"] == 1 + + +def test_pipeline_resume_skips_success_prefix( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + """resume 跳过连续成功步骤,从失败步骤继续。""" + from scheduler import pipeline + + state_path = tmp_path / "state.json" + # 预置状态:extractor 失败,dedup/llm 成功(模拟上次运行) + state = {"20260616": { + "extractor": {"status": "failed", "exit_code": 1}, + "dedup": {"status": "ok", "exit_code": 0}, + "llm": {"status": "ok", "exit_code": 0}, + }} + pipeline._save_pipeline_state(state, state_path) + + calls: list[str] = [] + + def _fake_run(cmd, timeout=None): # noqa: ARG001 + name = next(c.split(".")[-1] for c in cmd if "scripts.run_" in c) + calls.append(name) + return SimpleNamespace(returncode=0) + + monkeypatch.setattr(pipeline.subprocess, "run", _fake_run) + steps = ["extractor", "dedup", "llm"] + result = pipeline.run_pipeline( + "20260616", steps=steps, resume=True, state_path=state_path + ) + # 从 extractor 开始重跑全部(extractor 之后的 dedup/llm 需重跑以覆盖降级数据) + assert calls == ["run_extractor", "run_dedup", "run_event_extraction"] + assert all(s.success for s in result.steps) + + +def test_pipeline_resume_all_done_noop(tmp_path: Path) -> None: + """resume 且所有步骤均已成功时,不执行任何步骤。""" + from scheduler import pipeline + + state_path = tmp_path / "state.json" + state = {"20260616": { + "extractor": {"status": "ok", "exit_code": 0}, + "dedup": {"status": "ok", "exit_code": 0}, + "llm": {"status": "ok", "exit_code": 0}, + }} + pipeline._save_pipeline_state(state, state_path) + + steps = ["extractor", "dedup", "llm"] + result = pipeline.run_pipeline( + "20260616", steps=steps, resume=True, state_path=state_path + ) + assert result.steps == [] + assert result.all_success # 空步骤视为成功 + + +def test_pipeline_resume_missing_step_starts_from_first_missing( + tmp_path: Path, +) -> None: + """resume:部分步骤无历史记录时,从首个缺失步骤开始。""" + from scheduler import pipeline + + state_path = tmp_path / "state.json" + state = {"20260616": { + "extractor": {"status": "ok", "exit_code": 0}, + }} + pipeline._save_pipeline_state(state, state_path) + + idx = pipeline._resume_start_index( + ["extractor", "dedup", "llm"], "20260616", + pipeline._load_pipeline_state(state_path), + ) + assert idx == 1 # dedup 缺失 → 从它开始 + + +def test_once_rejects_resume_with_steps(monkeypatch: pytest.MonkeyPatch) -> None: + """--resume 与 --steps 同时使用时报错。""" + import scripts.run_scheduler as rs + + args = SimpleNamespace(steps="crawler,extractor", resume=True, + date="20260616") + rc = rs._once(args) + assert rc == 2 diff --git a/tests/test_llm.py b/tests/test_llm.py new file mode 100644 index 0000000..dd90824 --- /dev/null +++ b/tests/test_llm.py @@ -0,0 +1,500 @@ +"""M4 LLM 投资事件抽取测试。 + +不依赖真实 LLM API,所有调用通过 mock 注入响应。 +""" + +from __future__ import annotations + +import asyncio +import json +from datetime import datetime +from pathlib import Path +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from extractor import Article +from llm import ( + EVENT_TYPES, + EventExtraction, + ExtractedEvent, + LLMCallError, + PromptTemplate, + Sentiment, + extract_event, + extract_event_async, + load_llm_config, + parse_event_json, +) +from llm.client import LLMConfig +from llm.extractor import _extract_json_object + +# --------------------------------------------------------------------------- # +# fixtures +# --------------------------------------------------------------------------- # + +def _article( + *, + title: str = "宁德时代签订 100GWh 长期供货协议", + content: str = "宁德时代(300750)与某车企签 5 年 100GWh 协议,涉及金额超 1500 亿。" * 3, + publish_time: datetime | None = datetime(2026, 6, 16, 10, 0), +) -> Article: + return Article( + source_id="cls", + url="https://www.cls.cn/detail/1", + url_hash="abc1234567890000", + title=title, + content=content, + publish_time=publish_time, + word_count=len(content), + ) + + +@pytest.fixture +def fake_config() -> LLMConfig: + return LLMConfig( + provider="deepseek", + model="deepseek-chat", + api_key="sk-fake", + base_url="https://api.deepseek.com", + ) + + +def _mock_completion(content: str, prompt_tokens: int = 100, completion_tokens: int = 50) -> MagicMock: + """构造与 openai SDK 返回兼容的 mock 对象。""" + msg = MagicMock() + msg.content = content + choice = MagicMock() + choice.message = msg + usage = MagicMock() + usage.prompt_tokens = prompt_tokens + usage.completion_tokens = completion_tokens + resp = MagicMock() + resp.choices = [choice] + resp.usage = usage + return resp + + +# --------------------------------------------------------------------------- # +# EventExtraction 模型校验 +# --------------------------------------------------------------------------- # + +def test_event_extraction_minimal_valid() -> None: + e = EventExtraction(sentiment="positive", importance=4, event_type="重大合同") + assert e.sentiment == Sentiment.POSITIVE + assert e.importance == 4 + + +def test_event_extraction_normalizes_stock_codes() -> None: + e = EventExtraction( + stock_codes=["300750.sz", " 300750.SZ ", "abc", "12345", "600519"], + sentiment="positive", importance=3, event_type="其他", + ) + # 大小写 / 空白被规范;非法被过滤;去重 + assert e.stock_codes == ["300750.SZ", "600519"] + + +def test_event_extraction_filters_empty_lists() -> None: + e = EventExtraction( + stock_codes=[], company_names=["", " ", "宁德时代", "宁德时代"], + industries=[], + sentiment="neutral", importance=1, event_type="其他", + ) + assert e.company_names == ["宁德时代"] + assert e.stock_codes == [] + + +def test_event_extraction_rejects_importance_out_of_range() -> None: + with pytest.raises(Exception): # noqa: B017 + EventExtraction(sentiment="positive", importance=0, event_type="其他") + with pytest.raises(Exception): # noqa: B017 + EventExtraction(sentiment="positive", importance=6, event_type="其他") + + +def test_event_extraction_normalizes_blank_event_type() -> None: + e = EventExtraction(sentiment="neutral", importance=1, event_type=" ") + assert e.event_type == "其他" + + +def test_event_types_constant_includes_common() -> None: + for must in ["业绩预告", "合作签约", "监管处罚", "其他"]: + assert must in EVENT_TYPES + + +# --------------------------------------------------------------------------- # +# ExtractedEvent.sources 多源字段 +# --------------------------------------------------------------------------- # + +def test_extracted_event_sources_defaults_to_main_source() -> None: + """未提供 sources 时兜底为 [source_id](兼容旧产物)。""" + ev = ExtractedEvent( + source_id="cls", url="https://x/1", url_hash="h1", title="t", + event=EventExtraction(sentiment="positive", importance=3, event_type="重大合同"), + provider="deepseek", model="m", + ) + assert ev.sources == ["cls"] + + +def test_extracted_event_sources_keeps_main_first_and_dedup() -> None: + """sources 保主源居首、去重保序。""" + ev = ExtractedEvent( + source_id="cls", url="https://x/1", url_hash="h1", title="t", + sources=["sina", "cls", "eastmoney", "sina"], + event=EventExtraction(sentiment="neutral", importance=2, event_type="其他"), + provider="deepseek", model="m", + ) + assert ev.sources == ["cls", "sina", "eastmoney"] + + +# --------------------------------------------------------------------------- # +# JSON 提取与解析 +# --------------------------------------------------------------------------- # + +def test_extract_json_object_strips_fence() -> None: + s = '```json\n{"a": 1}\n```' + assert _extract_json_object(s) == '{"a": 1}' + + +def test_extract_json_object_picks_first_object() -> None: + s = '前置说明\n{"a": 1}\n更多文字' + assert _extract_json_object(s) == '{"a": 1}' + + +def test_extract_json_object_handles_nested() -> None: + s = '{"a": {"b": 2}}' + assert _extract_json_object(s) == '{"a": {"b": 2}}' + + +def test_parse_event_json_ok() -> None: + raw = json.dumps({ + "stock_codes": ["300750.SZ"], + "company_names": ["宁德时代"], + "industries": ["动力电池"], + "sentiment": "positive", + "importance": 5, + "event_type": "重大合同", + "summary": "签订长期供货协议", + }) + e = parse_event_json(raw) + assert e.sentiment == Sentiment.POSITIVE + assert e.stock_codes == ["300750.SZ"] + + +def test_parse_event_json_invalid_json_raises() -> None: + with pytest.raises(LLMCallError): + parse_event_json("not a json") + + +def test_parse_event_json_non_object_raises() -> None: + with pytest.raises(LLMCallError): + parse_event_json('["array"]') + + +def test_parse_event_json_schema_invalid_raises() -> None: + with pytest.raises(LLMCallError): + parse_event_json('{"importance": 99}') # 缺 sentiment + event_type 且 importance 越界 + + +# --------------------------------------------------------------------------- # +# PromptTemplate +# --------------------------------------------------------------------------- # + +def test_prompt_template_renders_placeholders(tmp_path: Path) -> None: + tpl_file = tmp_path / "tpl.md" + tpl_file.write_text( + "标题:{title}\n时间:{publish_time}\n源:{source_name}\n内容:\n{content}\nEND", + encoding="utf-8", + ) + tpl = PromptTemplate(tpl_file) + art = _article() + rendered = tpl.render(art) + assert "标题:" + art.title in rendered + assert "2026-06-16" in rendered + assert "源:财联社" in rendered or "源:cls" in rendered # source_name 默认空,落到 source_id + assert art.content[:30] in rendered + + +def test_prompt_template_truncates_long_content(tmp_path: Path) -> None: + tpl_file = tmp_path / "tpl.md" + tpl_file.write_text("{content}", encoding="utf-8") + tpl = PromptTemplate(tpl_file) + art = _article(content="字" * 20000) + rendered = tpl.render(art) + assert "[正文过长已截断]" in rendered + assert len(rendered) < 20000 + + +def test_prompt_template_default_path_loads() -> None: + """项目内置 prompts/event_extraction.md 必须可加载,作为回归保护。""" + real = Path("prompts/event_extraction.md") + if not real.is_file(): + pytest.skip("prompts/event_extraction.md 未找到") + tpl = PromptTemplate(real) + out = tpl.render(_article()) + assert "{title}" not in out + assert "{content}" not in out + + +# --------------------------------------------------------------------------- # +# extract_event(同步,带重试) +# --------------------------------------------------------------------------- # + +def test_extract_event_succeeds_first_try(fake_config: LLMConfig) -> None: + client = MagicMock() + raw = json.dumps({ + "stock_codes": ["300750.SZ"], + "company_names": ["宁德时代"], + "industries": ["动力电池"], + "sentiment": "positive", + "importance": 5, + "event_type": "重大合同", + "summary": "100GWh 合作", + }) + client.chat.completions.create = MagicMock(return_value=_mock_completion(raw)) + + result = extract_event(client, fake_config, _article()) + assert isinstance(result, ExtractedEvent) + assert result.attempts == 1 + assert result.event.sentiment == Sentiment.POSITIVE + assert result.provider == "deepseek" + assert result.prompt_tokens == 100 + + +def test_extract_event_retries_on_invalid_json(fake_config: LLMConfig) -> None: + """第 1/2 次返回非法 JSON,第 3 次成功。""" + valid = json.dumps({ + "sentiment": "neutral", "importance": 1, "event_type": "其他", + }) + client = MagicMock() + client.chat.completions.create = MagicMock(side_effect=[ + _mock_completion("not a json"), + _mock_completion('{"sentiment":"???"}'), # schema 校验失败 + _mock_completion(valid), + ]) + result = extract_event(client, fake_config, _article(), max_attempts=3) + assert result.attempts == 3 + assert client.chat.completions.create.call_count == 3 + + +def test_extract_event_gives_up_after_max(fake_config: LLMConfig) -> None: + client = MagicMock() + client.chat.completions.create = MagicMock( + return_value=_mock_completion("not a json") + ) + with pytest.raises(LLMCallError) as exc: + extract_event(client, fake_config, _article(), max_attempts=2) + assert exc.value.attempts == 2 + assert client.chat.completions.create.call_count == 2 + + +def test_extract_event_handles_network_exception(fake_config: LLMConfig) -> None: + client = MagicMock() + valid = json.dumps({ + "sentiment": "negative", "importance": 3, "event_type": "监管处罚", + }) + client.chat.completions.create = MagicMock(side_effect=[ + TimeoutError("net hang"), + _mock_completion(valid), + ]) + result = extract_event(client, fake_config, _article(), max_attempts=2) + assert result.attempts == 2 + assert result.event.sentiment == Sentiment.NEGATIVE + + +# --------------------------------------------------------------------------- # +# extract_event_async +# --------------------------------------------------------------------------- # + +@pytest.mark.asyncio +async def test_extract_event_async_succeeds(fake_config: LLMConfig) -> None: + client = MagicMock() + valid = json.dumps({ + "stock_codes": ["600519"], + "company_names": ["贵州茅台"], + "sentiment": "neutral", + "importance": 2, + "event_type": "财报披露", + "summary": "披露半年报", + }) + client.chat.completions.create = AsyncMock(return_value=_mock_completion(valid)) + + sem = asyncio.Semaphore(2) + result = await extract_event_async( + client, fake_config, _article(), semaphore=sem, + ) + assert result.event.stock_codes == ["600519"] + assert result.event.event_type == "财报披露" + + +@pytest.mark.asyncio +async def test_extract_event_async_retries(fake_config: LLMConfig) -> None: + valid = json.dumps({"sentiment": "positive", "importance": 4, "event_type": "其他"}) + client = MagicMock() + client.chat.completions.create = AsyncMock(side_effect=[ + ValueError("transient"), + _mock_completion(valid), + ]) + result = await extract_event_async( + client, fake_config, _article(), max_attempts=2, + ) + assert result.attempts == 2 + + +# --------------------------------------------------------------------------- # +# load_llm_config +# --------------------------------------------------------------------------- # + +def test_load_llm_config_deepseek_from_env(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LLM_PROVIDER", "deepseek") + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test-deepseek") + monkeypatch.setenv("DEEPSEEK_MODEL", "deepseek-v4-flash") + monkeypatch.delenv("LLM_MODEL", raising=False) + cfg = load_llm_config() + assert cfg.provider == "deepseek" + assert cfg.api_key == "sk-test-deepseek" + assert cfg.model == "deepseek-v4-flash" + assert "deepseek" in cfg.base_url + + +def test_load_llm_config_qwen_from_env(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LLM_PROVIDER", "qwen") + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-test-qwen") + monkeypatch.setenv("QWEN_MODEL", "qwen-plus") + monkeypatch.delenv("LLM_MODEL", raising=False) + cfg = load_llm_config() + assert cfg.provider == "qwen" + assert cfg.api_key == "sk-test-qwen" + assert cfg.model == "qwen-plus" + assert "dashscope" in cfg.base_url or "aliyuncs" in cfg.base_url + + +def test_load_llm_config_unknown_provider_raises(monkeypatch: pytest.MonkeyPatch) -> None: + with pytest.raises(ValueError): + load_llm_config(provider="anthropic") + + +def test_load_llm_config_missing_key_raises(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("DEEPSEEK_API_KEY", raising=False) + monkeypatch.setenv("DEEPSEEK_MODEL", "deepseek-v4-flash") + with pytest.raises(ValueError, match="API key"): + load_llm_config(provider="deepseek") + + +def test_load_llm_config_missing_model_raises(monkeypatch: pytest.MonkeyPatch) -> None: + """去掉内置默认模型后:未显式配置模型必须报错(不再回退 deepseek-chat)。""" + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test") + monkeypatch.delenv("DEEPSEEK_MODEL", raising=False) + monkeypatch.delenv("LLM_MODEL", raising=False) + with pytest.raises(ValueError, match="模型"): + load_llm_config(provider="deepseek") + + +# --------------------------------------------------------------------------- # +# load_llm_config —— configs/llm_models.yaml 场景配置 +# --------------------------------------------------------------------------- # + +def _patch_scene(monkeypatch: pytest.MonkeyPatch, cfg: dict) -> None: + """替换场景加载,模拟 configs/llm_models.yaml 中的某场景配置。""" + monkeypatch.setattr( + "llm.client.load_scene_config", + lambda scene: cfg if scene == "daily_report" else {}, + ) + + +def test_load_llm_config_scene_overrides_env(monkeypatch: pytest.MonkeyPatch) -> None: + """YAML 场景配置优先于 .env:provider / model / temperature / timeout / max_attempts。""" + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-env") + monkeypatch.setenv("DEEPSEEK_MODEL", "deepseek-env-model") + monkeypatch.setenv("QWEN_API_KEY", "sk-qwen") + _patch_scene(monkeypatch, { + "provider": "qwen", + "model": "qwen-max", + "temperature": 0.5, + "timeout_sec": 99, + "max_attempts": 5, + }) + cfg = load_llm_config(scene="daily_report") + assert cfg.provider == "qwen" + assert cfg.model == "qwen-max" + assert cfg.api_key == "sk-qwen" + assert cfg.temperature == 0.5 + assert cfg.timeout_sec == 99 + assert cfg.max_attempts == 5 + + +def test_load_llm_config_scene_blank_fields_fall_back_to_env( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """YAML 场景未配置的字段(如 model 留空)回退 .env,保持向后兼容。""" + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-env") + monkeypatch.setenv("DEEPSEEK_MODEL", "deepseek-env-model") + _patch_scene(monkeypatch, {"provider": "deepseek", "model": "", "temperature": 0.7}) + cfg = load_llm_config(scene="daily_report") + assert cfg.provider == "deepseek" + assert cfg.model == "deepseek-env-model" + assert cfg.api_key == "sk-env" + assert cfg.temperature == 0.7 + + +def test_load_llm_config_scene_api_key_env_name(monkeypatch: pytest.MonkeyPatch) -> None: + """api_key_env 指向自定义环境变量时,优先使用该变量。""" + monkeypatch.setenv("MY_CUSTOM_KEY", "sk-custom") + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-default") + monkeypatch.setenv("DEEPSEEK_MODEL", "deepseek-m") + _patch_scene(monkeypatch, { + "provider": "deepseek", + "model": "deepseek-scene-m", + "api_key_env": "MY_CUSTOM_KEY", + }) + cfg = load_llm_config(scene="daily_report") + assert cfg.api_key == "sk-custom" + assert cfg.model == "deepseek-scene-m" + + +def test_load_llm_config_scene_explicit_args_win(monkeypatch: pytest.MonkeyPatch) -> None: + """CLI/显式参数优先级最高,覆盖 YAML 场景。""" + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-env") + _patch_scene(monkeypatch, {"provider": "qwen", "model": "qwen-max"}) + monkeypatch.setenv("QWEN_API_KEY", "sk-qwen") + cfg = load_llm_config(provider="deepseek", model="deepseek-chat", scene="daily_report") + assert cfg.provider == "deepseek" + assert cfg.model == "deepseek-chat" + + +def test_load_llm_config_scene_missing_model_raises(monkeypatch: pytest.MonkeyPatch) -> None: + """场景与 .env 都未配置模型时必须报错(无内置兜底)。""" + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test") + monkeypatch.delenv("DEEPSEEK_MODEL", raising=False) + monkeypatch.delenv("LLM_MODEL", raising=False) + _patch_scene(monkeypatch, {"provider": "deepseek", "model": ""}) + with pytest.raises(ValueError, match="模型"): + load_llm_config(scene="daily_report") + + +def test_load_llm_config_real_yaml_parseable() -> None: + """真实 configs/llm_models.yaml 必须可解析且包含全部场景(回归保护)。""" + from configs.loader import load_defaults, load_scene_config + + for scene in ("event_extraction", "daily_report", "stock_report", "embedding"): + assert isinstance(load_scene_config(scene), dict) + assert isinstance(load_defaults(), dict) + + +def test_load_llm_config_temperature_zero_is_respected( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """temperature=0 是合法配置,不应被 or 链回退默认。""" + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-env") + monkeypatch.setenv("DEEPSEEK_MODEL", "deepseek-m") + _patch_scene(monkeypatch, {"provider": "deepseek", "model": "deepseek-m", "temperature": 0}) + cfg = load_llm_config(scene="daily_report") + assert cfg.temperature == 0.0 + + +def test_load_llm_config_scene_max_attempts_zero() -> None: + """max_attempts=0 由 _pick_int 显式处理。""" + from llm.client import _pick_int + + assert _pick_int({"max_attempts": 0}, "max_attempts", 3) == 0 + assert _pick_int({"max_attempts": ""}, "max_attempts", 3) == 3 + assert _pick_int({}, "max_attempts", 3) == 3 diff --git a/tests/test_mcp.py b/tests/test_mcp.py new file mode 100644 index 0000000..87ee7dc --- /dev/null +++ b/tests/test_mcp.py @@ -0,0 +1,129 @@ +"""M8 MCP 服务测试。 + +验证工具存在 + 格式化逻辑 + 降级行为,不依赖真实嵌入/检索。 +""" + +from __future__ import annotations + +from unittest.mock import patch + +from mcp_server.tools import ( + _fmt_results, + mcp, + search_company_news, + search_news, + search_sentiment_trend, + search_stock_events, +) + +# --------------------------------------------------------------------------- # +# 工具存在性 +# --------------------------------------------------------------------------- # + +def test_mcp_server_has_name() -> None: + assert mcp.name == "A股DeepResearch" + + +def test_all_five_tools_registered() -> None: + tool_names = [getattr(t, "name", "") for t in mcp._tool_manager._tools.values()] # type: ignore[union-attr] + expected = { + "search_news", "search_company_news", "search_industry_news", + "search_stock_events", "search_sentiment_trend", + } + assert set(tool_names) == expected + + +# --------------------------------------------------------------------------- # +# _fmt_results +# --------------------------------------------------------------------------- # + +def _hit(title: str = "测试标题", source: str = "cls", score: float = 0.9, + sentiment: str = "positive", stock_codes: list[str] | None = None, + company_names: list[str] | None = None, summary: str = "摘要", + publish_time: str = "2026-06-16T10:00:00", + url: str = "https://example.com/1") -> dict: + return { + "title": title, "url": url, "source": source, "score": score, + "publish_time": publish_time, + "event": { + "sentiment": sentiment, "importance": 4, "event_type": "重大合同", + "stock_codes": stock_codes or [], "company_names": company_names or [], + "industries": ["动力电池"], "summary": summary, + }, + } + + +def test_fmt_results_contains_title_and_source() -> None: + hits = [_hit("宁德时代签百亿合同", "cls")] + out = _fmt_results(hits, "宁德时代") + assert "百亿合同" in out + assert "cls" in out + assert "0.9" in out + + +def test_fmt_results_includes_event_fields() -> None: + hits = [_hit( + company_names=["宁德时代"], stock_codes=["300750"], + )] + out = _fmt_results(hits, "查询") + assert "300750" in out + assert "宁德时代" in out + assert "动力电池" in out + assert "摘要" in out + + +def test_fmt_results_empty_returns_hint() -> None: + out = _fmt_results([], "无结果") + assert "未找到" in out and "无结果" in out + + +def test_fmt_results_multiple_hits() -> None: + hits = [_hit(f"测试{i}") for i in range(3)] + out = _fmt_results(hits, "查询") + assert "共 3 条" in out + + +# --------------------------------------------------------------------------- # +# 工具:降级行为(无精确命中时走纯语义) +# --------------------------------------------------------------------------- # + +@patch("mcp_server.tools._search") +def test_search_company_news_falls_back_on_empty(mock_search) -> None: + """company 精确命中 0 条时,降级为无 filter 纯语义搜索。""" + mock_search.side_effect = [ + [], # 第一次:精确匹配 0 条 + [_hit("fallback")], # 降级: 纯语义 + ] + out = search_company_news("查询", company="不存在的公司") + assert "fallback" in out + + +@patch("mcp_server.tools._search") +def test_search_stock_events_normalizes_code(mock_search) -> None: + """stock_code 应去掉后缀,统一大写。""" + mock_search.return_value = [_hit("结果")] + out = search_stock_events("查询", stock_code="300750.SZ") + mock_search.assert_called() # code 应为 '300750' + assert "结果" in out + + +@patch("mcp_server.tools._search") +def test_search_sentiment_trend_includes_stats(mock_search) -> None: + mock_search.return_value = [ + _hit("a", sentiment="positive"), + _hit("b", sentiment="positive"), + _hit("c", sentiment="negative"), + _hit("d", sentiment="neutral"), + ] + out = search_sentiment_trend("查询", sentiment="all", top_k=10) + assert "利好 2" in out + assert "利空 1" in out + assert "中性 1" in out + assert "共 4 条" in out + + +@patch("mcp_server.tools._search") +def test_search_news_passthrough(mock_search) -> None: + mock_search.return_value = [_hit("结果")] + out = search_news("查询") + assert "结果" in out diff --git a/tests/test_report_builder.py b/tests/test_report_builder.py new file mode 100644 index 0000000..1916a41 --- /dev/null +++ b/tests/test_report_builder.py @@ -0,0 +1,285 @@ +"""日报结构化组装单元测试(_build_report_data,纯逻辑)。""" + +from __future__ import annotations + +import json +from datetime import date + +import pytest + +from scheduler.reporter import _build_report_data + + +def _fake_event(title: str, importance: int, event_type: str = "其他", + sentiment: str = "neutral", source_id: str = "cls", + url: str = "https://x.com/1") -> dict: + return { + "title": title, + "url": url, + "source_id": source_id, + "event": { + "stock_codes": [], + "company_names": [], + "industries": [], + "sentiment": sentiment, + "importance": importance, + "event_type": event_type, + "summary": f"{title}的摘要", + }, + } + + +class TestBuildReportData: + def test_sections_and_ranks(self) -> None: + news = { + "total": 2, "hi_threshold": 4, + "high": [_fake_event("新闻A", 5), _fake_event("新闻B", 4)], + "sentiments": {"neutral": 2}, "importances": {5: 1, 4: 1}, + "event_types": {"其他": 2}, + } + cninfo = { + "total": 1, "hi_threshold": 2, + "high": [_fake_event("公告C", 3, event_type="公告")], + "by_day": {"07月10日": 1}, "announcement": 1, "research": 0, "irm": 0, + } + pipeline = {"raw_total": 100, "proc": 90} + xwlb = {"items": [_fake_event("联播D", 4, event_type="新闻联播", source_id="xwlb")], + "date": "07月10日"} + + r = _build_report_data(news, cninfo, pipeline, "AI摘要", "20260710", xwlb=xwlb) + + assert r.report_date == date(2026, 7, 10) + assert r.report_type == "finance" + assert r.file_name == "" + assert r.ai_summary == "AI摘要" + assert [(e.section, e.rank) for e in r.events] == [ + ("news", 1), ("news", 2), ("cninfo", 1), ("xwlb", 1), + ] + assert r.events[0].source == "cls" + assert r.events[3].source == "xwlb" + + def test_stats_snapshot(self) -> None: + news = {"total": 1, "hi_threshold": 4, "high": [], "sentiments": {}, + "importances": {}, "event_types": {}} + cninfo = {"total": 0, "hi_threshold": 0, "high": [], "by_day": {}, + "announcement": 0, "research": 0, "irm": 0} + r = _build_report_data(news, cninfo, {"raw_total": 100}, "s", "20260710") + # stats 可 JSON 序列化(入库时 json.dumps) + json.dumps(r.stats, ensure_ascii=False) + assert r.stats["pipeline"] == {"raw_total": 100} + assert r.stats["news"]["total"] == 1 + assert "xwlb" not in r.stats + + def test_title_truncated(self) -> None: + news = {"total": 1, "hi_threshold": 4, + "high": [_fake_event("长" * 600, 4)], "sentiments": {}, + "importances": {}, "event_types": {}} + cninfo = {"total": 0, "hi_threshold": 0, "high": [], "by_day": {}, + "announcement": 0, "research": 0, "irm": 0} + r = _build_report_data(news, cninfo, {}, "s", "20260710") + assert len(r.events[0].title) == 512 + + def test_empty_events(self) -> None: + news = {"total": 0, "hi_threshold": 0, "high": [], "sentiments": {}, + "importances": {}, "event_types": {}} + cninfo = {"total": 0, "hi_threshold": 0, "high": [], "by_day": {}, + "announcement": 0, "research": 0, "irm": 0} + r = _build_report_data(news, cninfo, {}, None, "20260710") + assert r.events == [] + assert r.ai_summary is None + + +class TestLlmCallRetry: + """_llm_call 重试逻辑(纯逻辑,mock client)。""" + + @staticmethod + def _fake_client(failures: int): + """构造 mock client:前 failures 次抛 ConnectionError,之后成功。""" + from types import SimpleNamespace + + n = {"count": 0} + + class Completions: + def create(self, **kwargs): + n["count"] += 1 + if n["count"] <= failures: + raise ConnectionError("transient") + return SimpleNamespace( + choices=[SimpleNamespace( + message=SimpleNamespace(content="今日要点摘要"), + finish_reason="stop", + )] + ) + + return SimpleNamespace(chat=SimpleNamespace(completions=Completions())), n + + @staticmethod + def _cfg(): + from llm.client import LLMConfig + + return LLMConfig( + provider="deepseek", model="deepseek-v4-flash", + api_key="sk-test", base_url="https://api.deepseek.com", + temperature=0.3, + ) + + def test_success_first_try(self) -> None: + from scheduler.reporter import _llm_call + client, n = self._fake_client(0) + out = _llm_call(client, self._cfg(), "p") + assert out == "今日要点摘要" + assert n["count"] == 1 + + def test_retry_then_success(self, monkeypatch) -> None: + import scheduler.reporter as rep + monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 3) + monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01) + client, n = self._fake_client(2) # 前 2 次失败,第 3 次成功 + out = rep._llm_call(client, self._cfg(), "p") + assert out == "今日要点摘要" + assert n["count"] == 3 + + def test_exhausts_retries_raises(self, monkeypatch) -> None: + import scheduler.reporter as rep + monkeypatch.setattr(rep, "_LLM_RETRY_TIMES", 2) + monkeypatch.setattr(rep, "_LLM_RETRY_BACKOFF_SEC", 0.01) + client, n = self._fake_client(99) # 一直失败 + with pytest.raises(ConnectionError): + rep._llm_call(client, self._cfg(), "p") + assert n["count"] == 2 # 重试 2 次后放弃 + + def test_llm_summarize_single_chunk_passes_config(self) -> None: + """_llm_summarize → _llm_call 全链路:必须传 LLMConfig 而非 model 字符串。 + + 回归保护:生产日报曾因 _llm_call 收到 str 报 + 'str' object has no attribute 'model'。 + """ + from scheduler.reporter import _llm_summarize + client, n = self._fake_client(0) + out = _llm_summarize(client, self._cfg(), ["- 新闻A", "- 新闻B"], "20260812") + assert out == "今日要点摘要" + assert n["count"] == 1 + + def test_llm_summarize_multi_chunk_passes_config(self) -> None: + """多分块场景:每块 + 合并各调用一次 _llm_call,均传 config。""" + from scheduler.reporter import _llm_summarize + client, n = self._fake_client(0) + # 两条长行保证触发分块 + lines = ["- " + "长新闻内容" * 300, "- " + "长新闻内容" * 300] + out = _llm_summarize(client, self._cfg(), lines, "20260812") + assert out == "今日要点摘要" + assert n["count"] == 3 # 2 块 + 1 次合并 + + +class TestCollectXwlb: + """_collect_xwlb 取数逻辑:应查询日报前一日(已播出的联播),并跳过内容提要。""" + + def test_queries_previous_day_and_skips_toc(self, monkeypatch) -> None: + import json as _json + import urllib.request + + captured: dict[str, str] = {} + + def fake_urlopen(req, timeout=15): # noqa: ARG001 + captured["url"] = req.full_url + + class Resp: + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + def read(self): + return _json.dumps({"data": {"news": [ + {"daily_sub_id": 1, "news_title": "内容提要", "news_days": "2026-08-04", "news_improve": "开场白"}, + {"daily_sub_id": 2, "news_title": "联播要闻A", "news_days": "2026-08-04", "news_improve": "正文A"}, + ]}}).encode("utf-8") + + return Resp() + + monkeypatch.setattr(urllib.request, "urlopen", fake_urlopen) + from scheduler.reporter import _collect_xwlb + + result = _collect_xwlb("20260805") + # 查询的是前一日(20260804)而非当日 + assert "start_date=20260804" in captured["url"] + assert "end_date=20260804" in captured["url"] + # 跳过第 1 条内容提要 + assert len(result["items"]) == 1 + assert result["items"][0]["title"] == "联播要闻A" + assert result["source_date"] == "20260804" + assert result["date"] == "08月04日" + + +class TestCollectNewsEventsLookback: + """_collect_news_events 30 小时回溯逻辑。""" + + def test_filters_30h_and_excludes_cninfo(self, monkeypatch) -> None: + from datetime import datetime, timedelta + + import scheduler.reporter as rep + + now = datetime.now().astimezone() + + def fake_load(day_str: str) -> list[dict]: # noqa: ARG001 + def ev(title: str, hours_ago: float | None, source: str = "cls", + importance: int = 5, aware: bool = False) -> dict: + pt = None + if hours_ago is not None: + t = now - timedelta(hours=hours_ago) + pt = t.isoformat() if not aware else t.astimezone().isoformat() + return { + "title": title, "url": "u", "source_id": source, + "publish_time": pt, + "event": {"importance": importance, "sentiment": "neutral", + "event_type": "其他", "summary": "s"}, + } + + return [ + ev("窗口内新闻", 10), + ev("窗口内新闻带时区", 12, aware=True), + ev("窗口外旧闻", 40), + ev("无时间戳", None), + ev("公告排除", 5, source="cninfo", importance=2), + ] + + monkeypatch.setattr(rep, "_load_events_from_dir", fake_load) + result = rep._collect_news_events("20260805") + + # 两个日期目录各返回 5 条(共 10): 旧闻×2、公告×2 被滤, 保留 6 条 + assert result["total"] == 6 + titles = {e["title"] for e in result["high"]} + assert "窗口内新闻" in titles + assert "窗口内新闻带时区" in titles + assert "无时间戳" in titles + assert "窗口外旧闻" not in titles + assert "公告排除" not in titles + + +class TestLoadEventsSources: + """_load_events_from_dir 读取多源(sources)字段。""" + + def test_loads_sources_with_fallback(self, monkeypatch, tmp_path) -> None: + import json as _json + + import scheduler.reporter as rep + + ev_dir = tmp_path / "data" / "events" / "20260812" + ev_dir.mkdir(parents=True) + (ev_dir / "aaa.json").write_text(_json.dumps({ + "source_id": "cls", "sources": ["cls", "sina", "eastmoney"], + "title": "多源新闻", "url": "https://x/1", "event": {}, + }, ensure_ascii=False), encoding="utf-8") + (ev_dir / "bbb.json").write_text(_json.dumps({ + "source_id": "cls", "title": "旧产物无 sources", "url": "https://x/2", + "event": {}, + }, ensure_ascii=False), encoding="utf-8") + + monkeypatch.chdir(tmp_path) + events = rep._load_events_from_dir("20260812") + by_url = {e["url"]: e for e in events} + # 新产物:多源完整透传 + assert by_url["https://x/1"]["sources"] == ["cls", "sina", "eastmoney"] + # 旧产物:兜底 [source_id] + assert by_url["https://x/2"]["sources"] == ["cls"] diff --git a/tests/test_report_db.py b/tests/test_report_db.py new file mode 100644 index 0000000..3c8898c --- /dev/null +++ b/tests/test_report_db.py @@ -0,0 +1,58 @@ +"""日报数据模型单元测试。""" + +from __future__ import annotations + +from datetime import date, datetime + +import pytest +from pydantic import ValidationError + +from report_db.models import EventRow, ReportData + + +class TestEventRow: + def test_minimal(self) -> None: + ev = EventRow(section="news", rank=1, title="标题") + assert ev.importance is None + assert ev.sentiment is None + + def test_full(self) -> None: + ev = EventRow( + section="intl", rank=2, importance=4, event_type="地缘政治", + title="t", summary="s", sentiment="negative", source="ForexLive", + sources=["ForexLive", "新浪财经"], + url="https://x.com/1", + ) + assert ev.sentiment == "negative" + assert ev.sources == ["ForexLive", "新浪财经"] + + def test_sources_optional(self) -> None: + """sources 为可选项(旧数据无多源记录)。""" + ev = EventRow(section="news", rank=1, title="t") + assert ev.sources is None + + def test_missing_title_raises(self) -> None: + with pytest.raises(ValidationError): + EventRow(section="news", rank=1) # type: ignore[call-arg] + + +class TestReportData: + def test_defaults(self) -> None: + r = ReportData( + report_date=date(2026, 7, 11), + report_type="finance", + generated_at=datetime(2026, 7, 11, 7, 0), + ) + assert r.file_name == "" + assert r.stats == {} + assert r.events == [] + + def test_with_events(self) -> None: + r = ReportData( + report_date=date(2026, 7, 11), + report_type="finance", + generated_at=datetime(2026, 7, 11, 7, 0), + ai_summary="摘要", + events=[EventRow(section="xwlb", rank=1, title="t")], + ) + assert len(r.events) == 1 diff --git a/tests/test_report_import.py b/tests/test_report_import.py new file mode 100644 index 0000000..f6660ac --- /dev/null +++ b/tests/test_report_import.py @@ -0,0 +1,32 @@ +"""历史导入器单元测试(文件匹配/扫描逻辑,不依赖真实 DB)。""" + +from __future__ import annotations + +from pathlib import Path + +from report_import.importer import _match_file + + +class TestMatchFile: + def test_finance(self) -> None: + p = Path("20260711/finance_news_daily_20260710_0720.html") + assert _match_file(p, None, None) + assert _match_file(p, "20260710", None) + assert _match_file(p, None, "finance") + assert not _match_file(p, "20260711", None) # 文件名日期不含 20260711 + assert not _match_file(p, None, "intl") + + def test_no_timestamp_suffix(self) -> None: + # 早期文件无时间戳后缀,也应匹配 + p = Path("20260616/finance_news_daily_20260616.html") + assert _match_file(p, None, None) + assert _match_file(p, "20260616", "finance") + + def test_intl(self) -> None: + p = Path("20260711/intl_news_daily_20260711_070304.html") + assert _match_file(p, "20260711", "intl") + assert not _match_file(p, None, "finance") + + def test_non_report_ignored(self) -> None: + assert not _match_file(Path("20260711/002714.SZ_0724.html"), None, None) + assert not _match_file(Path("20260711/readme.md"), None, None) diff --git a/tests/test_report_parser.py b/tests/test_report_parser.py new file mode 100644 index 0000000..e94f99b --- /dev/null +++ b/tests/test_report_parser.py @@ -0,0 +1,98 @@ +"""历史日报解析器单元测试(基于真实样例 HTML)。""" + +from __future__ import annotations + +from datetime import date, datetime +from pathlib import Path + +import pytest + +from report_import.parser import ( + ReportParseError, + parse_finance_report, + parse_intl_report, + parse_report, +) + +FIXTURES = Path(__file__).parent / "fixtures" +FINANCE_HTML = (FIXTURES / "finance_news_daily_20260710_0720.html").read_text(encoding="utf-8") +INTL_HTML = (FIXTURES / "intl_news_daily_20260711_070304.html").read_text(encoding="utf-8") + + +class TestFinanceParse: + def test_metadata(self) -> None: + r = parse_finance_report(FINANCE_HTML, "finance_news_daily_20260710_0720.html") + assert r.report_date == date(2026, 7, 10) + assert r.report_type == "finance" + assert r.file_name == "finance_news_daily_20260710_0720.html" + assert r.generated_at == datetime(2026, 7, 11, 7, 20, 27) # header"生成于"优先 + + def test_ai_summary_lines(self) -> None: + r = parse_finance_report(FINANCE_HTML, "finance_news_daily_20260710_0720.html") + assert r.ai_summary is not None + assert len(r.ai_summary.splitlines()) >= 5 + assert "碳达峰" in r.ai_summary + + def test_sections_and_ranks(self) -> None: + r = parse_finance_report(FINANCE_HTML, "finance_news_daily_20260710_0720.html") + sections = {e.section for e in r.events} + assert sections == {"xwlb", "news", "cninfo"} + assert sum(1 for e in r.events if e.section == "xwlb") == 16 + assert sum(1 for e in r.events if e.section == "news") == 20 + assert sum(1 for e in r.events if e.section == "cninfo") == 20 + + def test_event_fields(self) -> None: + r = parse_finance_report(FINANCE_HTML, "finance_news_daily_20260710_0720.html") + xwlb = next(e for e in r.events if e.section == "xwlb") + assert xwlb.importance == 4 + assert xwlb.event_type == "新闻联播" + assert xwlb.sentiment == "neutral" + assert "张国清" in xwlb.title + news = next(e for e in r.events if e.section == "news" and e.source) + assert news.source # 新闻板块带来源 + + def test_stats_keys(self) -> None: + r = parse_finance_report(FINANCE_HTML, "finance_news_daily_20260710_0720.html") + for key in ("pipeline", "sources", "sentiment", "importance", "event_types"): + assert key in r.stats, f"缺少 stats.{key}" + assert r.stats["pipeline"]["M1 原始文章"] == 758 + + +class TestIntlParse: + def test_metadata(self) -> None: + r = parse_intl_report(INTL_HTML, "intl_news_daily_20260711_070304.html") + assert r.report_date == date(2026, 7, 11) + assert r.report_type == "intl" + assert r.generated_at == datetime(2026, 7, 11, 7, 3, 30) + + def test_events_and_source_from_small(self) -> None: + r = parse_intl_report(INTL_HTML, "intl_news_daily_20260711_070304.html") + assert len(r.events) == 19 + first = r.events[0] + assert first.section == "intl" + assert first.source == "investinglive.com" # 从摘要 [来源] 提取 + assert first.url.startswith("https://investinglive.com/") + assert first.importance == 4 + assert first.event_type == "地缘政治" + # 摘要中不应残留 [来源] 标记 + assert first.summary is not None and "[investinglive.com]" not in first.summary + + def test_stats_keys(self) -> None: + r = parse_intl_report(INTL_HTML, "intl_news_daily_20260711_070304.html") + for key in ("pipeline", "sentiment", "importance", "event_types", "source_dist"): + assert key in r.stats + assert r.stats["source_dist"][0]["来源"] == "ForexLive" + + +class TestParseReportDispatch: + def test_dispatch_finance(self) -> None: + r = parse_report(FINANCE_HTML, "finance_news_daily_20260710_0720.html") + assert r.report_type == "finance" + + def test_dispatch_intl(self) -> None: + r = parse_report(INTL_HTML, "intl_news_daily_20260711_070304.html") + assert r.report_type == "intl" + + def test_bad_filename_raises(self) -> None: + with pytest.raises(ReportParseError): + parse_report("", "not_a_report.html") diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py new file mode 100644 index 0000000..80b34ff --- /dev/null +++ b/tests/test_scheduler.py @@ -0,0 +1,132 @@ +"""M7 定时任务测试。 + +使用 mock subprocess.run,不依赖真实脚本执行。 +""" + +from __future__ import annotations + +import subprocess +from pathlib import Path +from unittest.mock import MagicMock, patch + +from scheduler import PipelineResult, StepResult, run_pipeline, run_step + +# --------------------------------------------------------------------------- # +# fixtures +# --------------------------------------------------------------------------- # + +def _mock_proc(returncode: int = 0, stderr: str = "") -> MagicMock: + p = MagicMock() + p.returncode = returncode + p.stderr = stderr + p.stdout = "ok" + return p + + +# --------------------------------------------------------------------------- # +# StepResult 模型 +# --------------------------------------------------------------------------- # + +def test_step_result_defaults() -> None: + sr = StepResult(name="crawler", success=True, elapsed_sec=12.3) + assert sr.name == "crawler" + assert sr.success is True + assert sr.elapsed_sec == 12.3 + assert sr.exit_code is None + + +def test_pipeline_result_all_success() -> None: + pr = PipelineResult(steps=[ + StepResult(name="crawler", success=True, elapsed_sec=1), + StepResult(name="extractor", success=True, elapsed_sec=2), + ]) + assert pr.all_success is True + + +def test_pipeline_result_partial_failure() -> None: + pr = PipelineResult(steps=[ + StepResult(name="crawler", success=True, elapsed_sec=1), + StepResult(name="extractor", success=False, elapsed_sec=0), + ]) + assert pr.all_success is False + + +# --------------------------------------------------------------------------- # +# run_step +# --------------------------------------------------------------------------- # + +def test_run_step_success() -> None: + with patch("subprocess.run", return_value=_mock_proc(returncode=0, + stderr="INFO | 完成: 成功 20/20")): + sr = run_step("crawler", "20260616") + assert sr.success is True + assert sr.exit_code == 0 + + +def test_run_step_failure() -> None: + with patch("subprocess.run", return_value=_mock_proc(returncode=1)): + sr = run_step("extractor", "20260616") + assert sr.success is False + assert sr.exit_code == 1 + assert "rc=1" in sr.tail_msg + + +def test_run_step_timeout() -> None: + with patch("subprocess.run", side_effect=subprocess.TimeoutExpired(cmd=["uv"], timeout=10)): + sr = run_step("llm", "20260616") + assert sr.success is False + assert "超时" in sr.tail_msg + + +def test_run_step_exception() -> None: + with patch("subprocess.run", side_effect=OSError("磁盘满")): + sr = run_step("embedding", "20260616") + assert sr.success is False + assert "磁盘满" in sr.tail_msg + + +def test_run_step_unknown_name() -> None: + sr = run_step("nonexistent", "20260616") + assert sr.success is False + assert "未知" in sr.tail_msg + + +# --------------------------------------------------------------------------- # +# run_pipeline +# --------------------------------------------------------------------------- # + +def test_run_pipeline_all_success() -> None: + with ( + patch("subprocess.run", return_value=_mock_proc(returncode=0)), + patch("scheduler.reporter.generate_report", return_value=Path("/tmp/r.html")), + ): + result = run_pipeline("20260616", steps=["crawler", "extractor", "dedup"]) + assert len(result.steps) == 3 + assert result.all_success is True + assert result.started_at is not None + assert result.finished_at is not None + + +def test_run_pipeline_continues_on_failure() -> None: + """中间步骤失败,后续继续执行(不阻断)。""" + call_count = {"n": 0} + + def _side_effect(*args, **kwargs): + call_count["n"] += 1 + if call_count["n"] == 2: # extractor 失败 + return _mock_proc(returncode=1, stderr="GNE 提取异常") + return _mock_proc(returncode=0, stderr="ok") + + with patch("subprocess.run", side_effect=_side_effect): + result = run_pipeline("20260616", steps=["crawler", "extractor", "dedup", "llm"]) + assert len(result.steps) == 4 + # extractor 失败,但后续仍执行 + assert result.steps[1].success is False + assert result.steps[2].success is True + + +def test_run_pipeline_custom_steps() -> None: + with patch("subprocess.run", return_value=_mock_proc(returncode=0, stderr="ok")): + result = run_pipeline("20260616", steps=["crawler", "extractor"]) + assert len(result.steps) == 2 + assert result.all_success is True diff --git a/tests/test_vectorstore.py b/tests/test_vectorstore.py new file mode 100644 index 0000000..04a2490 --- /dev/null +++ b/tests/test_vectorstore.py @@ -0,0 +1,243 @@ +"""M6 Qdrant 向量存储模块测试。 + +使用 qdrant-client 内存模式(:memory:),不依赖 Docker。 +""" + +from __future__ import annotations + +import pytest + +from vectorstore import ( + SearchFilter, + SearchResult, + VectorStore, + make_qdrant_client, +) +from vectorstore.client import _build_filter + +# --------------------------------------------------------------------------- # +# fixtures +# --------------------------------------------------------------------------- # + +@pytest.fixture +def store() -> VectorStore: + c = make_qdrant_client(memory=True) + s = VectorStore(c, collection_name="test_m6", vector_dim=4) + s.init_collection() + yield s + s.close() + + +def _point(id_: str, vector: list[float], **payload: object) -> dict: + return {"id": id_, "vector": vector, "payload": dict(payload)} + + +# --------------------------------------------------------------------------- # +# Collection 管理 +# --------------------------------------------------------------------------- # + +def test_init_collection_creates(store: VectorStore) -> None: + info = store.info() + assert info.exists is True + assert info.name == "test_m6" + + +def test_init_collection_idempotent(store: VectorStore) -> None: + """再次 init 不应报错,count 不变。""" + store.upsert([_point("a", [1.0, 0, 0, 0])]) + store.init_collection() # 不应重建 + assert store.count() == 1 + + +def test_init_collection_recreate_clears(store: VectorStore) -> None: + store.upsert([_point("a", [1.0, 0, 0, 0])]) + store.init_collection(recreate=True) + assert store.count() == 0 + + +def test_delete_collection(store: VectorStore) -> None: + store.delete_collection() + assert store.info().exists is False + # 再次 init 应恢复 + store.init_collection() + assert store.info().exists is True + + +# --------------------------------------------------------------------------- # +# upsert + count +# --------------------------------------------------------------------------- # + +def test_upsert_and_count(store: VectorStore) -> None: + store.upsert([ + _point("a", [1, 0, 0, 0], title="Article A"), + _point("b", [0, 1, 0, 0], title="Article B"), + ]) + assert store.count() == 2 + + +def test_upsert_idempotent(store: VectorStore) -> None: + """同 url_hash 再次 upsert 不应增加 count,数据被覆盖。""" + store.upsert([_point("a", [1, 0, 0, 0], title="Old")]) + store.upsert([_point("a", [0, 0, 0, 1], title="New")]) + assert store.count() == 1 + + +# --------------------------------------------------------------------------- # +# query - 语义检索 +# --------------------------------------------------------------------------- # + +def test_query_returns_score_desc(store: VectorStore) -> None: + store.upsert([ + _point("a", [1.0, 0, 0, 0], title="A"), + _point("b", [0.0, 1.0, 0, 0], title="B"), + _point("c", [0.0, 0, 1.0, 0], title="C"), + ]) + results = store.query(query_vector=[0.9, 0.1, 0, 0], top_k=2) + assert len(results) == 2 + assert results[0].url_hash == "a" + # score 应递减 + assert results[0].score >= results[1].score + + +def test_query_score_threshold(store: VectorStore) -> None: + store.upsert([ + _point("a", [1, 0, 0, 0], title="A"), + _point("b", [0, 1, 0, 0], title="B"), + ]) + # 只有 a 会匹配 + results = store.query(query_vector=[1, 0, 0, 0], top_k=10, score_threshold=0.9) + assert len(results) == 1 + assert results[0].url_hash == "a" + + +# --------------------------------------------------------------------------- # +# query - 结构化过滤 +# --------------------------------------------------------------------------- # + +def test_query_filter_by_source_id(store: VectorStore) -> None: + store.upsert([ + _point("a1", [1, 0, 0, 0], source_id="cls", title="CLS article"), + _point("a2", [0.9, 0.1, 0, 0], source_id="sina", title="Sina article"), + ]) + results = store.query( + query_vector=[1, 0, 0, 0], + filter=SearchFilter(source_id="cls"), + top_k=5, + ) + assert len(results) == 1 + assert results[0].source_id == "cls" + + +def test_query_filter_by_stock_codes(store: VectorStore) -> None: + store.upsert([ + _point("a", [1, 0, 0, 0], source_id="cls", + event={"stock_codes": ["300750"], "sentiment": "positive"}), + _point("b", [0.9, 0.1, 0, 0], source_id="sina", + event={"stock_codes": ["000001"], "sentiment": "neutral"}), + _point("c", [0.8, 0.2, 0, 0], source_id="sina", + event={"stock_codes": ["300750"], "sentiment": "negative"}), + ]) + results = store.query( + query_vector=[1, 0, 0, 0], + filter=SearchFilter(stock_codes=["300750"]), + top_k=5, + ) + assert len(results) == 2 + for r in results: + assert "300750" in (r.event or {}).get("stock_codes", []) + + +def test_query_filter_by_sentiment(store: VectorStore) -> None: + store.upsert([ + _point("a", [1, 0, 0, 0], source_id="cls", + event={"sentiment": "positive"}), + _point("b", [0, 1, 0, 0], source_id="cls", + event={"sentiment": "negative"}), + ]) + results = store.query( + query_vector=[1, 0, 0, 0], + filter=SearchFilter(sentiment="positive"), + top_k=5, + ) + assert len(results) >= 1 + assert all((r.event or {}).get("sentiment") == "positive" for r in results) + + +def test_query_filter_by_importance_min(store: VectorStore) -> None: + store.upsert([ + _point("a", [1, 0, 0, 0], source_id="cls", + event={"importance": 2}), + _point("b", [0, 1, 0, 0], source_id="cls", + event={"importance": 4}), + _point("c", [0, 0, 1, 0], source_id="cls", + event={"importance": 5}), + ]) + results = store.query( + query_vector=[0.5, 0.5, 0.5, 0], + filter=SearchFilter(importance_min=4), + top_k=5, + ) + assert all((r.event or {}).get("importance", 0) >= 4 for r in results) + + +def test_query_filter_by_industry(store: VectorStore) -> None: + store.upsert([ + _point("a", [1, 0, 0, 0], source_id="cls", + event={"industries": ["动力电池"]}), + _point("b", [0, 1, 0, 0], source_id="cls", + event={"industries": ["白酒"]}), + ]) + results = store.query( + query_vector=[1, 0, 0, 0], + filter=SearchFilter(industries=["动力电池"]), + top_k=5, + ) + assert len(results) >= 1 + for r in results: + assert "动力电池" in (r.event or {}).get("industries", []) + + +# --------------------------------------------------------------------------- # +# Filter 构建 +# --------------------------------------------------------------------------- # + +def test_build_filter_empty_returns_none() -> None: + assert _build_filter(SearchFilter()) is None + + +def test_build_filter_source_id() -> None: + f = _build_filter(SearchFilter(source_id="cls")) + assert f is not None and len(f.must) == 1 # type: ignore[arg-type] + + +def test_build_filter_date_range() -> None: + f = _build_filter(SearchFilter(publish_date_from="2026-06-01", publish_date_to="2026-06-30")) + assert f is not None + # range 应含 gte + lte + cond = f.must[0] # type: ignore[union-attr] + assert cond.key == "publish_time" + + +# --------------------------------------------------------------------------- # +# search_result 模型 +# --------------------------------------------------------------------------- # + +def test_search_result_short_summary() -> None: + r = SearchResult( + url_hash="abc", + score=0.95, + title="宁德时代签约 100GWh 协议", + url="https://x/1", + source_id="cls", + event={"stock_codes": ["300750"], "sentiment": "positive"}, + ) + s = r.short_summary() + assert "cls" in s and "0.9500" in s and "300750" in s + + +def test_search_result_handles_none_event() -> None: + r = SearchResult( + url_hash="abc", score=0.5, title="t", url="u", source_id="cls", event=None, + ) + s = r.short_summary() + assert "-" in s # 无 stock_code diff --git a/uv.lock b/uv.lock new file mode 100644 index 0000000..a9690c1 --- /dev/null +++ b/uv.lock @@ -0,0 +1,2917 @@ +version = 1 +revision = 3 +requires-python = "==3.11.*" +resolution-markers = [ + "sys_platform == 'win32'", + "sys_platform != 'win32'", +] + +[[package]] +name = "a-share-research" +version = "0.1.0" +source = { editable = "." } +dependencies = [ + { name = "apscheduler" }, + { name = "beautifulsoup4" }, + { name = "crawl4ai" }, + { name = "gne" }, + { name = "httpx" }, + { name = "loguru" }, + { name = "lxml" }, + { name = "markitdown", extra = ["all"] }, + { name = "mcp" }, + { name = "numpy" }, + { name = "openai" }, + { name = "pydantic" }, + { name = "pymysql" }, + { name = "pypdf" }, + { name = "python-dateutil" }, + { name = "python-dotenv" }, + { name = "pyyaml" }, + { name = "qdrant-client" }, + { name = "tenacity" }, +] + +[package.optional-dependencies] +local-embedding = [ + { name = "sentence-transformers" }, + { name = "torch" }, +] + +[package.dev-dependencies] +dev = [ + { name = "mypy" }, + { name = "pytest" }, + { name = "pytest-asyncio" }, + { name = "pytest-cov" }, + { name = "ruff" }, +] + +[package.metadata] +requires-dist = [ + { name = "apscheduler", specifier = ">=3.10" }, + { name = "beautifulsoup4", specifier = ">=4.12" }, + { name = "crawl4ai", specifier = ">=0.6,<0.8" }, + { name = "gne", specifier = ">=0.3" }, + { name = "httpx", specifier = ">=0.27" }, + { name = "loguru", specifier = ">=0.7" }, + { name = "lxml", specifier = ">=5.0" }, + { name = "markitdown", extras = ["all"], specifier = ">=0.1.5" }, + { name = "mcp", specifier = ">=1.0" }, + { name = "numpy", specifier = ">=1.26" }, + { name = "openai", specifier = ">=1.40" }, + { name = "pydantic", specifier = ">=2.7" }, + { name = "pymysql", specifier = ">=1.2.0" }, + { name = "pypdf", specifier = ">=6.13.3" }, + { name = "python-dateutil", specifier = ">=2.9" }, + { name = "python-dotenv", specifier = ">=1.0" }, + { name = "pyyaml", specifier = ">=6.0" }, + { name = "qdrant-client", specifier = ">=1.10" }, + { name = "sentence-transformers", marker = "extra == 'local-embedding'", specifier = ">=3.0" }, + { name = "tenacity", specifier = ">=9.0" }, + { name = "torch", marker = "extra == 'local-embedding'", specifier = ">=2.2" }, +] +provides-extras = ["local-embedding"] + +[package.metadata.requires-dev] +dev = [ + { name = "mypy", specifier = ">=1.10" }, + { name = "pytest", specifier = ">=8.0" }, + { name = "pytest-asyncio", specifier = ">=0.23" }, + { name = "pytest-cov", specifier = ">=5.0" }, + { name = "ruff", specifier = ">=0.5" }, +] + +[[package]] +name = "aiofiles" +version = "25.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/41/c3/534eac40372d8ee36ef40df62ec129bee4fdb5ad9706e58a29be53b2c970/aiofiles-25.1.0.tar.gz", hash = "sha256:a8d728f0a29de45dc521f18f07297428d56992a742f0cd2701ba86e44d23d5b2" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/bc/8a/340a1555ae33d7354dbca4faa54948d76d89a27ceef032c8c3bc661d003e/aiofiles-25.1.0-py3-none-any.whl", hash = "sha256:abe311e527c862958650f9438e859c1fa7568a141b22abcd015e120e86a85695" }, +] + +[[package]] +name = "aiohappyeyeballs" +version = "2.6.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/33/c6/61a2d7b7572279226bb2e7f61d7a19ca7c90da0329c93fa0d560cbf288d8/aiohappyeyeballs-2.6.2.tar.gz", hash = "sha256:e202810ee718bd01fc6ef49e8ea53d023d5cb6b581076d7925aa499fa55dbe64" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/5f/fc/a7bf5b6e4e617b45f90f2d9d2a68519c249c81dd4fc2658c7a2a61c4f4b7/aiohappyeyeballs-2.6.2-py3-none-any.whl", hash = "sha256:4708045e2d7a6c6bdf8aafa8ed39649eaf926a4543b54560659129e3365953c4" }, +] + +[[package]] +name = "aiohttp" +version = "3.14.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "aiohappyeyeballs" }, + { name = "aiosignal" }, + { name = "attrs" }, + { name = "frozenlist" }, + { name = "multidict" }, + { name = "propcache" }, + { name = "typing-extensions" }, + { name = "yarl" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/82/78/8ea7308cac6934de8c74a14f3d5f65d1c89287426688be79538d0e5c013d/aiohttp-3.14.1.tar.gz", hash = "sha256:307f2cff90a764d329e77040603fa032db89c5c24fdad50c4c15334cba744035" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/26/dd/bf526e6f0a1120dd6f2df2e97bacfe4d358f13d17a0ff5847301a1375a51/aiohttp-3.14.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:aa00140699487bd435fde4342d85c94cb256b7cd3a5b9c3396c67f19922afda2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8f/e1/a2872aa55495a70f61310d411541c6ee23812d9a884e000c716e1bc3edbf/aiohttp-3.14.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:1c1af67559445498b502030c35c59db59966f47041ca9de5b4e707f86bd10b5f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5b/e7/c60c7b209e509cc787de3cea0550a518538cfc08003e1c1e14c1c63fff71/aiohttp-3.14.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:d44ec478e713ee7f29b439f7eb8dc2b9d4079e11ae114d2c2ac3d5daf30516c8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5b/8d/614ace2f579702c9840ab1e1447fd8509e35b0b904f7196418fa2f57b25d/aiohttp-3.14.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d3b1a184a9a8f548a6b73f1e26b96b052193e4b3175ed7342aaf1151a1f00a04" }, + { url = "https://mirrors.aliyun.com/pypi/packages/49/e0/726e90f99542bf292f81a96a12cc4847deb86f3ccf62c6f4014a201f4d33/aiohttp-3.14.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:5f2504bc0322437c9a1ff6d3333ca56c7477b727c995f036b976ae17b98372c8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0b/4b/d176d5c4db9d33dacf0543102ea59503bc1d528af4cfd0b719949ca49389/aiohttp-3.14.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:73f05ea02013e02512c3bf42714f1208c57168c779cc6fe23516e4543089d0a6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/dc/d6/5a99b563690ea0cbed912ae94a2ce33993a5709a651a3a4fe761e7dd973a/aiohttp-3.14.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:797457503c2d426bee06eef808d07b31ede30b65e054444e7de64cad0061b7af" }, + { url = "https://mirrors.aliyun.com/pypi/packages/76/7f/a987b14a3859094b3cea3f4825219c3e5536242564af6e3f9c2f6c994eb2/aiohttp-3.14.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b821a1f7dedf7e37450654e620038ac3b2e81e8fa6ea269337e97101978ec730" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f1/1a/420e5c85a3e73349372ed22ce0b6af86bfa6ce16a4b20a64a2e94608c781/aiohttp-3.14.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:4cd96b5ba05d67ed0cf00b5b405c8cd99586d8e3481e8ee0a831057591af7621" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a7/80/18a592ed3be0a402cc03670bd72ee1f8563ddbe1d8d5542dbf868f274136/aiohttp-3.14.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:1d459b98a932296c6f0e94f87511a0b1b90a8a02c30a50e60a297619cd5a58ee" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ec/0b/8b3d5713373858ff71a617daf6e3b0e81ad63e79d09a3cf2f6b6b983939c/aiohttp-3.14.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:764457a7be60825fb770a644852ff717bcbb5042f189f2bd16df61a81b3f6573" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9f/49/fd564575cf225821d7ba5a117cb8bc27213d8a7e1811162afb43ae077039/aiohttp-3.14.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:f7a16ef45b081454ef844502d87a848876c490c4cb5c650c230f6ec79ed2c1e7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ed/1b/e850c9ae6fc91356552ae668bb6c51e93fa29c8aef13398a10b56678557f/aiohttp-3.14.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:2fbc3ed048b3475b9f0cbcb9978e9d2d3511acd91ead203af26ed9f0056004cf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/eb/94/3c337ba72451a89806ace6f75bddc92bafc5b8d53d90115a512858024b63/aiohttp-3.14.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:bedb0cd073cc2dc035e30aeb99444389d3cd2113afe4ef9fcd23d439f5bade85" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2b/9c/9c18cf367a0498212d9ba7daf990b504a5e8ae064cda4b504e2647c89c03/aiohttp-3.14.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:b6feea921016eb3d4e04d65fc4e9ca402d1a3801f562aef94989f54694917af3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b5/63/a251a9d2a6cb45065b2ddc0bde2b3dd10108740a9a42f632c66405a761a2/aiohttp-3.14.1-cp311-cp311-win32.whl", hash = "sha256:313701e488100074ce99850404ee36e741abf6330179fec908a1944ecf570126" }, + { url = "https://mirrors.aliyun.com/pypi/packages/17/ca/69274c51dcd6e8947d77b2806cf47a4a15f2c846e2cbeb1882547d3da283/aiohttp-3.14.1-cp311-cp311-win_amd64.whl", hash = "sha256:03ab4530fdcb3a543a122ba4b65ac9919da9fe9f78a03d328a6e38ff962f7aa5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2c/8a/c25904f77690c3688ec140f87591ef11a0cfe36bf3d5c0f1f38056fb62b3/aiohttp-3.14.1-cp311-cp311-win_arm64.whl", hash = "sha256:486f7d16ed54c39c2cbd7ca71fd8ba2b8bb7860df65bd7b6ed640bab96a38a8b" }, +] + +[[package]] +name = "aiosignal" +version = "1.4.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "frozenlist" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/61/62/06741b579156360248d1ec624842ad0edf697050bbaf7c3e46394e106ad1/aiosignal-1.4.0.tar.gz", hash = "sha256:f47eecd9468083c2029cc99945502cb7708b082c232f9aca65da147157b251c7" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/fb/76/641ae371508676492379f16e2fa48f4e2c11741bd63c48be4b12a6b09cba/aiosignal-1.4.0-py3-none-any.whl", hash = "sha256:053243f8b92b990551949e63930a839ff0cf0b0ebbe0597b0f3fb19e1a0fe82e" }, +] + +[[package]] +name = "aiosqlite" +version = "0.22.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/4e/8a/64761f4005f17809769d23e518d915db74e6310474e733e3593cfc854ef1/aiosqlite-0.22.1.tar.gz", hash = "sha256:043e0bd78d32888c0a9ca90fc788b38796843360c855a7262a532813133a0650" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/00/b7/e3bf5133d697a08128598c8d0abc5e16377b51465a33756de24fa7dee953/aiosqlite-0.22.1-py3-none-any.whl", hash = "sha256:21c002eb13823fad740196c5a2e9d8e62f6243bd9e7e4a1f87fb5e44ecb4fceb" }, +] + +[[package]] +name = "alphashape" +version = "1.3.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "click" }, + { name = "click-log" }, + { name = "networkx" }, + { name = "numpy" }, + { name = "rtree" }, + { name = "scipy" }, + { name = "shapely" }, + { name = "trimesh" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/2e/83/67ff905694df5b34a777123b59fdfd05998d5a31766f188aafbf5b340055/alphashape-1.3.1.tar.gz", hash = "sha256:7a27340afc5f8ed301577acec46bb0cf2bada5410045f7289142e735ef6977ec" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e4/ad/77fad9d6f974ec58d837cb49fb9b483d6227a420c4f908c3578633de1d47/alphashape-1.3.1-py2.py3-none-any.whl", hash = "sha256:96a5ddd5f09534a35f03a8916aeeaac00fe4d6bec2f9ad78f87f57be3007f795" }, +] + +[[package]] +name = "annotated-doc" +version = "0.0.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/57/ba/046ceea27344560984e26a590f90bc7f4a75b06701f653222458922b558c/annotated_doc-0.0.4.tar.gz", hash = "sha256:fbcda96e87e9c92ad167c2e53839e57503ecfda18804ea28102353485033faa4" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1e/d3/26bf1008eb3d2daa8ef4cacc7f3bfdc11818d111f7e2d0201bc6e3b49d45/annotated_doc-0.0.4-py3-none-any.whl", hash = "sha256:571ac1dc6991c450b25a9c2d84a3705e2ae7a53467b5d111c24fa8baabbed320" }, +] + +[[package]] +name = "annotated-types" +version = "0.7.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ee/67/531ea369ba64dcff5ec9c3402f9f51bf748cec26dde048a2f973a4eea7f5/annotated_types-0.7.0.tar.gz", hash = "sha256:aff07c09a53a08bc8cfccb9c85b05f1aa9a2a6f23728d790723543408344ce89" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/78/b6/6307fbef88d9b5ee7421e68d78a9f162e0da4900bc5f5793f6d3d0e34fb8/annotated_types-0.7.0-py3-none-any.whl", hash = "sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53" }, +] + +[[package]] +name = "anyio" +version = "4.14.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "idna" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/1c/b5/001890774a9552aff22502b8da382593109ce0c95314abaebbb116567545/anyio-4.14.0.tar.gz", hash = "sha256:b47c1f9ccf73e67021df785332508f99379c68fa7d0684e8e3492cb1d4b23f89" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ba/16/9826f089383c593cdfc4a6e5aca94d9e91ae1692c57af82c3b2aa5e810f7/anyio-4.14.0-py3-none-any.whl", hash = "sha256:dd9b7a2a9799ed6552fde617b2c5df02b7fdd7d88392fc48101e51bae46164d9" }, +] + +[[package]] +name = "apscheduler" +version = "3.11.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "tzlocal" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/07/12/3e4389e5920b4c1763390c6d371162f3784f86f85cd6d6c1bfe68eef14e2/apscheduler-3.11.2.tar.gz", hash = "sha256:2a9966b052ec805f020c8c4c3ae6e6a06e24b1bf19f2e11d91d8cca0473eef41" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/9f/64/2e54428beba8d9992aa478bb8f6de9e4ecaa5f8f513bcfd567ed7fb0262d/apscheduler-3.11.2-py3-none-any.whl", hash = "sha256:ce005177f741409db4e4dd40a7431b76feb856b9dd69d57e0da49d6715bfd26d" }, +] + +[[package]] +name = "ast-serialize" +version = "0.5.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/81/9d/09e27731bd5864a9ce04e3244074e674bb8936bf62b45e0357248717adac/ast_serialize-0.5.0.tar.gz", hash = "sha256:5880091bfe6f4f986f22866375c2e884843e7a0b6343ae41aeea659613d879b6" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e0/9e/dc2530acb3a60dc6e46d65abf27d1d9f86721694757906a148d90a6860de/ast_serialize-0.5.0-cp39-abi3-macosx_10_12_x86_64.whl", hash = "sha256:0668aa9459cfa8c9c49ddd2163ebcf43088ba045ef7492af6fe22e0098303101" }, + { url = "https://mirrors.aliyun.com/pypi/packages/26/0a/bd3d18a582f273d6c843d16bb9e22e9e16365ff7991e92f18f798e9f1224/ast_serialize-0.5.0-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:bf683d6363edf2b39eed6b6d4fe22d34b6203867a67e27134d9e2a2680c4bc4a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/40/ae/1f919100f8620887af58fcc381c61a1f218cdf89c6e155f87b213e61010a/ast_serialize-0.5.0-cp39-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9cc22cf0c9be65e71cf88fda130af60d61eb4a79370ad4cfe7900d48a4aa2211" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c6/ca/6376559dcce707cdbc1d0d9a13c8d3baaaa501e949ce0ebdc4230cd881aa/ast_serialize-0.5.0-cp39-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:f66173891548c9f2726bf27957b41cabce12fa679dc6da505ddbde4d4b3b31cf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/35/b2/a620e206b5aeb7efbf2710336df57d457cffbb3991076bbcc1147ef9abd4/ast_serialize-0.5.0-cp39-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:e42d729ef2be96a14efbad355093284739e3670ece3e534f82cc8832790911d9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fa/e0/4ad5c04c24a40481b2935ce9a0ccdb6023dc8b667167d06ae530cc3512f2/ast_serialize-0.5.0-cp39-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:b725026bafa801dbd7310eb13a75f0a2e370e7e51b2cb225f9d21fcfadf919ee" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b2/71/4d1d479aa56d0101c40e17720c3d6ac2af7269ea0487a80b18e7bfd1a5b7/ast_serialize-0.5.0-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:b54f60c1d78767a53b67eaa663f0dfac3afe606aa07f1301572f588b73d64809" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/4f/0de1bbe06f6edef9fde4ed12ca8e7b3ec7e6e2bd4e672c5af487f7957665/ast_serialize-0.5.0-cp39-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:27d51654fc240a1e87e742d353d98eb45b75f62f129086b3596ab53df2ac2a43" }, + { url = "https://mirrors.aliyun.com/pypi/packages/75/61/e00872439cfdddcc3c1b6cdaa6e5d904ba8e26a18807c67c4e14409d0ca8/ast_serialize-0.5.0-cp39-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:2782c36237c46dd1674542f2109740ea5ea485a169bf1431939ada0434e17934" }, + { url = "https://mirrors.aliyun.com/pypi/packages/76/8e/699a5b955f7926956c95e9e1d74132acad73c2fe7a426f94da89123c20aa/ast_serialize-0.5.0-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:1943db345233cc7194a470f13afa9c59772c0b123dea0c9414c4d4ca54369759" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a9/ae/d5b7626874478997adc7a29ab28accf21e596fb590c944290401dfd0b29e/ast_serialize-0.5.0-cp39-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:df1c00022cbbcb064bfaa505aa9c9295362443ce5dacb459d1331d3da353f887" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0c/ce/b59e02a82d9c4244d64cde502e0b00e83e38816abe19155ceb5437402c7f/ast_serialize-0.5.0-cp39-abi3-musllinux_1_2_i686.whl", hash = "sha256:cae65289fc456fde04af979a2be09302ef5d8ab92ef23e596d6746dc267ada27" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8b/38/d8d90042747d05aa08d4efcf1c99035a5f670a6bf4c214d31644392afbca/ast_serialize-0.5.0-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:239a4c354e8d676e9d94631d1d4a64edc6b266f86ff3a5a80aedd344f342c01d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/dd/51/5b840c4df7334104cecffa28f23904fe81ca89ca223d2450e288de39fd3c/ast_serialize-0.5.0-cp39-abi3-win32.whl", hash = "sha256:143a4ef63285a075871908fda3672dc21864b83a8ec3ee12304aa3e4c5387b9a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/41/11/ca5672c7d491825bc4cd6702dea106a6b60d928707712ec257c7833ae476/ast_serialize-0.5.0-cp39-abi3-win_amd64.whl", hash = "sha256:cf25572c526add400f26a4750dc6ce0c3bb93fc1f75e7ae0cad4ce4f2cd5c590" }, + { url = "https://mirrors.aliyun.com/pypi/packages/45/19/cc8bd127d28a43da249aa955cfd164cf8fd534e79e42cea96c4854d72fd0/ast_serialize-0.5.0-cp39-abi3-win_arm64.whl", hash = "sha256:92a31c9c20d25a076edaeec76b128a3535d74a24f340b9a8a7e96c9b86dc9642" }, +] + +[[package]] +name = "attrs" +version = "26.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/9a/8e/82a0fe20a541c03148528be8cac2408564a6c9a0cc7e9171802bc1d26985/attrs-26.1.0.tar.gz", hash = "sha256:d03ceb89cb322a8fd706d4fb91940737b6642aa36998fe130a9bc96c985eff32" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/64/b4/17d4b0b2a2dc85a6df63d1157e028ed19f90d4cd97c36717afef2bc2f395/attrs-26.1.0-py3-none-any.whl", hash = "sha256:c647aa4a12dfbad9333ca4e71fe62ddc36f4e63b2d260a37a8b83d2f043ac309" }, +] + +[[package]] +name = "azure-ai-documentintelligence" +version = "1.0.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "azure-core" }, + { name = "isodate" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/44/7b/8115cd713e2caa5e44def85f2b7ebd02a74ae74d7113ba20bdd41fd6dd80/azure_ai_documentintelligence-1.0.2.tar.gz", hash = "sha256:4d75a2513f2839365ebabc0e0e1772f5601b3a8c9a71e75da12440da13b63484" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d9/75/c9ec040f23082f54ffb1977ff8f364c2d21c79a640a13d1c1809e7fd6b1a/azure_ai_documentintelligence-1.0.2-py3-none-any.whl", hash = "sha256:e1fb446abbdeccc9759d897898a0fe13141ed29f9ad11fc705f951925822ed59" }, +] + +[[package]] +name = "azure-core" +version = "1.41.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "requests" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/a6/f3/b416179e408990df5db0d516283022dde0f5d0111d98c1a848e41853e81c/azure_core-1.41.0.tar.gz", hash = "sha256:f46ff5dfcd230f25cf1c19e8a34b8dc08a337b2503e268bb600a16c00db8ad5a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/5b/db/325c6d7312d2200251c52323878281045aaffcb5586612296484e4280eaa/azure_core-1.41.0-py3-none-any.whl", hash = "sha256:522b4011e8180b1a3dcd2024396a4e7fe9ac37fb8597db47163d230b5efe892d" }, +] + +[[package]] +name = "azure-identity" +version = "1.25.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "azure-core" }, + { name = "cryptography" }, + { name = "msal" }, + { name = "msal-extensions" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/c5/0e/3a63efb48aa4a5ae2cfca61ee152fbcb668092134d3eb8bfda472dd5c617/azure_identity-1.25.3.tar.gz", hash = "sha256:ab23c0d63015f50b630ef6c6cf395e7262f439ce06e5d07a64e874c724f8d9e6" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/49/9a/417b3a533e01953a7c618884df2cb05a71e7b68bdbce4fbdb62349d2a2e8/azure_identity-1.25.3-py3-none-any.whl", hash = "sha256:f4d0b956a8146f30333e071374171f3cfa7bdb8073adb8c3814b65567aa7447c" }, +] + +[[package]] +name = "beautifulsoup4" +version = "4.15.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "soupsieve" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/43/65/318323f98dbee45d42dff61d8f047181bc6f2268a9068cfad035a46be5af/beautifulsoup4-4.15.0.tar.gz", hash = "sha256:288e3ca7d54b06f2ac191970bc275c1939cb46d450b255bf6718b04aa37ab4f7" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/88/c6/92fcd42f1ba33e1184263f25bfabf3d27c383410470f169e4b8163bf9c17/beautifulsoup4-4.15.0-py3-none-any.whl", hash = "sha256:d6f88de62e1d4e38ecb1077eb9724cd0eff29d2a08ca16a401e9b9e93f117cf9" }, +] + +[[package]] +name = "brotli" +version = "1.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/f7/16/c92ca344d646e71a43b8bb353f0a6490d7f6e06210f8554c8f874e454285/brotli-1.2.0.tar.gz", hash = "sha256:e310f77e41941c13340a95976fe66a8a95b01e783d430eeaf7a2f87e0a57dd0a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7a/ef/f285668811a9e1ddb47a18cb0b437d5fc2760d537a2fe8a57875ad6f8448/brotli-1.2.0-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:15b33fe93cedc4caaff8a0bd1eb7e3dab1c61bb22a0bf5bdfdfd97cd7da79744" }, + { url = "https://mirrors.aliyun.com/pypi/packages/50/62/a3b77593587010c789a9d6eaa527c79e0848b7b860402cc64bc0bc28a86c/brotli-1.2.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:898be2be399c221d2671d29eed26b6b2713a02c2119168ed914e7d00ceadb56f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cd/e1/7fadd47f40ce5549dc44493877db40292277db373da5053aff181656e16e/brotli-1.2.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:350c8348f0e76fff0a0fd6c26755d2653863279d086d3aa2c290a6a7251135dd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/12/8b/1ed2f64054a5a008a4ccd2f271dbba7a5fb1a3067a99f5ceadedd4c1d5a7/brotli-1.2.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:2e1ad3fda65ae0d93fec742a128d72e145c9c7a99ee2fcd667785d99eb25a7fe" }, + { url = "https://mirrors.aliyun.com/pypi/packages/89/5a/7071a621eb2d052d64efd5da2ef55ecdac7c3b0c6e4f9d519e9c66d987ef/brotli-1.2.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:40d918bce2b427a0c4ba189df7a006ac0c7277c180aee4617d99e9ccaaf59e6a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/26/6d/0971a8ea435af5156acaaccec1a505f981c9c80227633851f2810abd252a/brotli-1.2.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:2a7f1d03727130fc875448b65b127a9ec5d06d19d0148e7554384229706f9d1b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f3/75/c1baca8b4ec6c96a03ef8230fab2a785e35297632f402ebb1e78a1e39116/brotli-1.2.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:9c79f57faa25d97900bfb119480806d783fba83cd09ee0b33c17623935b05fa3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0d/1a/23fcfee1c324fd48a63d7ebf4bac3a4115bdb1b00e600f80f727d850b1ae/brotli-1.2.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:844a8ceb8483fefafc412f85c14f2aae2fb69567bf2a0de53cdb88b73e7c43ae" }, + { url = "https://mirrors.aliyun.com/pypi/packages/36/e5/12904bbd36afeef53d45a84881a4810ae8810ad7e328a971ebbfd760a0b3/brotli-1.2.0-cp311-cp311-win32.whl", hash = "sha256:aa47441fa3026543513139cb8926a92a8e305ee9c71a6209ef7a97d91640ea03" }, + { url = "https://mirrors.aliyun.com/pypi/packages/02/8b/ecb5761b989629a4758c394b9301607a5880de61ee2ee5fe104b87149ebc/brotli-1.2.0-cp311-cp311-win_amd64.whl", hash = "sha256:022426c9e99fd65d9475dce5c195526f04bb8be8907607e27e747893f6ee3e24" }, +] + +[[package]] +name = "certifi" +version = "2026.5.20" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/f3/ce/ee2ecad540810a79593028e88299baeae54d346cc7a0d94b6199988b89b1/certifi-2026.5.20.tar.gz", hash = "sha256:69dea482ab64caa7b9f6aba1c6bf48bb6a5448d1c0f1b17ab42ad8c763a5344d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/59/8c/57e832b7af6d7c5abe66eb3fbe3a3a32f4d11ea23a1aa7131371035be991/certifi-2026.5.20-py3-none-any.whl", hash = "sha256:3c52e209ba0a4ad7aebe60436a4ab349c39e1e602e8c134221e546902ad25897" }, +] + +[[package]] +name = "cffi" +version = "2.0.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "pycparser", marker = "implementation_name != 'PyPy'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/eb/56/b1ba7935a17738ae8453301356628e8147c79dbb825bcbc73dc7401f9846/cffi-2.0.0.tar.gz", hash = "sha256:44d1b5909021139fe36001ae048dbdde8214afa20200eda0f64c068cac5d5529" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/12/4a/3dfd5f7850cbf0d06dc84ba9aa00db766b52ca38d8b86e3a38314d52498c/cffi-2.0.0-cp311-cp311-macosx_10_13_x86_64.whl", hash = "sha256:b4c854ef3adc177950a8dfc81a86f5115d2abd545751a304c5bcf2c2c7283cfe" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4f/8b/f0e4c441227ba756aafbe78f117485b25bb26b1c059d01f137fa6d14896b/cffi-2.0.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:2de9a304e27f7596cd03d16f1b7c72219bd944e99cc52b84d0145aefb07cbd3c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b1/b7/1200d354378ef52ec227395d95c2576330fd22a869f7a70e88e1447eb234/cffi-2.0.0-cp311-cp311-manylinux1_i686.manylinux2014_i686.manylinux_2_17_i686.manylinux_2_5_i686.whl", hash = "sha256:baf5215e0ab74c16e2dd324e8ec067ef59e41125d3eade2b863d294fd5035c92" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b8/56/6033f5e86e8cc9bb629f0077ba71679508bdf54a9a5e112a3c0b91870332/cffi-2.0.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:730cacb21e1bdff3ce90babf007d0a0917cc3e6492f336c2f0134101e0944f93" }, + { url = "https://mirrors.aliyun.com/pypi/packages/dc/7f/55fecd70f7ece178db2f26128ec41430d8720f2d12ca97bf8f0a628207d5/cffi-2.0.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:6824f87845e3396029f3820c206e459ccc91760e8fa24422f8b0c3d1731cbec5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/84/ef/a7b77c8bdc0f77adc3b46888f1ad54be8f3b7821697a7b89126e829e676a/cffi-2.0.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.whl", hash = "sha256:9de40a7b0323d889cf8d23d1ef214f565ab154443c42737dfe52ff82cf857664" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d7/91/500d892b2bf36529a75b77958edfcd5ad8e2ce4064ce2ecfeab2125d72d1/cffi-2.0.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:8941aaadaf67246224cee8c3803777eed332a19d909b47e29c9842ef1e79ac26" }, + { url = "https://mirrors.aliyun.com/pypi/packages/44/64/58f6255b62b101093d5df22dcb752596066c7e89dd725e0afaed242a61be/cffi-2.0.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:a05d0c237b3349096d3981b727493e22147f934b20f6f125a3eba8f994bec4a9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ab/49/fa72cebe2fd8a55fbe14956f9970fe8eb1ac59e5df042f603ef7c8ba0adc/cffi-2.0.0-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:94698a9c5f91f9d138526b48fe26a199609544591f859c870d477351dc7b2414" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0b/28/dd0967a76aab36731b6ebfe64dec4e981aff7e0608f60c2d46b46982607d/cffi-2.0.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:5fed36fccc0612a53f1d4d9a816b50a36702c28a2aa880cb8a122b3466638743" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2b/c0/015b25184413d7ab0a410775fdb4a50fca20f5589b5dab1dbbfa3baad8ce/cffi-2.0.0-cp311-cp311-win32.whl", hash = "sha256:c649e3a33450ec82378822b3dad03cc228b8f5963c0c12fc3b1e0ab940f768a5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ae/8f/dc5531155e7070361eb1b7e4c1a9d896d0cb21c49f807a6c03fd63fc877e/cffi-2.0.0-cp311-cp311-win_amd64.whl", hash = "sha256:66f011380d0e49ed280c789fbd08ff0d40968ee7b665575489afa95c98196ab5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/95/5c/1b493356429f9aecfd56bc171285a4c4ac8697f76e9bbbbb105e537853a1/cffi-2.0.0-cp311-cp311-win_arm64.whl", hash = "sha256:c6638687455baf640e37344fe26d37c404db8b80d037c3d29f58fe8d1c3b194d" }, +] + +[[package]] +name = "chardet" +version = "7.4.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/19/b6/9df434a8eeba2e6628c465a1dfa31034228ef79b26f76f46278f4ef7e49d/chardet-7.4.3.tar.gz", hash = "sha256:cc1d4eb92a4ec1c2df3b490836ffa46922e599d34ce0bb75cf41fd2bf6303d56" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/19/52/505c207f334d51e937cbaa27ff95776e16e2d120e13cbe491cd7b3a70b50/chardet-7.4.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:25a862cddc6a9ac07023e808aedd297115345fbaabc2690479481ddc0f980e09" }, + { url = "https://mirrors.aliyun.com/pypi/packages/14/4b/d3c79495dee4831b8bebca2790e72cb90f0c5849c940570a7c7e5b70b952/chardet-7.4.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:7005c88da26fd95d8abb8acbe6281d833e9a9181b03cf49b4546c4555389bd97" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b9/99/f6a822ad1bde25a4c38dc3e770485e78e0893dfd871cd6e18ed3ea3a795e/chardet-7.4.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:dc50f28bad067393cce0af9091052c3b8df7a23115afd8ba7b2e0947f0cef1f8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b1/10/31932775c94a86814f76b41c4a772b52abfb0e6125324f32c6da1196c297/chardet-7.4.3-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4c3da294de1a681097848ab58bd3f2771a674f8039d2d87a5538b28856b815e9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6c/63/0f43e3acf2c436fdb32a0f904aeb03a2904d2126eed34a042a194d235926/chardet-7.4.3-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:93c45e116dd51b66226a53ade3f9f635e870de5399b90e00ce45dcc311093bf4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5d/a6/e9b8f8a3e99602792b01fa7d0a731737615ab56d8bfd0b52935a0ef88b85/chardet-7.4.3-cp311-cp311-win_amd64.whl", hash = "sha256:ccc1f83ab4bcfb901cf39e0c4ba6bc6e726fc6264735f10e24ceb5cb47387578" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8c/6c/0a40afdb50a0fe041ab95553b835a8160b6cf0e81edf2ae2fe9f5224cbf9/chardet-7.4.3-py3-none-any.whl", hash = "sha256:1173b74051570cf08099d7429d92e4882d375ad4217f92a6e5240ccfb26f231e" }, +] + +[[package]] +name = "charset-normalizer" +version = "3.4.7" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e7/a1/67fe25fac3c7642725500a3f6cfe5821ad557c3abb11c9d20d12c7008d3e/charset_normalizer-3.4.7.tar.gz", hash = "sha256:ae89db9e5f98a11a4bf50407d4363e7b09b31e55bc117b4f7d80aab97ba009e5" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c2/d7/b5b7020a0565c2e9fa8c09f4b5fa6232feb326b8c20081ccded47ea368fd/charset_normalizer-3.4.7-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:7641bb8895e77f921102f72833904dcd9901df5d6d72a2ab8f31d04b7e51e4e7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5a/53/58c29116c340e5456724ecd2fff4196d236b98f3da97b404bc5e51ac3493/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:202389074300232baeb53ae2569a60901f7efadd4245cf3a3bf0617d60b439d7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b2/02/e8146dc6591a37a00e5144c63f29fb7c97a734ea8a111190783c0e60ab63/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:30b8d1d8c52a48c2c5690e152c169b673487a2a58de1ec7393196753063fcd5e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fb/73/77486c4cd58f1267bf17db420e930c9afa1b3be3fe8c8b8ebbebc9624359/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:532bc9bf33a68613fd7d65e4b1c71a6a38d7d42604ecf239c77392e9b4e8998c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a1/fa/f74eb381a7d94ded44739e9d94de18dc5edc9c17fb8c11f0a6890696c0a9/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2fe249cb4651fd12605b7288b24751d8bfd46d35f12a20b1ba33dea122e690df" }, + { url = "https://mirrors.aliyun.com/pypi/packages/dc/92/42bd3cefcf7687253fb86694b45f37b733c97f59af3724f356fa92b8c344/charset_normalizer-3.4.7-cp311-cp311-manylinux_2_31_armv7l.whl", hash = "sha256:65bcd23054beab4d166035cabbc868a09c1a49d1efe458fe8e4361215df40265" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4c/3d/069e7184e2aa3b3cddc700e3dd267413dc259854adc3380421c805c6a17d/charset_normalizer-3.4.7-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:08e721811161356f97b4059a9ba7bafb23ea5ee2255402c42881c214e173c6b4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/62/51/9d56feb5f2e7074c46f93e0ebdbe61f0848ee246e2f0d89f8e20b89ebb8f/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:e060d01aec0a910bdccb8be71faf34e7799ce36950f8294c8bf612cba65a2c9e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d2/59/893d8f99cc4c837dda1fe2f1139079703deb9f321aabcb032355de13b6c7/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:38c0109396c4cfc574d502df99742a45c72c08eff0a36158b6f04000043dbf38" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7d/1d/ee6f3be3464247578d1ed5c46de545ccc3d3ff933695395c402c21fa6b77/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:1c2a768fdd44ee4a9339a9b0b130049139b8ce3c01d2ce09f67f5a68048d477c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/54/bb/8fb0a946296ea96a488928bdce8ef99023998c48e4713af533e9bb98ef07/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:1a87ca9d5df6fe460483d9a5bbf2b18f620cbed41b432e2bddb686228282d10b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9a/bc/015b2387f913749f82afd4fcba07846d05b6d784dd16123cb66860e0237d/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:d635aab80466bc95771bb78d5370e74d36d1fe31467b6b29b8b57b2a3cd7d22c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/17/ab/63133691f56baae417493cba6b7c641571a2130eb7bceba6773367ab9ec5/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:ae196f021b5e7c78e918242d217db021ed2a6ace2bc6ae94c0fc596221c7f58d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/06/6d/3be70e827977f20db77c12a97e6a9f973631a45b8d186c084527e53e77a4/charset_normalizer-3.4.7-cp311-cp311-win32.whl", hash = "sha256:adb2597b428735679446b46c8badf467b4ca5f5056aae4d51a19f9570301b1ad" }, + { url = "https://mirrors.aliyun.com/pypi/packages/20/d9/5f67790f06b735d7c7637171bbfd89882ad67201891b7275e51116ed8207/charset_normalizer-3.4.7-cp311-cp311-win_amd64.whl", hash = "sha256:8e385e4267ab76874ae30db04c627faaaf0b509e1ccc11a95b3fc3e83f855c00" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ca/83/6413f36c5a34afead88ce6f66684d943d91f233d76dd083798f9602b75ae/charset_normalizer-3.4.7-cp311-cp311-win_arm64.whl", hash = "sha256:d4a48e5b3c2a489fae013b7589308a40146ee081f6f509e047e0e096084ceca1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/db/8f/61959034484a4a7c527811f4721e75d02d653a35afb0b6054474d8185d4c/charset_normalizer-3.4.7-py3-none-any.whl", hash = "sha256:3dce51d0f5e7951f8bb4900c257dad282f49190fdbebecd4ba99bcc41fef404d" }, +] + +[[package]] +name = "click" +version = "8.4.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/9b/98/518d8e5081007684232226f475082b30087d0f585e8457db087298259f49/click-8.4.1.tar.gz", hash = "sha256:918b5633eddf6b41c32d4f454bf0de810065c74e3f7dbf8ee5452f8be88d3e96" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c7/0d/67e5b4109ea4a837e80daa87c2c696711955e40449a97e8926672534def2/click-8.4.1-py3-none-any.whl", hash = "sha256:482be17c6991b8c19c5429a1e995d9b0efdbb63172824c41f99965dc0ade8ec2" }, +] + +[[package]] +name = "click-log" +version = "0.4.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "click" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/32/32/228be4f971e4bd556c33d52a22682bfe318ffe57a1ddb7a546f347a90260/click-log-0.4.0.tar.gz", hash = "sha256:3970f8570ac54491237bcdb3d8ab5e3eef6c057df29f8c3d1151a51a9c23b975" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ae/5a/4f025bc751087833686892e17e7564828e409c43b632878afeae554870cd/click_log-0.4.0-py2.py3-none-any.whl", hash = "sha256:a43e394b528d52112af599f2fc9e4b7cf3c15f94e53581f74fa6867e68c91756" }, +] + +[[package]] +name = "cobble" +version = "0.1.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/54/7a/a507c709be2c96e1bb6102eb7b7f4026c5e5e223ef7d745a17d239e9d844/cobble-0.1.4.tar.gz", hash = "sha256:de38be1539992c8a06e569630717c485a5f91be2192c461ea2b220607dfa78aa" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d5/e1/3714a2f371985215c219c2a70953d38e3eed81ef165aed061d21de0e998b/cobble-0.1.4-py3-none-any.whl", hash = "sha256:36c91b1655e599fd428e2b95fdd5f0da1ca2e9f1abb0bc871dec21a0e78a2b44" }, +] + +[[package]] +name = "colorama" +version = "0.4.6" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d8/53/6f443c9a4a8358a93a6792e2acffb9d9d5cb0a5cfd8802644b7b1c9a02e4/colorama-0.4.6.tar.gz", hash = "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6" }, +] + +[[package]] +name = "coloredlogs" +version = "15.0.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "humanfriendly" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/cc/c7/eed8f27100517e8c0e6b923d5f0845d0cb99763da6fdee00478f91db7325/coloredlogs-15.0.1.tar.gz", hash = "sha256:7c991aa71a4577af2f82600d8f8f3a89f936baeaf9b50a9c197da014e5bf16b0" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a7/06/3d6badcf13db419e25b07041d9c7b4a2c331d3f4e7134445ec5df57714cd/coloredlogs-15.0.1-py2.py3-none-any.whl", hash = "sha256:612ee75c546f53e92e70049c9dbfcc18c935a2b9a53b66085ce9ef6a6e5c0934" }, +] + +[[package]] +name = "coverage" +version = "7.14.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/54/fd/0ab2772530e946e1be1abd0bc09e647ec9b02e88f0867857601fefca8953/coverage-7.14.1.tar.gz", hash = "sha256:30c08f7d90415aa98b3c990385dea2939b0da55f38515e5b369b83655f8523be" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7d/d7/477ad149490e6cb849f28abea1dabb9c823cea72e7500c81b4240ce619c0/coverage-7.14.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:478b5bcd63c2e1357c5c7e16c070690df7b07f676b1c114d7b93e533c664309f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/91/82/a5eb47257c50601bb7b9a9d2857c67b7a3a85ad74180eb2c98bb1fbe0ce5/coverage-7.14.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:a24a81f9715ee42ef59a316cc11611c98fe23920f7c81861315c9f3ff4a230f4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/43/8b/78419b5391a5cb706b6544390507e469d83ffc9a8248b02c4011aceb9365/coverage-7.14.1-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:196a13319ad88d6d8ef5ab489ec4f44ddde2143c0c7d5b27786f6c3ffd56a7e1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/77/63/e77aaacd491182210d639636b7a8bba23ffffa9b82aa3762da9431855fa9/coverage-7.14.1-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:3d452fd08b5c72c5167c93e6867b5c08500bd40f2a21e1e854a500550b6cc36f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/65/1c/a022e3cfbec2ac241640003cb3a817e161d9c7f5aa9b49173756cdc03204/coverage-7.14.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:23bf7fa51ac02e07fc7c96849b82946da47ae862dc8f86d183b2a4864fc38129" }, + { url = "https://mirrors.aliyun.com/pypi/packages/61/d6/967e408aca4c1ceb88cb0cc677169110ae7f5995fb5eaf5fb1f5a1bb8f5d/coverage-7.14.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:bcaa50684dcaadfa599ac48f81103c756d791cfd85c97203d2217c593d48b860" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b8/be/869188f7fe28638078ec479331ace6dc5f7b40b7153eb616f47ab79404d8/coverage-7.14.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:4ea1c034f95c9b056e856b794630b17f9fa3d57e4800ff1e503d3be0f9c9078c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/07/aa/adb7d3b4278d690e68703abcd76ab1b948242e3668d921711551b78f9ddb/coverage-7.14.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:c7e057326434e441306226fbeb5d1aaf14a2637efe97ba668306635835f32ad7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/43/61/331c74103c62dcb0c4b9b3a0de9a61aca016208b0a90f109592a9f9ecc28/coverage-7.14.1-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:59baf88468dbc8d63b1887afd92bda52e40bb1561696e5819670601403810cec" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f6/b6/c5dae3c104d89be04828f61810e6b3473825482e4c288cc4ed04553e08ae/coverage-7.14.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:d34d75f892b3ab73ba11cab5442cce7b3e168fd64162b16f0e1e0d09c508edef" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ad/a1/2b9d5863e3b83c01ad8199e3c597802fbb3a9dc90b058885804c20296d31/coverage-7.14.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:3a56abc20a472baf0304c455721bc601477440d28ecfde8a03dde79ede07e0df" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7f/5e/0e511fbdb269359be26fe678a1c3fa1f2aa2a01573cc3f54268c8d6d4797/coverage-7.14.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:6a3cb83d1552c0cd1b4906655b6a33fd4a8473229633a901c6b73bf86914dee9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/85/10/e55307b622b3dd9671cb321824502dc10f93e72f2802b9946159a8edadeb/coverage-7.14.1-cp311-cp311-win32.whl", hash = "sha256:10274a1fbeb8ec5d72966e17bb198a3104257aca4ac09d98667c5f8aca8c8548" }, + { url = "https://mirrors.aliyun.com/pypi/packages/71/cf/107421693cfb71e4f1ca5bf70443f64d4161878068d07a3e51c7ad21d17b/coverage-7.14.1-cp311-cp311-win_amd64.whl", hash = "sha256:87ebdf787d4888e3f3f2d523eadc6e18c6d18c6d0eb173801a189641627fb37e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b8/1d/3e3644585eb29e9dafefb19555078529a4d7cce12bd21929664eea989277/coverage-7.14.1-cp311-cp311-win_arm64.whl", hash = "sha256:dd34767fa19848d35659ffc0a75314f58c7af3f1cd87ec521e8292a1238398a3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8a/3c/1a983b9a745d7f83d53f057bcc5bf79ba6a2bbc08266b3f0c7d6fe630c9b/coverage-7.14.1-py3-none-any.whl", hash = "sha256:a252f21c27e38347e60111a3266b03827422a7d5525951aceee313aa68bab1d2" }, +] + +[package.optional-dependencies] +toml = [ + { name = "tomli", marker = "python_full_version <= '3.11'" }, +] + +[[package]] +name = "crawl4ai" +version = "0.7.8" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "aiofiles" }, + { name = "aiohttp" }, + { name = "aiosqlite" }, + { name = "alphashape" }, + { name = "anyio" }, + { name = "beautifulsoup4" }, + { name = "brotli" }, + { name = "chardet" }, + { name = "click" }, + { name = "cssselect" }, + { name = "fake-useragent" }, + { name = "httpx", extra = ["http2"] }, + { name = "humanize" }, + { name = "lark" }, + { name = "litellm" }, + { name = "lxml" }, + { name = "nltk" }, + { name = "numpy" }, + { name = "patchright" }, + { name = "pillow" }, + { name = "playwright" }, + { name = "psutil" }, + { name = "pydantic" }, + { name = "pyopenssl" }, + { name = "python-dotenv" }, + { name = "pyyaml" }, + { name = "rank-bm25" }, + { name = "requests" }, + { name = "rich" }, + { name = "shapely" }, + { name = "snowballstemmer" }, + { name = "tf-playwright-stealth" }, + { name = "xxhash" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d6/8f/08133fcf6f4ae41aff02af8c5353af831e81bf925bc667ef2f8653abd7d2/crawl4ai-0.7.8.tar.gz", hash = "sha256:ed0189ca19ccc1dad349a37565d9dcbfed7897650e44d7c6ae25a37eef5ab44b" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/6b/b3/d2cc38bf9a3a285de5d2da6057e0a4809e41cd4344e2e88a00de19b51c7a/crawl4ai-0.7.8-py3-none-any.whl", hash = "sha256:a25a108e3e027022e98a6ab447d8261676c5e83717fa712502718ac47456b1dd" }, +] + +[[package]] +name = "cryptography" +version = "49.0.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "cffi", marker = "platform_python_implementation != 'PyPy'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/1f/99/d1c90d6041656cc6ee229dc99cd67fd0cd5aec3c5f7d72fffc27cc750054/cryptography-49.0.0.tar.gz", hash = "sha256:f89660a348f4f78a92366240a61404e337586ef7f5909a2fef59ca88ef505493" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/9b/22/adf66990e63584a68dfb50c24f48a125c07b1699899381c8151e63ed458c/cryptography-49.0.0-cp311-abi3-macosx_11_0_arm64.whl", hash = "sha256:966fe0e9c67490071f14c0d2b1cb2dfb3023c5ce39457343931415f08382f2db" }, + { url = "https://mirrors.aliyun.com/pypi/packages/09/41/3797cfaf69cae04a13ee78ebd83f0678d9c02b4779d21ce24445326f1a69/cryptography-49.0.0-cp311-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:36d1709f992593689b45bda411498d62c6e365f2ca00b84657d4dadd24de16db" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e6/8b/43011f7ebe515a8aa20d61f290a326cd890c2e738e16e59eaff8d9c3a412/cryptography-49.0.0-cp311-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:0e959b578856a3924bc0cbb710fc12c387b9412a951389f3ca61704a9e25f325" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4a/91/01ce7303a4579e6d3a6abef01bd322848e9ea7a219adcabc5048b9033571/cryptography-49.0.0-cp311-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:53ecee2e23f7169b6117e99fc8a944e5e50f79e69758a83b52a00cb98ab2b2d2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/62/99/a2c95cf8293f07491e9e27c20cc4dcd18176d944e674679adeb1d0173fd6/cryptography-49.0.0-cp311-abi3-manylinux_2_28_ppc64le.whl", hash = "sha256:2eda353d8a27bcbcaa4cbed18994a74ab4d19a2ca897db188ea269ab9b71419b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/20/2c/0622f20ff02b2ef32558733443805dc82fd4c275be01b2d19d14676f3a1b/cryptography-49.0.0-cp311-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:2afe9051da7ae7bd5905da5a949280c7d2bb75682e188f650a9d0f2756b834c6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a3/5b/c5246635d5fd3b64e0d45ae10e99fd32fe9676a79915ccfe5a61ba9af1a5/cryptography-49.0.0-cp311-abi3-manylinux_2_31_armv7l.whl", hash = "sha256:0b82e28ee398a386f0807bba7884d30f25218855690f45115831bcce5d90822c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/88/05563c7fe2e914e87d1a536d06fe83e66b4e1d95cb593e05aea375531da8/cryptography-49.0.0-cp311-abi3-manylinux_2_34_aarch64.whl", hash = "sha256:ccac2bfebc306b862133e3bb71f3f6ee8bb525240089b2d952e4144b3a6d5da7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c4/b6/d7696e4e890d6ae1469935164c9e5215c557671cb78d6e3f458ccceaa632/cryptography-49.0.0-cp311-abi3-manylinux_2_34_ppc64le.whl", hash = "sha256:d0527ce944105f257f605a827d6ebead966c752038b6e8656abb9c5edee6fc68" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a9/3c/f3ad17eecc1a57b0ba236dc01f90e783c51f4a2f35f64777cc4f47a184b2/cryptography-49.0.0-cp311-abi3-manylinux_2_34_x86_64.whl", hash = "sha256:cbc77da8c523d5abd028635ba850a6966fcee2c82e2bf65a41d1d8afe0f98be9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4f/01/339573cf1023163a400b0b5d16f6d507de413b9f60be6fd1b77feeaf6737/cryptography-49.0.0-cp311-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:b87e65d263b3e5d3bb92a57e2a6638e2f31110fa7aa890c7b2dbba42248d0a3f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/71/fd/577302e213a1be9468f92d1afef66fcf1ef83d516819d9992ca547f592bd/cryptography-49.0.0-cp311-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:66ec79c3904820572d7e987abdf304281f141d37ad9a489b8e97066e7b9b6459" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1f/09/f42b1d190c5ba75f72062a387f8030d1d75f6ab035788f1d9c4b01de6525/cryptography-49.0.0-cp311-abi3-win_amd64.whl", hash = "sha256:e5dfc1e64de5677cec922ffa8da89c546d0415bf6efdf081842e5d44c84e1f0e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/19/2a/5bb823f5bedcf80718cea7fbc95ec5515cca3769633c4b01a32be7f30e7c/cryptography-49.0.0-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:ec5e529fb80935c94fe7b729f9972b50e351a0e6b50aa294fd5cabb109fcc29a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3d/df/40577043ca124e17012f408ddddaeb213b856336ac82ddb3bc915f39e29f/cryptography-49.0.0-cp39-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:f78ff2c9ed8dc2d036b0f4d640e22522213d047c1b14e61205a7e55c80a494d4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2c/99/2d13299eb3dd27b02dcfaafcc91d6b5cb3329f7cbd6d8f51921acd566c1a/cryptography-49.0.0-cp39-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:35b151772baff2c74cba7fa290ceaff4c3b11c0c881eb93eb5dbc05a7cfbba18" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a5/4d/9c0cd02f95e2602dd5e563da149ee0830abef3537be8b34dc56281ebe27a/cryptography-49.0.0-cp39-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:0f21641cf4b30fca7aee061ced0ec7ad7b073518088b7c9969a297c0ae796c69" }, + { url = "https://mirrors.aliyun.com/pypi/packages/24/01/186c825898477d77e2324d5360fefe622ff1d8d1963ec0554e2cada8ec77/cryptography-49.0.0-cp39-abi3-manylinux_2_28_ppc64le.whl", hash = "sha256:9e82dcc8e56052715fb18b2429e3bca4823b1629136a2084fc45a9a5cecb9b64" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b8/7b/62cbbab75d0659865bf0273790031544a0b16c8072d258f9428dcd8190dc/cryptography-49.0.0-cp39-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:6f2debedf9ca60cf1d5bd466475638af5130f89965605cd818484d19987d3a21" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6c/72/3e798c064bc39e471008075d0f9bc9daf77a80879c092e4a8e170c585ed4/cryptography-49.0.0-cp39-abi3-manylinux_2_31_armv7l.whl", hash = "sha256:8c25ceb16df5b9435f3f6a9829204985b0e0cbee3b48aacd432c7d2c850b44d9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f0/ee/6fca21d1ac73e06f8bef71940abfd4d2f6472b4bca284d770f32bd4086f6/cryptography-49.0.0-cp39-abi3-manylinux_2_34_aarch64.whl", hash = "sha256:28d8b15e6275f12c8a207dc309dfa957903c927d08d0cc937ee3f63f200693cc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/67/d0/a5fcd3515f0bae49a7b6d0413cc1bdccdcc1fc0047037a0d480642cdc5d6/cryptography-49.0.0-cp39-abi3-manylinux_2_34_ppc64le.whl", hash = "sha256:6fc361c34fb6aac015ce19435876635e5c6d21db31998b0920f675f131e043b8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a0/84/84fe36f19caf857d61cb7fc9c63035a47ffabd84ea12d1d393148efa3615/cryptography-49.0.0-cp39-abi3-manylinux_2_34_x86_64.whl", hash = "sha256:2400ef9c9e2299a25614eb1dea3db54a69b1349efd043bfac9c67630d136df36" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6c/a0/db537264e234f7273a73ec020873d6d6b39dfd8a53db78b550ca8320440e/cryptography-49.0.0-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:67e1d20ad9ef3a563c59ef22e7a8a0b8210bd26604369ea4a30a7c66aefe504e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/93/77/8df9eb486495979bccecd1062e2eaf435250e84437040295b57d09048b0b/cryptography-49.0.0-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:42b0684e0e40cf26122427802486f6d93aea593612603a94fbf260c7eb1e9c1b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c2/e6/f60198ea8d9dfa15fff9ed4ca02ce362f6eadd9ba757dcc50634c4257b63/cryptography-49.0.0-cp39-abi3-win_amd64.whl", hash = "sha256:026ac7423e6fa66872d3bf889be5974507da3944f866f704fa200eadacd00001" }, + { url = "https://mirrors.aliyun.com/pypi/packages/63/d3/4a83af35d65e3fad632c926fad684c193ea4398569ccb0bbbc7fe8f5dc9a/cryptography-49.0.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:fc1e275c2f1d97b1a6450b8b0ea3ebfa6e087a611c2b26cb2404d48588abab7b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d6/a7/f9dac0ab7f80368c56993a7bf638ef9935f825c91902798481fac0898138/cryptography-49.0.0-pp311-pypy311_pp73-manylinux_2_28_aarch64.whl", hash = "sha256:c83782480a4a9da4d0feb51950131ba32e12e70813848b3343f6e18c28a66838" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d7/70/2ba3769dd0ae167e2f33dfa9592d45db6ff9a61d62ca1a5b3d1bdd09068f/cryptography-49.0.0-pp311-pypy311_pp73-manylinux_2_28_x86_64.whl", hash = "sha256:b39efa323140595abd3ecca8529d321ae50f55f3aa3ba9cc81ea56a6011953d5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/94/64/2923570ac1c0bd3a737aa366ac3abbbbde273042308b8cde95e2364a6e6a/cryptography-49.0.0-pp311-pypy311_pp73-manylinux_2_34_aarch64.whl", hash = "sha256:b47db11c2c3525083296069b98ac5221907455e989ae0c2e3008bde851921615" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ab/f8/614dc7e051418cfe53d55173c1e24c6b0085e89996fe90508c2fdf769aef/cryptography-49.0.0-pp311-pypy311_pp73-manylinux_2_34_x86_64.whl", hash = "sha256:084ef1af862eb07ec46d25f68689f2102a9fc0e05ce7b80f14f5fe51e4eef0f6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/aa/50/a9caea39ad19c431c1a3f8a31114df65b260cdfe67786b6c7e7c040c4c44/cryptography-49.0.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:be9fcb48a55f023493482827d4f459bd263cc20efde64f204b97c123201850c6" }, +] + +[[package]] +name = "cssselect" +version = "1.4.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ec/2e/cdfd8b01c37cbf4f9482eefd455853a3cf9c995029a46acd31dfaa9c1dd6/cssselect-1.4.0.tar.gz", hash = "sha256:fdaf0a1425e17dfe8c5cf66191d211b357cf7872ae8afc4c6762ddd8ac47fc92" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/20/0c/7bb51e3acfafd16c48875bf3db03607674df16f5b6ef8d056586af7e2b8b/cssselect-1.4.0-py3-none-any.whl", hash = "sha256:c0ec5c0191c8ee39fcc8afc1540331d8b55b0183478c50e9c8a79d44dbceb1d8" }, +] + +[[package]] +name = "cuda-bindings" +version = "13.3.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "cuda-pathfinder", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/51/6b/457ca12dad3ee9bfcc9a545cfd6b64b359ba49de40f776f6e028e678f262/cuda_bindings-13.3.1-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c5879712accf6e14bb01aa5e67440eb84998b8d104b509cc7a6dc0b8f656a474" }, + { url = "https://mirrors.aliyun.com/pypi/packages/95/7a/c5e3c34a409b148f5c0f5a4ea374158f95d488862c1dffedf9aa5c639df9/cuda_bindings-13.3.1-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:04436a9364059c84b8f9636f359eccda1cf814341f5b670c71d80d2f79dbc708" }, +] + +[[package]] +name = "cuda-pathfinder" +version = "1.5.5" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/11/c8/26f2e4aae92f11522a96043892ba39a90eac610d5242523aa863212bc1c7/cuda_pathfinder-1.5.5-py3-none-any.whl", hash = "sha256:0228c023f95d1480f143ef5c8922d27a2ab052087a942e81dc289c9eb8f91689" }, +] + +[[package]] +name = "cuda-toolkit" +version = "13.0.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/57/b2/453099f5f3b698d7d0eab38916aac44c7f76229f451709e2eb9db6615dcd/cuda_toolkit-13.0.2-py2.py3-none-any.whl", hash = "sha256:b198824cf2f54003f50d64ada3a0f184b42ca0846c1c94192fa269ecd97a66eb" }, +] + +[package.optional-dependencies] +cudart = [ + { name = "nvidia-cuda-runtime", marker = "sys_platform == 'linux'" }, +] +cufft = [ + { name = "nvidia-cufft", marker = "sys_platform == 'linux'" }, +] +cufile = [ + { name = "nvidia-cufile", marker = "sys_platform == 'linux'" }, +] +cupti = [ + { name = "nvidia-cuda-cupti", marker = "sys_platform == 'linux'" }, +] +curand = [ + { name = "nvidia-curand", marker = "sys_platform == 'linux'" }, +] +cusolver = [ + { name = "nvidia-cusolver", marker = "sys_platform == 'linux'" }, +] +cusparse = [ + { name = "nvidia-cusparse", marker = "sys_platform == 'linux'" }, +] +nvjitlink = [ + { name = "nvidia-nvjitlink", marker = "sys_platform == 'linux'" }, +] +nvrtc = [ + { name = "nvidia-cuda-nvrtc", marker = "sys_platform == 'linux'" }, +] +nvtx = [ + { name = "nvidia-nvtx", marker = "sys_platform == 'linux'" }, +] + +[[package]] +name = "defusedxml" +version = "0.7.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/0f/d5/c66da9b79e5bdb124974bfe172b4daf3c984ebd9c2a06e2b8a4dc7331c72/defusedxml-0.7.1.tar.gz", hash = "sha256:1bb3032db185915b62d7c6209c5a8792be6a32ab2fedacc84e01b52c51aa3e69" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/07/6c/aa3f2f849e01cb6a001cd8554a88d4c77c5c1a31c95bdf1cf9301e6d9ef4/defusedxml-0.7.1-py2.py3-none-any.whl", hash = "sha256:a352e7e428770286cc899e2542b6cdaedb2b4953ff269a210103ec58f6198a61" }, +] + +[[package]] +name = "distro" +version = "1.9.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/fc/f8/98eea607f65de6527f8a2e8885fc8015d3e6f5775df186e443e0964a11c3/distro-1.9.0.tar.gz", hash = "sha256:2fa77c6fd8940f116ee1d6b94a2f90b13b5ea8d019b98bc8bafdcabcdd9bdbed" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/12/b3/231ffd4ab1fc9d679809f356cebee130ac7daa00d6d6f3206dd4fd137e9e/distro-1.9.0-py3-none-any.whl", hash = "sha256:7bffd925d65168f85027d8da9af6bddab658135b840670a223589bc0c8ef02b2" }, +] + +[[package]] +name = "et-xmlfile" +version = "2.0.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d3/38/af70d7ab1ae9d4da450eeec1fa3918940a5fafb9055e934af8d6eb0c2313/et_xmlfile-2.0.0.tar.gz", hash = "sha256:dab3f4764309081ce75662649be815c4c9081e88f0837825f90fd28317d4da54" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c1/8b/5fe2cc11fee489817272089c4203e679c63b570a5aaeb18d852ae3cbba6a/et_xmlfile-2.0.0-py3-none-any.whl", hash = "sha256:7a91720bc756843502c3b7504c77b8fe44217c85c537d85037f0f536151b2caa" }, +] + +[[package]] +name = "fake-http-header" +version = "0.3.5" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e3/0b/2849c87d9f13766e29c0a2f4d31681aa72e035016b251ab19d99bde7b592/fake_http_header-0.3.5-py3-none-any.whl", hash = "sha256:cd05f4bebf1b7e38b5f5c03d7fb820c0c17e87d9614fbee0afa39c32c7a2ad3c" }, +] + +[[package]] +name = "fake-useragent" +version = "2.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/41/43/948d10bf42735709edb5ae51e23297d034086f17fc7279fef385a7acb473/fake_useragent-2.2.0.tar.gz", hash = "sha256:4e6ab6571e40cc086d788523cf9e018f618d07f9050f822ff409a4dfe17c16b2" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/51/37/b3ea9cd5558ff4cb51957caca2193981c6b0ff30bd0d2630ac62505d99d0/fake_useragent-2.2.0-py3-none-any.whl", hash = "sha256:67f35ca4d847b0d298187443aaf020413746e56acd985a611908c73dba2daa24" }, +] + +[[package]] +name = "fastuuid" +version = "0.14.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/c3/7d/d9daedf0f2ebcacd20d599928f8913e9d2aea1d56d2d355a93bfa2b611d7/fastuuid-0.14.0.tar.gz", hash = "sha256:178947fc2f995b38497a74172adee64fdeb8b7ec18f2a5934d037641ba265d26" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/98/f3/12481bda4e5b6d3e698fbf525df4443cc7dce746f246b86b6fcb2fba1844/fastuuid-0.14.0-cp311-cp311-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:73946cb950c8caf65127d4e9a325e2b6be0442a224fd51ba3b6ac44e1912ce34" }, + { url = "https://mirrors.aliyun.com/pypi/packages/59/19/2fc58a1446e4d72b655648eb0879b04e88ed6fa70d474efcf550f640f6ec/fastuuid-0.14.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:12ac85024637586a5b69645e7ed986f7535106ed3013640a393a03e461740cb7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/78/29/3c74756e5b02c40cfcc8b1d8b5bac4edbd532b55917a6bcc9113550e99d1/fastuuid-0.14.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:05a8dde1f395e0c9b4be515b7a521403d1e8349443e7641761af07c7ad1624b1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/52/96/d761da3fccfa84f0f353ce6e3eb8b7f76b3aa21fd25e1b00a19f9c80a063/fastuuid-0.14.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:09378a05020e3e4883dfdab438926f31fea15fd17604908f3d39cbeb22a0b4dc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fc/c2/f84c90167cc7765cb82b3ff7808057608b21c14a38531845d933a4637307/fastuuid-0.14.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:bbb0c4b15d66b435d2538f3827f05e44e2baafcc003dd7d8472dc67807ab8fd8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/af/7b/4bacd03897b88c12348e7bd77943bac32ccf80ff98100598fcff74f75f2e/fastuuid-0.14.0-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:cd5a7f648d4365b41dbf0e38fe8da4884e57bed4e77c83598e076ac0c93995e7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c0/a2/584f2c29641df8bd810d00c1f21d408c12e9ad0c0dafdb8b7b29e5ddf787/fastuuid-0.14.0-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:c0a94245afae4d7af8c43b3159d5e3934c53f47140be0be624b96acd672ceb73" }, + { url = "https://mirrors.aliyun.com/pypi/packages/24/68/c6b77443bb7764c760e211002c8638c0c7cce11cb584927e723215ba1398/fastuuid-0.14.0-cp311-cp311-musllinux_1_1_i686.whl", hash = "sha256:2b29e23c97e77c3a9514d70ce343571e469098ac7f5a269320a0f0b3e193ab36" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5a/87/93f553111b33f9bb83145be12868c3c475bf8ea87c107063d01377cc0e8e/fastuuid-0.14.0-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:1e690d48f923c253f28151b3a6b4e335f2b06bf669c68a02665bc150b7839e94" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9e/8c/a04d486ca55b5abb7eaa65b39df8d891b7b1635b22db2163734dc273579a/fastuuid-0.14.0-cp311-cp311-win32.whl", hash = "sha256:a6f46790d59ab38c6aa0e35c681c0484b50dc0acf9e2679c005d61e019313c24" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9c/b2/2d40bf00820de94b9280366a122cbaa60090c8cf59e89ac3938cf5d75895/fastuuid-0.14.0-cp311-cp311-win_amd64.whl", hash = "sha256:e150eab56c95dc9e3fefc234a0eedb342fac433dacc273cd4d150a5b0871e1fa" }, +] + +[[package]] +name = "filelock" +version = "3.29.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e6/dc/be6cbe99670cd6e4ad387123647cb08e0c32975e223f82551e914c5568a6/filelock-3.29.4.tar.gz", hash = "sha256:10cdb3656fc44541cdf30652a93fb10ec6b05325620eb316bd26893e4201538a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/13/37/a065dc3bd6e49423a6532c642ca7378d3f467b1ef44c2800c937af7f9739/filelock-3.29.4-py3-none-any.whl", hash = "sha256:dac1648087d5115554850d113e7dd8c83ab2d38e3435dde2d4f163847e57b767" }, +] + +[[package]] +name = "flatbuffers" +version = "25.12.19" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e8/2d/d2a548598be01649e2d46231d151a6c56d10b964d94043a335ae56ea2d92/flatbuffers-25.12.19-py2.py3-none-any.whl", hash = "sha256:7634f50c427838bb021c2d66a3d1168e9d199b0607e6329399f04846d42e20b4" }, +] + +[[package]] +name = "frozenlist" +version = "1.8.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/2d/f5/c831fac6cc817d26fd54c7eaccd04ef7e0288806943f7cc5bbf69f3ac1f0/frozenlist-1.8.0.tar.gz", hash = "sha256:3ede829ed8d842f6cd48fc7081d7a41001a56f1f38603f9d49bf3020d59a31ad" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/bc/03/077f869d540370db12165c0aa51640a873fb661d8b315d1d4d67b284d7ac/frozenlist-1.8.0-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:09474e9831bc2b2199fad6da3c14c7b0fbdd377cce9d3d77131be28906cb7d84" }, + { url = "https://mirrors.aliyun.com/pypi/packages/df/b5/7610b6bd13e4ae77b96ba85abea1c8cb249683217ef09ac9e0ae93f25a91/frozenlist-1.8.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:17c883ab0ab67200b5f964d2b9ed6b00971917d5d8a92df149dc2c9779208ee9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6e/ef/0e8f1fe32f8a53dd26bdd1f9347efe0778b0fddf62789ea683f4cc7d787d/frozenlist-1.8.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:fa47e444b8ba08fffd1c18e8cdb9a75db1b6a27f17507522834ad13ed5922b93" }, + { url = "https://mirrors.aliyun.com/pypi/packages/11/b1/71a477adc7c36e5fb628245dfbdea2166feae310757dea848d02bd0689fd/frozenlist-1.8.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:2552f44204b744fba866e573be4c1f9048d6a324dfe14475103fd51613eb1d1f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/45/7e/afe40eca3a2dc19b9904c0f5d7edfe82b5304cb831391edec0ac04af94c2/frozenlist-1.8.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:957e7c38f250991e48a9a73e6423db1bb9dd14e722a10f6b8bb8e16a0f55f695" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a6/aa/7416eac95603ce428679d273255ffc7c998d4132cfae200103f164b108aa/frozenlist-1.8.0-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:8585e3bb2cdea02fc88ffa245069c36555557ad3609e83be0ec71f54fd4abb52" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8b/3d/2a2d1f683d55ac7e3875e4263d28410063e738384d3adc294f5ff3d7105e/frozenlist-1.8.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:edee74874ce20a373d62dc28b0b18b93f645633c2943fd90ee9d898550770581" }, + { url = "https://mirrors.aliyun.com/pypi/packages/78/1e/2d5565b589e580c296d3bb54da08d206e797d941a83a6fdea42af23be79c/frozenlist-1.8.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c9a63152fe95756b85f31186bddf42e4c02c6321207fd6601a1c89ebac4fe567" }, + { url = "https://mirrors.aliyun.com/pypi/packages/aa/c3/65872fcf1d326a7f101ad4d86285c403c87be7d832b7470b77f6d2ed5ddc/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:b6db2185db9be0a04fecf2f241c70b63b1a242e2805be291855078f2b404dd6b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a0/76/ac9ced601d62f6956f03cc794f9e04c81719509f85255abf96e2510f4265/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:f4be2e3d8bc8aabd566f8d5b8ba7ecc09249d74ba3c9ed52e54dc23a293f0b92" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b9/49/ecccb5f2598daf0b4a1415497eba4c33c1e8ce07495eb07d2860c731b8d5/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:c8d1634419f39ea6f5c427ea2f90ca85126b54b50837f31497f3bf38266e853d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/53/4b/ddf24113323c0bbcc54cb38c8b8916f1da7165e07b8e24a717b4a12cbf10/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:1a7fa382a4a223773ed64242dbe1c9c326ec09457e6b8428efb4118c685c3dfd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a7/fb/9b9a084d73c67175484ba2789a59f8eebebd0827d186a8102005ce41e1ba/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:11847b53d722050808926e785df837353bd4d75f1d494377e59b23594d834967" }, + { url = "https://mirrors.aliyun.com/pypi/packages/95/a3/c8fb25aac55bf5e12dae5c5aa6a98f85d436c1dc658f21c3ac73f9fa95e5/frozenlist-1.8.0-cp311-cp311-win32.whl", hash = "sha256:27c6e8077956cf73eadd514be8fb04d77fc946a7fe9f7fe167648b0b9085cc25" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0a/f5/603d0d6a02cfd4c8f2a095a54672b3cf967ad688a60fb9faf04fc4887f65/frozenlist-1.8.0-cp311-cp311-win_amd64.whl", hash = "sha256:ac913f8403b36a2c8610bbfd25b8013488533e71e62b4b4adce9c86c8cea905b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5d/16/c2c9ab44e181f043a86f9a8f84d5124b62dbcb3a02c0977ec72b9ac1d3e0/frozenlist-1.8.0-cp311-cp311-win_arm64.whl", hash = "sha256:d4d3214a0f8394edfa3e303136d0575eece0745ff2b47bd2cb2e66dd92d4351a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9a/9a/e35b4a917281c0b8419d4207f4334c8e8c5dbf4f3f5f9ada73958d937dcc/frozenlist-1.8.0-py3-none-any.whl", hash = "sha256:0c18a16eab41e82c295618a77502e17b195883241c563b00f0aa5106fc4eaa0d" }, +] + +[[package]] +name = "fsspec" +version = "2026.6.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/10/a1/ae4e3e5003468d6391d2c77b6fa1cd73bd5d13511d81c642d7b28ac90ed4/fsspec-2026.6.0.tar.gz", hash = "sha256:f5bac145310fe30e16e1471bd6840b2d990d609e872251d7e674241822abf01a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e5/22/4222d7ddf3da30f363edaa98e329c2bce6c65497c9cb2810931c8b2c0fbc/fsspec-2026.6.0-py3-none-any.whl", hash = "sha256:02e0b71817df9b2169dc30a16832045764def1191b43dcff5bb85bdee212d2a1" }, +] + +[[package]] +name = "gne" +version = "0.4.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "lxml" }, + { name = "pyyaml" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/8e/41/d72fc42048fafcda9e5b88a3e1648cfeb0f1b59c813856b4d6422df70c90/gne-0.4.3.tar.gz", hash = "sha256:26ed77fc2b96d5e9dd28b288ebfdeaf6bf7f034a92b8f75067c64eefffd0b4a9" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/31/f5/07f65c68fab99b22b5b948d2790e2fe0d7ff4f444fb650f4a14c75855b16/gne-0.4.3-py3-none-any.whl", hash = "sha256:d4f53218ebc70fcc0e69f695894720d4f2c335ed1c77ad2bdf6c6cb0db4a7830" }, +] + +[[package]] +name = "greenlet" +version = "3.5.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/6d/6e/802acd792aebb2256fbbee8cacf2727faaeb6f240ac11008f09eae4414bc/greenlet-3.5.1.tar.gz", hash = "sha256:5a56aeb7d5d9cc4b3a735efb5095bd4b4f6f0e4f93e5ca876d0e2315137b7829" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/42/3c/ff890b466eaba2b0f5e6bdfff025f8c75f41b8ffdc3dbc3d24ad261e764a/greenlet-3.5.1-cp311-cp311-macosx_11_0_universal2.whl", hash = "sha256:73f78f9b9f0a5c06e5c946ba1e8e36f5114923b6be109ee618c54f079c3ea14f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/81/0e/5e5457be3d256918f6a4756f073548a3f0190836e2cc94aa6d0d617a940b/greenlet-3.5.1-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:a0cbed8bb44e23c5b199f888f4e4ce096b45ad9f25ff74a7ad0213875e936bb2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/e1/f89a21d58d308298e6f275f13a1b472ed96c680b601a371b08be6a725989/greenlet-3.5.1-cp311-cp311-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a203a8bd0acb0701653d3bbb26e404854a68674139ed5cbb778830f42b09bb33" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2c/f2/8fd452fd81adb9ec79c8275c1375702ab0fd6bee4952da12eaa09b9508d8/greenlet-3.5.1-cp311-cp311-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:6ebeb75c81211f5c702576cf81f315e77e23cfdb2c7c6fcb9dd143e6de35c360" }, + { url = "https://mirrors.aliyun.com/pypi/packages/75/de/af6cef182862d2ccd6975440d21c9058a77c3f9b469abf94e322dfd2e0e3/greenlet-3.5.1-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8a271fcd66c74615cda6a964fda3f304267a12e50a084472218a39bb0376f563" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ec/bc/c318aa9f3ffc77320fddcee3d892be957b42e2ff947198d9450b004f3a38/greenlet-3.5.1-cp311-cp311-manylinux_2_39_riscv64.whl", hash = "sha256:017a544f0385d441e88714160d089d6900ef46c9eff9d99b6715a5ef2d127747" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1a/c6/50e520283a9f19388a7326b05f9e8637e566003475eacaadad04f558c68d/greenlet-3.5.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:ded7b068c7c31c1a8657d4fd42d886b3e051ae29f88b80c5ff9d502257b0f071" }, + { url = "https://mirrors.aliyun.com/pypi/packages/21/1c/13abd1f4860d987fa5e1170a01930d6e6cd40d328de487a3c9fdaff0ffd0/greenlet-3.5.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:d0932b81d72f552ded9d810d00021b64d89f2195a91ce115b893f943b7a4ab3c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f5/56/5f332b7705545eac2dc01b4e9254d24a793f2656d55d5cc6b94ee59d22ae/greenlet-3.5.1-cp311-cp311-win_amd64.whl", hash = "sha256:88e300d136eac057b2397aa1cfd7328b4c87c7eb66a09c7bc6a1292234db474e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d9/a9/a3c2fa886c5b94863fb0e61b3bc14610b7aa94cf4f17f8741b11708305fc/greenlet-3.5.1-cp311-cp311-win_arm64.whl", hash = "sha256:cc6ab7e555c8a112ad3a76e368e86e12a2754bcae1652a5602e133ec7b635523" }, +] + +[[package]] +name = "grpcio" +version = "1.81.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b0/b5/1ff353970a87eda4c98251e34d2dfd214abd4982dc89119c9252a2a482d2/grpcio-1.81.1.tar.gz", hash = "sha256:6fa10a767143a5e82e8eaab53918af0cd8909a57a27f8cb2288b80a613ac671b" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/52/ea/1c2fa386b718ff493225e61cfc052ef400b4d6ffc54cbe261026432624b5/grpcio-1.81.1-cp311-cp311-linux_armv7l.whl", hash = "sha256:d71d30f2d92f67d944631c523713934fee37292469e182ebcd2c1dd8a64ce53f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2b/18/acf45fa8bd1bc5d7b0c2fd3dc4c209379fbd5bb396b440b68a83342226b7/grpcio-1.81.1-cp311-cp311-macosx_11_0_universal2.whl", hash = "sha256:b137f4bf3ada9dc44d411478decc6ff09a79ed30b306cd2abaa98408c3588137" }, + { url = "https://mirrors.aliyun.com/pypi/packages/48/d7/ee86a60699b7db039f772a2c4a7e4facc7138984ff42c0130933a0063884/grpcio-1.81.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:a3acb384427816dd5d470f47e62137b87f74da694faa8a50147012cf40df276a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/26/ee/d2de5e47378ffc207d476c230fea3be4d2601edbce9995f4fe45535d4896/grpcio-1.81.1-cp311-cp311-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:f9a0ebbe45c29b5e5866593c12b78bd9035f0f0f0d4bc8361680cd580d99db49" }, + { url = "https://mirrors.aliyun.com/pypi/packages/23/d6/abeda5c2b896a0b341584fe5ac411bbf72e197a9a374c355fb90965e08d2/grpcio-1.81.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:0a37165cc80b1a368384b383e63a4c38116a10467ae44c904d2d7468c4470ec2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/10/1c/1f0da7d590b4aeee006826ba568d0e419ca14b23e18f901a3da3e9fba613/grpcio-1.81.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:6282caffb41ec326d4cb67ca9cf53b739d1b2f975a2acb498c7418e9f7d9a416" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6a/81/5c505d508f7c887aa7982d21443a4126597c80d34b0bcf40f9cec576d7f3/grpcio-1.81.1-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:a35009284d0d3d5c2c9601c164a911b8b4331608d98a9a66d47d97bb2f522b70" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f7/b2/524847365122ee509ca17bcc4e092198b700e94af7bfd5bb5e6dd9f3ee66/grpcio-1.81.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:1b22c80559854b789a01fd89e8929b3798a156c0829b5282a8939f33ad4115ad" }, + { url = "https://mirrors.aliyun.com/pypi/packages/18/fa/07c037c50b006909d1d13a5848774f8aa7b242f70dc03a035c64eea0e6db/grpcio-1.81.1-cp311-cp311-win32.whl", hash = "sha256:428bec0161b48d8cf583c068591bc0016d0d9cfff52462b72b3884861ea768c5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/41/ed/6bff15376920942fac6b95b9802752b837437172c9e8fc2d3170546b89cc/grpcio-1.81.1-cp311-cp311-win_amd64.whl", hash = "sha256:30e825f6848d9f18bba350ed6c75c1b02a0b5184474a31db9a32b1fa66fd8c79" }, +] + +[[package]] +name = "h11" +version = "0.16.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/01/ee/02a2c011bdab74c6fb3c75474d40b3052059d95df7e73351460c8588d963/h11-0.16.0.tar.gz", hash = "sha256:4e35b956cf45792e4caa5885e69fba00bdbc6ffafbfa020300e549b208ee5ff1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/04/4b/29cac41a4d98d144bf5f6d33995617b185d14b22401f75ca86f384e87ff1/h11-0.16.0-py3-none-any.whl", hash = "sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86" }, +] + +[[package]] +name = "h2" +version = "4.3.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "hpack" }, + { name = "hyperframe" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/1d/17/afa56379f94ad0fe8defd37d6eb3f89a25404ffc71d4d848893d270325fc/h2-4.3.0.tar.gz", hash = "sha256:6c59efe4323fa18b47a632221a1888bd7fde6249819beda254aeca909f221bf1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/69/b2/119f6e6dcbd96f9069ce9a2665e0146588dc9f88f29549711853645e736a/h2-4.3.0-py3-none-any.whl", hash = "sha256:c438f029a25f7945c69e0ccf0fb951dc3f73a5f6412981daee861431b70e2bdd" }, +] + +[[package]] +name = "hf-xet" +version = "1.5.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/4b/2d/57fd21d84d93efb4bd0b962383790e19dd1bc053501b4264c97903b4e83e/hf_xet-1.5.1.tar.gz", hash = "sha256:51ef4500dab3764b41135ee1381a4b62ce56fc54d4c92b719b59e597d6df5bf6" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7a/d8/5e54cf37434759d1f4f2ba9b66077ff9d4c4e1f37b6bd7975da5c40d94ab/hf_xet-1.5.1-cp37-abi3-macosx_10_12_x86_64.whl", hash = "sha256:6abd35c3221eff63836618ddfb954dcf84798603f71d8e33e3ed7b04acfdbe6e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/35/94/4b2ecfbad8f8b04701a23aefb62f540b9137d058b7e1dbef16a32676f0e9/hf_xet-1.5.1-cp37-abi3-macosx_11_0_arm64.whl", hash = "sha256:94e761bbd266bf4c03cee73753916062665ce8365aa40ed321f45afcb934b41e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/de/cc/f99f4bc7295023d7bd9ebbfd51f75cc530ca262c1227666268b8208f4b77/hf_xet-1.5.1-cp37-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:892e3a3a3aecc12aded8b93cf4f9cd059282c7de0732f7d55026f3abdf474350" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cd/6e/21f7e5a2381278bd3b7b7a5a4d90038518bb6308a0c1daf5d9f8268bb178/hf_xet-1.5.1-cp37-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:a93df2039190502835b1db8cd7e178b0b7b889fe9ab51299d5ced26e0dd879a4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/35/0e/f992bb6927ac1cb30ef74e62268f551f338bc32b2191f7c96a44c6f7283e/hf_xet-1.5.1-cp37-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:0c97106032ef70467b4f6bc2d0ccc266d7613ee076afc56516c502f87ce1c4a6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fb/d1/90a498d05447980b977b1669246eeeeae4cfb0ea3e7a286eaba627f91bf9/hf_xet-1.5.1-cp37-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:6208adb15d192b90e4c2ad2a27ed864359b2cb0f2494eb6d7c7f3699ac02e2bf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/b6/20f99cfe97cc663a711f7b33cc21d4793e51968e9a26125b4afcd77315ba/hf_xet-1.5.1-cp37-abi3-win_amd64.whl", hash = "sha256:f7b3002f95d1c13e24bcb4537baa8f0eb3838957067c91bb4959bc004a6435f5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f9/fa/77453694888f03e5a8c8852d1514a0894d8e81c622d39edbaf308ea0dcf4/hf_xet-1.5.1-cp37-abi3-win_arm64.whl", hash = "sha256:93d090b57b211133f6c0dab0205ef5cb6d89162979ba75a74845045cc3063b8e" }, +] + +[[package]] +name = "hpack" +version = "4.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/2c/48/71de9ed269fdae9c8057e5a4c0aa7402e8bb16f2c6e90b3aa53327b113f8/hpack-4.1.0.tar.gz", hash = "sha256:ec5eca154f7056aa06f196a557655c5b009b382873ac8d1e66e79e87535f1dca" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/07/c6/80c95b1b2b94682a72cbdbfb85b81ae2daffa4291fbfa1b1464502ede10d/hpack-4.1.0-py3-none-any.whl", hash = "sha256:157ac792668d995c657d93111f46b4535ed114f0c9c8d672271bbec7eae1b496" }, +] + +[[package]] +name = "httpcore" +version = "1.0.9" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "certifi" }, + { name = "h11" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/06/94/82699a10bca87a5556c9c59b5963f2d039dbd239f25bc2a63907a05a14cb/httpcore-1.0.9.tar.gz", hash = "sha256:6e34463af53fd2ab5d807f399a9b45ea31c3dfa2276f15a2c3f00afff6e176e8" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7e/f5/f66802a942d491edb555dd61e3a9961140fd64c90bce1eafd741609d334d/httpcore-1.0.9-py3-none-any.whl", hash = "sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55" }, +] + +[[package]] +name = "httpx" +version = "0.28.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "anyio" }, + { name = "certifi" }, + { name = "httpcore" }, + { name = "idna" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b1/df/48c586a5fe32a0f01324ee087459e112ebb7224f646c0b5023f5e79e9956/httpx-0.28.1.tar.gz", hash = "sha256:75e98c5f16b0f35b567856f597f06ff2270a374470a5c2392242528e3e3e42fc" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/2a/39/e50c7c3a983047577ee07d2a9e53faf5a69493943ec3f6a384bdc792deb2/httpx-0.28.1-py3-none-any.whl", hash = "sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad" }, +] + +[package.optional-dependencies] +http2 = [ + { name = "h2" }, +] + +[[package]] +name = "httpx-sse" +version = "0.4.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/0f/4c/751061ffa58615a32c31b2d82e8482be8dd4a89154f003147acee90f2be9/httpx_sse-0.4.3.tar.gz", hash = "sha256:9b1ed0127459a66014aec3c56bebd93da3c1bc8bb6618c8082039a44889a755d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d2/fd/6668e5aec43ab844de6fc74927e155a3b37bf40d7c3790e49fc0406b6578/httpx_sse-0.4.3-py3-none-any.whl", hash = "sha256:0ac1c9fe3c0afad2e0ebb25a934a59f4c7823b60792691f779fad2c5568830fc" }, +] + +[[package]] +name = "huggingface-hub" +version = "1.19.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "click" }, + { name = "filelock" }, + { name = "fsspec" }, + { name = "hf-xet", marker = "platform_machine == 'AMD64' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'arm64' or platform_machine == 'x86_64'" }, + { name = "httpx" }, + { name = "packaging" }, + { name = "pyyaml" }, + { name = "tqdm" }, + { name = "typer" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/88/27/629cfe58c582f92ded066c4a07d1a057ff617118ab7973200f770bd853cb/huggingface_hub-1.19.0.tar.gz", hash = "sha256:fd771622182d40977272a923953ee3b1b13538f9f8a7f5d78398f10af0f1c0bd" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b2/a5/558da89f66464d8d0229ff497e8b8666977de2d8cf48c28a2862ecf1250f/huggingface_hub-1.19.0-py3-none-any.whl", hash = "sha256:1dc72e1f6b4d6df6b30eb72e57d00514ef453d660f04af2b87f0e67267f31ee0" }, +] + +[[package]] +name = "humanfriendly" +version = "10.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "pyreadline3", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/cc/3f/2c29224acb2e2df4d2046e4c73ee2662023c58ff5b113c4c1adac0886c43/humanfriendly-10.0.tar.gz", hash = "sha256:6b0b831ce8f15f7300721aa49829fc4e83921a9a301cc7f606be6686a2288ddc" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f0/0f/310fb31e39e2d734ccaa2c0fb981ee41f7bd5056ce9bc29b2248bd569169/humanfriendly-10.0-py2.py3-none-any.whl", hash = "sha256:1697e1a8a8f550fd43c2865cd84542fc175a61dcb779b6fee18cf6b6ccba1477" }, +] + +[[package]] +name = "humanize" +version = "4.15.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ba/66/a3921783d54be8a6870ac4ccffcd15c4dc0dd7fcce51c6d63b8c63935276/humanize-4.15.0.tar.gz", hash = "sha256:1dd098483eb1c7ee8e32eb2e99ad1910baefa4b75c3aff3a82f4d78688993b10" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c5/7b/bca5613a0c3b542420cf92bd5e5fb8ebd5435ce1011a091f66bb7693285e/humanize-4.15.0-py3-none-any.whl", hash = "sha256:b1186eb9f5a9749cd9cb8565aee77919dd7c8d076161cf44d70e59e3301e1769" }, +] + +[[package]] +name = "hyperframe" +version = "6.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/02/e7/94f8232d4a74cc99514c13a9f995811485a6903d48e5d952771ef6322e30/hyperframe-6.1.0.tar.gz", hash = "sha256:f630908a00854a7adeabd6382b43923a4c4cd4b821fcb527e6ab9e15382a3b08" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/48/30/47d0bf6072f7252e6521f3447ccfa40b421b6824517f82854703d0f5a98b/hyperframe-6.1.0-py3-none-any.whl", hash = "sha256:b03380493a519fce58ea5af42e4a42317bf9bd425596f7a0835ffce80f1a42e5" }, +] + +[[package]] +name = "idna" +version = "3.18" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/cd/63/9496c57188a2ee585e0f1db071d75089a11e98aa86eb99d9d7618fc1edce/idna-3.18.tar.gz", hash = "sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1e/5e/d4e9f1a599fb8e573b7b87160658329fbf28d19eac2718f51fc3def3aa5a/idna-3.18-py3-none-any.whl", hash = "sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2" }, +] + +[[package]] +name = "importlib-metadata" +version = "8.9.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "zipp" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e7/72/c600ae4f68c28fc19f9c31b9403053e5dbb8cace2e6842c7b7c3e4d42fe9/importlib_metadata-8.9.0.tar.gz", hash = "sha256:58850626cef4bd2df100378b0f2aea9724a7b92f10770d547725b047078f99ee" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7d/f9/97f2ca8bb3ec6e4b1d64f983ebe98b9a192faddff67fac3d6303a537e670/importlib_metadata-8.9.0-py3-none-any.whl", hash = "sha256:e0f761b6ea91ced3b0844c14c9d955224d538105921f8e6754c00f6ca79fba7f" }, +] + +[[package]] +name = "iniconfig" +version = "2.3.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/72/34/14ca021ce8e5dfedc35312d08ba8bf51fdd999c576889fc2c24cb97f4f10/iniconfig-2.3.0.tar.gz", hash = "sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12" }, +] + +[[package]] +name = "isodate" +version = "0.7.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/54/4d/e940025e2ce31a8ce1202635910747e5a87cc3a6a6bb2d00973375014749/isodate-0.7.2.tar.gz", hash = "sha256:4cd1aa0f43ca76f4a6c6c0292a85f40b35ec2e43e315b59f06e6d32171a953e6" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/15/aa/0aca39a37d3c7eb941ba736ede56d689e7be91cab5d9ca846bde3999eba6/isodate-0.7.2-py3-none-any.whl", hash = "sha256:28009937d8031054830160fce6d409ed342816b543597cece116d966c6d99e15" }, +] + +[[package]] +name = "jinja2" +version = "3.1.6" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "markupsafe" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/df/bf/f7da0350254c0ed7c72f3e33cef02e048281fec7ecec5f032d4aac52226b/jinja2-3.1.6.tar.gz", hash = "sha256:0137fb05990d35f1275a587e9aee6d56da821fc83491a0fb838183be43f66d6d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67" }, +] + +[[package]] +name = "jiter" +version = "0.15.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/66/b5/55f06bb281d92fb3cc86d14e1def2bd908bb77693183e7cb1f5a3c388b0c/jiter-0.15.0.tar.gz", hash = "sha256:4251acc80e2b7c9b7b8823456ea0fceeb0734dac2df7636d3c711b38476b5a76" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e4/13/daa722f5765c393576f466378f9dfd29d77c9bed939e0688f96afa3601ea/jiter-0.15.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:0f862193b8696249d22ec433e85fd2ab0ad9596bc3e45e6c0bc55e8aeba97be2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7f/82/2d2551829b082f4b6d82b9f939b031fb808a10aab1ec0664f82e150bb9a2/jiter-0.15.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:1303d4d68a9b051ea90502402063ecf3807da00ad2affa19ca1ae3b90b3c5f67" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2a/0a/8b1a51466f7fe9f31dbe4bc7e0ca848674f9825e0f737b929b97e8c60aa7/jiter-0.15.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:392b8ab019e5502d08aff85c6272209c24bc2cbe706ea82a56368f524236614a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f6/2a/e71dea19822e2e404e83992a08c1d6b9b617bb944f28c9c2fbd85d02c91e/jiter-0.15.0-cp311-cp311-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:773b6eb282ce11ee19f05f6b2d4404fa308e5bbd353b0b80a0262caad6db2cd7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c4/59/97e1fa539d124a509a00ab7f669289d1c1d236ecabf12948a18f16c91082/jiter-0.15.0-cp311-cp311-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:8d2c0c44d569ce0f2850f5c926f8caeb5f245fbc84475aeb36efccc2103e6dbd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d1/7a/4a68d331aef8cf2e2393c14a3aacb635c62aa86071b0229899fb5baaa907/jiter-0.15.0-cp311-cp311-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:032396229564bca02440396bd327710719f724f5e7b7e9f7a8eb3faa4a2c2281" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7b/7e/1c445c2b6f0e30a274dc8082e0c3c7825411cce80d726bccd697c98cc8d3/jiter-0.15.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f3d37768fce7f88dd2a8c6091f2325dea27d30d30d5c6e7a1c0f0af77723b708" }, + { url = "https://mirrors.aliyun.com/pypi/packages/00/94/e20d38984fc17a636371bffd2ae0f698124fdc8e75ef969cd2da6ba7cea7/jiter-0.15.0-cp311-cp311-manylinux_2_31_riscv64.whl", hash = "sha256:2c9cb907439d20bd0c7d7565ca01ee52234203208433749bae5b516907526928" }, + { url = "https://mirrors.aliyun.com/pypi/packages/94/fa/4d09f814779d0ea80a28ed8e4c6662ec9a4a8ecef0ac52190ebac6262d14/jiter-0.15.0-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:9100ddbec09741cc66feb0fc6773f8bdbd0e3c345689368f260082ff85dcc0cd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/54/9d/8eb5d4fb8bf7e93a75964a5da71a75c67c864baf7fa3f98598187b3c7e57/jiter-0.15.0-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:ae1b0d82ac2d987f9ea512b1c9adfcc71a28de3dea3a6039b54d76cffda9901e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e7/2c/5e07874e59e623a943a0acf1552a80d05b70f31b402287a8fc6d7ec634c7/jiter-0.15.0-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:8020c99ec13a7db2b6f96cbe82ef4721c88b426a4892f27478044af0284615ef" }, + { url = "https://mirrors.aliyun.com/pypi/packages/22/ed/d2d34422143474cadc15b60d482b1c35683dbc5c63c24346ddd0df09bcaf/jiter-0.15.0-cp311-cp311-win32.whl", hash = "sha256:42bfb257930800cf43e7c62c832402c704ab60797c992faf88d20e903eac8f32" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1d/7d/52778b930e5cc3e52a37d950b1c10494244308b4329b25a0ff0d88303a81/jiter-0.15.0-cp311-cp311-win_amd64.whl", hash = "sha256:860a74063284a2ae9bfedd694f299cc2c68e2696c5f3d440cc9d18bb81b9dd04" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3b/4f/d9b4067feb69b3fa6eb0488e1b59e2ad5b463fe39f59e527eab2aca00bb0/jiter-0.15.0-cp311-cp311-win_arm64.whl", hash = "sha256:37a10c377ce3a4a85f4a67f28b7afe093154cde77eaf248a72e856aa08b4d865" }, + { url = "https://mirrors.aliyun.com/pypi/packages/65/43/1fc62172aa98b50a7de9a25554060db510f85c89cfbed0dfe13e1907a139/jiter-0.15.0-graalpy311-graalpy242_311_native-macosx_10_12_x86_64.whl", hash = "sha256:411fa4dfa5a7ae3d11491027ffb9beadec3996010a986862db70d91abba1c750" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e8/c4/dd58fcd9e2df83666e5c1c1347bef58ce919cd8efc3ffa38aeea62ce493b/jiter-0.15.0-graalpy311-graalpy242_311_native-macosx_11_0_arm64.whl", hash = "sha256:2b0074e2f56eb2dacca1689760fd2852a068f85a0547a157b82cb4cafeb6768b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/39/86/b695e16f1180c07f43ea98e73ecd21cf63fa2e1b0c1103739013784d11ae/jiter-0.15.0-graalpy311-graalpy242_311_native-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:913d02d29c9606643418d9ccfc3b72492ab25a6bf7889934e09a3490f8d3438b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/34/56/55d76614af37fe3f22a3347d1e410d2a15da581997cb2da499a625000bb5/jiter-0.15.0-graalpy311-graalpy242_311_native-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:b15d3ec9b0449c40e85319bdb4caa8b77ab526e74f5532ed94bec15e2f66822c" }, +] + +[[package]] +name = "joblib" +version = "1.5.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/41/f2/d34e8b3a08a9cc79a50b2208a93dce981fe615b64d5a4d4abee421d898df/joblib-1.5.3.tar.gz", hash = "sha256:8561a3269e6801106863fd0d6d84bb737be9e7631e33aaed3fb9ce5953688da3" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl", hash = "sha256:5fc3c5039fc5ca8c0276333a188bbd59d6b7ab37fe6632daa76bc7f9ec18e713" }, +] + +[[package]] +name = "jsonschema" +version = "4.26.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "attrs" }, + { name = "jsonschema-specifications" }, + { name = "referencing" }, + { name = "rpds-py" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b3/fc/e067678238fa451312d4c62bf6e6cf5ec56375422aee02f9cb5f909b3047/jsonschema-4.26.0.tar.gz", hash = "sha256:0c26707e2efad8aa1bfc5b7ce170f3fccc2e4918ff85989ba9ffa9facb2be326" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/69/90/f63fb5873511e014207a475e2bb4e8b2e570d655b00ac19a9a0ca0a385ee/jsonschema-4.26.0-py3-none-any.whl", hash = "sha256:d489f15263b8d200f8387e64b4c3a75f06629559fb73deb8fdfb525f2dab50ce" }, +] + +[[package]] +name = "jsonschema-specifications" +version = "2025.9.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "referencing" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/19/74/a633ee74eb36c44aa6d1095e7cc5569bebf04342ee146178e2d36600708b/jsonschema_specifications-2025.9.1.tar.gz", hash = "sha256:b540987f239e745613c7a9176f3edb72b832a4ac465cf02712288397832b5e8d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/41/45/1a4ed80516f02155c51f51e8cedb3c1902296743db0bbc66608a0db2814f/jsonschema_specifications-2025.9.1-py3-none-any.whl", hash = "sha256:98802fee3a11ee76ecaca44429fda8a41bff98b00a0f2838151b113f210cc6fe" }, +] + +[[package]] +name = "lark" +version = "1.3.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/da/34/28fff3ab31ccff1fd4f6c7c7b0ceb2b6968d8ea4950663eadcb5720591a0/lark-1.3.1.tar.gz", hash = "sha256:b426a7a6d6d53189d318f2b6236ab5d6429eaf09259f1ca33eb716eed10d2905" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/82/3d/14ce75ef66813643812f3093ab17e46d3a206942ce7376d31ec2d36229e7/lark-1.3.1-py3-none-any.whl", hash = "sha256:c629b661023a014c37da873b4ff58a817398d12635d3bbb2c5a03be7fe5d1e12" }, +] + +[[package]] +name = "librt" +version = "0.11.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/40/08/9e7f6b5d2b5bed6ad055cdd5925f192bb403a51280f86b56554d9d0699a2/librt-0.11.0.tar.gz", hash = "sha256:075dc3ef4458a278e0195cbf6ac9d38808d9b906c5a6c7f7f79c3888276a3fb1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/fe/87/2bf31fe17587b29e3f93ec31421e2b1e1c3e349b8bf6c7c313dbad1d5340/librt-0.11.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:93d95bd45b7d58343d8b90d904450a545144eec19a002511163426f8ab1fae29" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cf/08/5c5bf772920b7ebac6e32bc91a643e0ab3870199c0b542356d3baa83970a/librt-0.11.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:4ee278c769a713638cdacd4c0436d72156e75df3ebc0166ab2b9dc43acc386c9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/06/20/662a03d254e5b000d838e8b345d83303ddb768c080fd488e40634c0fa66b/librt-0.11.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f230cb1cbc9faaa616f9a678f530ebcf186e414b6bcbd88b960e4ba1b92428d5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/de/f3/aa81523e45184c6ec23dc7f63263362ec55f80a09d424c012359ecbe7e35/librt-0.11.0-cp311-cp311-manylinux2014_i686.manylinux_2_17_i686.manylinux_2_28_i686.whl", hash = "sha256:5d63c855d86938d9de93e265c9bd8c705b51ec494de5738340ee93767a686e4b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6b/6f/59c74b560ca8853834d5501d589c8a2519f4184f273a085ffd0f37a1cc47/librt-0.11.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:993f028be9e96a08d31df3479ac80d99be374d17f3b78e4796b3fd3c913d4e89" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fe/7b/5aa4d2c9600a719401160bf7055417df0b2a47439b9d88286ce45e56b65f/librt-0.11.0-cp311-cp311-manylinux_2_34_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:258d73a0aa66a055e65b2e4d1b8cdb23b9d132c5bb915d9547d804fcaed116cc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d6/31/9143803d7da6856a69153785768c4936864430eec0fd9461c3ea527d9922/librt-0.11.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:0827efe7854718f04aaddf6496e96960a956e676fe1d0f04eb41511fd8ad06d5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2f/5a/bce08184488426bda4ccc2c4964ac048c8f68ae89bd7120082eef4233cfd/librt-0.11.0-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:7753e57d6e12d019c0d8786f1c09c709f4c3fcc57c3887b24e36e6c06ec938b7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/89/8c/bb5e213d254b7505a0e658da199d8ab719086632ce09eef311ab27976523/librt-0.11.0-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:11bd19822431cc21af9f27374e7ae2e58103c7d98bda823536a6c47f6bb2bb3d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9d/fb/541cdad5b1ab1300398c74c4c9a497b88e5074c21b1244c8f49731d3a284/librt-0.11.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:22bdf239b219d3993761a148ffa134b19e52e9989c84f845d5d7b71d70a17412" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8f/f2/464bb69295c320cb06bddb4f14a4ec67934ee14b2bffb12b19fb7ab287ba/librt-0.11.0-cp311-cp311-win32.whl", hash = "sha256:46c60b61e308eb535fbd6fa622b1ee1bb2815691c1ad9c98bf7b84952ec3bc8d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/e7/a17ee1788f9e4fbf548c19f4afa07c92089b9e24fef6cb2410863781ef4c/librt-0.11.0-cp311-cp311-win_amd64.whl", hash = "sha256:902e546ff044f579ff1c953ff5fce97b636fe9e3943996b2177710c6ef076f73" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cc/c7/6c766214f9f9903bcfcfbef97d807af8d8f5aa3502d247858ab17582d212/librt-0.11.0-cp311-cp311-win_arm64.whl", hash = "sha256:65ac3bc20f78aa0ee5ae84baa68917f89fef4af63e941084dd019a0d0e749f0c" }, +] + +[[package]] +name = "litellm" +version = "1.89.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "aiohttp" }, + { name = "click" }, + { name = "fastuuid" }, + { name = "httpx" }, + { name = "importlib-metadata" }, + { name = "jinja2" }, + { name = "jsonschema" }, + { name = "openai" }, + { name = "pydantic" }, + { name = "python-dotenv" }, + { name = "tiktoken" }, + { name = "tokenizers" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e8/1b/de398e6cef46d4cbc4395c895330f30eab05439bf6d0c5c53b0203661eeb/litellm-1.89.1.tar.gz", hash = "sha256:eb9292f90afe46dcce4bfef6bbeaa54494945813763e1c0917f4e9a1c584c1a4" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ec/24/81f03088876a13b628fdbaf746ee746e1dff127d63f339ef72f1a3801c91/litellm-1.89.1-py3-none-any.whl", hash = "sha256:a52a67625d89cb1787ef48c4b3c1ab9c2574ea304f56900bc631844297a13bd4" }, +] + +[[package]] +name = "loguru" +version = "0.7.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "win32-setctime", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/3a/05/a1dae3dffd1116099471c643b8924f5aa6524411dc6c63fdae648c4f1aca/loguru-0.7.3.tar.gz", hash = "sha256:19480589e77d47b8d85b2c827ad95d49bf31b0dcde16593892eb51dd18706eb6" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/0c/29/0348de65b8cc732daa3e33e67806420b2ae89bdce2b04af740289c5c6c8c/loguru-0.7.3-py3-none-any.whl", hash = "sha256:31a33c10c8e1e10422bfd431aeb5d351c7cf7fa671e3c4df004162264b28220c" }, +] + +[[package]] +name = "lxml" +version = "5.4.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/76/3d/14e82fc7c8fb1b7761f7e748fd47e2ec8276d137b6acfe5a4bb73853e08f/lxml-5.4.0.tar.gz", hash = "sha256:d12832e1dbea4be280b22fd0ea7c9b87f0d8fc51ba06e92dc62d52f804f78ebd" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/81/2d/67693cc8a605a12e5975380d7ff83020dcc759351b5a066e1cced04f797b/lxml-5.4.0-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:98a3912194c079ef37e716ed228ae0dcb960992100461b704aea4e93af6b0bb9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/73/53/b5a05ab300a808b72e848efd152fe9c022c0181b0a70b8bca1199f1bed26/lxml-5.4.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:0ea0252b51d296a75f6118ed0d8696888e7403408ad42345d7dfd0d1e93309a7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d8/cb/1a3879c5f512bdcd32995c301886fe082b2edd83c87d41b6d42d89b4ea4d/lxml-5.4.0-cp311-cp311-manylinux_2_12_i686.manylinux2010_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:b92b69441d1bd39f4940f9eadfa417a25862242ca2c396b406f9272ef09cdcaa" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f9/94/bbc66e42559f9d04857071e3b3d0c9abd88579367fd2588a4042f641f57e/lxml-5.4.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:20e16c08254b9b6466526bc1828d9370ee6c0d60a4b64836bc3ac2917d1e16df" }, + { url = "https://mirrors.aliyun.com/pypi/packages/66/95/34b0679bee435da2d7cae895731700e519a8dfcab499c21662ebe671603e/lxml-5.4.0-cp311-cp311-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:7605c1c32c3d6e8c990dd28a0970a3cbbf1429d5b92279e37fda05fb0c92190e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e0/5d/abfcc6ab2fa0be72b2ba938abdae1f7cad4c632f8d552683ea295d55adfb/lxml-5.4.0-cp311-cp311-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:ecf4c4b83f1ab3d5a7ace10bafcb6f11df6156857a3c418244cef41ca9fa3e44" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5a/78/6bd33186c8863b36e084f294fc0a5e5eefe77af95f0663ef33809cc1c8aa/lxml-5.4.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0cef4feae82709eed352cd7e97ae062ef6ae9c7b5dbe3663f104cd2c0e8d94ba" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3b/74/4d7ad4839bd0fc64e3d12da74fc9a193febb0fae0ba6ebd5149d4c23176a/lxml-5.4.0-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:df53330a3bff250f10472ce96a9af28628ff1f4efc51ccba351a8820bca2a8ba" }, + { url = "https://mirrors.aliyun.com/pypi/packages/24/0d/0a98ed1f2471911dadfc541003ac6dd6879fc87b15e1143743ca20f3e973/lxml-5.4.0-cp311-cp311-manylinux_2_28_ppc64le.whl", hash = "sha256:aefe1a7cb852fa61150fcb21a8c8fcea7b58c4cb11fbe59c97a0a4b31cae3c8c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/48/de/d4f7e4c39740a6610f0f6959052b547478107967362e8424e1163ec37ae8/lxml-5.4.0-cp311-cp311-manylinux_2_28_s390x.whl", hash = "sha256:ef5a7178fcc73b7d8c07229e89f8eb45b2908a9238eb90dcfc46571ccf0383b8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/07/8c/61763abd242af84f355ca4ef1ee096d3c1b7514819564cce70fd18c22e9a/lxml-5.4.0-cp311-cp311-manylinux_2_28_x86_64.whl", hash = "sha256:d2ed1b3cb9ff1c10e6e8b00941bb2e5bb568b307bfc6b17dffbbe8be5eecba86" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f9/c5/6d7e3b63e7e282619193961a570c0a4c8a57fe820f07ca3fe2f6bd86608a/lxml-5.4.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:72ac9762a9f8ce74c9eed4a4e74306f2f18613a6b71fa065495a67ac227b3056" }, + { url = "https://mirrors.aliyun.com/pypi/packages/71/4a/e60a306df54680b103348545706a98a7514a42c8b4fbfdcaa608567bb065/lxml-5.4.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:f5cb182f6396706dc6cc1896dd02b1c889d644c081b0cdec38747573db88a7d7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/27/f2/9754aacd6016c930875854f08ac4b192a47fe19565f776a64004aa167521/lxml-5.4.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:3a3178b4873df8ef9457a4875703488eb1622632a9cee6d76464b60e90adbfcd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/38/a2/0c49ec6941428b1bd4f280650d7b11a0f91ace9db7de32eb7aa23bcb39ff/lxml-5.4.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:e094ec83694b59d263802ed03a8384594fcce477ce484b0cbcd0008a211ca751" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7a/75/87a3963a08eafc46a86c1131c6e28a4de103ba30b5ae903114177352a3d7/lxml-5.4.0-cp311-cp311-win32.whl", hash = "sha256:4329422de653cdb2b72afa39b0aa04252fca9071550044904b2e7036d9d97fe4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fa/f9/1f0964c4f6c2be861c50db380c554fb8befbea98c6404744ce243a3c87ef/lxml-5.4.0-cp311-cp311-win_amd64.whl", hash = "sha256:fd3be6481ef54b8cfd0e1e953323b7aa9d9789b94842d0e5b142ef4bb7999539" }, +] + +[[package]] +name = "magika" +version = "0.6.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "click" }, + { name = "numpy" }, + { name = "onnxruntime" }, + { name = "python-dotenv" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/a3/f3/3d1dcdd7b9c41d589f5cff252d32ed91cdf86ba84391cfc81d9d8773571d/magika-0.6.3.tar.gz", hash = "sha256:7cc52aa7359af861957043e2bf7265ed4741067251c104532765cd668c0c0cb1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a2/e4/35c323beb3280482c94299d61626116856ac2d4ec16ecef50afc4fdd4291/magika-0.6.3-py3-none-any.whl", hash = "sha256:eda443d08006ee495e02083b32e51b98cb3696ab595a7d13900d8e2ef506ec9d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/25/8f/132b0d7cd51c02c39fd52658a5896276c30c8cc2fd453270b19db8c40f7e/magika-0.6.3-py3-none-macosx_11_0_arm64.whl", hash = "sha256:86901e64b05dde5faff408c9b8245495b2e1fd4c226e3393d3d2a3fee65c504b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c4/03/5ed859be502903a68b7b393b17ae0283bf34195cfcca79ce2dc25b9290e7/magika-0.6.3-py3-none-manylinux_2_28_x86_64.whl", hash = "sha256:3d9661eedbdf445ac9567e97e7ceefb93545d77a6a32858139ea966b5806fb64" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7b/9e/f8ee7d644affa3b80efdd623a3d75865c8f058f3950cb87fb0c48e3559bc/magika-0.6.3-py3-none-win_amd64.whl", hash = "sha256:e57f75674447b20cab4db928ae58ab264d7d8582b55183a0b876711c2b2787f3" }, +] + +[[package]] +name = "mammoth" +version = "1.11.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "cobble" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ed/3c/a58418d2af00f2da60d4a51e18cd0311307b72d48d2fffec36a97b4a5e44/mammoth-1.11.0.tar.gz", hash = "sha256:a0f59e442f34d5b6447f4b0999306cbf3e67aaabfa8cb516f878fb1456744637" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ca/54/2e39566a131b13f6d8d193f974cb6a34e81bb7cc2fa6f7e03de067b36588/mammoth-1.11.0-py2.py3-none-any.whl", hash = "sha256:c077ab0d450bd7c0c6ecd529a23bf7e0fa8190c929e28998308ff4eada3f063b" }, +] + +[[package]] +name = "markdown-it-py" +version = "4.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "mdurl" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/06/ff/7841249c247aa650a76b9ee4bbaeae59370dc8bfd2f6c01f3630c35eb134/markdown_it_py-4.2.0.tar.gz", hash = "sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b3/81/4da04ced5a082363ecfa159c010d200ecbd959ae410c10c0264a38cac0f5/markdown_it_py-4.2.0-py3-none-any.whl", hash = "sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a" }, +] + +[[package]] +name = "markdownify" +version = "1.2.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "beautifulsoup4" }, + { name = "six" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/3f/bc/c8c8eea5335341306b0fa7e1cb33c5e1c8d24ef70ddd684da65f41c49c92/markdownify-1.2.2.tar.gz", hash = "sha256:b274f1b5943180b031b699b199cbaeb1e2ac938b75851849a31fd0c3d6603d09" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/43/ce/f1e3e9d959db134cedf06825fae8d5b294bd368aacdd0831a3975b7c4d55/markdownify-1.2.2-py3-none-any.whl", hash = "sha256:3f02d3cc52714084d6e589f70397b6fc9f2f3a8531481bf35e8cc39f975e186a" }, +] + +[[package]] +name = "markitdown" +version = "0.1.5" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "beautifulsoup4" }, + { name = "charset-normalizer" }, + { name = "defusedxml" }, + { name = "magika" }, + { name = "markdownify" }, + { name = "requests" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/83/93/3b93c291c99d09f64f7535ba74c1c6a3507cf49cffd38983a55de6f834b6/markitdown-0.1.5.tar.gz", hash = "sha256:4c956ff1528bf15e1814542035ec96e989206d19d311bb799f4df973ecafc31a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b1/8b/fd7e042455a829a1ede0bc8e9e3061aa6c7c4cf745385526ef62ff1b5a5b/markitdown-0.1.5-py3-none-any.whl", hash = "sha256:5180a9a841e20fc01c2c09dbc5d039638429bbebcdc2af1b2615c3c427840434" }, +] + +[package.optional-dependencies] +all = [ + { name = "azure-ai-documentintelligence" }, + { name = "azure-identity" }, + { name = "lxml" }, + { name = "mammoth" }, + { name = "olefile" }, + { name = "openpyxl" }, + { name = "pandas" }, + { name = "pdfminer-six" }, + { name = "pdfplumber" }, + { name = "pydub" }, + { name = "python-pptx" }, + { name = "speechrecognition" }, + { name = "xlrd" }, + { name = "youtube-transcript-api" }, +] + +[[package]] +name = "markupsafe" +version = "3.0.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/7e/99/7690b6d4034fffd95959cbe0c02de8deb3098cc577c67bb6a24fe5d7caa7/markupsafe-3.0.3.tar.gz", hash = "sha256:722695808f4b6457b320fdc131280796bdceb04ab50fe1795cd540799ebe1698" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/08/db/fefacb2136439fc8dd20e797950e749aa1f4997ed584c62cfb8ef7c2be0e/markupsafe-3.0.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:1cc7ea17a6824959616c525620e387f6dd30fec8cb44f649e31712db02123dad" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e1/2e/5898933336b61975ce9dc04decbc0a7f2fee78c30353c5efba7f2d6ff27a/markupsafe-3.0.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:4bd4cd07944443f5a265608cc6aab442e4f74dff8088b0dfc8238647b8f6ae9a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1d/09/adf2df3699d87d1d8184038df46a9c80d78c0148492323f4693df54e17bb/markupsafe-3.0.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6b5420a1d9450023228968e7e6a9ce57f65d148ab56d2313fcd589eee96a7a50" }, + { url = "https://mirrors.aliyun.com/pypi/packages/30/ac/0273f6fcb5f42e314c6d8cd99effae6a5354604d461b8d392b5ec9530a54/markupsafe-3.0.3-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0bf2a864d67e76e5c9a34dc26ec616a66b9888e25e7b9460e1c76d3293bd9dbf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/19/ae/31c1be199ef767124c042c6c3e904da327a2f7f0cd63a0337e1eca2967a8/markupsafe-3.0.3-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:bc51efed119bc9cfdf792cdeaa4d67e8f6fcccab66ed4bfdd6bde3e59bfcbb2f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b2/76/7edcab99d5349a4532a459e1fe64f0b0467a3365056ae550d3bcf3f79e1e/markupsafe-3.0.3-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:068f375c472b3e7acbe2d5318dea141359e6900156b5b2ba06a30b169086b91a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a4/28/6e74cdd26d7514849143d69f0bf2399f929c37dc2b31e6829fd2045b2765/markupsafe-3.0.3-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:7be7b61bb172e1ed687f1754f8e7484f1c8019780f6f6b0786e76bb01c2ae115" }, + { url = "https://mirrors.aliyun.com/pypi/packages/62/7e/a145f36a5c2945673e590850a6f8014318d5577ed7e5920a4b3448e0865d/markupsafe-3.0.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0f/62/d9c46a7f5c9adbeeeda52f5b8d802e1094e9717705a645efc71b0913a0a8/markupsafe-3.0.3-cp311-cp311-win32.whl", hash = "sha256:0db14f5dafddbb6d9208827849fad01f1a2609380add406671a26386cdf15a19" }, + { url = "https://mirrors.aliyun.com/pypi/packages/83/8a/4414c03d3f891739326e1783338e48fb49781cc915b2e0ee052aa490d586/markupsafe-3.0.3-cp311-cp311-win_amd64.whl", hash = "sha256:de8a88e63464af587c950061a5e6a67d3632e36df62b986892331d4620a35c01" }, + { url = "https://mirrors.aliyun.com/pypi/packages/35/73/893072b42e6862f319b5207adc9ae06070f095b358655f077f69a35601f0/markupsafe-3.0.3-cp311-cp311-win_arm64.whl", hash = "sha256:3b562dd9e9ea93f13d53989d23a7e775fdfd1066c33494ff43f5418bc8c58a5c" }, +] + +[[package]] +name = "mcp" +version = "1.27.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "anyio" }, + { name = "httpx" }, + { name = "httpx-sse" }, + { name = "jsonschema" }, + { name = "pydantic" }, + { name = "pydantic-settings" }, + { name = "pyjwt", extra = ["crypto"] }, + { name = "python-multipart" }, + { name = "pywin32", marker = "sys_platform == 'win32'" }, + { name = "sse-starlette" }, + { name = "starlette" }, + { name = "typing-extensions" }, + { name = "typing-inspection" }, + { name = "uvicorn", marker = "sys_platform != 'emscripten'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/27/3c/347cf965d313f5d41764e7d46bea6ffe7d9ef13b983cc429b0340962a082/mcp-1.27.2.tar.gz", hash = "sha256:8e02db104096d1c25b28e64bde29a5c32b31bc241710213e12fd4d84985bdfef" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c9/11/252c6f971dc4f16af1d98a1c469d8ba523aab00d1bb76b4d3bc1ff32eacc/mcp-1.27.2-py3-none-any.whl", hash = "sha256:d6ff5160c6ca65d93013626efb3fc249de683c30b2d8570755ceddd490344de5" }, +] + +[[package]] +name = "mdurl" +version = "0.1.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d6/54/cfe61301667036ec958cb99bd3efefba235e65cdeb9c84d24a8293ba1d90/mdurl-0.1.2.tar.gz", hash = "sha256:bb413d29f5eea38f31dd4754dd7377d4465116fb207585f97bf925588687c1ba" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b3/38/89ba8ad64ae25be8de66a6d463314cf1eb366222074cfda9ee839c56a4b4/mdurl-0.1.2-py3-none-any.whl", hash = "sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8" }, +] + +[[package]] +name = "mpmath" +version = "1.3.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e0/47/dd32fa426cc72114383ac549964eecb20ecfd886d1e5ccf5340b55b02f57/mpmath-1.3.0.tar.gz", hash = "sha256:7a28eb2a9774d00c7bc92411c19a89209d5da7c4c9a9e227be8330a23a25b91f" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/43/e3/7d92a15f894aa0c9c4b49b8ee9ac9850d6e63b03c9c32c0367a13ae62209/mpmath-1.3.0-py3-none-any.whl", hash = "sha256:a0b2b9fe80bbcd81a6647ff13108738cfb482d481d826cc0e02f5b35e5c88d2c" }, +] + +[[package]] +name = "msal" +version = "1.37.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "cryptography" }, + { name = "pyjwt", extra = ["crypto"] }, + { name = "requests" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/9a/99/d840198ecf6e8057bbc937f129ae940404485d736cda73253bbff9537f01/msal-1.37.0.tar.gz", hash = "sha256:1b1672a33ee467c1d70b341bb16cafd51bb3c817147a95b93263794b03971bec" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/94/b0/d807279f4b55d16d1f120d5ac4344c6e39b56732e2a224d40bded7fd67ad/msal-1.37.0-py3-none-any.whl", hash = "sha256:dd17e95a7c71bce75e8108113438ba7c4a086b3bcad4f57a8c09b7af3d753c2d" }, +] + +[[package]] +name = "msal-extensions" +version = "1.3.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "msal" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/01/99/5d239b6156eddf761a636bded1118414d161bd6b7b37a9335549ed159396/msal_extensions-1.3.1.tar.gz", hash = "sha256:c5b0fd10f65ef62b5f1d62f4251d51cbcaf003fcedae8c91b040a488614be1a4" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/5e/75/bd9b7bb966668920f06b200e84454c8f3566b102183bc55c5473d96cb2b9/msal_extensions-1.3.1-py3-none-any.whl", hash = "sha256:96d3de4d034504e969ac5e85bae8106c8373b5c6568e4c8fa7af2eca9dbe6bca" }, +] + +[[package]] +name = "multidict" +version = "6.7.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/1a/c2/c2d94cbe6ac1753f3fc980da97b3d930efe1da3af3c9f5125354436c073d/multidict-6.7.1.tar.gz", hash = "sha256:ec6652a1bee61c53a3e5776b6049172c53b6aaba34f18c9ad04f82712bac623d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ce/f1/a90635c4f88fb913fbf4ce660b83b7445b7a02615bda034b2f8eb38fd597/multidict-6.7.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:7ff981b266af91d7b4b3793ca3382e53229088d193a85dfad6f5f4c27fc73e5d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a6/9b/267e64eaf6fc637a15b35f5de31a566634a2740f97d8d094a69d34f524a4/multidict-6.7.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:844c5bca0b5444adb44a623fb0a1310c2f4cd41f402126bb269cd44c9b3f3e1e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/dd/a4/d45caf2b97b035c57267791ecfaafbd59c68212004b3842830954bb4b02e/multidict-6.7.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:f2a0a924d4c2e9afcd7ec64f9de35fcd96915149b2216e1cb2c10a56df483855" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fd/d2/0a36c8473f0cbaeadd5db6c8b72d15bbceeec275807772bfcd059bef487d/multidict-6.7.1-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:8be1802715a8e892c784c0197c2ace276ea52702a0ede98b6310c8f255a5afb3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5d/16/8c65be997fd7dd311b7d39c7b6e71a0cb449bad093761481eccbbe4b42a2/multidict-6.7.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2e2d2ed645ea29f31c4c7ea1552fcfd7cb7ba656e1eafd4134a6620c9f5fdd9e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/01/fb/4dbd7e848d2799c6a026ec88ad39cf2b8416aa167fcc903baa55ecaa045c/multidict-6.7.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:95922cee9a778659e91db6497596435777bd25ed116701a4c034f8e46544955a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b6/8a/4a3a6341eac3830f6053062f8fbc9a9e54407c80755b3f05bc427295c2d0/multidict-6.7.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:6b83cabdc375ffaaa15edd97eb7c0c672ad788e2687004990074d7d6c9b140c8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f7/a2/dd575a69c1aa206e12d27d0770cdf9b92434b48a9ef0cd0d1afdecaa93c4/multidict-6.7.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:38fb49540705369bab8484db0689d86c0a33a0a9f2c1b197f506b71b4b6c19b0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5a/56/21b27c560c13822ed93133f08aa6372c53a8e067f11fbed37b4adcdac922/multidict-6.7.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:439cbebd499f92e9aa6793016a8acaa161dfa749ae86d20960189f5398a19144" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5a/a4/23466059dc3854763423d0ad6c0f3683a379d97673b1b89ec33826e46728/multidict-6.7.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:6d3bc717b6fe763b8be3f2bee2701d3c8eb1b2a8ae9f60910f1b2860c82b6c49" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1f/67/51dd754a3524d685958001e8fa20a0f5f90a6a856e0a9dcabff69be3dbb7/multidict-6.7.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:619e5a1ac57986dbfec9f0b301d865dddf763696435e2962f6d9cf2fdff2bb71" }, + { url = "https://mirrors.aliyun.com/pypi/packages/64/3f/036dfc8c174934d4b55d86ff4f978e558b0e585cef70cfc1ad01adc6bf18/multidict-6.7.1-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:0b38ebffd9be37c1170d33bc0f36f4f262e0a09bc1aac1c34c7aa51a7293f0b3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3d/20/6214d3c105928ebc353a1c644a6ef1408bc5794fcb4f170bb524a3c16311/multidict-6.7.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:10ae39c9cfe6adedcdb764f5e8411d4a92b055e35573a2eaa88d3323289ef93c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b1/e2/c653bc4ae1be70a0f836b82172d643fcf1dade042ba2676ab08ec08bff0f/multidict-6.7.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:25167cc263257660290fba06b9318d2026e3c910be240a146e1f66dd114af2b0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c8/11/a854b4154cd3bd8b1fd375e8a8ca9d73be37610c361543d56f764109509b/multidict-6.7.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:128441d052254f42989ef98b7b6a6ecb1e6f708aa962c7984235316db59f50fa" }, + { url = "https://mirrors.aliyun.com/pypi/packages/13/bf/9676c0392309b5fdae322333d22a829715b570edb9baa8016a517b55b558/multidict-6.7.1-cp311-cp311-win32.whl", hash = "sha256:d62b7f64ffde3b99d06b707a280db04fb3855b55f5a06df387236051d0668f4a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c9/68/f16a3a8ba6f7b6dc92a1f19669c0810bd2c43fc5a02da13b1cbf8e253845/multidict-6.7.1-cp311-cp311-win_amd64.whl", hash = "sha256:bdbf9f3b332abd0cdb306e7c2113818ab1e922dc84b8f8fd06ec89ed2a19ab8b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ac/ad/9dd5305253fa00cd3c7555dbef69d5bf4133debc53b87ab8d6a44d411665/multidict-6.7.1-cp311-cp311-win_arm64.whl", hash = "sha256:b8c990b037d2fff2f4e33d3f21b9b531c5745b33a49a7d6dbe7a177266af44f6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/81/08/7036c080d7117f28a4af526d794aab6a84463126db031b007717c1a6676e/multidict-6.7.1-py3-none-any.whl", hash = "sha256:55d97cc6dae627efa6a6e548885712d4864b81110ac76fa4e534c03819fa4a56" }, +] + +[[package]] +name = "mypy" +version = "2.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "ast-serialize" }, + { name = "librt", marker = "platform_python_implementation != 'PyPy'" }, + { name = "mypy-extensions" }, + { name = "pathspec" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/82/15/cca9d88503549ed6fedeaa1d448cdddd542ee8a490232d732e278036fbf2/mypy-2.1.0.tar.gz", hash = "sha256:81e76ad12c2d804512e9b13240d1588316531bfba07558286078bfbce9613633" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/0a/a1/639f3024794a2a15899cb90707fe02e044c4412794c39c5769fd3df2e2ef/mypy-2.1.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:a683016b16fe2f572dc04c72be7ee0504ac1605a265d0200f5cea695fb788f41" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3b/08/9a585dea4325f20d8b80dc78623fa50d1fd2173b710f6237afd6ba6ab39b/mypy-2.1.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:1a293c534adb55271fef24a26da04b855540a8c13cc07bc5917b9fd2c394f2ca" }, + { url = "https://mirrors.aliyun.com/pypi/packages/81/dc/7c42cc9c6cb01e8eb09961f1f738741d3e9c7e9d5c5b30ec69222625cd5f/mypy-2.1.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7406f4d048e71e576f5356d317e5b0a9e666dfd966bd99f9d14ca06e1a341538" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d4/fa/285946c33bce716e082c11dfeee9ee196eaf1f5042efb3581a31f9f205e4/mypy-2.1.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e0210d626fc8b31ccc90233754c7bc90e1f43205e85d96387f7db1285b55c398" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2b/83/82397f48af6c27e295d57979ded8490c9829040152cf7571b2f026aeb9a0/mypy-2.1.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:3712c20deed54e814eaaa825603bada8ea1c390670a397c95b98405347acc563" }, + { url = "https://mirrors.aliyun.com/pypi/packages/40/68/b02dec39057b88eb03dc0aa854732e26e8361f34f9d0e20c7614967d1eba/mypy-2.1.0-cp311-cp311-win_amd64.whl", hash = "sha256:fcaa0e479066e31f7cceb6a3bea39cb22b2ff51a6b2f24f193d19179ba17c389" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cf/a8/ea3dcbef31f99b634f2ee23bb0321cbc8c1b388b76a861eb849f13c347dc/mypy-2.1.0-cp311-cp311-win_arm64.whl", hash = "sha256:0b1a5260c95aa443083f9ed3592662941951bca3d4ca224a5dc517c38b7cf666" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0d/2a/13ca1f292f6db1b98ff495ef3467736b331621c5917cad984b7043e7348d/mypy-2.1.0-py3-none-any.whl", hash = "sha256:a663814603a5c563fb87a4f96fb473eeb30d1f5a4885afcf44f9db000a366289" }, +] + +[[package]] +name = "mypy-extensions" +version = "1.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/a2/6e/371856a3fb9d31ca8dac321cda606860fa4548858c0cc45d9d1d4ca2628b/mypy_extensions-1.1.0.tar.gz", hash = "sha256:52e68efc3284861e772bbcd66823fde5ae21fd2fdb51c62a211403730b916558" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/79/7b/2c79738432f5c924bef5071f933bcc9efd0473bac3b4aa584a6f7c1c8df8/mypy_extensions-1.1.0-py3-none-any.whl", hash = "sha256:1be4cccdb0f2482337c4743e60421de3a356cd97508abadd57d47403e94f5505" }, +] + +[[package]] +name = "narwhals" +version = "2.22.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/62/3c/c4ef2164a71c1a63d7f1ae411c4082c5fa872405106db60a4b7114989ad7/narwhals-2.22.1.tar.gz", hash = "sha256:d62920805a0a43b7ff8b54b0c0d3142d796f8a9301836ada37e573d6a33cbcd9" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/48/ca/36339329c4604adbcc99c899b7eb1ce1a555c499b6a6860757dc9bfed36d/narwhals-2.22.1-py3-none-any.whl", hash = "sha256:60567d774edf77db53906f89d9fbd164e66e56d66d388e1e6990f17ac33cfb53" }, +] + +[[package]] +name = "networkx" +version = "3.6.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/6a/51/63fe664f3908c97be9d2e4f1158eb633317598cfa6e1fc14af5383f17512/networkx-3.6.1.tar.gz", hash = "sha256:26b7c357accc0c8cde558ad486283728b65b6a95d85ee1cd66bafab4c8168509" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/9e/c9/b2622292ea83fbb4ec318f5b9ab867d0a28ab43c5717bb85b0a5f6b3b0a4/networkx-3.6.1-py3-none-any.whl", hash = "sha256:d47fbf302e7d9cbbb9e2555a0d267983d2aa476bac30e90dfbe5669bd57f3762" }, +] + +[[package]] +name = "nltk" +version = "3.9.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "click" }, + { name = "joblib" }, + { name = "regex" }, + { name = "tqdm" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/74/a1/b3b4adf15585a5bc4c357adde150c01ebeeb642173ded4d871e89468767c/nltk-3.9.4.tar.gz", hash = "sha256:ed03bc098a40481310320808b2db712d95d13ca65b27372f8a403949c8b523d0" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/9d/91/04e965f8e717ba0ab4bdca5c112deeab11c9e750d94c4d4602f050295d39/nltk-3.9.4-py3-none-any.whl", hash = "sha256:f2fa301c3a12718ce4a0e9305c5675299da5ad9e26068218b69d692fda84828f" }, +] + +[[package]] +name = "numpy" +version = "2.4.6" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d0/ad/fed0499ce6a338d2a03ebae59cd15093910c8875328855781952abf6c2fe/numpy-2.4.6.tar.gz", hash = "sha256:f3a3570c4a2a16746ac2c31a7c7c7b0c186b95ce902e33db6f28094ed7387dda" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b3/49/ec46835a70be8fa6446c495126ac84fdb28cb2558e1620ffb87a10c8b64c/numpy-2.4.6-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:0280e0356c0829a18d9de1cb7eee50ec22ca639878d7240307ca0943d73cd2c4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0e/0d/f5957185c0ee2f3e12f78715aa9e3b353fd83633316c8532b38faa37e3f6/numpy-2.4.6-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:110f8b71aacb688ec69062bb7f6938a0f8acb01b7c1c4beb453c65b6d234584d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ad/40/40a40ee0ddf7ceb782c49af278894b686e586d65d8c1889c8b5da01a3d7d/numpy-2.4.6-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:4cfe66903cc32a9921a6733d96b19bb6abf310397581bbad89c228f5abaf0ee8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/63/13/f9a8046535cb21deae82f8d03de9617e08882d274fad2539630761888228/numpy-2.4.6-cp311-cp311-macosx_14_0_x86_64.whl", hash = "sha256:8155154c7c691289fe18f510b5d4657c68c67989f293f0535a91360392ff6538" }, + { url = "https://mirrors.aliyun.com/pypi/packages/33/a8/6fa8c1a345a8c85dbb21932c447bee07c30a2c2a3f31e369c0a84b300147/numpy-2.4.6-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0ab0a9c4ffb1a6d95ef519fe4247dba8eb6b18ad93999f76b7f657039acabd47" }, + { url = "https://mirrors.aliyun.com/pypi/packages/02/03/74fe2a4cb3817d94d86402f2506554130a2f01414e299b5a843e5a8a957f/numpy-2.4.6-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:89cd468399cfd2504718f0ba50e410dca55a170b61a02ad92bb18c8a65186e93" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c5/80/3615be3313f7e7696609bc194b9f0101da809df79e859bdb84e0cd043f46/numpy-2.4.6-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:c2d37ab77531417474168eb79d6d80b14f821a966818505d03013d0833edb7a8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ca/ac/a691e0fe2675e370d0e08ff905adc49a1c8830e8cae03efe4477e92cd55d/numpy-2.4.6-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:f407cb6b8e9d6d8c626bc73c945db1706035af8fd632295547bf1c9e46d092d6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/15/a7/9bc1cd626d7bf6869bfedf27b91b6ab5dd607758bf8e959d6fa80c6a59cb/numpy-2.4.6-cp311-cp311-win32.whl", hash = "sha256:ddea102b48f9e339f3948bf22040944184627a30fdf7f858667673b9c5f033c8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c5/31/7fc6239c12bce7e931463251cca4426c465e1876ba3cc785402ef4dd8f4e/numpy-2.4.6-cp311-cp311-win_amd64.whl", hash = "sha256:1e254a00cdf42b1e4d5b3d68d33af63268d41340d8885df2ab6470f2e1500147" }, + { url = "https://mirrors.aliyun.com/pypi/packages/27/83/140f85a466595a16382996a1bf06b2b54bcd597488921b0c9daaeeda72af/numpy-2.4.6-cp311-cp311-win_arm64.whl", hash = "sha256:ed9749eef4cbd126da3dc1d6bcb3a57f5eb7ac6a6484146bdbf743f552dfc577" }, + { url = "https://mirrors.aliyun.com/pypi/packages/de/12/b422cc84439adc0d00de605bf4a308890ae5c26f2c71fbd73e5d08fbb0dd/numpy-2.4.6-pp311-pypy311_pp73-macosx_10_15_x86_64.whl", hash = "sha256:55cced7c52e981362f708ad635198e97a752dfba412cc03c23bbf3bd8d5cd662" }, + { url = "https://mirrors.aliyun.com/pypi/packages/44/53/f481bef68011740f8849418d82db07230e825013f31f4eef5ba5b805316a/numpy-2.4.6-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:d6da64deb6b8ed903e7560180a92f2d804ee1ba5eeb849ac2748b8c1aba1f6d7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7f/57/42ed575c10ced8af951d426bc4e1f8aff16fd851db33f067036215a7f860/numpy-2.4.6-pp311-pypy311_pp73-macosx_14_0_arm64.whl", hash = "sha256:68a5124b13fa6cc2086764a20005d30bc0548146f7f5322f02fce212ca14317f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6a/ef/f66cc724fcc36c1e364c67f51ae9146090b8b584f27d58b97fdae3edd737/numpy-2.4.6-pp311-pypy311_pp73-macosx_14_0_x86_64.whl", hash = "sha256:948424b06129ce883307e8cff868c31396d8dc7630a59c61d70d98dbe70f222c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1a/9c/c531f2293b91265d8b48e9b329f54fdd7ffae73cb4134ea10cca4237e9cc/numpy-2.4.6-pp311-pypy311_pp73-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5dbbdb29840ca3d91ee0fece42fc29278886d908280bfec0a5846c6f901a3eb0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1a/b0/413077f6b1153ed3cba361401c6783bbad6114804a000cc22eb71c13e190/numpy-2.4.6-pp311-pypy311_pp73-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8ad03c0965fb3c692200e74d458ca28c1dbb4ce96f9a479a8aa041ad5fabca02" }, + { url = "https://mirrors.aliyun.com/pypi/packages/15/ce/e5ec180bc41812edcd8daeb8639d205622c0e8c02259d8ab25a0201b3c2a/numpy-2.4.6-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:2803abfebfc990042cd494d8ce2d5f82e9d847af6d35ec486923aa19dbad5e73" }, +] + +[[package]] +name = "nvidia-cublas" +version = "13.1.1.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "nvidia-cuda-nvrtc", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a7/a1/0bd24ee8c8d03adac032fd2909426a00c88f8c57961b1277ded97f91119f/nvidia_cublas-13.1.1.3-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:b7a210458267ac818974c53038fbec2e969d5c99f305ab15c72522fa9f001dd5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3b/cd/154ca20c38269e05eff77c1464e6c1da89f50a6390b565e9d82e06bc11e1/nvidia_cublas-13.1.1.3-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:37936a16db8fe4ac1f065c2139360608a543a09275cb1a1af612e08cfa065436" }, +] + +[[package]] +name = "nvidia-cuda-cupti" +version = "13.0.85" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/2a/2a/80353b103fc20ce05ef51e928daed4b6015db4aaa9162ed0997090fe2250/nvidia_cuda_cupti-13.0.85-py3-none-manylinux_2_25_aarch64.whl", hash = "sha256:796bd679890ee55fb14a94629b698b6db54bcfd833d391d5e94017dd9d7d3151" }, + { url = "https://mirrors.aliyun.com/pypi/packages/33/6d/737d164b4837a9bbd202f5ae3078975f0525a55730fe871d8ed4e3b952b0/nvidia_cuda_cupti-13.0.85-py3-none-manylinux_2_25_x86_64.whl", hash = "sha256:4eb01c08e859bf924d222250d2e8f8b8ff6d3db4721288cf35d14252a4d933c8" }, +] + +[[package]] +name = "nvidia-cuda-nvrtc" +version = "13.0.88" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c3/68/483a78f5e8f31b08fb1bb671559968c0ca3a065ac7acabfc7cee55214fd6/nvidia_cuda_nvrtc-13.0.88-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:ad9b6d2ead2435f11cbb6868809d2adeeee302e9bb94bcf0539c7a40d80e8575" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b7/dc/6bb80850e0b7edd6588d560758f17e0550893a1feaf436807d64d2da040f/nvidia_cuda_nvrtc-13.0.88-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:d27f20a0ca67a4bb34268a5e951033496c5b74870b868bacd046b1b8e0c3267b" }, +] + +[[package]] +name = "nvidia-cuda-runtime" +version = "13.0.96" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/87/4f/17d7b9b8e285199c58ce28e31b5c5bbaa4d8271af06a89b6405258245de2/nvidia_cuda_runtime-13.0.96-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:ef9bcbe90493a2b9d810e43d249adb3d02e98dd30200d86607d8d02687c43f55" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2e/24/d1558f3b68b1d26e706813b1d10aa1d785e4698c425af8db8edc3dced472/nvidia_cuda_runtime-13.0.96-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:7f82250d7782aa23b6cfe765ecc7db554bd3c2870c43f3d1821f1d18aebf0548" }, +] + +[[package]] +name = "nvidia-cudnn-cu13" +version = "9.20.0.48" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "nvidia-cublas", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/56/c5/83384d846b2fd17c44bd499b36c75a45ed4f095fbbb2252294e89cea5c5c/nvidia_cudnn_cu13-9.20.0.48-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:e31454ae00094b0c55319d9d15b6fa2fc50a9e1c0f5c8c80fb75258234e731e1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6e/5e/edb9c0ae051602c3ccaffe424256463636d639e27d7f302dde9975ef9e7a/nvidia_cudnn_cu13-9.20.0.48-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:0c45dd8eeb50b603f07995b1b300c62ffe6a1980482b82b3bcf94a4ca9d49304" }, +] + +[[package]] +name = "nvidia-cufft" +version = "12.0.0.61" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "nvidia-nvjitlink", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/8b/ae/f417a75c0259e85c1d2f83ca4e960289a5f814ed0cea74d18c353d3e989d/nvidia_cufft-12.0.0.61-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:2708c852ef8cd89d1d2068bdbece0aa188813a0c934db3779b9b1faa8442e5f5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a8/2f/7b57e29836ea8714f81e9898409196f47d772d5ddedddf1592eadb8ab743/nvidia_cufft-12.0.0.61-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:6c44f692dce8fd5ffd3e3df134b6cdb9c2f72d99cf40b62c32dde45eea9ddad3" }, +] + +[[package]] +name = "nvidia-cufile" +version = "1.15.1.6" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/3f/70/4f193de89a48b71714e74602ee14d04e4019ad36a5a9f20c425776e72cd6/nvidia_cufile-1.15.1.6-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:08a3ecefae5a01c7f5117351c64f17c7c62efa5fffdbe24fc7d298da19cd0b44" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ab/73/cc4a14c9813a8a0d509417cf5f4bdaba76e924d58beb9864f5a7baceefbf/nvidia_cufile-1.15.1.6-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:bdc0deedc61f548bddf7733bdc216456c2fdb101d020e1ab4b88d232d5e2f6d1" }, +] + +[[package]] +name = "nvidia-curand" +version = "10.4.0.35" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1e/72/7c2ae24fb6b63a32e6ae5d241cc65263ea18d08802aaae087d9f013335a2/nvidia_curand-10.4.0.35-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:133df5a7509c3e292aaa2b477afd0194f06ce4ea24d714d616ff36439cee349a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a5/9f/be0a41ca4a4917abf5cb9ae0daff1a6060cc5de950aec0396de9f3b52bc5/nvidia_curand-10.4.0.35-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:1aee33a5da6e1db083fe2b90082def8915f30f3248d5896bcec36a579d941bfc" }, +] + +[[package]] +name = "nvidia-cusolver" +version = "12.0.4.66" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "nvidia-cublas", marker = "sys_platform != 'win32'" }, + { name = "nvidia-cusparse", marker = "sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c8/c3/b30c9e935fc01e3da443ec0116ed1b2a009bb867f5324d3f2d7e533e776b/nvidia_cusolver-12.0.4.66-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:02c2457eaa9e39de20f880f4bd8820e6a1cfb9f9a34f820eb12a155aa5bc92d2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5f/67/cba3777620cdacb99102da4042883709c41c709f4b6323c10781a9c3aa34/nvidia_cusolver-12.0.4.66-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:0a759da5dea5c0ea10fd307de75cdeb59e7ea4fcb8add0924859b944babf1112" }, +] + +[[package]] +name = "nvidia-cusparse" +version = "12.6.3.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "nvidia-nvjitlink", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f8/94/5c26f33738ae35276672f12615a64bd008ed5be6d1ebcb23579285d960a9/nvidia_cusparse-12.6.3.3-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:80bcc4662f23f1054ee334a15c72b8940402975e0eab63178fc7e670aa59472c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fa/18/623c77619c31d62efd55302939756966f3ecc8d724a14dab2b75f1508850/nvidia_cusparse-12.6.3.3-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:2b3c89c88d01ee0e477cb7f82ef60a11a4bcd57b6b87c33f789350b59759360b" }, +] + +[[package]] +name = "nvidia-cusparselt-cu13" +version = "0.8.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/46/e1/cdc1797eadf82d3a9a575a19b33fdc871a97edbec42c00b5b5e914f4aff4/nvidia_cusparselt_cu13-0.8.1-py3-none-manylinux2014_aarch64.whl", hash = "sha256:4dca476c50bf4780d46cd0bfbd82e2bc10a08e4fef7950917ce8d7578d22a23f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/34/7d/2661f2fb3ac4302f3a246f5fc030213ac60c1fe0bce84f9783dbd831dbb7/nvidia_cusparselt_cu13-0.8.1-py3-none-manylinux2014_x86_64.whl", hash = "sha256:786ce87568c303fadb5afcc7102d454cd3040d75f6f8626f5db460d1871f4dd0" }, +] + +[[package]] +name = "nvidia-nccl-cu13" +version = "2.29.7" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/72/0d/daf50d44177ee0cbc7ff0a0c91eb5ff676c82be42f9a970bc7597f440c3a/nvidia_nccl_cu13-2.29.7-py3-none-manylinux_2_18_aarch64.whl", hash = "sha256:674a12383e3c38a1bcccae7d4f3633b37852230b6047883cb2f4c2d1b36d9bf5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/67/f4/58e4e91b6919367c7aafb8e36fce9aad1a3047e536bf7e2fd560927d3a4c/nvidia_nccl_cu13-2.29.7-py3-none-manylinux_2_18_x86_64.whl", hash = "sha256:edd81538446786ec3b73972543e53bb43bcaf0bfc8ef76cb679fcc390ffe136d" }, +] + +[[package]] +name = "nvidia-nvjitlink" +version = "13.0.88" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/56/7a/123e033aaff487c77107195fa5a2b8686795ca537935a24efae476c41f05/nvidia_nvjitlink-13.0.88-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:13a74f429e23b921c1109976abefacc69835f2f433ebd323d3946e11d804e47b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ab/2c/93c5250e64df4f894f1cbb397c6fd71f79813f9fd79d7cd61de3f97b3c2d/nvidia_nvjitlink-13.0.88-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:e931536ccc7d467a98ba1d8b89ff7fa7f1fa3b13f2b0069118cd7f47bff07d0c" }, +] + +[[package]] +name = "nvidia-nvshmem-cu13" +version = "3.4.5" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/dc/0f/05cc9c720236dcd2db9c1ab97fff629e96821be2e63103569da0c9b72f19/nvidia_nvshmem_cu13-3.4.5-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:6dc2a197f38e5d0376ad52cd1a2a3617d3cdc150fd5966f4aee9bcebb1d68fe9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3c/35/a9bf80a609e74e3b000fef598933235c908fcefcef9026042b8e6dfde2a9/nvidia_nvshmem_cu13-3.4.5-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:290f0a2ee94c9f3687a02502f3b9299a9f9fe826e6d0287ee18482e78d495b80" }, +] + +[[package]] +name = "nvidia-nvtx" +version = "13.0.85" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c2/f3/d86c845465a2723ad7e1e5c36dcd75ddb82898b3f53be47ebd429fb2fa5d/nvidia_nvtx-13.0.85-py3-none-manylinux1_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:4936d1d6780fbe68db454f5e72a42ff64d1fd6397df9f363ae786930fd5c1cd4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a8/64/3708a90d1ebe202ffdeb7185f878a3c84d15c2b2c31858da2ce0583e2def/nvidia_nvtx-13.0.85-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:cb7780edb6b14107373c835bf8b72e7a178bac7367e23da7acb108f973f157a6" }, +] + +[[package]] +name = "olefile" +version = "0.47" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/69/1b/077b508e3e500e1629d366249c3ccb32f95e50258b231705c09e3c7a4366/olefile-0.47.zip", hash = "sha256:599383381a0bf3dfbd932ca0ca6515acd174ed48870cbf7fee123d698c192c1c" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/17/d3/b64c356a907242d719fc668b71befd73324e47ab46c8ebbbede252c154b2/olefile-0.47-py2.py3-none-any.whl", hash = "sha256:543c7da2a7adadf21214938bb79c83ea12b473a4b6ee4ad4bf854e7715e13d1f" }, +] + +[[package]] +name = "onnxruntime" +version = "1.20.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "coloredlogs" }, + { name = "flatbuffers" }, + { name = "numpy" }, + { name = "packaging" }, + { name = "protobuf" }, + { name = "sympy" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/95/8d/2634e2959b34aa8a0037989f4229e9abcfa484e9c228f99633b3241768a6/onnxruntime-1.20.1-cp311-cp311-macosx_13_0_universal2.whl", hash = "sha256:06bfbf02ca9ab5f28946e0f912a562a5f005301d0c419283dc57b3ed7969bb7b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a5/da/c44bf9bd66cd6d9018a921f053f28d819445c4d84b4dd4777271b0fe52a2/onnxruntime-1.20.1-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f6243e34d74423bdd1edf0ae9596dd61023b260f546ee17d701723915f06a9f7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/11/ac/4120dfb74c8e45cce1c664fc7f7ce010edd587ba67ac41489f7432eb9381/onnxruntime-1.20.1-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5eec64c0269dcdb8d9a9a53dc4d64f87b9e0c19801d9321246a53b7eb5a7d1bc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/12/f1/cefacac137f7bb7bfba57c50c478150fcd3c54aca72762ac2c05ce0532c1/onnxruntime-1.20.1-cp311-cp311-win32.whl", hash = "sha256:a19bc6e8c70e2485a1725b3d517a2319603acc14c1f1a017dda0afe6d4665b41" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2c/2d/2d4d202c0bcfb3a4cc2b171abb9328672d7f91d7af9ea52572722c6d8d96/onnxruntime-1.20.1-cp311-cp311-win_amd64.whl", hash = "sha256:8508887eb1c5f9537a4071768723ec7c30c28eb2518a00d0adcd32c89dea3221" }, +] + +[[package]] +name = "openai" +version = "2.41.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "anyio" }, + { name = "distro" }, + { name = "httpx" }, + { name = "jiter" }, + { name = "pydantic" }, + { name = "sniffio" }, + { name = "tqdm" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/40/36/4c926a91554483977608951360c18c2e911592785eb87a6437813f6123f7/openai-2.41.1.tar.gz", hash = "sha256:23d617a0432457ad844973bee8f540be9da90894f7c5686852d2d365da058f57" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/20/74/925d7b3892927e9804aaf58d374a45dc28e4420ff90e992272b77286343e/openai-2.41.1-py3-none-any.whl", hash = "sha256:a939565f350cb7443cb843b801b88c716ac8024b492fb94ca269d5f6b1bbefd6" }, +] + +[[package]] +name = "openpyxl" +version = "3.1.5" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "et-xmlfile" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/3d/f9/88d94a75de065ea32619465d2f77b29a0469500e99012523b91cc4141cd1/openpyxl-3.1.5.tar.gz", hash = "sha256:cf0e3cf56142039133628b5acffe8ef0c12bc902d2aadd3e0fe5878dc08d1050" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c0/da/977ded879c29cbd04de313843e76868e6e13408a94ed6b987245dc7c8506/openpyxl-3.1.5-py2.py3-none-any.whl", hash = "sha256:5282c12b107bffeef825f4617dc029afaf41d0ea60823bbb665ef3079dc79de2" }, +] + +[[package]] +name = "packaging" +version = "26.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d7/f1/e7a6dd94a8d4a5626c03e4e99c87f241ba9e350cd9e6d75123f992427270/packaging-26.2.tar.gz", hash = "sha256:ff452ff5a3e828ce110190feff1178bb1f2ea2281fa2075aadb987c2fb221661" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/df/b2/87e62e8c3e2f4b32e5fe99e0b86d576da1312593b39f47d8ceef365e95ed/packaging-26.2-py3-none-any.whl", hash = "sha256:5fc45236b9446107ff2415ce77c807cee2862cb6fac22b8a73826d0693b0980e" }, +] + +[[package]] +name = "pandas" +version = "3.0.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "numpy" }, + { name = "python-dateutil" }, + { name = "tzdata", marker = "sys_platform == 'emscripten' or sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/f8/87/4341c6252d1c47b08768c3d25ac487362bf403f0313ddae4a2a26c9b1b4c/pandas-3.0.3.tar.gz", hash = "sha256:696a4a00a2a2a35d4e5deb3fc946641b96c944f02230e4f76137fe35d806c4fc" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/42/16/b5c76b838fd9bf6ce84d3a53346b8874ec05c5f0040d75ef2c320100cd2a/pandas-3.0.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:455f6f8139d4282188f526868dbc3c828470e88a3d9d59a891bd46a455f21b98" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5a/b0/a4ffc4ae74d2d822200dcc46898987d8eb6032d1e2b219cae39da6f5cbcc/pandas-3.0.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:4e15135e2ee5df1063313e2425ceef8ac0f4ae775893815b0923651b806a5639" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2e/b2/3323601a52caee42c019e370090ca4544b241437240ca04f786cce82b0cf/pandas-3.0.3-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:05f1f1752b8533ea03f7f39a9c15b1a058d067bb48f4748948e7a8691e0510f2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/32/f1/bbecd2f867b97abebe0f9b53d750f862251b40337e061b36676ded3d920f/pandas-3.0.3-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8a1e45c80cceb3b4a21bc5939d52e8cbd8d9b7305309219d59e9754d9ce09e27" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7f/4f/eafabf2d5fae5adf143b4d18d3706c5efdc368a7c4eb1ee8a3eddabbd0f6/pandas-3.0.3-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:14da8316da4d0c5a77618425996bfb1248ca87fc2c1486e6fde4652bd18b5824" }, + { url = "https://mirrors.aliyun.com/pypi/packages/49/44/1eb20389301b57b19cc099a1c2f662501f72f08a65f912d05822613c1532/pandas-3.0.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:a55066a0505dae0ba2b50a46637db34b46f9094c65c5d4800794ef6335010938" }, + { url = "https://mirrors.aliyun.com/pypi/packages/eb/62/c321f13b5ba1819fc8dca456c7fce578da2dcfecff1abbf0eaddf8406c0f/pandas-3.0.3-cp311-cp311-win_amd64.whl", hash = "sha256:6674ab18ad8c57802867264b00e15e7bb904700cdd9046e3b2fa1fce237439ea" }, + { url = "https://mirrors.aliyun.com/pypi/packages/53/85/1b7f563ebc6357c27233a02a96b589bcce1fa9c6eb89fb4f0e56421d277e/pandas-3.0.3-cp311-cp311-win_arm64.whl", hash = "sha256:5cc09a68b3120e0f54870dede8287a7bb1fa463907e4fcec1ea77cab6179bf7a" }, +] + +[[package]] +name = "patchright" +version = "1.60.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "greenlet" }, + { name = "pyee" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/71/26/c1e858fd1acc63e410b3d33243955f36d2a0814487b97a7aa604ad2baffd/patchright-1.60.1-py3-none-macosx_10_13_x86_64.whl", hash = "sha256:e9492100d4e2a85ff92fc3a668dd16dee03f21df6e559c7b9f7c71e86ff48c6b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/55/dd/2dd8e4e02489ec8fd57ad93dec9ef444b6f42adcc4fe95df30237c92841d/patchright-1.60.1-py3-none-macosx_11_0_arm64.whl", hash = "sha256:20bd806df2469b451ccd2ea10f5f944ceb0e0d83c716f5752b0c956c1ee59476" }, + { url = "https://mirrors.aliyun.com/pypi/packages/54/cc/0fa0bedec61045fd9068682e3695557b622c483f829d341093879ec8dbd9/patchright-1.60.1-py3-none-macosx_11_0_universal2.whl", hash = "sha256:9fd15a64c0ca80740dc2a3f41cda336a06a2ed6068d0ab893172654290b06e6b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ce/c2/4b8f69de0a20d90792980c43c0e60b10b801e08cf0224ccc8a8266e1fffb/patchright-1.60.1-py3-none-manylinux1_x86_64.whl", hash = "sha256:547e7bfb813102309789cc42933780e5fdf7c4727de59fb2791e64bd1298a7f3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/db/fc/9fd6a70818cf0bc3b62574af483d762963cb65ca0a369826a0536faffe41/patchright-1.60.1-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:023945a2fd30219a284721ca36385bd44075ae7b53071dc1da38036b6dbe88ec" }, + { url = "https://mirrors.aliyun.com/pypi/packages/08/bc/81fb621e5ab4131e6f324c9ba6cb2a9f3b146c92f12484bec95efe8c8347/patchright-1.60.1-py3-none-win32.whl", hash = "sha256:05b98a6afdbe7e6645fe223009c47cc8e7859df55fd8ce9d8a9925b3389b0ee1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/74/8e/fff80350ed2c2c1f62145d667070799c60ca9e9b28d54e4f751f0b4f8da6/patchright-1.60.1-py3-none-win_amd64.whl", hash = "sha256:51b306ed55cd58f1bca24641458f5c9f7e86a1f1727dcffdead669cfe4c0a485" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ee/b8/b6d1bfe98a420c1ccfb2f23a7bb2b4cdb40170907bae638e7ae92abd287c/patchright-1.60.1-py3-none-win_arm64.whl", hash = "sha256:f795728c1e27fc226dbe203c1aec713a537f01be963474fe0f3691f5e6457f9f" }, +] + +[[package]] +name = "pathspec" +version = "1.1.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/5a/82/42f767fc1c1143d6fd36efb827202a2d997a375e160a71eb2888a925aac1/pathspec-1.1.1.tar.gz", hash = "sha256:17db5ecd524104a120e173814c90367a96a98d07c45b2e10c2f3919fff91bf5a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f1/d9/7fb5aa316bc299258e68c73ba3bddbc499654a07f151cba08f6153988714/pathspec-1.1.1-py3-none-any.whl", hash = "sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189" }, +] + +[[package]] +name = "pdfminer-six" +version = "20260107" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "charset-normalizer" }, + { name = "cryptography" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/34/a4/5cec1112009f0439a5ca6afa8ace321f0ab2f48da3255b7a1c8953014670/pdfminer_six-20260107.tar.gz", hash = "sha256:96bfd431e3577a55a0efd25676968ca4ce8fd5b53f14565f85716ff363889602" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/20/8b/28c4eaec9d6b036a52cb44720408f26b1a143ca9bce76cc19e8f5de00ab4/pdfminer_six-20260107-py3-none-any.whl", hash = "sha256:366585ba97e80dffa8f00cebe303d2f381884d8637af4ce422f1df3ef38111a9" }, +] + +[[package]] +name = "pdfplumber" +version = "0.11.10" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "pdfminer-six" }, + { name = "pillow" }, + { name = "pypdfium2" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/05/56/6f450312ba05a27d7713b73857c1a25100dbda04fbc1331b13fb227a607d/pdfplumber-0.11.10.tar.gz", hash = "sha256:b95b2d28c66efb0a794a83b88c6c6aea5987532a445d20a1cbcfa657022e6e57" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a2/9a/07d658e1e7fad860f1c541ab941348125dbdab773be3a0afaf32361866c7/pdfplumber-0.11.10-py3-none-any.whl", hash = "sha256:7741ea81bf165b474b153e6789d10d18e06b6ddcf3ec84289c3ef2fed6802580" }, +] + +[[package]] +name = "pillow" +version = "12.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/8c/21/c2bcdd5906101a30244eaffc1b6e6ce71a31bd0742a01eb89e660ebfac2d/pillow-12.2.0.tar.gz", hash = "sha256:a830b1a40919539d07806aa58e1b114df53ddd43213d9c8b75847eee6c0182b5" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/68/e1/748f5663efe6edcfc4e74b2b93edfb9b8b99b67f21a854c3ae416500a2d9/pillow-12.2.0-cp311-cp311-macosx_10_10_x86_64.whl", hash = "sha256:8be29e59487a79f173507c30ddf57e733a357f67881430449bb32614075a40ab" }, + { url = "https://mirrors.aliyun.com/pypi/packages/47/a1/d5ff69e747374c33a3b53b9f98cca7889fce1fd03d79cdc4e1bccc6c5a87/pillow-12.2.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:71cde9a1e1551df7d34a25462fc60325e8a11a82cc2e2f54578e5e9a1e153d65" }, + { url = "https://mirrors.aliyun.com/pypi/packages/df/21/e3fbdf54408a973c7f7f89a23b2cb97a7ef30c61ab4142af31eee6aebc88/pillow-12.2.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:f490f9368b6fc026f021db16d7ec2fbf7d89e2edb42e8ec09d2c60505f5729c7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d3/f1/00b7278c7dd52b17ad4329153748f87b6756ec195ff786c2bdf12518337d/pillow-12.2.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:8bd7903a5f2a4545f6fd5935c90058b89d30045568985a71c79f5fd6edf9b91e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ad/cf/220a5994ef1b10e70e85748b75649d77d506499352be135a4989c957b701/pillow-12.2.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3997232e10d2920a68d25191392e3a4487d8183039e1c74c2297f00ed1c50705" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e9/bd/e51a61b1054f09437acfbc2ff9106c30d1eb76bc1453d428399946781253/pillow-12.2.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e74473c875d78b8e9d5da2a70f7099549f9eb37ded4e2f6a463e60125bccd176" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6b/3d/45132c57d5fb4b5744567c3817026480ac7fc3ce5d4c47902bc0e7f6f853/pillow-12.2.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:56a3f9c60a13133a98ecff6197af34d7824de9b7b38c3654861a725c970c197b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7d/2e/9df2fc1e82097b1df3dce58dc43286aa01068e918c07574711fcc53e6fb4/pillow-12.2.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:90e6f81de50ad6b534cab6e5aef77ff6e37722b2f5d908686f4a5c9eba17a909" }, + { url = "https://mirrors.aliyun.com/pypi/packages/bd/2e/2941e42858ebb67e50ae741473de81c2984e6eff7b397017623c676e2e8d/pillow-12.2.0-cp311-cp311-win32.whl", hash = "sha256:8c984051042858021a54926eb597d6ee3012393ce9c181814115df4c60b9a808" }, + { url = "https://mirrors.aliyun.com/pypi/packages/69/42/836b6f3cd7f3e5fa10a1f1a5420447c17966044c8fbf589cc0452d5502db/pillow-12.2.0-cp311-cp311-win_amd64.whl", hash = "sha256:6e6b2a0c538fc200b38ff9eb6628228b77908c319a005815f2dde585a0664b60" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c2/88/549194b5d6f1f494b485e493edc6693c0a16f4ada488e5bd974ed1f42fad/pillow-12.2.0-cp311-cp311-win_arm64.whl", hash = "sha256:9a8a34cc89c67a65ea7437ce257cea81a9dad65b29805f3ecee8c8fe8ff25ffe" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4e/b7/2437044fb910f499610356d1352e3423753c98e34f915252aafecc64889f/pillow-12.2.0-pp311-pypy311_pp73-macosx_10_15_x86_64.whl", hash = "sha256:0538bd5e05efec03ae613fd89c4ce0368ecd2ba239cc25b9f9be7ed426b0af1f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f6/f4/8316e31de11b780f4ac08ef3654a75555e624a98db1056ecb2122d008d5a/pillow-12.2.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:394167b21da716608eac917c60aa9b969421b5dcbbe02ae7f013e7b85811c69d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d4/37/664fca7201f8bb2aa1d20e2c3d5564a62e6ae5111741966c8319ca802361/pillow-12.2.0-pp311-pypy311_pp73-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:5d04bfa02cc2d23b497d1e90a0f927070043f6cbf303e738300532379a4b4e0f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/49/62/5b0ed78fce87346be7a5cfcfaaad91f6a1f98c26f86bdbafa2066c647ef6/pillow-12.2.0-pp311-pypy311_pp73-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:0c838a5125cee37e68edec915651521191cef1e6aa336b855f495766e77a366e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c3/28/ec0fc38107fc32536908034e990c47914c57cd7c5a3ece4d8d8f7ffd7e27/pillow-12.2.0-pp311-pypy311_pp73-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4a6c9fa44005fa37a91ebfc95d081e8079757d2e904b27103f4f5fa6f0bf78c0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5e/8b/51b0eddcfa2180d60e41f06bd6d0a62202b20b59c68f5a132e615b75aecf/pillow-12.2.0-pp311-pypy311_pp73-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:25373b66e0dd5905ed63fa3cae13c82fbddf3079f2c8bf15c6fb6a35586324c1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/bc/60/5382c03e1970de634027cee8e1b7d39776b778b81812aaf45b694dfe9e28/pillow-12.2.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:bfa9c230d2fe991bed5318a5f119bd6780cda2915cca595393649fc118ab895e" }, +] + +[[package]] +name = "playwright" +version = "1.60.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "greenlet" }, + { name = "pyee" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/21/f0/832bd9677194908da118064eef20082f2791e3d18215cc6d9391ee2c5a67/playwright-1.60.0-py3-none-macosx_10_13_x86_64.whl", hash = "sha256:6a8cd0fec171fb3089e95e898c8bc8a6f35dea0b78b399e12fcc19427e91b1d7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/59/7b/e1d32ae8a3ed937ec2be3721c5f728b13d731a0b7c6442e0b3bec5094ac0/playwright-1.60.0-py3-none-macosx_11_0_arm64.whl", hash = "sha256:39b5420ba6145045b69ced4c5c47d4d9fe5bddfc8ff816c518913afcb25ec7a5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d7/bc/23de499ded6411c188a20c5a0dea6f0cd4ed5d2b3cc6042a5dbd3ed609aa/playwright-1.60.0-py3-none-macosx_11_0_universal2.whl", hash = "sha256:2581d0e6a3392c71f91b27460c7fd093356818dc430f48153896c8aeeaef7705" }, + { url = "https://mirrors.aliyun.com/pypi/packages/22/7b/1d679f4fced4ea94efadd17103856d8c565384f68382a1681264e46f5925/playwright-1.60.0-py3-none-manylinux1_x86_64.whl", hash = "sha256:1c2bfae7884fb3fb05b853290eab8f343d524e5016f2f1def702acbbdf14c93e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/84/c2/1528d267d4442bd2c6b8eaeab819dd52c2030bf80e89293f0ba1f687473b/playwright-1.60.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:43e66564125ee31b07a58cefb21e256d62d67d8d1713e6858df7a3019d8ed353" }, + { url = "https://mirrors.aliyun.com/pypi/packages/bb/4e/b008b6440a7a1624378041da94829956d4b8f7ab9ef5aad22d0dc3f2e26d/playwright-1.60.0-py3-none-win32.whl", hash = "sha256:ec94e416ea320711e0ad4bf185dcbf41833672961e90773e1885255d7db7b7e7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/55/f0/0541524133104f9cc20bf900870ff4a736b76a23483f3a55295ddfa58409/playwright-1.60.0-py3-none-win_amd64.whl", hash = "sha256:9566821ce6030a1f9e7146a24e19355ab0d98805fd0f9be50bb3d8fef1750c02" }, + { url = "https://mirrors.aliyun.com/pypi/packages/80/c8/210f282d278e4709cdd71b12a31af45a30a22ab3207b387e29b37e478713/playwright-1.60.0-py3-none-win_arm64.whl", hash = "sha256:6e4f6700a4c2250efff8e690a81d66e3855754fb587b6b87cf5c784014f91537" }, +] + +[[package]] +name = "pluggy" +version = "1.6.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/f9/e2/3e91f31a7d2b083fe6ef3fa267035b518369d9511ffab804f839851d2779/pluggy-1.6.0.tar.gz", hash = "sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746" }, +] + +[[package]] +name = "portalocker" +version = "3.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "pywin32", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/5e/77/65b857a69ed876e1951e88aaba60f5ce6120c33703f7cb61a3c894b8c1b6/portalocker-3.2.0.tar.gz", hash = "sha256:1f3002956a54a8c3730586c5c77bf18fae4149e07eaf1c29fc3faf4d5a3f89ac" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/4b/a6/38c8e2f318bf67d338f4d629e93b0b4b9af331f455f0390ea8ce4a099b26/portalocker-3.2.0-py3-none-any.whl", hash = "sha256:3cdc5f565312224bc570c49337bd21428bba0ef363bbcf58b9ef4a9f11779968" }, +] + +[[package]] +name = "propcache" +version = "0.5.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ec/44/c87281c333769159c50594f22610f77398a47ccbfbbf23074e744e86f87c/propcache-0.5.2.tar.gz", hash = "sha256:01c4fc7480cd0598bb4b57022df55b9ca296da7fc5a8760bd8451a7e63a7d427" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e7/f1/8a8cc1c2c7e7934ab77e0163414f736fadbc0f5e8dd9673b952355ac175b/propcache-0.5.2-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:74b70780220e2dd89175ca24b81b68b67c83db499ae611e7f2313cb329801c78" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c2/f4/651b1225e976bd1a2ba5cfba0c29d096581c2636b437e3a9a7ab6276270a/propcache-0.5.2-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:a4840ab0ae0216d952f4b53dc6d0b992bfc2bedbfe360bdd9b548bc184c08959" }, + { url = "https://mirrors.aliyun.com/pypi/packages/15/a8/8ede85d6aa1f79fc7dc2f8fd2c8d65920b8272c3892903c8a1affde48cfb/propcache-0.5.2-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:c6844ba6364fb12f403928a82cfd295ab103a2b315c77c747b2dbe4a41894ea7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7d/fe/b3551b41bbc2f5b5bb088fc6920567cd43101253e68fbaa261339eb96fe1/propcache-0.5.2-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2293949b855ce597f2826452d17c2d545fb5622379c4ea6fdf525e9b8e8a2511" }, + { url = "https://mirrors.aliyun.com/pypi/packages/83/27/ab851ebd1b7172e3e161f5f8d39e315d54a91bea246f01f4d872d3376aef/propcache-0.5.2-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:0fd59b5af35f74da48d905dcbad55449ba13be91823cb05a9bd590bbf5b61660" }, + { url = "https://mirrors.aliyun.com/pypi/packages/95/7d/466b3d18022e9897cbda9c735c493c5bd747d7a4c6f5ea1480b4cec434b6/propcache-0.5.2-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:29f9309a2e42b0d273be006fdb4be2d6c39a47f6f57d8fb1cf9f81481df81b66" }, + { url = "https://mirrors.aliyun.com/pypi/packages/27/1b/16ab7f2cf2041da2f60d156ba64c2484eadf9168075b4ff43c3ef60045af/propcache-0.5.2-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5aaa2b923c1944ac8febd6609cb373540a5563e7cbcb0fd770f75dace2eb817b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0a/67/bb777ffd907633563bf35fd859c4ce97b0512c32f4633cf5d1eb7c33512b/propcache-0.5.2-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:66ea454f095ddf5b6b14f56c064c0941c4788be11e18d2464cf643bf7203ff67" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b9/42/64f8d90b73fd9cdc1499b48057ff6d9cd2a98a25734c9bb62ecf07e87061/propcache-0.5.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:95f1e3f4760d404b13c9976c0229b2b49a3c8e2c62a9ce92efdd2b11ada75e3f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/eb/02/dba5bc03c9041f2092ea55a449caf5dfe68352c6654511b29ba0654ddb69/propcache-0.5.2-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:85341b12b9d55bad0bded24cac341bb34289469e03a11f3f583ea1cc1db0326c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/14/c0/43f649c7aa2a77a3b100d84e9dea3a483120ecb608bfe36ce49eaff517fe/propcache-0.5.2-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:26a4dca084132874e639895c3135dfad5eb20bae209f62d1aeb31b03e601c3c0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/83/c0/435dafd27f1cb4a495381dae60e25883ccfe4020bb72818e8184c1678092/propcache-0.5.2-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:3b199b9b2b3d6a7edf3183ba8a9a137a22b97f7df525feb5ae1eccf026d2a9c6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/53/ae/6e292df9135d659944e96cb3389258e4a663e5b2b5f6c217ef0ddc8d2f73/propcache-0.5.2-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:e59bc9e66329185b93dab73f210f1a37f81cb40f321501db8017c9aea15dba27" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0b/42/314ebc50d8159055411fd6b0bda322ff510e4b1f7d2e4927940ad0f6af20/propcache-0.5.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:552ffadf6ad409844bc5919c42a0a83d88314cedddaea0e41e80a8b8fffe881f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b8/9b/2da6dee38871c3c8772fabc2758325a5c9077d6d18c597737dc04dd884cd/propcache-0.5.2-cp311-cp311-win32.whl", hash = "sha256:cd416c1de191973c52ff1a12a57446bfc7642797b282d7caf2162d7d1b8aa9a0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/42/4e/f17363fb58c0afe05b067361cb6d86ed2d29de6506779a27547c4d183075/propcache-0.5.2-cp311-cp311-win_amd64.whl", hash = "sha256:44e488ef40dbb452700b2b1f8188934121f6648f52c295055662d2191959ff82" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c6/eb/6af6685077d22e8b33358d3c548e3282706a0b3cd85044ffba4e5dd08e3b/propcache-0.5.2-cp311-cp311-win_arm64.whl", hash = "sha256:54adaa85a22078d1e306304a40984dc5be99d599bf3dc0a24dc98f7daeab89ab" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3a/ed/1cdcab6ba3d6ab7feca11fc14f0eeea80755bb53ef4e892079f31b10a25f/propcache-0.5.2-py3-none-any.whl", hash = "sha256:be1ddfcbb376e3de5d2e2db1d58d6d67463e6b4f9f040c000de8e300295465fe" }, +] + +[[package]] +name = "protobuf" +version = "7.35.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/da/01/9ef0afd7999eb9badb3a768b4aedd78c86d4c65cfaf1958ab276199e76b4/protobuf-7.35.1.tar.gz", hash = "sha256:ce115a26fe0c39a2c29973d914d327e516a6455464489fe3cd1e51a1b354f81a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/10/03/8aeeb7458d22546bf64b5250ca1daeb5ff757d900e8e4a7476c6f0db843e/protobuf-7.35.1-cp310-abi3-macosx_10_9_universal2.whl", hash = "sha256:24f857477359a85c0c235261b8ba905fd51b2562f4a64ca1df5473f29850cbf6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/37/4b/dfb89eb0e652a1ff073c39a59fb5e3a83cfe9b57a2c83fa6d78270101767/protobuf-7.35.1-cp310-abi3-manylinux2014_aarch64.whl", hash = "sha256:11d6b0ec246892d85215b0a13ca6e0233cf5284b68f0ac02646427f4ff88a799" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0f/58/dc12f2cd484951524af6e3382c785869b9b3fb5e52ee95ae23add53ee8f9/protobuf-7.35.1-cp310-abi3-manylinux2014_s390x.whl", hash = "sha256:b73f9489a4b8b1c9cb1f8ed951c736392592edb24b9d6819f36d2e10b171d5b4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e4/be/5b3cfe508bfab6761414ff944e3366eb13be4fd71efcd69450f89ba39f43/protobuf-7.35.1-cp310-abi3-manylinux2014_x86_64.whl", hash = "sha256:74758715c53d7158fb76caf4f0cfdacc5329a4b1bb994f865d6cf302d413a1c4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d8/bc/6d6c7ba8709c85f8f2c390b2b118d6fb08a783676a572271851bf45a7d22/protobuf-7.35.1-cp310-abi3-win32.whl", hash = "sha256:353652e4efd0bca5b5fc2656abf8307ef351f0cf938c9eba09f0e09c20a25c30" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0a/19/8d0cb6f20a1ef7b18f1c8986ad5783f22f84cce39c6ce9a6e645ea55192e/protobuf-7.35.1-cp310-abi3-win_amd64.whl", hash = "sha256:230a75ddfc2de4806e56696ce9640c1cdfdb6543b7cfce98d42a4c0a0e7bdb87" }, + { url = "https://mirrors.aliyun.com/pypi/packages/19/c7/5f7c636ec43e0c545e28d1f1db71990108306f7bdcb89f069ba97e428e7f/protobuf-7.35.1-py3-none-any.whl", hash = "sha256:4bc97768d8fe4ad6743c8a19403e314511ed9f6d13205b687e52421c023ac1b9" }, +] + +[[package]] +name = "psutil" +version = "7.2.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/aa/c6/d1ddf4abb55e93cebc4f2ed8b5d6dbad109ecb8d63748dd2b20ab5e57ebe/psutil-7.2.2.tar.gz", hash = "sha256:0746f5f8d406af344fd547f1c8daa5f5c33dbc293bb8d6a16d80b4bb88f59372" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e7/36/5ee6e05c9bd427237b11b3937ad82bb8ad2752d72c6969314590dd0c2f6e/psutil-7.2.2-cp36-abi3-macosx_10_9_x86_64.whl", hash = "sha256:ed0cace939114f62738d808fdcecd4c869222507e266e574799e9c0faa17d486" }, + { url = "https://mirrors.aliyun.com/pypi/packages/80/c4/f5af4c1ca8c1eeb2e92ccca14ce8effdeec651d5ab6053c589b074eda6e1/psutil-7.2.2-cp36-abi3-macosx_11_0_arm64.whl", hash = "sha256:1a7b04c10f32cc88ab39cbf606e117fd74721c831c98a27dc04578deb0c16979" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b5/70/5d8df3b09e25bce090399cf48e452d25c935ab72dad19406c77f4e828045/psutil-7.2.2-cp36-abi3-manylinux2010_x86_64.manylinux_2_12_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:076a2d2f923fd4821644f5ba89f059523da90dc9014e85f8e45a5774ca5bc6f9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/63/65/37648c0c158dc222aba51c089eb3bdfa238e621674dc42d48706e639204f/psutil-7.2.2-cp36-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b0726cecd84f9474419d67252add4ac0cd9811b04d61123054b9fb6f57df6e9e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8e/13/125093eadae863ce03c6ffdbae9929430d116a246ef69866dad94da3bfbc/psutil-7.2.2-cp36-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:fd04ef36b4a6d599bbdb225dd1d3f51e00105f6d48a28f006da7f9822f2606d8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/04/78/0acd37ca84ce3ddffaa92ef0f571e073faa6d8ff1f0559ab1272188ea2be/psutil-7.2.2-cp36-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b58fabe35e80b264a4e3bb23e6b96f9e45a3df7fb7eed419ac0e5947c61e47cc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b4/90/e2159492b5426be0c1fef7acba807a03511f97c5f86b3caeda6ad92351a7/psutil-7.2.2-cp37-abi3-win_amd64.whl", hash = "sha256:eb7e81434c8d223ec4a219b5fc1c47d0417b12be7ea866e24fb5ad6e84b3d988" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8c/c7/7bb2e321574b10df20cbde462a94e2b71d05f9bbda251ef27d104668306a/psutil-7.2.2-cp37-abi3-win_arm64.whl", hash = "sha256:8c233660f575a5a89e6d4cb65d9f938126312bca76d8fe087b947b3a1aaac9ee" }, +] + +[[package]] +name = "pycparser" +version = "3.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/1b/7d/92392ff7815c21062bea51aa7b87d45576f649f16458d78b7cf94b9ab2e6/pycparser-3.0.tar.gz", hash = "sha256:600f49d217304a5902ac3c37e1281c9fe94e4d0489de643a9504c5cdfdfc6b29" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/0c/c3/44f3fbbfa403ea2a7c779186dc20772604442dde72947e7d01069cbe98e3/pycparser-3.0-py3-none-any.whl", hash = "sha256:b727414169a36b7d524c1c3e31839a521725078d7b2ff038656844266160a992" }, +] + +[[package]] +name = "pydantic" +version = "2.13.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "annotated-types" }, + { name = "pydantic-core" }, + { name = "typing-extensions" }, + { name = "typing-inspection" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/18/a5/b60d21ac674192f8ab0ba4e9fd860690f9b4a6e51ca5df118733b487d8d6/pydantic-2.13.4.tar.gz", hash = "sha256:c40756b57adaa8b1efeeced5c196f3f3b7c435f90e84ea7f443901bec8099ef6" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/fd/7b/122376b1fd3c62c1ed9dc80c931ace4844b3c55407b6fb2d199377c9736f/pydantic-2.13.4-py3-none-any.whl", hash = "sha256:45a282cde31d808236fd7ea9d919b128653c8b38b393d1c4ab335c62924d9aba" }, +] + +[[package]] +name = "pydantic-core" +version = "2.46.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/9d/56/921726b776ace8d8f5db44c4ef961006580d91dc52b803c489fafd1aa249/pydantic_core-2.46.4.tar.gz", hash = "sha256:62f875393d7f270851f20523dd2e29f082bcc82292d66db2b64ea71f64b6e1c1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/5c/fa/6d7708d2cfc1a832acb6aeb0cd16e801902df8a0f583bb3b4b527fde022e/pydantic_core-2.46.4-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:0e96592440881c74a213e5ad528e2b24d3d4f940de2766bed9010ab1d9e51594" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ae/6f/aa064a3e74b5745afbdf250594f38e7ead05e2d651bcb35994b9417a0d4d/pydantic_core-2.46.4-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:e0d65b8c354be7fb5f720c3caa8bc940bc2d20ce749c8e06135f07f8ed95dd7c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/43/3a/41114a9f7569b84b4d84e7a018c57c56347dac30c0d4a872946ec4e36c46/pydantic_core-2.46.4-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:7bfb192b3f4b9e8a89b6277b6ce787564f62cfd272055f6e685726b111dc7826" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ef/25/1ab42e8048fe551934d9884e8d64daa7e990ad386f310a15981aeb6a5b08/pydantic_core-2.46.4-cp311-cp311-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:9037063db01f09b09e237c282b6792bd4da634b5402c4e7f0c61effed7701a04" }, + { url = "https://mirrors.aliyun.com/pypi/packages/94/c2/1a934597ddf08da410385b3b7aae91956a5a76c635effef456074fad7e88/pydantic_core-2.46.4-cp311-cp311-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:fc010ab034c8c7452522748bf937df58020d256ccae0874463d1f4d01758af8e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/02/6d/9e8ad178c9c4df27ad3c8f25d1fe2a7ab0d2ba0559fad4aee5d3d1f16771/pydantic_core-2.46.4-cp311-cp311-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8c5dac79fa1614d1e06ca695109c6105923bd9c7d1d6c918d4e637b7e6b32fd3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/80/50/540cd3aeefc041beb111125c4bff779831a2111fc6b15a9138cda277d32c/pydantic_core-2.46.4-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f9fa868638bf362d3d138ea55829cefb3d5f4b0d7f142234382a15e2485dbec4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6b/a4/b440ad35f05f6a38f89fa0f149accb3f0e02be94ca5e15f3c449a61b4bc9/pydantic_core-2.46.4-cp311-cp311-manylinux_2_31_riscv64.whl", hash = "sha256:17299feefe090f2caa5b8e37222bb5f663e4935a8bfa6931d4102e5df1a9f398" }, + { url = "https://mirrors.aliyun.com/pypi/packages/99/61/de4f55db8dfd57bfdfa9a12ec90fe1b57c4f41062f7ca86f08586b3e0ac0/pydantic_core-2.46.4-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:4c63ebc82684aa89d9a3bcbd13d515b3be44250dc68dd3bd81526c1cb31286c3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f7/52/7c529d7bdb2d1068bd52f51fe32572c8301f9a4febf1948f10639f1436f5/pydantic_core-2.46.4-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:aaa2a54443eff1950ba5ddc6b6ccda0d9c84a364276a62f969bdf2a390650848" }, + { url = "https://mirrors.aliyun.com/pypi/packages/37/b3/7c40325848ba78247f2812dcf9c7274e38cd801820ca6dd9fe63bcfb0eb4/pydantic_core-2.46.4-cp311-cp311-musllinux_1_1_armv7l.whl", hash = "sha256:18e5ceec2ab67e6d5f1a9085e5a24c9c4e2ac4545730bfe668680bca05e555f3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d9/37/f913f81a657c865b75da6c0dbed79876073c2a43b5bd9edbe8da785e4d49/pydantic_core-2.46.4-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:a0f62d0a58f4e7da165457e995725421e0064f2255d8eccebc49f41bbc23b109" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c4/67/6acaa1be2567f9256b056d8477158cac7240813956ce86e49deae8e173b4/pydantic_core-2.46.4-cp311-cp311-win32.whl", hash = "sha256:041bde0a48fd37cf71cab1c9d56d3e8625a3793fef1f7dd232b3ff37e978ecda" }, + { url = "https://mirrors.aliyun.com/pypi/packages/aa/e6/c505f83dfeda9a2e5c995cfd872949e4d05e12f7feb3dca72f633daefa94/pydantic_core-2.46.4-cp311-cp311-win_amd64.whl", hash = "sha256:6f2eeda33a839975441c86a4119e1383c50b47faf0cbb5176985565c6bb02c33" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0f/da/7a263a96d965d9d0df5e8de8a475f33495451117035b09acb110288c381f/pydantic_core-2.46.4-cp311-cp311-win_arm64.whl", hash = "sha256:14f4c5d6db102bd796a627bbb3a17b4cf4574b9ae861d8b7c9a9661c6dd3362d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ee/a4/73995fd4ebbb46ba0ee51e6fa049b8f02c40daebb762208feda8a6b7894d/pydantic_core-2.46.4-graalpy311-graalpy242_311_native-macosx_10_12_x86_64.whl", hash = "sha256:14d4edf427bdcf950a8a02d7cb44a08614388dd6e1bdcbf4f67504fa7887da9c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fb/7f/f37d3a5e8bfcc2e403f5c57a730f2d815693fb42119e8ea48b3789335af1/pydantic_core-2.46.4-graalpy311-graalpy242_311_native-macosx_11_0_arm64.whl", hash = "sha256:0ce40cd7b21210e99342afafbd4d0f76d784eb5b1d60f3bdc566be4983c6c73b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/15/3c/d7eb777b3ff43e8433a4efb39a17aa8fd98a4ee8561a24a67ef5db07b2d6/pydantic_core-2.46.4-graalpy311-graalpy242_311_native-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:90884113d8b48f760e9587002789ddd741e76ab9f89518cd1e43b1f1a52ec44b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/63/87/70b9f40170a81afd55ca26c9b2acb25c20d64bcfbf888fafecb3ba077d4c/pydantic_core-2.46.4-graalpy311-graalpy242_311_native-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:66ce7632c22d837c95301830e111ad0128a32b8207533b60896a96c4915192ea" }, + { url = "https://mirrors.aliyun.com/pypi/packages/11/cb/428de0385b6c8d44b716feba566abfacfbd23ee3c4439faa789a1456242f/pydantic_core-2.46.4-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:0c563b08bca408dc7f65f700633d8442fffb2421fc47b8101377e9fd65051ff0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0b/b5/6a17bdadd0fc1f170adfd05a20d37c832f52b117b4d9131da1f41bb097ce/pydantic_core-2.46.4-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:db06ffe51636ffe9ca531fe9023dd64bdd794be8754cb5df57c5498ae5b518a7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2a/dc/03734d80e362cd43ef65428e9de77c730ce7f2f11c60d2b1e1b39f0fbf99/pydantic_core-2.46.4-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:133878133d271ade3d41d1bfb2a45ec38dbdbda40bc065921c6b04e4630127e2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/de/df/5e5ffc085ed07cc22d298134d3d911c63e91f6a0eb91fe646750a3209910/pydantic_core-2.46.4-pp311-pypy311_pp73-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:9bc519fbf2b7578398853d815009ae5e4d4603d12f4e3f91da8c06852d3da3e9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/81/44/6e112a4253e56f5705467cbab7ab5e91ee7398ba3d56d358635958893d3e/pydantic_core-2.46.4-pp311-pypy311_pp73-musllinux_1_1_aarch64.whl", hash = "sha256:c7a7bd4e39e8e4c12c39cd480356842b6a8a06e41b23a55a5e3e191718838ddf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ac/ad/5565071e937d8e752842ac241463944c9eb14c87e2d269f2658a5bd05e98/pydantic_core-2.46.4-pp311-pypy311_pp73-musllinux_1_1_armv7l.whl", hash = "sha256:d396ec2b979760aaf3218e76c24e65bd0aca24983298653b3a9d7a45f9e47b30" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4f/c3/66883a5cec183e7fba4d024b4cbbe61851a63750ef606b0afecc46d1f2bf/pydantic_core-2.46.4-pp311-pypy311_pp73-musllinux_1_1_x86_64.whl", hash = "sha256:86e1a4418c6cd97d60c95c71164158eaf7324fae7b0923264016baa993eba6fc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4b/2d/69abac8f838090bbecd5df894befb2c2619e7996a98ddb949db9f3b93225/pydantic_core-2.46.4-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983" }, +] + +[[package]] +name = "pydantic-settings" +version = "2.14.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "pydantic" }, + { name = "python-dotenv" }, + { name = "typing-inspection" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/07/60/1d1e59c9c90d54591469ada7d268251f71c24bdb765f1a8a832cee8c6653/pydantic_settings-2.14.1.tar.gz", hash = "sha256:e874d3bec7e787b0c9958277956ed9b4dd5de6a80e162188fdaff7c5e26fd5fa" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ae/8d/f1af3832f5e6eb13ba94ee809e72b8ecb5eef226d27ee0bef7d963d943c7/pydantic_settings-2.14.1-py3-none-any.whl", hash = "sha256:6e3c7edfd8277687cdc598f56e5cff0e9bfff0910a3749deaa8d4401c3a2b9de" }, +] + +[[package]] +name = "pydub" +version = "0.25.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/fe/9a/e6bca0eed82db26562c73b5076539a4a08d3cffd19c3cc5913a3e61145fd/pydub-0.25.1.tar.gz", hash = "sha256:980a33ce9949cab2a569606b65674d748ecbca4f0796887fd6f46173a7b0d30f" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a6/53/d78dc063216e62fc55f6b2eebb447f6a4b0a59f55c8406376f76bf959b08/pydub-0.25.1-py2.py3-none-any.whl", hash = "sha256:65617e33033874b59d87db603aa1ed450633288aefead953b30bded59cb599a6" }, +] + +[[package]] +name = "pyee" +version = "13.0.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/8b/04/e7c1fe4dc78a6fdbfd6c337b1c3732ff543b8a397683ab38378447baa331/pyee-13.0.1.tar.gz", hash = "sha256:0b931f7c14535667ed4c7e0d531716368715e860b988770fc7eb8578d1f67fc8" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a0/c4/b4d4827c93ef43c01f599ef31453ccc1c132b353284fc6c87d535c233129/pyee-13.0.1-py3-none-any.whl", hash = "sha256:af2f8fede4171ef667dfded53f96e2ed0d6e6bd7ee3bb46437f77e3b57689228" }, +] + +[[package]] +name = "pygments" +version = "2.20.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/c3/b2/bc9c9196916376152d655522fdcebac55e66de6603a76a02bca1b6414f6c/pygments-2.20.0.tar.gz", hash = "sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f4/7e/a72dd26f3b0f4f2bf1dd8923c85f7ceb43172af56d63c7383eb62b332364/pygments-2.20.0-py3-none-any.whl", hash = "sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176" }, +] + +[[package]] +name = "pyjwt" +version = "2.13.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/3b/81/58d0ac84e1ef3a3843791d6954d94c0b33d526c75eeb1efbce9d0a4c4077/pyjwt-2.13.0.tar.gz", hash = "sha256:41571c89ca91598c79e8ef18a2d07367d4810fbbd6f637794879baf1b7703423" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a3/5e/ecf12fdb62546d64385c158514e9b2b671f7832108ef2ecd2020ce0af2d1/pyjwt-2.13.0-py3-none-any.whl", hash = "sha256:66adcc2aff09b3f1bbd95fc1e1577df8ac8723c978552fd43304c8a290ac5728" }, +] + +[package.optional-dependencies] +crypto = [ + { name = "cryptography" }, +] + +[[package]] +name = "pymysql" +version = "1.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/c9/bc/1c6a92f385940f727daeecf3bacaf186e03875dff57197801046c583bcf0/pymysql-1.2.0.tar.gz", hash = "sha256:6c7b17ca686988104d7426c27895b455cdeea3e9d3ceb1270f0c3704fead8c33" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c4/bd/2534e130295c8cfd4f0a2e31623baab7502278f1e97bcfe61db75656a77f/pymysql-1.2.0-py3-none-any.whl", hash = "sha256:62169ce6d5510f08e140c5e7990ee884a9764024e4a9a27b2cc11f1099322ae0" }, +] + +[[package]] +name = "pyopenssl" +version = "26.3.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "cryptography" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/74/b7/da07bae88f5a9506b4def6f2f4903cf4c3b8831e560dba8fa18ca08f758f/pyopenssl-26.3.0.tar.gz", hash = "sha256:589de7fae1c9ea670d18422ed00fc04da787bbde8e1454aea872aa57b49ad341" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/54/18/1dd71c9b43192ab83f1d531ad6002dc81108ac36c475f79fb7a295abe2f4/pyopenssl-26.3.0-py3-none-any.whl", hash = "sha256:46367f8f66b92271e6d218da9c87607e1ef5a0bc5c8dea5bb3db82f395c385a3" }, +] + +[[package]] +name = "pypdf" +version = "6.13.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/17/18/9947cc201af9ccf76720fd3347bf4f70eb882ce3fcf4cb05f7443e4cf871/pypdf-6.13.3.tar.gz", hash = "sha256:f3cb822769725f1bac658c406cfc9460399043f3750c2d3e4650e0a85eacabd7" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/94/56/2967e621598987905fb8cdfadd8f8de6b5c68c9351f0523c4df8409f28f1/pypdf-6.13.3-py3-none-any.whl", hash = "sha256:c6e3f86afb625791510b02ad5480e94b63970bb957df75d44657c282ecc52224" }, +] + +[[package]] +name = "pypdfium2" +version = "5.10.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/1d/78/d9b45abb97a3686643f7c6472a5f7688f2013a373226121dc76b9debbacf/pypdfium2-5.10.1.tar.gz", hash = "sha256:f257d2011eb43c846b7e9f5a802e28646b29732763e4a35dd6ca76f9be580538" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/98/49/0b85fcc0d236582143a25cf275b0b4f5d786f51eb07a89ec9c79d43efe18/pypdfium2-5.10.1-py3-none-android_23_arm64_v8a.whl", hash = "sha256:13abf7a9f5e0ddebc8bbcccea5f13ae5abe8a298ea219e125b0fc24c1d2171b4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/20/d2/2f522c5b2ad5166edf256bb4dbab97de5d07b2573f8a5630ddfcd1f8d4ee/pypdfium2-5.10.1-py3-none-android_23_armeabi_v7a.whl", hash = "sha256:a0dc52b56631e2f7edcdb22bae3b155aa840bec32c5bd05781e90baaecee88e7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a7/be/477548c026c2badfdbf4afc3358b7135121fb5bec2e2effcb67e3a674d0b/pypdfium2-5.10.1-py3-none-macosx_12_0_arm64.whl", hash = "sha256:ebb9e63f92d15fc41b359fe7a187233dfae37548800e1fa09cb2fc466ac89951" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8c/9b/0131c7f711b62c6edd6b200e9eb6340be6de4f6dc5baae625e3394d8d5fb/pypdfium2-5.10.1-py3-none-macosx_12_0_x86_64.whl", hash = "sha256:d04f2050b6b32bb18624688b600543342a4ab3aacf64bf66521a5af72bbc7de1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9d/3d/bddfceb6e67e54d6dd1ab6c0f1feff796a89596e40f6345a3b4cc6a3d408/pypdfium2-5.10.1-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0de53d2710ca9509fc2812340acc57c3f043697609068592d87de654f8cabf44" }, + { url = "https://mirrors.aliyun.com/pypi/packages/58/08/dedeb25c6645fc8a5eda24f54e0b1d083b7334ababf5f518bb939d729cc0/pypdfium2-5.10.1-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:10a26ce04795f8ec079e81c707fcb5737061e8a78025babc3d6e36642e9c903a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cc/48/69e1fc8b1216005243c6415183bbf6de1cda3f5a06758b3fa4a26a7385c6/pypdfium2-5.10.1-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:be6d2a8d1bcfd777188e7aa55a25c83f34d54e8350ad8810fb32017787d9b0d9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/72/7f/132455a58ad736d76815c6cd1307532c3f433299d945b9d2f8cc2387c309/pypdfium2-5.10.1-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:cf4d2527f79f31c550490cc74c9f32e19385a944630a3ef4cd4d9b6f961fbf77" }, + { url = "https://mirrors.aliyun.com/pypi/packages/bf/8c/4d5804eca598bbe894e0a9a510807e221c1623e7276ce6b68fa2660dc933/pypdfium2-5.10.1-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3ba127750bc3f4161461538d532d74491cd976f584f1753a6cee9cb821338ec1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d1/7f/baac59bf14ff914d97789ee0368c22ae233aa857aa9c0726bcb515dbc4d7/pypdfium2-5.10.1-py3-none-manylinux_2_27_s390x.manylinux_2_28_s390x.whl", hash = "sha256:80f30517ee089dfbbc6e9de6da365b6e8c0ce8f80c40b7d4025a3b1d3bbd8a70" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d0/d7/a5d58a0bcba31a0e37ed636a76ef3d2d215733f28af61637287d061b0c54/pypdfium2-5.10.1-py3-none-manylinux_2_34_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:ad5f5de15febb788c6eb3853e58aceda9cd8c5187c92472abaeecf9558deb0cf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/da/1a/98eebd14b36812176297cf765d504ebeebc982f895c8a3a9fbd2717797de/pypdfium2-5.10.1-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:8947fa3cd808da33960bdf8ff9e5247aa94ccd94f0b09bb3402e99498d83bcc9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1c/20/f2e124d607b8bb90a9f1ce976afff38c70e215cd7ea86af784cb2e8a19dd/pypdfium2-5.10.1-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:139a6387a3a2652f288e53164268bb03af4fa221d78484ee18407053a60082a3" }, + { url = "https://mirrors.aliyun.com/pypi/packages/59/ef/469ea87f668a32ff3280ea15e522e3a7858d2c80f1ecc320ece1244624c9/pypdfium2-5.10.1-py3-none-musllinux_1_2_i686.whl", hash = "sha256:172ff3e10358d66456e27fb0b8b5098e28ec24e51072eb4b5b86077d550e21bb" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c3/48/c7ed3001f0c5e28114c98bf918c8121e422a65ba7323e1e12f3f28e2d278/pypdfium2-5.10.1-py3-none-musllinux_1_2_ppc64le.whl", hash = "sha256:09fda0609dc4749c9865a0315c447d6f2583a580693bcabd69980d3fbb22ad51" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b2/bc/00b731bfc1fdc0f3c7d91108fceb54978ff4f5a92336820e5ed6c12535ff/pypdfium2-5.10.1-py3-none-musllinux_1_2_riscv64.whl", hash = "sha256:6b01adcfe9aaf7a635a59bed5687fd4fd7b0da292664f050d4ebd2bfa5c70584" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9a/e8/3ad242233f657c19092a8c83684c8154b8d02a826535a6a2591d91aa5dde/pypdfium2-5.10.1-py3-none-musllinux_1_2_s390x.whl", hash = "sha256:610e14c37d2090b826dccc0604fc7e7612c0ce591190780ae228abdb9abb971e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/80/3f/9d0747a02ac3021ed3db7ac27c5187d97e78b0253a4bdfbf386666968474/pypdfium2-5.10.1-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:f2803020952afa57e1e148adc19369b835154e1fa251a1ca85008ab6466a710f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9e/41/7d5187e9527eae81890a3b13193442112ef788d61ddaa68a6d59ac447bec/pypdfium2-5.10.1-py3-none-win32.whl", hash = "sha256:8702bb4f01ddfc8e7757b41b4c2c8392ac17c9f0234476e1e69672ea7c6d6aa0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/16/1d/c62bd59dd8345cc4b640f942f465633f6b07b859d01ddb648610a7bf5c7c/pypdfium2-5.10.1-py3-none-win_amd64.whl", hash = "sha256:58da5b51fb7884c7d21a05062ab13edb011d1a08dfd9694f3d5d685df62796b9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/10/d5/21bac39125df8a93e99c04583486b58a62b5997d6b3541e3ad0f69053392/pypdfium2-5.10.1-py3-none-win_arm64.whl", hash = "sha256:e3301c2f7a66fb8cb57dba857d0c9e90215e178f6602a87c5a306cd98513dab8" }, +] + +[[package]] +name = "pyreadline3" +version = "3.5.6" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b6/6d/f94028646d7bbe6d9d873c47ee7c246f2d29129d253f0d96cb6fcab70733/pyreadline3-3.5.6.tar.gz", hash = "sha256:61e53218b99656091ddb077df9e71f25850e72e030b6183b39c9b7e6e4f4a9bf" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f7/5e/35c856e186b74678c24927847ad9895a51f1bc02a0c6126477a6c6040064/pyreadline3-3.5.6-py3-none-any.whl", hash = "sha256:8449b734232e42a5dcd74048e39b60db2839a4c38cf3ae2bf7707d58b5389c0d" }, +] + +[[package]] +name = "pytest" +version = "9.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "iniconfig" }, + { name = "packaging" }, + { name = "pluggy" }, + { name = "pygments" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/84/0e/b5858858d74958632c49b72cb25a3976ff9f632397626715be71c89d3971/pytest-9.1.0.tar.gz", hash = "sha256:41dd9148c08072446394cefd3d79701701335a9f4cae69ba92e39f6c7f5c061c" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/8b/5a/ba30a81239b909821b3153e303e7def45178bf353da4f72380e6c5e8793b/pytest-9.1.0-py3-none-any.whl", hash = "sha256:8ebb0e7888bdf2bdfc602ec51f8f62d50200af37356c74e503c79a94f5c81f32" }, +] + +[[package]] +name = "pytest-asyncio" +version = "1.4.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "pytest" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/43/7c/d36d04db312ecf4298932ef77e6e4a9e8ad017906e24e34f0b0c361a2473/pytest_asyncio-1.4.0.tar.gz", hash = "sha256:c6c0d2259945122819f171a32ecea2c349ead889ee28176caaf492143424be42" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/03/e2/08a497ef684b88559c9cc5f4ad53a37e7b99e727094a86d6ea32536d5d3c/pytest_asyncio-1.4.0-py3-none-any.whl", hash = "sha256:933ca923a23075a87fb7070c0ec272a6848489824d887c85c812670932835aa1" }, +] + +[[package]] +name = "pytest-cov" +version = "7.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "coverage", extra = ["toml"] }, + { name = "pluggy" }, + { name = "pytest" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b1/51/a849f96e117386044471c8ec2bd6cfebacda285da9525c9106aeb28da671/pytest_cov-7.1.0.tar.gz", hash = "sha256:30674f2b5f6351aa09702a9c8c364f6a01c27aae0c1366ae8016160d1efc56b2" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/9d/7a/d968e294073affff457b041c2be9868a40c1c71f4a35fcc1e45e5493067b/pytest_cov-7.1.0-py3-none-any.whl", hash = "sha256:a0461110b7865f9a271aa1b51e516c9a95de9d696734a2f71e3e78f46e1d4678" }, +] + +[[package]] +name = "python-dateutil" +version = "2.9.0.post0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "six" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/66/c0/0c8b6ad9f17a802ee498c46e004a0eb49bc148f2fd230864601a86dcf6db/python-dateutil-2.9.0.post0.tar.gz", hash = "sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ec/57/56b9bcc3c9c6a792fcbaf139543cee77261f3651ca9da0c93f5c1221264b/python_dateutil-2.9.0.post0-py2.py3-none-any.whl", hash = "sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427" }, +] + +[[package]] +name = "python-dotenv" +version = "1.2.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/82/ed/0301aeeac3e5353ef3d94b6ec08bbcabd04a72018415dcb29e588514bba8/python_dotenv-1.2.2.tar.gz", hash = "sha256:2c371a91fbd7ba082c2c1dc1f8bf89ca22564a087c2c287cd9b662adde799cf3" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/0b/d7/1959b9648791274998a9c3526f6d0ec8fd2233e4d4acce81bbae76b44b2a/python_dotenv-1.2.2-py3-none-any.whl", hash = "sha256:1d8214789a24de455a8b8bd8ae6fe3c6b69a5e3d64aa8a8e5d68e694bbcb285a" }, +] + +[[package]] +name = "python-multipart" +version = "0.0.32" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/5b/42/55c32bb9b12693c092ad250a0e82edb5b31ddeda6eb772de5f308b3804ad/python_multipart-0.0.32.tar.gz", hash = "sha256:be54b7f3fa167bb83e4fcd936b887b708f4e57fe75911c02aebf53efaf8d938e" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e1/04/e8135ebd1ad02c56ec633277529b2602ff99ff634be76cdba5744cf554fd/python_multipart-0.0.32-py3-none-any.whl", hash = "sha256:ff6d3f776f16878c894e52e107296ffc890e913c611b1a4ec6c44e2821fe2e23" }, +] + +[[package]] +name = "python-pptx" +version = "1.0.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "lxml" }, + { name = "pillow" }, + { name = "typing-extensions" }, + { name = "xlsxwriter" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/52/a9/0c0db8d37b2b8a645666f7fd8accea4c6224e013c42b1d5c17c93590cd06/python_pptx-1.0.2.tar.gz", hash = "sha256:479a8af0eaf0f0d76b6f00b0887732874ad2e3188230315290cd1f9dd9cc7095" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d9/4f/00be2196329ebbff56ce564aa94efb0fbc828d00de250b1980de1a34ab49/python_pptx-1.0.2-py3-none-any.whl", hash = "sha256:160838e0b8565a8b1f67947675886e9fea18aa5e795db7ae531606d68e785cba" }, +] + +[[package]] +name = "pywin32" +version = "312" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1f/f5/10a6e845a00fc5e7afd0a988b744f403d4d57162a28d160a093c4d9322f0/pywin32-312-cp311-cp311-win32.whl", hash = "sha256:17948aeadbdb091f0ced6ef0841620794e68327b94ee415571c1203594b7215c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/35/c4/dcd2d62b5944b6d5db53413a5899016ccd57ffcb7278f3f81655d25d2027/pywin32-312-cp311-cp311-win_amd64.whl", hash = "sha256:d11417d84412f859b722fad0841b3614459ed0047f7542d8362e77884f6b6e8a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b7/56/3cbb433fe4501cdba2eb9040f56a4e1a8243faa4186b25295564d1a7a79d/pywin32-312-cp311-cp311-win_arm64.whl", hash = "sha256:b2200a054ca6d6625c4842fc56a4976a4b47f96b73dbe5538c3f813a80359f47" }, +] + +[[package]] +name = "pyyaml" +version = "6.0.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/05/8e/961c0007c59b8dd7729d542c61a4d537767a59645b82a0b521206e1e25c2/pyyaml-6.0.3.tar.gz", hash = "sha256:d76623373421df22fb4cf8817020cbb7ef15c725b9d5e45f17e189bfc384190f" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/6d/16/a95b6757765b7b031c9374925bb718d55e0a9ba8a1b6a12d25962ea44347/pyyaml-6.0.3-cp311-cp311-macosx_10_13_x86_64.whl", hash = "sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/16/19/13de8e4377ed53079ee996e1ab0a9c33ec2faf808a4647b7b4c0d46dd239/pyyaml-6.0.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0c/62/d2eb46264d4b157dae1275b573017abec435397aa59cbcdab6fc978a8af4/pyyaml-6.0.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/10/cb/16c3f2cf3266edd25aaa00d6c4350381c8b012ed6f5276675b9eba8d9ff4/pyyaml-6.0.3-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:850774a7879607d3a6f50d36d04f00ee69e7fc816450e5f7e58d7f17f1ae5c00" }, + { url = "https://mirrors.aliyun.com/pypi/packages/71/60/917329f640924b18ff085ab889a11c763e0b573da888e8404ff486657602/pyyaml-6.0.3-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b8bb0864c5a28024fac8a632c443c87c5aa6f215c0b126c449ae1a150412f31d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/dd/6f/529b0f316a9fd167281a6c3826b5583e6192dba792dd55e3203d3f8e655a/pyyaml-6.0.3-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:1d37d57ad971609cf3c53ba6a7e365e40660e3be0e5175fa9f2365a379d6095a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f2/6a/b627b4e0c1dd03718543519ffb2f1deea4a1e6d42fbab8021936a4d22589/pyyaml-6.0.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:37503bfbfc9d2c40b344d06b2199cf0e96e97957ab1c1b546fd4f87e53e5d3e4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/45/91/47a6e1c42d9ee337c4839208f30d9f09caa9f720ec7582917b264defc875/pyyaml-6.0.3-cp311-cp311-win32.whl", hash = "sha256:8098f252adfa6c80ab48096053f512f2321f0b998f98150cea9bd23d83e1467b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/da/e3/ea007450a105ae919a72393cb06f122f288ef60bba2dc64b26e2646fa315/pyyaml-6.0.3-cp311-cp311-win_amd64.whl", hash = "sha256:9f3bfb4965eb874431221a3ff3fdcddc7e74e3b07799e0e84ca4a0f867d449bf" }, +] + +[[package]] +name = "qdrant-client" +version = "1.18.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "grpcio" }, + { name = "httpx", extra = ["http2"] }, + { name = "numpy" }, + { name = "portalocker" }, + { name = "protobuf" }, + { name = "pydantic" }, + { name = "urllib3" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/65/45/5b1bdd15a3c7730eefb9c113600829e20d689b82b5a23f9e07d107094004/qdrant_client-1.18.0.tar.gz", hash = "sha256:52e8ece1a7d40519801bf0b70713bfa0f6b7ae28c7275bbe0b0286fbed7f6db4" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d6/10/c437bd2ac41ef30d3019063e6ce537dc111e9214473b337ee88f7fa6359a/qdrant_client-1.18.0-py3-none-any.whl", hash = "sha256:093aa8cf8a420ee3ad2a68b007e1378d7992b2600e0b53c193fc172674f659cd" }, +] + +[[package]] +name = "rank-bm25" +version = "0.2.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/fc/0a/f9579384aa017d8b4c15613f86954b92a95a93d641cc849182467cf0bb3b/rank_bm25-0.2.2.tar.gz", hash = "sha256:096ccef76f8188563419aaf384a02f0ea459503fdf77901378d4fd9d87e5e51d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/2a/21/f691fb2613100a62b3fa91e9988c991e9ca5b89ea31c0d3152a3210344f9/rank_bm25-0.2.2-py3-none-any.whl", hash = "sha256:7bd4a95571adadfc271746fa146a4bcfd89c0cf731e49c3d1ad863290adbe8ae" }, +] + +[[package]] +name = "referencing" +version = "0.37.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "attrs" }, + { name = "rpds-py" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/22/f5/df4e9027acead3ecc63e50fe1e36aca1523e1719559c499951bb4b53188f/referencing-0.37.0.tar.gz", hash = "sha256:44aefc3142c5b842538163acb373e24cce6632bd54bdb01b21ad5863489f50d8" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/2c/58/ca301544e1fa93ed4f80d724bf5b194f6e4b945841c5bfd555878eea9fcb/referencing-0.37.0-py3-none-any.whl", hash = "sha256:381329a9f99628c9069361716891d34ad94af76e461dcb0335825aecc7692231" }, +] + +[[package]] +name = "regex" +version = "2026.5.9" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/dc/0e/49aee608ad09480e7fd276898c99ec6192985fa331abe4eb3a986094490b/regex-2026.5.9.tar.gz", hash = "sha256:a8234aa23ec39894bfe4a3f1b85616a7032481964a13ac6fc9f10de4f6fca270" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c2/dc/c1f2df4027e82fc54b5a473e4b250f5139faca49a0fbe29a48668d228f34/regex-2026.5.9-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:ccf5249114cc3e772ecdd88a98a86eca0fd74c61ce32a94743758c083fc05d48" }, + { url = "https://mirrors.aliyun.com/pypi/packages/03/d2/59f01110660081cce9c0bc30ebd0b5ee250dacf658e3248ed92f01e0e8ee/regex-2026.5.9-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:46f1326ca6e65b0879d23ca302c0f2415aad42ff0309b9c818e7949fe19a41d8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/58/b6/14b2c84ff90ddb370c81d27503f4a0fcf071496416f4855f6cc8c5d81c35/regex-2026.5.9-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:ef31cbfe458e21c6122ba8150ff060e0c7789ed0d26eb423f25472584920b555" }, + { url = "https://mirrors.aliyun.com/pypi/packages/03/d0/4db86529117320de0c84afd90e70bb47434625875e34fcef9d8c127c5b16/regex-2026.5.9-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:992604d02e6d9c6d786c24a706a71ecffe1020fc1ef264044474cd81fa2c3919" }, + { url = "https://mirrors.aliyun.com/pypi/packages/07/78/fe4800cd322f862ecffd2d553409b20d80650e5ed71b9d178f853d020b82/regex-2026.5.9-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c9411dd64ca95477225734a93dfc8583b51916b8d5942f99d6cac21e09965451" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b5/d0/b3618a895dd8feb897c61bb2954edd265e1767d82a01d53065d5871127a3/regex-2026.5.9-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:3dd4a3ff360dfb836fecdb93a4598f9d6e2ac81e3e397125145c6221bf58cf4c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/33/6f/1481597e859ef19508b345eec4afd1416ed6e6b459c75a64026ef193aecf/regex-2026.5.9-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2a661a7d270a61f7cf460caee8b9fa2d5ef9e5c681234bcb9e0fe14f488e7dfc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/73/59/955734c803f59108deccba3597ae440c76b62a652733c0006e6243758420/regex-2026.5.9-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f079e50a0d3cc3cd5091fa9ff45869a2e6b2cd35895731edafb0327901a8d86d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/68/8f/70c04a236d651c81881dac42ef8538bddda6121434509d0a22d9e601503b/regex-2026.5.9-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:4ebe8f0b5ec5a5024dc4a4c59f444c4e9afc5f2abdbb8962065b75d27fb971f9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1d/96/05c7434d88185e5d27fe54aeb74df86bd77cd79f52f0b4eae54faa8fea70/regex-2026.5.9-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:97cf3bc1b7d7d2306772ec07366c80d9df00ff79e79cea32898883a646d2fae2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4e/c1/6e3d8202d981f3117004bf341ee74893ba4ba8a9fbaf4b94615846550a08/regex-2026.5.9-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:0f9eede6a5cbdc02d4978090186390936e1776a7d1359b21e41014c609880bcf" }, + { url = "https://mirrors.aliyun.com/pypi/packages/93/c7/e7737f1526b3fb32bd4c337fd6c71c3ebb5c8296fc34d11197e0955d2e35/regex-2026.5.9-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:01f0f5f55f4b64dacec85dc116d3c05fd23ad3ff037bbc73a2085775953c2611" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a5/27/0daffb1a535bb39f422c3d200f4ab023c71110ad66a32b366bee708baba0/regex-2026.5.9-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:1268eddd8486dc561d08eee1156e40aa3a8fe10f4bdec8fa653b455fcbffd12c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ce/fc/294fe4fac4f2ed67207b17471815870c1c45b3a489e08e0ac96daea16ef6/regex-2026.5.9-cp311-cp311-win32.whl", hash = "sha256:8676474c07469d6f33dd1085ca2cd45f65785f32518f2b20e36d9953ca07f994" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d0/b0/8dce459f6245bcf8f6e9f23ac9569f1a0f15c131cc0745e82b43226204cf/regex-2026.5.9-cp311-cp311-win_amd64.whl", hash = "sha256:246de9d60aa3f8538b519834dd95cbf276ea263d6a7bd5a3666dc3fa0230505b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/db/8d/f9aeff6ad63a3ef720386f2907e6d34a35a510a6e498ebad28b0fb3f6ab6/regex-2026.5.9-cp311-cp311-win_arm64.whl", hash = "sha256:d726ca3f0d76969bf1e8e477d160d3d666bbf999f6860bd314889e5345782046" }, +] + +[[package]] +name = "requests" +version = "2.34.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "certifi" }, + { name = "charset-normalizer" }, + { name = "idna" }, + { name = "urllib3" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ac/c3/e2a2b89f2d3e2179abd6d00ebd70bff6273f37fb3e0cc209f48b39d00cbf/requests-2.34.2.tar.gz", hash = "sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a0/f4/c67b0b3f1b9245e8d266f0f112c500d50e5b4e83cb6f3b71b6528104182a/requests-2.34.2-py3-none-any.whl", hash = "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0" }, +] + +[[package]] +name = "rich" +version = "15.0.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "markdown-it-py" }, + { name = "pygments" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/c0/8f/0722ca900cc807c13a6a0c696dacf35430f72e0ec571c4275d2371fca3e9/rich-15.0.0.tar.gz", hash = "sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/82/3b/64d4899d73f91ba49a8c18a8ff3f0ea8f1c1d75481760df8c68ef5235bf5/rich-15.0.0-py3-none-any.whl", hash = "sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb" }, +] + +[[package]] +name = "rpds-py" +version = "2026.5.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/2e/43/25a8dcd3feedd735039a8f0b5b7e3b118232b5eae288c4fd9ab200d41094/rpds_py-2026.5.1.tar.gz", hash = "sha256:07b24fea40541e28570e5b795a4a38fbdcd12550c06bd0748005ecc8116ca256" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/4f/a0/acf8b6fc20bfdcd3a45bd3f57680fb198e157b7e997b9123b10763798bd2/rpds_py-2026.5.1-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:3397a5ed7174dc2786bb214030232fc36fe8e5584fec43a9952cc542b1a12036" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b6/95/f8203fd997484b1690a6869cd0e503b6c3c6be55b0ecc36d1a491fe742f0/rpds_py-2026.5.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:99ab6ba7bfa2cb0f96a04e3652355bf04e3f51aceb1e943b8541dab7ba4828cc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/33/8c/b47326ad2f0be545a5e5c1a55937a12afaea7d392ba2837bb9680f57e6c9/rpds_py-2026.5.1-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:d0efbe45632665e53e3db8fe1e5692db58fc5cb9bab4459d570b83efefe11164" }, + { url = "https://mirrors.aliyun.com/pypi/packages/22/0b/e83bbd97ffac6f6389b605cd4e1c8ac5761dc7e977769c9255d8c5adb7bd/rpds_py-2026.5.1-cp311-cp311-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:01d17b29c0c23d82b1f4751147ec49cf451f1fc2554eb9ef5f957e55d2656ead" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fd/0e/d285d1bc8864245919c61e1ca82263e4a66d337759c3a4cef72766ff9afc/rpds_py-2026.5.1-cp311-cp311-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:7559f72b94ae52659086c595dfa017cde03155f7832071d30959049052cb3ece" }, + { url = "https://mirrors.aliyun.com/pypi/packages/86/06/ccb2109a1e543437b5e43816f2b43b9554cc6783145528a4e3711e05c011/rpds_py-2026.5.1-cp311-cp311-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:9e25b7088f9ccbfc0dfcaa52bf969300ca229e10ecf758974ebcbb080a4b37bb" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3d/33/237173db1cfef10105b3839a24de00eb8d2a523711add4632447cdf0aedd/rpds_py-2026.5.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:613fc4ee9eaef26dc5840666214dd6fbcebcf32f46e76f4abc473059f4e13dda" }, + { url = "https://mirrors.aliyun.com/pypi/packages/97/64/1eae54e34d5161f9969295e80bd6b62a55f2b6ac5f2a5b60d02c2140e758/rpds_py-2026.5.1-cp311-cp311-manylinux_2_31_riscv64.whl", hash = "sha256:85264a90ff4c05c1568dd65f5921c837614b67c60358fb4c17df3b7f2e90690a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d8/34/5bb334a5a0f65d77869217c4654f34c78a7d11b93938a3c076a2edeafc52/rpds_py-2026.5.1-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:fe71bca7d547acb17027c7fd1624ff8aae623499c498d3e7011182c4de5c25e0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/16/0f/007ec21283b5b040b4ec3bd95e0402591e22bfa7d5c93dfe01c465c2d2d7/rpds_py-2026.5.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:a05fa4f41f37ec97c9c260441a940450a192f78d774d2b097eee1379f1e1246a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ff/10/5437c94508169b6b22d8418fef7a66e9ffb5f3b9e9c94460f2eedafe06ff/rpds_py-2026.5.1-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:df1d2a1996755b24b9ecee92cb4d36c28f86f464a6a173349c26bab41e94b8c2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e0/d5/9937dce4d6bda74157b954e7d1460db05a22f5929dccfeeba1ed27a93df0/rpds_py-2026.5.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:8895840ac4809e5f60c88fd07617cd71326e73d6e5a8aa783c5c0f7c24985de2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6c/31/750617dd0ae1752471bf43f9e41d263398fae7cde7849d23b8574a70e617/rpds_py-2026.5.1-cp311-cp311-win32.whl", hash = "sha256:3684a59b158a7683aaeb8e25352e9a9dd2122cec78f2d8530266e4f91b4c7b3f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3c/bb/3dcab0e1d9516303f2eb672a5d6f62eca5a69e2886301e9c8c54b520c39b/rpds_py-2026.5.1-cp311-cp311-win_amd64.whl", hash = "sha256:7bd530e6a530bb3ea892f194fafa455f3516ac25ecf7143fd33c09be62b0470a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/49/d6/c6bbf5cb1cf12b9732df8074b57f6ef8341ba884c95d40632ae8bddb44e4/rpds_py-2026.5.1-cp311-cp311-win_arm64.whl", hash = "sha256:0a5ae4dbe43c1076983b72616496919872ae7bbe7a1e21cc48336bc3154d130b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/42/56/3fe0fb34820ff667be791b3a3c22b85e8bcba54e9c832f47438c191fa7be/rpds_py-2026.5.1-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:edf2765d84e42447f112ad877af8fe1db0089aaec5b28e88d6eab45e7fe99cea" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8b/f2/3eb9ccdb9f143b8c9b003978898cb497f942a324c077401e6b8834238e63/rpds_py-2026.5.1-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:ad3773236e95f7f33991eb125224b7da66f206504d032a253a02da7e134519fb" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a7/24/dbda232bc4f3ed732120692ab0d2c8402cb020516556d8bee622dcef2413/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a04df86b3f0fade39ec8fd0e0aab089b1da9fbd2b48df778a57ef96f5e7d38df" }, + { url = "https://mirrors.aliyun.com/pypi/packages/40/30/32e769839a358f78810c234f160f2cc21d1e4e47e1c0e0e0d535be5a0219/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:6142dbd80c4df62a5d899f0d616d417f84e0bc8d32526c8e5589019d75d028a7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ab/86/ec84d243aadb3b34b71dd26a010d0930b2d284ff5fc9a69fec53810ee6fd/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:0b35217adefe87f2fe4db7e9766cabe84744bfe9616d9667be18988928c7f2dc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/74/25/b60e52686bbff777a64f9e4f4d3dd57980dc846913777177a2c92e4937aa/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:b95d5e11fc712b752081183a55a244c03cd00570489edd7014d8899f8ceb8162" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9b/c7/b3a6a588cc2219510ef3f42e207483a93950bedd1e3a0fd4015c95cff9e5/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:141c9498daf2ace9eda35d2b0e376f9ea8b058d84f2aef4f96fccfd449a2f251" }, + { url = "https://mirrors.aliyun.com/pypi/packages/31/00/c7dba3fc8a3da8cb3f6db1eb3386be4d79c2e97c6890d20eb9ac66ae8c43/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_31_riscv64.whl", hash = "sha256:6f249f8b860a200ad35193af961183ebe9132710484e6f6ce0cf89fd83c63a9a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/93/dd/472ba494c70753f93745992c99855bee0636daf74e6984e5e003f150316f/rpds_py-2026.5.1-pp311-pypy311_pp73-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:e4abbf391a70be864920858bf360f4fb380577c9a0f732438a1996726e2c195b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1d/6f/93831a3bfe789542ed0c1d0d74b78b440f055d6dc3ea4640eba2d95e6e23/rpds_py-2026.5.1-pp311-pypy311_pp73-musllinux_1_2_aarch64.whl", hash = "sha256:c74005a7bb87752acf351c93897ec63ad77a07a0da7ecad9c050e32e7286ba34" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1f/ff/0b3d604614ffc77522c6b288fdbce68957eb583da1002aa65ba38ac0ee40/rpds_py-2026.5.1-pp311-pypy311_pp73-musllinux_1_2_i686.whl", hash = "sha256:8213afbe8a3a906fb9acb2014423fe3359ee783d0bf90995f70623a3217bfa6c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ea/ea/e7b0251441da9adfeaebcf29601d10f2a1455fcf0772fae9e7e19032bd96/rpds_py-2026.5.1-pp311-pypy311_pp73-musllinux_1_2_x86_64.whl", hash = "sha256:8c43a8a973270fd173bf48cdf80bbe66312421cba68d40845034f174f2389049" }, +] + +[[package]] +name = "rtree" +version = "1.4.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/95/09/7302695875a019514de9a5dd17b8320e7a19d6e7bc8f85dcfb79a4ce2da3/rtree-1.4.1.tar.gz", hash = "sha256:c6b1b3550881e57ebe530cc6cffefc87cd9bf49c30b37b894065a9f810875e46" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/04/d9/108cd989a4c0954e60b3cdc86fd2826407702b5375f6dfdab2802e5fed98/rtree-1.4.1-py3-none-macosx_10_9_x86_64.whl", hash = "sha256:d672184298527522d4914d8ae53bf76982b86ca420b0acde9298a7a87d81d4a4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f3/cf/2710b6fd6b07ea0aef317b29f335790ba6adf06a28ac236078ed9bd8a91d/rtree-1.4.1-py3-none-macosx_11_0_arm64.whl", hash = "sha256:a7e48d805e12011c2cf739a29d6a60ae852fb1de9fc84220bbcef67e6e595d7d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/55/e1/4d075268a46e68db3cac51846eb6a3ab96ed481c585c5a1ad411b3c23aad/rtree-1.4.1-py3-none-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:efa8c4496e31e9ad58ff6c7df89abceac7022d906cb64a3e18e4fceae6b77f65" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d1/75/e5d44be90525cd28503e7f836d077ae6663ec0687a13ba7810b4114b3668/rtree-1.4.1-py3-none-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:12de4578f1b3381a93a655846900be4e3d5f4cd5e306b8b00aa77c1121dc7e8c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fd/85/b8684f769a142163b52859a38a486493b05bafb4f2fb71d4f945de28ebf9/rtree-1.4.1-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:b558edda52eca3e6d1ee629042192c65e6b7f2c150d6d6cd207ce82f85be3967" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e9/a4/c2292b95246b9165cc43a0c3757e80995d58bc9b43da5cb47ad6e3535213/rtree-1.4.1-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:f155bc8d6bac9dcd383481dee8c130947a4866db1d16cb6dff442329a038a0dc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/74/25/5282c8270bfcd620d3e73beb35b40ac4ab00f0a898d98ebeb41ef0989ec8/rtree-1.4.1-py3-none-win_amd64.whl", hash = "sha256:efe125f416fd27150197ab8521158662943a40f87acab8028a1aac4ad667a489" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3f/50/0a9e7e7afe7339bd5e36911f0ceb15fed51945836ed803ae5afd661057fd/rtree-1.4.1-py3-none-win_arm64.whl", hash = "sha256:3d46f55729b28138e897ffef32f7ce93ac335cb67f9120125ad3742a220800f0" }, +] + +[[package]] +name = "ruff" +version = "0.15.17" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/8c/a9/3abdf488f1bf3d24c699415e454ed554a6350d5d89ce183be1ee0a3361ac/ruff-0.15.17.tar.gz", hash = "sha256:2ec446937fd16c8c4de2674a209cc5af64d9c6f17d21fbf1151054fa0bcf5219" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/db/4d/e11259f5da07cb6afb2d074c31bf09da9671993f7329d4f15d2fdc458301/ruff-0.15.17-py3-none-linux_armv6l.whl", hash = "sha256:d9feddb927fc68bd295f5eebc587a7e42cfaf9b65f60ca4a2386febff575da8f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/29/3e/772d679e1a0dc058e58875bd2c0cb713a0530877b4a76fee3c7966df0d49/ruff-0.15.17-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:25805a226d741c47d274a35ad5c10a7dde175fcddfa511d7cf3da0a21eb3eab7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/68/58/bd41f7688b2fd5623012605130ed70e60aa7f2244baa3d5066bdd61530c8/ruff-0.15.17-py3-none-macosx_11_0_arm64.whl", hash = "sha256:f6ad73b14c2d18a3bf8ad7cb6974294d7f613a7898604826058e6ac64918ef4d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d8/5b/733371013fcf1ec339e477ece6ab42bfe10bdd9bba8ee88a9516aa56bfc0/ruff-0.15.17-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:6ba0c1e4f95bcb3869d0d30cbd5917071ef2e28665abfec970cdab0492c713ed" }, + { url = "https://mirrors.aliyun.com/pypi/packages/bd/cc/6f24251cc0252f7239391ccb85833f320efad14ebe5b443943f37ced6332/ruff-0.15.17-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:81647960f10bff57d2e51cadd0c3950fe598400c852863a038720ef5b8cca91e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/68/dd/0d10c17ce1a1624d6fc3156309c3f834fdb5dfaad026ec90c85684f3990e/ruff-0.15.17-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:0e01a84ddbc8c16c23055ba3924476850f1bbc1917cebbb9376665a63e74260d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2f/91/556bfb156f6144f355e831c23db00b2fc4120f86b3ce81cc5f7fd2df51f3/ruff-0.15.17-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:84fe9f653152f8f294f9f7e03bf3a453d8b4a27f7a59c78c8666167f2b17b96c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/88/82/8b5999aa13355e926f06d9f42a32dcca862f623bf0363785ff89d607dffd/ruff-0.15.17-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8c0fe88a7676e7a05b73174d4d4a59cb2ac21ff8263583f87a81a6018475a978" }, + { url = "https://mirrors.aliyun.com/pypi/packages/11/93/f10377bb04109ca0e8cbc483ff1982c54b6d418210041776f93e8cdc7fa9/ruff-0.15.17-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ecfc3c7878fff94633ab0348524e093f9ce3243080416dd7d14f8ba400174719" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c7/a6/eeeae7f7d5493df41649ab3db92f086b2d0a30199e4efdf8e3dd7a033f24/ruff-0.15.17-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:b8461180b22420b1bdc289909410930761629fddf2a5aaf60fae1ab26cedc4c4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/32/88/5991ce565129a24dd4a00db1254b3b5db2e53018cbe4018ea5a89738e727/ruff-0.15.17-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:6eccbe50a038b503e7140b441aa9c7fc8c1f36edf23ebef9f4165c2f28f568b7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f5/1d/0fdd248313425f55223968af04b0a42125466a8d88d21c1d99c6af0a51e8/ruff-0.15.17-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:382fc0521025f5a8ad447d8bdd523545d0d7646adb718eb1c2dac5065ec27c0f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9e/0e/072e8260deb9461062ce9311ced27a8e541229a6ffd483013dd37661e43e/ruff-0.15.17-py3-none-musllinux_1_2_i686.whl", hash = "sha256:456d41fcd1b2777ad63f09a6e7121d43f7b688bbc76a800c10f7f8fb1f912c3f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ab/b4/55060a34163121498014696b5f656db5b8c6963768f227dbf0d76b311073/ruff-0.15.17-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:b1a04bcc94ae6194e9db05d16ad31f298a7194bfbcb08258bbe589cee1d587b8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/49/71/9b29d6b87cef468d697f43c6a91e3fae4a80185779d7d5a4ef27d173439f/ruff-0.15.17-py3-none-win32.whl", hash = "sha256:596065960ab1ff593f744220c9fe6580eda00a95003cffa9f4048bb5b1bf0392" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3d/b2/8fc77f3723228836fa5d12497eb71c808f83782e10d058d2b15cfa14640b/ruff-0.15.17-py3-none-win_amd64.whl", hash = "sha256:6769e5fa1710b179b92e0bfa5a51735b35baea9013dadb06d5f44cbcf9547084" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2d/c7/c53e8dbff9c9dc4b7928773421ae294a5d28fcb8dcda1a089579d3a7e510/ruff-0.15.17-py3-none-win_arm64.whl", hash = "sha256:f3be1fbb34bcdfd146240d8fb92a709d4c2c8191348580a3c044ec60fa0b4456" }, +] + +[[package]] +name = "safetensors" +version = "0.8.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/45/06/f955dbbb1859e3bd23c8ac6141af5106e7ad5fedec4a3a6e3d60f94b7001/safetensors-0.8.0.tar.gz", hash = "sha256:fabaf3e0f18a6618d9b36560682562157f77c2b71fcffc7b432be2baed9d753d" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/39/a0/f718cda65b05407d228f97602cf60dca269c979867aa5beb25410de26cd3/safetensors-0.8.0-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:c554f85858e05226d3c2828e32395e677434685d6d94594a41643361c5e837f0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f5/b1/fa7c600e7dceae12e9606c7578cbc9ff1e1ed55844883ee5c92205e86226/safetensors-0.8.0-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:c80201d22cbf405b80647a60ada77bba06c8fba2da2743ba1e89cdcc39a81f25" }, + { url = "https://mirrors.aliyun.com/pypi/packages/09/7d/65a7de0af421317bb36a067241e4235fff194eed60b961ed6d3f59a3fc60/safetensors-0.8.0-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:7a46e5ff292c356d6991e60942ba7f79817682d3a2cef0702136448cb9c4d235" }, + { url = "https://mirrors.aliyun.com/pypi/packages/91/4f/3175c9d75634e0e0dda0082794193521035edd7c70a6f212bf33ca06ddf4/safetensors-0.8.0-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:4124502b78f03534117c848f87a39b8f31e577b15eff423bf8bfb95f2a8c30d0" }, + { url = "https://mirrors.aliyun.com/pypi/packages/20/87/846c289e7aa2299eff406335717cf43ce8777194ece8aad75772e0411615/safetensors-0.8.0-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:7bc0a787ba8a35be368ee3574edfa2b1ad389eebd0a72e482ae275490e3f6c98" }, + { url = "https://mirrors.aliyun.com/pypi/packages/76/22/8d64d9df2c45d5ded401df889d0ad90882804ca172d79ec4f0df8f727fe0/safetensors-0.8.0-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:040070828e36dc8e122178bbbd5830ff9e97920affb84cbe0f46442497bed358" }, + { url = "https://mirrors.aliyun.com/pypi/packages/28/50/f203ff3a3ddfe19308efc83c5a3a29ed02bf786732ec35e68bf9162f3365/safetensors-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:fd6f3f93c9a0a7cc2788ee63fb763353d4bd2e89b0751bc78fcf7dda00bea774" }, + { url = "https://mirrors.aliyun.com/pypi/packages/46/fb/cdaed17ceb2948784fd9c36b6fd3e951b608547cea81a48e8ee6f8cfdfcb/safetensors-0.8.0-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:fcdd41ec4628fee5799f807c73c353629130fbd942aa23d83c623dd6c9d52d78" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0d/49/1e15de264dcc3b77943d2d0c56a95809956883b1c2d6d585c792523f180b/safetensors-0.8.0-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:8e9f537aa183a38ace122d27303dcd986b26bd2a7591f9181d7f0c396f4677ca" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2a/43/bf38443278eab4b1be1fce2931e2b012ad9cb7df52ada751d0aab8f7659a/safetensors-0.8.0-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:87eec7ffed2b809f05a398a8becb7d013f19f7837cd15d9748580d6cf30dbaf4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/72/e3/68cd3fa5b48488e84add63e04cb12f3bc28ae4638c06d4508c6e88823d0e/safetensors-0.8.0-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:4a95ae2b05d7726d751da4ebf626a2ca782b706e101bd894c95bc2450b1cffcc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/29/4b/1c19c509d56e01f4fbb3d0a2e597450f6cc04d1d56cf52defb0a62dfd715/safetensors-0.8.0-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:3ae091f16662658bdc019a4ff6cb4c085bb7d725eb5978b183ffd265863b6d2d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/27/43/41c1621732edd934d868a00d1b891584c892a7b62a9aab82ea5a0a5623ee/safetensors-0.8.0-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:8e080062fcde23be189565e1c3305d16751a218ecf9412c8601e64204eb6f846" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8e/3f/73ccf82579412b4a71c4ca673f10b5f1f888d7cf5af7fe24f27d30307be4/safetensors-0.8.0-cp310-abi3-win32.whl", hash = "sha256:2ddf52eac562eda224f99acfa7889d02968c1fd59a5b011ae7d8137c37e9c02d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1b/6d/3fba214c1e5e0f69991677ec3bc17023f0421776975e1de0c682dca475e2/safetensors-0.8.0-cp310-abi3-win_amd64.whl", hash = "sha256:096ec1a98435df7beb08853bb5aa9081a84f23d0adc67ed1a0a10550f608373f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8d/fc/7eedc3510d97878876e32774eebbeb61c43f148a96e915c84229a3e967aa/safetensors-0.8.0-cp310-abi3-win_arm64.whl", hash = "sha256:f7838e5135a406ad3e02efdcb8cf2e5397d368b0154537c4fec682dbc544d452" }, +] + +[[package]] +name = "scikit-learn" +version = "1.9.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "joblib" }, + { name = "narwhals" }, + { name = "numpy" }, + { name = "scipy" }, + { name = "threadpoolctl" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/fa/6f/37092bdb25f712817231799fc5674d8e704066a8a70c1d2d40517e18b4ab/scikit_learn-1.9.0.tar.gz", hash = "sha256:8833266989d3a5110178a9fae30783675460724d0e1efb13b14901d2c660c557" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f5/be/e844fd9586e66540a15b71924d17a6cbc1bb749e81ddd0a796bcdba4c055/scikit_learn-1.9.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:9db6f4d34e68c8899e4cab27fdf8eafe6ed21f2ba52ceb25ea250cd237f8e47b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/42/e2/ff880f62677a17d035817d543cb0fc8727d01eccbee81c5f7fc733a9d856/scikit_learn-1.9.0-cp311-cp311-macosx_12_0_arm64.whl", hash = "sha256:f401448645a3e7bc115aa3c094097865155b34bff1cba8101857d9104e99074c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/25/64/eb40435e1a508ab1b4e284ce43ae80f6a162e5be5e38ed5a6fab467a9ea4/scikit_learn-1.9.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:fd3a8ef0c758555a3b23c03adaa858af32f7736785ded50ad5991f59c4ed03fa" }, + { url = "https://mirrors.aliyun.com/pypi/packages/8d/da/4810a28e473185429e45a57eebcc91fc991b33d889cc0676063e671db03d/scikit_learn-1.9.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f7e254636164090da847715a27f8e5478feb98c40a9e0ee90cbd277de9e5ceb8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/3b/67/be3d369f40d8178ba3bd86635d132e08cb5329b023e4669d9426d84bc007/scikit_learn-1.9.0-cp311-cp311-win_amd64.whl", hash = "sha256:5dc1818c77575d149e25fce9ef82dd7b7263ae372f03494158668ad632a69759" }, + { url = "https://mirrors.aliyun.com/pypi/packages/37/79/a733f02dc2118da7e77a134b34f39f40201a353311b011d20859d2db3556/scikit_learn-1.9.0-cp311-cp311-win_arm64.whl", hash = "sha256:366652351f092b219c248f1e72821e841960a63d8f358f1dcfd54dc1cbdbbc28" }, +] + +[[package]] +name = "scipy" +version = "1.17.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/7a/97/5a3609c4f8d58b039179648e62dd220f89864f56f7357f5d4f45c29eb2cc/scipy-1.17.1.tar.gz", hash = "sha256:95d8e012d8cb8816c226aef832200b1d45109ed4464303e997c5b13122b297c0" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/df/75/b4ce781849931fef6fd529afa6b63711d5a733065722d0c3e2724af9e40a/scipy-1.17.1-cp311-cp311-macosx_10_14_x86_64.whl", hash = "sha256:1f95b894f13729334fb990162e911c9e5dc1ab390c58aa6cbecb389c5b5e28ec" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f7/58/bccc2861b305abdd1b8663d6130c0b3d7cc22e8d86663edbc8401bfd40d4/scipy-1.17.1-cp311-cp311-macosx_12_0_arm64.whl", hash = "sha256:e18f12c6b0bc5a592ed23d3f7b891f68fd7f8241d69b7883769eb5d5dfb52696" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/ee/18146b7757ed4976276b9c9819108adbc73c5aad636e5353e20746b73069/scipy-1.17.1-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:a3472cfbca0a54177d0faa68f697d8ba4c80bbdc19908c3465556d9f7efce9ee" }, + { url = "https://mirrors.aliyun.com/pypi/packages/ec/e6/cef1cf3557f0c54954198554a10016b6a03b2ec9e22a4e1df734936bd99c/scipy-1.17.1-cp311-cp311-macosx_14_0_x86_64.whl", hash = "sha256:766e0dc5a616d026a3a1cffa379af959671729083882f50307e18175797b3dfd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4d/60/8804678875fc59362b0fb759ab3ecce1f09c10a735680318ac30da8cd76b/scipy-1.17.1-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:744b2bf3640d907b79f3fd7874efe432d1cf171ee721243e350f55234b4cec4c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/09/7d/af933f0f6e0767995b4e2d705a0665e454d1c19402aa7e895de3951ebb04/scipy-1.17.1-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:43af8d1f3bea642559019edfe64e9b11192a8978efbd1539d7bc2aaa23d92de4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b4/3d/7ccbbdcbb54c8fdc20d3b6930137c782a163fa626f0aef920349873421ba/scipy-1.17.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:cd96a1898c0a47be4520327e01f874acfd61fb48a9420f8aa9f6483412ffa444" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e8/19/f926cb11c42b15ba08e3a71e376d816ac08614f769b4f47e06c3580c836a/scipy-1.17.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:4eb6c25dd62ee8d5edf68a8e1c171dd71c292fdae95d8aeb3dd7d7de4c364082" }, + { url = "https://mirrors.aliyun.com/pypi/packages/95/da/0d1df507cf574b3f224ccc3d45244c9a1d732c81dcb26b1e8a766ae271a8/scipy-1.17.1-cp311-cp311-win_amd64.whl", hash = "sha256:d30e57c72013c2a4fe441c2fcb8e77b14e152ad48b5464858e07e2ad9fbfceff" }, + { url = "https://mirrors.aliyun.com/pypi/packages/68/7f/bdd79ceaad24b671543ffe0ef61ed8e659440eb683b66f033454dcee90eb/scipy-1.17.1-cp311-cp311-win_arm64.whl", hash = "sha256:9ecb4efb1cd6e8c4afea0daa91a87fbddbce1b99d2895d151596716c0b2e859d" }, +] + +[[package]] +name = "sentence-transformers" +version = "5.5.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "huggingface-hub" }, + { name = "numpy" }, + { name = "scikit-learn" }, + { name = "scipy" }, + { name = "torch" }, + { name = "tqdm" }, + { name = "transformers" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/cf/d4/7ef93157485e978c016f49da05363c1e4e7237beb5343b64b5631101f0f1/sentence_transformers-5.5.1.tar.gz", hash = "sha256:02b7740dfc60bdbbcb6061625f5d97a5c1a4e2d3baac5f9391b912bb5eae2290" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/bf/03/ee99a6b030e7a2e056547729f8a4709dd93e13d9c6f07590f74c395c4017/sentence_transformers-5.5.1-py3-none-any.whl", hash = "sha256:4fe11d433badc5282d32f7fc08bc714216b7a5aca426f9df77a45a554756deb7" }, +] + +[[package]] +name = "setuptools" +version = "81.0.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/0d/1c/73e719955c59b8e424d015ab450f51c0af856ae46ea2da83eba51cc88de1/setuptools-81.0.0.tar.gz", hash = "sha256:487b53915f52501f0a79ccfd0c02c165ffe06631443a886740b91af4b7a5845a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e1/e3/c164c88b2e5ce7b24d667b9bd83589cf4f3520d97cad01534cd3c4f55fdb/setuptools-81.0.0-py3-none-any.whl", hash = "sha256:fdd925d5c5d9f62e4b74b30d6dd7828ce236fd6ed998a08d81de62ce5a6310d6" }, +] + +[[package]] +name = "shapely" +version = "2.1.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/4d/bc/0989043118a27cccb4e906a46b7565ce36ca7b57f5a18b78f4f1b0f72d9d/shapely-2.1.2.tar.gz", hash = "sha256:2ed4ecb28320a433db18a5bf029986aa8afcfd740745e78847e330d5d94922a9" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/8f/8d/1ff672dea9ec6a7b5d422eb6d095ed886e2e523733329f75fdcb14ee1149/shapely-2.1.2-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:91121757b0a36c9aac3427a651a7e6567110a4a67c97edf04f8d55d4765f6618" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4f/ce/28fab8c772ce5db23a0d86bf0adaee0c4c79d5ad1db766055fa3dab442e2/shapely-2.1.2-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:16a9c722ba774cf50b5d4541242b4cce05aafd44a015290c82ba8a16931ff63d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/70/8b/868b7e3f4982f5006e9395c1e12343c66a8155c0374fdc07c0e6a1ab547d/shapely-2.1.2-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:cc4f7397459b12c0b196c9efe1f9d7e92463cbba142632b4cc6d8bbbbd3e2b09" }, + { url = "https://mirrors.aliyun.com/pypi/packages/13/02/58b0b8d9c17c93ab6340edd8b7308c0c5a5b81f94ce65705819b7416dba5/shapely-2.1.2-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:136ab87b17e733e22f0961504d05e77e7be8c9b5a8184f685b4a91a84efe3c26" }, + { url = "https://mirrors.aliyun.com/pypi/packages/af/61/8e389c97994d5f331dcffb25e2fa761aeedfb52b3ad9bcdd7b8671f4810a/shapely-2.1.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:16c5d0fc45d3aa0a69074979f4f1928ca2734fb2e0dde8af9611e134e46774e7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d3/d4/9b2a9fe6039f9e42ccf2cb3e84f219fd8364b0c3b8e7bbc857b5fbe9c14c/shapely-2.1.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:6ddc759f72b5b2b0f54a7e7cde44acef680a55019eb52ac63a7af2cf17cb9cd2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/16/f6/9840f6963ed4decf76b08fd6d7fed14f8779fb7a62cb45c5617fa8ac6eab/shapely-2.1.2-cp311-cp311-win32.whl", hash = "sha256:2fa78b49485391224755a856ed3b3bd91c8455f6121fee0db0e71cefb07d0ef6" }, + { url = "https://mirrors.aliyun.com/pypi/packages/38/1e/3f8ea46353c2a33c1669eb7327f9665103aa3a8dfe7f2e4ef714c210b2c2/shapely-2.1.2-cp311-cp311-win_amd64.whl", hash = "sha256:c64d5c97b2f47e3cd9b712eaced3b061f2b71234b3fc263e0fcf7d889c6559dc" }, +] + +[[package]] +name = "shellingham" +version = "1.5.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/58/15/8b3609fd3830ef7b27b655beb4b4e9c62313a4e8da8c676e142cc210d58e/shellingham-1.5.4.tar.gz", hash = "sha256:8dbca0739d487e5bd35ab3ca4b36e11c4078f3a234bfce294b0a0291363404de" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e0/f9/0595336914c5619e5f28a1fb793285925a8cd4b432c9da0a987836c7f822/shellingham-1.5.4-py2.py3-none-any.whl", hash = "sha256:7ecfff8f2fd72616f7481040475a65b2bf8af90a56c89140852d1120324e8686" }, +] + +[[package]] +name = "six" +version = "1.17.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/94/e7/b2c673351809dca68a0e064b6af791aa332cf192da575fd474ed7d6f16a2/six-1.17.0.tar.gz", hash = "sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b7/ce/149a00dd41f10bc29e5921b496af8b574d8413afcd5e30dfa0ed46c2cc5e/six-1.17.0-py2.py3-none-any.whl", hash = "sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274" }, +] + +[[package]] +name = "sniffio" +version = "1.3.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/a2/87/a6771e1546d97e7e041b6ae58d80074f81b7d5121207425c964ddf5cfdbd/sniffio-1.3.1.tar.gz", hash = "sha256:f4324edc670a0f49750a81b895f35c3adb843cca46f0530f79fc1babb23789dc" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2" }, +] + +[[package]] +name = "snowballstemmer" +version = "2.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/44/7b/af302bebf22c749c56c9c3e8ae13190b5b5db37a33d9068652e8f73b7089/snowballstemmer-2.2.0.tar.gz", hash = "sha256:09b16deb8547d3412ad7b590689584cd0fe25ec8db3be37788be3810cbf19cb1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ed/dc/c02e01294f7265e63a7315fe086dd1df7dacb9f840a804da846b96d01b96/snowballstemmer-2.2.0-py2.py3-none-any.whl", hash = "sha256:c8e1716e83cc398ae16824e5572ae04e0d9fc2c6b985fb0f900f5f0c96ecba1a" }, +] + +[[package]] +name = "soupsieve" +version = "2.8.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/47/2c/0a5f6f8ee0d5589e48c7640213ed5175d52cf540a06725b628cc1a45d6ce/soupsieve-2.8.4.tar.gz", hash = "sha256:e121fd02e975c695e4e9e8774a5ee35d74714b59307868dcc5319ad2d9e3328e" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/5e/f5/0c41cb68dcae6b7de4fac4188a3a9589e21fb31df21ea3a2e888db95e6c9/soupsieve-2.8.4-py3-none-any.whl", hash = "sha256:e7e6b0769c8f51ed59acab6e994b00621096cfb1c640a7509295987388fbaf65" }, +] + +[[package]] +name = "speechrecognition" +version = "3.17.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ae/91/442c0ec260ad94ec1f6cae74fa065811f0e4416e4dc38f658ee1052498e4/speechrecognition-3.17.0.tar.gz", hash = "sha256:bd7e609c2ebea1680e75fc5dfbd8b44388c316f134c8d44fa2a30c0f50b346be" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/aa/e7/13e260a9cb53a40177783a882ebdfa437b2414fa21ca6f1cb8d9043b3fc9/speechrecognition-3.17.0-py3-none-any.whl", hash = "sha256:754f2cd9d7fbeff5e05ad91b906350cb7983fd2ef82002d91e117ac44aa94efa" }, +] + +[[package]] +name = "sse-starlette" +version = "3.4.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "anyio" }, + { name = "starlette" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/f7/2b/58abc2d1fd397e7dde08e947e05c884d8ef2f78d5e2588c17a12d42d6994/sse_starlette-3.4.4.tar.gz", hash = "sha256:07e0fa0460138baf25cdd5fb28683472c3995dc1642225191b3832d62526bcb0" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/dc/67/805710444ea8cc75fbf70b920ed431a560c4bf9c57f7d5a3117213189399/sse_starlette-3.4.4-py3-none-any.whl", hash = "sha256:3f4dd50d8aed2771a091f3a83000323fc3844541c16b4fe585ae2420cc6df973" }, +] + +[[package]] +name = "starlette" +version = "1.3.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "anyio" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/eb/e3/7c1dc7381d9f8ab7d854328ebfa884e62cb3f3d8549ddfd37c7814f42afa/starlette-1.3.1.tar.gz", hash = "sha256:05d0213193f2fbaae60e2ecb593b4add4262ad4e46536b54abe36f11a71724e0" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ec/bb/2799cc2ede3ed41131f8975621e7213dfc7ef4acbbaadfa440f32500c370/starlette-1.3.1-py3-none-any.whl", hash = "sha256:c7372aae11c3c3f26a42df7bd626cec2f47d03483d261d369516a615a53714c6" }, +] + +[[package]] +name = "sympy" +version = "1.14.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "mpmath" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/83/d3/803453b36afefb7c2bb238361cd4ae6125a569b4db67cd9e79846ba2d68c/sympy-1.14.0.tar.gz", hash = "sha256:d3d3fe8df1e5a0b42f0e7bdf50541697dbe7d23746e894990c030e2b05e72517" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/a2/09/77d55d46fd61b4a135c444fc97158ef34a095e5681d0a6c10b75bf356191/sympy-1.14.0-py3-none-any.whl", hash = "sha256:e091cc3e99d2141a0ba2847328f5479b05d94a6635cb96148ccb3f34671bd8f5" }, +] + +[[package]] +name = "tenacity" +version = "9.1.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/47/c6/ee486fd809e357697ee8a44d3d69222b344920433d3b6666ccd9b374630c/tenacity-9.1.4.tar.gz", hash = "sha256:adb31d4c263f2bd041081ab33b498309a57c77f9acf2db65aadf0898179cf93a" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/d7/c1/eb8f9debc45d3b7918a32ab756658a0904732f75e555402972246b0b8e71/tenacity-9.1.4-py3-none-any.whl", hash = "sha256:6095a360c919085f28c6527de529e76a06ad89b23659fa881ae0649b867a9d55" }, +] + +[[package]] +name = "tf-playwright-stealth" +version = "1.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "fake-http-header" }, + { name = "playwright" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d6/6b/32bb58c65991f91aeaaf7473b650175d9d4af5dd383983d177d49ccba08d/tf_playwright_stealth-1.2.0.tar.gz", hash = "sha256:7bb8d32d3e60324fbf6b9eeae540b8cd9f3b9e07baeb33b025dbc98ad47658ba" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/11/3d/2653f4cf49660bb44eeac8270617cc4c0287d61716f249f55053f0af0724/tf_playwright_stealth-1.2.0-py3-none-any.whl", hash = "sha256:26ee47ee89fa0f43c606fe37c188ea3ccd36f96ea90c01d167b768df457e7886" }, +] + +[[package]] +name = "threadpoolctl" +version = "3.6.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b7/4d/08c89e34946fce2aec4fbb45c9016efd5f4d7f24af8e5d93296e935631d8/threadpoolctl-3.6.0.tar.gz", hash = "sha256:8ab8b4aa3491d812b623328249fab5302a68d2d71745c8a4c719a2fcaba9f44e" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/32/d5/f9a850d79b0851d1d4ef6456097579a9005b31fea68726a4ae5f2d82ddd9/threadpoolctl-3.6.0-py3-none-any.whl", hash = "sha256:43a0b8fd5a2928500110039e43a5eed8480b918967083ea48dc3ab9f13c4a7fb" }, +] + +[[package]] +name = "tiktoken" +version = "0.13.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "regex" }, + { name = "requests" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e4/e5/5f3cb2159769d0f4324c0e9e87f9de3c4b1cd45848a96b2eb3566ad5ca77/tiktoken-0.13.0.tar.gz", hash = "sha256:c9435714c3a84c2319499de9a300c0e604449dd0799ff246458b3bb6a7f433c1" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1a/4c/1bc81f4cd53e827c4ee67ca951b5935724716049452d8dfa09b8b82372bb/tiktoken-0.13.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:7bfe1849caa65d1e1d9871817170ec497bbb7984e182012e1bdce72f66608cdb" }, + { url = "https://mirrors.aliyun.com/pypi/packages/75/91/10b9c7076bc02c246c853201fdbbe300a4b8c5ed7b84c25f7403f4e32655/tiktoken-0.13.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:91c180fe255bd5a86d8316210d2833a1d4d33d026cd86a67812f4773743c8d26" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4e/e4/fceae98015fab47fcd49b8bd7f46145bcd187a47e0add1e5378ed67ef980/tiktoken-0.13.0-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:059c8ecf554eb5b41e6e054ba467b871b03277d267dee7244380aca4359747d4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f9/39/fe42ad00de01a8c4a49ad8649a2c8a316835a9cad5961b11d21eac0020a5/tiktoken-0.13.0-cp311-cp311-manylinux_2_28_x86_64.whl", hash = "sha256:36217497eaffc158607a3b26f065300db2aefd43b115263f3b9688ce38146173" }, + { url = "https://mirrors.aliyun.com/pypi/packages/03/c4/ccee1ecccca107e9a16efcecdeeb964c325305038554d466ece65b42338f/tiktoken-0.13.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:303f7d91b4fce3baddbcde05c139091d4caa5026ac7214c1dc7ff7a71ee429ff" }, + { url = "https://mirrors.aliyun.com/pypi/packages/9d/03/cd0cba295522b91eb55c6b2704f1df895f8226cfe60ab10d4d51d0cc9e69/tiktoken-0.13.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:5d48843bee149630eb735a99e1f4a85b47308d21868ea63163f6e87768d3cfed" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7e/25/a10efd564402d82c2ff50d12057353ace447aa8007deceaa48641f63d35c/tiktoken-0.13.0-cp311-cp311-win_amd64.whl", hash = "sha256:fc1c44cd37b43fc46bae593129164f4f281e82ea116b57a85aa81bda57eafc94" }, +] + +[[package]] +name = "tokenizers" +version = "0.22.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "huggingface-hub" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/73/6f/f80cfef4a312e1fb34baf7d85c72d4411afde10978d4657f8cdd811d3ccc/tokenizers-0.22.2.tar.gz", hash = "sha256:473b83b915e547aa366d1eee11806deaf419e17be16310ac0a14077f1e28f917" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/92/97/5dbfabf04c7e348e655e907ed27913e03db0923abb5dfdd120d7b25630e1/tokenizers-0.22.2-cp39-abi3-macosx_10_12_x86_64.whl", hash = "sha256:544dd704ae7238755d790de45ba8da072e9af3eea688f698b137915ae959281c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2e/47/174dca0502ef88b28f1c9e06b73ce33500eedfac7a7692108aec220464e7/tokenizers-0.22.2-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:1e418a55456beedca4621dbab65a318981467a2b188e982a23e117f115ce5001" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d6/84/7990e799f1309a8b87af6b948f31edaa12a3ed22d11b352eaf4f4b2e5753/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2249487018adec45d6e3554c71d46eb39fa8ea67156c640f7513eb26f318cec7" }, + { url = "https://mirrors.aliyun.com/pypi/packages/78/59/09d0d9ba94dcd5f4f1368d4858d24546b4bdc0231c2354aa31d6199f0399/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:25b85325d0815e86e0bac263506dd114578953b7b53d7de09a6485e4a160a7dd" }, + { url = "https://mirrors.aliyun.com/pypi/packages/47/50/b3ebb4243e7160bda8d34b731e54dd8ab8b133e50775872e7a434e524c28/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:bfb88f22a209ff7b40a576d5324bf8286b519d7358663db21d6246fb17eea2d5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e0/fa/89f4cb9e08df770b57adb96f8cbb7e22695a4cb6c2bd5f0c4f0ebcf33b66/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:1c774b1276f71e1ef716e5486f21e76333464f47bece56bbd554485982a9e03e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/64/04/ca2363f0bfbe3b3d36e95bf67e56a4c88c8e3362b658e616d1ac185d47f2/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:df6c4265b289083bf710dff49bc51ef252f9d5be33a45ee2bed151114a56207b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/2e/76/932be4b50ef6ccedf9d3c6639b056a967a86258c6d9200643f01269211ca/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:369cc9fc8cc10cb24143873a0d95438bb8ee257bb80c71989e3ee290e8d72c67" }, + { url = "https://mirrors.aliyun.com/pypi/packages/1d/28/5f9f5a4cc211b69e89420980e483831bcc29dade307955cc9dc858a40f01/tokenizers-0.22.2-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:29c30b83d8dcd061078b05ae0cb94d3c710555fbb44861139f9f83dcca3dc3e4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6c/fb/66e2da4704d6aadebf8cb39f1d6d1957df667ab24cff2326b77cda0dcb85/tokenizers-0.22.2-cp39-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:37ae80a28c1d3265bb1f22464c856bd23c02a05bb211e56d0c5301a435be6c1a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/16/04/fed398b05caa87ce9b1a1bb5166645e38196081b225059a6edaff6440fac/tokenizers-0.22.2-cp39-abi3-musllinux_1_2_i686.whl", hash = "sha256:791135ee325f2336f498590eb2f11dc5c295232f288e75c99a36c5dbce63088a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/05/a1/d62dfe7376beaaf1394917e0f8e93ee5f67fea8fcf4107501db35996586b/tokenizers-0.22.2-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:38337540fbbddff8e999d59970f3c6f35a82de10053206a7562f1ea02d046fa5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fd/18/a545c4ea42af3df6effd7d13d250ba77a0a86fb20393143bbb9a92e434d4/tokenizers-0.22.2-cp39-abi3-win32.whl", hash = "sha256:a6bf3f88c554a2b653af81f3204491c818ae2ac6fbc09e76ef4773351292bc92" }, + { url = "https://mirrors.aliyun.com/pypi/packages/65/71/0670843133a43d43070abeb1949abfdef12a86d490bea9cd9e18e37c5ff7/tokenizers-0.22.2-cp39-abi3-win_amd64.whl", hash = "sha256:c9ea31edff2968b44a88f97d784c2f16dc0729b8b143ed004699ebca91f05c48" }, + { url = "https://mirrors.aliyun.com/pypi/packages/72/f4/0de46cfa12cdcbcd464cc59fde36912af405696f687e53a091fb432f694c/tokenizers-0.22.2-cp39-abi3-win_arm64.whl", hash = "sha256:9ce725d22864a1e965217204946f830c37876eee3b2ba6fc6255e8e903d5fcbc" }, +] + +[[package]] +name = "tomli" +version = "2.4.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/22/de/48c59722572767841493b26183a0d1cc411d54fd759c5607c4590b6563a6/tomli-2.4.1.tar.gz", hash = "sha256:7c7e1a961a0b2f2472c1ac5b69affa0ae1132c39adcb67aba98568702b9cc23f" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f4/11/db3d5885d8528263d8adc260bb2d28ebf1270b96e98f0e0268d32b8d9900/tomli-2.4.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:f8f0fc26ec2cc2b965b7a3b87cd19c5c6b8c5e5f436b984e85f486d652285c30" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6d/f7/675db52c7e46064a9aa928885a9b20f4124ecb9bc2e1ce74c9106648d202/tomli-2.4.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:4ab97e64ccda8756376892c53a72bd1f964e519c77236368527f758fbc36a53a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/61/71/81c50943cf953efa35bce7646caab3cf457a7d8c030b27cfb40d7235f9ee/tomli-2.4.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:96481a5786729fd470164b47cdb3e0e58062a496f455ee41b4403be77cb5a076" }, + { url = "https://mirrors.aliyun.com/pypi/packages/48/c1/f41d9cb618acccca7df82aaf682f9b49013c9397212cb9f53219e3abac37/tomli-2.4.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5a881ab208c0baf688221f8cecc5401bd291d67e38a1ac884d6736cbcd8247e9" }, + { url = "https://mirrors.aliyun.com/pypi/packages/22/e4/5a816ecdd1f8ca51fb756ef684b90f2780afc52fc67f987e3c61d800a46d/tomli-2.4.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:47149d5bd38761ac8be13a84864bf0b7b70bc051806bc3669ab1cbc56216b23c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6b/49/2b2a0ef529aa6eec245d25f0c703e020a73955ad7edf73e7f54ddc608aa5/tomli-2.4.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:ec9bfaf3ad2df51ace80688143a6a4ebc09a248f6ff781a9945e51937008fcbc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/83/bd/6c1a630eaca337e1e78c5903104f831bda934c426f9231429396ce3c3467/tomli-2.4.1-cp311-cp311-win32.whl", hash = "sha256:ff2983983d34813c1aeb0fa89091e76c3a22889ee83ab27c5eeb45100560c049" }, + { url = "https://mirrors.aliyun.com/pypi/packages/42/59/71461df1a885647e10b6bb7802d0b8e66480c61f3f43079e0dcd315b3954/tomli-2.4.1-cp311-cp311-win_amd64.whl", hash = "sha256:5ee18d9ebdb417e384b58fe414e8d6af9f4e7a0ae761519fb50f721de398dd4e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b8/83/dceca96142499c069475b790e7913b1044c1a4337e700751f48ed723f883/tomli-2.4.1-cp311-cp311-win_arm64.whl", hash = "sha256:c2541745709bad0264b7d4705ad453b76ccd191e64aa6f0fc66b69a293a45ece" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7b/61/cceae43728b7de99d9b847560c262873a1f6c98202171fd5ed62640b494b/tomli-2.4.1-py3-none-any.whl", hash = "sha256:0d85819802132122da43cb86656f8d1f8c6587d54ae7dcaf30e90533028b49fe" }, +] + +[[package]] +name = "torch" +version = "2.12.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "cuda-bindings", marker = "sys_platform == 'linux'" }, + { name = "cuda-toolkit", extra = ["cudart", "cufft", "cufile", "cupti", "curand", "cusolver", "cusparse", "nvjitlink", "nvrtc", "nvtx"], marker = "sys_platform == 'linux'" }, + { name = "filelock" }, + { name = "fsspec" }, + { name = "jinja2" }, + { name = "networkx" }, + { name = "nvidia-cublas", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cudnn-cu13", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cusparselt-cu13", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nccl-cu13", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvshmem-cu13", marker = "sys_platform == 'linux'" }, + { name = "setuptools" }, + { name = "sympy" }, + { name = "triton", marker = "sys_platform == 'linux'" }, + { name = "typing-extensions" }, +] +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/18/62/131124fb95df03811b8260d1d43dcc5ee85ea1a344b964613d7efe77fb08/torch-2.12.0-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:10802fd383bbfed646212e765a72c37d2185205d4f26eb197a254e8ac7ddcb25" }, + { url = "https://mirrors.aliyun.com/pypi/packages/12/9c/dda0dbd547dc549839824135f223792fd0e725f28ed0715dda366b7acaa2/torch-2.12.0-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:c12592630aef72feaf18bd3f197ef587bbfa21131b31c38b23ab2e55fce92e36" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e2/d2/a7dd5a3f9bdaa7842124e8e2359202b317c48d47d2fc5816fafdf2049adb/torch-2.12.0-cp311-cp311-manylinux_2_28_x86_64.whl", hash = "sha256:415c1b8d0412f67551c8e89a2daca0fb3e56694af0281ba155eaa9da481f58b4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/12/1b/a61ce2004f9ab0ea8964a6e6168133a127795667639e2ff4f8f2bdb16a65/torch-2.12.0-cp311-cp311-win_amd64.whl", hash = "sha256:dd37188ea325042cb1f6cafa56822b11ada2520c04791a52629b0af25bdfbfd9" }, +] + +[[package]] +name = "tqdm" +version = "4.68.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/85/05/0d5260f1f1ca784f4a4a0def9cbe6affe587f5b4025328d446c3d67765f4/tqdm-4.68.2.tar.gz", hash = "sha256:89c230e8dbc67c7615c142487111222f878c77427ea09549960f62389e258add" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/eb/75/1a0392bcc21c44dcdf87b3cf2d137e7829be2c083a1e38d44efca3d57a16/tqdm-4.68.2-py3-none-any.whl", hash = "sha256:d4240441fb5353290b87d6a85968c9decc131a99b8c7faa28269d829de669ede" }, +] + +[[package]] +name = "transformers" +version = "5.12.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "huggingface-hub" }, + { name = "numpy" }, + { name = "packaging" }, + { name = "pyyaml" }, + { name = "regex" }, + { name = "safetensors" }, + { name = "tokenizers" }, + { name = "tqdm" }, + { name = "typer" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/aa/7c/8240f612819718100a9346dc28dea6a11370c3ca9c8c6eabadd3dea4ef29/transformers-5.12.1.tar.gz", hash = "sha256:679ee731c8225347889ad4fb3b2c926a62e9da3b7d284e9d12c791da7272466b" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/df/56/bbd60dd8668055803bf8ba55a81f9b8a8b31497f620109a9671d26a2076d/transformers-5.12.1-py3-none-any.whl", hash = "sha256:2a5e109d2021265df7098ffbb738295acaf5ad256f12cbc586db2ea4dcbb1a8a" }, +] + +[[package]] +name = "trimesh" +version = "4.12.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/79/37/5cb90f04990260d2caceb6093560c6cefafca1ec522c1e43be01ca658244/trimesh-4.12.2.tar.gz", hash = "sha256:c8ca31571ac00b112e4e160e66a2d4c3491df321f056bd33806be0485d1af9d9" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/05/98/716a473cfb24750858ddd5d14e6527539dd206583a46408d08eeb2844a75/trimesh-4.12.2-py3-none-any.whl", hash = "sha256:b5b5afa63c5272345f2858f7676bc8c217dc8a89f4fadf6193fe10a81b5ff2aa" }, +] + +[[package]] +name = "triton" +version = "3.7.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/b8/c1/5d842314bb6c78442cc60437928781701c6050b8d479bc2a1aed691d37ca/triton-3.7.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:a9e71fc392675fac364e0ecf4ef3f76f85b7f5433a16f4c3c5fe5f05a52c85fe" }, + { url = "https://mirrors.aliyun.com/pypi/packages/13/31/8315ea5f8dd18e60970b3022e3a8b93fd37e0b784fbbef86e10c8e6e5ca1/triton-3.7.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:22bacffce443f54593dd20f05294d5a40622e0ea9ab632816f87154504356221" }, +] + +[[package]] +name = "typer" +version = "0.25.1" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "annotated-doc" }, + { name = "click" }, + { name = "rich" }, + { name = "shellingham" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/e4/51/9aed62104cea109b820bbd6c14245af756112017d309da813ef107d42e7e/typer-0.25.1.tar.gz", hash = "sha256:9616eb8853a09ffeabab1698952f33c6f29ffdbceb4eaeecf571880e8d7664cc" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/3f/f9/2b3ff4e56e5fa7debfaf9eb135d0da96f3e9a1d5b27222223c7296336e5f/typer-0.25.1-py3-none-any.whl", hash = "sha256:75caa44ed46a03fb2dab8808753ffacdbfea88495e74c85a28c5eefcf5f39c89" }, +] + +[[package]] +name = "typing-extensions" +version = "4.15.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/72/94/1a15dd82efb362ac84269196e94cf00f187f7ed21c242792a923cdb1c61f/typing_extensions-4.15.0.tar.gz", hash = "sha256:0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/18/67/36e9267722cc04a6b9f15c7f3441c2363321a3ea07da7ae0c0707beb2a9c/typing_extensions-4.15.0-py3-none-any.whl", hash = "sha256:f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548" }, +] + +[[package]] +name = "typing-inspection" +version = "0.4.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/55/e3/70399cb7dd41c10ac53367ae42139cf4b1ca5f36bb3dc6c9d33acdb43655/typing_inspection-0.4.2.tar.gz", hash = "sha256:ba561c48a67c5958007083d386c3295464928b01faa735ab8547c5692e87f464" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/dc/9b/47798a6c91d8bdb567fe2698fe81e0c6b7cb7ef4d13da4114b41d239f65d/typing_inspection-0.4.2-py3-none-any.whl", hash = "sha256:4ed1cacbdc298c220f1bd249ed5287caa16f34d44ef4e9c3d0cbad5b521545e7" }, +] + +[[package]] +name = "tzdata" +version = "2026.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/ba/19/1b9b0e29f30c6d35cb345486df41110984ea67ae69dddbc0e8a100999493/tzdata-2026.2.tar.gz", hash = "sha256:9173fde7d80d9018e02a662e168e5a2d04f87c41ea174b139fbef642eda62d10" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/ce/e4/dccd7f47c4b64213ac01ef921a1337ee6e30e8c6466046018326977efd95/tzdata-2026.2-py2.py3-none-any.whl", hash = "sha256:bbe9af844f658da81a5f95019480da3a89415801f6cc966806612cc7169bffe7" }, +] + +[[package]] +name = "tzlocal" +version = "5.4" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "tzdata", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/d8/52/ee2e6d7031687c5bad28363148cb72f2bbf38201d2e220671bd9fb830bc2/tzlocal-5.4.tar.gz", hash = "sha256:41e1293f80d4b5ff38dff222601a8fbd06b4fdcaf25e224704047ad26a39af54" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1d/70/5771c9ecbdb7cc0c3f3bbded7e0fa7911ee8e872ce5b5dc48ce7dce21a11/tzlocal-5.4-py3-none-any.whl", hash = "sha256:024d11221ff83453eae1f608f09b145b9779e1345d08c15404ce8ff7917cf629" }, +] + +[[package]] +name = "urllib3" +version = "2.7.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/53/0c/06f8b233b8fd13b9e5ee11424ef85419ba0d8ba0b3138bf360be2ff56953/urllib3-2.7.0.tar.gz", hash = "sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/7f/3e/5db95bcf282c52709639744ca2a8b149baccf648e39c8cc87553df9eae0c/urllib3-2.7.0-py3-none-any.whl", hash = "sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897" }, +] + +[[package]] +name = "uvicorn" +version = "0.49.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "click" }, + { name = "h11" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/c4/1f/fa18009dea8469069cca78a4e877a008ab78f08b064bfc9ab891579077ff/uvicorn-0.49.0.tar.gz", hash = "sha256:ebf4271aa580d9de97f93192d4595176df6e91f9aae919ca73e4fc07df1e66a3" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/88/fa/e1388bbcf24ef3274f45c0c1c7b501fd14971037c1b6ee23610553307497/uvicorn-0.49.0-py3-none-any.whl", hash = "sha256:ba3d14c3ee7e41c6c654c46c9eb489d33213cdd30aa1696eab1374337c13f68f" }, +] + +[[package]] +name = "win32-setctime" +version = "1.2.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b3/8f/705086c9d734d3b663af0e9bb3d4de6578d08f46b1b101c2442fd9aecaa2/win32_setctime-1.2.0.tar.gz", hash = "sha256:ae1fdf948f5640aae05c511ade119313fb6a30d7eabe25fef9764dca5873c4c0" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/e1/07/c6fe3ad3e685340704d314d765b7912993bcb8dc198f0e7a89382d37974b/win32_setctime-1.2.0-py3-none-any.whl", hash = "sha256:95d644c4e708aba81dc3704a116d8cbc974d70b3bdb8be1d150e36be6e9d1390" }, +] + +[[package]] +name = "xlrd" +version = "2.0.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/07/5a/377161c2d3538d1990d7af382c79f3b2372e880b65de21b01b1a2b78691e/xlrd-2.0.2.tar.gz", hash = "sha256:08b5e25de58f21ce71dc7db3b3b8106c1fa776f3024c54e45b45b374e89234c9" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/1a/62/c8d562e7766786ba6587d09c5a8ba9f718ed3fa8af7f4553e8f91c36f302/xlrd-2.0.2-py2.py3-none-any.whl", hash = "sha256:ea762c3d29f4cca48d82df517b6d89fbce4db3107f9d78713e48cd321d5c9aa9" }, +] + +[[package]] +name = "xlsxwriter" +version = "3.2.9" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/46/2c/c06ef49dc36e7954e55b802a8b231770d286a9758b3d936bd1e04ce5ba88/xlsxwriter-3.2.9.tar.gz", hash = "sha256:254b1c37a368c444eac6e2f867405cc9e461b0ed97a3233b2ac1e574efb4140c" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/3a/0c/3662f4a66880196a590b202f0db82d919dd2f89e99a27fadef91c4a33d41/xlsxwriter-3.2.9-py3-none-any.whl", hash = "sha256:9a5db42bc5dff014806c58a20b9eae7322a134abb6fce3c92c181bfb275ec5b3" }, +] + +[[package]] +name = "xxhash" +version = "3.7.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/24/2f/e183a1b407002f5af81822bee18b61cdb94b8670208ef34734d8d2b8ebe9/xxhash-3.7.0.tar.gz", hash = "sha256:6cc4eefbb542a5d6ffd6d70ea9c502957c925e800f998c5630ecc809d6702bae" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/3b/f4/7bd35089ff1f8e2c96baa2dce05775a122aacd2e3830a73165e27a4d0848/xxhash-3.7.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:fdc7d06929ae28dda98297a18eef7b0fd38991a3b405d8d7b55c9ef24c296958" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a3/26/4e00c88a6a2c8a759cfb77d2a9a405f901e8aa66e60ef1fd0aeb35edda48/xxhash-3.7.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:ea6daa712f4e094a30830cf01e9b47d03b24d05cc9dab8609f0d9a9db8454712" }, + { url = "https://mirrors.aliyun.com/pypi/packages/82/2f/eeb942c17a5a761a8f01cb9180a0b76bfb62a2c39e6f46b1f9001899027a/xxhash-3.7.0-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:9e6c0d843f1daf85ea23aeb053579135552bde575b7b98af20bfc667b6e4548d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0e/fd/96f132c08b1e5951c68691d3b9ec351ec2edc028f6a01fcd294f46b9d9f0/xxhash-3.7.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:363c139bf15e1ac5f136b981d3c077eb551299b1effede7f12faa010b8590a60" }, + { url = "https://mirrors.aliyun.com/pypi/packages/82/89/d4e92b796c5ed052d29ed324dbfc1dc1188e0c4bf64bebbf0f8fc20698df/xxhash-3.7.0-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:a778b25874cb0f862eaab5986bff4ca49ffb0def7c0a34c237b948b3c6c775b2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/40/f1/81fc4361921dc6e557a9c60cb3712f36d244d06eeeb71cd2f4252ac42678/xxhash-3.7.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:3e1860f1e43d40e9d904cf22d93e587ea42e010ebce4160877e46bcab4bc232a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6a/d0/afeddd4cff50a332f50d4b8a2e8857673153ab0564ef472fcdeb0b5430df/xxhash-3.7.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:9122ad6f867c4a0f5e655f5c3bdf89103852009dbb442a3d23e688b9e699e800" }, + { url = "https://mirrors.aliyun.com/pypi/packages/f7/d0/3c91e4e6a05ca4d7df8e39ec3a75b713609258ec84705ab34be6430826a1/xxhash-3.7.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d7d9110d0c3fb02679972837a033251fd186c529aa62f19c132fc909c74052b8" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4e/3a/a6b0772d9801dd4bea4ca4fd34734d6e9b51a711c8a611a24a79de26a878/xxhash-3.7.0-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:347a93f2b4ce67ce61959665e32a7447c380f8347e55e100daa23766baacf0e5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/6c/f8/cf8e31fd7282230fe7367cd501a2e75b4b67b222bfc7eacccfc20d2652cb/xxhash-3.7.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:acbb48679ddf3852c45280c10ff10d52ca2cd1da2e552fb81db1ff786c75d0e4" }, + { url = "https://mirrors.aliyun.com/pypi/packages/cc/f0/fd36cc4a81bf52ee5633275daae2b93dd958aace67fd4f5d466ec83b5f35/xxhash-3.7.0-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:fe14c356f8b23ad811dc026077a6d4abccdaa7bce5ca98579605550657b6fcfb" }, + { url = "https://mirrors.aliyun.com/pypi/packages/08/e1/67f5d9c9369be42eaf99ba02c01bf14c5ecd67087b02567960bfcee43b63/xxhash-3.7.0-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:f420ad3d41e38194353a498bbc9561fd5a9973a27b536ce46d8583479cf44335" }, + { url = "https://mirrors.aliyun.com/pypi/packages/50/17/a4c865ca22d2da6b1bc7d739bf88cab209533cf52ba06ca9da27c3039bee/xxhash-3.7.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:693d02c6dc7d1aa0a45921d54cd8c1ff629e09dfdc2238471507af1f7a1c6f04" }, + { url = "https://mirrors.aliyun.com/pypi/packages/49/8b/453b35810d697abac3c96bde3528bece685869227da274eb80a4a4d4a119/xxhash-3.7.0-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:14bf7a54e43825ec131ee7fe3c60e142e7c2c1e676ad0f93fc893432d15414af" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b5/ad/4eed7eab07fd3ee6678f416190f0413d097ab5d7c1278906bf1e9549d789/xxhash-3.7.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:ae3a39a4d96bdb6f8d154fd7f490c4ad06f0532fcd2bb656052a9a7762cf5d31" }, + { url = "https://mirrors.aliyun.com/pypi/packages/d3/4e/fd6f8a680ba248fdb83054fa71a8bfa3891225200de1708b888ef2c49829/xxhash-3.7.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:1cc07c639e3a77ef1d32987464d3e408565b8a3be57b545d3542b191054d9923" }, + { url = "https://mirrors.aliyun.com/pypi/packages/50/7c/8cb34b3bed4f44ca6827a534d50833f9bc6c006e83b0eb410ac9fa0793bd/xxhash-3.7.0-cp311-cp311-win32.whl", hash = "sha256:3281ba1d1e60ee7a382a7b958513ba03c2c0d5fcbd9a6f7517c0a81251a23422" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0b/47/a49767bd7b40782bedae9ff0721bfe1d7e4dd9dc1585dea684e57ba67c20/xxhash-3.7.0-cp311-cp311-win_amd64.whl", hash = "sha256:a7f25baec4c5d851d40718d6fae52285b31683093d4ff5207e63ab306ccf14a5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/7c/c6/3957bfacfb706bd687be246dfa8dd60f8df97c44186d229f7fd6e26c4b7e/xxhash-3.7.0-cp311-cp311-win_arm64.whl", hash = "sha256:4c2454448ce847c72635827bb75c15c5a3434b03ee1afd28cb6dc6fb2597d830" }, + { url = "https://mirrors.aliyun.com/pypi/packages/54/c1/e57ac7317b1f58a92bab692da6d497e2a7ce44735b224e296347a7ecc754/xxhash-3.7.0-pp311-pypy311_pp73-macosx_10_15_x86_64.whl", hash = "sha256:ad3aa71e12ee634f22b39a0ff439357583706e50765f17f05550f92dbf128a23" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4f/4e/075559bd712bc62e84915ea46bbee859f935d285659082c129bdbff679dd/xxhash-3.7.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:5de686e73690cdaf72b96d4fa083c230ec9020bcc2627ce6316138e2cf2fe2d1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/92/ca/a9c78cb384d4b033b0c58196bd5c8509873cabe76389e195127b0302a741/xxhash-3.7.0-pp311-pypy311_pp73-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:7fbec49f5341bbdea0c471f7d1e2fb41ae8925af9b6f28025c28defd8eb94274" }, + { url = "https://mirrors.aliyun.com/pypi/packages/bd/b1/dfe2629f7c77eb2fa234c72ff537cdd64939763df704e256446ed364a16d/xxhash-3.7.0-pp311-pypy311_pp73-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:48b542c347c2089f43dc5a6db31d2a6f3cdb04ee33505ec6e9f653834dbb0bde" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e7/f7/5a484afce0f48dd8083208b42e4911f290a82c7b52458ef2927e4d421a45/xxhash-3.7.0-pp311-pypy311_pp73-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:a169a036bed0995e090d1493b283cc2cc8a6f5046821086b843abefff80643bc" }, + { url = "https://mirrors.aliyun.com/pypi/packages/0f/5f/4acfcd490db9780cf36c58534d828003c564cde5350220a1c783c4d10776/xxhash-3.7.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:ec101643395d7f21405b640f728f6f627e6986557027d740f2f9b220955edafe" }, +] + +[[package]] +name = "yarl" +version = "1.24.2" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "idna" }, + { name = "multidict" }, + { name = "propcache" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/79/12/1e8f37460ea0f7eb59c221fdaf0ed75e7ac43e97f8093b9c6f411df50a78/yarl-1.24.2.tar.gz", hash = "sha256:9ac374123c6fd7abf64d1fec93962b0bd4ee2c19751755a762a72dd96c0378f8" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/c5/c5/1ce244152ff2839645e7cae92f90e7bafcb2c52bea7ff586ac714f14f5df/yarl-1.24.2-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:36348bebb147b83818b9d7e673ea4debc75970afc6ffdc7e3975ad05ce5a58c1" }, + { url = "https://mirrors.aliyun.com/pypi/packages/87/5a/00f36967203ed89cb3acd2c8ed526cc3fed9418eb70ce128160a911c8499/yarl-1.24.2-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:1a97e42c8a2233f2f279ecadd9e4a037bcb5d813b78435e8eedd4db5a9e9708c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/31/d0/1fb0c1cd27288f39f6974da4318c32768d72c9890984541fdf1e2e32a51d/yarl-1.24.2-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:8d027d56f1035e339d1001ac33eceab5b2ec8e42e449787bb75e289fb9a5cd1d" }, + { url = "https://mirrors.aliyun.com/pypi/packages/03/ce/d4a646508bed2f8dec6435b40166fe9308dd191262033d3f307b2bbcaecd/yarl-1.24.2-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0a6377060e7927187a42b7eb202090cbe2b34933a4eeaf90e3bd9e33432e5cae" }, + { url = "https://mirrors.aliyun.com/pypi/packages/4b/07/b3278e82d8bc41485bcf6d856cd0433262593de615b1d3dc43bd3f5bead4/yarl-1.24.2-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:17076578bce0049a5ce57d14ad1bded391b68a3b213e9b81b0097b090244999a" }, + { url = "https://mirrors.aliyun.com/pypi/packages/17/5b/4cee6e7c92e487bebe7afc797da0aa54a248ab4e776a68fe369ec29665a5/yarl-1.24.2-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:50713f1d4d6be6375bb178bb43d140ee1acb8abe589cd723320b7925a275be1e" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5c/82/111076571545a7d4f9cca3fbd5c6f40615af58642be09f12328f48022468/yarl-1.24.2-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:34263e2fa8fb5bb63a0d97706cda38edbad62fddb58c7f12d6acbc092812aa50" }, + { url = "https://mirrors.aliyun.com/pypi/packages/b6/ec/08f671f69a444d704aeecebf92af659b67b97a869942411d0a578b08c334/yarl-1.24.2-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:49016d82f032b1bd1e10b01078a7d29ae71bf468eeae0ea22df8bab691e60003" }, + { url = "https://mirrors.aliyun.com/pypi/packages/e5/86/ce41e7a7a199340b2330d52b60f25c4074b6636dd0e60b1a80d31a9db042/yarl-1.24.2-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:3f6d2c216318f8f32038ca3f72501ba08536f0fd18a36e858836b121b2deed9f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c4/5d/31be8a729531ab3e55ac3e7e5c800be8c89ea98947f418b2f6ea259fb6ee/yarl-1.24.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:08d3a33218e0c64393e7610284e770409a9c31c429b078bcb24096ed0a783b8f" }, + { url = "https://mirrors.aliyun.com/pypi/packages/47/9b/b57afb22b386ae87ac9940f09878b98d8c333f89113e6fc96fcf4ca9eb64/yarl-1.24.2-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:5d699376c4ca3cba49bbfae3a05b5b70ded572937171ce1e0b8d87118e2ba294" }, + { url = "https://mirrors.aliyun.com/pypi/packages/a3/4f/06348c27c8389256c313e8a57d796808fc0264c915dd5e7cfd3c0e314dc7/yarl-1.24.2-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:a1cab588b4fa14bea2e55ebea27478adfb05372f47573738e1acc4a36c0b05d2" }, + { url = "https://mirrors.aliyun.com/pypi/packages/5f/1c/284f307b298e4a17b7943b07d9d7ecc4151537f8d137ba51f3bb6c31ca20/yarl-1.24.2-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:ec87ccc31bd21db7ad009d8572c127c1000f268517618a4cc09adba3c2a7f21c" }, + { url = "https://mirrors.aliyun.com/pypi/packages/c8/bf/0de123bec8619e45c80cbded9085f61b5b4a9eddb8abe6d25d28ee1ec866/yarl-1.24.2-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:d1dd47a22843b212baa8d74f37796815d43bd046b42a0f41e9da433386c3136b" }, + { url = "https://mirrors.aliyun.com/pypi/packages/90/af/0248eb065e51129d2a9b2436cd1b5c772c19a6b04e5b6a186955671e3319/yarl-1.24.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7b54b9c67c2b06bd7b9a77253d242124b9c95d2c02def5a1144001ee547dd9d5" }, + { url = "https://mirrors.aliyun.com/pypi/packages/21/3c/f960d7a65ef97d8ba9b424fb5128796a4bc710fc6df2ddbbd7dfdc3bbd20/yarl-1.24.2-cp311-cp311-win_amd64.whl", hash = "sha256:f8fdbcff8b2c7c9284e60c196f693588598ddcee31e11c18e14949ce44519d45" }, + { url = "https://mirrors.aliyun.com/pypi/packages/03/1a/49fb03750e4de4d2284cd5b885a383133c34eef45bd59631b2bb8b7e81e8/yarl-1.24.2-cp311-cp311-win_arm64.whl", hash = "sha256:b32c37a7a337e90822c45797bf3d79d60875cfcccd3ecc80e9f453d87026c122" }, + { url = "https://mirrors.aliyun.com/pypi/packages/fd/4d/4b880086bd0d3e034d25647be1d830afc3e3f610e98c4ab3490af6b1b6d5/yarl-1.24.2-py3-none-any.whl", hash = "sha256:2783d9226db8797636cd6896e4de81feed252d1db72265686c9558d97a4d94b9" }, +] + +[[package]] +name = "youtube-transcript-api" +version = "1.0.3" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +dependencies = [ + { name = "defusedxml" }, + { name = "requests" }, +] +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b0/32/f60d87a99c05a53604c58f20f670c7ea6262b55e0bbeb836ffe4550b248b/youtube_transcript_api-1.0.3.tar.gz", hash = "sha256:902baf90e7840a42e1e148335e09fe5575dbff64c81414957aea7038e8a4db46" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/f0/44/40c03bb0f8bddfb9d2beff2ed31641f52d96c287ba881d20e0c074784ac2/youtube_transcript_api-1.0.3-py3-none-any.whl", hash = "sha256:d1874e57de65cf14c9d7d09b2b37c814d6287fa0e770d4922c4cd32a5b3f6c47" }, +] + +[[package]] +name = "zipp" +version = "4.1.0" +source = { registry = "https://mirrors.aliyun.com/pypi/simple/" } +sdist = { url = "https://mirrors.aliyun.com/pypi/packages/b9/d8/eab98a517c14134c0b2eb4e2387bc5f457334293ec5d2dd3857ec2966802/zipp-4.1.0.tar.gz", hash = "sha256:4cb57381f544315db7688e976e922a2b18cdb513d21cc194eb42232ba2a3e602" } +wheels = [ + { url = "https://mirrors.aliyun.com/pypi/packages/3a/13/547360d81e6d88d58492968ffda9f9542854f11310ee556fef14260cc886/zipp-4.1.0-py3-none-any.whl", hash = "sha256:25ad4e16390cd314347dd8f1de67a2ac538ae658ed4ab9db16029c07c188e97f" }, +] diff --git a/vectorstore/__init__.py b/vectorstore/__init__.py new file mode 100644 index 0000000..dee8632 --- /dev/null +++ b/vectorstore/__init__.py @@ -0,0 +1,25 @@ +"""Qdrant 向量知识库模块 (M6)。 + +公共 API: + - VectorStore: Qdrant 封装(init / upsert / query / info / count) + - SearchFilter / SearchResult / CollectionInfo: 数据模型 + - make_qdrant_client: 工厂(内存/Docker) +""" + +from .client import ( + DEFAULT_COLLECTION, + DEFAULT_VECTOR_DIM, + VectorStore, + make_qdrant_client, +) +from .models import CollectionInfo, SearchFilter, SearchResult + +__all__ = [ + "DEFAULT_COLLECTION", + "DEFAULT_VECTOR_DIM", + "CollectionInfo", + "SearchFilter", + "SearchResult", + "VectorStore", + "make_qdrant_client", +] diff --git a/vectorstore/client.py b/vectorstore/client.py new file mode 100644 index 0000000..596afbe --- /dev/null +++ b/vectorstore/client.py @@ -0,0 +1,326 @@ +"""Qdrant 客户端封装 (M6)。 + +核心: + - 连接:本地文件模式(默认,嵌入运行无需 Docker)或 HTTP 远程模式 + - 初始化 Collection:1024 维 / 余弦距离 + - upsert:幂等写入(url_hash 做 point ID) + - query:语义检索 + 结构化过滤 + - info / delete / count:运维辅助 +""" + +from __future__ import annotations + +import os +import uuid +from datetime import datetime +from pathlib import Path +from typing import Any + +from loguru import logger +from qdrant_client import QdrantClient +from qdrant_client.http.models import ( + DatetimeRange, + Distance, + FieldCondition, + Filter, + MatchAny, + MatchValue, + PointStruct, + Range, + VectorParams, +) + +from .models import CollectionInfo, SearchFilter, SearchResult + +# 默认配置 +DEFAULT_HOST = "localhost" +DEFAULT_PORT = 6333 +DEFAULT_COLLECTION = "a_share_news" +DEFAULT_VECTOR_DIM = 1024 +DEFAULT_DISTANCE = Distance.COSINE + +# UUID namespace for url_hash -> UUID conversion (确定性,便于幂等 upsert) +_UUID_NAMESPACE = uuid.UUID("a1b2c3d4-e5f6-7890-abcd-ef1234567890") + + +def url_hash_to_uuid(url_hash: str) -> str: + """把 16 位 hex url_hash 转为 UUID 字符串(point ID 要求)。 + + 使用 uuid5 保证确定性——相同 url_hash 总是得到相同 UUID。 + """ + return str(uuid.uuid5(_UUID_NAMESPACE, url_hash)) + + +def _read_env(key: str, default: str | None = None) -> str | None: + val = os.environ.get(key) + if val is None or val.strip() == "": + return default + return val.strip() + + +def make_qdrant_client( + host: str | None = None, + port: int | None = None, + *, + memory: bool = False, + path: str | None = None, +) -> QdrantClient: + """构造 QdrantClient。 + + 模式优先级: + 1. memory=True -> 内存模式(测试用) + 2. path 非空 -> 本地文件模式(嵌入运行,无需 Docker,默认 data/qdrant_storage) + 3. host/port -> 远程 HTTP 模式(需要单独 Qdrant 服务) + + 树莓派 5 ARM64 的 Docker Qdrant 不兼容 16K 页内核, + 推荐默认用本地文件模式。 + """ + if memory: + logger.debug("Qdrant 内存模式") + return QdrantClient(location=":memory:") + + # 本地文件模式:显式 path 或 host 未指定时默认走文件 + if path is not None or (host is None and port is None): + use_path = path or str(DEFAULT_STORAGE_PATH) + logger.debug("Qdrant 本地文件模式: {}", use_path) + return QdrantClient(path=use_path) + + h = host or _read_env("QDRANT_HOST", DEFAULT_HOST) or DEFAULT_HOST + p = int(port or int(_read_env("QDRANT_PORT", str(DEFAULT_PORT)) or DEFAULT_PORT)) # type: ignore[arg-type] + api_key = _read_env("QDRANT_API_KEY") or None + url = f"http://{h}:{p}" + logger.debug("Qdrant HTTP {} (key={})", url, "yes" if api_key else "no") + return QdrantClient(url=url, api_key=api_key, timeout=10) + +# 默认持久化目录 +DEFAULT_STORAGE_PATH = Path("data/qdrant_storage") + + +# --------------------------------------------------------------------------- # +# 客户端封装 +# --------------------------------------------------------------------------- # + +class VectorStore: + """Qdrant 向量知识库封装。 + + 线程不安全,批处理串行使用即可。 + """ + + def __init__( + self, + client: QdrantClient, + collection_name: str | None = None, + vector_dim: int = DEFAULT_VECTOR_DIM, + ) -> None: + self._c = client + self.collection_name = collection_name or ( + _read_env("QDRANT_COLLECTION", DEFAULT_COLLECTION) or DEFAULT_COLLECTION + ) + self.vector_dim = vector_dim + + # ------------------------------------------------------------------ # + # Collection 管理 + # ------------------------------------------------------------------ # + + def init_collection(self, *, recreate: bool = False) -> None: + """创建 collection(已存在时若 recreate 则重建)。 + + 幂等:已存在且非 recreate 时直接返回。 + """ + exists = self._c.collection_exists(self.collection_name) + if exists and not recreate: + logger.debug("Collection {} 已存在,跳过初始化", self.collection_name) + return + if exists and recreate: + logger.warning("重建 collection {}", self.collection_name) + self._c.delete_collection(self.collection_name) + exists = False + + self._c.create_collection( + collection_name=self.collection_name, + vectors_config=VectorParams( + size=self.vector_dim, + distance=DEFAULT_DISTANCE, + ), + ) + logger.info( + "已创建 collection {} (dim={} distance={})", + self.collection_name, self.vector_dim, DEFAULT_DISTANCE.name, + ) + + def delete_collection(self) -> None: + if self._c.collection_exists(self.collection_name): + self._c.delete_collection(self.collection_name) + logger.info("已删除 collection {}", self.collection_name) + + def info(self) -> CollectionInfo: + exists = self._c.collection_exists(self.collection_name) + if not exists: + return CollectionInfo(name=self.collection_name, exists=False, vectors_count=0) + c_info = self._c.get_collection(self.collection_name) + return CollectionInfo( + name=self.collection_name, + exists=True, + vectors_count=c_info.points_count or 0, + indexed_vectors_count=getattr(c_info, "indexed_vectors_count", None), + segments_count=getattr(c_info, "segments_count", None), + ) + + def count(self) -> int: + try: + return self._c.count(self.collection_name).count + except Exception: # noqa: BLE001 - collection 不存在时优雅退化 + return 0 + + # ------------------------------------------------------------------ # + # 数据写入(幂等 upsert) + # ------------------------------------------------------------------ # + + def upsert( + self, + points: list[dict[str, Any]], + *, + batch_size: int = 100, + ) -> int: + """批量幂等写入。 + + 参数: + points: 每个 dict 包含: + id (str) point ID(用 url_hash) + vector (list[float]) 嵌入向量 + payload (dict) 任意结构化数据 + batch_size: 每批写入条数。 + + 返回: 写入条数。 + """ + structs = [ + PointStruct( + id=url_hash_to_uuid(p["id"]), + vector=p["vector"], + payload={"url_hash": p["id"], **(p.get("payload") or {})}, + ) + for p in points + ] + total = len(structs) + for i in range(0, total, batch_size): + chunk = structs[i : i + batch_size] + self._c.upsert(collection_name=self.collection_name, points=chunk) + logger.debug("upsert 批 {}/{} ({} 条)", i // batch_size + 1, (total + batch_size - 1) // batch_size, len(chunk)) + logger.info("upsert 完成: {} 条 -> collection {}", total, self.collection_name) + return total + + # ------------------------------------------------------------------ # + # 检索 + # ------------------------------------------------------------------ # + + def query( + self, + query_vector: list[float], + *, + top_k: int = 10, + filter: SearchFilter | None = None, + score_threshold: float | None = None, + ) -> list[SearchResult]: + """语义检索 + 可选结构化过滤。 + + 参数: + query_vector: 嵌入向量(需与 collection 维度一致)。 + top_k: 返回条数。 + filter: 结构化过滤(AND 关系)。 + score_threshold: 最低余弦相似度。 + + 返回: 列表按 score 降序。 + """ + # 构建 Qdrant Filter + q_filter = _build_filter(filter) + hits = self._c.query_points( + collection_name=self.collection_name, + query=query_vector, + query_filter=q_filter, + limit=top_k, + score_threshold=score_threshold, + with_payload=True, + with_vectors=False, + ) + results: list[SearchResult] = [] + for p in hits.points: + payload = p.payload or {} + pt_raw = payload.get("publish_time") + publish_time = ( + datetime.fromisoformat(pt_raw) + if isinstance(pt_raw, str) and pt_raw + else None + ) + results.append(SearchResult( + url_hash=payload.get("url_hash") or "", + score=p.score if p.score is not None else 0.0, + title=payload.get("title") or "", + url=payload.get("url") or "", + source_id=payload.get("source_id") or "", + publish_time=publish_time, + event=payload.get("event"), + char_count=payload.get("char_count"), + word_count=payload.get("word_count"), + )) + logger.debug( + "检索完成 top_k={} filter={} -> {} 条", + top_k, filter, len(results), + ) + return results + + def close(self) -> None: + self._c.close() + + def __enter__(self) -> VectorStore: + return self + + def __exit__(self, *_: object) -> None: + self.close() + + +# --------------------------------------------------------------------------- # +# Filter 构建 +# --------------------------------------------------------------------------- # + +def _build_filter(f: SearchFilter | None) -> Filter | None: + """把 SearchFilter 转换为 Qdrant Filter。""" + if f is None: + return None + conditions: list[FieldCondition] = [] + + if f.source_id: + conditions.append(FieldCondition(key="source_id", match=MatchValue(value=f.source_id))) + if f.source_ids: + conditions.append(FieldCondition(key="source_id", match=MatchAny(any=f.source_ids))) + if f.stock_codes: + conditions.append(FieldCondition(key="event.stock_codes", match=MatchAny(any=f.stock_codes))) + if f.company_names: + conditions.append(FieldCondition(key="event.company_names", match=MatchAny(any=f.company_names))) + if f.industries: + conditions.append(FieldCondition(key="event.industries", match=MatchAny(any=f.industries))) + if f.sentiment: + conditions.append(FieldCondition(key="event.sentiment", match=MatchValue(value=f.sentiment))) + if f.importance_min is not None: + conditions.append(FieldCondition(key="event.importance", range=Range(gte=f.importance_min))) + if f.event_types: + conditions.append(FieldCondition(key="event.event_type", match=MatchAny(any=f.event_types))) + if f.publish_date_from or f.publish_date_to: + try: + range_kwargs: dict[str, datetime] = {} + if f.publish_date_from: + range_kwargs["gte"] = datetime.fromisoformat(f.publish_date_from + "T00:00:00") + if f.publish_date_to: + range_kwargs["lte"] = datetime.fromisoformat(f.publish_date_to + "T23:59:59") + conditions.append(FieldCondition( + key="publish_time", + range=DatetimeRange(**range_kwargs), + )) + except ValueError: + logger.warning( + "filter 日期格式错误 from={!r} to={!r},跳过日期过滤", + f.publish_date_from, f.publish_date_to, + ) + + if not conditions: + return None + return Filter(must=conditions) diff --git a/vectorstore/models.py b/vectorstore/models.py new file mode 100644 index 0000000..66cbd27 --- /dev/null +++ b/vectorstore/models.py @@ -0,0 +1,54 @@ +"""Qdrant 向量存储模块的数据模型 (M6)。""" + +from __future__ import annotations + +from datetime import datetime +from typing import Any + +from pydantic import BaseModel, Field + + +class SearchFilter(BaseModel): + """可选检索过滤条件,全部为 AND 关系。""" + + source_id: str | None = None + source_ids: list[str] | None = None + stock_codes: list[str] | None = Field(default=None, description="match any") + company_names: list[str] | None = Field(default=None, description="match any") + industries: list[str] | None = Field(default=None, description="match any") + sentiment: str | None = None # positive / neutral / negative + importance_min: int | None = None # >= N + event_types: list[str] | None = None # match any + publish_date_from: str | None = None # YYYY-MM-DD + publish_date_to: str | None = None # YYYY-MM-DD + + +class SearchResult(BaseModel): + """单条检索结果。""" + + url_hash: str + score: float + title: str + url: str + source_id: str + publish_time: datetime | None = None + event: dict[str, Any] | None = None # EventExtraction 展开的 dict + char_count: int | None = None + word_count: int | None = None + + def short_summary(self) -> str: + codes = ( + ",".join((self.event or {}).get("stock_codes", [])) + if self.event else "-" + ) + return f"[{self.source_id}] score={self.score:.4f} 《{self.title[:40]}》 {codes}" + + +class CollectionInfo(BaseModel): + """Collection 概览信息。""" + + name: str + exists: bool + vectors_count: int + indexed_vectors_count: int | None = None + segments_count: int | None = None