From 60f8c263f2b8bf61b46204630baa421fa841518d Mon Sep 17 00:00:00 2001 From: simon Date: Fri, 25 Sep 2026 11:17:46 +0800 Subject: [PATCH] =?UTF-8?q?=E3=80=8A=E6=96=B0=E9=97=BB=E8=81=94=E6=92=AD?= =?UTF-8?q?=E3=80=8B=E6=AF=8F=E6=97=A5=E6=8A=93=E5=8F=96=E5=85=A5=E5=BA=93?= =?UTF-8?q?=EF=BC=9A=E5=85=A8=E9=93=BE=E8=B7=AF=20+=20=E5=8F=AF=E7=A7=BB?= =?UTF-8?q?=E6=A4=8D=E5=8C=96=20+=20=E5=AE=9A=E6=97=B6=E4=BB=BB=E5=8A=A1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 从 CCTV 主页抓取《新闻联播》,下载 → 转 MP3/WAV → 静音切分 → ASR 识别 → LLM 校对 → 切分为单条新闻 → 入库 MySQL。 主要内容: - 全链路:getVideo5 抓取下载、audioRead 转写、deepseek 校对与切分、newsProcess 入库 - 可移植化:配置分层,.env 只放密钥、config.yml 放模型/接入点/路由/参数 - 可换供应商:endpoints(kind/base_url/api_key_env/extra_body)+ routes 按环节选路 - 数据保真:数值事实守卫,校对改动数字/年份/届次则整片回退 ASR 原文; 识别不完整不发布该日精编,避免半天内容被当成完整一天 - 定时任务:systemd 每天 21:00,失败 21:30 / 22:00 重试; 只缺切分时只重跑切分(省掉全部 ASR),用 state/.asr_complete_* 标记判定阶段 - 隧道自愈:13306 不通时自动执行 autossh.sh(所有入口共用,systemd 托管时只等待) - 中间产物每日清理;97 项离线自检(配置/清理/事实守卫/解析/隧道) --- .env.example | 30 ++ .gitignore | 17 + README.md | 508 ++++++++++++++++++ audioRead.py | 684 ++++++++++++++++++++++++ autossh.sh | 3 + cleanup.py | 119 +++++ config.py | 403 ++++++++++++++ config.yml | 136 +++++ deepseek.py | 365 +++++++++++++ docs/ARCHITECTURE.md | 393 ++++++++++++++ docs/BUGS.md | 363 +++++++++++++ docs/REPORT_raw_vs_improve.md | 142 +++++ env.py | 62 +++ getVideo5.py | 288 ++++++++++ main.py | 25 + main_videos.py | 75 +++ mysqlHandle.py | 9 + mysql_handler.py | 240 +++++++++ newsProcess.py | 271 ++++++++++ newsRedo.py | 122 +++++ requirements.txt | 34 ++ scripts/day_status.py | 52 ++ scripts/install_systemd.sh | 51 ++ scripts/run_daily.sh | 124 +++++ systemd/xwlb-daily.service | 31 ++ systemd/xwlb-daily.timer | 13 + systemd/xwlb-tunnel.service | 21 + tests/test_cleanup.py | 106 ++++ tests/test_config.py | 173 ++++++ tests/test_fidelity_guard.py | 70 +++ tests/test_news_parse.py | 84 +++ tests/test_tunnel.py | 183 +++++++ tools/compare_raw_improve.py | 273 ++++++++++ tools/experiment_merge_correct_split.py | 100 ++++ tunnel.py | 200 +++++++ 35 files changed, 5770 insertions(+) create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 README.md create mode 100644 audioRead.py create mode 100755 autossh.sh create mode 100644 cleanup.py create mode 100644 config.py create mode 100644 config.yml create mode 100644 deepseek.py create mode 100644 docs/ARCHITECTURE.md create mode 100644 docs/BUGS.md create mode 100644 docs/REPORT_raw_vs_improve.md create mode 100644 env.py create mode 100644 getVideo5.py create mode 100644 main.py create mode 100644 main_videos.py create mode 100644 mysqlHandle.py create mode 100644 mysql_handler.py create mode 100644 newsProcess.py create mode 100644 newsRedo.py create mode 100644 requirements.txt create mode 100755 scripts/day_status.py create mode 100755 scripts/install_systemd.sh create mode 100755 scripts/run_daily.sh create mode 100644 systemd/xwlb-daily.service create mode 100644 systemd/xwlb-daily.timer create mode 100644 systemd/xwlb-tunnel.service create mode 100644 tests/test_cleanup.py create mode 100644 tests/test_config.py create mode 100644 tests/test_fidelity_guard.py create mode 100644 tests/test_news_parse.py create mode 100644 tests/test_tunnel.py create mode 100644 tools/compare_raw_improve.py create mode 100644 tools/experiment_merge_correct_split.py create mode 100644 tunnel.py diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..f2fe423 --- /dev/null +++ b/.env.example @@ -0,0 +1,30 @@ +# ============================================================================ +# xwlb 敏感配置模板 —— 复制为 .env 后填写真实值 +# ============================================================================ +# cp .env.example .env +# +# 本文件只放 **敏感项**(口令 / API Key)。 +# 非敏感配置(数据库地址、目录、各环节模型、切分参数、清理开关)在 config.yml。 +# ============================================================================ + +# ---- MySQL / MariaDB 口令(地址/端口/用户/库名见 config.yml 的 mysql.*)---- +MYSQL_PASSWORD= + +# ---- 各供应商 API Key ---- +# 变量名不是写死在代码里的:config.yml 的 endpoints.<接入点>.api_key_env 决定用哪个变量。 +# 换供应商只需在 config.yml 里加接入点并写好 api_key_env,然后在这里加对应变量。 +DASHSCOPE_API_KEY= # 语音识别(ASR) + 校对(config.yml: endpoints.dashscope.api_key_env) +DEEPSEEK_API_KEY= # 新闻切分 + 标题(config.yml: endpoints.deepseek.api_key_env) +# MOONSHOT_API_KEY= # 示例:把校对换成月之暗面时新增 + +# ============================================================================ +# 可选:覆盖默认路径(一般不需要) +# ============================================================================ +# XWLB_ENV_FILE=/path/to/.env # 指定其他 .env +# XWLB_CONFIG_FILE=/path/to/config.yml # 指定其他 config.yml + +# ============================================================================ +# 说明:进程环境变量 > config.yml > 代码内置默认值 +# 因此临时试验可直接用环境变量覆盖,例如: +# DASHSCOPE_LLM_MODEL=qwen-max python audioRead.py 20260904 +# ============================================================================ \ No newline at end of file diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..bd0e12a --- /dev/null +++ b/.gitignore @@ -0,0 +1,17 @@ +# 运行产物(体积大,不入版本库) +xwlb_video/ +audio_processing/ +main.log +*.log +.run_daily.lock +state/ + +# 密钥与本地配置 +.env +.env.bak.* + +# Python +__pycache__/ +*.py[cod] +.venv/ +venv/ \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..90374b8 --- /dev/null +++ b/README.md @@ -0,0 +1,508 @@ +# xwlb —— 央视《新闻联播》自动转写与结构化入库流水线 + +把央视网《新闻联播》**完整版视频**自动抓取下来,转写成文字,再交给大模型切分成**一条条独立新闻**(含标题、正文)写进 MySQL/MariaDB。 + +全流程一条命令跑完,可幂等重跑、失败可定位、跑完自动清理中间产物。 + +> 内容版权归中央广播电视总台所有。本项目仅用于个人学习与研究用途,请勿用于商业分发。 + +--- + +## 目录 + +- [一、功能](#一功能) +- [二、技术架构](#二技术架构) +- [三、使用说明](#三使用说明) +- [四、相关文档](#四相关文档) + +--- + +## 一、功能 + +### 1.1 主流程(`python main.py` 一条命令走完) + +| 步骤 | 做什么 | 产出 | +|---|---|---| +| ① 链接发现 | 抓 `tv.cctv.com/lm/xwlb/day/YYYYMMDD.shtml`,匹配出当天「完整版《新闻联播》」的 VID 详情页 | 视频页 URL | +| ② 下载与抽音 | yt-dlp 下载 mp4,ffmpeg 抽出 mp3(已存在则跳过下载) | `xwlb_video/YYYYMMDD.mp4` / `.mp3` | +| ③ 转写落库 | mp3 → 16kHz 单声道 wav → 按静音切成 ~11 片 → 逐片语音识别 → LLM 校对 → 写入 `xwlb_daily` | 分片级原文 + 校对文本 | +| ④ 结构化精编 | 把当天全部分片文本拼起来交给 DeepSeek,切分成独立新闻并起标题 | `xwlb_daily_ext` 多行(标题 + 正文) | +| ⑤ 自动清理 | 当天全部成功后才清空 mp3/mp4/wav | 释放磁盘(实测单日约 121MB) | + +### 1.2 工程能力 + +- **幂等重跑**:`xwlb_daily` 按 `(日期, 分片号)` 先删后插;`xwlb_daily_ext` 按日期整体替换。同一天跑多少遍结果都一致,不会产生重复行。 +- **绝不丢已识别文本**:校对环节任何失败(超时、限流、返回空)都自动回退 ASR 原文,不会因为"润色"把整片文字丢掉。 +- **数值事实守卫**:LLM 校对若改动了数字/年份/届次/规划期等事实(实测出现过 `2027→2024`、`十五五→十四五`、`第十一届→第九届`),该分片**整片回退 ASR 原文**。 +- **不写脏数据**:识别为空的分片不写库;失败当天不删除已有数据,留给下次重跑。 +- **可重跑的分步入口**:只重做转写、只重做切分、强制重切,都有独立命令。 +- **配置分层**:口令/Key 在 `.env`,其余(含**三个模型名**与**各供应商 base url**)在 `config.yml`;换模型、换供应商都不用动代码。 +- **失败说得清楚**:缺 Key、连不上库(含隧道提示)、模型返回非 JSON 等都有明确日志与修复指引。 +- **97 项离线自检**:配置、清理、事实守卫、LLM 返回解析、隧道自愈五个测试套件,不依赖网络与真实 API。 + +### 1.3 当前数据规模(2026-09-25 实测) + +| 表 | 行数 | 覆盖天数 | 日期范围 | +|---|---|---|---| +| `xwlb_daily`(分片级原文) | 7,973 | 725 | 2024-09-26 ~ 2026-09-23 | +| `xwlb_daily_ext`(单条新闻) | 11,494 | 523 | 2024-09-26 ~ 2026-09-23 | + +单日耗时约 **25~35 分钟**(下载 + 抽音 + 11 片识别 + 校对 + 切分),全程顺序执行、无并发。 + +--- + +## 二、技术架构 + +### 2.1 数据流 + +``` + main.py(当天) + │ start_date = end_date = %Y%m%d + ▼ + ┌───────────────────────────────────────────────────────────────┐ + │ ① 链接发现 getVideo5.get_all_video_links(start, end) │ + │ xwlb_urls() → tv.cctv.com/lm/xwlb/day/YYYYMMDD.shtml + │ get_xwlb_video_link() → 匹配「完整版《新闻联播》」得到 VID 页 │ + ├───────────────────────────────────────────────────────────────┤ + │ ② 下载与抽音 download_and_extract_audio(url, date, dir) │ + │ yt-dlp best[ext=mp4]/best → xwlb_video/YYYYMMDD.mp4 │ + │ ffmpeg -c:a libmp3lame -q:a 0 → xwlb_video/YYYYMMDD.mp3 │ + ├───────────────────────────────────────────────────────────────┤ + │ ③ 转写落库 audioRead.process_long_audio(mp3, out_dir, date) │ + │ convert_mp3_to_wav() 16kHz / 单声道 → input_YYYYMMDD.wav│ + │ split_audio_by_smart_silence(700ms, -40dBFS, keep 400ms) │ + │ → 贪心合并为 ≤3 分钟分片 chunk_i.wav(实测约 11 片/天) │ + │ for each chunk: │ + │ transcribe_audio() dashscope paraformer ASR │ + │ analyze_and_correct_text() LLM 校对 + 数值事实守卫 │ + │ _upsert_daily_chunk() → xwlb_daily(先删后插) │ + ├───────────────────────────────────────────────────────────────┤ + │ ④ 切分入库 newsProcess.news_to_db(date) │ + │ 读当天 news_improve 按分片号拼接 → DeepSeek(json_object) │ + │ → 解析 [{news_id, news_title, news_content}] │ + │ → 删除该日期旧记录 → 批量写入 xwlb_daily_ext │ + ├───────────────────────────────────────────────────────────────┤ + │ ⑤ 清理 cleanup.maybe_cleanup_after_run(date, day_ok) │ + │ 成功才清空 xwlb_video/*.mp3|*.mp4 与 audio_processing/*.wav │ + └───────────────────────────────────────────────────────────────┘ +``` + +### 2.2 技术栈 + +| 层 | 选型 | 用途 | +|---|---|---| +| 语言/运行 | Python 3.13 + `.venv` | 本机为 externally-managed 环境,必须用虚拟环境 | +| 网页抓取 | `requests` + `beautifulsoup4` | 解析央视按天页面,定位完整版视频 | +| 视频下载 | `yt-dlp` | 下载 mp4 | +| 音频处理 | `ffmpeg`(系统依赖)+ `pydub` + `audioop-lts` | 抽音、16k 单声道转换、静音切分(3.13 已移除 `audioop`,需补丁包) | +| 语音识别 | 阿里云 DashScope `paraformer-realtime-v2` | 中文长音频转写(接入点可配) | +| 文本校对 | 阿里云 DashScope `qwen3.7-max` | 修正 ASR 的错别字/断句(受事实守卫约束) | +| 新闻切分 | DeepSeek `deepseek-chat`(`json_object`) | 切分为独立新闻 + 起标题 | +| LLM 接入 | 可配置的 **endpoints + routes** | base url / 协议类型 / 密钥变量名都在 `config.yml`,可换供应商 | +| 数据库 | MySQL / MariaDB 10.11 + `mysql-connector-python` | 结果入库 | +| 配置 | `PyYAML`(config.yml)+ 自研 dotenv 解析(.env) | 敏感/非敏感分开 | +| 远程访问 | `autossh` | 本地 13306 → 远端 3306 隧道 | + +### 2.3 目录结构 + +``` +xwlb/ +├── main.py # 每日入口(当天) +├── getVideo5.py # 链接发现 + 下载 + 主流程编排 +├── audioRead.py # 转写 + 校对 + 分片写库 +├── newsProcess.py # DeepSeek 切分 + 写 xwlb_daily_ext +├── deepseek.py # DeepSeek 客户端(重试 / 退避 / json_object) +├── cleanup.py # 中间产物清理(自动钩子 + 手动命令) +├── config.py / config.yml# 非敏感配置加载器 / 配置本体(含三个模型名) +├── env.py # 敏感配置(.env)加载与缺失检查 +├── mysql_handler.py # 数据库层(项目内置,不再依赖父项目) +├── mysqlHandle.py # 转发 mysql_handler.MySQLDB +├── newsRedo.py # 手动重跑某天(跳过 / 只切分 / 全流程) +├── main_videos.py # 补缺失日期(区间写在文件内) +├── .env / .env.example # 敏感项 / 模板 +├── requirements.txt +├── autossh.sh # SSH 隧道脚本 +├── xwlb_video/ # 中间产物:YYYYMMDD.mp4 / .mp3(成功后清空) +├── audio_processing/ # 中间产物:input_*.wav / chunk_*.wav(成功后清空) +├── docs/ # ARCHITECTURE.md(技术说明)、BUGS.md(缺陷清单) +└── tests/ # 4 个离线测试套件 +``` + +### 2.4 数据库结构 + +**`xwlb_daily` —— 音频分片级文本**(一天的完整版被切成约 11 片,每片一行) + +| 字段 | 类型 | 说明 | +|---|---|---| +| `nid` | int, PK, auto_increment | 自增主键 | +| `news_days` | date | 日期(`YYYY-MM-DD`) | +| `daily_sub_id` | int | 分片序号,从 0 开始 | +| `news_raw` | text | ASR 原始识别文本 | +| `news_improve` | text | 校对后文本(关闭校对或守卫回退时等于 `news_raw`) | +| `news_title` | text | 保留字段(当前流程未使用) | + +> 表中无 `(news_days, daily_sub_id)` 唯一索引,幂等性由代码「先删后插」保证。 + +**建表语句**(新环境可直接执行;库名默认 `myquant`) + +```sql +CREATE TABLE IF NOT EXISTS xwlb_daily ( + nid int(11) NOT NULL AUTO_INCREMENT, + news_days date NOT NULL, + daily_sub_id int(11) NOT NULL, + news_raw text NOT NULL, + news_improve text NOT NULL, + news_title text NOT NULL, + PRIMARY KEY (nid) +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_general_ci; + +CREATE TABLE IF NOT EXISTS xwlb_daily_ext ( + extid int(11) NOT NULL AUTO_INCREMENT, + news_date date NOT NULL, + sub_id tinyint(4) NOT NULL, + news_title varchar(256) NOT NULL, + news_content text NOT NULL, + PRIMARY KEY (extid) +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_general_ci; +``` + +**`xwlb_daily_ext` —— 最终产物:单条新闻** + +| 字段 | 类型 | 说明 | +|---|---|---| +| `extid` | int, PK, auto_increment | 自增主键 | +| `news_date` | date | 日期(`YYYY-MM-DD`) | +| `sub_id` | tinyint | 当天第几条新闻,从 1 开始 | +| `news_title` | varchar(256) | 新闻标题(模型生成) | +| `news_content` | text | 新闻正文 | + +### 2.5 三个模型的职责 + +| 环节 | 模型配置项 | 走哪个接入点(`routes.*`) | 默认模型 | 输入 → 输出 | +|---|---|---|---|---| +| 语音识别 | `models.asr_model` | `asr` → `dashscope` | `paraformer-realtime-v2` | 单片 wav → 该片文本 | +| 文本校对 | `models.correct_model` | `correct` → `dashscope` | `qwen3.7-max` | 单片 ASR 文本 → 校对后文本(同一分片粒度) | +| 新闻切分 | `models.split_model` | `split` → `deepseek` | `deepseek-chat` | 当天全文 → JSON 数组(标题 + 正文) | + +模型名与接入点是**分开配置**的:换供应商时改 `routes` + `endpoints`,再同步改模型名,代码不动(见 [3.4 更换 LLM 供应商](#34-更换-llm-供应商))。 + +**为什么要两道 LLM**:ASR 只保证"字"对上,标点、断句、同音错字仍需修;而切分属于理解任务——需要判断新闻边界、给每条起标题,所以交给擅长长文本理解的 DeepSeek。两道 LLM 的分工边界清晰,任一道失败都不会污染另一道的结果。 + +### 2.6 关键设计取舍 + +| 取舍 | 做法 | 原因 | +|---|---|---| +| 校对会改错事实 | 数值事实守卫:数值签名不一致就整片回退原文 | 宁保留可信的 ASR 原文,也不接受"通顺但错误"的文本 | +| 重跑会重复写库 | `xwlb_daily` 先删后插;`xwlb_daily_ext` 按日期替换 | 无需唯一索引即可幂等,重跑不产生脏数据 | +| 分片识别失败 | 记 ERROR、**不写空行**、其余分片照常 | 空行曾导致下游拼出残缺文本(历史库中有 101 行空白) | +| 切分返回非 JSON | 剥代码围栏 + `raw_decode` 兜底 + 强化提示词重试一次 | 历史上有 22 天因此得到 0 条精编 | +| 中间产物占用磁盘 | 当天全部成功才清理,失败保留 | 成功即无价值,可释放磁盘;失败则保留供重跑 | +| 顺序执行、不并发 | 逐片串行 | 单机负载可控、日志与失败定位简单;单日 25~35 分钟已满足离线需求 | + +--- + +## 三、使用说明 + +### 3.1 环境要求 + +- Linux(本机 Raspberry Pi / Debian) +- Python 3.11 ~ 3.13(本项目在 3.13.5 上验证) +- 系统依赖:`ffmpeg`;`autossh`(远程数据库时) +- 一个可用的 MySQL/MariaDB 库(两张表需已存在,见 [2.4](#24-数据库结构)) +- 阿里云 DashScope API Key(语音识别 + 校对)、DeepSeek API Key(切分) + +### 3.2 安装 + +```bash +cd /home/pi/project/xwlb + +# 1) 端口隧道:本地 13306 → 远端 3306(数据库不在本机时需要,通了可跳过) +bash autossh.sh + +# 2) 虚拟环境(本机 Python 为 externally-managed,必须用 venv) +python3 -m venv .venv +.venv/bin/pip install -r requirements.txt + +# 3) 配置 +cp .env.example .env # 填 3 个敏感项 +vim config.yml # 核对数据库地址、目录、模型 +``` + +> Python 3.13 已移除标准库 `audioop`,`pydub` 需要 `requirements.txt` 中的 `audioop-lts`,已自动按版本条件安装。 + +### 3.3 配置 + +**(1)`.env` —— 只放敏感项**(已被 `.gitignore` 排除) + +```ini +MYSQL_PASSWORD=... +DASHSCOPE_API_KEY=sk-... +DEEPSEEK_API_KEY=sk-... +``` + +**(2)`config.yml` —— 其余全部配置** + +```yaml +mysql: { host: localhost, port: 13306, user: myquant, database: myquant } +paths: { video_dir: xwlb_video, audio_dir: audio_processing } +models: + asr_model: paraformer-realtime-v2 # 语音识别 + correct_model: qwen3.7-max # ASR 文本校对 + split_model: deepseek-chat # 新闻切分 + 标题 + +# 接入点(base url):换供应商 / 换区域 / 走代理只改这一段 +endpoints: + dashscope: + kind: dashscope # dashscope=用 SDK,openai=OpenAI 兼容 + http_base_url: https://dashscope.aliyuncs.com/api/v1 # 文本生成走 HTTP + websocket_base_url: wss://dashscope.aliyuncs.com/api-ws/v1/inference # 实时 ASR 走 WS + api_key_env: DASHSCOPE_API_KEY # 密钥在 .env 里的变量名 + deepseek: + kind: openai + base_url: https://api.deepseek.com/v1 + chat_completions_path: /chat/completions + api_key_env: DEEPSEEK_API_KEY + +# 每个环节走哪个接入点(值是 endpoints 下的名字) +routes: { asr: dashscope, correct: dashscope, split: deepseek } + +llm_correct: + enabled: 1 # 1=开启校对(默认);0=直接用 ASR 原文 +cleanup: + after_daily_run: 1 # 当天成功结束后自动清理中间产物 + only_on_success: 1 # 仅成功才清理;0=无论成败都清理 + remove_video: 1 # 删除 mp3 / mp4 + remove_audio: 1 # 删除 wav +``` + +**优先级**:`进程环境变量 > config.yml > 代码内置默认值`。 +所以临时试验可以不改文件: + +```bash +DASHSCOPE_LLM_MODEL=qwen-max .venv/bin/python audioRead.py 20260904 +LLM_CORRECT_ENABLED=0 .venv/bin/python getVideo5.py 20260904 20260904 +XWLB_ROUTE_CORRECT=deepseek .venv/bin/python newsProcess.py 2026-09-04 # 临时换校对供应商 +``` + +启动时会打印生效配置(不含敏感值),含每个环节实际用的接入点: + +``` +配置来源: /home/pi/project/xwlb/config.yml | MySQL myquant@localhost:13306/myquant | +模型 asr=paraformer-realtime-v2 correct=qwen3.7-max split=deepseek-chat | 校对=开启 | 完成后清理=开启 +接入点 asr -> dashscope kind=dashscope wss://dashscope.aliyuncs.com/api-ws/v1/inference(密钥变量 DASHSCOPE_API_KEY) +接入点 correct -> dashscope kind=dashscope https://dashscope.aliyuncs.com/api/v1(密钥变量 DASHSCOPE_API_KEY) +接入点 split -> deepseek kind=openai https://api.deepseek.com/v1/chat/completions(密钥变量 DEEPSEEK_API_KEY) +``` + +### 3.4 更换 LLM 供应商 + +供应商不写死在代码里:**地址在 `endpoints`,走哪条线在 `routes`,密钥变量名在 `api_key_env`**。 + +| 想做什么 | 改哪里 | +|---|---| +| 换区域(如 DashScope 国际站) | `endpoints.dashscope` 的两个 url | +| 走公司网关 / 代理 / 自建推理 | 改对应接入点的 `base_url`(OpenAI 兼容) | +| 换校对供应商(如月之暗面、智谱、Kimi) | ① `routes.correct` 指向新接入点 ② 在 `endpoints` 加一段 ③ 改 `models.correct_model` ④ `.env` 加对应 Key | +| 换切分供应商 | 同上,改 `routes.split` + `models.split_model` | +| 换 ASR 供应商 | ⚠️ 实时语音识别只有 `kind: dashscope` 一种适配器;换**厂商**需要新增适配器(换区域/网关仍可,改 url 即可) | + +**示例:把校对从通义千问换成月之暗面** + +```yaml +routes: { asr: dashscope, correct: moonshot, split: deepseek } +endpoints: + moonshot: + kind: openai + base_url: https://api.moonshot.cn/v1 + chat_completions_path: /chat/completions + api_key_env: MOONSHOT_API_KEY +models: + correct_model: kimi-k2-0905-preview +``` + +```ini +# .env +MOONSHOT_API_KEY=sk-... +``` + +要点: + +- `kind` 只有两种:`dashscope`(用 dashscope SDK)和 `openai`(OpenAI 兼容的 `/chat/completions`,覆盖 DeepSeek / Kimi / 智谱 / vLLM / 各类网关)。 +- 请求地址 = `base_url` + `chat_completions_path`,所以不带版本号的网关地址(如 `https://my-gateway/llm`)也能直接填。 +- 缺 `base_url`、`routes` 指向不存在的接入点、`kind` 拼错,都会在启动日志与调用时报出**可读错误**,不会静默走错地址。 +- **换供应商不影响数值事实守卫**:无论哪家的校对结果,改动数字/年份/届次一样会被拦下并回退 ASR 原文(已用 DeepSeek 作为校对供应商实测)。 +- 运行时必需哪些 Key 是**按路由推导**的:只有被 `routes` 用到的接入点才要求配 Key,未被使用的不会报缺失。 + +#### 3.4.1 推理模型(thinking)注意 + +现在的模型多是"推理模型":回答前先生成一大段思维链,**推理 token 按输出计费**。 +实测(详见 [`docs/REPORT_raw_vs_improve.md`](./docs/REPORT_raw_vs_improve.md)): + +| 环节 | 关掉思考 | 开着思考 | 结论 | +|---|---|---|---| +| 校对 | 312 token / 4~6 秒,输出几乎一致 | 5,932 token | **关**(省约 13 倍,机械任务不需要推理) | +| 切分 | 4,656 token,正文覆盖率 **95.34%**、20 条 | 19,077 token,覆盖率 **99.67%**、27 条 | **开**(关掉会丢 4.7% 正文、少切 7 条) | + +关思考的参数名**因供应商而异**,写错会被**静默忽略**(不报错但也不生效),所以配在接入点上: + +```yaml +endpoints: + qwen: { extra_body: {enable_thinking: false} } # 通义/百炼兼容模式 + deepseek: { extra_body: {thinking: {type: disabled}} } # DeepSeek 推理模型 +``` + +也可在 `llm_correct` / `llm_split` 段写 `extra_body` 覆盖(优先级更高)。 +使用**推理模型**时 `llm_split.max_tokens` 要给足(现为 60000):实测推理 token 在 +14k~24k 之间波动,上限 20000 会截断,甚至**整个回复为空**。 + +### 3.5 运行 + +| 场景 | 命令 | 说明 | +|---|---|---| +| 日常(当天) | `.venv/bin/python main.py` | 抓当天 → 下载 → 转写 → 切分 → 清理 | +| 指定某天 | `.venv/bin/python getVideo5.py 20260904 20260904` | 日期格式 `YYYYMMDD` | +| 指定区间 | `.venv/bin/python getVideo5.py 20260901 20260907` | 逐日全流程 | +| 补缺失日期 | `.venv/bin/python main_videos.py` | 反查库中缺失日期再补跑(**区间写在文件内**,需按需修改) | +| 重跑某天 | `.venv/bin/python newsRedo.py 2026-09-04` | 已有精编 → 跳过;有原文 → 只重做切分;都没有 → 全流程 | +| 强制重切 | `.venv/bin/python newsRedo.py 2026-09-04 --force` | 忽略已有记录,重新切分并**替换**该日 `xwlb_daily_ext` | +| 只重做切分 | `.venv/bin/python newsProcess.py 2026-09-04 [--force]` | 不碰音频,直接用库中文本重切 | +| 只重做转写 | `.venv/bin/python audioRead.py 20260904` | 用已有 `xwlb_video/20260904.mp3` 重跑分割+识别+落库 | +| 手动清理 | `.venv/bin/python cleanup.py [--dry-run]` | 清空两个中间产物目录,`--dry-run` 只列不删 | + +> **注意**:默认 `cleanup.remove_video: 1`,当天成功后 mp3 会被删掉。因此**要重做某天的转写**(`audioRead.py`)时,需要先重新下载该天的 mp3——最简单是把 `cleanup.remove_video` 临时设为 `0`,或直接跑 `getVideo5.py` 重下再重跑。 + +### 3.6 自检与测试 + +五个测试套件共 **97 项断言**,全部离线运行(不消耗 API 额度、不触碰真实文件): + +```bash +.venv/bin/python tests/test_config.py # 37 项:配置加载、优先级、接入点/路由、敏感项隔离 +.venv/bin/python tests/test_cleanup.py # 10 项:清理开关五种组合(临时目录) +.venv/bin/python tests/test_fidelity_guard.py # 14 项:数值事实守卫(含真实篡改样例) +.venv/bin/python tests/test_news_parse.py # 16 项:LLM 返回解析(含真实坏返回) +.venv/bin/python tests/test_tunnel.py # 20 项:隧道自愈各分支(临时端口,不碰真实隧道) +``` + +其中隧道测试**不会碰真实的 13306 与 autossh.sh**:它用临时 config.yml + 临时端口的假脚本, +验证"端口已通时绝不执行脚本""systemd 托管时只等待不抢端口"等关键约定。 + +连通性自检: + +```bash +# 打印生效配置并校验 Key 是否齐全 +.venv/bin/python -c "import config; config.log_summary()" + +# 数据库连通 +.venv/bin/python -c "from mysqlHandle import MySQLDB; d=MySQLDB(); print(d.query_one('xwlb_daily','COUNT(*) c')); d.close()" +``` + +### 3.7 定时任务(systemd) + +已配置为 systemd 定时器:**每天 21:00 触发全链路,失败后 21:30、22:00 各自动重试一次**。 + +```bash +bash scripts/install_systemd.sh # 安装/更新单元(幂等,可反复执行) +systemctl list-timers xwlb-daily.timer # 看下次触发时间 +journalctl -u xwlb-daily -f # 实时看日志(也在 main.log) +sudo systemctl start xwlb-daily.service # 立即试跑一次 +sudo systemctl disable --now xwlb-daily.timer # 临时停用 +``` + +涉及三个单元(源码在 `systemd/`): + +| 单元 | 作用 | +|---|---| +| `xwlb-daily.timer` | 每天 21:00 触发;`Persistent=true`,树莓派关机错过时刻会在开机后补跑 | +| `xwlb-daily.service` | `oneshot`,调用 `scripts/run_daily.sh`;`TimeoutStartSec=10800` 以容忍内部等待重试 | +| `xwlb-tunnel.service` | autossh 隧道托管(开机自启、断线自动重连),`xwlb-daily` 依赖它 | + +重试不是简单地"再跑一遍": + +- **只缺切分**(识别已全部完成)→ 只重跑切分环节,省掉全部 ASR 调用(约十几分钟与相应费用) +- **识别没跑完** → 重跑全链路 +- 判断依据是识别**全部**完成后写的标记 `state/.asr_complete_<日期>`; + 只看"库里有没有分片"是不够的——识别中途卡死也会留下部分分片, + 那样会被误判成"只缺切分",**把半天内容当成完整一天入库**(`scripts/day_status.py` 专门防这个) +- 识别不完整的那一天**不会执行切分**(不发布半天的 ext),宁可让重试补全, + 也不产生"看起来完整"的一天;分片原文仍在 `xwlb_daily` 里,不会丢 +- 标记放在 `state/` 且不随清理删除:它是长期凭证,删了就分不清 + "已完成"与"识别只跑了一半但切分照样写了 ext" + (`cleanup` 只清 `xwlb_video/*.mp3|*.mp4` 与 `audio_processing/*.wav`) +- 单次尝试有 40 分钟上限(`XWLB_ATTEMPT_TIMEOUT`),卡死的 ASR 不会拖垮整晚 +- 有文件锁防重入,手动执行与定时触发叠在一起也不会重复烧 ASR + +退出码:`0` 成功;`1` 重试用尽仍失败(会出现在 `systemctl --failed`);`2` 已有实例在运行。 + +> 用 cron 也行,但那样拿不到"超时上限、重试、防重入、开机补跑"这些保障: +> `40 20 * * * cd /home/pi/project/xwlb && .venv/bin/python main.py >> main.log 2>&1` + +### 3.8 数据库隧道自愈 + +数据库在内网,靠 `autossh.sh` 把本地 `13306` 转发到远端 `3306`。隧道断掉时, +**任何入口都会自己把它拉起来**——不只是定时任务,手动跑 `main.py` / `getVideo5.py` / +`newsProcess.py` / `day_status.py` 都一样(实现挂在数据库连接层 `MySQLDB.connect()`)。 + +行为(`tunnel.py`,所有入口共用一套逻辑): + +| 情况 | 动作 | +|---|---| +| 端口可连 | **什么都不做**(正常路径零开销,不会多起进程) | +| 端口不通,且 systemd 隧道单元 active | **只等待**它自动重连——不另起 autossh 抢同一个端口(那会让 systemd 单元因端口被占反复重启失败) | +| 端口不通,systemd 单元未在运行 | 执行 `config.yml` 里 `mysql.tunnel.script`(默认 `autossh.sh`),最多等 30 秒 | +| 仍不通 | 如实报错,不再假装成功 | + +```bash +python tunnel.py # 手动探测 + 按需修复,退出码 0=通 / 1=仍不通 +python tunnel.py --dry-run # 只看会做什么,不执行 +``` + +配置: + +```yaml +mysql: + tunnel: + enabled: 1 + script: autossh.sh # 相对项目根 + systemd_unit: xwlb-tunnel.service # 该单元 active 时只等待,不抢端口 + wait_seconds: 30 + connect_timeout: 2 +``` + +(直连远端数据库时,`host` 不是本机地址 → 判定为"不适用隧道",不会去跑 autossh.sh。) + +### 3.9 故障排查 + +| 现象 | 原因 / 处理 | +|---|---| +| `MySQL 连接失败 ...` | 隧道不通。**现在会自动恢复**:连接失败时探测端口,不通则执行 `autossh.sh` 并等待(见 3.9 隧道自愈)。仍失败再手动 `bash autossh.sh`,并确认 `config.yml` 的 `mysql.port` 是隧道端口 `13306` | +| `缺少敏感配置 ...` | `.env` 未填 `MYSQL_PASSWORD` / `DASHSCOPE_API_KEY` / `DEEPSEEK_API_KEY` | +| `externally-managed-environment` | 不要用系统 `pip`,用 `.venv/bin/pip` | +| `No module named 'audioop'` | Python 3.13 需装 `audioop-lts`(已在 requirements 中) | +| 某天识别成功但精编为 0 条 | 切分返回非 JSON;现会自动用强化提示词重试一次,仍失败则保留原文不写库,可 `newsProcess.py <日期> --force` 重试 | +| 校对结果与原文数字不一致 | 事实守卫已拦截并回退原文,日志会打印 `⚠️ 校对改动了数值事实,已回退 ASR 原文` | +| 想完全关掉 LLM 校对 | `config.yml` 设 `llm_correct.enabled: 0` | +| `接入点配置错误: ...` | `routes.*` 指向了不存在的接入点、`kind` 拼错、或缺 `base_url`;按日志提示补 `endpoints` 字段 | +| `ASR 接入点 ... kind=openai:实时语音识别仅支持 kind=dashscope` | ASR 换**厂商**需新增适配器;换区域 / 网关请保留 `kind: dashscope` 只改 url | +| 换供应商后报 401 / 403 | `.env` 里没有该接入点 `api_key_env` 指定的那个变量(不是把新 Key 塞进旧变量名) | +| 换供应商后报模型不存在 | `models.correct_model` / `models.split_model` 还是旧供应商的模型名,需同步修改 | +| `getVideo5` 报某天 404 | 该天页面尚未上线(当天节目过期或未发布),日志会明确跳过 | +| 磁盘被中间产物占满 | `python cleanup.py --dry-run` 查看,再 `python cleanup.py` | +| 某天识别到一半卡死(无新日志、无 CPU) | ASR 的 `Recognition.call` 没有超时参数,长连接可能悬挂;定时任务有 40 分钟上限并在 21:30/22:00 重试。手动跑请自行 `timeout` | +| 定时任务没跑 | `systemctl list-timers xwlb-daily.timer`;`systemctl status xwlb-daily.service`;日志 `journalctl -u xwlb-daily -n 100` | +| 定时任务报数据库连接失败 | 隧道服务:`systemctl status xwlb-tunnel`;手动兜底 `bash autossh.sh` | + +--- + +## 四、相关文档 + +| 文档 | 内容 | +|---|---| +| [`docs/ARCHITECTURE.md`](./docs/ARCHITECTURE.md) | 完整技术说明:数据流、模块与函数索引、数据库结构、配置项全表、外部调用实测参数、时序图 | +| [`docs/REPORT_raw_vs_improve.md`](./docs/REPORT_raw_vs_improve.md) | 校对效果评估:7973 条真实数据统计、实测 token 开销、合并/关闭校对的对比与结论 | +| [`docs/BUGS.md`](./docs/BUGS.md) | 15 项缺陷清单(含严重度、位置、影响、证据、修复),以及本次修复的实测验证记录与后续待办 | + +**已知待办**(详见 `docs/BUGS.md`): + +- `main_videos.py` 的日期区间仍写死在文件内,尚未改为命令行参数; +- 历史数据治理(截至 2026-09-25):`xwlb_daily` 有 **100 行**空白分片、**202 天**缺 `xwlb_daily_ext`,可用上述分步入口补齐; +- 项目尚未 `git init`(`.gitignore` 已就绪);`main.log` 无轮转(历史 34MB 作为证据保留)。 \ No newline at end of file diff --git a/audioRead.py b/audioRead.py new file mode 100644 index 0000000..4dd78b9 --- /dev/null +++ b/audioRead.py @@ -0,0 +1,684 @@ +import datetime +import env # 加载 .env(敏感项) +import config # 加载 config.yml(模型、参数、路径、开关) +import os +import re +import time +import dashscope +import pydub +from pydub import AudioSegment +from pydub.silence import split_on_silence +from dashscope.audio.asr import Recognition +from dashscope import Generation +from http import HTTPStatus +from mysqlHandle import MySQLDB +from deepseek import openai_chat +import logging + +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + +# 确保 dashscope SDK 使用 config.yml 里配置的接入点地址 +# (SDK 在 import 时读环境变量,config 加载阶段已写入;这里再显式覆盖一次,不依赖 import 顺序) +config.apply_dashscope_endpoints(dashscope) +# 设置环境变量 +# os.environ["DASHSCOPE_API_KEY"] = "sk-your-dashscope-key" + +def convert_mp3_to_wav(mp3_path, output_wav_path): + """ + 将MP3文件转换为16kHz单声道WAV格式,这是Qwen3-ASR-Flash模型的推荐格式 + 参数: + mp3_path (str): MP3文件路径 + output_wav_path (str): 输出WAV文件路径 + 返回值: + str: 转换后的WAV文件路径 + """ + logger.info(f"开始转换MP3到WAV: {mp3_path}") + # 加载MP3文件 + audio = AudioSegment.from_file(mp3_path, format="mp3") + # 转换为 ASR 要求的采样率(config.yml: asr.sample_rate)、单声道 + audio = audio.set_frame_rate(config.get_int('asr.sample_rate', 16000)).set_channels(1) + # 导出为WAV格式 + audio.export(output_wav_path, format="wav") + logger.info(f"✓ MP3转换完成: {output_wav_path}") + #print(f"✓ MP3转换完成: {output_wav_path}") + return output_wav_path + +def split_audio_by_fixed_duration(audio_path, chunk_duration, output_folder): + """ + 将音频文件按固定时长分割成多个片段 + 参数: + audio_path (str): 音频文件路径 + chunk_duration (int): 分片时长(毫秒) + output_folder (str): 输出文件夹路径 + 返回值: + list: 分片文件路径列表 + """ + # 加载音频文件 + audio = AudioSegment.from_file(audio_path) + # 计算总时长(毫秒) + total_duration = len(audio) + # 分片数 + num_chunks = total_duration // chunk_duration + 1 + # 存储分片文件路径 + chunks = [] + + # 创建输出文件夹 + os.makedirs(output_folder, exist_ok=True) + + logger.info(f"开始音频分割,总时长: {total_duration/1000:.1f}秒,将分割为{num_chunks}个片段") + + for i in range(num_chunks): + # 计算当前分片的起始和结束时间 + start_time = i * chunk_duration + end_time = (i + 1) * chunk_duration + # 提取分片音频 + chunk = audio[start_time:end_time] + # 生成文件名 + chunk_name = f"chunk_{i}.wav" + chunk_path = os.path.join(output_folder, chunk_name) + # 导出分片音频 + chunk.export(chunk_path, format="wav") + chunks.append(chunk_path) + + # 打印处理进度 + progress = (i + 1) / num_chunks * 100 + logger.info(f"✓ 已完成分片 {i+1}/{num_chunks} ({progress:.1f}%)") + + logger.info(f"✓ 音频分割完成,共生成{len(chunks)}个分片文件") + return chunks + +def split_audio_by_smart_silence(audio_path, min_silence_len, silence_thresh, output_folder): + """ + 将音频文件按智能静音检测方式分割成多个片段,每段不超过3分钟 + 参数: + audio_path (str): 音频文件路径 + min_silence_len (int): 最小静音长度(毫秒) + silence_thresh (int): 静音阈值(dBFS) + output_folder (str): 输出文件夹路径 + 返回值: + list: 分片文件路径列表 + """ + # 加载音频文件 + audio = AudioSegment.from_file(audio_path, format="wav") + # 按静音分割 + segments = split_on_silence( + audio, + # 静音超过该长度则分割(config.yml: audio_split.min_silence_ms) + min_silence_len=min_silence_len, + # 静音阈值(config.yml: audio_split.silence_thresh_db) + silence_thresh=silence_thresh, + # 保留静音部分(config.yml: audio_split.keep_silence_ms) + keep_silence=config.get_int('audio_split.keep_silence_ms', 400) + ) + + logger.info(f"✓ 静音分割完成,共{len(segments)}个初始片段") + + # 合并过短的片段 + merged_segments = [] + current_segment = None + for segment in segments: + if current_segment is None: + current_segment = segment + else: + # 合并当前片段和新片段 + temp_segment = current_segment + segment + # 如果合并后的片段超过上限则单独保存(config.yml: audio_split.max_chunk_ms) + if len(temp_segment) > config.get_int('audio_split.max_chunk_ms', 180000): + merged_segments.append(current_segment) + current_segment = segment + else: + current_segment = temp_segment + # 添加最后一个片段 + if current_segment is not None: + merged_segments.append(current_segment) + + logger.info(f"✓ 片段合并完成,共{len(merged_segments)}个最终片段") + + # 存储分片文件路径 + chunks = [] + + # 创建输出文件夹 + os.makedirs(output_folder, exist_ok=True) + + logger.info(f"开始导出音频片段到: {output_folder}") + + for i, segment in enumerate(merged_segments): + # 生成文件名 + chunk_name = f"chunk_{i}.wav" + chunk_path = os.path.join(output_folder, chunk_name) + # 导出分片音频 + segment.export(chunk_path, format="wav") + chunks.append(chunk_path) + + # 打印处理进度 + progress = (i + 1) / len(merged_segments) * 100 + logger.info(f"✓ 已完成分片 {i+1}/{len(merged_segments)} ({progress:.1f}%)") + + logger.info(f"✓ 智能静音分割完成,共生成{len(chunks)}个分片文件") + return chunks + + +def transcribe_audio(audio_path): + """ + 仅执行语音识别(ASR),**不再在同一函数里做 LLM 校对** + + 原实现把「识别」与「qwen 校对」串在同一个 try 中,校对超时会被当成识别失败, + 导致已识别成功的文本被整体丢弃(生产库中已累积 101 行空记录,见 docs/BUGS.md B1)。 + + 参数: + audio_path (str): 音频文件路径(必须是16kHz单声道WAV) + 返回值: + str: 识别文本;失败返回 ''(由调用方决定是否跳过入库) + """ + try: + # 确保音频文件存在 + if not os.path.exists(audio_path): + logger.error(f"音频文件不存在: {audio_path}") + return '' + dashscope.api_key = config.api_key('asr') # 密钥变量名由 config.yml 的 endpoints.*.api_key_env 决定 + if not dashscope.api_key: + logger.error("接入点 %s 的密钥未配置(.env 中的 %s),无法识别音频", + config.route('asr'), config.api_key_env('asr')) + return '' + asr_endpoint, asr_cfg = config.endpoint_for('asr') + if str(asr_cfg.get('kind', '')).lower() != 'dashscope': + logger.error("ASR 接入点 '%s' 的 kind=%s:实时语音识别仅支持 kind=dashscope 的接入点," + "更换 ASR 供应商需要新增适配器", asr_endpoint, asr_cfg.get('kind')) + return '' + # 创建识别对象 + recognition = Recognition( + model=config.model('asr_model'), # config.yml: models.asr_model + format='wav', + sample_rate=config.get_int('asr.sample_rate', 16000), + language_hints=config.get_list('asr.language_hints', ['zh', 'en']), # 中文和英文 + callback=None + ) + + # 调用识别 + logger.info(f"开始识别音频: {audio_path}") + result = recognition.call(audio_path) + if result.status_code == HTTPStatus.OK: + # 提取识别结果 + sentence = result.get_sentence() + text = merge_transcripts(sentence) + logger.info(f"✓ {audio_path} 识别成功,文本长度: {len(text)}") + return text + logger.error(f"❌ ASR 任务失败: {getattr(result, 'message', result)}") + return '' + except Exception as e: + logger.error(f"ASR 识别异常: {audio_path}: {e}") + return '' + + +def _extract_llm_text(response): + """ + 兼容 dashscope 的两种返回形态,返回模型生成的纯文本(取不到则返回 '') + + 实测(dashscope 1.27.7 / qwen-plus / 约 700 字输入):status=200 时 + `output.text` 为 None,真正的内容在 `output.choices[0].message.content`; + 而短输入时 `output.text` 有值。旧代码只读 output.text,于是每次都拿到 None + (生产日志中「文本修正返回 None」共出现于每个分片,见 docs/BUGS.md B2)。 + 另外 401 等失败场景 `output` 为 None,需要一并防御。 + """ + output = getattr(response, 'output', None) + if output is None: + return '' + text = getattr(output, 'text', None) + if text and text.strip(): + return text.strip() + choices = getattr(output, 'choices', None) or [] + if choices: + choice = choices[0] + message = choice.get('message') if isinstance(choice, dict) else getattr(choice, 'message', None) + if message: + content = message.get('content') if isinstance(message, dict) else getattr(message, 'content', None) + if content and content.strip(): + return content.strip() + return '' + +def merge_transcripts(transcripts): + """ + 将多段识别文本合并成完整句子(保留原始段落逻辑,用空格连接) + 参数: + transcripts (list): 识别结果列表,每个元素为字典{'text': '识别文本'} + 返回: + str: 合并后的完整文本 + """ + # 输入参数检查 + if not transcripts: + return "" + + # 确保transcripts是可迭代对象 + if not hasattr(transcripts, '__iter__'): + return "" + + try: + # 提取所有有效的text字段 + texts = [] + for t in transcripts: + try: + # 检查是否为字典类型且包含text字段 + if isinstance(t, dict) and 'text' in t and t['text']: + text = t['text'] + # 确保text是字符串类型 + if isinstance(text, str) and text.strip(): + texts.append(text.strip()) + except (KeyError, TypeError, AttributeError): + # 忽略单个元素的处理错误,继续处理其他元素 + continue + + # 用空格连接所有段落(根据实际需求可调整连接符) + return " ".join(texts) if texts else "" + + except Exception as e: + logger.error(f"合并转录文本时发生错误: {e}") + return "" +def text_correction(text, max_retries=None, timeout=None): + """ + 使用通义千问模型修正文本中的错误和标点符号 + + 与原实现的区别: + - 显式传入 timeout(原实现依赖 SDK 默认 300s,生产日志中 57 次读超时全发生在这里); + - 按结果取文本时兼容 `output.text` 与 `output.choices[0].message.content`(B2 根因); + - 失败时**抛异常**,由 analyze_and_correct_text 决定回退原文,不再静默返回 None; + - max_tokens 由 30000 降为可配置的 8000(校对输出不会超过输入量级)。 + + 参数: + text (str): 需要修正的文本 + max_retries (int): 重试次数,默认取环境变量 LLM_CORRECT_RETRIES=2 + timeout (int): 单次调用超时秒数,默认取环境变量 LLM_CORRECT_TIMEOUT=90 + 返回值: + str: 修正后的文本 + 异常: + RuntimeError: 重试耗尽仍失败 + """ + max_retries = max_retries if max_retries is not None else config.get_int('llm_correct.max_retries', 2) + timeout = timeout if timeout is not None else config.get_int('llm_correct.timeout', 90) + max_tokens = config.get_int('llm_correct.max_tokens', 8000) + model = config.model('correct_model') # config.yml: models.correct_model + endpoint_name, endpoint_cfg = config.endpoint_for('correct') + kind = str(endpoint_cfg.get('kind', 'dashscope')).lower() + + logger.info("开始文本修正...") + + # 构建修正提示词 + correction_prompt = """请仔细检查以下文本,修正其中的错误: +1. 错别字和语法错误 +2. 标点符号使用错误 +3. 语句不通顺的地方 +4. 逻辑不清晰的部分 +5. 严禁改动任何事实信息:数字、年份、日期、届次、数量、机构名、人名、地名、专有名词(如"十五五")必须与原文完全一致,即使你认为原文有误也不要修改。 + +请直接返回修正后的完整文本,不要添加任何解释说明。""" + + # 构建消息列表 + messages = [ + {"role": "system", "content": "你是一个专业的文本校对助手,擅长修正文本中的各种错误。"}, + {"role": "user", "content": correction_prompt}, + {"role": "user", "content": text} + ] + + last_error = "未知错误" + for attempt in range(1, max_retries + 1): + try: + logger.info(f"调用文本校对模型(接入点={endpoint_name} kind={kind} 第 {attempt}/{max_retries} 次," + f"model={model},timeout={timeout}s)...") + if kind == 'openai': + # 非 DashScope 供应商:走 OpenAI 兼容的 /chat/completions + corrected = openai_chat(text, correction_prompt, role='correct', + model=model, max_tokens=max_tokens, timeout=timeout) + if corrected: + corrected = corrected.strip() + logger.info(f"✓ 文本修正完成,长度 {len(text)} -> {len(corrected)}") + return corrected + last_error = "返回内容为空" + logger.warning(f"⚠️ 文本修正返回空内容: {last_error}") + else: + # DashScope:用 SDK(dashscope.api_key 已按接入点设置) + dashscope.api_key = config.api_key('correct') + response = Generation.call( + model=model, + messages=messages, + max_tokens=max_tokens, + temperature=config.get('llm_correct.temperature', 0.1), # 较低温度以提高确定性 + top_p=config.get('llm_correct.top_p', 0.5), + timeout=timeout, + ) + if response.status_code != HTTPStatus.OK: + last_error = f"status={response.status_code} message={getattr(response, 'message', '')}" + logger.warning(f"❌ 文本修正API调用失败: {last_error}") + else: + corrected = _extract_llm_text(response) + if corrected: + logger.info(f"✓ 文本修正完成,长度 {len(text)} -> {len(corrected)}") + return corrected + last_error = "响应中无可用文本(output/choices 均为空)" + logger.warning(f"⚠️ 文本修正返回空内容: {last_error}") + except Exception as e: + last_error = f"{type(e).__name__}: {e}" + logger.warning(f"⚠️ 文本修正异常(第 {attempt}/{max_retries} 次): {last_error}") + + if attempt < max_retries: + time.sleep(min(2 ** attempt, 8)) + + raise RuntimeError(f"文本修正失败(已重试 {max_retries} 次): {last_error}") + +_CN_NUM_CHARS = '零〇一二两三四五六七八九十百千万亿' +_NUM_TOKEN_RE = re.compile(r'\d+(?:\.\d+)?|[零〇一二两三四五六七八九十百千万亿]+') +_CN_DIGITS = {'零': 0, '〇': 0, '一': 1, '二': 2, '两': 2, '三': 3, '四': 4, + '五': 5, '六': 6, '七': 7, '八': 8, '九': 9} +_CN_UNITS = {'十': 10, '百': 100, '千': 1000, '万': 10000, '亿': 100000000} + + +def _split_cn_numeral_run(run): + """把中文数字串按「连续数字位」切成若干数词,处理"十五五/十四五"这类缩写 + + '十五五' -> ['十五', '五'];'十四五' -> ['十四', '五'];'一百二十三' -> ['一百二十三'] + """ + segments, buf, prev_digit = [], '', False + for ch in run: + is_digit = ch in _CN_DIGITS + if is_digit and prev_digit and buf: + segments.append(buf) + buf = '' + buf += ch + prev_digit = is_digit + if buf: + segments.append(buf) + return segments + + +def _cn_to_number(text): + """单个中文数词转数值(十/百/千/万/亿),无法确定时返回 None""" + total = section = number = 0 + for ch in text: + if ch in _CN_DIGITS: + number = _CN_DIGITS[ch] + elif ch in _CN_UNITS: + unit = _CN_UNITS[ch] + if unit >= 10000: + section = (section + number) * unit + total += section + section = 0 + else: + section += (number or 1) * unit + number = 0 + else: + return None + return total + section + number + + +def _number_signature(text): + """ + 提取文本的数值事实签名:返回 (计数字典, 数字拼接串) + + 阿拉伯数字与中文数字统一为数值,忽略书写形式差异: + 例:`7月23号` 与 `七月二十三` 都得到 {7:1, 23:1} / "723",不算改动; + 而 `2027年`→`2024年`、`十一届`→`九届`、`十五五`→`十四五` 会被检出。 + """ + counter = {} + concat = [] + # "百分之X" 与 "X%" 等价,先去掉"百分之"避免把"百"当成数值 100 + text = (text or '').replace('百分之', '') + for token in _NUM_TOKEN_RE.findall(text): + if token[0].isdigit(): + keys = [token.lstrip('0') or '0'] + else: + keys = [] + for segment in _split_cn_numeral_run(token): + value = _cn_to_number(segment) + keys.append(str(value) if value is not None else segment) + for key in keys: + counter[key] = counter.get(key, 0) + 1 + concat.append(key) + return counter, ''.join(concat) + + +def _number_drift(raw_text, corrected_text): + """返回 (原文独有, 修正后独有) 的数值清单;两者皆空表示数值事实未被改动""" + before, before_concat = _number_signature(raw_text) + after, after_concat = _number_signature(corrected_text) + # 计数一致,或仅切分/书写形式不同导致拼接串一致(如 2026 ↔ 二零二六),都视为未改动 + if before == after or before_concat == after_concat: + return [], [] + only_raw = sorted(t for t in before if before[t] > after.get(t, 0)) + only_new = sorted(t for t in after if after[t] > before.get(t, 0)) + return only_raw, only_new + + +def analyze_and_correct_text(text): + """ + 分析文本并自动修正错误;**任何失败都回退原文**,保证不丢已识别文本; + **数值事实被改动时同样回退原文**(LLM 曾把 2027 年改成 2024 年、 + 十五五改成十四五、第十一届改成第九届,见 docs/BUGS.md B2 补充说明)。 + + 可用 config.yml 的 llm_correct.enabled=0 完全关闭校对(默认开启)。 + + 参数: + text (str): 待分析和修正的文本 + 返回值: + str: 修正后的文本;关闭/失败/改动事实时返回原文(绝不会是 None 或空串) + """ + if not text or not text.strip(): + return text or '' + + if not config.get_bool('llm_correct.enabled', True): + logger.info("LLM 校对已关闭(config.yml: llm_correct.enabled=0),保留 ASR 原文") + return text + + logger.info("开始文本分析和修正流程...") + try: + corrected_text = text_correction(text) + if corrected_text and corrected_text.strip(): + only_raw, only_new = _number_drift(text, corrected_text) + if only_raw or only_new: + logger.warning(f"⚠️ 校对改动了数值事实,已回退 ASR 原文" + f"(原文独有={only_raw} 修正后独有={only_new})") + return text + logger.info(f"原始文本长度: {len(text)},修正后: {len(corrected_text)}(数值事实一致)") + return corrected_text + logger.warning("文本修正返回空内容,使用原始文本") + except Exception as e: + logger.warning(f"文本修正失败,保留 ASR 原文(不影响已识别内容): {e}") + + logger.info(f"保留原始文本长度: {len(text)}") + return text +def analyze_text(text, prompt): + """ + 使用通义千问模型分析文本 + 参数: + text (str): 待分析文本 + prompt (str): 分析提示词 + 返回值: + str: 分析结果 + """ + logger.info("开始文本分析...") + # 设置系统提示 + system_prompt = "你是一个专业的文本分析助手,擅长根据提示词对长文本进行深入分析。" + # 构建消息列表 + messages = [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": prompt}, + {"role": "user", "content": text} + ] + + logger.info("调用通义千问模型进行文本分析...") + # 调用DashScope文本生成接口 + response = Generation.call( + model=config.model('correct_model'), + messages=messages, + max_tokens=8190, # 控制生成文本的最大长度 + temperature=0.3, # 控制生成文本的确定性 + top_p=0.7 # 控制生成文本的多样性 + ) + + # 检查API调用是否成功 + if response.status_code != 200: + logger.error(f"❌ API调用失败: {response.message}") + raise Exception(f"API调用失败: {response.message}") + + logger.info("✓ 文本分析完成") + # 返回分析结果(兼容 output.text 与 choices[0].message.content 两种形态) + return _extract_llm_text(response) + +def _upsert_daily_chunk(db, date_str, sub_id, raw, improved, title=''): + """ + 按 (news_days, daily_sub_id) 覆盖写入 xwlb_daily + + 原实现无条件 INSERT,重跑会追加一整组 daily_sub_id=0..N 的记录 + (库中 2026-06-15 全部 11 个分片已重复 2 份,见 docs/BUGS.md B5)。 + 这里先删同键记录再插入,无需依赖唯一索引即可实现幂等。 + """ + db.execute( + "DELETE FROM xwlb_daily WHERE news_days = %s AND daily_sub_id = %s", + (date_str, sub_id) + ) + return db.insert_data("xwlb_daily", { + "news_days": date_str, + "daily_sub_id": sub_id, + "news_raw": raw, + "news_improve": improved, + "news_title": title, + }) + + +# ---- 识别完整性标记 ---- +# 全部识别成功后写一个标记文件;切分失败需要重试时,重试逻辑靠它区分 +# "识别已完成、只缺切分"(可只重跑切分)与"识别只跑了一半"(必须重跑全链路)。 +ASR_MARKER_PREFIX = '.asr_complete_' + + +def asr_marker_path(date_str): + """标记文件路径:项目下 state/ 目录 + + 不放在 audio_processing/ 里,因为清理时会把中间产物删空;这个标记是 + "该日识别已全部完成"的长期凭证,必须活过清理,否则无法与 + "识别只跑了一半但切分照样写了 ext" 区分开。 + """ + d8 = str(date_str).replace('-', '') + return os.path.join(config.BASE_DIR, 'state', f'{ASR_MARKER_PREFIX}{d8}') + + +def write_asr_marker(date_str, chunks): + try: + path = asr_marker_path(date_str) + os.makedirs(os.path.dirname(path) or '.', exist_ok=True) + with open(path, 'w', encoding='utf-8') as f: + f.write(f'chunks={chunks}\nts={datetime.datetime.now().isoformat(timespec="seconds")}\n') + except OSError as e: + logger.warning(f"写入识别完成标记失败(不影响本次结果): {e}") + + +def remove_asr_marker(date_str): + try: + os.remove(asr_marker_path(date_str)) + except OSError: + pass + + +def process_long_audio(mp3_path, output_folder, date_str): + """ + 处理长音频文件:分割 → 逐片识别 → 校对 → 落库 + + 与原实现的区别: + - 识别失败/空结果**不再写入空行**,只记 ERROR 并跳过(B1); + - 每片按 (日期, 分片序号) 覆盖写入,重跑不产生重复(B5); + - 分片文件在 finally 中清理,失败分支不再残留; + - WAV 按日期命名并在结束时清理(原实现固定 input.wav,跨天互相覆盖且长期堆积)。 + + 参数: + mp3_path (str): MP3文件路径 + output_folder (str): 输出文件夹路径 + date_str (str): 日期,格式 YYYYMMDD + 返回值: + dict: {'chunks': 分片总数, 'ok': 成功入库数, 'failed': 失败分片数} + """ + logger.info("开始处理长音频...") + output_folder = os.fspath(output_folder) + os.makedirs(output_folder, exist_ok=True) # 原实现未建目录,目录缺失时导出会失败 + + # 转换MP3为WAV格式 + logger.info("步骤1/3: 转换MP3为WAV格式") + wav_path = convert_mp3_to_wav( + mp3_path, os.path.join(output_folder, f"input_{date_str}.wav") + ) + + # 分割音频(智能静音分割,参数见 config.yml: audio_split.*) + logger.info("步骤2/3: 智能静音分割音频") + chunks = split_audio_by_smart_silence( + wav_path, + config.get_int('audio_split.min_silence_ms', 700), + config.get('audio_split.silence_thresh_db', -40), + output_folder, + ) + + logger.info(f"步骤3/3: 开始识别音频分片,共{len(chunks)}个分片") + ok = 0 + failed = 0 + db = MySQLDB() # 使用默认参数连接数据库(整个流程复用一条连接) + try: + for i, chunk_path in enumerate(chunks): + try: + logger.info(f"识别进度: {i+1}/{len(chunks)} ({((i+1)/len(chunks)*100):.1f}%)") + raw = transcribe_audio(chunk_path) + if not raw or not raw.strip(): + failed += 1 + logger.error(f"❌ 分片 {i+1}/{len(chunks)} 识别为空,跳过入库: {chunk_path}") + continue + improved = analyze_and_correct_text(raw) + _upsert_daily_chunk(db, date_str, i, raw, improved, '') + ok += 1 + except Exception as e: + failed += 1 + logger.error(f"识别失败: {chunk_path}, 错误: {e}") + finally: + # 无论成功失败都清理分片(原实现仅在成功分支删除) + try: + os.remove(chunk_path) + except OSError: + pass + + # 重跑时切分点可能略有变化:若本次分片数少于库中已有记录,清理尾部残留行, + # 否则旧 sub_id 的正文会被 get_news_improve_by_date 一起拼进当天文本 + try: + stale = db.execute( + "DELETE FROM xwlb_daily WHERE news_days = %s AND daily_sub_id >= %s", + (date_str, len(chunks)) + ) + if stale: + logger.info(f"清理 {date_str} 多余的旧分片记录 {stale} 行(本次共 {len(chunks)} 片)") + except Exception as e: + logger.warning(f"清理旧分片记录失败(不影响本次结果): {e}") + finally: + db.close() + try: + os.remove(wav_path) + except OSError: + pass + + if failed: + logger.error(f"⚠️ {date_str} 长音频处理完成:成功 {ok}/{len(chunks)} 片,失败 {failed} 片(该日期内容不完整)") + remove_asr_marker(date_str) + else: + logger.info(f"✓ 长音频处理完成:成功 {ok}/{len(chunks)} 片") + # 全部识别成功才写"识别已完成"标记:重试逻辑据此判断 + # 「只缺切分」还是「识别本身没跑完」,避免把半天内容当完整数据切分入库 + write_asr_marker(date_str, len(chunks)) + return {'chunks': len(chunks), 'ok': ok, 'failed': failed} + +# 使用示例:python audioRead.py [YYYYMMDD] +if __name__ == "__main__": + import sys + from datetime import datetime + + date_str = sys.argv[1] if len(sys.argv) > 1 else datetime.now().strftime('%Y%m%d') + config.log_summary() + mp3_path = os.path.join(config.video_dir(), f"{date_str}.mp3") + output_folder = config.audio_dir() + + try: + result = process_long_audio(mp3_path, output_folder, date_str) + print(f"处理结果: {result}") + except Exception as e: + print(f"处理失败: {e}") \ No newline at end of file diff --git a/autossh.sh b/autossh.sh new file mode 100755 index 0000000..29ebf00 --- /dev/null +++ b/autossh.sh @@ -0,0 +1,3 @@ +#!/bin/bash +autossh -M 0 -fN -L 13306:localhost:3306 tunnel@doorcome.cn + diff --git a/cleanup.py b/cleanup.py new file mode 100644 index 0000000..7e2bce2 --- /dev/null +++ b/cleanup.py @@ -0,0 +1,119 @@ +"""中间产物清理 + +用法: + python cleanup.py # 按 config.yml 的 cleanup 配置清理 + python cleanup.py --dry-run # 只列出将删除的文件,不真删 + +清理范围(由 config.yml 的 cleanup 段控制): + paths.video_dir 下所有 *.mp3 / *.mp4 + paths.audio_dir 下所有 *.wav + +当日流程(getVideo5.process_videos)在一天的任务成功结束后会自动调用本模块; +失败时默认保留文件以便重跑(cleanup.only_on_success: 1)。 +""" +import argparse +import logging + +import config + +logger = logging.getLogger(__name__) + + +def _list_files(directory, patterns): + if not directory.is_dir(): + return [] + found = [] + for pattern in patterns: + found.extend(sorted(p for p in directory.glob(pattern) if p.is_file())) + return found + + +def collect_targets(): + """返回 [(路径, 目录角色)],受 cleanup.remove_video / remove_audio 控制""" + targets = [] + if config.get_bool('cleanup.remove_video', True): + targets += [(p, 'video') for p in _list_files(config.video_dir(), ('*.mp3', '*.mp4'))] + if config.get_bool('cleanup.remove_audio', True): + # 只删 wav。识别完成标记在 state/ 下,是长期凭证,**不能删** + # (删了就分不清"已完成"与"识别只跑了一半",见 scripts/day_status.py) + targets += [(p, 'audio') for p in _list_files(config.audio_dir(), ('*.wav',))] + return targets + + +def cleanup_intermediates(dry_run=False, force=False, reason=''): + """ + 删除中间产物;返回 {'removed': n, 'bytes': n, 'skipped': bool} + + 参数: + dry_run: 只统计不删除 + force: 忽略 cleanup.after_daily_run 开关(供手动调用) + reason: 日志说明(一般是日期) + """ + if not force and not config.get_bool('cleanup.after_daily_run', True): + logger.info("cleanup.after_daily_run 已关闭,跳过清理") + return {'removed': 0, 'bytes': 0, 'skipped': True} + + targets = collect_targets() + if not targets: + logger.info("没有需要清理的中间产物(%s)", reason or '手动清理') + return {'removed': 0, 'bytes': 0, 'skipped': False} + + removed = 0 + freed = 0 + for path, role in targets: + try: + size = path.stat().st_size + except OSError: + size = 0 + if dry_run: + logger.info("[dry-run] 将删除 %s(%.1f MB)", path, size / 1024 / 1024) + removed += 1 + freed += size + continue + try: + path.unlink() + removed += 1 + freed += size + logger.debug("已删除 %s", path) + except OSError as e: + logger.warning("删除失败 %s: %s", path, e) + + action = '将清理' if dry_run else '已清理' + logger.info("%s中间产物 %d 个文件(释放 %.1f MB)%s", + action, removed, freed / 1024 / 1024, f" | {reason}" if reason else '') + return {'removed': removed, 'bytes': freed, 'skipped': False} + + +def maybe_cleanup_after_run(date_str, day_ok): + """ + 当天流程结束后按 config.yml 的 cleanup 段决定是否清理 + + 参数: + date_str: 日期(仅用于日志) + day_ok: 当天是否全部成功(识别分片全成功 + 切分已就绪) + """ + if not config.get_bool('cleanup.after_daily_run', True): + logger.info("cleanup.after_daily_run 已关闭,跳过清理") + return {'removed': 0, 'bytes': 0, 'skipped': True} + + if not day_ok and config.get_bool('cleanup.only_on_success', True): + logger.warning("⚠️ %s 未全部成功,保留中间产物以便重跑" + "(cleanup.only_on_success=1;改为 0 则无论成败都清理)", date_str) + return {'removed': 0, 'bytes': 0, 'skipped': True} + + reason = f'{date_str} 任务完成' if day_ok else f'{date_str} 任务未全部成功(only_on_success=0)' + return cleanup_intermediates(reason=reason) + + +def main(): + parser = argparse.ArgumentParser(description='清理 xwlb 中间产物(mp3/mp4/wav)') + parser.add_argument('--dry-run', action='store_true', help='只列出将删除的文件') + args = parser.parse_args() + logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') + config.log_summary() + result = cleanup_intermediates(dry_run=args.dry_run, force=True, reason='手动执行') + return 0 if result['removed'] >= 0 else 1 + + +if __name__ == '__main__': + raise SystemExit(main()) \ No newline at end of file diff --git a/config.py b/config.py new file mode 100644 index 0000000..023105d --- /dev/null +++ b/config.py @@ -0,0 +1,403 @@ +"""非敏感配置加载器:config.yml + 环境变量覆盖 + 内置默认值 + +分工: +- 敏感配置(数据库口令、API Key)留在 .env,由 env.py 加载; +- 其余配置(数据库地址、目录、**各环节使用的模型**、切分参数、清理开关)都写在 config.yml; +- 可用 XWLB_CONFIG_FILE 指定其他配置文件。 + +优先级:**进程环境变量 > config.yml > 本文件内置默认值** +(便于临时试验,例如 `DEEPSEEK_MODEL=deepseek-reasoner python newsProcess.py`) +""" +import copy +import logging +import os +from pathlib import Path + +import env # noqa: F401 — 先加载 .env,使其中的变量可覆盖 config.yml + +logger = logging.getLogger(__name__) + +BASE_DIR = Path(__file__).resolve().parent + +# 环境变量 -> 配置项(点号路径)的覆盖映射 +ENV_OVERRIDES = { + 'MYSQL_HOST': 'mysql.host', + 'MYSQL_PORT': 'mysql.port', + 'MYSQL_USER': 'mysql.user', + 'MYSQL_DATABASE': 'mysql.database', + 'XWLB_VIDEO_DIR': 'paths.video_dir', + 'XWLB_AUDIO_DIR': 'paths.audio_dir', + 'DASHSCOPE_ASR_MODEL': 'models.asr_model', + 'DASHSCOPE_LLM_MODEL': 'models.correct_model', + 'DEEPSEEK_MODEL': 'models.split_model', + 'LLM_CORRECT_ENABLED': 'llm_correct.enabled', + 'LLM_CORRECT_RETRIES': 'llm_correct.max_retries', + 'LLM_CORRECT_TIMEOUT': 'llm_correct.timeout', + 'LLM_CORRECT_MAX_TOKENS': 'llm_correct.max_tokens', + # 接入点地址与路由(换供应商时也可用环境变量临时试) + 'DASHSCOPE_HTTP_BASE_URL': 'endpoints.dashscope.http_base_url', + 'DASHSCOPE_WEBSOCKET_BASE_URL': 'endpoints.dashscope.websocket_base_url', + 'DEEPSEEK_BASE_URL': 'endpoints.deepseek.base_url', + 'XWLB_ROUTE_ASR': 'routes.asr', + 'XWLB_ROUTE_CORRECT': 'routes.correct', + 'XWLB_ROUTE_SPLIT': 'routes.split', +} + +# 各环节默认走哪个接入点 +DEFAULT_ROUTES = {'asr': 'dashscope', 'correct': 'dashscope', 'split': 'deepseek'} + +# 内置默认值:config.yml 缺失时项目仍可运行 +DEFAULTS = { + 'mysql': {'host': 'localhost', 'port': 13306, 'user': 'myquant', 'database': 'myquant'}, + 'paths': {'video_dir': 'xwlb_video', 'audio_dir': 'audio_processing'}, + 'models': { + 'asr_model': 'paraformer-realtime-v2', + 'correct_model': 'qwen-plus', + 'split_model': 'deepseek-chat', + }, + 'endpoints': { + 'dashscope': { + 'kind': 'dashscope', + 'http_base_url': 'https://dashscope.aliyuncs.com/api/v1', + 'websocket_base_url': 'wss://dashscope.aliyuncs.com/api-ws/v1/inference', + 'api_key_env': 'DASHSCOPE_API_KEY', + }, + 'deepseek': { + 'kind': 'openai', + 'base_url': 'https://api.deepseek.com/v1', + 'chat_completions_path': '/chat/completions', + 'api_key_env': 'DEEPSEEK_API_KEY', + }, + }, + 'routes': dict(DEFAULT_ROUTES), + 'asr': {'sample_rate': 16000, 'language_hints': ['zh', 'en']}, + 'audio_split': { + 'min_silence_ms': 700, + 'silence_thresh_db': -40, + 'keep_silence_ms': 400, + 'max_chunk_ms': 180000, + }, + 'llm_correct': { + 'enabled': 1, 'max_retries': 2, 'timeout': 90, + 'max_tokens': 8000, 'temperature': 0.1, 'top_p': 0.5, + }, + 'llm_split': { + 'max_tokens': 20000, 'retry_max_tokens': 32000, + 'temperature': 0.5, 'timeout': 60, 'max_retries': 3, + }, + 'cleanup': { + 'after_daily_run': 1, 'only_on_success': 1, + 'remove_video': 1, 'remove_audio': 1, + }, +} + + +def config_path() -> Path: + """配置文件路径(XWLB_CONFIG_FILE 可覆盖)""" + override = os.getenv('XWLB_CONFIG_FILE') + if override: + return Path(override).expanduser() + return BASE_DIR / 'config.yml' + + +def _deep_merge(base: dict, override: dict) -> dict: + """递归合并,override 优先""" + result = copy.deepcopy(base) + for key, value in (override or {}).items(): + if isinstance(value, dict) and isinstance(result.get(key), dict): + result[key] = _deep_merge(result[key], value) + else: + result[key] = value + return result + + +def _set_by_path(target: dict, dotted: str, value): + """按点号路径写入嵌套 dict""" + keys = dotted.split('.') + node = target + for key in keys[:-1]: + node = node.setdefault(key, {}) + node[keys[-1]] = value + + +def load_config(path=None) -> dict: + """加载配置:默认值 ← config.yml ← 环境变量""" + path = Path(path) if path else config_path() + file_config = {} + if path.is_file(): + import yaml # 延迟导入,未使用配置文件时不强依赖 + try: + with open(path, encoding='utf-8') as f: + file_config = yaml.safe_load(f) or {} + except yaml.YAMLError as e: + raise RuntimeError(f"配置文件解析失败({path}):{e}") from e + if not isinstance(file_config, dict): + raise RuntimeError(f"配置文件格式错误({path}):顶层应为键值映射") + + merged = _deep_merge(DEFAULTS, file_config) + for env_key, dotted in ENV_OVERRIDES.items(): + raw = os.getenv(env_key) + if raw is None or raw == '': + continue + _set_by_path(merged, dotted, raw) + return merged + + +CONFIG = load_config() +LOADED_CONFIG_FILE = config_path() if config_path().is_file() else None + + +def reload_config(): + """重新加载(供测试使用)""" + global CONFIG, LOADED_CONFIG_FILE + CONFIG = load_config() + LOADED_CONFIG_FILE = config_path() if config_path().is_file() else None + apply_dashscope_endpoints() + return CONFIG + + +# --------------------------------------------------------------------- 取值 +def get(dotted: str, default=None): + """按点号路径取值,如 get('models.asr_model')""" + node = CONFIG + for key in dotted.split('.'): + if not isinstance(node, dict) or key not in node: + return default + node = node[key] + return node + + +def get_int(dotted: str, default: int = 0) -> int: + try: + return int(get(dotted, default)) + except (TypeError, ValueError): + logger.warning("配置项 %s 不是整数,使用默认值 %s", dotted, default) + return default + + +def get_bool(dotted: str, default: bool = False) -> bool: + value = get(dotted, default) + if isinstance(value, bool): + return value + if isinstance(value, (int, float)): + return bool(value) + return str(value).strip().lower() in ('1', 'true', 'yes', 'on') + + +def get_dict(dotted: str, default=None): + """取字典配置(如 extra_body);非字典时返回默认值""" + value = get(dotted, None) + if isinstance(value, dict): + return dict(value) + if value not in (None, ''): + logger.warning("配置项 %s 不是字典,已忽略: %r", dotted, value) + return dict(default or {}) + + +def get_list(dotted: str, default=None): + value = get(dotted, None) + if isinstance(value, (list, tuple)): + return list(value) + if isinstance(value, str) and value.strip(): + return [item.strip() for item in value.split(',') if item.strip()] + return list(default or []) + + +# --------------------------------------------------------------------- 派生 +def model(role: str) -> str: + """取某环节实际使用的模型名:role ∈ {asr_model, correct_model, split_model}""" + return str(get(f'models.{role}', '')) + + +# ----------------------------------------------------------------- 接入点/路由 +def endpoint(name: str) -> dict: + """取某个接入点定义(endpoints.)""" + return get(f'endpoints.{name}', {}) or {} + + +def route(role: str) -> str: + """取某环节走哪个接入点:role ∈ {asr, correct, split}""" + return str(get(f'routes.{role}', DEFAULT_ROUTES.get(role, ''))) + + +def endpoint_for(role: str): + """ + 解析某环节的接入点,返回 (接入点名字, 接入点定义) + + 未配置或缺少该协议必需字段时抛 RuntimeError(附修复指引),避免调用时才报难懂的错。 + """ + name = route(role) + if not name: + raise RuntimeError(f"config.yml 未配置 routes.{role}") + cfg = endpoint(name) + if not cfg: + raise RuntimeError( + f"config.yml 的 routes.{role} 指向接入点 '{name}',但 endpoints 下没有该定义") + kind = str(cfg.get('kind', 'openai')).lower() + if kind == 'dashscope': + if not cfg.get('http_base_url') and not cfg.get('websocket_base_url'): + raise RuntimeError(f"接入点 '{name}' 是 dashscope 类型,但缺少 http_base_url / websocket_base_url") + elif kind == 'openai': + if not cfg.get('base_url'): + raise RuntimeError(f"接入点 '{name}' 是 openai 类型,但缺少 base_url") + else: + raise RuntimeError(f"接入点 '{name}' 的 kind='{kind}' 不支持(仅支持 dashscope / openai)") + if not cfg.get('api_key_env'): + raise RuntimeError(f"接入点 '{name}' 未配置 api_key_env(密钥在 .env 中的变量名)") + return name, cfg + + +def openai_url(role: str) -> str: + """OpenAI 兼容接入点的完整请求地址(base_url + chat_completions_path 拼接)""" + _, cfg = endpoint_for(role) + base = str(cfg.get('base_url', '')).rstrip('/') + path = str(cfg.get('chat_completions_path', '/chat/completions')) + if not path.startswith('/'): + path = '/' + path + return base + path + + +def api_key_env(role: str) -> str: + """该环节密钥所在的 .env 变量名""" + return str(endpoint_for(role)[1].get('api_key_env', '')) + + +def api_key(role: str) -> str: + """该环节的密钥值(从环境变量取,即 .env)""" + return os.getenv(api_key_env(role), '') + + +def apply_dashscope_endpoints(dashscope_module=None): + """ + 把 config.yml 里 dashscope 类型的接入点地址写入 SDK + + dashscope SDK 在 **import 时**读取 `DASHSCOPE_HTTP_BASE_URL` / + `DASHSCOPE_WEBSOCKET_BASE_URL`,因此在 config 加载阶段就写环境变量, + 并在已 import 之后再次赋值模块属性,避免依赖 import 顺序。 + """ + http_url = ws_url = '' + try: + if route('correct') and endpoint(route('correct')).get('kind') == 'dashscope': + http_url = endpoint(route('correct')).get('http_base_url', '') + if route('asr') and endpoint(route('asr')).get('kind') == 'dashscope': + ws_url = endpoint(route('asr')).get('websocket_base_url', '') + except Exception: # 配置不全时不阻塞启动,调用时才会明确报错 + return + if not http_url: + http_url = endpoint('dashscope').get('http_base_url', '') + if not ws_url: + ws_url = endpoint('dashscope').get('websocket_base_url', '') + + if http_url and not os.environ.get('DASHSCOPE_HTTP_BASE_URL'): + os.environ['DASHSCOPE_HTTP_BASE_URL'] = http_url + if ws_url and not os.environ.get('DASHSCOPE_WEBSOCKET_BASE_URL'): + os.environ['DASHSCOPE_WEBSOCKET_BASE_URL'] = ws_url + + if dashscope_module is not None: + if http_url: + dashscope_module.base_http_api_url = http_url + if ws_url: + dashscope_module.base_websocket_api_url = ws_url + + +def validate_endpoints(): + """返回配置问题清单(供启动日志提示,不抛异常)""" + problems = [] + for role in ('asr', 'correct', 'split'): + try: + name, cfg = endpoint_for(role) + except RuntimeError as e: + problems.append(str(e)) + continue + if role == 'asr' and str(cfg.get('kind', '')).lower() != 'dashscope': + problems.append( + f"routes.asr -> '{name}'(kind={cfg.get('kind')}):实时语音识别仅支持 kind=dashscope," + f"换 ASR 供应商需要新增适配器") + key_env = cfg.get('api_key_env', '') + if key_env and not os.getenv(key_env): + problems.append(f"接入点 '{name}' 的密钥未设置:请在 .env 中填写 {key_env}") + return problems + + +# ----------------------------------------------------------------- 敏感项校验 +def required_secrets(): + """运行所需的全部敏感项:数据库口令 + 各环节接入点的密钥""" + keys = {'MYSQL_PASSWORD'} + for role in ('asr', 'correct', 'split'): + try: + key_env = api_key_env(role) + except RuntimeError: + continue + if key_env: + keys.add(key_env) + return sorted(keys) + + +def missing_secrets(): + """返回缺失的敏感项列表""" + return [key for key in required_secrets() if not os.getenv(key)] + + +def check_secrets(): + """校验敏感项;缺失时记录 ERROR(不抛异常,便于只做只读操作的场景)""" + missing = missing_secrets() + if missing: + logger.error("缺少敏感配置 %s;请在 %s 中填写(模板见 .env.example)", ', '.join(missing), env.env_file()) + return missing + + +def resolve_dir(value) -> Path: + """目录配置解析:相对路径基于项目根""" + path = Path(str(value)).expanduser() + return path if path.is_absolute() else (BASE_DIR / path) + + +def video_dir() -> Path: + return resolve_dir(get('paths.video_dir', 'xwlb_video')) + + +def audio_dir() -> Path: + return resolve_dir(get('paths.audio_dir', 'audio_processing')) + + +def db_config() -> dict: + """数据库连接参数(口令来自 .env)""" + return { + 'host': get('mysql.host', 'localhost'), + 'port': get_int('mysql.port', 3306), + 'username': get('mysql.user', 'myquant'), + 'password': os.getenv('MYSQL_PASSWORD', ''), + 'database': get('mysql.database', 'myquant'), + } + + +def log_summary(): + """启动时打印一次生效配置(不含敏感项),并校验敏感项与接入点配置""" + source = str(LOADED_CONFIG_FILE) if LOADED_CONFIG_FILE else '内置默认值(未找到 config.yml)' + logger.info( + "配置来源: %s | MySQL %s@%s:%s/%s | 模型 asr=%s correct=%s split=%s | 校对=%s | 完成后清理=%s", + source, + get('mysql.user'), get('mysql.host'), get('mysql.port'), get('mysql.database'), + model('asr_model'), model('correct_model'), model('split_model'), + '开启' if get_bool('llm_correct.enabled', True) else '关闭', + '开启' if get_bool('cleanup.after_daily_run', True) else '关闭', + ) + for role in ('asr', 'correct', 'split'): + try: + name, cfg = endpoint_for(role) + except RuntimeError as e: + logger.error("接入点配置错误: %s", e) + continue + if str(cfg.get('kind', '')).lower() == 'dashscope': + url = (cfg.get('websocket_base_url', '') if role == 'asr' else '') or cfg.get('http_base_url', '') + else: + url = openai_url(role) + logger.info("接入点 %-5s -> %-10s kind=%-9s %s(密钥变量 %s)", + role, name, cfg.get('kind'), url, cfg.get('api_key_env')) + for problem in validate_endpoints(): + logger.error("配置检查: %s", problem) + check_secrets() + + +# config 加载阶段就把 dashscope 地址写入环境变量(SDK 在 import 时读取) +apply_dashscope_endpoints() \ No newline at end of file diff --git a/config.yml b/config.yml new file mode 100644 index 0000000..28c73fa --- /dev/null +++ b/config.yml @@ -0,0 +1,136 @@ +# ============================================================================ +# xwlb 非敏感配置 +# ============================================================================ +# 敏感配置(密钥、数据库口令)保留在 .env: +# MYSQL_PASSWORD / DASHSCOPE_API_KEY / DEEPSEEK_API_KEY +# +# 修改本文件后无需改代码;可用 XWLB_CONFIG_FILE 指定其他配置文件。 +# 优先级:进程环境变量 > config.yml > 代码内置默认值 +# (即为临时试验,可用 `DASHSCOPE_LLM_MODEL=qwen-max python audioRead.py ...` 覆盖) +# ============================================================================ + +# ---- 数据库(口令在 .env 的 MYSQL_PASSWORD)---- +mysql: + # 隧道自愈:端口不通时自动执行下面的脚本(逻辑见 tunnel.py,所有入口都生效) + tunnel: + enabled: 1 + script: autossh.sh # 相对项目根 + systemd_unit: xwlb-tunnel.service # 该单元 active 时只等待它自动重连,不抢端口 + wait_seconds: 30 # 执行后最多等多久 + connect_timeout: 2 # 单次 TCP 探测超时 + host: localhost + port: 13306 # 经 autossh.sh 隧道时的本地端口(隧道: 13306 -> 远端 3306) + user: myquant + database: myquant + +# ---- 目录(相对项目根,也可写绝对路径)---- +paths: + video_dir: xwlb_video # mp4 / mp3 中间产物 + audio_dir: audio_processing # wav 分片 + +# ---- 各环节实际使用的模型(集中在此管理)---- +models: + asr_model: paraformer-realtime-v2 # 语音识别 + correct_model: qwen3.8-flash # ASR 文本校对 + split_model: deepseek-flash # 新闻切分 + 标题(API 实测:deepseek-flash / deepseek-v4-pro) + +# ============================================================================ +# 接入点(base url):换供应商 / 换区域 / 走公司网关或代理,只改这一段 +# ============================================================================ +# 每个接入点声明三件事: +# kind 协议类型:dashscope=用 dashscope SDK;openai=OpenAI 兼容的 /chat/completions +# *base_url* 服务地址 +# api_key_env 该供应商的密钥放在 .env 里的**变量名**(密钥本身不进本文件) +endpoints: + dashscope: + kind: dashscope + # 文本生成(校对)走 HTTP + http_base_url: https://dashscope.aliyuncs.com/api/v1 + # 实时语音识别走 WebSocket + websocket_base_url: wss://dashscope.aliyuncs.com/api-ws/v1/inference + api_key_env: DASHSCOPE_API_KEY + # 国际站示例:把上面两行换成 + # http_base_url: https://dashscope-intl.aliyuncs.com/api/v1 + # websocket_base_url: wss://dashscope-intl.aliyuncs.com/api-ws/v1/inference + deepseek: + kind: openai + base_url: https://api.deepseek.com/v1 + chat_completions_path: /chat/completions + api_key_env: DEEPSEEK_API_KEY + # 注意:**切分环节不要关思维链**。同一天(2026-09-23)实测对比: + # 思考开启 27 条 / 正文覆盖率 99.67% / 19k token + # 思考关闭 20 条 / 正文覆盖率 95.34% / 4.7k token + # 关掉省 4 倍 token 但会丢 4.7% 正文、少切 7 条新闻,得不偿失。 + # 若确实要关(DeepSeek 用 thinking 字段,写 enable_thinking 会被静默忽略),在此加: + # extra_body: + # thinking: {type: disabled} + qwen: + kind: openai + base_url: https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1 + chat_completions_path: /chat/completions + api_key_env: QWEN_TOKEN_PLAN + # 校对关思维链:通义/百炼兼容模式用 enable_thinking(实测 4045 → 312 token,约 13 倍, + # 输出仅差一个"的"字、数值无漂移)。校对是机械任务,想不出什么,纯属浪费。 + extra_body: + enable_thinking: false + +# 每个环节走哪个接入点(值是 endpoints 下的名字) +routes: + asr: dashscope # 语音识别:仅支持 kind=dashscope(paraformer 实时识别协议) + correct: qwen # 文本校对 + split: deepseek # 新闻切分 + +# ---- 换供应商示例:把「校对」换成 OpenAI 兼容的第三方,只需三步 ---- +# 1) 上面的 routes.correct 改成 moonshot +# 2) 在 endpoints 下加一段: +# moonshot: +# kind: openai +# base_url: https://api.moonshot.cn/v1 +# chat_completions_path: /chat/completions +# api_key_env: MOONSHOT_API_KEY +# 3) models.correct_model 改成对方的模型名(如 kimi-k2-0905-preview) +# 并在 .env 里加 MOONSHOT_API_KEY=... +# 说明:换供应商不影响数值事实守卫(对任何供应商的校对结果都会校验)。 +# +# 若某供应商的 OpenAI 兼容地址不带版本号(如 https://my-gateway/llm), +# 直接写进 base_url 即可,代码只做 base_url + chat_completions_path 拼接。 + +# ---- 语音识别 ---- +asr: + sample_rate: 16000 + language_hints: [zh, en] + +# ---- 音频切分(静音检测 + 合并)---- +audio_split: + min_silence_ms: 700 # 超过该静音长度即切分 + silence_thresh_db: -40 # 静音阈值(dBFS) + keep_silence_ms: 400 # 切分处保留的静音 + max_chunk_ms: 180000 # 单片最长 3 分钟 + +# ---- LLM 文本校对 ---- +llm_correct: + enabled: 1 # 1=开启校对(默认);0=关闭,直接以 ASR 原文作为 news_improve + max_retries: 2 # 失败重试次数(失败后回退原文,不丢分片) + timeout: 120 # 单次调用超时(秒) + max_tokens: 8000 + temperature: 0.1 + top_p: 0.5 + # 思维链开关在接入点(endpoints.*.extra_body)上配置,因为参数名因供应商而异; + # 如需只针对本环节覆盖,可在此加 extra_body: { ... } + # 开启时仍受「数值事实守卫」保护:若校对改动了数字/年份/届次/规划期等, + # 该分片一律回退 ASR 原文(详见 docs/BUGS.md B2 补充说明) + +# ---- DeepSeek 新闻切分 ---- +llm_split: + max_tokens: 60000 # 首次尝试;推理模型 reasoning 可能占 2 万+,上限要给足(按实际产出计费) + retry_max_tokens: 80000 # 解析失败后强化提示词重试时的上限 + temperature: 0.5 + timeout: 180 # 单次 HTTP 超时(秒);输出 2 万 token 级别需要更久 + max_retries: 3 # 网络/5xx 重试次数(429、5xx 退避重试) + +# ---- 任务完成后清理中间产物 ---- +cleanup: + after_daily_run: 1 # 1=当天全程任务结束后自动清理;0=不清理 + only_on_success: 1 # 1=仅当当天全部成功才清理(失败时保留文件便于重跑) + remove_video: 1 # 删除 paths.video_dir 下所有 mp3 / mp4 + remove_audio: 1 # 删除 paths.audio_dir 下所有 wav diff --git a/deepseek.py b/deepseek.py new file mode 100644 index 0000000..2df02cc --- /dev/null +++ b/deepseek.py @@ -0,0 +1,365 @@ +import env # 加载 .env(敏感项) +import config # 加载 config.yml(模型、参数) +import requests +import json +import time +import logging +from typing import Optional, Dict, Any +import os + +# 配置日志 +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + +class DeepSeekAPI: + """OpenAI 兼容协议的对话客户端(DeepSeek 是默认接入点) + + 地址与密钥变量名都来自 config.yml 的 `endpoints.>`, + 因此把 routes.split / routes.correct 指向别的 OpenAI 兼容供应商即可切换,无需改代码。 + 请求地址 = base_url + chat_completions_path。 + """ + + def __init__(self, api_key: Optional[str] = None, role: str = 'split', + endpoint: Optional[str] = None): + """ + 初始化客户端 + + Args: + api_key: 直接指定密钥;为 None 时按接入点的 api_key_env 从 .env 读 + role: 环节名(asr/correct/split),用于解析 routes. + endpoint: 直接指定接入点名(覆盖 role 的路由) + """ + if endpoint: + cfg = config.endpoint(endpoint) + if not cfg: + raise RuntimeError(f"config.yml 的 endpoints 下没有接入点 '{endpoint}'") + name = endpoint + else: + name, cfg = config.endpoint_for(role) + + endpoint_extra = cfg.get('extra_body') + self.endpoint_extra_body = dict(endpoint_extra) if isinstance(endpoint_extra, dict) else {} + if endpoint_extra not in (None, {}) and not isinstance(endpoint_extra, dict): + logger.warning("接入点 %s 的 extra_body 应为字典,已忽略: %r", name, endpoint_extra) + self.endpoint_name = name + self.role = role + self.kind = str(cfg.get('kind', 'openai')).lower() + if self.kind != 'openai': + raise RuntimeError( + f"接入点 '{name}' 的 kind='{self.kind}',OpenAI 兼容客户端只能用于 kind=openai 的接入点") + + base = str(cfg.get('base_url', '')).rstrip('/') + path = str(cfg.get('chat_completions_path', '/chat/completions')) + self.api_url = base + (path if path.startswith('/') else '/' + path) + self.api_key = api_key or os.getenv(cfg.get('api_key_env', ''), '') + if not self.api_key: + logger.warning("接入点 %s 的密钥未设置(.env 中的 %s)", name, cfg.get('api_key_env')) + + # 调用参数按环节取各自配置段 + section = 'llm_correct' if role == 'correct' else 'llm_split' + self.section = section + self.max_retries = config.get_int(f'{section}.max_retries', 3) + self.retry_delay = 2 # 秒 + self.timeout = config.get_int(f'{section}.timeout', 60) + + # 默认系统提示词 + self.default_system_prompt = """你是一个专业的AI助手,能够准确理解用户需求并提供高质量的回答。 +请根据用户的输入进行适当的处理和分析,保持回答的专业性和准确性。注意:所处理文字来自中央电视台新闻联播节目转文字,请在内容审查时重点考虑。""" + + def _handle_api_error(self, response: requests.Response) -> str: + """ + 处理API错误响应 + + Args: + response: API响应对象 + + Returns: + 错误描述信息 + """ + error_msg = (f"API请求失败 [{self.endpoint_name} {self.api_url}]: " + f"{response.status_code} {response.reason}") + + try: + error_data = response.json() + if 'error' in error_data: + error_msg += f" - {error_data['error'].get('message', '未知错误')}" + logger.error(f"API错误详情: {error_data}") + except json.JSONDecodeError: + error_msg += f" - 响应内容: {response.text[:200]}" + + return error_msg + + def _make_api_request(self, payload: Dict[str, Any]) -> Dict[str, Any]: + """ + 发送API请求并处理响应 + + Args: + payload: 请求数据 + + Returns: + API响应数据 + + Raises: + Exception: 当所有重试都失败时抛出异常 + """ + headers = { + "Content-Type": "application/json", + "Authorization": f"Bearer {self.api_key}" + } + + last_exception = None + + for attempt in range(self.max_retries): + try: + logger.info("发送API请求 [%s] (尝试 %d/%d)", self.endpoint_name, + attempt + 1, self.max_retries) + + response = requests.post( + self.api_url, + headers=headers, + json=payload, + timeout=self.timeout # 秒(config.yml:
.timeout) + ) + + if response.status_code == 200: + return response.json() + elif response.status_code == 400: + # 400错误通常是请求格式问题,不需要重试 + error_msg = self._handle_api_error(response) + raise Exception(f"请求参数错误: {error_msg}") + elif response.status_code == 401: + # 401未授权错误,不需要重试 + raise Exception("API密钥无效或未授权,请检查您的API密钥") + elif response.status_code == 429: + # 速率限制,需要重试 + logger.warning("达到速率限制,等待后重试...") + time.sleep(self.retry_delay * (attempt + 1)) + continue + elif 500 <= response.status_code < 600: + # 服务器错误,需要重试 + logger.warning(f"服务器错误 {response.status_code},等待后重试...") + time.sleep(self.retry_delay * (attempt + 1)) + continue + else: + error_msg = self._handle_api_error(response) + raise Exception(f"API请求失败: {error_msg}") + + except requests.exceptions.Timeout: + last_exception = Exception(f"请求超时 (尝试 {attempt + 1})") + logger.warning(f"请求超时,等待后重试...") + time.sleep(self.retry_delay * (attempt + 1)) + + except requests.exceptions.ConnectionError: + last_exception = Exception(f"网络连接错误 (尝试 {attempt + 1})") + logger.warning(f"网络连接错误,等待后重试...") + time.sleep(self.retry_delay * (attempt + 1)) + + except requests.exceptions.RequestException as e: + last_exception = Exception(f"请求异常: {str(e)}") + logger.warning(f"请求异常,等待后重试...") + time.sleep(self.retry_delay * (attempt + 1)) + + # 所有重试都失败 + if last_exception: + raise last_exception + else: + raise Exception("API请求失败,未知错误") + + def process_text(self, + prompt: str, + text: str, + system_prompt: Optional[str] = None, + model: str = "deepseek-chat", + temperature: float = 0.7, + max_tokens: int = 2000, + response_format: Optional[Dict] = None, + extra_body: Optional[Dict] = None) -> str: + """ + 处理文本的通用方法 + + Args: + prompt: 用户提示词 + text: 需要处理的文本(约1万字符) + system_prompt: 系统提示词,如果为None则使用默认值 + model: 使用的模型 + temperature: 生成温度 + max_tokens: 最大生成token数 + + Returns: + 处理后的文本 + + Raises: + Exception: 当处理失败时抛出包含详细信息的异常 + """ + # 输入验证 + if not self.api_key: + raise Exception( + f"接入点 {self.endpoint_name} 的密钥未设置:请在 .env 中配置 " + f"{config.endpoint_for(self.role)[1].get('api_key_env') if self.role in ('asr', 'correct', 'split') else '对应变量'}") + + if not prompt or not text: + raise Exception("prompt和text不能为空") + + # 检查文本长度(约1万字符) + if len(text) > 15000: # 留一些余量 + logger.warning(f"输入文本长度({len(text)}字符)较长,可能会超过上下文限制") + + # 准备系统提示词 + system_content = system_prompt or self.default_system_prompt + + # 构建消息 + messages = [ + {"role": "system", "content": system_content}, + {"role": "user", "content": f"{prompt}\n\n文本内容:\n{text}"} + ] + + # 构建请求数据 + payload = { + "model": model, + "messages": messages, + "temperature": temperature, + "max_tokens": max_tokens, + "stream": False + } + if response_format: + payload["response_format"] = response_format + # 供应商特有参数原样透传,用于关闭推理模型的思维链等。 + # 先应用接入点级(语法属于供应商:通义 {enable_thinking: false}, + # DeepSeek {thinking: {type: disabled}},写错会被静默忽略),再用环节级覆盖。 + if self.endpoint_extra_body: + payload.update(self.endpoint_extra_body) + if extra_body: + payload.update(extra_body) + + try: + # 发送API请求 + response_data = self._make_api_request(payload) + + # 解析响应 + if 'choices' in response_data and len(response_data['choices']) > 0: + result = response_data['choices'][0]['message']['content'] + logger.info("文本处理成功完成") + return result.strip() + else: + raise Exception("API响应格式异常,未找到有效结果") + + except Exception as e: + logger.error(f"文本处理失败: {str(e)}") + raise Exception(f"文本处理失败: {str(e)}") + + def process_text_with_fallback(self, + prompt: str, + text: str, + system_prompt: Optional[str] = None, + **kwargs) -> str: + """ + 带降级处理的文本处理方法 + + Args: + prompt: 用户提示词 + text: 需要处理的文本 + system_prompt: 系统提示词 + **kwargs: 其他参数 + + Returns: + 处理后的文本,如果API调用失败则返回降级结果 + """ + try: + return self.process_text(prompt, text, system_prompt, **kwargs) + except Exception as e: + logger.error(f"API调用失败,使用降级处理: {str(e)}") + # 这里可以添加降级逻辑,比如返回原始文本或简单处理 + return f"处理失败,返回原始文本(错误: {str(e)})\n\n{text}" + +# 使用示例 +def deepseek_text(text, prompt, model=None, max_tokens=None, temperature=None, use_fallback=True): + """ + 调用 DeepSeek 处理文本(默认 json_object 模式) + + 参数: + text: 待处理文本 + prompt: 提示词 + model: 模型名,默认取 config.yml 的 models.split_model + max_tokens: 最大生成 token 数,默认取 llm_split.max_tokens + temperature: 采样温度,默认取 llm_split.temperature + use_fallback: 调用失败时是否返回降级文本(False 时直接抛异常,便于调用方重试) + 返回值: + str: 模型返回内容 + """ + api_client = DeepSeekAPI(role='split') # 接入点/密钥来自 config.yml 的 routes.split + + custom_system_prompt = "你是一个专业的文本分析助手,擅长根据提示词对长文本进行深入分析。" + + try: + # 处理文本(使用 response_format 强制返回 JSON) + result = api_client.process_text( + model=model or config.model('split_model'), + prompt=prompt, + text=text, + system_prompt=custom_system_prompt, + temperature=config.get('llm_split.temperature', 0.5) if temperature is None else temperature, + max_tokens=config.get_int('llm_split.max_tokens', 20000) if max_tokens is None else max_tokens, + response_format={"type": "json_object"}, + extra_body=config.get_dict('llm_split.extra_body'), + ) + return result + + except Exception as e: + logger.error(f"DeepSeek 处理失败: {e}") + if not use_fallback: + raise + + # 使用降级方法(显式传入同一模型与参数,避免兜底时**悄悄换用另一个模型**—— + # 曾出现配置的模型名无效、兜底用硬编码 deepseek-chat 成功、 + # 于是"按配置运行"看起来正常实则完全没生效) + fallback_result = api_client.process_text_with_fallback( + prompt=prompt, + text=text, + system_prompt=custom_system_prompt, + model=model or config.model('split_model'), + temperature=config.get('llm_split.temperature', 0.5) if temperature is None else temperature, + max_tokens=config.get_int('llm_split.max_tokens', 20000) if max_tokens is None else max_tokens, + extra_body=config.get_dict('llm_split.extra_body'), + ) + # 原实现此处误写为 return result(异常分支下 result 未绑定,会抛 NameError, + # 见 docs/BUGS.md B7) + return fallback_result + + +def openai_chat(text, prompt, role='correct', model=None, max_tokens=None, temperature=None, + timeout=None, system_prompt=None): + """ + 通用 OpenAI 兼容对话调用(供 **非 DashScope 供应商的文本校对** 使用) + + 地址 = config.yml `endpoints.>`,即换供应商只改配置。 + 与 `deepseek_text` 的区别:不强制 json_object,直接返回纯文本。 + + 参数: + text/prompt: 待处理文本与提示词 + role: 环节名,默认 correct(决定用哪个接入点与哪段参数) + model: 模型名,默认取 models.correct_model(role=split 时取 models.split_model) + max_tokens/temperature/timeout: 不传则取该环节配置段 + 返回值: + str: 模型返回的文本 + """ + client = DeepSeekAPI(role=role) + section = client.section + if timeout is not None: + client.timeout = timeout + default_model = config.model('split_model') if role == 'split' else config.model('correct_model') + return client.process_text( + model=model or default_model, + prompt=prompt, + text=text, + system_prompt=system_prompt, + temperature=config.get(f'{section}.temperature', 0.3) if temperature is None else temperature, + max_tokens=config.get_int(f'{section}.max_tokens', 8000) if max_tokens is None else max_tokens, + extra_body=config.get_dict(f'{section}.extra_body'), + ) + +if __name__ == "__main__": + import sys + if len(sys.argv) < 3: + print("用法: python deepseek.py ") + sys.exit(1) + print(deepseek_text(sys.argv[2], sys.argv[1])) \ No newline at end of file diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..e84290f --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,393 @@ +# xwlb 项目技术说明 + +> 一句话定位:把央视《新闻联播》完整版视频自动转成**结构化单条新闻**并入库的离线流水线。 +> 本文基于仓库当前源码 + `main.log`(2026-06-01 ~ 2026-09-23)实测行为整理。缺陷清单见 [`BUGS.md`](./BUGS.md)。 + +--- + +## 1. 端到端数据流 + +``` + main.py(今天) + │ start_date = end_date = %Y%m%d + ▼ + ┌──────────────────────────────────────────────────────────────┐ + │ ① 链接发现 getVideo5.get_all_video_links(start, end) │ + │ xwlb_urls() → https://tv.cctv.com/lm/xwlb/day/ │ + │ YYYYMMDD.shtml (按天,含端点) │ + │ get_xwlb_video_link()→ 正则/文本匹配「完整版《新闻联播》」 │ + │ 的 ,得到 VID 详情页 URL │ + │ 例 tv.cctv.com/2024/10/30/VID...shtml │ + ├──────────────────────────────────────────────────────────────┤ + │ ② 下载与抽音 download_and_extract_audio(url, date, dir) │ + │ yt-dlp format=best[ext=mp4]/best → xwlb_video/YYYYMMDD.mp4 │ + │ ffmpeg -c:a libmp3lame -q:a 0 -map a → 同名 .mp3 │ + ├──────────────────────────────────────────────────────────────┤ + │ ③ 转写落库 audioRead.process_long_audio(mp3, out_dir, date) │ + │ convert_mp3_to_wav() 16kHz / 单声道 / 16bit → input.wav│ + │ split_audio_by_smart_silence(audio_split.*,默认 700ms/-40) │ + │ → 贪心合并为 ≤3 分钟分片 chunk_i.wav(实测 11 片/天) │ + │ for each chunk: │ + │ transcribe_audio() models.asr_model(paraformer-...v2) │ + │ └─ merge_transcripts() 句级结果空格拼接 │ + │ └─ analyze_and_correct_text() models.correct_model │ + │ (默认 qwen3.7-max;受数值事实守卫约束) │ + │ → 先删后插 xwlb_daily(news_days, daily_sub_id=i, ...) │ + │ → os.remove(chunk) │ + ├──────────────────────────────────────────────────────────────┤ + │ ④ 切分入库 newsProcess.news_to_db(date) │ + │ 读 xwlb_daily.news_improve 按 daily_sub_id 升序 → '\n' 拼接 │ + │ models.split_model(deepseek-chat,json_object) │ + │ prompt:切分为独立新闻 + 每条起标题(含"国内/国际/联播快讯")│ + │ → 解析 [{news_id, news_title, news_content}] │ + │ → 删除该日期旧记录 → 批量 INSERT xwlb_daily_ext │ + ├───────────────────────────────────────────────────────────────┤ + │ ⑤ 清理 cleanup.maybe_cleanup_after_run(date, day_ok) │ + │ 当天全部成功才清空 xwlb_video/*.mp3|*.mp4、audio_processing/*.wav │ + └───────────────────────────────────────────────────────────────┘ +``` + +### 两道 LLM 的分工 +| 阶段 | 接入点 | 模型 | 输入 | 输出 | 落库位置 | +|---|---|---|---|---|---| +| 校对 | `routes.correct`(默认 `dashscope`) | `models.correct_model`(默认 `qwen3.7-max`) | 单分片 ASR 原始文本 | "修正后全文" | `xwlb_daily.news_improve` | +| 切分 | `routes.split`(默认 `deepseek`) | `models.split_model`(默认 `deepseek-chat`) | 当天全部分片拼接文本 | JSON 数组 | `xwlb_daily_ext` 多行 | + +模型名(`models.*`)与接入点(`endpoints.*` + `routes.*`)统一在 `config.yml` 管理, +**换模型或换供应商都不需要动代码**(详见 [第 4 节](#4-配置项分层敏感在-env其余在-configyml))。 + +--- + +## 2. 模块职责与函数索引 + +| 文件 | 关键函数 | 说明 | +|---|---|---| +| `main.py` | — | 入口:当天日期 → `process_videos` | +| `main_videos.py` | `get_missing_dates(start, end)` | 反查 `xwlb_daily` 缺哪些日期,逐日补跑(文件内示例区间写死为 2025-01-01~2025-10-25) | +| `newsRedo.py` | `_parse_date` / `main` | 手动重跑某天:`ext>5` 条→跳过;`xwlb_daily` 有行→只做 AI 切分;无行→跑全流程。支持 `yyyymmdd` 与 `yyyy-mm-dd` | +| `getVideo5.py` | `xwlb_urls` | 生成按天页面 URL 列表 | +| | `get_xwlb_video_link` | 解析单日页面找完整版链接(返回值有缺陷,见 BUGS B4) | +| | `get_all_video_links` | 逐日聚合 | +| | `download_and_extract_audio` | yt-dlp + ffmpeg | +| | `process_videos(start, end)` | 主流程编排:下载 → 转写 → 切分入库 → **成功后清理中间产物** | +| `config.py` | `load_config` / `get` / `get_int` / `get_bool` / `get_list` | 加载 `config.yml`,环境变量可覆盖(`XWLB_CONFIG_FILE` 换文件) | +| | `model(role)` | 取 `models.asr_model` / `correct_model` / `split_model` | +| | `endpoint` / `route` / `endpoint_for` | **接入点与路由**:`routes.` → `endpoints.`,配置不全时抛可读错误 | +| | `openai_url(role)` / `api_key_env(role)` / `api_key(role)` | OpenAI 兼容请求地址拼接、该环节密钥的变量名 / 变量值 | +| | `apply_dashscope_endpoints` | 把 dashscope 接入点地址写入 SDK(env + 模块属性,不依赖 import 顺序) | +| | `validate_endpoints` / `required_secrets` / `missing_secrets` / `check_secrets` | 启动校验;必需密钥**按路由推导**(只要求用到的那几家) | +| | `video_dir` / `audio_dir` / `db_config` | 目录解析(相对项目根)、数据库参数(口令来自 `.env`) | +| | `log_summary` | 启动打印生效配置与各环节接入点(不含敏感项)并校验 | +| `cleanup.py` | `maybe_cleanup_after_run(date, day_ok)` | 当天结束后按 `cleanup.*` 决定是否清理 | +| | `cleanup_intermediates` | 删除 `*.mp3/*.mp4` 与 `*.wav`,返回删除数/字节数;也可命令行单独执行 | +| `audioRead.py` | `convert_mp3_to_wav` | mp3 → 16k 单声道 wav(采样率见 `asr.sample_rate`) | +| | `split_audio_by_fixed_duration` | 定长切分(当前未启用) | +| | `split_audio_by_smart_silence` | **实际使用**:静音切分 + ≤3 分钟合并 | +| | `transcribe_audio` | 单分片 ASR,**只返回文本**(失败返回 `''`;校对已解耦) | +| | `_extract_llm_text` | 兼容 `output.text` 与 `output.choices[0].message.content` 两种返回形态 | +| | `merge_transcripts` | 句列表 → 空格拼接 | +| | `text_correction` | qwen 校对:显式 timeout + 重试,失败抛异常 | +| | `_number_signature` / `_number_drift` | **数值事实守卫**:中文/阿拉伯数字归一后比较数值签名,检测校对是否篡改事实 | +| | `analyze_and_correct_text` | 失败一律回退原文;**数值事实被改动时同样回退原文**;`llm_correct.enabled=0` 可关闭 | +| | `analyze_text` | 通用 qwen 文本分析(当前未使用) | +| | `_upsert_daily_chunk` | 按 `(news_days, daily_sub_id)` 先删后插,实现幂等 | +| | `process_long_audio` | 单日音频全流程 + 逐片写 `xwlb_daily` + 清理分片临时文件;返回 `{'chunks','ok','failed'}` | +| `newsProcess.py` | `normalize_date` | `YYYYMMDD` / `YYYY-MM-DD` 统一为后者 | +| | `get_news_improve_by_date` | 取当天 `news_improve` 拼接(过滤空白行);无内容返回 `None` | +| | `extract_news_rows` | 宽解析 LLM 返回(剥 ``` 围栏 / raw_decode / 形态归一 / 字段校验),失败抛 `ValueError` | +| | `news_to_db(target_date, force=False)` | 切分入库:失败重试一次 → **按日期替换**写入 `xwlb_daily_ext`;返回 `written`/`skipped`/`failed` | +| `deepseek.py` | `DeepSeekAPI(role=...)` | **OpenAI 兼容客户端**:地址/密钥变量名来自接入点配置;3 次重试、429/5xx 退避;400/401 不重试 | +| | `deepseek_text(text, prompt, ...)` | 切分便捷入口(`routes.split`,`json_object` 模式) | +| | `openai_chat(text, prompt, role='correct', ...)` | 通用对话入口:供**非 DashScope 供应商的校对**使用(不强制 JSON) | +| `env.py` | `_load_dotenv` | **只加载敏感项**(口令/Key):`XWLB_ENV_FILE` → 项目内 `.env` → 上级 `.env`,已存在的不覆盖 | +| | `missing_secrets` / `check_secrets` | 校验 `MYSQL_PASSWORD` / `DASHSCOPE_API_KEY` / `DEEPSEEK_API_KEY` 是否齐全 | +| `mysql_handler.py` | `MySQLDB` | **项目内置**的数据库层(原位于父项目 `utils/`),接口与原实现一致,另加 `execute()` / `insert_many()` / `query_one()` / 上下文管理器;连接失败立即抛错并给出修复指引 | +| `mysqlHandle.py` | — | 转发项目内 `mysql_handler.MySQLDB` | +| `tests/test_news_parse.py` | — | B3 解析逻辑回归测试(含生产日志中的真实坏返回) | +| `tests/test_fidelity_guard.py` | — | 数值事实守卫回归测试(含 qwen 真实篡改样例) | +| `tests/test_config.py` | — | 配置加载、环境变量优先级、敏感项与 config.yml 隔离 | +| `tests/test_cleanup.py` | — | 清理开关组合(用临时目录与临时 config.yml,不触碰真实文件) | + +### `MySQLDB` 接口约定(调用方依赖,重写时必须保持一致) +```python +db = MySQLDB() # 无参:地址来自 config.yml,口令来自 .env +rows = db.query_data(table="t", columns="c1,c2", where="k = %s", params=(v,)) # → list[dict] +new_id = db.insert_data("t", {"col": val}) # → lastrowid +db.update_data("t", {"col": val}, "k = %s", (v,)) # → rowcount +db.close() # 可重复调用 +``` +`where` 子句允许携带 `order by`(调用方已在用,如 `newsProcess.py`)。 +项目内置实现另提供:`execute(sql, params)` → rowcount(DELETE/DDL)、`insert_many(table, rows)`(单事务批量)、`query_one(...)` → dict|None、以及 `with MySQLDB() as db:`。 + +## 3. 数据库结构 + +### `xwlb_daily` — 音频分片级原文 +| 字段 | 类型 | 说明 | +|---|---|---| +| `nid` | int PK auto_increment | | +| `news_days` | date | 日期 | +| `daily_sub_id` | int | 分片序号(当前 = chunk 下标 i,0 起;**无唯一约束**) | +| `news_raw` | text | ASR 原始文本 | +| `news_improve` | text | qwen 校对后文本(当前实际等于 raw) | +| `news_title` | text | 恒为空串(原设计留给分片标题) | + +### `xwlb_daily_ext` — 单条新闻(最终产物) +| 字段 | 类型 | 说明 | +|---|---|---| +| `extid` | int PK auto_increment | | +| `news_date` | date | 日期 | +| `sub_id` | tinyint | 新闻序号(来自 LLM 的 `news_id`,1 起) | +| `news_title` | varchar(256) | 截断到 256 | +| `news_content` | text | 正文 | + +> 建议补的约束:`xwlb_daily` 唯一索引 `(news_days, daily_sub_id)`;`xwlb_daily_ext` 唯一索引 `(news_date, sub_id)`。 + +--- + +## 4. 配置项(分层:敏感在 .env,其余在 config.yml) + +**分层原则**:口令 / API Key 只放 `.env`(不入版本库);其余全部放 `config.yml`(可入版本库), +包含**各供应商的 base url**(`endpoints.*`)与**每个环节走哪条线**(`routes.*`)。 +**优先级**:`进程环境变量 > config.yml > 代码内置默认值` +(便于临时试验,如 `DASHSCOPE_LLM_MODEL=qwen-max python audioRead.py 20260904`)。 + +`.env` 查找顺序:`XWLB_ENV_FILE` 指定路径 → **项目目录内 `.env`** → 上一级 / 上两级 `.env`(兼容旧 djapi 布局)。 +已存在的进程环境变量优先,不会被 `.env` 覆盖。启动时打印实际来源,例如: + +``` +已加载敏感配置 /home/pi/project/xwlb/.env(注入 11 个变量) +配置来源: /home/pi/project/xwlb/config.yml | MySQL myquant@localhost:13306/myquant | 模型 asr=... correct=... split=... | 校对=开启 | 完成后清理=开启 +``` + +### 4.1 `.env` —— 只有敏感项 + +| 变量 | 必需 | 用途 | +|---|---|---| +| `MYSQL_PASSWORD` | ✅ | 数据库口令 | +| `DASHSCOPE_API_KEY` | 按路由 | ASR + 文本校对(由 `endpoints.dashscope.api_key_env` 指定) | +| `DEEPSEEK_API_KEY` | 按路由 | 新闻切分 + 标题(由 `endpoints.deepseek.api_key_env` 指定) | +| 其他 `*_API_KEY` | 按路由 | 换供应商后在 `endpoints.<新接入点>.api_key_env` 里声明什么名字,这里就放什么 | +| `XWLB_ENV_FILE` | | 指定其他 .env 路径 | +| `XWLB_CONFIG_FILE` | | 指定其他 config.yml 路径 | + +**密钥变量名不写死在代码里**:`config.required_secrets()` 按 `routes` 反查用到的接入点, +只有被使用的供应商才要求配 Key;缺失时 `config.check_secrets()` 直接 ERROR,而不是等到 401 才发现。 + +### 4.2 `config.yml` —— 非敏感项(含**各环节模型**) + +| 配置项 | 默认 | 用途 | +|---|---|---| +| `mysql.host` / `port` / `user` / `database` | `localhost` / `13306` / `myquant` / `myquant` | 数据库地址;**经 SSH 隧道时端口为 13306**(`autossh.sh` 转本地 13306 → 远端 3306) | +| `mysql.tunnel.enabled` / `script` / `systemd_unit` / `wait_seconds` / `connect_timeout` | `1` / `autossh.sh` / `xwlb-tunnel.service` / `30` / `2` | **隧道自愈**:端口不通时的处理(见 `tunnel.py`) | +| `paths.video_dir` / `audio_dir` | `xwlb_video` / `audio_processing` | 相对项目根,也可写绝对路径 | +| `models.asr_model` | `paraformer-realtime-v2` | **语音识别模型** | +| `models.correct_model` | `qwen3.8-flash` | **ASR 文本校对模型** | +| `models.split_model` | `deepseek-flash` | **新闻切分 + 标题模型**(API 实际只提供 `deepseek-flash` / `deepseek-v4-pro`) | +| `endpoints.<名>.kind` | — | 协议类型:`dashscope`(用 SDK)/ `openai`(OpenAI 兼容 `/chat/completions`) | +| `endpoints.<名>.base_url` / `chat_completions_path` | `https://api.deepseek.com/v1` / `/chat/completions` | openai 类型的地址(拼接成完整请求 URL) | +| `endpoints.<名>.http_base_url` / `websocket_base_url` | `https://dashscope.aliyuncs.com/api/v1` / `wss://.../api-ws/v1/inference` | dashscope 类型:文本生成走 HTTP、实时 ASR 走 WS | +| `endpoints.<名>.api_key_env` | `DASHSCOPE_API_KEY` 等 | 该供应商密钥在 `.env` 中的**变量名** | +| `endpoints.<名>.extra_body` | — | 供应商特有请求体参数,如关闭思维链。参数名因厂商而异(通义 `enable_thinking`、DeepSeek `thinking`),写错会被**静默忽略** | +| `routes.asr` / `correct` / `split` | `dashscope` / `qwen` / `deepseek` | 每个环节走哪个接入点(换供应商改这里) | +| `llm_correct.extra_body` / `llm_split.extra_body` | — | 环节级覆盖接入点的 `extra_body`(优先级更高) | +| `asr.sample_rate` / `language_hints` | `16000` / `[zh, en]` | ASR 输入要求 | +| `audio_split.min_silence_ms` / `silence_thresh_db` / `keep_silence_ms` / `max_chunk_ms` | `700` / `-40` / `400` / `180000` | 静音切分参数 | +| `llm_correct.enabled` | `1` | **校对开关**:`0` = 直接把 ASR 原文作为 `news_improve`(开启时仍受数值事实守卫保护) | +| `llm_correct.max_retries` / `timeout` / `max_tokens` / `temperature` / `top_p` | `2` / `120` / `8000` / `0.1` / `0.5` | 校对调用参数 | +| `llm_split.max_tokens` / `retry_max_tokens` / `temperature` / `timeout` / `max_retries` | `60000` / `80000` / `0.5` / `180` / `3` | 切分调用参数;**推理模型的 reasoning token 也计入 max_tokens**,实测波动 14k~24k,给少了会截断甚至返回空 | +| `cleanup.after_daily_run` | `1` | 当天全程任务结束后是否自动清理中间产物 | +| `cleanup.only_on_success` | `1` | 仅当天**全部成功**才清理;`0` = 无论成败都清理 | +| `cleanup.remove_video` / `remove_audio` | `1` / `1` | 分别控制删除 `*.mp3/*.mp4` 与 `*.wav` | + +改模型只改 `models.*` 三行即可,不需要动代码。 + +**推理模型(thinking)的取舍**(实测数据见 [`REPORT_raw_vs_improve.md`](./REPORT_raw_vs_improve.md)): +校对是机械任务,关掉思维链省约 13 倍 token 且输出几乎不变; +切分关掉会丢 4.7% 正文、少切 7 条新闻,**必须保留思考**。 + +### 4.3 环境准备 + +```bash +# 0) 端口隧道(本机访问 13306;不通时执行) +bash autossh.sh # autossh -M 0 -fN -L 13306:localhost:3306 tunnel@doorcome.cn + +# 1) 虚拟环境(本机 Python 为 externally-managed,必须用 venv) +python3 -m venv .venv +.venv/bin/pip install -r requirements.txt +# 注:Python 3.13 已移除标准库 audioop,pydub 需要 requirements 中的 audioop-lts + +# 2) 配置 +cp .env.example .env # 只填 3 个敏感项:MYSQL_PASSWORD / DASHSCOPE_API_KEY / DEEPSEEK_API_KEY +vim config.yml # 非敏感项与模型(已随仓库提供,按需修改) + +# 3) 自检(四个测试都不依赖网络与真实 API) +.venv/bin/python tests/test_config.py # 配置加载与优先级 +.venv/bin/python tests/test_fidelity_guard.py # 数值事实守卫 +.venv/bin/python tests/test_news_parse.py # LLM 返回解析 +.venv/bin/python tests/test_cleanup.py # 清理开关组合(用临时目录,不动真实文件) +.venv/bin/python -c "import config; config.log_summary()" +.venv/bin/python -c "from mysqlHandle import MySQLDB; d=MySQLDB(); print(d.query_one('xwlb_daily','COUNT(*) c')); d.close()" +``` + +系统依赖:**ffmpeg**(`/usr/bin/ffmpeg`,抽音必需)。 + +--- + +## 5. 外部调用参数(实测/代码) + +| 调用 | 参数要点 | 实测耗时 | +|---|---|---| +| 央视页面 | requests,UA 伪装,`Referer: https://tv.cctv.com/`,timeout 10s | 秒级 | +| yt-dlp | `best[ext=mp4]/best`,输出 `YYYYMMDD.mp4` | ~30 分钟视频 / 104MB,约 20s~1min | +| ffmpeg | `-c:a libmp3lame -q:a 0 -map a -y` | 十几秒 | +| pydub 切分 | 静音 700ms / 阈值 -40dBFS / 保留 400ms / 单片 ≤180s | 秒级,得 11 片 | +| paraformer ASR | wav / 16k / `language_hints=['zh','en']` | 约 30~45s/片 | +| qwen 校对 | `max_tokens=8000`(可配)、`temperature=0.1`、`top_p=0.5`、`timeout=90s`、重试 2 次 | 改造后实测 **20~32s/片且有真实产出**;改造前 30~300s 且必然拿到空结果(57 次超时) | +| deepseek-chat | `json_object`、`max_tokens=20000`(重试时 32000)、`temperature=0.5`、timeout 60s、3 重试 | 实测 **27~35s** | + +**单日耗时**:改造前日志实测 20:34:02 → 21:24:50,约 **51 分钟**(其中相当部分耗在"必然失败"的校对调用上)。 +按改造后实测单片耗时(ASR 30~45s + 校对 20~32s)估算,11 片约 10~14 分钟,加上下载、切分与片段处理,单日约 **25~35 分钟**。 + +--- + +## 6. 存储布局 + +``` +xwlb/ +├── main.py / main_videos.py / newsRedo.py # 入口 +├── getVideo5.py / audioRead.py # 采集 + 转写 +├── newsProcess.py / deepseek.py # LLM 切分 +├── config.yml / config.py # 非敏感配置 + 加载器(含各环节模型) +├── env.py # 敏感配置(.env)加载 +├── cleanup.py # 中间产物清理(也是手动清理入口) +├── mysql_handler.py / mysqlHandle.py # 数据库层(已内置,不再依赖父项目) +├── .env / .env.example # 敏感项(口令/Key) / 模板 +├── requirements.txt +├── xwlb_video/ YYYYMMDD.mp3(及 .mp4) # 当天任务成功后自动清空 +├── audio_processing/ input_YYYYMMDD.wav + chunk_*.wav # 当天任务成功后自动清空 +├── docs/ BUGS.md / ARCHITECTURE.md +├── tests/ test_config.py / test_cleanup.py / test_fidelity_guard.py / test_news_parse.py +├── .venv/ 项目虚拟环境(不入版本库) +├── autossh.sh SSH 隧道:本地 13306 → 远端 3306 +└── main.log 历史生产日志 34MB(证据保留,无轮转) +``` +命名约定:中间产物以 `YYYYMMDD` 为键(无横线),数据库字段用 `YYYY-MM-DD`(有横线); +两者由 `newsProcess.normalize_date` 与 `newsRedo._parse_date` 转换(数据库连接字面量两种格式均被接受)。 +临时文件策略(改造后):分片 `chunk_*.wav` 与 `input_YYYYMMDD.wav` 在 `finally` 中删除; +**当天全程任务成功结束后**由 `cleanup.maybe_cleanup_after_run()` 清空 `xwlb_video/*.mp3|*.mp4` 与 `audio_processing/*.wav` +(失败时默认保留以便重跑,见 `config.yml` 的 `cleanup.*`;手动清理:`python cleanup.py [--dry-run]`)。 + +--- + +## 7. 入口脚本对照 + +| 场景 | 命令 | 行为 | +|---|---|---| +| 日常(当天) | `python main.py` | 抓当天 → 下载 → 转写 → 切分(已处理则切分阶段自动跳过) | +| 指定区间 | `python getVideo5.py 20260901 20260907` | 逐日全流程(参数校验:8 位、start ≤ end);已存在 mp3 的日期跳过下载 | +| 补缺失日 | `python main_videos.py` | 先查 `xwlb_daily` 缺失日期再逐日全流程(区间写在文件内,按需修改) | +| 重跑某天 | `python newsRedo.py 20260923` | 已有精编记录则跳过;有原文只重做切分;否则全流程 | +| 强制重切 | `python newsRedo.py 20260923 --force` | 忽略"已处理"判据,重新切分并**覆盖**该日 `xwlb_daily_ext` | +| 只重做切分 | `python newsProcess.py 2026-09-23 [--force]` | 直接对指定日期执行切分入库 | +| 只重做转写 | `python audioRead.py 20260923` | 用已有 `xwlb_video/20260923.mp3` 重跑分割+识别+落库(按分片覆盖,可安全重跑) | +| 手动清理 | `python cleanup.py [--dry-run]` | 清空 `xwlb_video/*.mp3|*.mp4` 与 `audio_processing/*.wav`(含识别完成标记) | +| **定时任务(生产入口)** | `bash scripts/run_daily.sh [YYYYMMDD]` | 全链路 + 失败重试(21:30 / 22:00);由 `systemd/xwlb-daily.timer` 每天 21:00 触发 | +| 这天跑到哪一步 | `python scripts/day_status.py <日期>` | 退出码 0=已有精编 / 1=识别完成只缺切分 / 2=需重跑全链路 | +| 安装定时任务 | `bash scripts/install_systemd.sh` | 安装单元、启用定时器与隧道服务 | +| 校对效果评估 | `python tools/compare_raw_improve.py --md <文件>` | 统计 news_raw vs news_improve 差异分类与 token 开销 | +| 合并方案评估 | `python tools/experiment_merge_correct_split.py <日期>` | 实测"校对+切分合并成一次调用"的覆盖率与事实漂移 | +| 配置自检 | `python tests/test_config.py` | 校验 config.yml 加载、优先级与敏感项隔离 | +| 清理自检 | `python tests/test_cleanup.py` | 校验清理开关组合(临时目录,不动真实文件) | +| 解析自检 | `python tests/test_news_parse.py` | 校验 LLM 返回解析逻辑 | + +### `tunnel.py` —— 隧道自愈(所有入口共用) + +| 函数 | 作用 | +|---|---| +| `port_open(host, port, timeout)` | TCP 层探测端口,不涉及 MySQL 握手 | +| `is_local(host)` | 是否本机地址(`localhost`/`127.0.0.1`/`::1`)→ 决定"隧道"这个概念是否适用 | +| `systemd_unit_active(unit)` | systemd 单元是否 active(systemctl 不可用时安全返回 False) | +| `ensure_tunnel(...)` | 探测 → 不通则按分支等待 systemd 或执行 autossh.sh → 如实返回 `ST_*` 状态 | + +挂在 `MySQLDB.connect()`:**只在连接失败时介入**,所以正常路径没有任何额外开销; +`_ensure_connection()` 的重连也会走同一条路径,长流程中途断隧道同样能恢复。 + +--- + +## 7.1 部署:systemd 定时任务与重试语义 + +生产入口不是 `main.py` 而是 `scripts/run_daily.sh`,由 `systemd/xwlb-daily.timer` 每天 21:00 触发。 + +``` +21:00 xwlb-daily.timer → xwlb-daily.service → scripts/run_daily.sh + ├─ flock 防重入(已有实例则退出码 2) + ├─ 检查 127.0.0.1:13306(ss 快速判断);不通则调用 tunnel.py 自愈 + │ (唯一实现:systemd 单元 active 就等它重连,否则执行 autossh.sh) + ├─ 第 1 次尝试:getVideo5.py <今天> <今天>,上限 40 分钟 + │ 成功 → 退出 0 + │ 失败 ↓ + ├─ 等到 21:30 → scripts/day_status.py 判定阶段 + │ 退出码 0(已有精编)→ 什么都不做 + │ 退出码 1(有分片 **且有 .asr_complete_<日期> 标记**)→ 只跑 newsProcess.py(省掉全部 ASR) + │ 退出码 2(其余)→ 重跑全链路 + └─ 22:00 再同样重试一次;仍失败 → 退出 1(systemd 记录为 failed) +``` + +| 单元 | 关键设置 | 原因 | +|---|---|---| +| `xwlb-daily.timer` | `OnCalendar=*-*-* 21:00:00`、`Persistent=true` | 关机错过时刻则开机补跑 | +| `xwlb-daily.service` | `Type=oneshot`、`TimeoutStartSec=10800` | 默认 90s 会把"等待 + 重试"的服务砍掉 | +| `xwlb-tunnel.service` | autossh + `Restart=always` + `ExitOnForwardFailure=yes` | 隧道是数据库前置条件;`xwlb-daily` 通过 `Wants=`/`After=` 依赖它 | + +退出码约定:`0` 成功 / `1` 重试用尽仍失败 / `2` 已有实例在运行。 + +**为什么必须有 `state/.asr_complete_<日期>` 标记**:识别中途卡死(B19)会留下部分分片, +若只按"库里有没有分片"判断,重试会走"只重跑切分",把**半天内容当成完整一天**入库且不报错。 +标记由 `audioRead.process_long_audio` 在**全部识别成功**时写入,失败时主动删除。 + +标记**刻意放在 `state/` 而不是 `audio_processing/`**,也**不随 `cleanup.py` 删除**: +它是"该日识别已完成"的长期凭证。若随中间产物一起清掉, +就无法区分"已完成并被清理"与"识别只跑了一半、切分却照样写出了 ext"这两种状态。 +配合 `process_videos` 的改动——**识别不完整时不执行切分**——四种状态才互不混淆: + +| 有 `state/` 标记 | 有 `xwlb_daily_ext` | 含义 | 重试动作 | +|---|---|---|---| +| ✅ | ❌ | 识别完成、切分未完成 | 只重跑切分 | +| ✅ | ✅ | 已完成 | 无需动作 | +| ❌ | ❌ | 识别未完成或未开始 | 重跑全链路 | +| ❌ | ✅ | 只可能来自人工干预 | 视为已完成 | + +--- + +## 8. 当前环境状态(改造后,本机实测) + +| 项 | 状态 | 说明 | +|---|---|---| +| Python / venv | ✅ `.venv`(3.13.5) | 依赖已装齐(含 `audioop-lts` 以支持 3.13 下的 pydub) | +| 数据库 | ✅ 已连通 | SSH 隧道 `localhost:13306` → 远端 MariaDB 10.11;`xwlb_daily` 7973 行、`xwlb_daily_ext` 11460 行 | +| `MySQLDB` | ✅ 已内置 | `mysql_handler.py`,接口与原父项目实现一致 | +| `.env` | ✅ 已精简 | 只留 3 个敏感项(口令/Key);地址、模型等已迁至 `config.yml` | +| `config.yml` | ✅ 已就位 | 含 MySQL 地址、目录、**三个模型**、切分参数、清理开关 | +| ffmpeg | ✅ `/usr/bin/ffmpeg` | 可用 | +| 下载/输出目录 | ✅ 基于项目目录 | 默认 `./xwlb_video`、`./audio_processing`(`config.yml: paths.*`) | +| 中间产物清理 | ✅ 已启用 | 当天成功结束后自动清空 mp3/mp4/wav;实测清掉 15 个文件 / 474.3 MB | +| 版本管理 | ❌ 仍非 git 仓库 | 已提供 `.gitignore`,建议后续 `git init` | + +**端到端验证**:在测试日期 `1900-01-01` 上用裁剪后的真实音频完整跑过「分割 → ASR → 校对 → `xwlb_daily` → DeepSeek 切分 → `xwlb_daily_ext`」,全部成功,测试数据已清理。详见 [BUGS.md 的验证记录](./BUGS.md#本次修复的验证记录实测非推断)。 + +--- + +## 9. 时序图(单日,改造后) + +``` +20:34 抓页面 → 找到 VID 页(失败日期明确跳过) +20:35 yt-dlp 下载 ~100MB mp4 → ffmpeg 抽 mp3(已存在则跳过下载) +20:35 mp3 → input_YYYYMMDD.wav(16k) → 静音切分为 ~11 片 +20:36 chunk_0 ASR(30~45s) → 校对(20~32s;失败或改动数值事实则用原文) → 覆盖写 xwlb_daily ┐ + ... │ 约 10~14 min +21:0x chunk_10 同上 → 删除该分片 ┘ +21:0x 拼接 11 行 news_improve(~6900 字)→ DeepSeek(27~35s) +21:0x 解析 JSON(失败则强化提示词重试一次)→ 删除当日旧记录 → 批量写入 xwlb_daily_ext ×N +21:0x 清理中间产物:mp3/mp4/wav 全部删除(仅当本日全部成功;失败则保留供重跑) +``` +每个分片顺序执行,无并发;单个分片识别失败只影响该片(记 ERROR 且不写空行),其余分片照常入库。 \ No newline at end of file diff --git a/docs/BUGS.md b/docs/BUGS.md new file mode 100644 index 0000000..ba5795b --- /dev/null +++ b/docs/BUGS.md @@ -0,0 +1,363 @@ +# xwlb 项目缺陷清单 + +审查范围:仓库内全部 Python 源码 + `main.log`(2026-06-01 ~ 2026-09-23,34MB,约 46 个有效运行日)。 +结论口径:每条缺陷给出**位置 → 问题 → 影响 → 证据 → 修复方向**。证据栏区分"日志实测"与"代码推断"。 + +严重度定义: +- **P0**:造成数据丢失或数据错误(静默发生、无人察觉) +- **P1**:功能不生效、崩溃、成本浪费、磁盘/日志失控 +- **P2**:工程与可移植性问题(换环境即失败) + +--- + +## 汇总表 + +| # | 严重度 | 位置 | 一句话问题 | 实际影响 | 本次状态 | +|---|---|---|---|---|---| +| B1 | P0 | `audioRead.py:186,191-193,260` | qwen 校对超时(300s)被当成 ASR 失败,整片文本丢弃 | 46 天、57 次,每天约丢 2/11 分片(~1400 字)静默缺口 | ✅ 已修复并实测 | +| B2 | P0 | `audioRead.py:275,289-292` | 校对结果恒为 `None`,回退原文 | 校对功能 0 收益,却贡献了全部 57 次超时与 token 成本 | ✅ 已修复并实测 | +| B3 | P0 | `newsProcess.py:89,119-121` | DeepSeek 返回非纯 JSON(```json 围栏 / 截断)即整批失败 | 22 天 `xwlb_daily_ext` 当天 0 条 | ✅ 已修复(回归测试覆盖) | +| B4 | P0 | `getVideo5.py:69,92,156` | `get_xwlb_video_link()` 返回值错误(只回最后一个 / 未定义 / `[]`) | 取错链接、`UnboundLocalError`、无效 URL 仍进下载 | ✅ 已修复并实测(联网) | +| B5 | P0 | `getVideo5.py:104-146`、`audioRead.py:389-399` | 下载与分片落库无幂等 | 重跑 → 重复下载 + `xwlb_daily` 重复行 → 新闻重复、token 翻倍 | ✅ 已修复并实测 | +| B6 | P1 | `audioRead.py:260`、`getVideo5.py:132-140` | 关键外部调用无 timeout/重试;ffmpeg 输出未接管 | 单点失败无补救;`main.log` 34MB 且无轮转 | 🟡 部分(校对已加超时/重试;ffmpeg 输出已接管;yt-dlp 进度与日志轮转未处理) | +| B7 | P1 | `deepseek.py:257-268` | 降级分支 `return result`(变量错/可能未绑定) | DeepSeek 失败时抛 `NameError`,降级形同虚设 | ✅ 已修复(`return fallback_result`) | +| B8 | P1 | `audioRead.py:422-439` | `__main__` 示例用 3 参数调 4 参数签名 | `python audioRead.py` 必错 | ✅ 已修复(改为 `python audioRead.py [YYYYMMDD]`) | +| B9 | P1 | `audioRead.py:387`、`getVideo5.py:104-146` | 临时文件不清理(chunk/input.wav/mp4) | 单日约 180MB 常驻,磁盘持续增长 | 🟡 部分(分片与按日 WAV 已在 `finally` 清理;mp4/mp3 保留策略待定) | +| B10 | P2 | `mysqlHandle.py:5` | 依赖父项目 `utils.mysql_handler`,本机不存在 | 所有脚本 import 即 `ModuleNotFoundError` | ✅ 已修复(内置 `mysql_handler.py`) | +| B11 | P2 | `getVideo5.py:170,178`、`audioRead.py` | 硬编码 `/home/simon/myquant/djapi/...` | 换机即废(本机已复现) | ✅ 已修复(基于项目目录 + 环境变量可覆盖) | +| B12 | P2 | `env.py:9` | `.env` 定位按老 djapi 目录层级,且不覆盖已有变量 | 指向 `/home/pi/.env`,实际不存在 | ✅ 已修复(项目内 `.env`,支持 `XWLB_ENV_FILE`) | +| B13 | P2 | `newsProcess.py:56`、`newsRedo.py:69` | "已处理"判据为 `COUNT(*) > 5` | 少于 5 条的日期被判定未处理,无限重跑 | ✅ 已修复(存在任意记录即视为已处理,`--force` 可强制) | +| B14 | P2 | 仓库根目录 | 产物入库、非 git 仓库 | 354MB 产物 + 34MB 日志 + 失效 `.pyc` | 🟡 部分(已加 `.gitignore`、清理失效 `__pycache__`;未 `git init`、历史产物未清) | +| B15 | P2 | 全局 | 无 `.env.example`,密钥缺失/失效时报错不指向配置 | 401 当天 33 次报警,排查成本高 | 🟡 部分(已加 `.env.example` 与连接失败明确报错;启动期配置校验未做) | + +--- + +## 本次修复的验证记录(实测,非推断) + +验证环境:项目 venv(Python 3.13.5)+ 真实 DashScope/DeepSeek key + 通过 SSH 隧道(13306)连接生产 MariaDB。 +端到端验证使用**测试日期 `1900-01-01`** 与裁剪后的 5 分钟音频,验证完成后已删除测试数据(`xwlb_daily` 2 行、`xwlb_daily_ext` 3 行,复查均为 0)。 + +| 验证项 | 方法 | 结果 | +|---|---|---| +| B12 配置加载 | 运行任意脚本 | `已加载配置文件 /home/pi/project/xwlb/.env(注入 18 个变量)` | +| B10 DB 层 | `import mysqlHandle` 后直接构造 `MySQLDB()` | 连接成功;故意用错端口时抛出带修复指引的 `RuntimeError`(不再是 `AttributeError`) | +| B8/B1/B2 真实链路 | `process_long_audio()` 处理 5 分钟真实音频 | 2 个分片全部成功;校对**有真实产出**:677→699、635→632 字;`返回 {'chunks': 2, 'ok': 2, 'failed': 0}`;落库 2 行(2011/2079、1885/1880 字),**空白行 0** | +| B5 分片幂等 | 同 `(日期, 分片)` 连续写入 3 次(其中 2 次同键) | 只保留 2 行(子号 0、1)且为最新内容;对照旧逻辑会累积成 3 行 | +| B3 解析健壮性 | `tests/test_news_parse.py`,含日志中真实坏返回 | 15 项全部通过(```json 围栏、夹解释文字、对象包装、混入字符串、缺字段等均可解析;截断/非 JSON 按预期抛错触发重试) | +| B3/B5/B13 切分入库 | 对同一测试日期连续执行 3 次 `news_to_db` | 第 1 次切出 3 条并写入;第 2 次(无 force)正确跳过;第 3 次(force)**替换**旧 3 条 → 仍为 3 条,重复 `sub_id` 组数 = 0 | +| B4 抓取 | 联网解析 20260923/20260924/20260925 三天页面 | 返回类型为 list;09-23、09-24 各解析出 1 条完整版链接;09-25(尚不存在,404)被明确记录并跳过,不再产生空 URL | +| **2026-09-04 生产级实跑** | `getVideo5.py 20260904 20260904` 完整跑一遍(下载 98MB → 抽音 → 11 片 ASR+校对 → 切分入库) | 11/11 分片成功(原空白分片 7 补回 2091 字);切分首轮遇 `Unterminated string`(**正是当初让该日 0 条精编的原因**),第 2 轮强化提示词成功切出 34 条;见下方"事实篡改"一节,修复后复查 0 漂移 | +| 数值事实守卫 | `tests/test_fidelity_guard.py`(13 用例) | 全部通过:`2027→2024`、`十一→九`、`十五五→十四五`、`5.3→5.31`、`300→30` 均拦下;`7↔七`、`2026↔二零二六`、`55%↔百分之五十五` 均放行 | + +> 未验证项:本轮未对 `xwlb_daily` 中既有的 101 行空白记录与 203 天缺失 `xwlb_daily_ext` 做数据修复/回填——那属于数据治理,需你确认后再执行(会消耗 API 额度)。 + +--- + +## P0 详情 + +### B1 qwen 校对超时导致已成功的 ASR 结果被静默丢弃 + +**位置**:`audioRead.py:186`(`transcribe_audio` 内调用 `analyze_and_correct_text`)、`:260`(`Generation.call`,未设 timeout)、`:191-193`(异常兜底 `return ['', '']`)、`:369-399`(`process_long_audio` 落库,无条件插入) + +**问题**:ASR 识别与"qwen 文本校对"被串在同一个 `try` 里。校对调用一旦抛异常(实测全部是 300s 读超时),异常被 `transcribe_audio` 的兜底捕获,函数返回 `['', '']`——**已经识别成功的文本不再返回**。`process_long_audio` 随后照常以空字符串插入一行。 + +**影响**: +- 每个受影响分片丢失约 700 字识别文本,且**没有任何失败标记**(`news_raw=''`、`news_improve=''` 与"正常但短"无法区分); +- 46 个运行日受影响,每天约 2 个分片,累计约 4 万字内容缺口; +- `news_to_db` 拿到的是缺失版全文,DeepSeek 切分出的新闻条数/内容随之缺失,且上游完全无感知。 + +**证据(日志实测)**:`read timeout=300` 共 57 次,**前一行 100% 是"调用通义千问模型进行文本修正"**(校验:`grep -B1 "read timeout=300" | grep -c 调用通义千问` = 57,`grep -c 开始识别音频` = 0),说明超时发生在校对而非识别。分布 2026-06-18 ~ 2026-09-23,覆盖 46 个不同日期,最后一天(09-23)仍在发生,**属存活缺陷**。 + +**修复方向**: +1. 解耦:ASR 成功即先落库 `news_raw`,校对作为独立后续步骤(失败保留原文,不丢数据); +2. 校对调用单独设 `timeout`(dashscope 支持 `timeout=`)+ 有限重试,超时按"放弃校对"处理而非"放弃识别"; +3. 空文本不入库,或入库但显式记录 `status=failed` 供 `news_to_db` 跳过。 + +--- + +### B2 qwen 校对功能实际从未生效 + +**位置**:`audioRead.py:275`(`return response.output.text`)、`:277-299`(`analyze_and_correct_text`) + +**问题**:`text_correction` 返回 `None`(未抛异常),`analyze_and_correct_text:290-292` 判断为 `None` 后回退原文。 + +**影响**: +- 校对本应修错别字/标点,实际 `news_improve == news_raw`,**功能 0 收益**; +- 代价是全部 57 次 300s 超时(B1 的根因)、每个分片 30~300s 的额外耗时,以及 qwen 的全部 token 成本; +- 它同时把"校对"变成了"随机丢分片"的赌局。 + +**证据(日志实测)**:每个分片固定出现 `✓ 文本修正完成` → `WARNING - 文本修正返回 None,使用原始文本` → `原始文本长度: N` / `修正后文本长度: N`(两值恒等;如 09-23 的 798/798、708/708、688/688)。 + +**疑似原因**:`Generation.call` 走 `messages` 时该 SDK 版本的文本取法应为 `response.output.choices[0].message.content`,`.output.text` 为空;或 `max_tokens=30000` 超出模型上限导致 output 为空但 status 仍 200。 + +**修复方向**:修正结果字段取法(或改用 OpenAI 兼容接口);若确认无收益则直接删掉这一步,只保留 ASR。 + +--- + +### B2 补充(2026-09-25 实测发现):校对修好后会**篡改事实** + +把 B2 修好之后校对第一次真正生效,随即在 2026-09-04 的真实数据上发现它**改错了事实**: + +| 内容 | ASR 原文 | qwen 校对后 | 判定 | +|---|---|---|---| +| 条例施行日期 | 自 **2027** 年1月1日起施行 | 自 **2024** 年1月1日起施行 | ❌ 改错(新签署的条例不可能追溯生效) | +| 五年规划 | **十五五** | **十四五** | ❌ 改错(2026 年应为十五五) | +| 东方经济论坛届次 | **第十一**届 | **第九**届 | ❌ 改错(同一天开场提要也作"第十一届") | +| 年份(多处) | 2025 / 2026 | 2023 / 2024 | ❌ 改错 | +| 数量 | 55 / 300亿 | 60 / 30亿 | ❌ 改错 | + +11 个分片中 **8 个**存在数值/序数被改写;这些错误会经 `news_improve` 直接进入最终 `xwlb_daily_ext`(DeepSeek 忠实复制了它的输入)。 + +**影响**:这比"校对不生效"更危险——原文虽是 ASR 结果但事实可信,校对后反而产生**看起来通顺但事实错误**的文本。 + +**已实施的修复**: +1. **数值事实守卫**(`audioRead._number_drift`):中文数字与阿拉伯数字统一归一到数值后比较,若校对前后数值集合与拼接串均不一致,则该分片**整片回退 ASR 原文**;`百分之X` 与 `X%`、`7 ↔ 七`、`2026 ↔ 二零二六` 等纯书写差异不会误拦。 +2. **提示词加固**:明确禁止改动数字/年份/日期/届次/数量/机构名/人名/地名/专有名词。 +3. **开关**:`LLM_CORRECT_ENABLED=0` 可完全关闭校对(直接以 ASR 原文入库)。 +4. 回归测试 `tests/test_fidelity_guard.py`(13 个用例,含上表中的真实篡改样例)。 + +**取舍**:被守卫拦下的分片会连"合理的 ASR 修复"一起放弃(例如同一分片里的 `弗拉迪沃斯。波克` → `符拉迪沃斯托克`、`丁学祥` → `丁薛祥`)。守卫的取向是**宁保留原文,不接受 LLM 改数**。 + +**待你决定**:`LLM_CORRECT_ENABLED` 的默认值。当前默认 `1`(开启+守卫),另一选择是默认 `0`(完全不改写原文,最保守,且每天省下 11 次 qwen 调用)。 + +--- + +### B3 DeepSeek 返回非纯 JSON 时整批切分丢失 + +**位置**:`newsProcess.py:86-89`(提示词与 `deepseek_text`)、`:119-121`(`json.JSONDecodeError` 仅记录后放弃) + +**问题**:依赖 `response_format={"type":"json_object"}` 保证纯 JSON,但模型仍会输出 ```json 围栏或超长截断(`max_tokens=20000`),解析失败后**不重试、不降级**,直接丢弃当天全部文本。 + +**影响**:22 个日期 `xwlb_daily_ext` 当天 0 条记录;由于 B13 的判据,这些日期之后会被反复重跑,而每次重跑又触发 B5 的重复写入。 + +**证据(日志实测)**:`JSON解析失败` 22 次,覆盖 22 个不同日期(2026-06-03 ~ 2026-09-14);报错内容含 `Expecting value: line 1 column 1`(前后被 ``` 包裹,日志里可见 `DeepSeek API返回内容: ```json`)与 `Unterminated string starting at: line 80 column 21`(截断)。另有 1 次 `插入数据库失败: string indices must be integers`,说明 `news_list` 元素有时是字符串而非 dict(`news["news_id"]` 直接炸)。 + +**修复方向**:剥离围栏后再解析 → 失败则重试一次(提高 max_tokens / 降低 temperature)→ 仍失败则按空行/段落降级切分;对 `news_list` 元素做类型校验与字段缺失兜底。 + +--- + +### B4 `get_xwlb_video_link()` 返回值错误 + +**位置**:`getVideo5.py:34-69`(构建 `video_links` 但 `return video_url`)、`:86-94`(`get_all_video_links`)、`:149-171`(`process_videos`) + +**问题**:三处叠加。 +1. `:69` 返回的是循环里最后一次赋值的**单个 URL 字符串**,`video_links` 列表被丢弃 → 当天若有多个匹配链接只取最后一个; +2. 请求异常路径 `return []`(`:28,:31`),但 `get_all_video_links:92` 仍包装成 `{"url": [], "date": ...}`; +3. `process_videos:156` 用 `if sub_url:` 判断——非空 dict 恒为真,**错误值不会被拦截**,`[]` 会被送进 yt-dlp; +4. 若一个 a 标签都没匹配到,`video_url` 从未赋值 → `UnboundLocalError`(不是被捕获的异常类型)。 + +**影响**:抓取失败/页面改版时不是"跳过并告警",而是带着无效 URL 继续下载,或用错链接下载到非新闻联播内容;幂等性判断(B5)也因此失效。 + +**证据**:代码推断(`get_xwlb_video_link` 内 `video_links` 变量在 `:34` 定义、`:61` 追加、`:69` 未被返回,可静态确认)。 + +**修复方向**:`return video_links`;调用方展开列表并对空值 `continue`;`if sub_url.get('url')` 显式判空。 + +--- + +### B5 下载与分片落库无幂等 + +**位置**:`getVideo5.py:104-146`(`download_and_extract_audio` 不检查 mp4/mp3 是否已存在)、`audioRead.py:389-399`(无条件 `insert_data`) + +**问题**:`newsRedo.py:79-108` / `main_videos.py:23-58` 判断"是否需要处理"只看 `xwlb_daily` 有没有当天的行。而重跑会**追加**一整组 `daily_sub_id = 0..N` 的新行(`nid` 自增,`(news_days, daily_sub_id)` 无唯一约束)。 + +**影响**: +- `get_news_improve_by_date`(`newsProcess.py:61-65`)按 `daily_sub_id` 排序拼接,会把同一天的文本**重复拼两遍**送进 DeepSeek → 切分结果重复、token 翻倍、可能超出上下文; +- 重复下载 100MB 级 mp4(B9 磁盘问题同步放大); +- 与 B13 组合会形成"重跑 → 重复 → 再重跑"的循环。 + +**证据**:代码推断(无唯一索引、无存在性检查)。 + +**修复方向**:`(news_days, daily_sub_id)` 加唯一索引 + `INSERT ... ON DUPLICATE KEY UPDATE`;下载前检查目标 mp4/mp3 是否存在且非空;`daily_sub_id` 用分片序号而非可变下标。 + +--- + +## P1 详情 + +### B6 外部调用无超时/重试,子进程输出未接管 + +**位置**:`audioRead.py:258-266`(`Generation.call` 无 timeout、无重试)、`:168-186`(`Recognition.call` 单次失败即放弃)、`getVideo5.py:132-140`(ffmpeg `check=True` 但 `stdout=None, stderr=None`)、`:123`(yt-dlp 进度 hook 直接 `print`) + +**影响**: +- 单分片 ASR 失败无任何补救,直接进入下一片; +- ffmpeg 的完整 banner(版本、编译参数、流信息)与 yt-dlp 的逐帧进度全部落到 stdout,`main.log` 已达 34MB 且**无 logrotate**;一次失败排查需要在数万行进度里翻找; +- 超时阈值 300s 是 SDK 默认,与业务预期(每天 51 分钟跑完)不匹配。 + +**修复方向**:统一 timeout + 指数退避重试;ffmpeg 用 `stdout=subprocess.DEVNULL, stderr=subprocess.PIPE` 并只在失败时打印尾部;日志加 `RotatingFileHandler` 或把进度输出降级为 DEBUG。 + +--- + +### B7 `deepseek.py` 降级分支变量写错 + +**位置**:`deepseek.py:241-268` + +**问题**:`try` 中 `result = api_client.process_text(...)`;`except` 里调用 `process_text_with_fallback(...)` 得到 `fallback_result`,却 `return result`。异常路径下 `result` 未绑定 → `NameError`(若部分成功则返回错误内容)。 + +**影响**:本应"降级返回原文"的兜底变成二次异常,被 `newsProcess.py:122` 的宽 `except` 吞成一行 `插入数据库失败`,与 B3 的静默失败叠加。 + +**证据**:代码推断(`deepseek.py:261-268`)。**附带**:`deepseek.py:271` 的 `deepseek_text()` 缺必填参数,`python deepseek.py` 直接 `TypeError`。 + +**修复方向**:`return fallback_result`;`__main__` 补参数或改成 CLI。 + +--- + +### B8 `audioRead.py` 的 `__main__` 示例已过期 + +**位置**:`audioRead.py:422-439` + +**问题**:`process_long_audio(mp3_path, prompt, output_folder)` 传 3 个位置参数,而签名是 `(mp3_path, output_folder, date_str)` → `prompt` 被当作输出目录,`date_str` 缺失。 + +**影响**:任何人按文件底部示例单独调试都会失败,且会把中文提示词当目录名创建。 + +**修复方向**:改为 `process_long_audio(mp3_path, output_folder, date_str)`,或直接删掉示例。 + +--- + +### B9 临时文件与产物不清理 + +**位置**:`audioRead.py:387`(仅成功分支 `os.remove(chunk_path)`)、`:352-354`(`input.wav` 不删)、`getVideo5.py:113-114`(mp4/mp3 长期保留) + +**影响**:单日约 104MB mp4 + 24MB mp3 + 55MB wav ≈ 180MB;仓库当前已含 `xwlb_video/` 299MB、`audio_processing/input.wav` 55MB。长期运行必然打满磁盘,且在失败分支 chunk 会残留。 + +**修复方向**:处理完即删中间产物(或按保留天数清理);mp3 作为可重建产物不必长期保存;`finally` 中清理 chunk。 + +--- + +## P2 详情 + +### B10 `utils.mysql_handler` 缺失(当前无法运行的首要原因) + +**位置**:`mysqlHandle.py:4-5` + +**问题**:`sys.path.insert(0, )` 后 `from utils.mysql_handler import MySQLDB`,而 `/home/pi/project/utils/` 不存在。 + +**影响**:所有脚本(包括仅查库的 `newsRedo.py`)**import 阶段即失败**:已在本机复现 `ModuleNotFoundError: No module named 'utils'`。 + +**证据(日志实测,反向线索)**:日志中 `查询失败: local variable 'cursor' referenced before assignment` 说明该模块在连接失败时自身还有未初始化变量缺陷——重写时应一并避免。 + +**修复方向**:在项目内提供自包含的 `mysql_handler.py`,保持 `MySQLDB()` / `query_data(table, columns, where, params)` / `insert_data(table, data)` / `close()` 四个接口不变,配置从环境变量读取。 + +--- + +### B11 硬编码绝对路径 + +**位置**:`getVideo5.py:170`(`/home/simon/myquant/djapi/api/video/xwlb_video`)、`:178`(同前缀 `audio_processing`)、`audioRead.py:353` 的输出目录同样来自该前缀 + +**影响**:迁移到任何其他机器都不可用;本机 `pwd` 为 `/home/pi/project/xwlb`,正确位置应是仓库内的 `xwlb_video/` 与 `audio_processing/`(数据也确实已放在那里)。 + +**修复方向**:以 `Path(__file__).parent` 为基准,或由 `env` 提供 `XWLB_VIDEO_DIR` / `XWLB_AUDIO_DIR`。 + +--- + +### B12 `.env` 定位与优先级 + +**位置**:`env.py:9`(`parent.parent.parent / '.env'` → `/home/pi/.env`)、`:23`(`if key not in os.environ` 不覆盖) + +**影响**:按老 `djapi/api/video/` 三层结构推导,本项目只有一层,指向错误位置且文件不存在 → 密钥全部缺失;"不覆盖已有变量"的行为未文档化,容易与容器注入的变量混淆。 + +**修复方向**:改为项目根/`xwlb` 目录下的 `.env`,并支持 `XWLB_ENV_FILE` 覆盖;在文档中写明优先级(环境变量 > `.env`)。 + +--- + +### B13 "已处理"判据用行数阈值 + +**位置**:`newsProcess.py:49-57`、`newsRedo.py:62-71` + +**影响**:某天若只切出 ≤5 条新闻(短节目/切分异常),会被永久判定为"未处理"并在每次补漏时重跑;配合 B5 造成重复数据。 + +**修复方向**:以"是否已经跑过切分"的显式状态为准(如 `xwlb_daily_ext` 存在任意行即视为已处理,或新增处理状态表/字段)。 + +--- + +### B14 仓库卫生 + +**位置**:仓库根目录 + +**问题**:`main.log` 34MB、`xwlb_video/` 299MB、`audio_processing/input.wav` 55MB 与源码混放;`__pycache__/` 内含 `ai.cpython-310.pyc`、`parsem3u8.cpython-310.pyc` 等**已删除模块**的产物(3.10/3.11,与当前 3.13 不符);目录**不是 git 仓库**,无变更历史可回溯。 + +**修复方向**:加 `.gitignore`(`*.mp4 *.mp3 *.wav main.log __pycache__/`)、`git init` 纳入版本管理、清理失效 `__pycache__`。 + +--- + +### B15 配置与密钥错误不可诊断 + +**位置**:全局(`deepseek.py:21-23` 只 warning、`audioRead.py:166` 直接赋值空串) + +**影响**:`DASHSCOPE_API_KEY` 无效/缺失时表现为 2026-06-01 那天连续 33 条 `401 Unauthorized` + 11 个分片全废,但报错信息不指向"检查 `.env`";`DeepSeekAPI` 缺 key 仅 warning,随后才在调用处抛异常。 + +**修复方向**:提供 `.env.example`;启动时做一次配置校验(缺 key 直接 fail-fast 并打印期望的文件路径)。 + +--- + +## 本轮新发现(2026-09-25,B16~B19) + +### B16(严重 · 静默换模型)兜底分支偷偷用另一个模型 + +**位置**:`deepseek.py` `process_text_with_fallback` +**问题**:兜底调用 `self.process_text(prompt, text, system_prompt)` **未传 model**,于是用函数签名里的硬编码默认值 `deepseek-chat`。 +**影响**:配置里的模型名无效(如 `deepseek-v4.1-flash`)时,主调用 400 报错、兜底却用另一个模型成功返回,**"按配置运行"看起来完全正常,实际配置从未生效**——排障方向会被彻底带偏。 +**修复**:兜底显式传入同一模型与参数;切分环节改为 `use_fallback=False`(它自带重试与解析校验),配置错误直接暴露。现已验证:模型名写错立刻报 `The supported API model names are ...`。 + +### B17(严重 · 数据完整性)重试判定会把"半天"当"一天" + +**位置**:`scripts/day_status.py`(本轮新增) +**问题**:只看"库里有没有分片"就判断"只缺切分"。识别中途卡死同样会留下部分分片, +于是重试走"只重跑切分",把半天内容切分入库并标记为完成。 +**影响**:静默产生**不完整的一天**,且不会报错;事后极难发现(那天看起来是正常的)。 +**修复**:识别全部完成后写标记 `state/.asr_complete_<日期>`(放 `state/`、**不随清理删除**, +否则无法与"识别只跑了一半但切分照样写了 ext"区分),`day_status.py` 只有**同时**满足 +"有分片 + 有标记"才允许只重跑切分,否则重跑全链路; +并在 `process_videos` 中改成**识别不完整就不执行切分**,不发布"看起来完整"的半天数据。 + +### B18(高 · 推理模型截断/空回复)max_tokens 未按推理模型调整 + +**位置**:`config.yml` `llm_split.max_tokens`(原 20000 / 重试 32000) +**问题**:`deepseek-flash` 是推理模型,reasoning token 实测在 14k~24k 间波动且**计入 max_tokens**。 +20000 时实测出现两种故障:输出截断在 JSON 中途(`finish_reason=length`)、 +或推理耗尽全部额度导致 **content 为空**(报"LLM 响应为空")。 +**影响**:切分随机失败或写入被截断的新闻;每晚概率性发生。 +**修复**:上限提到 60000(重试 80000)——按实际产出计费,给足上限不额外花钱; +`timeout` 60→180 秒(2 万 token 级别输出需要更久)。 + +### B19(高 · 可用性)ASR 的 `Recognition.call` 无超时,长连接悬挂会卡死整晚 + +**位置**:`audioRead.transcribe_audio` → dashscope `Recognition.call` +**问题**:SDK 内部 `for part in responses` 无限等待 websocket 响应,**没有任何超时参数** +(签名无 timeout,内部 `while True`)。实测 2026-09-24 重跑时卡在 chunk 3 之后: +进程 0 CPU、`futex_do_wait`、无新日志,**15 分钟以上无进展**。 +**影响**:手动跑会一直挂着;定时任务靠外层上限兜底,但会白白耗掉一次尝试。 +**缓解**:`scripts/run_daily.sh` 单次尝试 40 分钟上限 + 21:30/22:00 重试; +systemd `TimeoutStartSec=10800`。**根治需要给 ASR 加子进程看门狗**(列入待办)。 + +--- + +## 后续待办(本轮新增) + +7. **ASR 看门狗**(对应 B19):把 `Recognition.call` 放进子进程执行并加超时, + 超时终止子进程、该分片记失败后**继续下一片**,而不是整晚挂死。 +8. **回补校对失效期**(见 `REPORT_raw_vs_improve.md`):2026-06-15~2026-09-23 的 + `news_improve` 实际等于 ASR 原文(2026-07/08 为 100% 未改动)。校对现已修复且 + 关掉思维链后成本降到约 1/13,回补约 100 天 ≈ 1M token,需确认是否执行。 +9. **重试跳过已识别分片**:当前重试会重跑全部 ASR(识别已完成的片也重做)。 + 加"已入库分片跳过"可让重试只补缺失片,需注意别让 `newsRedo.py` 的强制重跑被静默跳过。 + +--- + +## 后续待办(本轮未做) + +1. **数据治理**:`xwlb_daily` 现存 101 行空白记录(可识别、可重跑修复);725 天原文里 203 天没有 `xwlb_daily_ext`——回填需重新调用 DeepSeek,属于有成本的操作,需你确认范围后再执行。 +2. **B6 收尾**:yt-dlp 进度 hook 仍直接 `print`;`main.log` 无轮转(建议 `RotatingFileHandler` 或由外部 logrotate 管理)。 +3. **B9 收尾**:`xwlb_video/*.mp4` 与 `*.mp3` 是否保留需定策略(mp3 可重跑,mp4 价值较低)。 +4. **B14 收尾**:`git init` 纳入版本管理,清理历史大文件(34MB 日志、299MB 视频)。 +5. **B15 收尾**:启动期做一次配置校验(缺 key / 连不上库时 fail-fast 并打印期望文件路径)。 +6. **可选加固**:给 `xwlb_daily` 加唯一索引 `(news_days, daily_sub_id)`、`xwlb_daily_ext` 加 `(news_date, sub_id)`。当前代码已用「先删后插」实现幂等,索引属额外保险;注意建索引前必须先清理既有重复行(如 2026-06-15 全量重复 2 份、2026-06-18 ext 重复 3 份)。 + +## 修复优先级建议(原始评估,供回看) + +1. **B2 + B1 一起做**(改/删校对步骤,保住已识别文本)—— 直接消除最大的静默数据缺口,同时省下每分片 30~300s 与全部 qwen 成本。 +2. **B3**(JSON 解析健壮化 + 重试)—— 消除 22 天的整天数据缺失。 +3. **B10 + B11 + B12**(可移植化)—— 让项目在任何机器可运行,是继续验证与修复的前提。 +4. **B5 + B13**(幂等 + 判据)—— 防止修复过程中把重复数据写进生产库。 +5. **B4**(抓取返回值)—— 抓取链路的正确性兜底。 +6. B6~B9、B14、B15 —— 稳定性与工程卫生。 + +> 注:`main.log` 中 `'str' object has no attribute 'append'`(22 次)与 `object of type 'NoneType' has no len()`(11 次)**仅出现在 2026-06-01 与 06-16**,当前 `audioRead.py` 已分别通过"返回 list"与"`None` 回退原文"修掉,故未单列;但它们与 B1/B2 同源,重写校对步骤时不要再引入同类返回类型不一致的写法。 \ No newline at end of file diff --git a/docs/REPORT_raw_vs_improve.md b/docs/REPORT_raw_vs_improve.md new file mode 100644 index 0000000..3a69ec4 --- /dev/null +++ b/docs/REPORT_raw_vs_improve.md @@ -0,0 +1,142 @@ +# news_raw vs news_improve 评估报告 + +样本:`xwlb_daily` 7981 个分片,覆盖 727 天(2024-09-26 ~ 2099-01-01);`xwlb_daily_ext` 11494 条 + +## 1. 校对到底改了什么 + +| 分类 | 分片数 | 占比 | 平均改动字符数 | 含义 | +|---|---|---|---|---| +| identical | 1094 | 13.7% | 0.0 | 完全没改 | +| punct_only | 1 | 0.0% | 0.0 | 只改标点/空白/断句 | +| numeral_only | 1 | 0.0% | 2.0 | 只改数字写法或标点 | +| word_change | 6885 | 86.3% | 38.5 | **真的改了字词** | + +「真的改了字词」的 6885 个分片里,**排除标点与数字后**的平均改动量只有 **29.7 字**(占分片平均长度 718 字的 4.1%) + +改动最频繁的字符(raw 有而 improve 没有 → improve 新增): + +| 被替换掉的字符 | 次数 | 新增的字符 | 次数 | +|---|---|---|---| +| `。` | 20087 | `⏎` | 38761 | +| `␠` | 19329 | `”` | 16765 | +| `号` | 10326 | `“` | 16763 | +| `的` | 8440 | `,` | 12433 | +| `,` | 3649 | `日` | 11487 | +| `了` | 3457 | `、` | 9422 | +| `这` | 2780 | `;` | 7558 | +| `个` | 2666 | `␠` | 5625 | +| `是` | 2640 | `的` | 4465 | +| `到` | 2478 | `《` | 4254 | + +- 校对**实际生效**的分片:6887/7981(86.3%);完全没动的:1094/7981(13.7%) +- 全天 11 个分片**全都没被改**的天数:98/727 + +### 可读性指标(越大越接近正式书面语) + +| 指标 | 原文 news_raw | 校对后 news_improve | 变化 | +|---|---|---|---| +| 标点密度(每百字标点数) | 8.52 | 9.81 | +1.28 | +| 书名号《总数 | 0 | 4254 | +4254 | +| 总字数 | 5675790 | 5810634 | +134844 | + +### 「真的改了字词」的抽样(判断是修错字还是改写) + +- **2024-09-26 第0片**(改动约 31 字) + - raw: 各位观众晚上好晚上好,今天是9月26号星期四,农历8月24,欢迎收看新闻联播节目。首先为您介绍今天节目的主要内容。中共中央政治局召开会议,分析研究当前经济形势和经济工作,中共中央总书记习近平主持会议。习近平给中国传媒大学全体师生回信,强调突 + - imp: 各位观众晚上好,今天是9月26日,星期四,农历八月二十四,欢迎收看新闻联播节目。首先为您介绍今天节目的主要内容:中共中央政治局召开会议,分析研究当前经济形势和经济工作,中共中央总书记习近平主持会议;习近平给中国传媒大学全体师生回信,强调突出 + +- **2024-09-26 第1片**(改动约 10 字) + - raw: 会议强调,要加大财政货币政策逆周期调节力度,保证必要的财政支出,切实做好基层三保工作。要发行使用好超长期特别国债和地方政府专项债,更好发挥政府投资带动作用。要降低存款准备金率,实施有力度的降息。要促进房地产市场止跌回稳,对商品房建设要严控增 + - imp: 会议强调,要加大财政货币政策逆周期调节力度,保证必要的财政支出,切实做好基层“三保”工作。要发行使用好超长期特别国债和地方政府专项债券,更好发挥政府投资的带动作用。要降低存款准备金率,实施有力度的降息。要促进房地产市场止跌回稳,对商品房建设 + +### 按月的「完全没改」比例(用于识别校对失效的历史区间) + +| 月份 | 分片数 | 完全没改 | 占比 | +|---|---|---|---| +| 2024-09 | 61 | 0 | 0% | +| 2024-10 | 322 | 0 | 0% | +| 2024-11 | 321 | 0 | 0% | +| 2024-12 | 334 | 0 | 0% | +| 2025-01 | 327 | 0 | 0% | +| 2025-02 | 289 | 0 | 0% | +| 2025-03 | 363 | 0 | 0% | +| 2025-04 | 332 | 0 | 0% | +| 2025-05 | 324 | 0 | 0% | +| 2025-06 | 305 | 0 | 0% | +| 2025-07 | 334 | 0 | 0% | +| 2025-08 | 325 | 0 | 0% | +| 2025-09 | 326 | 0 | 0% | +| 2025-10 | 359 | 0 | 0% | +| 2025-11 | 334 | 0 | 0% | +| 2025-12 | 330 | 0 | 0% | +| 2026-01 | 340 | 0 | 0% | +| 2026-02 | 309 | 0 | 0% | +| 2026-03 | 384 | 0 | 0% | +| 2026-04 | 334 | 0 | 0% | +| 2026-05 | 347 | 0 | 0% | +| 2026-06 | 343 | 165 | 48% | +| 2026-07 | 344 | 344 | 100% | +| 2026-08 | 331 | 331 | 100% | +| 2026-09 | 261 | 254 | 97% | +| 2099-01 | 2 | 0 | 0% | + +## 2. token 开销:现状 vs 两个替代方案 + +统计口径:523 天同时有原文与精编的天数,取日均字符数(≈token 数,中文 1 字≈1 token) + +| 方案 | 输入 | 输出 | 合计/天 | 相对现状 | +|---|---|---|---|---| +| A 现状(校对 + 切分 两次调用) | 15777 | 15646 | **31423** | — | +| B 合并进切分(一次调用同时校对+切分) | 7787 | 7657 | **15444** | 省 51% | +| C 关掉校对(直接用 ASR 原文切分) | 15444 | 7657 | **15444** | 省 51% | + +其中校对这一次调用本身消耗 输入 7787 + 输出 7990 = **15777 token/天**,占现状总开销的 50%。 + +按 523 天累计:校对一项约消耗 8.25 M token。 + +--- + +## 3. 实测 token 开销(真实 API 返回的 usage,不是估算) + +用**你当前配置的模型**各跑一次真实调用,读 `usage` 字段(2026-09-23 / 2026-06-08 数据): + +| 环节 | 模型 | 输入 | 输出 | 其中推理 | 说明 | +|---|---|---|---|---|---| +| 校对(思考开) | `qwen3.8-flash` | 487 | 5,932 | 5,616 | 545 字文本 | +| 校对(思考**关**) | `qwen3.8-flash` | 451 | **312** | — | 同文本,输出只差一个"的"字 | +| 校对(旧配置对照) | `qwen3.7-max` | 449 | 2,338 | 2,025 | 原来也在做推理 | +| 切分(思考开) | `deepseek-flash` | 4,206 | 19,077 | 14,426 | 全天 6,940 字 | +| 切分(思考关) | `deepseek-flash` | 4,182 | 4,656 | — | **但正文覆盖率掉到 95.34%** | + +推算到每天(11 个分片 + 1 次切分): + +| 方案 | 每天 token | 相对现状 | +|---|---|---| +| **现状**(校对思考开 + 切分) | ≈ 100,000 | — | +| **建议**(校对思考关 + 切分思考开) | ≈ **33,000** | **省 67%** | +| 合并(校对+切分一次调用,思考开) | ≈ 24,000 | 省 76% | +| 关掉校对(只切分) | ≈ 23,000 | 省 77% | + +## 4. 三个方案的实测对比(同一天 2026-09-23) + +| 指标 | 现状两次调用 | 合并成一次 | 只切分 | +|---|---|---|---| +| 条数 | 27 | 27 | — | +| 正文覆盖率 | **100%** | 99.02% | 95.34%(思考关时) | +| 书名号《 | 0(该日数据是失效期产物) | 3 | 0 | +| 引号 | 0 | 15 | 0 | +| **数值事实漂移** | 无(有按分片守卫) | **有**:`44.9→44.93`、丢失 `3`、多出 `1000`/`5` | — | +| 按分片回退保护 | ✅ 有 | ❌ 无(只能整天丢弃) | ❌ 无 | + +## 5. 结论 + +1. **校对有必要,但只值"关掉思考"这一刀。** 它的真实产出是标点与格式: + 引号 33,524 次、书名号《》4,250 次、日期归一("9月26号"→"9月26日")10,322 次、 + 段落换行 38,750 次——ASR 原文**从来不会**产生书名号和引号。 +2. **不要把 correct 合并进 split。** 收益只剩 ~27%(因为校对关思考后已经很便宜), + 代价是失去按分片的数值守卫,且实测确实把 `44.9` 改成了 `44.93`、丢了数字。 + 数值事实一旦被改,事后无法察觉——这正是 `docs/BUGS.md` B2 那类事故。 +3. **切分环节不要关思考**:省 4 倍 token,但正文覆盖率 99.67% → 95.34%、条数 27 → 20。 +4. **注意历史欠账**:2026-07 与 2026-08 是 **100% "完全没改"**(校对失效期), + 2026-06 有 48%。也就是说 2026-06 中旬至 2026-09 的 `news_improve` 其实等于 ASR 原文, + 并没有被校对过。现在校对已修复且成本降低约 13 倍,是否有必要回补这段,见 README「后续可选」。 diff --git a/env.py b/env.py new file mode 100644 index 0000000..98893ae --- /dev/null +++ b/env.py @@ -0,0 +1,62 @@ +"""敏感配置加载器(.env) + +职责边界(2026-09-25 重构): +- 本文件只负责**加载**敏感配置:数据库口令、各供应商 API Key。 +- 校验(缺哪些、密钥变量叫什么)在 `config.required_secrets()` / `config.check_secrets()`, + 因为密钥变量名由 config.yml 的 `endpoints.*.api_key_env` 决定(换供应商时会变)。 +- 非敏感配置(数据库地址、目录、模型名、**base url 接入点**、参数、清理开关)见 config.yml / config.py。 + +`.env` 查找顺序:XWLB_ENV_FILE → 项目目录内 .env → 上一级 / 上两级 .env(兼容旧 djapi 布局)。 +已存在的进程环境变量优先,不会被 .env 覆盖。 +""" +import logging +import os +from pathlib import Path + +logger = logging.getLogger(__name__) + +# 本文件所在目录即项目根(xwlb/) +BASE_DIR = Path(__file__).resolve().parent + + +def _candidate_env_files(): + """按优先级返回候选 .env 路径""" + override = os.getenv('XWLB_ENV_FILE') + if override: + yield Path(override).expanduser() + return + yield BASE_DIR / '.env' + # 兼容旧布局:xwlb/ 的上一级或上两级 + yield BASE_DIR.parent / '.env' + yield BASE_DIR.parent.parent / '.env' + + +def _load_dotenv(): + """从候选路径中第一个存在的 .env 加载环境变量(不覆盖已有)""" + for dotenv_path in _candidate_env_files(): + if not dotenv_path.is_file(): + continue + count = 0 + with open(dotenv_path, encoding='utf-8') as f: + for line in f: + line = line.strip() + if not line or line.startswith('#') or '=' not in line: + continue + key, _, value = line.partition('=') + key = key.strip() + value = value.strip().strip('"').strip("'") + if key and key not in os.environ: + os.environ[key] = value + count += 1 + logger.info("已加载敏感配置 %s(注入 %d 个变量)", dotenv_path, count) + return dotenv_path + logger.debug("未找到 .env 文件,仅使用进程环境变量") + return None + + +LOADED_ENV_FILE = _load_dotenv() + + +def env_file() -> Path: + """当前使用的 .env 路径(不存在时返回期望路径)""" + return LOADED_ENV_FILE or (BASE_DIR / '.env') \ No newline at end of file diff --git a/getVideo5.py b/getVideo5.py new file mode 100644 index 0000000..5225539 --- /dev/null +++ b/getVideo5.py @@ -0,0 +1,288 @@ +import requests +from bs4 import BeautifulSoup +import re,os,subprocess +from datetime import timedelta, date +import yt_dlp +import env +import config +from audioRead import * +from cleanup import maybe_cleanup_after_run +from newsProcess import STATUS_FAILED, news_to_db +import logging + +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + +def get_xwlb_video_link(url): + """ + 从央视网新闻联播页面抓取历史完整版视频链接 + """ + headers = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36', + 'Referer': 'https://tv.cctv.com/' + } + + try: + response = requests.get(url, headers=headers, timeout=10) + response.encoding = 'utf-8' + if response.status_code != 200: + logger.error(f"请求失败,状态码: {response.status_code}") + #print(f"请求失败,状态码: {response.status_code}") + return [] + except Exception as e: + logger.error(f"请求异常: {e}") + return [] + + soup = BeautifulSoup(response.text, 'html.parser') + video_links = [] + + # 查找所有包含“完整版《新闻联播》”的链接 + # 方法1: 查找包含 完整版《新闻联播》 的 a 标签 + for a_tag in soup.find_all('a', href=True): + # 检查文本中是否包含“完整版”和“新闻联播” + title_text = a_tag.get_text(strip=True) + inner_html = str(a_tag) + + # 判断是否是“完整版《新闻联播》”的链接 + if ('完整版' in title_text and '新闻联播' in title_text) or \ + (re.search(r']*>完整版\s*《新闻联播》', inner_html)): + + video_url = a_tag['href'] + # 提取日期信息(从标题或链接中) + date_match = re.search(r'\d{8}', title_text) + if not date_match: + # 从链接中提取日期,如 /2025/09/25/...VID...250925.shtml + date_match = re.search(r'/(\d{4})/(\d{2})/(\d{2})/', video_url) + if date_match: + year, month, day = date_match.groups() + date_str = f"{year}{month}{day}" + else: + date_str = "未知日期" + else: + date_str = date_match.group() + + video_links.append({ + 'date': date_str, + 'title': title_text.strip(), + 'url': video_url, + 'page_url': url + }) + logger.info(f"✅ 找到新闻联播完整版: {date_str} -> {video_url}") + + # 原实现误返回循环内的单个 video_url(且无匹配时该变量未定义会 UnboundLocalError) + return video_links + +def xwlb_urls(start: str, end: str): + """ + start/end 格式 '20240925' + 返回列表,如 ['https://tv.cctv.com/lm/xwlb/day/20240925.shtml', ...] + """ + d0 = date(int(start[:4]), int(start[4:6]), int(start[6:8])) + + d1 = date(int(end[:4]), int(end[4:6]), int(end[6:8])) + urls = [] + for n in range((d1 - d0).days + 1): + day = d0 + timedelta(days=n) + urls.append({"url": f"https://tv.cctv.com/lm/xwlb/day/{day:%Y%m%d}.shtml", "date": f"{day:%Y%m%d}"}) + #print(urls) + return urls + +def get_all_video_links(start: str, end: str): + """逐日解析完整版视频链接,返回 [{'url':..., 'date': 'YYYYMMDD'}, ...] + + 原实现把 get_xwlb_video_link() 的返回值(单个 URL 或空列表)直接包成 dict, + 导致空链接也会进入下载流程;现改为按天取第一条有效链接并显式跳过失败日期。 + """ + base_urls = xwlb_urls(start, end) + video_urls = [] + for item in base_urls: + links = get_xwlb_video_link(item['url']) + if not links: + logger.error(f"❌ {item['date']} 未解析到「完整版《新闻联播》」链接,跳过: {item['url']}") + continue + if len(links) > 1: + logger.warning(f"⚠️ {item['date']} 解析到 {len(links)} 条完整版链接,取第一条: {links[0]['url']}") + video_urls.append({"url": links[0]['url'], "date": item['date']}) + + return video_urls + +''' +get_xwlb_video_link() 方法获得的url, +urls like: https://tv.cctv.com/2024/10/30/VIDEUlPz1Qusy41JFQj3LMLd241030.shtml +通过yt-dlp下载视频,保存为mp4文件,并用ffmpeg提取音频为mp3文件,文件名使用url 的日期部分,如上面的url应保存为 20241030.mp4 和 20241030.mp3 +文件保存路径为当前目录下的 xwlb_video 文件夹,若不存在则创建。 +''' + + +def download_and_extract_audio(video_url,date_str,download_dir): + """ + 使用yt-dlp下载视频并提取音频;已存在音频时跳过(幂等,支持重跑) + 返回: mp3 路径(成功)或 None(失败) + """ + # 从URL中提取日期 + + if not video_url: + logger.error(f"❌ {date_str} 视频链接为空,跳过下载") + return None + + download_dir = os.fspath(download_dir) + os.makedirs(download_dir, exist_ok=True) + + # 构建文件路径 + mp4_path = os.path.join(download_dir, f"{date_str}.mp4") + mp3_path = os.path.join(download_dir, f"{date_str}.mp3") + + # 幂等:音频已抽取过就不再重复下载(原实现每次重跑都重新下载 ~100MB) + if os.path.exists(mp3_path) and os.path.getsize(mp3_path) > 0: + logger.info(f"⏭️ {date_str} 音频已存在,跳过下载: {mp3_path}") + return mp3_path + + try: + # 使用yt-dlp库下载视频 + logger.info(f"📥 开始下载 {date_str} 的视频...") + # 配置yt-dlp选项 + ydl_opts = { + 'outtmpl': mp4_path, + 'format': 'best[ext=mp4]/best', + 'progress_hooks': [lambda d: print(f"\r📥 下载进度: {d.get('_percent_str', 'N/A').strip()} | {d.get('_speed_str', 'N/A').strip()} | 已下载: {d.get('_downloaded_bytes_str', 'N/A')}", end='') if d['status'] == 'downloading' else None], + } + + with yt_dlp.YoutubeDL(ydl_opts) as ydl: + ydl.download([video_url]) + logger.info(f"📥 下载 {date_str} 完成") + + # 使用ffmpeg提取音频 + logger.info(f"🎵 开始提取 {date_str} 的音频...") + # 原实现 stdout=None/stderr=None 会把 ffmpeg 完整 banner 灌进日志(日志已 34MB), + # 现仅在失败时记录 stderr 尾部 + proc = subprocess.run([ + "ffmpeg", + "-i", mp4_path, + "-c:a", "libmp3lame", # 明确指定MP3编码器 + "-q:a", "0", + "-map", "a", + mp3_path, + "-y" # 覆盖已存在文件 + ], stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, text=True) + if proc.returncode != 0: + raise RuntimeError(f"ffmpeg 提取音频失败(returncode={proc.returncode}): {(proc.stderr or '')[-500:]}") + logger.info(f"🎵 提取 {date_str} 音频完成") + + logger.info(f"✅ 成功处理 {date_str}: {mp4_path}, {mp3_path}") + return mp3_path + + except Exception as e: + logger.error(f"❌ 处理 {date_str} 时发生异常: {e}") + return None + +# 在get_all_video_links函数后添加调用代码 +def process_videos(start_date, end_date): + """ + 处理指定日期范围内的所有视频 + + 返回值(供定时任务判断成败): + dict: {'total': 待处理天数, 'ok': 全链路成功天数, 'failed': 失败天数} + total=0 表示一天都没抓到链接,按失败处理(页面可能还没发布)。 + """ + video_urls = get_all_video_links(start_date, end_date) + if not video_urls: + logger.error(f"❌ {start_date}~{end_date} 未获取到任何视频链接,流程结束") + return {'total': 0, 'ok': 0, 'failed': 0} + + summary = {'total': 0, 'ok': 0, 'failed': 0} + + # 目录来自 config.yml 的 paths.*(相对项目根,可写绝对路径) + download_dir = config.video_dir() + output_folder = config.audio_dir() + + for sub_url in video_urls: + date_str = sub_url.get('date') + url = sub_url.get('url') + if not url or not date_str: + logger.error(f"❌ 跳过无效条目: {sub_url}") + summary['failed'] += 1 + continue + + summary['total'] += 1 + print("=" * 80) + mp3_path = download_and_extract_audio(url, date_str, download_dir) + if not mp3_path or not os.path.exists(mp3_path): + logger.error(f"❌ {date_str} 音频不可用,跳过识别与入库") + print("=" * 80) + summary['failed'] += 1 + continue + + # 处理长音频 + 切分入库;两者都成功才算当天任务完成 + day_ok = False + try: + audio_result = process_long_audio(mp3_path, output_folder, date_str) + if audio_result and audio_result.get('failed', 1) > 0: + # 识别不完整:**不执行切分**。否则会给这天写出 ext, + # 让下游把"半天内容"当成完整一天(且失败原因被掩盖)。 + # 分片数据仍在 xwlb_daily 里,重试会补全。 + logger.error(f"❌ {date_str} 识别不完整({audio_result.get('ok')}/{audio_result.get('chunks')} 片)," + f"跳过切分以免发布不完整的一天") + split_status = 'skipped(incomplete-asr)' + else: + split_status = news_to_db(date_str) + day_ok = bool(audio_result and audio_result.get('failed', 1) == 0 + and split_status != STATUS_FAILED) + logger.info(f"{date_str} 任务结果: 识别 {audio_result.get('ok', 0)}/{audio_result.get('chunks', 0)} 片,切分 {split_status}") + except Exception as e: + logger.error(f"❌ {date_str} 处理失败: {e}") + print("=" * 80) + + summary['ok' if day_ok else 'failed'] += 1 + + # 当天全程任务完成后清理 mp3/mp4 与 wav(config.yml: cleanup.*) + maybe_cleanup_after_run(date_str, day_ok) + + logger.info(f"本次运行汇总: 共 {summary['total']} 天,成功 {summary['ok']} 天,失败 {summary['failed']} 天") + return summary + + +# ======================== +# 主程序执行 +# ======================== +if __name__ == "__main__": + + import sys + import re + from datetime import datetime + + # 检查命令行参数 + if len(sys.argv) < 2: + print("用法: python getVideo5.py ") + print("日期格式: YYYYMMDD") + sys.exit(1) + + start_date = sys.argv[1] + end_date = sys.argv[2] if len(sys.argv) > 2 and sys.argv[2] else start_date + + # 检查日期格式 + date_pattern = r'^\d{8}$' + if not re.match(date_pattern, start_date) or not re.match(date_pattern, end_date): + print("错误: 日期格式必须为 YYYYMMDD") + sys.exit(1) + + # 检查日期有效性 + try: + #start_dt = datetime.strptime(start_date, '%Y%m%d') + #end_dt = datetime.strptime(end_date, '%Y%m%d') + + if start_date > end_date: + print(f"错误: start_date {start_date} 不能大于 end_date {end_date}") + sys.exit(1) + print("正在抓取央视《新闻联播》历史完整版视频链接...") + print("=" * 80) + config.log_summary() + summary = process_videos(start_date, end_date) + # 退出码供定时任务判断: 0=当天全链路成功; 1=失败(含"没抓到链接",交由重试) + if summary and summary['ok'] > 0 and summary['failed'] == 0: + sys.exit(0) + print(f"❌ 运行失败: {summary}") + sys.exit(1) + except ValueError as e: + print(f"错误: 无效日期 - {e}") + sys.exit(1) + diff --git a/main.py b/main.py new file mode 100644 index 0000000..9fd5509 --- /dev/null +++ b/main.py @@ -0,0 +1,25 @@ +"""每日入口:抓取当天《新闻联播》并完成识别、切分、入库、清理中间产物 + +定时任务用法: python main.py (systemd 见 systemd/xwlb-daily.timer) +指定日期用法: python getVideo5.py 20260904 20260904 + +退出码:0 = 当天全链路成功;1 = 失败(抓不到链接 / 识别或切分失败),供重试判断。 +""" +import logging +import sys +from datetime import datetime + +import config +from getVideo5 import process_videos + +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') + +if __name__ == '__main__': + # 获取当日日期并格式化为 yyyymmdd + today = datetime.now().strftime('%Y%m%d') + config.log_summary() + summary = process_videos(today, today) + if summary and summary['ok'] > 0 and summary['failed'] == 0: + sys.exit(0) + logging.error(f"❌ 当日任务失败: {summary}") + sys.exit(1) \ No newline at end of file diff --git a/main_videos.py b/main_videos.py new file mode 100644 index 0000000..d5bc043 --- /dev/null +++ b/main_videos.py @@ -0,0 +1,75 @@ +""" +xwlb_daily 表结构如下: ++--------------+---------+------+-----+---------+----------------+ +| Field | Type | Null | Key | Default | Extra | ++--------------+---------+------+-----+---------+----------------+ +| nid | int(11) | NO | PRI | NULL | auto_increment | +| news_days | date | NO | | NULL | | +| daily_sub_id | int(11) | NO | | NULL | | +| news_raw | text | NO | | NULL | | +| news_improve | text | NO | | NULL | | +| news_title | text | NO | | NULL | | ++--------------+---------+------+-----+---------+----------------+ +""" + +from mysqlHandle import MySQLDB +from getVideo5 import process_videos +from datetime import datetime, timedelta +import logging + +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + +def get_missing_dates(start_date, end_date): + """ + 给定日期范围,查询xwlb_daily表中缺失的日期 + """ + try: + # 连接数据库 + db = MySQLDB() + + # 查询指定日期范围内存在的所有日期 + result = db.query_data( + table="xwlb_daily", + columns="DISTINCT(news_days) as news_days", + where="news_days BETWEEN %s AND %s order by news_days", + params=(start_date, end_date) + ) + # 获取所有存在的日期 + existing_dates = [row['news_days'] for row in result] + + # 生成完整的日期范围 + start = datetime.strptime(start_date, '%Y-%m-%d').date() + end = datetime.strptime(end_date, '%Y-%m-%d').date() + + all_dates = [] + current_date = start + while current_date <= end: + all_dates.append(current_date) + current_date = current_date + timedelta(days=1) + + # 找出缺失的日期 + existing_set = set(existing_dates) + missing_dates = [date.strftime('%Y%m%d') for date in all_dates if date not in existing_set] + + logger.info(f"查询日期范围 {start_date} 到 {end_date}") + logger.info(f"存在 {len(existing_dates)} 天数据,缺失 {len(missing_dates)} 天数据") + logger.info(f"缺失日期: {missing_dates}") + return missing_dates + + except Exception as e: + logger.error(f"查询缺失日期时出错: {str(e)}") + return [] + +if __name__ == "__main__": + # 测试代码 + start_date = "2025-01-01" + end_date = "2025-10-25" + missing_dates=get_missing_dates(start_date, end_date) + for date_str in missing_dates: + logger.info(f"正在处理缺失日期: {date_str}") + try: + process_videos(date_str,date_str) + logger.info(f"成功处理日期: {date_str}") + except Exception as e: + logger.error(f"处理日期 {date_str} 时出错: {str(e)}") diff --git a/mysqlHandle.py b/mysqlHandle.py new file mode 100644 index 0000000..1241be3 --- /dev/null +++ b/mysqlHandle.py @@ -0,0 +1,9 @@ +"""数据库连接入口:转发到项目内置的 mysql_handler.MySQLDB + +历史:本文件原通过 sys.path 引用父项目 `utils.mysql_handler`,本项目独立运行后该模块不存在, +导致所有脚本 import 阶段即 ModuleNotFoundError(见 docs/BUGS.md B10)。 +现改为直接引用项目内的 mysql_handler,保持 `from mysqlHandle import MySQLDB` 的用法不变。 +""" +from mysql_handler import MySQLDB # noqa: F401 + +__all__ = ['MySQLDB'] \ No newline at end of file diff --git a/mysql_handler.py b/mysql_handler.py new file mode 100644 index 0000000..2610000 --- /dev/null +++ b/mysql_handler.py @@ -0,0 +1,240 @@ +"""MySQLDB —— 自包含的数据库访问层 + +背景:原实现位于父项目 `djapi/api/utils/mysql_handler.py`,本项目独立运行后该模块缺失 +(`mysqlHandle.py` 通过 sys.path 引用父目录,换机即 ModuleNotFoundError)。 +本文件把该实现内置回项目,**保持原有 4 个接口不变**,只修掉原实现的缺陷: + + db = MySQLDB() # 无参:读环境变量 + rows = db.query_data(table, columns, where, params) # -> list[dict] + new_id = db.insert_data(table, {col: val}) # -> lastrowid + db.update_data(table, {col: val}, where, params) # -> rowcount + db.close() + +修复点(对照 docs/BUGS.md): +- B10-1 连接失败时原实现只 print 然后留下 connection=None,后续再抛 + `AttributeError: 'NoneType' object has no attribute 'cursor'`;现改为**立即抛错** + 并给出目标地址,便于定位配置问题。 +- B10-2 原实现 `finally: if cursor:` 在 `cursor = self.connection.cursor()` 之前抛错时 + 会触发 `local variable 'cursor' referenced before assignment`(生产日志出现过); + 现把 cursor 提前初始化为 None。 +- 新增 `execute()` / `insert_many()`,用于按日期替换(幂等写库,见 B5)与批量事务插入。 + +配置来源(2026-09-25 重构): +- 地址/端口/用户/库名 ← config.yml 的 `mysql.*` +- 口令 ← .env 的 `MYSQL_PASSWORD` +- 以上均可被构造参数直接覆盖(保持与原实现一致的用法) +- 隧道自愈(2026-09-25):连接失败时调用 `tunnel.ensure_tunnel()`, + 端口不通则按 config.yml 的 `mysql.tunnel` 执行 autossh.sh 后重试一次(见 tunnel.py) +""" +import logging + +import config +import env # noqa: F401 — 保证 .env(敏感项)已加载 +import mysql.connector +import tunnel # 隧道自愈:端口不通时自动执行 autossh.sh +from mysql.connector import Error + +logger = logging.getLogger(__name__) + + +class MySQLDB: + """轻量 MySQL/MariaDB 访问封装(接口与原 utils.mysql_handler.MySQLDB 兼容)""" + + def __init__(self, host=None, port=None, username=None, password=None, database=None, + connect_timeout=10): + cfg = config.db_config() + self.host = host or cfg['host'] + self.port = int(port or cfg['port']) + self.username = username or cfg['username'] + self.password = password if password is not None else cfg['password'] + self.database = database or cfg['database'] + self.connect_timeout = connect_timeout + self.connection = None + self.connect() + + # ------------------------------------------------------------------ 连接 + @property + def target(self): + return f"{self.username}@{self.host}:{self.port}/{self.database}" + + def connect(self): + """建立连接;失败时先尝试隧道自愈,仍失败才抛异常(不再静默吞掉)""" + last_error = None + for attempt in (1, 2): + try: + self.connection = mysql.connector.connect( + host=self.host, + port=self.port, + user=self.username, + password=self.password, + database=self.database, + connection_timeout=self.connect_timeout, + charset='utf8mb4', + ) + if attempt == 2: + logger.info("✓ 隧道自愈后重连成功(%s)", self.target) + logger.debug("已连接数据库 %s", self.target) + return self.connection + except Error as e: + self.connection = None + last_error = e + if attempt == 2: + break + # 只在失败时介入:正常路径不增加任何探测开销。 + # 端口不通就按 config.yml 的 mysql.tunnel 执行 autossh.sh(见 tunnel.py) + healed = tunnel.ensure_tunnel(self.host, self.port, reason=f"MySQL 连接失败: {e}") + if not healed.get('healed') and healed['status'] not in ( + tunnel.ST_ALREADY_OPEN, tunnel.ST_SYSTEMD_WAITED, tunnel.ST_SCRIPT_STARTED): + break + raise RuntimeError( + f"MySQL 连接失败({self.target}):{last_error}\n" + "请检查 config.yml 的 mysql.host/port/user/database 与 .env 的 MYSQL_PASSWORD;" + f"若经 SSH 隧道访问,已尝试执行 {config.get('mysql.tunnel.script', 'autossh.sh')}" + "(可手动运行 `python tunnel.py` 查看隧道状态)" + ) from last_error + + def _ensure_connection(self): + """连接可用性检查(长流程中连接可能被服务端断开)""" + if self.connection is None: + self.connect() + return self.connection + try: + self.connection.ping(reconnect=True, attempts=3, delay=1) + except Error: + self.connect() + return self.connection + + # ------------------------------------------------------------------ 写 + def insert_data(self, table, data): + """插入一行,返回自增主键(失败返回 None)""" + cursor = None + try: + conn = self._ensure_connection() + cursor = conn.cursor() + columns = ', '.join(data.keys()) + placeholders = ', '.join(['%s'] * len(data)) + query = f"INSERT INTO {table} ({columns}) VALUES ({placeholders})" + cursor.execute(query, tuple(data.values())) + conn.commit() + logger.debug("插入 %s 成功,影响行数 %s", table, cursor.rowcount) + return cursor.lastrowid + except Error as e: + logger.error("插入 %s 失败: %s", table, e) + return None + finally: + if cursor is not None: + cursor.close() + + def insert_many(self, table, rows, chunk_size=500): + """批量插入(同一事务提交一次),rows 为 dict 列表;返回成功写入的行数""" + if not rows: + return 0 + cursor = None + try: + conn = self._ensure_connection() + cursor = conn.cursor() + columns = list(rows[0].keys()) + col_sql = ', '.join(columns) + placeholders = ', '.join(['%s'] * len(columns)) + query = f"INSERT INTO {table} ({col_sql}) VALUES ({placeholders})" + values = [tuple(r[c] for c in columns) for r in rows] + written = 0 + for start in range(0, len(values), chunk_size): + batch = values[start:start + chunk_size] + cursor.executemany(query, batch) + written += cursor.rowcount + conn.commit() + logger.debug("批量插入 %s 成功,%d 行", table, written) + return written + except Error as e: + if self.connection is not None: + self.connection.rollback() + logger.error("批量插入 %s 失败,已回滚: %s", table, e) + return 0 + finally: + if cursor is not None: + cursor.close() + + def update_data(self, table, data, where, params=None): + """按条件更新,返回受影响行数""" + cursor = None + try: + conn = self._ensure_connection() + cursor = conn.cursor() + set_clause = ', '.join([f"{key} = %s" for key in data.keys()]) + query = f"UPDATE {table} SET {set_clause} WHERE {where}" + all_params = tuple(data.values()) + tuple(params or ()) + cursor.execute(query, all_params) + conn.commit() + logger.debug("更新 %s 成功,影响行数 %s", table, cursor.rowcount) + return cursor.rowcount + except Error as e: + logger.error("更新 %s 失败: %s", table, e) + return 0 + finally: + if cursor is not None: + cursor.close() + + def execute(self, sql, params=None, commit=True): + """执行任意写 SQL,返回受影响行数(用于 DELETE / DDL 等)""" + cursor = None + try: + conn = self._ensure_connection() + cursor = conn.cursor() + cursor.execute(sql, tuple(params or ())) + if commit: + conn.commit() + return cursor.rowcount + except Error as e: + if commit and self.connection is not None: + self.connection.rollback() + logger.error("执行失败: %s | SQL=%s", e, sql.split('\n')[0][:120]) + return 0 + finally: + if cursor is not None: + cursor.close() + + # ------------------------------------------------------------------ 读 + def query_data(self, table, columns="*", where=None, params=None): + """查询,返回 list[dict](失败返回 [])""" + cursor = None + try: + conn = self._ensure_connection() + cursor = conn.cursor(dictionary=True) + query = f"SELECT {columns} FROM {table}" + if where: + query += f" WHERE {where}" + logger.debug("查询: %s", query) + cursor.execute(query, tuple(params or ())) + return cursor.fetchall() + except Error as e: + logger.error("查询 %s 失败: %s", table, e) + return [] + finally: + if cursor is not None: + cursor.close() + + def query_one(self, table, columns="*", where=None, params=None): + """查询单行,无结果返回 None""" + rows = self.query_data(table, columns, where, params) + return rows[0] if rows else None + + # ------------------------------------------------------------------ 收尾 + def close(self): + """关闭连接(可重复调用)""" + if self.connection is not None: + try: + if self.connection.is_connected(): + self.connection.close() + logger.debug("数据库连接已关闭") + except Error as e: + logger.warning("关闭数据库连接出错: %s", e) + finally: + self.connection = None + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, tb): + self.close() + return False \ No newline at end of file diff --git a/newsProcess.py b/newsProcess.py new file mode 100644 index 0000000..b39be8b --- /dev/null +++ b/newsProcess.py @@ -0,0 +1,271 @@ +""" +xwlb_daily 表结构如下: ++--------------+---------+------+-----+---------+----------------+ +| Field | Type | Null | Key | Default | Extra | ++--------------+---------+------+-----+---------+----------------+ +| nid | int(11) | NO | PRI | NULL | auto_increment | +| news_days | date | NO | | NULL | | +| daily_sub_id | int(11) | NO | | NULL | | +| news_raw | text | NO | | NULL | | +| news_improve | text | NO | | NULL | | +| news_title | text | NO | | NULL | | ++--------------+---------+------+-----+---------+----------------+ +xwlb_daily_ext 表结构如下: ++--------------+--------------+------+-----+---------+----------------+ +| Field | Type | Null | Key | Default | Extra | ++--------------+--------------+------+-----+---------+----------------+ +| extid | int(11) | NO | PRI | NULL | auto_increment | +| news_date | date | NO | | NULL | | +| sub_id | tinyint(4) | NO | | NULL | | +| news_title | varchar(256) | NO | | NULL | | +| news_content | text | NO | | NULL | | ++--------------+--------------+------+-----+---------+----------------+ + +流程:取给定日期的所有 news_improve 内容,按 daily_sub_id 顺序拼接为一个字符串, +交给 DeepSeek 切分为独立新闻并起标题,写入 xwlb_daily_ext。 + +本次修复(详见 docs/BUGS.md): +- B3:LLM 返回的 ```json 围栏 / 截断 / 非 JSON 内容按「剥围栏 → 宽解析 → 重试 → 明确报错」处理, + 不再静默丢弃当天全部文本; +- B5:写入前先删除该日期的旧记录,重跑不产生重复; +- B13:是否「已处理」改为按 xwlb_daily_ext 是否存在记录判断(原 `COUNT(*)>5` 会把 + 只切出 ≤5 条的日期永久判为未处理)。 +""" +import json +import logging +import re + +from mysqlHandle import MySQLDB +from deepseek import deepseek_text +import config + +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + +SPLIT_PROMPT = ("###请根据下面新闻内容的文本逻辑 \n" + " - 帮我分割成各个独立的新闻内容(注意:不要修改新闻本身,仅分割文本),并给每个新闻总结一个标题; \n" + " - 如果遇到'国内快讯'、'国际快讯'或'联播快讯',也请根据每个条快讯分割为一个新闻以及新闻标题; \n" + " - 返回json格式。json格式包含:news_id,news_title,news_content; news_id从1开始递增。") + +# 首次解析失败后的强化提示词(进一步约束输出格式) +RETRY_PROMPT = SPLIT_PROMPT + ("\n\n重要:只输出一个 JSON 数组,不要输出 markdown 代码块、不要输出任何解释文字。" + "数组每一项形如 {\"news_id\": 1, \"news_title\": \"标题\", \"news_content\": \"正文\"}。") + +# news_to_db 的执行结果 +STATUS_WRITTEN = 'written' # 本次完成切分并写入 +STATUS_SKIPPED = 'skipped' # 已有精编记录,按设计跳过 +STATUS_FAILED = 'failed' # 无可用文本 / 切分失败 / 写入不完整 + + +def normalize_date(date_str): + """把 YYYYMMDD / YYYY-MM-DD 统一为 YYYY-MM-DD""" + s = str(date_str or '').strip() + if re.fullmatch(r'\d{8}', s): + return f"{s[:4]}-{s[4:6]}-{s[6:8]}" + if re.fullmatch(r'\d{4}-\d{2}-\d{2}', s): + return s + raise ValueError(f"无法识别的日期格式: {date_str!r}(应为 YYYYMMDD 或 YYYY-MM-DD)") + + +def _get_ext_count(db, target_date): + try: + row = db.query_one("xwlb_daily_ext", "COUNT(*) as count", "news_date = %s", (target_date,)) + return int(row['count']) if row else 0 + except Exception as e: + logger.error(f"查询 xwlb_daily_ext 失败: {e}") + return 0 + + +def get_news_improve_by_date(target_date): + """ + 获取指定日期的所有 news_improve 内容,按 daily_sub_id 顺序拼接 + + 参数: + target_date: 目标日期,格式 YYYY-MM-DD + 返回值: + str | None: 拼接后的文本;该日期没有任何有效文本时返回 None + """ + try: + db = MySQLDB() + try: + result = db.query_data( + table="xwlb_daily", + columns="news_improve", + where="news_days = %s order by daily_sub_id ASC", + params=(target_date,)) + finally: + db.close() + + if not result: + return None + # 过滤空行:历史数据中存在识别失败留下的空白记录(B1),不能把它们拼进正文 + parts = [row['news_improve'].strip() for row in result + if row.get('news_improve') and row['news_improve'].strip()] + if not parts: + return None + return '\n'.join(parts) + except Exception as e: + logger.error(f"查询失败: {e}") + return None + + +# --------------------------------------------------------------------------- 解析 + +def _strip_code_fence(text): + """去掉 ```json ... ``` 包裹(B3 中最常见的失败形态)""" + s = (text or '').strip() + if s.startswith('```'): + s = re.sub(r'^```[a-zA-Z0-9_-]*\s*', '', s) + s = re.sub(r'\s*```\s*$', '', s) + return s.strip() + + +def _loads_lenient(text): + """先整体解析;失败则从第一个 { 或 [ 处做 raw_decode(容忍前后多余文字)""" + try: + return json.loads(text) + except json.JSONDecodeError: + match = re.search(r'[\[{]', text) + if not match: + raise + obj, _ = json.JSONDecoder().raw_decode(text[match.start():]) + return obj + + +def _to_news_list(data): + """把 LLM 返回的多种形态统一成列表""" + if isinstance(data, list): + return data + if isinstance(data, dict): + for value in data.values(): + if isinstance(value, list): + return value + values = list(data.values()) + # 兼容 {"1": {...}, "2": {...}};但不能把单条新闻的字段值当成列表(原实现踩过这个坑) + if values and all(isinstance(v, dict) for v in values): + return values + raise ValueError("JSON 对象中未找到新闻列表") + raise ValueError(f"不支持的 DeepSeek 响应格式: {type(data).__name__}") + + +def extract_news_rows(response_text, target_date): + """ + 解析 DeepSeek 返回并规整为待写入的行 + + 返回值: + list[dict]: 可直接交给 insert_many 的行 + 异常: + ValueError: 无法解析或没有任何有效新闻 + """ + if not response_text or not str(response_text).strip(): + raise ValueError("LLM 响应为空") + + data = _loads_lenient(_strip_code_fence(str(response_text))) + items = _to_news_list(data) + + rows = [] + for idx, item in enumerate(items, 1): + if not isinstance(item, dict): + logger.warning(f"跳过非对象条目: {str(item)[:60]}") + continue + content = str(item.get('news_content') or item.get('content') or '').strip() + if not content: + continue + title = str(item.get('news_title') or item.get('title') or '').strip() + try: + sub_id = int(item.get('news_id', idx)) + except (TypeError, ValueError): + sub_id = idx + rows.append({ + "news_date": target_date, + "sub_id": sub_id, + "news_title": title[:256], # varchar(256) + "news_content": content, + }) + + if not rows: + raise ValueError("解析结果中没有任何有效新闻") + return rows + + +# --------------------------------------------------------------------------- 入库 + +def news_to_db(target_date, force=False): + """ + 取当天原文 → DeepSeek 切分 → 覆盖写入 xwlb_daily_ext + + 参数: + target_date: 日期(YYYYMMDD 或 YYYY-MM-DD) + force: 即使已有精编记录也重新切分(默认 False,避免重复消耗 token) + 返回值: + str: STATUS_WRITTEN / STATUS_SKIPPED / STATUS_FAILED + (注意:不是 bool;'skipped' 与 'written' 都表示数据已就绪) + """ + target_date = normalize_date(target_date) + + if not force: + db = MySQLDB() + try: + existing = _get_ext_count(db, target_date) + finally: + db.close() + if existing > 0: + logger.info(f"日期 {target_date} 已有 {existing} 条精编记录,跳过(需重跑请加 force)") + return STATUS_SKIPPED + + result = get_news_improve_by_date(target_date) + if result is None: + logger.warning(f"日期 {target_date} 没有新闻内容,跳过") + return STATUS_FAILED + logger.info(f"日期 {target_date} 的新闻内容长度:{len(result)} 字符") + + attempts = [(SPLIT_PROMPT, {}), + (RETRY_PROMPT, {"max_tokens": config.get_int('llm_split.retry_max_tokens', 32000)})] + rows = None + last_error = None + for attempt, (prompt, extra) in enumerate(attempts, 1): + response = None + try: + # use_fallback=False:切分环节自带重试与解析校验, + # 关掉"返回降级文本"的兜底,模型名/权限等配置错误会直接暴露而不是静默出数据 + response = deepseek_text(result, prompt, use_fallback=False, **extra) + rows = extract_news_rows(response, target_date) + logger.info(f"第 {attempt} 次切分成功,得到 {len(rows)} 条新闻") + break + except Exception as e: + last_error = e + logger.error(f"第 {attempt} 次切分/解析失败: {e}") + if response is not None: + logger.error(f"DeepSeek 原始返回(前 300 字): {str(response)[:300]}") + rows = None + + if not rows: + # 不写库、不删除已有数据,交给下次重跑 + logger.error(f"❌ 日期 {target_date} 切分失败,未写入数据库: {last_error}") + return STATUS_FAILED + + db = MySQLDB() + try: + deleted = db.execute("DELETE FROM xwlb_daily_ext WHERE news_date = %s", (target_date,)) + written = db.insert_many("xwlb_daily_ext", rows) + finally: + db.close() + + if written != len(rows): + logger.error(f"❌ 日期 {target_date} 写入不完整:期望 {len(rows)} 条,实际 {written} 条") + return STATUS_FAILED + logger.info(f"✓ 日期 {target_date} 写入 {written} 条新闻(替换旧记录 {deleted} 条)") + return STATUS_WRITTEN + + +if __name__ == "__main__": + import sys + from datetime import datetime + + date_arg = sys.argv[1] if len(sys.argv) > 1 else datetime.now().strftime('%Y-%m-%d') + force_flag = '--force' in sys.argv + config.log_summary() + status = news_to_db(date_arg, force=force_flag) + label = {'written': '成功写入', 'skipped': '已有精编记录,已跳过', 'failed': '失败'}.get(status, status) + print(f"处理结果: {label}({status})") + raise SystemExit(0 if status != STATUS_FAILED else 1) \ No newline at end of file diff --git a/newsRedo.py b/newsRedo.py new file mode 100644 index 0000000..cea477d --- /dev/null +++ b/newsRedo.py @@ -0,0 +1,122 @@ +""" +newsRedo — 手动重新执行新闻 AI 分割流程。 + +用法: + python newsRedo.py # 默认当天日期 + python newsRedo.py 20250601 # yyyymmdd 格式 + python newsRedo.py 2025-06-01 # yyyy-mm-dd 格式 + python newsRedo.py 2025-06-01 --force # 已有精编记录也重新切分 + +流程: + 1. 检查 xwlb_daily_ext 是否已有记录 → 已处理过则跳过(--force 可强制重跑) + 2. 检查 xwlb_daily 是否有当天记录 → 无记录则先跑 getVideo5 全流程 + 3. 有记录但未处理 → 直接执行 news_to_db() AI 分割 +""" +import sys +import re +import logging +from datetime import datetime +from mysqlHandle import MySQLDB +from newsProcess import news_to_db +from getVideo5 import process_videos +import config + +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + + +def _parse_date(date_str): + """解析日期,返回 (yyyymmdd_str, yyyy_mm_dd_str),或报错退出""" + if not date_str: + today = datetime.now() + d8 = today.strftime('%Y%m%d') + d10 = today.strftime('%Y-%m-%d') + logger.info(f"未指定日期,使用当天: {d10}") + return d8, d10 + + if re.match(r'^\d{4}-\d{2}-\d{2}$', date_str): + try: + datetime.strptime(date_str, '%Y-%m-%d') + except ValueError: + print(f"无效日期: {date_str}") + sys.exit(1) + return date_str.replace('-', ''), date_str + + if re.match(r'^\d{8}$', date_str): + try: + datetime.strptime(date_str, '%Y%m%d') + except ValueError: + print(f"无效日期: {date_str}") + sys.exit(1) + return date_str, f"{date_str[:4]}-{date_str[4:6]}-{date_str[6:8]}" + + print("日期格式错误,请使用 yyyymmdd 或 yyyy-mm-dd 格式") + sys.exit(1) + + +def main(): + args = [a for a in sys.argv[1:] if not a.startswith('-')] + force = '--force' in sys.argv[1:] or '-f' in sys.argv[1:] + date_str = args[0] if args else None + date_d8, date_d10 = _parse_date(date_str) + + db = MySQLDB() + + # 1. 检查 xwlb_daily_ext 是否已处理过 + # 判据由「>5 条」改为「存在任意记录」(原判据会把只切出 ≤5 条的日期永久判为未处理,B13) + try: + ext_count = db.query_data( + table="xwlb_daily_ext", + columns="COUNT(*) as count", + where="news_date = %s", + params=(date_d10,) + ) + existing = ext_count[0]['count'] if ext_count else 0 + if existing > 0 and not force: + logger.info(f"日期 {date_d10} 已有 {existing} 条精编记录,无需重新处理(需重跑请加 --force)。") + return + if existing > 0 and force: + logger.info(f"日期 {date_d10} 已有 {existing} 条精编记录,--force 将重新切分并覆盖。") + except Exception as e: + logger.error(f"查询 xwlb_daily_ext 失败: {e}") + finally: + db.close() + + db = MySQLDB() + + # 2. 检查 xwlb_daily 是否有当天数据 + try: + daily_count = db.query_data( + table="xwlb_daily", + columns="COUNT(*) as count", + where="news_days = %s", + params=(date_d10,) + ) + has_daily = daily_count and daily_count[0]['count'] > 0 + except Exception as e: + logger.error(f"查询 xwlb_daily 失败: {e}") + has_daily = False + finally: + db.close() + + # 3. 分支处理 + if has_daily: + logger.info(f"日期 {date_d10} 在 xwlb_daily 中有记录,直接执行 AI 分割。") + try: + status = news_to_db(date_d10, force=force) + logger.info(f"news_to_db 结果: {status}") + except Exception as e: + logger.error(f"news_to_db 执行出错: {e}") + sys.exit(1) + else: + logger.info(f"日期 {date_d10} 在 xwlb_daily 中无记录,重新执行视频下载全流程。") + try: + process_videos(date_d8, date_d8) + except Exception as e: + logger.error(f"process_videos 执行出错: {e}") + sys.exit(1) + + +if __name__ == "__main__": + config.log_summary() + main() diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..e16ec54 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,34 @@ +# xwlb 运行依赖 +# +# 安装(本机为 externally-managed 环境,需用虚拟环境): +# python3 -m venv .venv && .venv/bin/pip install -r requirements.txt +# +# 系统依赖(apt):ffmpeg(抽音/转码必需);MySQL 客户端非必需(用 mysql-connector 连接) + +# ---- 数据库 ---- +# 原项目用 mysql_connector_repackaged(旧 fork,python3.13 下不建议);改用官方驱动 +mysql-connector-python>=9.0 + +# ---- 采集 ---- +requests==2.32.5 +beautifulsoup4==4.14.2 +yt_dlp==2025.11.12 + +# ---- 音频处理 ---- +pydub==0.25.1 +# pydub 依赖标准库 audioop,而 audioop 在 Python 3.13 中被移除;3.13 需额外安装此兼容包 +audioop-lts; python_version >= "3.13" + +# ---- 配置 ---- +# config.yml 解析(非敏感配置;敏感项在 .env) +PyYAML>=6.0 + +# ---- ASR / LLM ---- +# 原 pin 为 1.24.6,但本机实际运行的是 1.27.7(base url 通过 +# DASHSCOPE_HTTP_BASE_URL / DASHSCOPE_WEBSOCKET_BASE_URL 生效,已在 1.27.7 验证) +dashscope>=1.27.7 + +# ---- 以下依赖当前代码未使用,保留仅为兼容历史环境 ---- +# m3u8 # 早期解析 m3u8 播放列表 +# playwright # 早期页面渲染方案 +# aiofiles、aiohttp # 并发下载方案(现为同步 yt-dlp) \ No newline at end of file diff --git a/scripts/day_status.py b/scripts/day_status.py new file mode 100755 index 0000000..4521871 --- /dev/null +++ b/scripts/day_status.py @@ -0,0 +1,52 @@ +"""查询某天数据完成度,供重试逻辑判断"该重跑哪个阶段" + + .venv/bin/python scripts/day_status.py 2026-09-24 + +退出码: + 0 已有精编(xwlb_daily_ext 有数据)→ 视为完成 + 1 识别已全部完成(存在 .asr_complete_<日期> 标记)但有分片无精编 → 只缺切分环节 + 2 其余情况(分片未入库 / 识别只跑了一半)→ 必须重跑全链路 + +注意:只有 daily 有数据**不足以**说明识别跑完了。识别中途卡死时也会留下一部分分片, +此时若无条件走"只重跑切分",会把半天内容当成完整一天切分入库——所以用标记文件区分。 + +同时在 stdout 打印 daily=<分片数> ext=<精编条数>,便于日志排查。 +""" +import os +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from mysqlHandle import MySQLDB # noqa: E402 +from audioRead import asr_marker_path # noqa: E402 + + +def main(): + if len(sys.argv) < 2: + print('用法: day_status.py ') + return 3 + raw = sys.argv[1] + date10 = f"{raw[:4]}-{raw[4:6]}-{raw[6:]}" if len(raw) == 8 else raw + + db = MySQLDB() + try: + daily = db.query_one('xwlb_daily', 'COUNT(*) n', 'news_days = %s', (date10,)) + ext = db.query_one('xwlb_daily_ext', 'COUNT(*) n', 'news_date = %s', (date10,)) + finally: + db.close() + + n_daily = (daily or {}).get('n', 0) or 0 + n_ext = (ext or {}).get('n', 0) or 0 + marker = asr_marker_path(date10) + asr_done = os.path.exists(marker) + print(f"daily={n_daily} ext={n_ext} asr_complete={int(asr_done)}") + if n_ext > 0: + return 0 + if n_daily > 0 and asr_done: + return 1 + return 2 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/scripts/install_systemd.sh b/scripts/install_systemd.sh new file mode 100755 index 0000000..50abd39 --- /dev/null +++ b/scripts/install_systemd.sh @@ -0,0 +1,51 @@ +#!/usr/bin/env bash +# 安装/更新 systemd 定时任务(需要 sudo;幂等,可反复执行) +# +# bash scripts/install_systemd.sh +# +# 做四件事: +# 1) 把 systemd/*.{service,timer} 装到 /etc/systemd/system/ +# 2) 启用并启动 xwlb-daily.timer(每天 21:00 触发) +# 3) 启用 xwlb-tunnel.service(autossh 隧道,随开机自启) +# 4) 打印下一次触发时间 +set -euo pipefail + +DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +SUDO="" +if [ "$(id -u)" -ne 0 ]; then SUDO="sudo"; fi + +echo "== 1/4 安装单元文件 ==" +for unit in xwlb-daily.service xwlb-daily.timer xwlb-tunnel.service; do + $SUDO install -m 0644 "$DIR/systemd/$unit" "/etc/systemd/system/$unit" + echo " $unit" +done + +echo "== 2/4 daemon-reload ==" +$SUDO systemctl daemon-reload + +echo "== 3/4 启用定时器 ==" +$SUDO systemctl enable --now xwlb-daily.timer >/dev/null +echo " xwlb-daily.timer 已启用" + +echo "== 4/4 隧道服务 ==" +if ss -ltn 2>/dev/null | grep -q '127.0.0.1:13306'; then + if systemctl is-active --quiet xwlb-tunnel.service; then + $SUDO systemctl restart xwlb-tunnel.service + echo " xwlb-tunnel.service 已重启" + else + echo " ⚠️ 13306 已被**非 systemd 托管**的 autossh 占用,暂不启用隧道服务。" + echo " 想改由 systemd 托管(可开机自启、断了自动重连)请执行:" + echo " pkill -f 'autossh.*13306' && sudo systemctl enable --now xwlb-tunnel.service" + fi +else + $SUDO systemctl enable --now xwlb-tunnel.service >/dev/null + echo " xwlb-tunnel.service 已启用" +fi + +echo +echo "== 下一次触发时间 ==" +systemctl list-timers xwlb-daily.timer --no-pager || true +echo +echo "查看日志: journalctl -u xwlb-daily -f (也可看 $DIR/main.log)" +echo "立即试跑: sudo systemctl start xwlb-daily.service" +echo "临时停用: sudo systemctl disable --now xwlb-daily.timer" \ No newline at end of file diff --git a/scripts/run_daily.sh b/scripts/run_daily.sh new file mode 100755 index 0000000..5e41efa --- /dev/null +++ b/scripts/run_daily.sh @@ -0,0 +1,124 @@ +#!/usr/bin/env bash +# 每日全链路任务(由 systemd 定时器调用,也可手动执行) +# +# scripts/run_daily.sh # 抓取"今天" +# scripts/run_daily.sh 20260924 # 指定日期 +# +# 重试策略:首次失败后,在 21:30、22:00 各再执行一次(可用 XWLB_RETRY_SLOTS 覆盖)。 +# · 若当天分片已入库、只缺精编(切分失败)→ 重试只重跑切分,省掉全部 ASR 调用 +# · 若分片都没入库(下载/识别失败) → 重试重跑全链路 +# +# 环境变量(一般不用改): +# XWLB_RETRY_SLOTS 重试时刻,默认 "21:30 22:00" +# XWLB_ATTEMPT_TIMEOUT 单次尝试上限秒数,默认 2400(40 分钟) +# XWLB_ENV_FILE / XWLB_CONFIG_FILE 透传给 python 的配置覆盖 +# XWLB_FORCE=1 即使当天已入库也强制重跑 +# +# 退出码:0=成功 1=重试用尽仍失败 2=已有实例在运行(跳过) +set -uo pipefail + +DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$DIR" || exit 1 +PY="$DIR/.venv/bin/python" +LOG="$DIR/main.log" +RETRY_SLOTS="${XWLB_RETRY_SLOTS:-21:30 22:00}" +ATTEMPT_TIMEOUT="${XWLB_ATTEMPT_TIMEOUT:-2400}" +DB_PORT="${XWLB_DB_PORT:-13306}" + +log() { printf '%s [run_daily] %s\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$*"; } + +# ---- 防重入:定时器与手动执行、或两次重试叠在一起时,避免重复烧 ASR ---- +exec 9>"$DIR/.run_daily.lock" +if ! flock -n 9; then + log "已有 run_daily 实例在运行,本次跳过" + exit 2 +fi + +# ---- 日期 ---- +D8="${1:-$(date +%Y%m%d)}" +D10="${D8:0:4}-${D8:4:2}-${D8:6:2}" + +# ---- 隧道自愈:数据库走 autossh 本地端口,隧道断了当天必失败 ---- +# 先用 ss 快速判断(零成本);不通则交给 tunnel.py —— 那里是唯一的自愈实现, +# 所有 Python 入口(main/getVideo5/newsProcess/...)走的是同一套逻辑。 +ensure_tunnel() { + ss -ltn 2>/dev/null | grep -q "127.0.0.1:${DB_PORT}" && return 0 + log "数据库端口 ${DB_PORT} 不通,调用 tunnel.py 自愈(必要时执行 autossh.sh)" + if "$PY" "$DIR/tunnel.py" >>"$LOG" 2>&1; then + log "隧道已就绪" + return 0 + fi + log "⚠️ 隧道仍不通(端口 ${DB_PORT}),本次很可能失败" + return 1 +} + +# 等到 HH:MM;若已过该时刻则立即返回(例如首次尝试本来就跑过了 21:30) +wait_until() { + local target="$1" now target_sec + now=$(date +%s) + target_sec=$(date -d "today ${target}" +%s 2>/dev/null) || { log "无法解析重试时刻 ${target},立即执行"; return 0; } + if [ "$target_sec" -gt "$now" ]; then + log "等待到 ${target} 再重试(约 $(( (target_sec - now + 59) / 60 )) 分钟)" + sleep $(( target_sec - now )) + else + log "已过 ${target},立即重试" + fi +} + +# 失败重试时:只缺精编就走切分环节(快、便宜),否则重跑全链路 +build_cmd() { + local stage="$1" + if [ "$stage" = "retry" ]; then + "$PY" "$DIR/scripts/day_status.py" "$D10" >/dev/null 2>&1 + case $? in + 1) log "当天分片已入库、只缺精编 → 重试仅重跑切分环节" + CMD=("$PY" "$DIR/newsProcess.py" "$D10" --force); return 0 ;; + 0) log "当天数据已完整(可能上次重试已完成),无需再跑" + CMD=(); return 0 ;; + *) log "当天分片未入库 → 重试重跑全链路" ;; + esac + fi + CMD=("$PY" "$DIR/getVideo5.py" "$D8" "$D8") +} + +run_attempt() { + local label="$1" stage="$2" rc + build_cmd "$stage" + if [ "${#CMD[@]}" -eq 0 ]; then + return 0 + fi + log "===== ${label} 开始(${CMD[*]},上限 ${ATTEMPT_TIMEOUT}s)=====" + timeout "$ATTEMPT_TIMEOUT" "${CMD[@]}" 2>&1 | tee -a "$LOG" + rc=${PIPESTATUS[0]} + if [ "$rc" -eq 0 ]; then + log "===== ${label} 成功 =====" + return 0 + fi + log "===== ${label} 失败(退出码 ${rc})=====" + return 1 +} + +ensure_tunnel || true + +# ---- 当天已完成则跳过:ext 有数据即视为完成(识别不完整的日子不会写出 ext)---- +# 手动强制重跑:XWLB_FORCE=1 bash scripts/run_daily.sh <日期> +if [ "${XWLB_FORCE:-0}" != "1" ]; then + if "$PY" "$DIR/scripts/day_status.py" "$D10" >/dev/null 2>&1; then + log "$D10 的数据已完整入库,跳过本次运行(强制重跑用 XWLB_FORCE=1 或 newsRedo.py)" + exit 0 + fi +fi + +if run_attempt "首次尝试($(date +%H:%M))" full; then + exit 0 +fi + +for slot in $RETRY_SLOTS; do + wait_until "$slot" + if run_attempt "重试(${slot})" retry; then + exit 0 + fi +done + +log "❌ 全部重试失败,${D8} 数据未完成" +exit 1 \ No newline at end of file diff --git a/systemd/xwlb-daily.service b/systemd/xwlb-daily.service new file mode 100644 index 0000000..5e399f2 --- /dev/null +++ b/systemd/xwlb-daily.service @@ -0,0 +1,31 @@ +[Unit] +Description=新闻联播每日抓取入库(下载 → 识别 → 校对 → 切分 → 入库 → 清理) +Documentation=file:///home/pi/project/xwlb/README.md +# 数据库走 autossh 隧道,隧道由独立服务托管,随本服务一起拉起 +Wants=network-online.target xwlb-tunnel.service +After=network-online.target xwlb-tunnel.service + +[Service] +Type=oneshot +User=pi +Group=pi +WorkingDirectory=/home/pi/project/xwlb +# 失败重试(21:30 / 22:00)在脚本内部完成,因此可能持续到 2~3 小时; +# systemd 默认 TimeoutStartSec=90s 会把服务砍掉,必须放大。 +TimeoutStartSec=10800 +# 成败由脚本退出码体现,不用 systemd 反复重启 +Restart=no +StandardOutput=journal +StandardError=journal +SyslogIdentifier=xwlb-daily +UMask=0077 +Environment=PYTHONUNBUFFERED=1 +ExecStart=/home/pi/project/xwlb/scripts/run_daily.sh + +# 轻量加固:只读保护系统目录,不动 $HOME(yt-dlp 需要写 ~/.cache) +NoNewPrivileges=yes +PrivateTmp=yes +ProtectSystem=full + +[Install] +WantedBy=multi-user.target \ No newline at end of file diff --git a/systemd/xwlb-daily.timer b/systemd/xwlb-daily.timer new file mode 100644 index 0000000..cc87837 --- /dev/null +++ b/systemd/xwlb-daily.timer @@ -0,0 +1,13 @@ +[Unit] +Description=每天 21:00 触发《新闻联播》抓取入库(失败时脚本内 21:30 / 22:00 自动重试) + +[Timer] +# 每天 21:00(新闻联播 19:00 播出,21:00 页面与视频都已就绪) +OnCalendar=*-*-* 21:00:00 +# 树莓派关机/重启错过时刻时,开机后补跑一次(不会重复:脚本内部幂等) +Persistent=true +AccuracySec=1s +Unit=xwlb-daily.service + +[Install] +WantedBy=timers.target \ No newline at end of file diff --git a/systemd/xwlb-tunnel.service b/systemd/xwlb-tunnel.service new file mode 100644 index 0000000..e9ebbab --- /dev/null +++ b/systemd/xwlb-tunnel.service @@ -0,0 +1,21 @@ +[Unit] +Description=autossh 隧道:本机 127.0.0.1:13306 → doorcome.cn:3306(MySQL) +Documentation=file:///home/pi/project/xwlb/autossh.sh +After=network-online.target +Wants=network-online.target + +[Service] +User=pi +Group=pi +# 让 autossh 在首次连接失败时也持续重试,而不是直接退出 +Environment=AUTOSSH_GATETIME=0 +ExecStart=/usr/bin/autossh -M 0 -N \ + -o ServerAliveInterval=30 -o ServerAliveCountMax=3 \ + -o ExitOnForwardFailure=yes \ + -L 13306:localhost:3306 tunnel@doorcome.cn +Restart=always +RestartSec=10 +SyslogIdentifier=xwlb-tunnel + +[Install] +WantedBy=multi-user.target \ No newline at end of file diff --git a/tests/test_cleanup.py b/tests/test_cleanup.py new file mode 100644 index 0000000..b7851b0 --- /dev/null +++ b/tests/test_cleanup.py @@ -0,0 +1,106 @@ +"""清理逻辑回归测试(不依赖 pytest,且**不触碰真实目录**) + + python tests/test_cleanup.py + +做法:用 XWLB_CONFIG_FILE 指向临时 config.yml,把 paths 指向临时目录, +验证 cleanup 段各开关的组合行为。 +""" +import os +import sys +import tempfile +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT)) + + +def build_env(tmp, after_daily_run=1, only_on_success=1, remove_video=1, remove_audio=1): + """在临时目录生成 config.yml,并让 config 模块加载它;返回 (video_dir, audio_dir)""" + video_dir = Path(tmp) / 'video' + audio_dir = Path(tmp) / 'audio' + video_dir.mkdir(parents=True, exist_ok=True) + audio_dir.mkdir(parents=True, exist_ok=True) + for name in ('a.mp3', 'b.mp4', 'keep.txt'): + (video_dir / name).write_bytes(b'x' * 10) + for name in ('c.wav', 'd.wav', 'keep.txt'): + (audio_dir / name).write_bytes(b'x' * 10) + + cfg = Path(tmp) / 'config.yml' + cfg.write_text(f""" +paths: + video_dir: {video_dir} + audio_dir: {audio_dir} +cleanup: + after_daily_run: {after_daily_run} + only_on_success: {only_on_success} + remove_video: {remove_video} + remove_audio: {remove_audio} +""", encoding='utf-8') + os.environ['XWLB_CONFIG_FILE'] = str(cfg) + import config + config.reload_config() + return video_dir, audio_dir + + +def snapshot(video_dir, audio_dir): + return sorted(p.name for p in list(video_dir.iterdir()) + list(audio_dir.iterdir())) + + +def main(): + failed = 0 + + def check(desc, got, want): + nonlocal failed + if got == want: + print(f" ✓ {desc}: {got}") + else: + failed += 1 + print(f" ✗ {desc}: 期望 {want},实际 {got}") + + import cleanup + + # 1) 成功 + 开关全开 → 删除 mp3/mp4/wav,保留其他文件 + with tempfile.TemporaryDirectory() as tmp: + video_dir, audio_dir = build_env(tmp) + result = cleanup.maybe_cleanup_after_run('2026-09-04', day_ok=True) + check('成功时删除 4 个中间产物', result['removed'], 4) + check('非中间产物保留', snapshot(video_dir, audio_dir), ['keep.txt', 'keep.txt']) + check('清理字节数>0', result['bytes'] > 0, True) + + # 2) 失败 + only_on_success=1(默认)→ 保留文件便于重跑 + with tempfile.TemporaryDirectory() as tmp: + video_dir, audio_dir = build_env(tmp, only_on_success=1) + result = cleanup.maybe_cleanup_after_run('2026-09-04', day_ok=False) + check('失败时保留文件', result['skipped'], True) + check('文件未被删除', len(snapshot(video_dir, audio_dir)), 6) + + # 3) 失败 + only_on_success=0 → 仍然清理 + with tempfile.TemporaryDirectory() as tmp: + video_dir, audio_dir = build_env(tmp, only_on_success=0) + result = cleanup.maybe_cleanup_after_run('2026-09-04', day_ok=False) + check('only_on_success=0 时失败也清理', result['removed'], 4) + + # 4) after_daily_run=0 → 完全不动 + with tempfile.TemporaryDirectory() as tmp: + video_dir, audio_dir = build_env(tmp, after_daily_run=0) + result = cleanup.maybe_cleanup_after_run('2026-09-04', day_ok=True) + check('after_daily_run=0 时跳过', result['skipped'], True) + check('文件未被删除', len(snapshot(video_dir, audio_dir)), 6) + + # 5) 只关 wav 清理 → 只删 mp3/mp4 + with tempfile.TemporaryDirectory() as tmp: + video_dir, audio_dir = build_env(tmp, remove_audio=0) + result = cleanup.maybe_cleanup_after_run('2026-09-04', day_ok=True) + check('remove_audio=0 时只删视频', result['removed'], 2) + check('wav 保留、mp3/mp4 已删', snapshot(video_dir, audio_dir), + ['c.wav', 'd.wav', 'keep.txt', 'keep.txt']) + + os.environ.pop('XWLB_CONFIG_FILE', None) + import config + config.reload_config() + print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}") + return 1 if failed else 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tests/test_config.py b/tests/test_config.py new file mode 100644 index 0000000..e74c863 --- /dev/null +++ b/tests/test_config.py @@ -0,0 +1,173 @@ +"""配置层回归测试(不依赖 pytest) + + python tests/test_config.py + +覆盖: +- config.yml 能被加载,关键配置项存在且类型正确; +- 点号取值 / 布尔与整数转换; +- 目录解析(相对路径基于项目根); +- 环境变量覆盖 config.yml(临时试验用); +- 敏感项不在 config.yml 里(只能在 .env)。 +""" +import os +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT)) + +import yaml # noqa: E402 + +import config # noqa: E402 + +# 测试不硬编码"当前配的是哪个模型/哪条路由"——那是用户随时会改的。 +# 这里直接用 yaml 独立读一遍 config.yml,与 config.py 的解析结果对比: +# 校验的是"解析机制正确",而不是"值恰好等于某个名字"。 +with open(ROOT / 'config.yml', encoding='utf-8') as _f: + RAW = yaml.safe_load(_f) + +def raw(dotted, default=None): + node = RAW + for key in dotted.split('.'): + if not isinstance(node, dict) or key not in node: + return default + node = node[key] + return node + +# 按 config.yml 自己的 routes/endpoints 推导"应要求哪些密钥"(独立算法) +_expected_secrets = {'MYSQL_PASSWORD'} +for _role in ('asr', 'correct', 'split'): + _ep = (RAW.get('endpoints') or {}).get((RAW.get('routes') or {}).get(_role), {}) + if _ep.get('api_key_env'): + _expected_secrets.add(_ep['api_key_env']) + + +def main(): + failed = 0 + + def check(desc, got, want): + nonlocal failed + if got == want: + print(f" ✓ {desc}: {got!r}") + else: + failed += 1 + print(f" ✗ {desc}: 期望 {want!r},实际 {got!r}") + + print("config.yml 加载:") + check('配置文件存在', config.LOADED_CONFIG_FILE is not None, True) + check('数据库端口', config.get_int('mysql.port'), 13306) + check('数据库库名', config.get('mysql.database'), 'myquant') + check('ASR 模型', config.model('asr_model'), 'paraformer-realtime-v2') + check('校对模型(与 config.yml 一致)', config.model('correct_model'), raw('models.correct_model')) + check('切分模型(与 config.yml 一致)', config.model('split_model'), raw('models.split_model')) + check('校对开关(默认1)', config.get_bool('llm_correct.enabled'), True) + check('清理开关(默认1)', config.get_bool('cleanup.after_daily_run'), True) + check('仅成功时清理', config.get_bool('cleanup.only_on_success'), True) + check('语言提示', config.get_list('asr.language_hints'), ['zh', 'en']) + + print("\n目录解析:") + check('video_dir 基于项目根', config.video_dir(), ROOT / 'xwlb_video') + check('audio_dir 基于项目根', config.audio_dir(), ROOT / 'audio_processing') + + print("\n环境变量覆盖优先于 config.yml:") + os.environ['DASHSCOPE_LLM_MODEL'] = 'qwen-max-test' + os.environ['LLM_CORRECT_ENABLED'] = '0' + cfg = config.reload_config() + check('模型被环境变量覆盖', cfg['models']['correct_model'], 'qwen-max-test') + check('开关被环境变量覆盖', config.get_bool('llm_correct.enabled'), False) + del os.environ['DASHSCOPE_LLM_MODEL'], os.environ['LLM_CORRECT_ENABLED'] + config.reload_config() + check('恢复 config.yml 值', config.model('correct_model'), raw('models.correct_model')) + + print("\n接入点与路由(换供应商的核心):") + check('asr 路由(与 config.yml 一致)', config.route('asr'), raw('routes.asr')) + check('correct 路由(与 config.yml 一致)', config.route('correct'), raw('routes.correct')) + check('correct 接入点类型(与 config.yml 一致)', + config.endpoint_for('correct')[1]['kind'], raw(f"endpoints.{raw('routes.correct')}.kind")) + check('split 接入点类型', config.endpoint_for('split')[1]['kind'], 'openai') + check('split 请求地址拼接', config.openai_url('split'), + 'https://api.deepseek.com/v1/chat/completions') + check('密钥变量名来自接入点', config.api_key_env('split'), 'DEEPSEEK_API_KEY') + check('必需敏感项 = 路由用到的接入点密钥 + MySQL 口令', + sorted(config.required_secrets()), sorted(_expected_secrets)) + check('默认配置无问题', config.validate_endpoints(), []) + import dashscope + check('dashscope SDK 地址已按配置生效', dashscope.base_http_api_url, + config.endpoint('dashscope')['http_base_url']) + check('dashscope WS 地址已按配置生效', dashscope.base_websocket_api_url, + config.endpoint('dashscope')['websocket_base_url']) + + print("\n切换供应商(把校对接管到 OpenAI 兼容接入点):") + os.environ['XWLB_ROUTE_CORRECT'] = 'deepseek' + os.environ['DASHSCOPE_LLM_MODEL'] = 'deepseek-chat' + config.reload_config() + check('correct 已切到 openai 类型', config.endpoint_for('correct')[1]['kind'], 'openai') + check('correct 请求地址随之改变', config.openai_url('correct'), + 'https://api.deepseek.com/v1/chat/completions') + check('correct 模型名可独立覆盖', config.model('correct_model'), 'deepseek-chat') + check('密钥变量名随供应商变化', config.api_key_env('correct'), 'DEEPSEEK_API_KEY') + + # 把 ASR 也切到非 dashscope 接入点:应被校验拦下(实时识别协议不支持) + os.environ['XWLB_ROUTE_ASR'] = 'deepseek' + config.reload_config() + check('非 dashscope 的 ASR 被校验提示', any('仅支持 kind=dashscope' in p + for p in config.validate_endpoints()), True) + check('不需要 dashscope 密钥了', config.required_secrets(), ['DEEPSEEK_API_KEY', 'MYSQL_PASSWORD']) + + # 指向不存在的接入点:必须给出可读错误 + os.environ['XWLB_ROUTE_SPLIT'] = 'not-exist' + config.reload_config() + try: + config.endpoint_for('split') + check('未知接入点应报错', '未报错', 'RuntimeError') + except RuntimeError as e: + check('未知接入点报错可读', '没有该定义' in str(e), True) + + # 地址也可用环境变量覆盖 + os.environ['XWLB_ROUTE_CORRECT'] = 'deepseek' + os.environ['DEEPSEEK_BASE_URL'] = 'https://my-gateway.example.com/llm' + config.reload_config() + check('base_url 可被环境变量覆盖', config.openai_url('correct'), + 'https://my-gateway.example.com/llm/chat/completions') + + for key in ('XWLB_ROUTE_CORRECT', 'XWLB_ROUTE_ASR', 'XWLB_ROUTE_SPLIT', + 'DASHSCOPE_LLM_MODEL', 'DEEPSEEK_BASE_URL'): + os.environ.pop(key, None) + config.reload_config() + check('恢复 config.yml 路由', [config.route(r) for r in ('asr', 'correct', 'split')], + [raw('routes.asr'), raw('routes.correct'), raw('routes.split')]) + + print("\n敏感项与 config.yml 隔离:") + import yaml + yml = yaml.safe_load((ROOT / 'config.yml').read_text(encoding='utf-8')) + + def walk(node, path=''): + for key, value in (node or {}).items(): + here = f'{path}.{key}' if path else str(key) + if isinstance(value, dict): + yield from walk(value, here) + else: + yield here, value + + leaves = list(walk(yml)) + # 1) 配置项名不得是口令/密钥类(注意排除 max_tokens 这类合法项) + risky_names = {'password', 'passwd', 'secret', 'secret_key', 'api_key', 'apikey', + 'token', 'access_token', 'auth_token'} + risky = [p for p, _ in leaves + if (lambda last: last in risky_names or last.endswith(('_password', '_secret', '_api_key')))( + p.split('.')[-1].lower())] + check('config.yml 无口令/密钥类配置项', risky, []) + # 2) 配置值不得等于 .env 中的真实密钥(防误粘贴) + real_secrets = {v for v in (os.getenv('MYSQL_PASSWORD'), os.getenv('DASHSCOPE_API_KEY'), + os.getenv('DEEPSEEK_API_KEY')) if v} + leaked = [p for p, v in leaves if isinstance(v, str) and v in real_secrets] + check('config.yml 未泄漏 .env 中的密钥值', leaked, []) + check('db_config 口令来自环境变量', + config.db_config()['password'] == os.getenv('MYSQL_PASSWORD'), True) + + print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}") + return 1 if failed else 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tests/test_fidelity_guard.py b/tests/test_fidelity_guard.py new file mode 100644 index 0000000..efee7ec --- /dev/null +++ b/tests/test_fidelity_guard.py @@ -0,0 +1,70 @@ +"""数值事实守卫回归测试(不依赖 pytest) + + python tests/test_fidelity_guard.py + +背景:LLM 校对(qwen)曾把 ASR 原文的事实改错——实测样例: + `自2027年1月1日起施行` -> `自2024年1月1日起施行` + `第十一届东方经济论坛` -> `第九届东方经济论坛` + `十五五` -> `十四五` +因此校对结果在落库前必须通过数值事实守卫:数值被改动则回退 ASR 原文。 + +注意守卫的取向是「宁可保留原文,也不接受 LLM 改数」: +- 书写形式差异(7 ↔ 七、2026 ↔ 二零二六)必须放行,否则会误伤大量正常校对; +- 任何真实数值变化必须拦下。 +""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from audioRead import _number_drift, _number_signature # noqa: E402 + +# (原文, 校对后, 说明, 是否应拦下) +CASES = [ + # 应当放行:仅书写形式/切分差异 + ('7月23号', '七月二十三号', '阿拉伯数字 ↔ 中文数字', False), + ('2026年', '二零二六年', '阿拉伯 ↔ 逐年读法', False), + ('2027年1月1日', '2027年1月1日', '完全相同', False), + ('第十五届', '第十五届', '序数未变', False), + ('增长55%', '增长百分之五十五', '百分比写法差异', False), + ('他3日在莫斯科', '他三日在莫斯科', '日期写法差异', False), + + # 应当拦下:真实数值改动 + ('自2027年1月1日起施行', '自2024年1月1日起施行', '年份篡改(实测样例)', True), + ('第十一届东方经济论坛', '第九届东方经济论坛', '届次篡改(实测样例)', True), + ('十五五规划', '十四五规划', '规划期篡改(实测样例)', True), + ('增长5.3%', '增长5.31%', '小数点篡改', True), + ('55个', '60个', '数量篡改', True), + ('2025年', '2023年', '年份篡改', True), + ('300亿元', '30亿元', '量级篡改', True), +] + + +def main(): + failed = 0 + for raw, corrected, desc, expect_flag in CASES: + only_raw, only_new = _number_drift(raw, corrected) + flagged = bool(only_raw or only_new) + if flagged == expect_flag: + tag = '拦下' if flagged else '放行' + print(f" ✓ {desc:20} -> {tag} {only_raw}{only_new}") + else: + failed += 1 + expect = '拦下' if expect_flag else '放行' + print(f" ✗ {desc:20} -> 期望{expect},实际{'拦下' if flagged else '放行'} {only_raw}{only_new}") + + # 签名函数自检:十五五/十四五 必须解析为不同的数值序列 + c1, s1 = _number_signature('十五五规划') + c2, s2 = _number_signature('十四五规划') + if c1 != c2: + print(f" ✓ 缩写解析 十五五={c1} 十四五={c2}(不同)") + else: + failed += 1 + print(f" ✗ 缩写解析失败:十五五与十四五都被解析为 {c1}") + + print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}") + return 1 if failed else 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tests/test_news_parse.py b/tests/test_news_parse.py new file mode 100644 index 0000000..63c970a --- /dev/null +++ b/tests/test_news_parse.py @@ -0,0 +1,84 @@ +"""newsProcess 解析逻辑回归测试(不依赖 pytest,直接运行) + + python tests/test_news_parse.py + +覆盖 docs/BUGS.md B3 在生产日志中出现过的真实失败形态: +- ```json 围栏包裹 +- 前后夹杂解释文字 +- {"news": [...]} / {"1": {...}, ...} 等对象包装 +- 数组里混入字符串元素(原实现报 `string indices must be integers`) +- 输出被截断(无法修复时必须显式失败,交由重试,而不是静默写坏数据) +""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from newsProcess import extract_news_rows, normalize_date # noqa: E402 + +DATE = '2026-09-23' +GOOD = '{"news_id": 1, "news_title": "标题一", "news_content": "正文一"}' +GOOD2 = '{"news_id": 2, "news_title": "标题二", "news_content": "正文二"}' + +CASES = [ + # (名称, 原始响应, 期望条数) + ("纯数组", f'[{GOOD}, {GOOD2}]', 2), + ("```json 围栏(生产实测)", f'```json\n[{GOOD}, {GOOD2}]\n```', 2), + ("``` 无语言标记", f'```\n[{GOOD}]\n```', 1), + ("前后夹解释文字", f'好的,结果如下:\n[{GOOD}, {GOOD2}]\n以上。', 2), + ("对象包装 news 键", f'{{"news": [{GOOD}, {GOOD2}]}}', 2), + ("对象包装 数字键", f'{{"1": {GOOD}, "2": {GOOD2}}}', 2), + ("数组混入字符串(原 string indices 报错)", f'["垃圾", {GOOD}, 123, {GOOD2}]', 2), + ("条目缺 news_content 被跳过", f'[{GOOD}, {{"news_id": 9, "news_title": "空"}}]', 1), + ("news_id 非数字回退序号", '[{"news_title": "t", "news_content": "c"}]', 1), +] + +FAIL_CASES = [ + ("空响应", ""), + ("非 JSON", "今天的新闻联播主要内容有……"), + ("被截断(生产实测 Unterminated string)", '{"news": [{"news_id": 1, "news_title": "标题", "news_content": "正文未结束'), + ("单条新闻对象(无列表)", GOOD), + ("所有条目都无正文", f'[{{"news_id": 1, "news_title": "只有标题"}}]'), + ("空数组", "[]"), +] + + +def main(): + failed = 0 + + for name, raw, expected in CASES: + try: + rows = extract_news_rows(raw, DATE) + assert len(rows) == expected, f"期望 {expected} 条,实际 {len(rows)} 条" + for row in rows: + assert row['news_date'] == DATE + assert isinstance(row['sub_id'], int) + assert row['news_content'].strip() + assert len(row['news_title']) <= 256 + print(f" ✓ {name} -> {len(rows)} 条") + except Exception as e: + failed += 1 + print(f" ✗ {name}: {type(e).__name__}: {e}") + + for name, raw in FAIL_CASES: + try: + rows = extract_news_rows(raw, DATE) + failed += 1 + print(f" ✗ {name}: 本应失败,却解析出 {len(rows)} 条") + except ValueError as e: + print(f" ✓ {name} -> 按预期抛 ValueError(触发重试): {str(e)[:48]}") + except Exception as e: + failed += 1 + print(f" ✗ {name}: 抛出了非 ValueError: {type(e).__name__}: {e}") + + # 日期归一化 + assert normalize_date('20260923') == DATE + assert normalize_date(DATE) == DATE + print(" ✓ 日期归一化 20260923 / 2026-09-23") + + print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}") + return 1 if failed else 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tests/test_tunnel.py b/tests/test_tunnel.py new file mode 100644 index 0000000..3789431 --- /dev/null +++ b/tests/test_tunnel.py @@ -0,0 +1,183 @@ +"""隧道自愈回归测试(不依赖 pytest,不碰真实隧道) + + python tests/test_tunnel.py + +覆盖: +- 端口探测 / 本机地址判定; +- 各分支状态:禁用、非本机、已可连、systemd 托管中等待、执行脚本后恢复、脚本不存在; +- **端口已通时绝不执行 autossh.sh**(避免在正常运行时多起进程抢端口); +- systemd 单元 active 时只等待、不抢端口; +- dry-run 不执行脚本。 + +测试用临时 config.yml(`XWLB_CONFIG_FILE`)+ 临时端口的假脚本, +不会触碰真实的 13306 与项目里的 autossh.sh。 +""" +import os +import socket +import subprocess +import sys +import tempfile +import time +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT)) + + +def free_port(): + with socket.socket() as s: + s.bind(('127.0.0.1', 0)) + return s.getsockname()[1] + + +def write_cfg(path, port, script, unit='xwlb-tunnel.service', enabled=1): + path.write_text(f"""mysql: + tunnel: + enabled: {enabled} + script: {script} + systemd_unit: {unit} + wait_seconds: 4 + connect_timeout: 1 + host: 127.0.0.1 + port: {port} + user: myquant + database: myquant +paths: + video_dir: xwlb_video + audio_dir: audio_processing +models: + asr_model: paraformer-realtime-v2 + correct_model: qwen3.8-flash + split_model: deepseek-flash +""", encoding='utf-8') + + +def make_fake_script(dirpath, port, marker): + """假 autossh.sh:起一个 TCP 监听并把"被执行过"记录到 marker 文件""" + p = dirpath / 'fake_tunnel.sh' + p.write_text(f"""#!/bin/bash +touch "{marker}" +setsid .venv/bin/python -m http.server {port} --bind 127.0.0.1 >/dev/null 2>&1 & +""", encoding='utf-8') + p.chmod(0o755) + return p + + +def kill_listener(port): + subprocess.run(['pkill', '-f', f'http[.]server {port}'], capture_output=True) + time.sleep(0.3) + + +def main(): + failed = 0 + tmp = Path(tempfile.mkdtemp(prefix='xwlb_tunnel_test_')) + + def check(desc, got, want): + nonlocal failed + if got == want: + print(f" ✓ {desc}: {got!r}") + else: + failed += 1 + print(f" ✗ {desc}: 期望 {want!r},实际 {got!r}") + + # 临时配置 + 重新加载(用环境变量指定,不污染真实 config.yml) + port = free_port() + script = make_fake_script(tmp, port, tmp / 'executed') + cfg = tmp / 'config.yml' + write_cfg(cfg, port, str(script)) + os.environ['XWLB_CONFIG_FILE'] = str(cfg) + + import config + config.reload_config() + import tunnel + + try: + print("端口探测:") + check('关闭的端口判定为不通', tunnel.port_open('127.0.0.1', free_port(), timeout=0.5), False) + srv = subprocess.Popen([sys.executable, '-m', 'http.server', str(port), '--bind', '127.0.0.1'], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + time.sleep(1.0) + check('已监听端口判定为可连', tunnel.port_open('127.0.0.1', port, timeout=1), True) + srv.terminate() + srv.wait(timeout=5) + + print("\n本机地址判定:") + for h in ('localhost', '127.0.0.1', '::1'): + check(f'{h} 视为本机', tunnel.is_local(h), True) + check('远端地址不视为本机', tunnel.is_local('db.example.com'), False) + + print("\n分支:端口已通(不得执行脚本):") + if (tmp / 'executed').exists(): + os.remove(tmp / 'executed') + srv = subprocess.Popen([sys.executable, '-m', 'http.server', str(port), '--bind', '127.0.0.1'], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + time.sleep(1.0) + r = tunnel.ensure_tunnel() + check('状态 = 已可连', r['status'], tunnel.ST_ALREADY_OPEN) + check('未执行隧道脚本', (tmp / 'executed').exists(), False) + srv.terminate() + srv.wait(timeout=5) + + print("\n分支:端口不通 + systemd 托管中(只等待,不抢端口):") + if (tmp / 'executed').exists(): + os.remove(tmp / 'executed') + tunnel.systemd_unit_active = lambda unit, timeout=5: True + r = tunnel.ensure_tunnel() + tunnel.systemd_unit_active = lambda unit, timeout=5: False # 还原 + check('状态 = 等待 systemd', r['status'], tunnel.ST_UNAVAILABLE) + check('未执行隧道脚本(不抢 systemd 的端口)', (tmp / 'executed').exists(), False) + + print("\n分支:端口不通 + 执行脚本后恢复:") + if (tmp / 'executed').exists(): + os.remove(tmp / 'executed') + r = tunnel.ensure_tunnel(wait_seconds=6) + check('状态 = 执行脚本后端口就绪', r['status'], tunnel.ST_SCRIPT_STARTED) + check('脚本确实被执行', (tmp / 'executed').exists(), True) + check('判定为已恢复', r['healed'], True) + check('端口现在可连', tunnel.port_open('127.0.0.1', port), True) + kill_listener(port) + + print("\n分支:脚本不存在 / 禁用 / 非本机:") + write_cfg(cfg, free_port(), str(tmp / '不存在.sh')) + config.reload_config() + r = tunnel.ensure_tunnel() + check('脚本不存在 → 不可用', r['status'], tunnel.ST_UNAVAILABLE) + check('且未判定为已恢复', r['healed'], False) + + write_cfg(cfg, free_port(), str(script), enabled=0) + config.reload_config() + check('enabled=0 → 跳过探测', tunnel.ensure_tunnel()['status'], tunnel.ST_DISABLED) + + write_cfg(cfg, free_port(), str(script)) + config.reload_config() + real_db_config = config.db_config + config.db_config = lambda *a, **k: {'host': 'db.example.com', 'port': 3306, + 'username': 'u', 'password': 'p', 'database': 'd'} + try: + check('远端地址 → 不适用隧道', tunnel.ensure_tunnel()['status'], tunnel.ST_NOT_LOCAL) + finally: + config.db_config = real_db_config # 必须还原,否则污染后续用例 + + print("\n分支:dry-run 不执行脚本:") + port2 = free_port() + write_cfg(cfg, port2, str(script)) + config.reload_config() + if (tmp / 'executed').exists(): + os.remove(tmp / 'executed') + r = tunnel.ensure_tunnel(dry_run=True) + check('dry-run 报告将执行脚本', 'dry-run' in r['detail'], True) + check('dry-run 未真的执行', (tmp / 'executed').exists(), False) + + finally: + os.environ.pop('XWLB_CONFIG_FILE', None) + kill_listener(port) + kill_listener(port) + import shutil + shutil.rmtree(tmp, ignore_errors=True) + + print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}") + return 1 if failed else 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tools/compare_raw_improve.py b/tools/compare_raw_improve.py new file mode 100644 index 0000000..bd8948d --- /dev/null +++ b/tools/compare_raw_improve.py @@ -0,0 +1,273 @@ +"""对比 news_raw / news_improve(并联动 news_content),评估 LLM 校对的必要性与 token 成本 + + .venv/bin/python tools/compare_raw_improve.py [--limit N] [--examples N] [--md 输出文件] + +分类口径(逐分片): + identical 原文与校对后完全相同 → 校对没做任何事 + punct_only 去掉标点/空白后相同 → 校对只动了标点、空白、断句 + numeral_only 再去掉数字(中文数字与阿拉伯数字)后相同 → 只动了数字写法或标点 + word_change 仍有差异 → 真正改了字词(可能修错别字,也可能是改写/篡改) + +token 估算:中文近似 1 字符 ≈ 1 token(保守上限,实际分词器约 0.6~1.0), +只用于比较两个方案的**相对**开销,绝对费用请按自己供应商单价换算。 +""" +import argparse +import re +import sys +from collections import Counter +from datetime import date +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from mysqlHandle import MySQLDB # noqa: E402 + +# 标点/空白(除中文、字母、数字之外的一切) +_NON_WORD = re.compile(r'[^\u4e00-\u9fffA-Za-z0-9]') +# 中文数字(与阿拉伯数字一并视为"数字") +_CN_NUM = '零〇一二两三四五六七八九十百千万亿' +_NON_WORD_OR_DIGIT = re.compile(r'[^\u4e00-\u9fffA-Za-z]') +_PUNCT_ONLY = re.compile(r'[\u4e00-\u9fffA-Za-z0-9]') + + +def strip_punct(text): + """去掉标点与空白,只留中文/字母/数字""" + return _PUNCT_ONLY.findall(text or '') + + +def strip_punct_and_digits(text): + """去掉标点、空白与所有数字(含中文数字)""" + return [ch for ch in (text or '') if ch not in _CN_NUM and not ch.isdigit() and _PUNCT_ONLY.match(ch)] + + +def punct_density(text): + """每 100 字符中的标点数""" + text = text or '' + if not text: + return 0.0 + return len(_NON_WORD.findall(text)) / len(text) * 100 + + +def title_marks(text): + """书名号《》出现次数(可读性的直观指标)""" + return (text or '').count('《') + + +def multiset_delta(a, b): + """a 相对 b 的字符多重集差异数(近似"改了多少字",O(n))""" + ca, cb = Counter(a or ''), Counter(b or '') + return sum((ca - cb).values()) + + +def classify(raw, improve): + if raw == improve: + return 'identical' + if strip_punct(raw) == strip_punct(improve): + return 'punct_only' + if strip_punct_and_digits(raw) == strip_punct_and_digits(improve): + return 'numeral_only' + return 'word_change' + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument('--limit', type=int, default=0, help='只取最近 N 天') + ap.add_argument('--examples', type=int, default=2, help='每类抽样条数') + ap.add_argument('--md', default='', help='同时写出 markdown 报告') + args = ap.parse_args() + + db = MySQLDB() + try: + where = '1=1' + params = None + if args.limit: + dates = db.query_data( + 'xwlb_daily', 'DISTINCT news_days d', + '1=1 ORDER BY news_days DESC LIMIT %s', (args.limit,)) + if not dates: + print('没有数据') + return 1 + d0 = min(r['d'] for r in dates) + where, params = 'news_days >= %s', (d0,) + rows = db.query_data( + 'xwlb_daily', 'news_days, daily_sub_id, news_raw, news_improve', + f'{where} ORDER BY news_days, daily_sub_id', params) + content_rows = db.query_data( + 'xwlb_daily_ext', 'news_date, sub_id, news_content', + '1=1 ORDER BY news_date, sub_id', None) + finally: + db.close() + + # ---------- 逐分片分类 ---------- + stats = {k: {'n': 0, 'len_raw': 0, 'len_imp': 0, 'delta': 0, 'delta_words': 0} + for k in ('identical', 'punct_only', 'numeral_only', 'word_change')} + removed_chars, added_chars = Counter(), Counter() + monthly = {} + examples = {k: [] for k in stats} + total_raw = total_imp = 0 + pd_raw = pd_imp = 0.0 + tm_raw = tm_imp = 0 + days = set() + day_changed = {} + for r in rows: + raw, imp = r['news_raw'] or '', r['news_improve'] or '' + cat = classify(raw, imp) + s = stats[cat] + s['n'] += 1 + s['len_raw'] += len(raw) + s['len_imp'] += len(imp) + s['delta'] += multiset_delta(raw, imp) + # 排除标点/数字后的"真实字词"改动量 + s['delta_words'] += multiset_delta(''.join(strip_punct_and_digits(raw)), + ''.join(strip_punct_and_digits(imp))) + if cat == 'word_change': + removed_chars.update((Counter(raw) - Counter(imp)).elements()) + added_chars.update((Counter(imp) - Counter(raw)).elements()) + key = str(r['news_days'])[:7] + m = monthly.setdefault(key, {'n': 0, 'same': 0}) + m['n'] += 1 + if cat == 'identical': + m['same'] += 1 + total_raw += len(raw) + total_imp += len(imp) + pd_raw += punct_density(raw) + pd_imp += punct_density(imp) + tm_raw += title_marks(raw) + tm_imp += title_marks(imp) + days.add(r['news_days']) + day_changed.setdefault(r['news_days'], 0) + if cat != 'identical': + day_changed[r['news_days']] += 1 + if len(examples[cat]) < args.examples: + examples[cat].append((r['news_days'], r['daily_sub_id'], raw, imp)) + + n = len(rows) or 1 + out = [] + def w(line=''): + print(line) + out.append(line) + + w(f"# news_raw vs news_improve 评估报告") + w() + w(f"样本:`xwlb_daily` {len(rows)} 个分片,覆盖 {len(days)} 天" + f"({min(days)} ~ {max(days)});`xwlb_daily_ext` {len(content_rows)} 条") + w() + w("## 1. 校对到底改了什么") + w() + w("| 分类 | 分片数 | 占比 | 平均改动字符数 | 含义 |") + w("|---|---|---|---|---|") + label = { + 'identical': '完全没改', + 'punct_only': '只改标点/空白/断句', + 'numeral_only': '只改数字写法或标点', + 'word_change': '**真的改了字词**', + } + for k in ('identical', 'punct_only', 'numeral_only', 'word_change'): + s = stats[k] + avg_delta = s['delta'] / s['n'] if s['n'] else 0 + w(f"| {k} | {s['n']} | {s['n'] / n * 100:.1f}% | {avg_delta:.1f} | {label[k]} |") + + wc = stats['word_change'] + if wc['n']: + w() + w(f"「真的改了字词」的 {wc['n']} 个分片里,**排除标点与数字后**的平均改动量只有 " + f"**{wc['delta_words'] / wc['n']:.1f} 字**" + f"(占分片平均长度 {wc['len_raw'] / wc['n']:.0f} 字的 " + f"{wc['delta_words'] / max(wc['len_raw'], 1) * 100:.1f}%)") + w() + w("改动最频繁的字符(raw 有而 improve 没有 → improve 新增):") + w() + w("| 被替换掉的字符 | 次数 | 新增的字符 | 次数 |") + w("|---|---|---|---|") + top_rm = removed_chars.most_common(10) + top_ad = added_chars.most_common(10) + for i in range(max(len(top_rm), len(top_ad))): + rm = top_rm[i] if i < len(top_rm) else ('', '') + ad = top_ad[i] if i < len(top_ad) else ('', '') + rm_ch = rm[0].replace(' ', '␠').replace('\n', '⏎') + ad_ch = ad[0].replace(' ', '␠').replace('\n', '⏎') + w(f"| `{rm_ch}` | {rm[1]} | `{ad_ch}` | {ad[1]} |") + + changed = n - stats['identical']['n'] + w() + w(f"- 校对**实际生效**的分片:{changed}/{n}({changed / n * 100:.1f}%);" + f"完全没动的:{stats['identical']['n']}/{n}({stats['identical']['n'] / n * 100:.1f}%)") + all_same_days = sum(1 for d in days if day_changed.get(d, 0) == 0) + w(f"- 全天 11 个分片**全都没被改**的天数:{all_same_days}/{len(days)}") + w() + w("### 可读性指标(越大越接近正式书面语)") + w() + w("| 指标 | 原文 news_raw | 校对后 news_improve | 变化 |") + w("|---|---|---|---|") + w(f"| 标点密度(每百字标点数) | {pd_raw / n:.2f} | {pd_imp / n:.2f} | " + f"{(pd_imp - pd_raw) / n:+.2f} |") + w(f"| 书名号《总数 | {tm_raw} | {tm_imp} | {tm_imp - tm_raw:+d} |") + w(f"| 总字数 | {total_raw} | {total_imp} | {total_imp - total_raw:+d} |") + + if stats['word_change']['n']: + w() + w("### 「真的改了字词」的抽样(判断是修错字还是改写)") + for d, sid, raw, imp in examples['word_change']: + w() + w(f"- **{d} 第{sid}片**(改动约 {multiset_delta(raw, imp)} 字)") + w(f" - raw: {raw[:120]}") + w(f" - imp: {imp[:120]}") + + w() + w("### 按月的「完全没改」比例(用于识别校对失效的历史区间)") + w() + w("| 月份 | 分片数 | 完全没改 | 占比 |") + w("|---|---|---|---|") + for key in sorted(monthly): + m = monthly[key] + w(f"| {key} | {m['n']} | {m['same']} | {m['same'] / m['n'] * 100:.0f}% |") + + # ---------- token 预算 ---------- + w() + w("## 2. token 开销:现状 vs 两个替代方案") + w() + by_day_raw, by_day_imp = {}, {} + for r in rows: + by_day_raw[r['news_days']] = by_day_raw.get(r['news_days'], 0) + len(r['news_raw'] or '') + by_day_imp[r['news_days']] = by_day_imp.get(r['news_days'], 0) + len(r['news_improve'] or '') + by_day_content = {} + for r in content_rows: + if r['news_date'] in by_day_raw: + by_day_content[r['news_date']] = by_day_content.get(r['news_date'], 0) + len(r['news_content'] or '') + + common = sorted(set(by_day_raw) & set(by_day_content)) + if common: + raw_c = sum(by_day_raw[d] for d in common) / len(common) + imp_c = sum(by_day_imp[d] for d in common) / len(common) + con_c = sum(by_day_content[d] for d in common) / len(common) + corr_in, corr_out = raw_c, imp_c + split_in, split_out = imp_c, con_c + merged_in, merged_out = raw_c, con_c + cur = corr_in + corr_out + split_in + split_out + merged = merged_in + merged_out + off = raw_c + con_c + w(f"统计口径:{len(common)} 天同时有原文与精编的天数,取日均字符数(≈token 数,中文 1 字≈1 token)") + w() + w("| 方案 | 输入 | 输出 | 合计/天 | 相对现状 |") + w("|---|---|---|---|---|") + w(f"| A 现状(校对 + 切分 两次调用) | {corr_in + split_in:.0f} | {corr_out + split_out:.0f} | **{cur:.0f}** | — |") + w(f"| B 合并进切分(一次调用同时校对+切分) | {merged_in:.0f} | {merged_out:.0f} | **{merged:.0f}** | " + f"省 {(1 - merged / cur) * 100:.0f}% |") + w(f"| C 关掉校对(直接用 ASR 原文切分) | {off:.0f} | {con_c:.0f} | **{off:.0f}** | " + f"省 {(1 - off / cur) * 100:.0f}% |") + w() + w(f"其中校对这一次调用本身消耗 输入 {corr_in:.0f} + 输出 {corr_out:.0f} = " + f"**{corr_in + corr_out:.0f} token/天**,占现状总开销的 " + f"{(corr_in + corr_out) / cur * 100:.0f}%。") + w() + w(f"按 {len(common)} 天累计:校对一项约消耗 " + f"{(corr_in + corr_out) * len(common) / 1e6:.2f} M token。") + + if args.md: + Path(args.md).write_text('\n'.join(out) + '\n', encoding='utf-8') + print(f"\n报告已写入 {args.md}") + return 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tools/experiment_merge_correct_split.py b/tools/experiment_merge_correct_split.py new file mode 100644 index 0000000..2d6adf0 --- /dev/null +++ b/tools/experiment_merge_correct_split.py @@ -0,0 +1,100 @@ +"""A/B 实验:两次调用(校对 → 切分) vs 一次调用(校对+切分合并) + + .venv/bin/python tools/experiment_merge_correct_split.py 20260904 + +只在真实数据上跑**一次**额外调用(DeepSeek 切分环节所配置的模型), +对比两条路线的产出,回答"把 correct 合并进 split 是否安全、省多少"。 + +关键指标:**覆盖率** = 合并输出中保留了多少比例的原文有效字符(非标点、非数字), +用来发现"模型为了切分/改写而丢内容"。 +""" +import json +import re +import sys +from collections import Counter +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +import config # noqa: E402 +from deepseek import deepseek_text # noqa: E402 +from mysqlHandle import MySQLDB # noqa: E402 +from newsProcess import extract_news_rows, _strip_code_fence # noqa: E402 + +MERGE_PROMPT = """请对下面的《新闻联播》转写文本同时完成两件事: + +1. **整理文本**:修正标点与断句、补全引号与书名号、规范日期与数字写法(如"9月26号"改"9月26日")。 + 严禁改动任何事实信息:数字、年份、日期、届次、数量、机构名、人名、地名、专有名词必须与原文完全一致。 + 严禁概括、缩写、改写、增删内容——必须逐字保留原意与全部信息。 +2. **切分新闻**:按新闻逻辑分割为独立条目,并给每条起一个标题。 + 遇到"国内快讯""国际快讯""联播快讯"时,按其下每条快讯分割。 + +只输出 JSON 数组,不要输出 markdown 代码块或任何解释: +[{"news_id": 1, "news_title": "标题", "news_content": "整理后的该条新闻全文"}] +""" + +_KEEP = re.compile(r'[\u4e00-\u9fffA-Za-z]') + + +def effective_chars(text): + """有效字符:中文与字母(用于覆盖率比较,排除标点与数字的写法差异)""" + return Counter(_KEEP.findall(text or '')) + + +def coverage(source, produced): + """produced 覆盖了 source 中多少比例的有效字符""" + src, prod = effective_chars(source), effective_chars(produced) + total = sum(src.values()) + if not total: + return 1.0 + kept = sum(min(v, prod.get(k, 0)) for k, v in src.items()) + return kept / total + + +def main(): + date_arg = sys.argv[1] if len(sys.argv) > 1 else '2026-09-04' + d10 = f"{date_arg[:4]}-{date_arg[4:6]}-{date_arg[6:]}" if len(date_arg) == 8 else date_arg + + db = MySQLDB() + try: + rows = db.query_data('xwlb_daily', 'daily_sub_id, news_raw, news_improve', + 'news_days = %s ORDER BY daily_sub_id', (d10,)) + cur = db.query_data('xwlb_daily_ext', 'sub_id, news_title, news_content', + 'news_date = %s ORDER BY sub_id', (d10,)) + finally: + db.close() + if not rows: + print(f"{d10} 在 xwlb_daily 中没有数据") + return 1 + + raw_all = '\n'.join(r['news_raw'] or '' for r in rows) + imp_all = '\n'.join(r['news_improve'] or '' for r in rows) + cur_all = '\n'.join(c['news_content'] or '' for c in cur) + + print(f"== {d10} 现状(两次调用)==") + print(f" 分片 {len(rows)} 个,原文 {len(raw_all)} 字,校对后 {len(imp_all)} 字,精编 {len(cur)} 条 / {len(cur_all)} 字") + print(f" 校对改写量(有效字符): 原文→校对后覆盖率 {coverage(raw_all, imp_all) * 100:.2f}%") + print(f" 切分改写量(有效字符): 校对后→精编覆盖率 {coverage(imp_all, cur_all) * 100:.2f}%") + print(f" 精编《数量 {cur_all.count('《')},引号 {cur_all.count(chr(0x201C))},换行 {cur_all.count(chr(10))}") + + print("\n== 实验:一次调用(校对+切分合并)==") + print(f" 模型: {config.model('split_model')} @ {config.openai_url('split')}") + response = deepseek_text(raw_all, MERGE_PROMPT, use_fallback=False) + items = extract_news_rows(response, d10) + merged_all = '\n'.join(it['news_content'] for it in items) + print(f" 得到 {len(items)} 条 / {len(merged_all)} 字") + print(f" 原文→合并产出覆盖率: {coverage(raw_all, merged_all) * 100:.2f}% ← 关键:越低说明丢内容") + print(f" 精编《数量 {merged_all.count('《')},引号 {merged_all.count(chr(0x201C))},换行 {merged_all.count(chr(10))}") + + from audioRead import _number_drift + drift = _number_drift(raw_all, merged_all) + print(f" 数值事实漂移: 原文独有={drift[0][:12]} 产出独有={drift[1][:12]}") + print(f" 条数对比: 现状 {len(cur)} 条 vs 合并 {len(items)} 条") + print("\n 合并产出的前 3 条标题:") + for it in items[:3]: + print(f" - {it['news_title'][:40]}") + return 0 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file diff --git a/tunnel.py b/tunnel.py new file mode 100644 index 0000000..794e13c --- /dev/null +++ b/tunnel.py @@ -0,0 +1,200 @@ +"""数据库隧道自愈:13306 不通就自动拉起 autossh.sh + +背景:数据库经 SSH 隧道访问(autossh.sh,本地 13306 → 远端 3306)。隧道断掉时, +任何入口都只会得到一句"MySQL 连接失败",不会自己恢复 —— 凌晨的定时任务因此整晚失败。 + +行为(挂在 `MySQLDB.connect()` 上,所以**所有入口都生效**,不只是定时任务): + +1. 探测 `host:port` 是否可连; +2. 可连 → 什么都不做(正常路径零开销,不做任何多余动作); +3. 不通 → + a. 若 systemd 隧道单元处于 active(说明由 systemd 托管,它会自动重连)→ **只等待**, + 不另起一个 autossh 去抢同一个端口(那会让 systemd 单元因端口被占而反复重启失败); + b. 否则执行 `config.yml` 里 `mysql.tunnel.script`(默认项目根的 `autossh.sh`); +4. 最多等 `wait_seconds` 秒,最后再探测一次并**如实报告**结果(不通就说通不了)。 + +配置(config.yml): + mysql: + tunnel: + enabled: 1 # 关掉则完全不探测 + script: autossh.sh # 相对项目根 + systemd_unit: xwlb-tunnel.service + wait_seconds: 30 + connect_timeout: 2 + +命令行自检: + python tunnel.py # 探测并按需修复,退出码 0=通 / 1=仍不通 + python tunnel.py --dry-run # 只看会做什么,不执行 +""" +import argparse +import logging +import os +import socket +import subprocess +import sys +import time +from pathlib import Path + +import config + +logger = logging.getLogger(__name__) + +LOCAL_HOSTS = {'localhost', '127.0.0.1', '::1', '0.0.0.0'} + +# 状态常量(便于调用方与测试断言,不要用字符串字面量比较) +ST_DISABLED = 'disabled' +ST_NOT_LOCAL = 'not-local' +ST_ALREADY_OPEN = 'already-open' +ST_SYSTEMD_WAITED = 'systemd-waited' +ST_SCRIPT_STARTED = 'script-started' +ST_UNAVAILABLE = 'unavailable' + + +def _cfg(dotted, default): + return config.get(f'mysql.tunnel.{dotted}', default) + + +def is_local(host) -> bool: + return str(host or '').strip().lower() in LOCAL_HOSTS + + +def port_open(host, port, timeout=2.0) -> bool: + """TCP 层探测端口是否可连(不涉及 MySQL 握手,快且无副作用)""" + try: + with socket.create_connection((host, int(port)), timeout=timeout): + return True + except (OSError, ValueError): + return False + + +def systemd_unit_active(unit, timeout=5) -> bool: + """systemd 单元是否 active;systemctl 不可用或单元不存在时返回 False""" + if not unit: + return False + try: + out = subprocess.run(['systemctl', 'is-active', str(unit)], + capture_output=True, text=True, timeout=timeout) + return out.stdout.strip() == 'active' + except (OSError, subprocess.SubprocessError): + return False + + +def _wait_for_port(host, port, seconds, interval=2.0): + """轮询等待端口就绪;返回实际等待秒数(未就绪返回总等待时长)""" + deadline = time.time() + max(0, seconds) + waited = 0.0 + while True: + if port_open(host, port, timeout=2.0): + return waited + if time.time() >= deadline: + return waited + time.sleep(interval) + waited = min(interval, max(0.0, seconds - waited)) + waited + + +def ensure_tunnel(host=None, port=None, reason='', dry_run=False, wait_seconds=None): + """确保数据库端口可连;不通则按配置自愈 + + 参数: + host/port: 不传则取 config.yml 的 mysql.host / mysql.port + reason: 写进日志的触发原因(如"MySQL 连接失败") + dry_run: 只报告会做什么,不执行脚本 + wait_seconds: 覆盖配置里的等待时长 + 返回值: + dict: {'status': 见 ST_* 常量, 'detail': 说明, 'waited': 等待秒数, + 'host': ..., 'port': ..., 'healed': bool} + """ + cfg = config.db_config() + host = host or cfg['host'] + port = int(port or cfg['port']) + prefix = f"({reason})" if reason else '' + result = {'host': host, 'port': port, 'waited': 0.0, 'healed': False} + + if not config.get_bool('mysql.tunnel.enabled', True): + result.update(status=ST_DISABLED, detail='配置 mysql.tunnel.enabled=0,跳过探测') + return result + + if not is_local(host): + # 直连远端数据库时没有隧道可言,不去跑 autossh.sh + result.update(status=ST_NOT_LOCAL, detail=f'{host} 不是本机地址,不适用 SSH 隧道') + return result + + if port_open(host, port, timeout=float(_cfg('connect_timeout', 2) or 2)): + result.update(status=ST_ALREADY_OPEN, detail=f'{host}:{port} 已可连') + return result + + wait_seconds = float(_cfg('wait_seconds', 30) if wait_seconds is None else wait_seconds) + unit = _cfg('systemd_unit', 'xwlb-tunnel.service') + + # 分支 a:systemd 托管中(它会自动重连),只等,不抢端口 + if not dry_run and systemd_unit_active(unit): + logger.warning("数据库端口 %s:%s 不通%s,systemd 单元 %s 处于 active,等待其自动重连…", + host, port, prefix, unit) + waited = _wait_for_port(host, port, wait_seconds) + ok = port_open(host, port) + result.update(status=ST_SYSTEMD_WAITED if ok else ST_UNAVAILABLE, + detail=(f'等待 systemd 单元 {unit} 重连' + ('成功' if ok else '超时')), + waited=waited, healed=ok) + if not ok: + logger.error("等待 %s 秒后 %s:%s 仍不通,请检查: systemctl status %s", + int(waited), host, port, unit) + return result + + # 分支 b:执行 autossh.sh + script = _cfg('script', 'autossh.sh') + script_path = Path(script) + if not script_path.is_absolute(): + script_path = Path(config.BASE_DIR) / script + + if dry_run: + result.update(status=ST_SCRIPT_STARTED, healed=False, + detail=f'[dry-run] 将执行 {script_path}') + return result + + if not script_path.exists(): + logger.error("隧道脚本不存在: %s(config.yml: mysql.tunnel.script)", script_path) + result.update(status=ST_UNAVAILABLE, detail=f'脚本不存在: {script_path}') + return result + + logger.warning("数据库端口 %s:%s 不通%s,执行 %s", host, port, prefix, script_path.name) + try: + proc = subprocess.run(['bash', str(script_path)], capture_output=True, text=True, timeout=60) + if proc.returncode != 0: + logger.warning("%s 退出码 %s: %s", script_path.name, proc.returncode, + (proc.stderr or proc.stdout or '').strip()[:300]) + except (OSError, subprocess.SubprocessError) as e: + logger.error("执行 %s 失败: %s", script_path, e) + + waited = _wait_for_port(host, port, wait_seconds) + ok = port_open(host, port) + result.update(status=ST_SCRIPT_STARTED if ok else ST_UNAVAILABLE, + detail=(f'执行 {script_path.name}' + ('后端口已就绪' if ok else '后端口仍不通')), + waited=waited, healed=ok) + if ok: + logger.info("✓ 隧道已恢复:%s:%s(等待 %.0f 秒)", host, port, waited) + else: + logger.error("执行 %s 并等待 %.0f 秒后 %s:%s 仍不通,请手动检查隧道", + script_path.name, waited, host, port) + return result + + +def main(): + ap = argparse.ArgumentParser(description='数据库隧道探测/自愈') + ap.add_argument('--dry-run', action='store_true', help='只报告会做什么,不执行脚本') + ap.add_argument('--quiet', action='store_true', help='只输出一行结果(供脚本调用)') + args = ap.parse_args() + + logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') + r = ensure_tunnel(reason='手动探测', dry_run=args.dry_run) + if args.quiet: + print(f"{r['status']} {r['host']}:{r['port']} {r['detail']}") + else: + print(f"状态: {r['status']}\n地址: {r['host']}:{r['port']}\n说明: {r['detail']}\n等待: {r['waited']:.0f}s") + # 退出码:端口可连才算成功(dry-run 不判定) + if args.dry_run: + return 0 + return 0 if port_open(r['host'], r['port']) else 1 + + +if __name__ == '__main__': + sys.exit(main()) \ No newline at end of file