Files
xwlb/tools/experiment_merge_correct_split.py
simon 60f8c263f2 《新闻联播》每日抓取入库:全链路 + 可移植化 + 定时任务
从 CCTV 主页抓取《新闻联播》,下载 → 转 MP3/WAV → 静音切分 → ASR 识别
→ LLM 校对 → 切分为单条新闻 → 入库 MySQL。

主要内容:
- 全链路:getVideo5 抓取下载、audioRead 转写、deepseek 校对与切分、newsProcess 入库
- 可移植化:配置分层,.env 只放密钥、config.yml 放模型/接入点/路由/参数
- 可换供应商:endpoints(kind/base_url/api_key_env/extra_body)+ routes 按环节选路
- 数据保真:数值事实守卫,校对改动数字/年份/届次则整片回退 ASR 原文;
  识别不完整不发布该日精编,避免半天内容被当成完整一天
- 定时任务:systemd 每天 21:00,失败 21:30 / 22:00 重试;
  只缺切分时只重跑切分(省掉全部 ASR),用 state/.asr_complete_* 标记判定阶段
- 隧道自愈:13306 不通时自动执行 autossh.sh(所有入口共用,systemd 托管时只等待)
- 中间产物每日清理;97 项离线自检(配置/清理/事实守卫/解析/隧道)
2026-09-25 11:17:46 +08:00

100 lines
4.6 KiB
Python

"""A/B 实验:两次调用(校对 → 切分) vs 一次调用(校对+切分合并)
.venv/bin/python tools/experiment_merge_correct_split.py 20260904
只在真实数据上跑**一次**额外调用(DeepSeek 切分环节所配置的模型),
对比两条路线的产出,回答"把 correct 合并进 split 是否安全、省多少"。
关键指标:**覆盖率** = 合并输出中保留了多少比例的原文有效字符(非标点、非数字),
用来发现"模型为了切分/改写而丢内容"。
"""
import json
import re
import sys
from collections import Counter
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
import config # noqa: E402
from deepseek import deepseek_text # noqa: E402
from mysqlHandle import MySQLDB # noqa: E402
from newsProcess import extract_news_rows, _strip_code_fence # noqa: E402
MERGE_PROMPT = """请对下面的《新闻联播》转写文本同时完成两件事:
1. **整理文本**:修正标点与断句、补全引号与书名号、规范日期与数字写法(如"9月26号"改"9月26日")。
严禁改动任何事实信息:数字、年份、日期、届次、数量、机构名、人名、地名、专有名词必须与原文完全一致。
严禁概括、缩写、改写、增删内容——必须逐字保留原意与全部信息。
2. **切分新闻**:按新闻逻辑分割为独立条目,并给每条起一个标题。
遇到"国内快讯""国际快讯""联播快讯"时,按其下每条快讯分割。
只输出 JSON 数组,不要输出 markdown 代码块或任何解释:
[{"news_id": 1, "news_title": "标题", "news_content": "整理后的该条新闻全文"}]
"""
_KEEP = re.compile(r'[\u4e00-\u9fffA-Za-z]')
def effective_chars(text):
"""有效字符:中文与字母(用于覆盖率比较,排除标点与数字的写法差异)"""
return Counter(_KEEP.findall(text or ''))
def coverage(source, produced):
"""produced 覆盖了 source 中多少比例的有效字符"""
src, prod = effective_chars(source), effective_chars(produced)
total = sum(src.values())
if not total:
return 1.0
kept = sum(min(v, prod.get(k, 0)) for k, v in src.items())
return kept / total
def main():
date_arg = sys.argv[1] if len(sys.argv) > 1 else '2026-09-04'
d10 = f"{date_arg[:4]}-{date_arg[4:6]}-{date_arg[6:]}" if len(date_arg) == 8 else date_arg
db = MySQLDB()
try:
rows = db.query_data('xwlb_daily', 'daily_sub_id, news_raw, news_improve',
'news_days = %s ORDER BY daily_sub_id', (d10,))
cur = db.query_data('xwlb_daily_ext', 'sub_id, news_title, news_content',
'news_date = %s ORDER BY sub_id', (d10,))
finally:
db.close()
if not rows:
print(f"{d10} 在 xwlb_daily 中没有数据")
return 1
raw_all = '\n'.join(r['news_raw'] or '' for r in rows)
imp_all = '\n'.join(r['news_improve'] or '' for r in rows)
cur_all = '\n'.join(c['news_content'] or '' for c in cur)
print(f"== {d10} 现状(两次调用)==")
print(f" 分片 {len(rows)} 个,原文 {len(raw_all)} 字,校对后 {len(imp_all)} 字,精编 {len(cur)} 条 / {len(cur_all)} 字")
print(f" 校对改写量(有效字符): 原文→校对后覆盖率 {coverage(raw_all, imp_all) * 100:.2f}%")
print(f" 切分改写量(有效字符): 校对后→精编覆盖率 {coverage(imp_all, cur_all) * 100:.2f}%")
print(f" 精编《数量 {cur_all.count('《')},引号 {cur_all.count(chr(0x201C))},换行 {cur_all.count(chr(10))}")
print("\n== 实验:一次调用(校对+切分合并)==")
print(f" 模型: {config.model('split_model')} @ {config.openai_url('split')}")
response = deepseek_text(raw_all, MERGE_PROMPT, use_fallback=False)
items = extract_news_rows(response, d10)
merged_all = '\n'.join(it['news_content'] for it in items)
print(f" 得到 {len(items)} 条 / {len(merged_all)} 字")
print(f" 原文→合并产出覆盖率: {coverage(raw_all, merged_all) * 100:.2f}% ← 关键:越低说明丢内容")
print(f" 精编《数量 {merged_all.count('《')},引号 {merged_all.count(chr(0x201C))},换行 {merged_all.count(chr(10))}")
from audioRead import _number_drift
drift = _number_drift(raw_all, merged_all)
print(f" 数值事实漂移: 原文独有={drift[0][:12]} 产出独有={drift[1][:12]}")
print(f" 条数对比: 现状 {len(cur)} 条 vs 合并 {len(items)} 条")
print("\n 合并产出的前 3 条标题:")
for it in items[:3]:
print(f" - {it['news_title'][:40]}")
return 0
if __name__ == '__main__':
sys.exit(main())