"""A/B 实验:两次调用(校对 → 切分) vs 一次调用(校对+切分合并) .venv/bin/python tools/experiment_merge_correct_split.py 20260904 只在真实数据上跑**一次**额外调用(DeepSeek 切分环节所配置的模型), 对比两条路线的产出,回答"把 correct 合并进 split 是否安全、省多少"。 关键指标:**覆盖率** = 合并输出中保留了多少比例的原文有效字符(非标点、非数字), 用来发现"模型为了切分/改写而丢内容"。 """ import json import re import sys from collections import Counter from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) import config # noqa: E402 from deepseek import deepseek_text # noqa: E402 from mysqlHandle import MySQLDB # noqa: E402 from newsProcess import extract_news_rows, _strip_code_fence # noqa: E402 MERGE_PROMPT = """请对下面的《新闻联播》转写文本同时完成两件事: 1. **整理文本**:修正标点与断句、补全引号与书名号、规范日期与数字写法(如"9月26号"改"9月26日")。 严禁改动任何事实信息:数字、年份、日期、届次、数量、机构名、人名、地名、专有名词必须与原文完全一致。 严禁概括、缩写、改写、增删内容——必须逐字保留原意与全部信息。 2. **切分新闻**:按新闻逻辑分割为独立条目,并给每条起一个标题。 遇到"国内快讯""国际快讯""联播快讯"时,按其下每条快讯分割。 只输出 JSON 数组,不要输出 markdown 代码块或任何解释: [{"news_id": 1, "news_title": "标题", "news_content": "整理后的该条新闻全文"}] """ _KEEP = re.compile(r'[\u4e00-\u9fffA-Za-z]') def effective_chars(text): """有效字符:中文与字母(用于覆盖率比较,排除标点与数字的写法差异)""" return Counter(_KEEP.findall(text or '')) def coverage(source, produced): """produced 覆盖了 source 中多少比例的有效字符""" src, prod = effective_chars(source), effective_chars(produced) total = sum(src.values()) if not total: return 1.0 kept = sum(min(v, prod.get(k, 0)) for k, v in src.items()) return kept / total def main(): date_arg = sys.argv[1] if len(sys.argv) > 1 else '2026-09-04' d10 = f"{date_arg[:4]}-{date_arg[4:6]}-{date_arg[6:]}" if len(date_arg) == 8 else date_arg db = MySQLDB() try: rows = db.query_data('xwlb_daily', 'daily_sub_id, news_raw, news_improve', 'news_days = %s ORDER BY daily_sub_id', (d10,)) cur = db.query_data('xwlb_daily_ext', 'sub_id, news_title, news_content', 'news_date = %s ORDER BY sub_id', (d10,)) finally: db.close() if not rows: print(f"{d10} 在 xwlb_daily 中没有数据") return 1 raw_all = '\n'.join(r['news_raw'] or '' for r in rows) imp_all = '\n'.join(r['news_improve'] or '' for r in rows) cur_all = '\n'.join(c['news_content'] or '' for c in cur) print(f"== {d10} 现状(两次调用)==") print(f" 分片 {len(rows)} 个,原文 {len(raw_all)} 字,校对后 {len(imp_all)} 字,精编 {len(cur)} 条 / {len(cur_all)} 字") print(f" 校对改写量(有效字符): 原文→校对后覆盖率 {coverage(raw_all, imp_all) * 100:.2f}%") print(f" 切分改写量(有效字符): 校对后→精编覆盖率 {coverage(imp_all, cur_all) * 100:.2f}%") print(f" 精编《数量 {cur_all.count('《')},引号 {cur_all.count(chr(0x201C))},换行 {cur_all.count(chr(10))}") print("\n== 实验:一次调用(校对+切分合并)==") print(f" 模型: {config.model('split_model')} @ {config.openai_url('split')}") response = deepseek_text(raw_all, MERGE_PROMPT, use_fallback=False) items = extract_news_rows(response, d10) merged_all = '\n'.join(it['news_content'] for it in items) print(f" 得到 {len(items)} 条 / {len(merged_all)} 字") print(f" 原文→合并产出覆盖率: {coverage(raw_all, merged_all) * 100:.2f}% ← 关键:越低说明丢内容") print(f" 精编《数量 {merged_all.count('《')},引号 {merged_all.count(chr(0x201C))},换行 {merged_all.count(chr(10))}") from audioRead import _number_drift drift = _number_drift(raw_all, merged_all) print(f" 数值事实漂移: 原文独有={drift[0][:12]} 产出独有={drift[1][:12]}") print(f" 条数对比: 现状 {len(cur)} 条 vs 合并 {len(items)} 条") print("\n 合并产出的前 3 条标题:") for it in items[:3]: print(f" - {it['news_title'][:40]}") return 0 if __name__ == '__main__': sys.exit(main())