Files
xwlb/tests/test_news_parse.py
simon 60f8c263f2 《新闻联播》每日抓取入库:全链路 + 可移植化 + 定时任务
从 CCTV 主页抓取《新闻联播》,下载 → 转 MP3/WAV → 静音切分 → ASR 识别
→ LLM 校对 → 切分为单条新闻 → 入库 MySQL。

主要内容:
- 全链路:getVideo5 抓取下载、audioRead 转写、deepseek 校对与切分、newsProcess 入库
- 可移植化:配置分层,.env 只放密钥、config.yml 放模型/接入点/路由/参数
- 可换供应商:endpoints(kind/base_url/api_key_env/extra_body)+ routes 按环节选路
- 数据保真:数值事实守卫,校对改动数字/年份/届次则整片回退 ASR 原文;
  识别不完整不发布该日精编,避免半天内容被当成完整一天
- 定时任务:systemd 每天 21:00,失败 21:30 / 22:00 重试;
  只缺切分时只重跑切分(省掉全部 ASR),用 state/.asr_complete_* 标记判定阶段
- 隧道自愈:13306 不通时自动执行 autossh.sh(所有入口共用,systemd 托管时只等待)
- 中间产物每日清理;97 项离线自检(配置/清理/事实守卫/解析/隧道)
2026-09-25 11:17:46 +08:00

84 lines
3.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""newsProcess 解析逻辑回归测试(不依赖 pytest,直接运行)
python tests/test_news_parse.py
覆盖 docs/BUGS.md B3 在生产日志中出现过的真实失败形态:
- ```json 围栏包裹
- 前后夹杂解释文字
- {"news": [...]} / {"1": {...}, ...} 等对象包装
- 数组里混入字符串元素(原实现报 `string indices must be integers`)
- 输出被截断(无法修复时必须显式失败,交由重试,而不是静默写坏数据)
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from newsProcess import extract_news_rows, normalize_date # noqa: E402
DATE = '2026-09-23'
GOOD = '{"news_id": 1, "news_title": "标题一", "news_content": "正文一"}'
GOOD2 = '{"news_id": 2, "news_title": "标题二", "news_content": "正文二"}'
CASES = [
# (名称, 原始响应, 期望条数)
("纯数组", f'[{GOOD}, {GOOD2}]', 2),
("```json 围栏(生产实测)", f'```json\n[{GOOD}, {GOOD2}]\n```', 2),
("``` 无语言标记", f'```\n[{GOOD}]\n```', 1),
("前后夹解释文字", f'好的,结果如下:\n[{GOOD}, {GOOD2}]\n以上。', 2),
("对象包装 news 键", f'{{"news": [{GOOD}, {GOOD2}]}}', 2),
("对象包装 数字键", f'{{"1": {GOOD}, "2": {GOOD2}}}', 2),
("数组混入字符串(原 string indices 报错)", f'["垃圾", {GOOD}, 123, {GOOD2}]', 2),
("条目缺 news_content 被跳过", f'[{GOOD}, {{"news_id": 9, "news_title": "空"}}]', 1),
("news_id 非数字回退序号", '[{"news_title": "t", "news_content": "c"}]', 1),
]
FAIL_CASES = [
("空响应", ""),
("非 JSON", "今天的新闻联播主要内容有……"),
("被截断(生产实测 Unterminated string)", '{"news": [{"news_id": 1, "news_title": "标题", "news_content": "正文未结束'),
("单条新闻对象(无列表)", GOOD),
("所有条目都无正文", f'[{{"news_id": 1, "news_title": "只有标题"}}]'),
("空数组", "[]"),
]
def main():
failed = 0
for name, raw, expected in CASES:
try:
rows = extract_news_rows(raw, DATE)
assert len(rows) == expected, f"期望 {expected} 条,实际 {len(rows)} 条"
for row in rows:
assert row['news_date'] == DATE
assert isinstance(row['sub_id'], int)
assert row['news_content'].strip()
assert len(row['news_title']) <= 256
print(f" ✓ {name} -> {len(rows)} 条")
except Exception as e:
failed += 1
print(f" ✗ {name}: {type(e).__name__}: {e}")
for name, raw in FAIL_CASES:
try:
rows = extract_news_rows(raw, DATE)
failed += 1
print(f" ✗ {name}: 本应失败,却解析出 {len(rows)} 条")
except ValueError as e:
print(f" ✓ {name} -> 按预期抛 ValueError(触发重试): {str(e)[:48]}")
except Exception as e:
failed += 1
print(f" ✗ {name}: 抛出了非 ValueError: {type(e).__name__}: {e}")
# 日期归一化
assert normalize_date('20260923') == DATE
assert normalize_date(DATE) == DATE
print(" ✓ 日期归一化 20260923 / 2026-09-23")
print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}")
return 1 if failed else 0
if __name__ == '__main__':
sys.exit(main())