从 CCTV 主页抓取《新闻联播》,下载 → 转 MP3/WAV → 静音切分 → ASR 识别 → LLM 校对 → 切分为单条新闻 → 入库 MySQL。 主要内容: - 全链路:getVideo5 抓取下载、audioRead 转写、deepseek 校对与切分、newsProcess 入库 - 可移植化:配置分层,.env 只放密钥、config.yml 放模型/接入点/路由/参数 - 可换供应商:endpoints(kind/base_url/api_key_env/extra_body)+ routes 按环节选路 - 数据保真:数值事实守卫,校对改动数字/年份/届次则整片回退 ASR 原文; 识别不完整不发布该日精编,避免半天内容被当成完整一天 - 定时任务:systemd 每天 21:00,失败 21:30 / 22:00 重试; 只缺切分时只重跑切分(省掉全部 ASR),用 state/.asr_complete_* 标记判定阶段 - 隧道自愈:13306 不通时自动执行 autossh.sh(所有入口共用,systemd 托管时只等待) - 中间产物每日清理;97 项离线自检(配置/清理/事实守卫/解析/隧道)
84 lines
3.3 KiB
Python
84 lines
3.3 KiB
Python
"""newsProcess 解析逻辑回归测试(不依赖 pytest,直接运行)
|
||
|
||
python tests/test_news_parse.py
|
||
|
||
覆盖 docs/BUGS.md B3 在生产日志中出现过的真实失败形态:
|
||
- ```json 围栏包裹
|
||
- 前后夹杂解释文字
|
||
- {"news": [...]} / {"1": {...}, ...} 等对象包装
|
||
- 数组里混入字符串元素(原实现报 `string indices must be integers`)
|
||
- 输出被截断(无法修复时必须显式失败,交由重试,而不是静默写坏数据)
|
||
"""
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||
|
||
from newsProcess import extract_news_rows, normalize_date # noqa: E402
|
||
|
||
DATE = '2026-09-23'
|
||
GOOD = '{"news_id": 1, "news_title": "标题一", "news_content": "正文一"}'
|
||
GOOD2 = '{"news_id": 2, "news_title": "标题二", "news_content": "正文二"}'
|
||
|
||
CASES = [
|
||
# (名称, 原始响应, 期望条数)
|
||
("纯数组", f'[{GOOD}, {GOOD2}]', 2),
|
||
("```json 围栏(生产实测)", f'```json\n[{GOOD}, {GOOD2}]\n```', 2),
|
||
("``` 无语言标记", f'```\n[{GOOD}]\n```', 1),
|
||
("前后夹解释文字", f'好的,结果如下:\n[{GOOD}, {GOOD2}]\n以上。', 2),
|
||
("对象包装 news 键", f'{{"news": [{GOOD}, {GOOD2}]}}', 2),
|
||
("对象包装 数字键", f'{{"1": {GOOD}, "2": {GOOD2}}}', 2),
|
||
("数组混入字符串(原 string indices 报错)", f'["垃圾", {GOOD}, 123, {GOOD2}]', 2),
|
||
("条目缺 news_content 被跳过", f'[{GOOD}, {{"news_id": 9, "news_title": "空"}}]', 1),
|
||
("news_id 非数字回退序号", '[{"news_title": "t", "news_content": "c"}]', 1),
|
||
]
|
||
|
||
FAIL_CASES = [
|
||
("空响应", ""),
|
||
("非 JSON", "今天的新闻联播主要内容有……"),
|
||
("被截断(生产实测 Unterminated string)", '{"news": [{"news_id": 1, "news_title": "标题", "news_content": "正文未结束'),
|
||
("单条新闻对象(无列表)", GOOD),
|
||
("所有条目都无正文", f'[{{"news_id": 1, "news_title": "只有标题"}}]'),
|
||
("空数组", "[]"),
|
||
]
|
||
|
||
|
||
def main():
|
||
failed = 0
|
||
|
||
for name, raw, expected in CASES:
|
||
try:
|
||
rows = extract_news_rows(raw, DATE)
|
||
assert len(rows) == expected, f"期望 {expected} 条,实际 {len(rows)} 条"
|
||
for row in rows:
|
||
assert row['news_date'] == DATE
|
||
assert isinstance(row['sub_id'], int)
|
||
assert row['news_content'].strip()
|
||
assert len(row['news_title']) <= 256
|
||
print(f" ✓ {name} -> {len(rows)} 条")
|
||
except Exception as e:
|
||
failed += 1
|
||
print(f" ✗ {name}: {type(e).__name__}: {e}")
|
||
|
||
for name, raw in FAIL_CASES:
|
||
try:
|
||
rows = extract_news_rows(raw, DATE)
|
||
failed += 1
|
||
print(f" ✗ {name}: 本应失败,却解析出 {len(rows)} 条")
|
||
except ValueError as e:
|
||
print(f" ✓ {name} -> 按预期抛 ValueError(触发重试): {str(e)[:48]}")
|
||
except Exception as e:
|
||
failed += 1
|
||
print(f" ✗ {name}: 抛出了非 ValueError: {type(e).__name__}: {e}")
|
||
|
||
# 日期归一化
|
||
assert normalize_date('20260923') == DATE
|
||
assert normalize_date(DATE) == DATE
|
||
print(" ✓ 日期归一化 20260923 / 2026-09-23")
|
||
|
||
print(f"\n结果: {'全部通过' if failed == 0 else f'{failed} 项失败'}")
|
||
return 1 if failed else 0
|
||
|
||
|
||
if __name__ == '__main__':
|
||
sys.exit(main()) |