Files
news/scripts/run_xwlb.py
T
simon 7ea8925209 fix: 修复 xwlb 日期错位与重复行问题(方案 A)
- run_xwlb: 新增 --date(处理日)参数,落盘改为处理日目录,下游 M2-M6 打通
- run_xwlb: 幂等落盘(按 url_hash 去重,重写 index 替代裸追加)
- pipeline: xwlb 步骤传 --date {date}
- run_extractor: xwlb publish_time 从 url 兜底解析真实播出日
- 新增 tests/test_run_xwlb.py 9 个测试
- 清理 60 天历史死数据(用户决策 A1)
- docs: 文档重构收尾(删除 deployment.md,已在服务器直接修改)
2026-08-22 19:11:32 +08:00

174 lines
6.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""M1 新闻联播 API 抓取脚本。
《新闻联播》每天 19:00 播出:日报/知识库在早上生成时,当日报表只能引用
前一天晚上已播出的联播,因此本脚本固定抓取「处理日 - 1 天」的节目,
但**落盘到「处理日」目录** data/raw/xwlb/{处理日}/。
日期语义说明:
- 处理日 = pipeline 本次批次的日期(--date,默认今天),与管道其它步骤一致;
- 数据日 = 处理日 - 1(昨晚 19:00 已播出的联播),业务约束来源:
当天的日报需要前一天晚上的新闻联播内容;
- 目录名使用处理日,使 M2→M6(extractor/dedup/llm/embedding/qdrant)
按同一日期扫目录即可处理 xwlb 数据,与其它新闻源完全一致。
幂等:同一天被定时任务多次触发(如 07:00/12:00/18:00/22:00)时,
数据日相同 → 抓取结果相同;按 url_hash 去重后只保留一份,
index.jsonl 重写为去重后的完整集(不再追加产生重复行)。
用法:
uv run python -m scripts.run_xwlb # 处理日 = 今天
uv run python -m scripts.run_xwlb --date 20260823 # 指定处理日
uv run python -m scripts.run_xwlb --output-root data/raw
"""
from __future__ import annotations
import argparse
import hashlib
import json
import sys
import urllib.request
from datetime import date, datetime, timedelta
from pathlib import Path
from loguru import logger
def _setup_logger(level: str) -> None:
logger.remove()
logger.add(
sys.stderr,
level=level,
format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}",
)
def _url_hash(s: str) -> str:
return hashlib.sha1(s.encode("utf-8")).hexdigest()[:16]
def _load_existing_index(index_path: Path) -> dict[str, str]:
"""读取已有 index.jsonl,返回 {url_hash: 原始行},用于幂等去重。
已存在同 url_hash 的条目视为已落盘,跳过写入;行内容原样保留。
文件不存在或损坏行忽略,不影响本次运行。
"""
existing: dict[str, str] = {}
if not index_path.is_file():
return existing
for line in index_path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except json.JSONDecodeError:
logger.warning("跳过 index.jsonl 非法行: {}", line[:80])
continue
key = rec.get("url_hash") or rec.get("url")
if key:
existing[key] = line
return existing
def main() -> int:
parser = argparse.ArgumentParser(description="新闻联播 API 抓取 (M1-xwlb)")
parser.add_argument(
"--date",
default=date.today().strftime("%Y%m%d"),
help="处理日 YYYYMMDD(知识库批次日,默认今日);抓取该日前一天播出的联播",
)
parser.add_argument("--output-root", default="data/raw")
parser.add_argument("--log-level", default="INFO")
args = parser.parse_args()
_setup_logger(args.log_level)
# 处理日 → 数据日(前一晚已播出的联播)
try:
process_day = datetime.strptime(args.date, "%Y%m%d").date()
except ValueError:
logger.error("--date 格式错误: {!r},应为 YYYYMMDD", args.date)
return 2
target_day = (process_day - timedelta(days=1)).strftime("%Y%m%d")
api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={target_day}&end_date={target_day}"
logger.info(
"请求新闻联播 API: {} (处理日={}, 数据日={})",
api_url, args.date, target_day,
)
try:
req = urllib.request.Request(api_url)
with urllib.request.urlopen(req, timeout=15) as resp:
body = json.loads(resp.read().decode("utf-8"))
except Exception as e:
logger.error("API 请求失败: {}", e)
return 2
raw_news = body.get("data", {}).get("news", [])
if not raw_news:
logger.warning("{} 无新闻联播数据", target_day)
return 0
# 输出目录: data/raw/xwlb/{处理日}/(目录 = pipeline 批次日)
out_dir = Path(args.output_root) / "xwlb" / args.date
out_dir.mkdir(parents=True, exist_ok=True)
index_path = out_dir / "index.jsonl"
# 幂等: 已有条目去重(同一天多次调度不会重复写入)
existing = _load_existing_index(index_path)
saved = 0
skipped = 0
dup_rows = 0
for n in raw_news:
sid = n.get("daily_sub_id", 0)
title = n.get("news_title", "")
content = n.get("news_improve", "")
news_day = n.get("news_days", target_day)
fake_url = f"xwlb://{news_day}/{sid}"
h = _url_hash(fake_url)
if h in existing:
skipped += 1
continue # 幂等: 已落盘过,跳过写入
html_file = f"{h}.html"
html_content = f"""<!DOCTYPE html>
<html><head><meta charset="utf-8"><title>{title}</title></head>
<body><article><h1>{title}</h1><div class="article-content">{content}</div></article></body>
</html>"""
(out_dir / f"{h}.html").write_text(html_content, encoding="utf-8")
meta = {
"source_id": "xwlb",
"stage": "article",
"url": fake_url,
"success": True,
"status_code": 200,
"title": title,
"error": None,
"fetched_at": datetime.now().isoformat(),
"attempts": 1,
"url_hash": h,
"html_file": html_file,
}
existing[h] = json.dumps(meta, ensure_ascii=False)
saved += 1
# 重写 index.jsonl 为去重后的完整集(原子写,替代裸追加)
if saved or existing:
tmp = index_path.with_suffix(".jsonl.tmp")
tmp.write_text("\n".join(existing.values()) + "\n", encoding="utf-8")
tmp.replace(index_path)
logger.info(
"新闻联播 数据日 {} 抓取完成: 新增 {} 条, 跳过已存在 {} 条 -> {}",
target_day, saved, skipped, out_dir,
)
return 0
if __name__ == "__main__":
raise SystemExit(main())