- run_xwlb: 新增 --date(处理日)参数,落盘改为处理日目录,下游 M2-M6 打通
- run_xwlb: 幂等落盘(按 url_hash 去重,重写 index 替代裸追加)
- pipeline: xwlb 步骤传 --date {date}
- run_extractor: xwlb publish_time 从 url 兜底解析真实播出日
- 新增 tests/test_run_xwlb.py 9 个测试
- 清理 60 天历史死数据(用户决策 A1)
- docs: 文档重构收尾(删除 deployment.md,已在服务器直接修改)
174 lines
6.0 KiB
Python
174 lines
6.0 KiB
Python
"""M1 新闻联播 API 抓取脚本。
|
||
|
||
《新闻联播》每天 19:00 播出:日报/知识库在早上生成时,当日报表只能引用
|
||
前一天晚上已播出的联播,因此本脚本固定抓取「处理日 - 1 天」的节目,
|
||
但**落盘到「处理日」目录** data/raw/xwlb/{处理日}/。
|
||
|
||
日期语义说明:
|
||
- 处理日 = pipeline 本次批次的日期(--date,默认今天),与管道其它步骤一致;
|
||
- 数据日 = 处理日 - 1(昨晚 19:00 已播出的联播),业务约束来源:
|
||
当天的日报需要前一天晚上的新闻联播内容;
|
||
- 目录名使用处理日,使 M2→M6(extractor/dedup/llm/embedding/qdrant)
|
||
按同一日期扫目录即可处理 xwlb 数据,与其它新闻源完全一致。
|
||
|
||
幂等:同一天被定时任务多次触发(如 07:00/12:00/18:00/22:00)时,
|
||
数据日相同 → 抓取结果相同;按 url_hash 去重后只保留一份,
|
||
index.jsonl 重写为去重后的完整集(不再追加产生重复行)。
|
||
|
||
用法:
|
||
uv run python -m scripts.run_xwlb # 处理日 = 今天
|
||
uv run python -m scripts.run_xwlb --date 20260823 # 指定处理日
|
||
uv run python -m scripts.run_xwlb --output-root data/raw
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import hashlib
|
||
import json
|
||
import sys
|
||
import urllib.request
|
||
from datetime import date, datetime, timedelta
|
||
from pathlib import Path
|
||
|
||
from loguru import logger
|
||
|
||
|
||
def _setup_logger(level: str) -> None:
|
||
logger.remove()
|
||
logger.add(
|
||
sys.stderr,
|
||
level=level,
|
||
format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}",
|
||
)
|
||
|
||
|
||
def _url_hash(s: str) -> str:
|
||
return hashlib.sha1(s.encode("utf-8")).hexdigest()[:16]
|
||
|
||
|
||
def _load_existing_index(index_path: Path) -> dict[str, str]:
|
||
"""读取已有 index.jsonl,返回 {url_hash: 原始行},用于幂等去重。
|
||
|
||
已存在同 url_hash 的条目视为已落盘,跳过写入;行内容原样保留。
|
||
文件不存在或损坏行忽略,不影响本次运行。
|
||
"""
|
||
existing: dict[str, str] = {}
|
||
if not index_path.is_file():
|
||
return existing
|
||
for line in index_path.read_text(encoding="utf-8").splitlines():
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
try:
|
||
rec = json.loads(line)
|
||
except json.JSONDecodeError:
|
||
logger.warning("跳过 index.jsonl 非法行: {}", line[:80])
|
||
continue
|
||
key = rec.get("url_hash") or rec.get("url")
|
||
if key:
|
||
existing[key] = line
|
||
return existing
|
||
|
||
|
||
def main() -> int:
|
||
parser = argparse.ArgumentParser(description="新闻联播 API 抓取 (M1-xwlb)")
|
||
parser.add_argument(
|
||
"--date",
|
||
default=date.today().strftime("%Y%m%d"),
|
||
help="处理日 YYYYMMDD(知识库批次日,默认今日);抓取该日前一天播出的联播",
|
||
)
|
||
parser.add_argument("--output-root", default="data/raw")
|
||
parser.add_argument("--log-level", default="INFO")
|
||
args = parser.parse_args()
|
||
|
||
_setup_logger(args.log_level)
|
||
|
||
# 处理日 → 数据日(前一晚已播出的联播)
|
||
try:
|
||
process_day = datetime.strptime(args.date, "%Y%m%d").date()
|
||
except ValueError:
|
||
logger.error("--date 格式错误: {!r},应为 YYYYMMDD", args.date)
|
||
return 2
|
||
target_day = (process_day - timedelta(days=1)).strftime("%Y%m%d")
|
||
api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={target_day}&end_date={target_day}"
|
||
|
||
logger.info(
|
||
"请求新闻联播 API: {} (处理日={}, 数据日={})",
|
||
api_url, args.date, target_day,
|
||
)
|
||
try:
|
||
req = urllib.request.Request(api_url)
|
||
with urllib.request.urlopen(req, timeout=15) as resp:
|
||
body = json.loads(resp.read().decode("utf-8"))
|
||
except Exception as e:
|
||
logger.error("API 请求失败: {}", e)
|
||
return 2
|
||
|
||
raw_news = body.get("data", {}).get("news", [])
|
||
if not raw_news:
|
||
logger.warning("{} 无新闻联播数据", target_day)
|
||
return 0
|
||
|
||
# 输出目录: data/raw/xwlb/{处理日}/(目录 = pipeline 批次日)
|
||
out_dir = Path(args.output_root) / "xwlb" / args.date
|
||
out_dir.mkdir(parents=True, exist_ok=True)
|
||
index_path = out_dir / "index.jsonl"
|
||
|
||
# 幂等: 已有条目去重(同一天多次调度不会重复写入)
|
||
existing = _load_existing_index(index_path)
|
||
saved = 0
|
||
skipped = 0
|
||
dup_rows = 0
|
||
|
||
for n in raw_news:
|
||
sid = n.get("daily_sub_id", 0)
|
||
title = n.get("news_title", "")
|
||
content = n.get("news_improve", "")
|
||
news_day = n.get("news_days", target_day)
|
||
|
||
fake_url = f"xwlb://{news_day}/{sid}"
|
||
h = _url_hash(fake_url)
|
||
|
||
if h in existing:
|
||
skipped += 1
|
||
continue # 幂等: 已落盘过,跳过写入
|
||
|
||
html_file = f"{h}.html"
|
||
html_content = f"""<!DOCTYPE html>
|
||
<html><head><meta charset="utf-8"><title>{title}</title></head>
|
||
<body><article><h1>{title}</h1><div class="article-content">{content}</div></article></body>
|
||
</html>"""
|
||
(out_dir / f"{h}.html").write_text(html_content, encoding="utf-8")
|
||
|
||
meta = {
|
||
"source_id": "xwlb",
|
||
"stage": "article",
|
||
"url": fake_url,
|
||
"success": True,
|
||
"status_code": 200,
|
||
"title": title,
|
||
"error": None,
|
||
"fetched_at": datetime.now().isoformat(),
|
||
"attempts": 1,
|
||
"url_hash": h,
|
||
"html_file": html_file,
|
||
}
|
||
existing[h] = json.dumps(meta, ensure_ascii=False)
|
||
saved += 1
|
||
|
||
# 重写 index.jsonl 为去重后的完整集(原子写,替代裸追加)
|
||
if saved or existing:
|
||
tmp = index_path.with_suffix(".jsonl.tmp")
|
||
tmp.write_text("\n".join(existing.values()) + "\n", encoding="utf-8")
|
||
tmp.replace(index_path)
|
||
|
||
logger.info(
|
||
"新闻联播 数据日 {} 抓取完成: 新增 {} 条, 跳过已存在 {} 条 -> {}",
|
||
target_day, saved, skipped, out_dir,
|
||
)
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main()) |