fix: 修复 xwlb 日期错位与重复行问题(方案 A)

- run_xwlb: 新增 --date(处理日)参数,落盘改为处理日目录,下游 M2-M6 打通
- run_xwlb: 幂等落盘(按 url_hash 去重,重写 index 替代裸追加)
- pipeline: xwlb 步骤传 --date {date}
- run_extractor: xwlb publish_time 从 url 兜底解析真实播出日
- 新增 tests/test_run_xwlb.py 9 个测试
- 清理 60 天历史死数据(用户决策 A1)
- docs: 文档重构收尾(删除 deployment.md,已在服务器直接修改)
This commit is contained in:
2026-08-22 19:11:32 +08:00
parent 65ead54b4f
commit 7ea8925209
5 changed files with 306 additions and 19 deletions
+21
View File
@@ -115,9 +115,30 @@ def _process_one(rec: dict, raw_dir: Path, out_dir: Path,
logger.warning("提取失败 {} {}: {}", rec["source_id"], rec["url"], e.reason)
return None
# xwlb 的假 HTML 无时间节点,从 url(xwlb://YYYY-MM-DD/sid)兜底解析真实播出日
if src_id == "xwlb" and article.publish_time is None:
article = _fill_xwlb_publish_time(article)
return _save_article(article, out_dir)
def _fill_xwlb_publish_time(article: Article) -> Article:
"""从 xwlb url 解析播出日期填充 publish_time,失败时原样返回。"""
import re as _re
m = _re.match(r"xwlb://(\d{4}-\d{2}-\d{2})", article.url)
if not m:
return article
try:
pt = datetime.strptime(m.group(1), "%Y-%m-%d")
except ValueError:
return article
return article.model_copy(update={
"publish_time": pt,
"publish_time_raw": m.group(1),
})
def _process_cninfo(rec: dict, html_path: Path, out_dir: Path) -> Article | None:
"""处理 cninfo 公告记录:从 meta JSON 解析结构化数据。"""
import re
+85 -18
View File
@@ -1,10 +1,24 @@
"""M1 新闻联播 API 抓取脚本。
从 doorcome API /api/xwlbFine/ 获取 AI 精编的新闻联播条目,
转换为与 Web 抓取兼容的格式(data/raw/xwlb/{date}/),
供 M2-M6 管道统一处理。
《新闻联播》每天 19:00 播出:日报/知识库在早上生成时,当日报表只能引用
前一天晚上已播出的联播,因此本脚本固定抓取「处理日 - 1 天」的节目,
但**落盘到「处理日」目录** data/raw/xwlb/{处理日}/。
新闻联播晚间播出,始终抓取前一天数据,不依赖 --date 参数。
日期语义说明:
- 处理日 = pipeline 本次批次的日期(--date,默认今天),与管道其它步骤一致;
- 数据日 = 处理日 - 1(昨晚 19:00 已播出的联播),业务约束来源:
当天的日报需要前一天晚上的新闻联播内容;
- 目录名使用处理日,使 M2→M6(extractor/dedup/llm/embedding/qdrant)
按同一日期扫目录即可处理 xwlb 数据,与其它新闻源完全一致。
幂等:同一天被定时任务多次触发(如 07:00/12:00/18:00/22:00)时,
数据日相同 → 抓取结果相同;按 url_hash 去重后只保留一份,
index.jsonl 重写为去重后的完整集(不再追加产生重复行)。
用法:
uv run python -m scripts.run_xwlb # 处理日 = 今天
uv run python -m scripts.run_xwlb --date 20260823 # 指定处理日
uv run python -m scripts.run_xwlb --output-root data/raw
"""
from __future__ import annotations
@@ -33,19 +47,56 @@ def _url_hash(s: str) -> str:
return hashlib.sha1(s.encode("utf-8")).hexdigest()[:16]
def _load_existing_index(index_path: Path) -> dict[str, str]:
"""读取已有 index.jsonl,返回 {url_hash: 原始行},用于幂等去重。
已存在同 url_hash 的条目视为已落盘,跳过写入;行内容原样保留。
文件不存在或损坏行忽略,不影响本次运行。
"""
existing: dict[str, str] = {}
if not index_path.is_file():
return existing
for line in index_path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except json.JSONDecodeError:
logger.warning("跳过 index.jsonl 非法行: {}", line[:80])
continue
key = rec.get("url_hash") or rec.get("url")
if key:
existing[key] = line
return existing
def main() -> int:
parser = argparse.ArgumentParser(description="新闻联播 API 抓取 (M1-xwlb)")
parser.add_argument(
"--date",
default=date.today().strftime("%Y%m%d"),
help="处理日 YYYYMMDD(知识库批次日,默认今日);抓取该日前一天播出的联播",
)
parser.add_argument("--output-root", default="data/raw")
parser.add_argument("--log-level", default="INFO")
args = parser.parse_args()
_setup_logger(args.log_level)
# 新闻联播晚间播出,始终抓取前一天
day_str = (date.today() - timedelta(days=1)).strftime("%Y%m%d")
api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={day_str}&end_date={day_str}"
# 处理日 → 数据日(前一晚已播出的联播)
try:
process_day = datetime.strptime(args.date, "%Y%m%d").date()
except ValueError:
logger.error("--date 格式错误: {!r},应为 YYYYMMDD", args.date)
return 2
target_day = (process_day - timedelta(days=1)).strftime("%Y%m%d")
api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={target_day}&end_date={target_day}"
logger.info("请求新闻联播 API: {}", api_url)
logger.info(
"请求新闻联播 API: {} (处理日={}, 数据日={})",
api_url, args.date, target_day,
)
try:
req = urllib.request.Request(api_url)
with urllib.request.urlopen(req, timeout=15) as resp:
@@ -56,26 +107,34 @@ def main() -> int:
raw_news = body.get("data", {}).get("news", [])
if not raw_news:
logger.warning("{} 无新闻联播数据", day_str)
logger.warning("{} 无新闻联播数据", target_day)
return 0
# 准备输出目录
out_dir = Path(args.output_root) / "xwlb" / day_str
# 输出目录: data/raw/xwlb/{处理日}/(目录 = pipeline 批次日)
out_dir = Path(args.output_root) / "xwlb" / args.date
out_dir.mkdir(parents=True, exist_ok=True)
index_path = out_dir / "index.jsonl"
# 幂等: 已有条目去重(同一天多次调度不会重复写入)
existing = _load_existing_index(index_path)
saved = 0
skipped = 0
dup_rows = 0
for n in raw_news:
sid = n.get("daily_sub_id", 0)
title = n.get("news_title", "")
content = n.get("news_improve", "")
news_day = n.get("news_days", day_str)
news_day = n.get("news_days", target_day)
fake_url = f"xwlb://{news_day}/{sid}"
h = _url_hash(fake_url)
html_file = f"{h}.html"
if h in existing:
skipped += 1
continue # 幂等: 已落盘过,跳过写入
html_file = f"{h}.html"
html_content = f"""<!DOCTYPE html>
<html><head><meta charset="utf-8"><title>{title}</title></head>
<body><article><h1>{title}</h1><div class="article-content">{content}</div></article></body>
@@ -95,13 +154,21 @@ def main() -> int:
"url_hash": h,
"html_file": html_file,
}
with index_path.open("a", encoding="utf-8") as f:
f.write(json.dumps(meta, ensure_ascii=False) + "\n")
existing[h] = json.dumps(meta, ensure_ascii=False)
saved += 1
logger.info("新闻联播 {} 抓取完成: {} 条 -> {}", day_str, saved, out_dir)
# 重写 index.jsonl 为去重后的完整集(原子写,替代裸追加)
if saved or existing:
tmp = index_path.with_suffix(".jsonl.tmp")
tmp.write_text("\n".join(existing.values()) + "\n", encoding="utf-8")
tmp.replace(index_path)
logger.info(
"新闻联播 数据日 {} 抓取完成: 新增 {} 条, 跳过已存在 {} 条 -> {}",
target_day, saved, skipped, out_dir,
)
return 0
if __name__ == "__main__":
raise SystemExit(main())
raise SystemExit(main())