fix: 修复 xwlb 日期错位与重复行问题(方案 A)
- run_xwlb: 新增 --date(处理日)参数,落盘改为处理日目录,下游 M2-M6 打通
- run_xwlb: 幂等落盘(按 url_hash 去重,重写 index 替代裸追加)
- pipeline: xwlb 步骤传 --date {date}
- run_extractor: xwlb publish_time 从 url 兜底解析真实播出日
- 新增 tests/test_run_xwlb.py 9 个测试
- 清理 60 天历史死数据(用户决策 A1)
- docs: 文档重构收尾(删除 deployment.md,已在服务器直接修改)
This commit is contained in:
@@ -115,9 +115,30 @@ def _process_one(rec: dict, raw_dir: Path, out_dir: Path,
|
||||
logger.warning("提取失败 {} {}: {}", rec["source_id"], rec["url"], e.reason)
|
||||
return None
|
||||
|
||||
# xwlb 的假 HTML 无时间节点,从 url(xwlb://YYYY-MM-DD/sid)兜底解析真实播出日
|
||||
if src_id == "xwlb" and article.publish_time is None:
|
||||
article = _fill_xwlb_publish_time(article)
|
||||
|
||||
return _save_article(article, out_dir)
|
||||
|
||||
|
||||
def _fill_xwlb_publish_time(article: Article) -> Article:
|
||||
"""从 xwlb url 解析播出日期填充 publish_time,失败时原样返回。"""
|
||||
import re as _re
|
||||
|
||||
m = _re.match(r"xwlb://(\d{4}-\d{2}-\d{2})", article.url)
|
||||
if not m:
|
||||
return article
|
||||
try:
|
||||
pt = datetime.strptime(m.group(1), "%Y-%m-%d")
|
||||
except ValueError:
|
||||
return article
|
||||
return article.model_copy(update={
|
||||
"publish_time": pt,
|
||||
"publish_time_raw": m.group(1),
|
||||
})
|
||||
|
||||
|
||||
def _process_cninfo(rec: dict, html_path: Path, out_dir: Path) -> Article | None:
|
||||
"""处理 cninfo 公告记录:从 meta JSON 解析结构化数据。"""
|
||||
import re
|
||||
|
||||
+85
-18
@@ -1,10 +1,24 @@
|
||||
"""M1 新闻联播 API 抓取脚本。
|
||||
|
||||
从 doorcome API /api/xwlbFine/ 获取 AI 精编的新闻联播条目,
|
||||
转换为与 Web 抓取兼容的格式(data/raw/xwlb/{date}/),
|
||||
供 M2-M6 管道统一处理。
|
||||
《新闻联播》每天 19:00 播出:日报/知识库在早上生成时,当日报表只能引用
|
||||
前一天晚上已播出的联播,因此本脚本固定抓取「处理日 - 1 天」的节目,
|
||||
但**落盘到「处理日」目录** data/raw/xwlb/{处理日}/。
|
||||
|
||||
新闻联播晚间播出,始终抓取前一天数据,不依赖 --date 参数。
|
||||
日期语义说明:
|
||||
- 处理日 = pipeline 本次批次的日期(--date,默认今天),与管道其它步骤一致;
|
||||
- 数据日 = 处理日 - 1(昨晚 19:00 已播出的联播),业务约束来源:
|
||||
当天的日报需要前一天晚上的新闻联播内容;
|
||||
- 目录名使用处理日,使 M2→M6(extractor/dedup/llm/embedding/qdrant)
|
||||
按同一日期扫目录即可处理 xwlb 数据,与其它新闻源完全一致。
|
||||
|
||||
幂等:同一天被定时任务多次触发(如 07:00/12:00/18:00/22:00)时,
|
||||
数据日相同 → 抓取结果相同;按 url_hash 去重后只保留一份,
|
||||
index.jsonl 重写为去重后的完整集(不再追加产生重复行)。
|
||||
|
||||
用法:
|
||||
uv run python -m scripts.run_xwlb # 处理日 = 今天
|
||||
uv run python -m scripts.run_xwlb --date 20260823 # 指定处理日
|
||||
uv run python -m scripts.run_xwlb --output-root data/raw
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -33,19 +47,56 @@ def _url_hash(s: str) -> str:
|
||||
return hashlib.sha1(s.encode("utf-8")).hexdigest()[:16]
|
||||
|
||||
|
||||
def _load_existing_index(index_path: Path) -> dict[str, str]:
|
||||
"""读取已有 index.jsonl,返回 {url_hash: 原始行},用于幂等去重。
|
||||
|
||||
已存在同 url_hash 的条目视为已落盘,跳过写入;行内容原样保留。
|
||||
文件不存在或损坏行忽略,不影响本次运行。
|
||||
"""
|
||||
existing: dict[str, str] = {}
|
||||
if not index_path.is_file():
|
||||
return existing
|
||||
for line in index_path.read_text(encoding="utf-8").splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
logger.warning("跳过 index.jsonl 非法行: {}", line[:80])
|
||||
continue
|
||||
key = rec.get("url_hash") or rec.get("url")
|
||||
if key:
|
||||
existing[key] = line
|
||||
return existing
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="新闻联播 API 抓取 (M1-xwlb)")
|
||||
parser.add_argument(
|
||||
"--date",
|
||||
default=date.today().strftime("%Y%m%d"),
|
||||
help="处理日 YYYYMMDD(知识库批次日,默认今日);抓取该日前一天播出的联播",
|
||||
)
|
||||
parser.add_argument("--output-root", default="data/raw")
|
||||
parser.add_argument("--log-level", default="INFO")
|
||||
args = parser.parse_args()
|
||||
|
||||
_setup_logger(args.log_level)
|
||||
|
||||
# 新闻联播晚间播出,始终抓取前一天
|
||||
day_str = (date.today() - timedelta(days=1)).strftime("%Y%m%d")
|
||||
api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={day_str}&end_date={day_str}"
|
||||
# 处理日 → 数据日(前一晚已播出的联播)
|
||||
try:
|
||||
process_day = datetime.strptime(args.date, "%Y%m%d").date()
|
||||
except ValueError:
|
||||
logger.error("--date 格式错误: {!r},应为 YYYYMMDD", args.date)
|
||||
return 2
|
||||
target_day = (process_day - timedelta(days=1)).strftime("%Y%m%d")
|
||||
api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={target_day}&end_date={target_day}"
|
||||
|
||||
logger.info("请求新闻联播 API: {}", api_url)
|
||||
logger.info(
|
||||
"请求新闻联播 API: {} (处理日={}, 数据日={})",
|
||||
api_url, args.date, target_day,
|
||||
)
|
||||
try:
|
||||
req = urllib.request.Request(api_url)
|
||||
with urllib.request.urlopen(req, timeout=15) as resp:
|
||||
@@ -56,26 +107,34 @@ def main() -> int:
|
||||
|
||||
raw_news = body.get("data", {}).get("news", [])
|
||||
if not raw_news:
|
||||
logger.warning("{} 无新闻联播数据", day_str)
|
||||
logger.warning("{} 无新闻联播数据", target_day)
|
||||
return 0
|
||||
|
||||
# 准备输出目录
|
||||
out_dir = Path(args.output_root) / "xwlb" / day_str
|
||||
# 输出目录: data/raw/xwlb/{处理日}/(目录 = pipeline 批次日)
|
||||
out_dir = Path(args.output_root) / "xwlb" / args.date
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
index_path = out_dir / "index.jsonl"
|
||||
|
||||
# 幂等: 已有条目去重(同一天多次调度不会重复写入)
|
||||
existing = _load_existing_index(index_path)
|
||||
saved = 0
|
||||
skipped = 0
|
||||
dup_rows = 0
|
||||
|
||||
for n in raw_news:
|
||||
sid = n.get("daily_sub_id", 0)
|
||||
title = n.get("news_title", "")
|
||||
content = n.get("news_improve", "")
|
||||
news_day = n.get("news_days", day_str)
|
||||
news_day = n.get("news_days", target_day)
|
||||
|
||||
fake_url = f"xwlb://{news_day}/{sid}"
|
||||
h = _url_hash(fake_url)
|
||||
html_file = f"{h}.html"
|
||||
|
||||
if h in existing:
|
||||
skipped += 1
|
||||
continue # 幂等: 已落盘过,跳过写入
|
||||
|
||||
html_file = f"{h}.html"
|
||||
html_content = f"""<!DOCTYPE html>
|
||||
<html><head><meta charset="utf-8"><title>{title}</title></head>
|
||||
<body><article><h1>{title}</h1><div class="article-content">{content}</div></article></body>
|
||||
@@ -95,13 +154,21 @@ def main() -> int:
|
||||
"url_hash": h,
|
||||
"html_file": html_file,
|
||||
}
|
||||
with index_path.open("a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(meta, ensure_ascii=False) + "\n")
|
||||
existing[h] = json.dumps(meta, ensure_ascii=False)
|
||||
saved += 1
|
||||
|
||||
logger.info("新闻联播 {} 抓取完成: {} 条 -> {}", day_str, saved, out_dir)
|
||||
# 重写 index.jsonl 为去重后的完整集(原子写,替代裸追加)
|
||||
if saved or existing:
|
||||
tmp = index_path.with_suffix(".jsonl.tmp")
|
||||
tmp.write_text("\n".join(existing.values()) + "\n", encoding="utf-8")
|
||||
tmp.replace(index_path)
|
||||
|
||||
logger.info(
|
||||
"新闻联播 数据日 {} 抓取完成: 新增 {} 条, 跳过已存在 {} 条 -> {}",
|
||||
target_day, saved, skipped, out_dir,
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user