"""存储管理:index.jsonl 读写、输出目录管理""" import json import logging import re from pathlib import Path from crawler.models import ArticleItem, CrawlResult from crawler.utils import get_news_day logger = logging.getLogger(__name__) def _infer_news_day(result: CrawlResult) -> str | None: """从本次抓取结果中的文件路径推断新闻日 YYYYMMDD。 优先使用成功文章的 html_path / md_path 中的日期目录, 避免 crawl 跨天边界时 write_index 与文件落盘日期不一致。 """ for article in result.articles: if article.status != "success": continue path = article.html_path or article.md_path or "" m = re.search(r"/(\d{8})/", path) if m: return m.group(1) return None def write_index_jsonl(result: CrawlResult) -> Path: """将单源抓取结果写入 index.jsonl Args: result: 单源抓取结果 Returns: index 文件路径 """ today = _infer_news_day(result) or get_news_day() out_dir = Path(f"data/raw/{result.source_id}/{today}") out_dir.mkdir(parents=True, exist_ok=True) index_path = out_dir / "index.jsonl" # 追加写入(同一天多次抓取合并) existing: set[str] = set() if index_path.exists(): with open(index_path, encoding="utf-8") as f: for line in f: line = line.strip() if not line: continue try: item = json.loads(line) existing.add(item.get("url_hash", "")) except json.JSONDecodeError: continue written = 0 with open(index_path, "a", encoding="utf-8") as f: for article in result.articles: if article.status != "success": continue if article.url_hash in existing: continue # 跳过已存在的 record = article.model_dump() f.write(json.dumps(record, ensure_ascii=False) + "\n") existing.add(article.url_hash) written += 1 logger.info("[%s] index.jsonl 写入 %d 条(跳过重复 %d 条)", result.source_id, written, len(result.articles) - written) return index_path def load_index( source_id: str, date_str: str | None = None, ) -> list[ArticleItem]: """读取指定源/日期的 index.jsonl Args: source_id: 新闻源 ID date_str: 日期字符串 YYYYMMDD,默认当前新闻日 Returns: ArticleItem 列表 """ if date_str is None: date_str = get_news_day() index_path = Path(f"data/raw/{source_id}/{date_str}/index.jsonl") if not index_path.exists(): logger.warning("index 文件不存在: %s", index_path) return [] articles: list[ArticleItem] = [] with open(index_path, encoding="utf-8") as f: for line in f: line = line.strip() if not line: continue try: data = json.loads(line) articles.append(ArticleItem(**data)) except (json.JSONDecodeError, Exception) as e: logger.warning("解析 index 行失败: %s", e) return articles