- docs: 删除 CLAUDE.md / continuation.md / english-news-plan.md 及旧版 intlnews_usage.*,
统一迁移到 docs/{README,architecture,quickstart,usage,pipeline,configuration,deployment,development,faq}.md
- README: 精简为仓库入口,指向 docs/
- configs/sources.yaml: 更新注释指向新文档
- .env.example: 修正 DashScope Embedding 端点说明
Pipeline 修复:
- dedup/llm/embedding/vectorstore/reporter: 过滤 M2 no_content / 空正文,避免污染下游与 Qdrant
- dedup/pipeline: 改为先写唯一文件再写指纹,避免崩溃导致文章永久丢失
- crawler/orchestrator: sources_crawled 改为“尝试数”,成功数 = crawled - failed
- crawler/storage: write_index_jsonl 从文章路径推断日期,修复跨天/测试路径问题
- scheduler/pipeline: STEP_TIMEOUTS 实际生效(SIGALRM)
- scheduler/reporter: emb_count 排除 index.json;日报跳过无原文事件
- vectorstore/pipeline: payload 增加 source_ids;--recreate --all 时空日期也重建 collection
- app/cli: extract/dedup/translate/embed/index/pipeline 支持 --date;embed/index 支持 --all;crawl 全源失败返回非零
- scripts: domestic_full/crawl_8g/crawl_2g/pipeline 安全加载 .env;M1 全失败不标记且最终退出码=1
113 lines
3.2 KiB
Python
113 lines
3.2 KiB
Python
"""存储管理:index.jsonl 读写、输出目录管理"""
|
|
|
|
import json
|
|
import logging
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from crawler.models import ArticleItem, CrawlResult
|
|
from crawler.utils import get_news_day
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _infer_news_day(result: CrawlResult) -> str | None:
|
|
"""从本次抓取结果中的文件路径推断新闻日 YYYYMMDD。
|
|
|
|
优先使用成功文章的 html_path / md_path 中的日期目录,
|
|
避免 crawl 跨天边界时 write_index 与文件落盘日期不一致。
|
|
"""
|
|
for article in result.articles:
|
|
if article.status != "success":
|
|
continue
|
|
path = article.html_path or article.md_path or ""
|
|
m = re.search(r"/(\d{8})/", path)
|
|
if m:
|
|
return m.group(1)
|
|
return None
|
|
|
|
|
|
def write_index_jsonl(result: CrawlResult) -> Path:
|
|
"""将单源抓取结果写入 index.jsonl
|
|
|
|
Args:
|
|
result: 单源抓取结果
|
|
|
|
Returns:
|
|
index 文件路径
|
|
"""
|
|
today = _infer_news_day(result) or get_news_day()
|
|
out_dir = Path(f"data/raw/{result.source_id}/{today}")
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
index_path = out_dir / "index.jsonl"
|
|
|
|
# 追加写入(同一天多次抓取合并)
|
|
existing: set[str] = set()
|
|
if index_path.exists():
|
|
with open(index_path, encoding="utf-8") as f:
|
|
for line in f:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
item = json.loads(line)
|
|
existing.add(item.get("url_hash", ""))
|
|
except json.JSONDecodeError:
|
|
continue
|
|
|
|
written = 0
|
|
with open(index_path, "a", encoding="utf-8") as f:
|
|
for article in result.articles:
|
|
if article.status != "success":
|
|
continue
|
|
if article.url_hash in existing:
|
|
continue # 跳过已存在的
|
|
|
|
record = article.model_dump()
|
|
f.write(json.dumps(record, ensure_ascii=False) + "\n")
|
|
existing.add(article.url_hash)
|
|
written += 1
|
|
|
|
logger.info("[%s] index.jsonl 写入 %d 条(跳过重复 %d 条)",
|
|
result.source_id, written,
|
|
len(result.articles) - written)
|
|
return index_path
|
|
|
|
|
|
def load_index(
|
|
source_id: str,
|
|
date_str: str | None = None,
|
|
) -> list[ArticleItem]:
|
|
"""读取指定源/日期的 index.jsonl
|
|
|
|
Args:
|
|
source_id: 新闻源 ID
|
|
date_str: 日期字符串 YYYYMMDD,默认当前新闻日
|
|
|
|
Returns:
|
|
ArticleItem 列表
|
|
"""
|
|
if date_str is None:
|
|
date_str = get_news_day()
|
|
|
|
index_path = Path(f"data/raw/{source_id}/{date_str}/index.jsonl")
|
|
|
|
if not index_path.exists():
|
|
logger.warning("index 文件不存在: %s", index_path)
|
|
return []
|
|
|
|
articles: list[ArticleItem] = []
|
|
with open(index_path, encoding="utf-8") as f:
|
|
for line in f:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
data = json.loads(line)
|
|
articles.append(ArticleItem(**data))
|
|
except (json.JSONDecodeError, Exception) as e:
|
|
logger.warning("解析 index 行失败: %s", e)
|
|
|
|
return articles
|