refactor: 清理历史 AI Agent 文档残留 + 重构 docs/ + Pipeline 健壮性修复
- docs: 删除 CLAUDE.md / continuation.md / english-news-plan.md 及旧版 intlnews_usage.*,
统一迁移到 docs/{README,architecture,quickstart,usage,pipeline,configuration,deployment,development,faq}.md
- README: 精简为仓库入口,指向 docs/
- configs/sources.yaml: 更新注释指向新文档
- .env.example: 修正 DashScope Embedding 端点说明
Pipeline 修复:
- dedup/llm/embedding/vectorstore/reporter: 过滤 M2 no_content / 空正文,避免污染下游与 Qdrant
- dedup/pipeline: 改为先写唯一文件再写指纹,避免崩溃导致文章永久丢失
- crawler/orchestrator: sources_crawled 改为“尝试数”,成功数 = crawled - failed
- crawler/storage: write_index_jsonl 从文章路径推断日期,修复跨天/测试路径问题
- scheduler/pipeline: STEP_TIMEOUTS 实际生效(SIGALRM)
- scheduler/reporter: emb_count 排除 index.json;日报跳过无原文事件
- vectorstore/pipeline: payload 增加 source_ids;--recreate --all 时空日期也重建 collection
- app/cli: extract/dedup/translate/embed/index/pipeline 支持 --date;embed/index 支持 --all;crawl 全源失败返回非零
- scripts: domestic_full/crawl_8g/crawl_2g/pipeline 安全加载 .env;M1 全失败不标记且最终退出码=1
This commit is contained in:
@@ -95,13 +95,15 @@ async def crawl_all_sources(
|
||||
|
||||
# 串行执行每个源
|
||||
for source in sources:
|
||||
# sources_crawled 表示“已尝试的源数量”,包含成功和失败的源;
|
||||
# 因此成功数 = sources_crawled - sources_failed。
|
||||
stats.sources_crawled += 1
|
||||
result = await _crawl_single_source_with_storage(source)
|
||||
|
||||
if isinstance(result, Exception):
|
||||
logger.error("源抓取异常: %s", result)
|
||||
stats.sources_failed += 1
|
||||
else:
|
||||
stats.sources_crawled += 1
|
||||
stats.total_articles += result.total_success
|
||||
stats.results.append(result)
|
||||
if result.error:
|
||||
|
||||
+18
-1
@@ -2,6 +2,7 @@
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from crawler.models import ArticleItem, CrawlResult
|
||||
@@ -10,6 +11,22 @@ from crawler.utils import get_news_day
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _infer_news_day(result: CrawlResult) -> str | None:
|
||||
"""从本次抓取结果中的文件路径推断新闻日 YYYYMMDD。
|
||||
|
||||
优先使用成功文章的 html_path / md_path 中的日期目录,
|
||||
避免 crawl 跨天边界时 write_index 与文件落盘日期不一致。
|
||||
"""
|
||||
for article in result.articles:
|
||||
if article.status != "success":
|
||||
continue
|
||||
path = article.html_path or article.md_path or ""
|
||||
m = re.search(r"/(\d{8})/", path)
|
||||
if m:
|
||||
return m.group(1)
|
||||
return None
|
||||
|
||||
|
||||
def write_index_jsonl(result: CrawlResult) -> Path:
|
||||
"""将单源抓取结果写入 index.jsonl
|
||||
|
||||
@@ -19,7 +36,7 @@ def write_index_jsonl(result: CrawlResult) -> Path:
|
||||
Returns:
|
||||
index 文件路径
|
||||
"""
|
||||
today = get_news_day()
|
||||
today = _infer_news_day(result) or get_news_day()
|
||||
out_dir = Path(f"data/raw/{result.source_id}/{today}")
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user