refactor: 清理历史 AI Agent 文档残留 + 重构 docs/ + Pipeline 健壮性修复
- docs: 删除 CLAUDE.md / continuation.md / english-news-plan.md 及旧版 intlnews_usage.*,
统一迁移到 docs/{README,architecture,quickstart,usage,pipeline,configuration,deployment,development,faq}.md
- README: 精简为仓库入口,指向 docs/
- configs/sources.yaml: 更新注释指向新文档
- .env.example: 修正 DashScope Embedding 端点说明
Pipeline 修复:
- dedup/llm/embedding/vectorstore/reporter: 过滤 M2 no_content / 空正文,避免污染下游与 Qdrant
- dedup/pipeline: 改为先写唯一文件再写指纹,避免崩溃导致文章永久丢失
- crawler/orchestrator: sources_crawled 改为“尝试数”,成功数 = crawled - failed
- crawler/storage: write_index_jsonl 从文章路径推断日期,修复跨天/测试路径问题
- scheduler/pipeline: STEP_TIMEOUTS 实际生效(SIGALRM)
- scheduler/reporter: emb_count 排除 index.json;日报跳过无原文事件
- vectorstore/pipeline: payload 增加 source_ids;--recreate --all 时空日期也重建 collection
- app/cli: extract/dedup/translate/embed/index/pipeline 支持 --date;embed/index 支持 --all;crawl 全源失败返回非零
- scripts: domestic_full/crawl_8g/crawl_2g/pipeline 安全加载 .env;M1 全失败不标记且最终退出码=1
This commit is contained in:
+12
-3
@@ -10,7 +10,7 @@ from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
from crawler.utils import get_news_day
|
||||
from dedup.deduper import Deduper
|
||||
from dedup.deduper import Deduper, article_to_fingerprint
|
||||
from dedup.models import DedupResult
|
||||
from extractor.models import ProcessedArticle
|
||||
|
||||
@@ -59,7 +59,12 @@ def _load_processed_articles(
|
||||
continue
|
||||
try:
|
||||
data = json.loads(json_file.read_text(encoding="utf-8"))
|
||||
articles.append(ProcessedArticle(**data))
|
||||
article = ProcessedArticle(**data)
|
||||
# M2 可能写入 no_content / failed:这些文章没有有效正文,
|
||||
# 不应进入去重、翻译、向量化等下游环节。
|
||||
if article.status != "success" or not article.content.strip():
|
||||
continue
|
||||
articles.append(article)
|
||||
except (json.JSONDecodeError, Exception) as e:
|
||||
logger.warning("解析 processed JSON 失败 %s: %s", json_file, e)
|
||||
|
||||
@@ -99,7 +104,9 @@ def dedup_source(
|
||||
dup_count = 0
|
||||
|
||||
for article in articles:
|
||||
result = deduper.ingest(article)
|
||||
# 先只读判断,不写指纹;等唯一文件落盘成功后再写指纹,
|
||||
# 避免“指纹已入库但唯一文件未生成”导致该文章后续被当重复丢弃。
|
||||
result = deduper.check(article)
|
||||
|
||||
if result.is_duplicate:
|
||||
dup_count += 1
|
||||
@@ -121,6 +128,8 @@ def dedup_source(
|
||||
article.model_dump_json(indent=2, ensure_ascii=False),
|
||||
encoding="utf-8",
|
||||
)
|
||||
# 唯一文件落盘成功后再写指纹
|
||||
deduper.store.upsert(article_to_fingerprint(article))
|
||||
logger.debug("[%s] ✅ %s (%d words)",
|
||||
source_id, article.title[:40], article.word_count)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user