refactor: 清理历史 AI Agent 文档残留 + 重构 docs/ + Pipeline 健壮性修复
- docs: 删除 CLAUDE.md / continuation.md / english-news-plan.md 及旧版 intlnews_usage.*,
统一迁移到 docs/{README,architecture,quickstart,usage,pipeline,configuration,deployment,development,faq}.md
- README: 精简为仓库入口,指向 docs/
- configs/sources.yaml: 更新注释指向新文档
- .env.example: 修正 DashScope Embedding 端点说明
Pipeline 修复:
- dedup/llm/embedding/vectorstore/reporter: 过滤 M2 no_content / 空正文,避免污染下游与 Qdrant
- dedup/pipeline: 改为先写唯一文件再写指纹,避免崩溃导致文章永久丢失
- crawler/orchestrator: sources_crawled 改为“尝试数”,成功数 = crawled - failed
- crawler/storage: write_index_jsonl 从文章路径推断日期,修复跨天/测试路径问题
- scheduler/pipeline: STEP_TIMEOUTS 实际生效(SIGALRM)
- scheduler/reporter: emb_count 排除 index.json;日报跳过无原文事件
- vectorstore/pipeline: payload 增加 source_ids;--recreate --all 时空日期也重建 collection
- app/cli: extract/dedup/translate/embed/index/pipeline 支持 --date;embed/index 支持 --all;crawl 全源失败返回非零
- scripts: domestic_full/crawl_8g/crawl_2g/pipeline 安全加载 .env;M1 全失败不标记且最终退出码=1
This commit is contained in:
+45
-1
@@ -4,6 +4,7 @@
|
||||
"""
|
||||
|
||||
import logging
|
||||
import signal
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime
|
||||
@@ -30,6 +31,47 @@ STEP_TIMEOUTS: dict[str, int] = {
|
||||
}
|
||||
|
||||
|
||||
class _StepTimeout(Exception):
|
||||
"""步骤超时专用异常,避免与业务 TimeoutError 混淆。"""
|
||||
|
||||
|
||||
def _run_step_with_timeout(func, name: str, date_str: str, timeout: int | None) -> "StepResult":
|
||||
"""在支持 SIGALRM 的主线程中为单步执行添加超时保护。"""
|
||||
if timeout is None:
|
||||
return func(date_str)
|
||||
|
||||
started = datetime.now()
|
||||
|
||||
if not hasattr(signal, "SIGALRM"):
|
||||
return func(date_str)
|
||||
|
||||
def _handler(signum, frame): # noqa: ARG001
|
||||
raise _StepTimeout(f"step {name} timed out after {timeout}s")
|
||||
|
||||
try:
|
||||
old_handler = signal.getsignal(signal.SIGALRM)
|
||||
except (ValueError, OSError):
|
||||
# 非主线程无法设置信号处理器,直接不启用超时
|
||||
return func(date_str)
|
||||
|
||||
signal.signal(signal.SIGALRM, _handler)
|
||||
signal.setitimer(signal.ITIMER_REAL, timeout)
|
||||
try:
|
||||
return func(date_str)
|
||||
except _StepTimeout:
|
||||
elapsed = (datetime.now() - started).total_seconds()
|
||||
return StepResult(
|
||||
name=name,
|
||||
success=False,
|
||||
elapsed_sec=elapsed,
|
||||
message=f"超时(>{timeout}s)",
|
||||
started_at=started,
|
||||
)
|
||||
finally:
|
||||
signal.setitimer(signal.ITIMER_REAL, 0)
|
||||
signal.signal(signal.SIGALRM, old_handler)
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""单步执行结果。"""
|
||||
@@ -210,7 +252,9 @@ def run_pipeline(
|
||||
continue
|
||||
|
||||
logger.info("── 步骤 %s 开始 ──", name)
|
||||
sr = func(date_str)
|
||||
sr = _run_step_with_timeout(
|
||||
func, name, date_str, STEP_TIMEOUTS.get(name)
|
||||
)
|
||||
result.steps.append(sr)
|
||||
|
||||
flag = "✅" if sr.success else "❌"
|
||||
|
||||
@@ -210,6 +210,9 @@ def _load_events_window(now: datetime) -> list[dict]:
|
||||
continue
|
||||
try:
|
||||
data = json.loads(fp.read_text(encoding="utf-8"))
|
||||
# 防御性过滤:M2 no_content 残留不应进入日报统计/摘要。
|
||||
if not (data.get("content_en") or "").strip():
|
||||
continue
|
||||
# 时间过滤:publish_time 在 25 小时内
|
||||
pt_str = data.get("publish_time", "")
|
||||
pt = _try_parse_time(pt_str)
|
||||
@@ -267,7 +270,11 @@ def _collect_stats_window(now: datetime) -> dict:
|
||||
deduped += len(list(dedup_dir.glob("*.json")))
|
||||
emb_dir = Path(f"data/embeddings/{day_str}")
|
||||
if emb_dir.is_dir():
|
||||
emb_count += len(list(emb_dir.glob("*.json")))
|
||||
# 不把 index.json 计入实际向量文章数
|
||||
emb_count += len([
|
||||
f for f in emb_dir.glob("*.json")
|
||||
if f.name != "index.json"
|
||||
])
|
||||
for idx in Path("data/raw").glob(f"*/{day_str}/index.jsonl"):
|
||||
src = idx.parent.parent.name
|
||||
n = sum(1 for _ in open(idx, encoding="utf-8"))
|
||||
|
||||
Reference in New Issue
Block a user