- docs: 删除 CLAUDE.md / continuation.md / english-news-plan.md 及旧版 intlnews_usage.*,
统一迁移到 docs/{README,architecture,quickstart,usage,pipeline,configuration,deployment,development,faq}.md
- README: 精简为仓库入口,指向 docs/
- configs/sources.yaml: 更新注释指向新文档
- .env.example: 修正 DashScope Embedding 端点说明
Pipeline 修复:
- dedup/llm/embedding/vectorstore/reporter: 过滤 M2 no_content / 空正文,避免污染下游与 Qdrant
- dedup/pipeline: 改为先写唯一文件再写指纹,避免崩溃导致文章永久丢失
- crawler/orchestrator: sources_crawled 改为“尝试数”,成功数 = crawled - failed
- crawler/storage: write_index_jsonl 从文章路径推断日期,修复跨天/测试路径问题
- scheduler/pipeline: STEP_TIMEOUTS 实际生效(SIGALRM)
- scheduler/reporter: emb_count 排除 index.json;日报跳过无原文事件
- vectorstore/pipeline: payload 增加 source_ids;--recreate --all 时空日期也重建 collection
- app/cli: extract/dedup/translate/embed/index/pipeline 支持 --date;embed/index 支持 --all;crawl 全源失败返回非零
- scripts: domestic_full/crawl_8g/crawl_2g/pipeline 安全加载 .env;M1 全失败不标记且最终退出码=1
61 lines
1.8 KiB
Bash
Executable File
61 lines
1.8 KiB
Bash
Executable File
#!/bin/bash
|
||
# =============================================
|
||
# 国内服务器:2G headless stealth 抓取
|
||
# =============================================
|
||
# 适用场景:
|
||
# - 日常轻量补充抓取
|
||
# - RSS 优先 + headless stealth Web 回退
|
||
# - 可选 SOCKS5 代理(修改 profiles/2g_headless.yaml 中 proxy.enabled)
|
||
#
|
||
# 用法:
|
||
# bash scripts/domestic_crawl_2g.sh # 抓取全部源
|
||
# bash scripts/domestic_crawl_2g.sh reuters # 只抓取指定源
|
||
# =============================================
|
||
set -euo pipefail
|
||
|
||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||
PROJECT_DIR="$(dirname "$SCRIPT_DIR")"
|
||
cd "$PROJECT_DIR"
|
||
|
||
LOG() { echo "[$(date '+%Y-%m-%d %H:%M:%S')] $*"; }
|
||
|
||
LOG "══════ 国内 2G headless stealth 抓取 ══════"
|
||
LOG "Profile: 2g_headless"
|
||
LOG "内存上限: 1800 MB"
|
||
LOG "浏览器模式: stealth headless"
|
||
|
||
# ── 加载 .env ──
|
||
if [ -f .env ]; then
|
||
set -a
|
||
. ./.env
|
||
set +a
|
||
fi
|
||
|
||
# ── 设置 Profile ──
|
||
export EN_NEWS_PROFILE="2g_headless"
|
||
|
||
# ── 执行抓取 ──
|
||
SOURCE_ARG=""
|
||
if [ $# -ge 1 ]; then
|
||
SOURCE_ARG="--source $1"
|
||
LOG "目标源: $1"
|
||
else
|
||
LOG "目标源: 全部启用源"
|
||
fi
|
||
|
||
if PYTHONPATH=. .venv/bin/python3 -c "
|
||
from crawler.orchestrator import run_crawl_sync
|
||
import sys
|
||
source_filter = sys.argv[1] if len(sys.argv) > 1 else None
|
||
stats = run_crawl_sync(source_filter=source_filter)
|
||
print(f'抓取完成: {stats.sources_crawled} 源, {stats.total_articles} 篇')
|
||
if stats.sources_crawled > 0 and stats.sources_failed >= stats.sources_crawled:
|
||
print(f'❌ 所有源抓取失败(sources_failed={stats.sources_failed})', file=sys.stderr)
|
||
sys.exit(1)
|
||
" "$1"; then
|
||
LOG "══════ 2G 抓取完成 ✅ ══════"
|
||
else
|
||
LOG "❌ 2G 抓取失败:本次未标记完成"
|
||
exit 1
|
||
fi
|