feat: 日报改为每日一次 + 新增 LLM 使用/成本/缓存文档

- domestic_full.sh / pipeline.sh 新增 --no-report 参数:跳过日报生成,
  保持 M1 抓取 + M2→M6 管道不变,用于 12:00/18:00 只抓不生成日报
- crontab: 07:00 全量含日报,12:00/18:00 --no-report(日报每日一次,降低 LLM 费用)
- docs/llm-models.md: 各 LLM 调用点分别用的 provider/model 事实清单
- docs/llm-cost-analysis.md: 真实 token 消耗与费用构成、降本建议
- docs/llm-cache-optimization.md: DeepSeek 上下文缓存机制与命中率提升设计
- docs/README.md 增加上述三个文档索引
This commit is contained in:
2026-09-04 10:38:33 +08:00
parent 9e4f5b4c75
commit c1ab751d95
6 changed files with 347 additions and 16 deletions
+17 -10
View File
@@ -7,9 +7,11 @@
# 每天 06:00 / 12:00 / 18:00 / 22:00 各执行一次
# =============================================
# 用法:
# ./scripts/domestic_full.sh # 全新执行
# ./scripts/domestic_full.sh --resume # 从中断处继续(跳过已完成步骤,
# # 步骤状态见 data/run_state/{date}.state)
# ./scripts/domestic_full.sh # 全新执行(含日报)
# ./scripts/domestic_full.sh --resume # 从中断处继续(跳过已完成步骤,
# # 步骤状态见 data/run_state/{date}.state)
# ./scripts/domestic_full.sh --no-report # 只抓取 + M2→M6,跳过日报生成
# ./scripts/domestic_full.sh --no-report --resume
# =============================================
set -euo pipefail
@@ -19,10 +21,12 @@ cd "$PROJECT_DIR"
# ── 参数解析 ──
RESUME=0
NO_REPORT=0
for arg in "$@"; do
case "$arg" in
--resume) RESUME=1 ;;
*) echo "未知参数: ${arg}(支持 --resume)" >&2; exit 1 ;;
--no-report) NO_REPORT=1 ;;
*) echo "未知参数: ${arg}(支持 --resume / --no-report)" >&2; exit 1 ;;
esac
done
@@ -36,7 +40,7 @@ mkdir -p "$LOG_DIR"
FULL_LOG="$LOG_DIR/domestic_full_$(date +%Y%m%d_%H%M%S).log"
: > "$FULL_LOG"
LOG "══════ 国内全流程开始(resume=${RESUME})═══════"
LOG "══════ 国内全流程开始(resume=${RESUME} no_report=${NO_REPORT})═══════"
LOG "全流程日志: ${FULL_LOG}(终端实时显示完整进度)"
# ── 加载 .env ──
@@ -62,13 +66,16 @@ if step_should_run M1_crawl; then
fi
fi
# ── 2. M2→M6 管道(含日报)──
LOG "━━━ [2/2] M2→M6 管道 + 日报 ━━━"
if [ "$RESUME" = "1" ]; then
bash "$SCRIPT_DIR/pipeline.sh" --resume 2>&1 | tee -a "$FULL_LOG"
# ── 2. M2→M6 管道(+ 可选日报)──
if [ "$NO_REPORT" = "1" ]; then
LOG "━━━ [2/2] M2→M6 管道(已跳过日报,--no-report)━━━"
else
bash "$SCRIPT_DIR/pipeline.sh" 2>&1 | tee -a "$FULL_LOG"
LOG "━━━ [2/2] M2→M6 管道 + 日报 ━━━"
fi
PIPELINE_ARGS=""
[ "$RESUME" = "1" ] && PIPELINE_ARGS="$PIPELINE_ARGS --resume"
[ "$NO_REPORT" = "1" ] && PIPELINE_ARGS="$PIPELINE_ARGS --no-report"
bash "$SCRIPT_DIR/pipeline.sh" $PIPELINE_ARGS 2>&1 | tee -a "$FULL_LOG"
if [ "$M1_FAILED" -eq 1 ]; then
LOG "⚠️ 全流程结束,但 M1 抓取未成功,退出码=1"
+14 -6
View File
@@ -3,8 +3,10 @@
# 国内服务器:全链路管道 M2 → M3 → M4 → M5 → M6 → 日报
# =============================================
# 用法:
# ./scripts/pipeline.sh # 全新执行(步骤级不跳过)
# ./scripts/pipeline.sh --resume # 从中断处继续(跳过已完成步骤)
# ./scripts/pipeline.sh # 全新执行(步骤级不跳过)
# ./scripts/pipeline.sh --resume # 从中断处继续(跳过已完成步骤)
# ./scripts/pipeline.sh --no-report # 只跑 M2→M6,跳过日报生成
# ./scripts/pipeline.sh --no-report --resume
# 前提:domestic_sync.sh 已完成
# =============================================
# 各步骤"跳过已处理文件"(文件级增量,由各 Python 模块自带):
@@ -24,10 +26,12 @@ cd "$PROJECT_DIR"
# ── 参数解析 ──
RESUME=0
NO_REPORT=0
for arg in "$@"; do
case "$arg" in
--resume) RESUME=1 ;;
*) echo "未知参数: ${arg}(支持 --resume)" >&2; exit 1 ;;
--no-report) NO_REPORT=1 ;;
*) echo "未知参数: ${arg}(支持 --resume / --no-report)" >&2; exit 1 ;;
esac
done
@@ -40,7 +44,7 @@ LOG_DIR="logs"
mkdir -p "$LOG_DIR"
STEP_LOG="$LOG_DIR/pipeline_$(date +%Y%m%d_%H%M%S).log"
: > "$STEP_LOG"
LOG "══════ 全链路管道开始(resume=${RESUME})═══════"
LOG "══════ 全链路管道开始(resume=${RESUME} no_report=${NO_REPORT})═══════"
LOG "步骤日志: ${STEP_LOG}(终端实时显示完整进度)"
# ── AI 模型信息读取(供阶段 banner 显性展示供应商/模型)──
@@ -113,11 +117,15 @@ print(f'M6: {stats[\"ingested\"]}/{stats[\"total\"]} 条, {stats[\"elapsed_sec\"
"
# ── 日报(AI 大模型)──
LOG "━━━ 日报生成(AI 大模型: $(ai_model_info daily_report))━━━"
step_run report .venv/bin/python3 -c "
if [ "$NO_REPORT" = "1" ]; then
LOG "━━━ 日报生成(已跳过,--no-report)━━━"
else
LOG "━━━ 日报生成(AI 大模型: $(ai_model_info daily_report))━━━"
step_run report .venv/bin/python3 -c "
from scheduler.reporter import generate_report
report_id = generate_report()
print(f'日报: report_id={report_id}' if report_id is not None else '日报: 无数据/失败')
"
fi
LOG "══════ 全链路管道完成 ✅ ═══════"