feat: 脚本执行显性输出当前阶段与 AI 模型供应商/名称

- pipeline.sh 每阶段 banner(━━━ M4 翻译+事件抽取(AI 大模型: deepseek / deepseek-v4-flash)━━━)
- ai_model_info() 从 system.yaml 读取场景模型(translation/daily_report/embedding)
- Python 层:M4/M5/日报 显性打印 provider/model(场景标注)
- 已同步 pi5 实测:客户端初始化日志含供应商/模型
This commit is contained in:
2026-08-12 11:20:51 +08:00
parent bdd936a3e4
commit 012cb615d3
7 changed files with 65 additions and 12 deletions
+31 -3
View File
@@ -43,10 +43,33 @@ STEP_LOG="$LOG_DIR/pipeline_$(date +%Y%m%d_%H%M%S).log"
LOG "══════ 全链路管道开始(resume=${RESUME})═══════"
LOG "步骤日志: ${STEP_LOG}(终端实时显示完整进度)"
# ── AI 模型信息读取(供阶段 banner 显性展示供应商/模型)──
ai_model_info() {
# $1: 场景名(translation / daily_report)或 "embedding"
# 输出格式: "供应商 / 模型"
.venv/bin/python3 -c "
import sys, yaml
scene = sys.argv[1]
raw = yaml.safe_load(open('configs/system.yaml', encoding='utf-8'))
if scene == 'embedding':
cfg = raw.get('embedding', {})
p = cfg.get('provider', '?')
m = cfg.get('dashscope_model', '?')
else:
base = raw.get('llm', {})
sc = (raw.get('llm_scenes', {}) or {}).get(scene, {})
merged = {**base, **sc}
p = merged.get('provider', '?')
m = merged.get('model') or merged.get(p + '_model', '?')
print(f'{p} / {m}')
" "$1"
}
# 加载 .env
export $(grep -v '^#' .env | grep -v '^$' | xargs 2>/dev/null || true)
# ── M2: 正文提取 ──
LOG "━━━ M2 正文提取 ━━━"
step_run M2_extract .venv/bin/python3 -c "
from extractor.pipeline import process_all_sources
stats = process_all_sources()
@@ -54,20 +77,23 @@ print(f'M2: {stats[\"total_articles\"]} 篇, {stats[\"elapsed_sec\"]:.0f}s')
"
# ── M3: 去重 ──
LOG "━━━ M3 三层去重 ━━━"
step_run M3_dedup .venv/bin/python3 -c "
from dedup.pipeline import dedup_all_sources
stats = dedup_all_sources()
print(f'M3: 唯一 {stats[\"unique\"]}/重复 {stats[\"duplicate\"]}, {stats[\"elapsed_sec\"]:.0f}s')
"
# ── M4: 翻译+事件 ──
# ── M4: 翻译+事件(AI 大模型)──
LOG "━━━ M4 翻译+事件抽取(AI 大模型: $(ai_model_info translation))━━━"
step_run M4_translate .venv/bin/python3 -c "
from llm.pipeline import translate_all_deduped
stats = translate_all_deduped()
print(f'M4: {stats[\"success\"]}/{stats[\"total\"]} 篇, {stats[\"elapsed_sec\"]:.0f}s')
"
# ── M5: 向量生成 ──
# ── M5: 向量生成(AI 大模型)──
LOG "━━━ M5 向量生成(AI 大模型: $(ai_model_info embedding))━━━"
step_run M5_embed .venv/bin/python3 -c "
from embedding.pipeline import embed_all_events
stats = embed_all_events()
@@ -75,13 +101,15 @@ print(f'M5: {stats[\"success\"]}/{stats[\"total\"]} 篇, {stats[\"elapsed_sec\"]
"
# ── M6: Qdrant 入库 ──
LOG "━━━ M6 Qdrant 入库 ━━━"
step_run M6_index .venv/bin/python3 -c "
from vectorstore.pipeline import ingest_all_embeddings
stats = ingest_all_embeddings()
print(f'M6: {stats[\"ingested\"]}/{stats[\"total\"]} 条, {stats[\"elapsed_sec\"]:.0f}s')
"
# ── 日报 ──
# ── 日报(AI 大模型)──
LOG "━━━ 日报生成(AI 大模型: $(ai_model_info daily_report))━━━"
step_run report .venv/bin/python3 -c "
from scheduler.reporter import generate_report
report_id = generate_report()