feat: 大模型使用场景化配置与去重多源记录

- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
  可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
2026-08-12 07:57:10 +08:00
parent 0c032196d2
commit 3c65701449
21 changed files with 886 additions and 80 deletions
+40 -2
View File
@@ -3,10 +3,14 @@
注意:SimHash 是 64 位无符号整数,SQLite INTEGER 是 64 位有符号
(范围 [-2^63, 2^63-1])。直接存可能溢出/转负数,虽然 XOR 仍然
正确但语义混乱。这里统一存为 16 位 hex TEXT,避免符号问题。
source_ids 列存 JSON 数组文本(同一内容组全部来源);旧库无此列时
自动 ALTER TABLE 迁移,旧数据读取时回退为 [source_id]。
"""
from __future__ import annotations
import json
import sqlite3
from datetime import datetime, timedelta
from pathlib import Path
@@ -27,13 +31,17 @@ CREATE TABLE IF NOT EXISTS fingerprints (
url TEXT NOT NULL,
title TEXT NOT NULL,
publish_date TEXT,
ingested_at TEXT NOT NULL
ingested_at TEXT NOT NULL,
source_ids TEXT
);
CREATE INDEX IF NOT EXISTS idx_content_hash ON fingerprints(content_hash);
CREATE INDEX IF NOT EXISTS idx_publish_date ON fingerprints(publish_date);
CREATE INDEX IF NOT EXISTS idx_source_id ON fingerprints(source_id);
"""
# 兼容旧库:为已存在但缺少 source_ids 列的表补列
_ALTER_SQL = "ALTER TABLE fingerprints ADD COLUMN source_ids TEXT"
def _to_hex(simhash: int) -> str:
return f"{simhash:016x}"
@@ -43,6 +51,25 @@ def _from_hex(hex_str: str) -> int:
return int(hex_str, 16)
def _to_sources_json(source_ids: list[str]) -> str:
return json.dumps(source_ids, ensure_ascii=False)
def _from_sources_json(raw: str | None, fallback: str) -> list[str]:
"""解析 source_ids 列;NULL/损坏时回退 [主源]。"""
if not raw:
return [fallback]
try:
val = json.loads(raw)
except (TypeError, ValueError):
return [fallback]
if isinstance(val, list) and val:
# 保证主源在首位(兼容手改/旧数据)
cleaned = [s for s in val if s and s != fallback]
return [fallback, *cleaned]
return [fallback]
def _row_to_fp(row: sqlite3.Row) -> Fingerprint:
return Fingerprint(
url_hash=row["url_hash"],
@@ -53,6 +80,7 @@ def _row_to_fp(row: sqlite3.Row) -> Fingerprint:
title=row["title"],
publish_date=row["publish_date"],
ingested_at=datetime.fromisoformat(row["ingested_at"]),
source_ids=_from_sources_json(row["source_ids"], row["source_id"]),
)
@@ -67,8 +95,17 @@ class FingerprintStore:
)
self._conn.row_factory = sqlite3.Row
self._conn.executescript(_SCHEMA_SQL)
self._migrate_source_ids()
logger.debug("打开指纹库: {}", self.db_path)
def _migrate_source_ids(self) -> None:
"""旧库兼容:为缺少 source_ids 列的表补列(新库无需执行)。"""
try:
self._conn.execute(_ALTER_SQL)
logger.info("指纹库迁移:为 fingerprints 表新增 source_ids 列")
except sqlite3.OperationalError:
logger.debug("source_ids 列已存在,跳过迁移")
def close(self) -> None:
self._conn.close()
@@ -148,7 +185,7 @@ class FingerprintStore:
self._conn.execute(
"INSERT OR REPLACE INTO fingerprints "
"(url_hash, content_hash, simhash_hex, source_id, url, title, "
" publish_date, ingested_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
" publish_date, ingested_at, source_ids) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
(
fp.url_hash,
fp.content_hash,
@@ -158,6 +195,7 @@ class FingerprintStore:
fp.title,
fp.publish_date,
fp.ingested_at.isoformat(),
_to_sources_json(fp.source_ids),
),
)