feat: 大模型使用场景化配置与去重多源记录

- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
  可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
2026-08-12 07:57:10 +08:00
parent 0c032196d2
commit 3c65701449
21 changed files with 886 additions and 80 deletions
+83 -6
View File
@@ -2,9 +2,12 @@
输入: data/processed/{source}/{YYYYMMDD}/*.json (M2 产物)
输出:
- 指纹库:data/dedup/fingerprints.sqlite3
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json
- 指纹库:data/dedup/fingerprints.sqlite3 (source_ids 列记录多源)
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json (含 sources 多源字段)
- 多源记录:data/deduped/{YYYYMMDD}/sources.json
{url_hash: [source_id, ...]},一条唯一新闻的全部来源
- 重复记录:data/deduped/{YYYYMMDD}/duplicates.jsonl
(含 matched_source_id / matched_source_ids)
用法:
uv run python -m scripts.run_dedup # 处理今日全部源
@@ -56,14 +59,60 @@ def _list_source_dirs(processed_root: Path) -> list[str]:
return sorted(p.name for p in processed_root.iterdir() if p.is_dir())
def _write_unique(
url_hash: str,
article: Article,
uniques_dir: Path,
sources_map: dict[str, list[str]],
) -> None:
"""写 uniques JSON,附加 sources 多源字段(向后兼容:下游 Pydantic 忽略多余字段)。"""
data = json.loads(article.model_dump_json())
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [article.source_id])))
out_path = uniques_dir / f"{url_hash}.json"
out_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
def _update_unique_sources(
url_hash: str,
uniques_dir: Path,
sources_map: dict[str, list[str]],
) -> None:
"""仅更新已存在 uniques 文件的 sources 字段(不覆盖原文内容)。
跨日命中时对应 uniques 文件在历史日期目录,不在本次处理范围,以指纹库为准。
"""
uniq_path = uniques_dir / f"{url_hash}.json"
if not uniq_path.is_file():
return
try:
data = json.loads(uniq_path.read_text(encoding="utf-8"))
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [])))
uniq_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
except (json.JSONDecodeError, OSError) as e:
logger.warning("更新 uniques 多源失败 {}: {}", uniq_path, e)
def _merge_sources(url_hash: str, new_source: str, sources_map: dict[str, list[str]]) -> None:
"""把新源并入 url_hash 的源列表(去重保序,主源居首)。"""
cur = sources_map.setdefault(url_hash, [])
if new_source not in cur:
cur.append(new_source)
def _process_source_day(
source_id: str,
day: str,
processed_root: Path,
out_root: Path,
deduper: Deduper,
sources_map: dict[str, list[str]],
) -> tuple[int, int, Counter]:
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。"""
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。
sources_map: 本次去重涉及内容组的 url_hash -> 全部来源列表(跨源累积,
最终写入 data/deduped/{day}/sources.json,供「显示新闻源」使用;
跨日命中的历史内容组也会记录,权威多源以指纹库 source_ids 列为准)。
"""
src_dir = processed_root / source_id / day
if not src_dir.is_dir():
logger.info("源 {} 日期 {} 无 processed 目录,跳过", source_id, day)
@@ -92,6 +141,11 @@ def _process_source_day(
dup_cnt += 1
if result.matched_layer is not None:
layer_cnt[result.matched_layer.value] += 1
# 记录多源:把被去重文章的源并入对应唯一新闻
if result.matched_url_hash:
_merge_sources(result.matched_url_hash, article.source_id, sources_map)
# 若该唯一新闻文件在当天目录,同步更新其 sources 字段
_update_unique_sources(result.matched_url_hash, uniques_dir, sources_map)
dup_f.write(
json.dumps(
{
@@ -107,6 +161,8 @@ def _process_source_day(
"matched_url": result.matched_url,
"matched_url_hash": result.matched_url_hash,
"matched_title": result.matched_title,
"matched_source_id": result.matched_source_id,
"matched_source_ids": result.all_source_ids,
"hamming_distance": result.hamming_distance,
},
ensure_ascii=False,
@@ -115,8 +171,8 @@ def _process_source_day(
)
else:
uniq_cnt += 1
out_path = uniques_dir / f"{article.url_hash}.json"
out_path.write_text(article.model_dump_json(indent=2), encoding="utf-8")
sources_map[article.url_hash] = [article.source_id]
_write_unique(article.url_hash, article, uniques_dir, sources_map)
total = uniq_cnt + dup_cnt
rate = dup_cnt / max(total, 1)
@@ -173,14 +229,35 @@ def main() -> int:
total_uniq = 0
total_dup = 0
total_layers: Counter = Counter()
# 当天唯一新闻 url_hash -> 全部来源列表(跨源累积,多源记录)
sources_map: dict[str, list[str]] = {}
for src in sources:
u, d, lc = _process_source_day(
src, args.date, processed_root, out_root, deduper
src, args.date, processed_root, out_root, deduper, sources_map
)
total_uniq += u
total_dup += d
total_layers.update(lc)
# 多源记录汇总:data/deduped/{day}/sources.json
# {url_hash: [source_id, ...]},配合 uniques/{url_hash}.json 的 sources 字段
# 与指纹库 source_ids 列,提供「一条唯一新闻多个来源」的完整记录。
sources_path = out_root / args.date / "sources.json"
sources_path.write_text(
json.dumps(
{k: v for k, v in sources_map.items() if v},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
logger.info(
"多源记录已写入 {} ({} 条唯一新闻,{} 条含多源)",
sources_path,
len(sources_map),
sum(1 for v in sources_map.values() if len(v) > 1),
)
total = total_uniq + total_dup
rate = total_dup / max(total, 1)
logger.info(