feat: 大模型使用场景化配置与去重多源记录
- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
+83
-6
@@ -2,9 +2,12 @@
|
||||
|
||||
输入: data/processed/{source}/{YYYYMMDD}/*.json (M2 产物)
|
||||
输出:
|
||||
- 指纹库:data/dedup/fingerprints.sqlite3
|
||||
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json
|
||||
- 指纹库:data/dedup/fingerprints.sqlite3 (source_ids 列记录多源)
|
||||
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json (含 sources 多源字段)
|
||||
- 多源记录:data/deduped/{YYYYMMDD}/sources.json
|
||||
{url_hash: [source_id, ...]},一条唯一新闻的全部来源
|
||||
- 重复记录:data/deduped/{YYYYMMDD}/duplicates.jsonl
|
||||
(含 matched_source_id / matched_source_ids)
|
||||
|
||||
用法:
|
||||
uv run python -m scripts.run_dedup # 处理今日全部源
|
||||
@@ -56,14 +59,60 @@ def _list_source_dirs(processed_root: Path) -> list[str]:
|
||||
return sorted(p.name for p in processed_root.iterdir() if p.is_dir())
|
||||
|
||||
|
||||
def _write_unique(
|
||||
url_hash: str,
|
||||
article: Article,
|
||||
uniques_dir: Path,
|
||||
sources_map: dict[str, list[str]],
|
||||
) -> None:
|
||||
"""写 uniques JSON,附加 sources 多源字段(向后兼容:下游 Pydantic 忽略多余字段)。"""
|
||||
data = json.loads(article.model_dump_json())
|
||||
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [article.source_id])))
|
||||
out_path = uniques_dir / f"{url_hash}.json"
|
||||
out_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _update_unique_sources(
|
||||
url_hash: str,
|
||||
uniques_dir: Path,
|
||||
sources_map: dict[str, list[str]],
|
||||
) -> None:
|
||||
"""仅更新已存在 uniques 文件的 sources 字段(不覆盖原文内容)。
|
||||
|
||||
跨日命中时对应 uniques 文件在历史日期目录,不在本次处理范围,以指纹库为准。
|
||||
"""
|
||||
uniq_path = uniques_dir / f"{url_hash}.json"
|
||||
if not uniq_path.is_file():
|
||||
return
|
||||
try:
|
||||
data = json.loads(uniq_path.read_text(encoding="utf-8"))
|
||||
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [])))
|
||||
uniq_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
except (json.JSONDecodeError, OSError) as e:
|
||||
logger.warning("更新 uniques 多源失败 {}: {}", uniq_path, e)
|
||||
|
||||
|
||||
def _merge_sources(url_hash: str, new_source: str, sources_map: dict[str, list[str]]) -> None:
|
||||
"""把新源并入 url_hash 的源列表(去重保序,主源居首)。"""
|
||||
cur = sources_map.setdefault(url_hash, [])
|
||||
if new_source not in cur:
|
||||
cur.append(new_source)
|
||||
|
||||
|
||||
def _process_source_day(
|
||||
source_id: str,
|
||||
day: str,
|
||||
processed_root: Path,
|
||||
out_root: Path,
|
||||
deduper: Deduper,
|
||||
sources_map: dict[str, list[str]],
|
||||
) -> tuple[int, int, Counter]:
|
||||
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。"""
|
||||
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。
|
||||
|
||||
sources_map: 本次去重涉及内容组的 url_hash -> 全部来源列表(跨源累积,
|
||||
最终写入 data/deduped/{day}/sources.json,供「显示新闻源」使用;
|
||||
跨日命中的历史内容组也会记录,权威多源以指纹库 source_ids 列为准)。
|
||||
"""
|
||||
src_dir = processed_root / source_id / day
|
||||
if not src_dir.is_dir():
|
||||
logger.info("源 {} 日期 {} 无 processed 目录,跳过", source_id, day)
|
||||
@@ -92,6 +141,11 @@ def _process_source_day(
|
||||
dup_cnt += 1
|
||||
if result.matched_layer is not None:
|
||||
layer_cnt[result.matched_layer.value] += 1
|
||||
# 记录多源:把被去重文章的源并入对应唯一新闻
|
||||
if result.matched_url_hash:
|
||||
_merge_sources(result.matched_url_hash, article.source_id, sources_map)
|
||||
# 若该唯一新闻文件在当天目录,同步更新其 sources 字段
|
||||
_update_unique_sources(result.matched_url_hash, uniques_dir, sources_map)
|
||||
dup_f.write(
|
||||
json.dumps(
|
||||
{
|
||||
@@ -107,6 +161,8 @@ def _process_source_day(
|
||||
"matched_url": result.matched_url,
|
||||
"matched_url_hash": result.matched_url_hash,
|
||||
"matched_title": result.matched_title,
|
||||
"matched_source_id": result.matched_source_id,
|
||||
"matched_source_ids": result.all_source_ids,
|
||||
"hamming_distance": result.hamming_distance,
|
||||
},
|
||||
ensure_ascii=False,
|
||||
@@ -115,8 +171,8 @@ def _process_source_day(
|
||||
)
|
||||
else:
|
||||
uniq_cnt += 1
|
||||
out_path = uniques_dir / f"{article.url_hash}.json"
|
||||
out_path.write_text(article.model_dump_json(indent=2), encoding="utf-8")
|
||||
sources_map[article.url_hash] = [article.source_id]
|
||||
_write_unique(article.url_hash, article, uniques_dir, sources_map)
|
||||
|
||||
total = uniq_cnt + dup_cnt
|
||||
rate = dup_cnt / max(total, 1)
|
||||
@@ -173,14 +229,35 @@ def main() -> int:
|
||||
total_uniq = 0
|
||||
total_dup = 0
|
||||
total_layers: Counter = Counter()
|
||||
# 当天唯一新闻 url_hash -> 全部来源列表(跨源累积,多源记录)
|
||||
sources_map: dict[str, list[str]] = {}
|
||||
for src in sources:
|
||||
u, d, lc = _process_source_day(
|
||||
src, args.date, processed_root, out_root, deduper
|
||||
src, args.date, processed_root, out_root, deduper, sources_map
|
||||
)
|
||||
total_uniq += u
|
||||
total_dup += d
|
||||
total_layers.update(lc)
|
||||
|
||||
# 多源记录汇总:data/deduped/{day}/sources.json
|
||||
# {url_hash: [source_id, ...]},配合 uniques/{url_hash}.json 的 sources 字段
|
||||
# 与指纹库 source_ids 列,提供「一条唯一新闻多个来源」的完整记录。
|
||||
sources_path = out_root / args.date / "sources.json"
|
||||
sources_path.write_text(
|
||||
json.dumps(
|
||||
{k: v for k, v in sources_map.items() if v},
|
||||
ensure_ascii=False,
|
||||
indent=2,
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
logger.info(
|
||||
"多源记录已写入 {} ({} 条唯一新闻,{} 条含多源)",
|
||||
sources_path,
|
||||
len(sources_map),
|
||||
sum(1 for v in sources_map.values() if len(v) > 1),
|
||||
)
|
||||
|
||||
total = total_uniq + total_dup
|
||||
rate = total_dup / max(total, 1)
|
||||
logger.info(
|
||||
|
||||
Reference in New Issue
Block a user