feat: 大模型使用场景化配置与去重多源记录

- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
  可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
2026-08-12 07:57:10 +08:00
parent 0c032196d2
commit 3c65701449
21 changed files with 886 additions and 80 deletions
+33 -5
View File
@@ -35,7 +35,10 @@ def _publish_date(article: Article) -> str | None:
def article_to_fingerprint(article: Article) -> Fingerprint:
"""构造 Fingerprint(用于 ingest 写入或对外只读)。"""
"""构造 Fingerprint(用于 ingest 写入或对外只读)。
source_ids 初始为 [article.source_id],后续重复文章命中时由 ingest 合并。
"""
return Fingerprint(
url_hash=article.url_hash,
content_hash=content_hash(article.content),
@@ -45,6 +48,7 @@ def article_to_fingerprint(article: Article) -> Fingerprint:
title=article.title,
publish_date=_publish_date(article),
ingested_at=datetime.now(),
source_ids=[article.source_id],
)
@@ -80,7 +84,7 @@ class Deduper:
# ------------------------------------------------------------------ #
def check(self, article: Article) -> DedupResult:
"""三层判重(只读)。"""
"""三层判重(只读)。命中时附带匹配指纹的多源信息(all_source_ids)。"""
fp = article_to_fingerprint(article)
# L1: URL hash
@@ -93,6 +97,8 @@ class Deduper:
matched_url_hash=existing.url_hash,
matched_url=existing.url,
matched_title=existing.title,
matched_source_id=existing.source_id,
all_source_ids=existing.source_ids,
)
# L2: 内容 hash
@@ -105,6 +111,8 @@ class Deduper:
matched_url_hash=existing.url_hash,
matched_url=existing.url,
matched_title=existing.title,
matched_source_id=existing.source_id,
all_source_ids=existing.source_ids,
)
# L3: SimHash 模糊
@@ -129,20 +137,40 @@ class Deduper:
matched_url_hash=best_match.url_hash,
matched_url=best_match.url,
matched_title=best_match.title,
matched_source_id=best_match.source_id,
all_source_ids=best_match.source_ids,
hamming_distance=best_dist,
)
return DedupResult(url_hash=fp.url_hash, is_duplicate=False)
def ingest(self, article: Article) -> DedupResult:
"""判重 + 不重复则入库。"""
"""判重 + 不重复则入库。
命中重复时,把当前文章的 source_id 合并进匹配指纹的 source_ids
(记录同一内容组的全部来源),并更新 all_source_ids 后返回。
"""
result = self.check(article)
if not result.is_duplicate:
fp = article_to_fingerprint(article)
self.store.upsert(fp)
logger.debug("入库: {} {}", fp.url_hash, fp.title[:30])
else:
logger.debug("命中重复: {}", result.short_summary())
return result
# 重复:合并来源到匹配指纹(主源保持首位,Fingerprint validator 负责去重)
if result.matched_url_hash and article.source_id not in result.all_source_ids:
matched = self.store.get_by_url_hash(result.matched_url_hash)
if matched is not None:
merged = [*matched.source_ids, article.source_id]
self.store.upsert(matched.model_copy(update={"source_ids": merged}))
result = result.model_copy(
update={"all_source_ids": merged}
)
logger.debug(
"合并来源 {} -> {} ({} 个源)",
article.source_id, result.matched_url_hash, len(merged),
)
logger.debug("命中重复: {}", result.short_summary())
return result
def stats(self) -> DedupStats: