feat: 大模型使用场景化配置与去重多源记录
- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
+33
-5
@@ -35,7 +35,10 @@ def _publish_date(article: Article) -> str | None:
|
||||
|
||||
|
||||
def article_to_fingerprint(article: Article) -> Fingerprint:
|
||||
"""构造 Fingerprint(用于 ingest 写入或对外只读)。"""
|
||||
"""构造 Fingerprint(用于 ingest 写入或对外只读)。
|
||||
|
||||
source_ids 初始为 [article.source_id],后续重复文章命中时由 ingest 合并。
|
||||
"""
|
||||
return Fingerprint(
|
||||
url_hash=article.url_hash,
|
||||
content_hash=content_hash(article.content),
|
||||
@@ -45,6 +48,7 @@ def article_to_fingerprint(article: Article) -> Fingerprint:
|
||||
title=article.title,
|
||||
publish_date=_publish_date(article),
|
||||
ingested_at=datetime.now(),
|
||||
source_ids=[article.source_id],
|
||||
)
|
||||
|
||||
|
||||
@@ -80,7 +84,7 @@ class Deduper:
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
def check(self, article: Article) -> DedupResult:
|
||||
"""三层判重(只读)。"""
|
||||
"""三层判重(只读)。命中时附带匹配指纹的多源信息(all_source_ids)。"""
|
||||
fp = article_to_fingerprint(article)
|
||||
|
||||
# L1: URL hash
|
||||
@@ -93,6 +97,8 @@ class Deduper:
|
||||
matched_url_hash=existing.url_hash,
|
||||
matched_url=existing.url,
|
||||
matched_title=existing.title,
|
||||
matched_source_id=existing.source_id,
|
||||
all_source_ids=existing.source_ids,
|
||||
)
|
||||
|
||||
# L2: 内容 hash
|
||||
@@ -105,6 +111,8 @@ class Deduper:
|
||||
matched_url_hash=existing.url_hash,
|
||||
matched_url=existing.url,
|
||||
matched_title=existing.title,
|
||||
matched_source_id=existing.source_id,
|
||||
all_source_ids=existing.source_ids,
|
||||
)
|
||||
|
||||
# L3: SimHash 模糊
|
||||
@@ -129,20 +137,40 @@ class Deduper:
|
||||
matched_url_hash=best_match.url_hash,
|
||||
matched_url=best_match.url,
|
||||
matched_title=best_match.title,
|
||||
matched_source_id=best_match.source_id,
|
||||
all_source_ids=best_match.source_ids,
|
||||
hamming_distance=best_dist,
|
||||
)
|
||||
|
||||
return DedupResult(url_hash=fp.url_hash, is_duplicate=False)
|
||||
|
||||
def ingest(self, article: Article) -> DedupResult:
|
||||
"""判重 + 不重复则入库。"""
|
||||
"""判重 + 不重复则入库。
|
||||
|
||||
命中重复时,把当前文章的 source_id 合并进匹配指纹的 source_ids
|
||||
(记录同一内容组的全部来源),并更新 all_source_ids 后返回。
|
||||
"""
|
||||
result = self.check(article)
|
||||
if not result.is_duplicate:
|
||||
fp = article_to_fingerprint(article)
|
||||
self.store.upsert(fp)
|
||||
logger.debug("入库: {} {}", fp.url_hash, fp.title[:30])
|
||||
else:
|
||||
logger.debug("命中重复: {}", result.short_summary())
|
||||
return result
|
||||
|
||||
# 重复:合并来源到匹配指纹(主源保持首位,Fingerprint validator 负责去重)
|
||||
if result.matched_url_hash and article.source_id not in result.all_source_ids:
|
||||
matched = self.store.get_by_url_hash(result.matched_url_hash)
|
||||
if matched is not None:
|
||||
merged = [*matched.source_ids, article.source_id]
|
||||
self.store.upsert(matched.model_copy(update={"source_ids": merged}))
|
||||
result = result.model_copy(
|
||||
update={"all_source_ids": merged}
|
||||
)
|
||||
logger.debug(
|
||||
"合并来源 {} -> {} ({} 个源)",
|
||||
article.source_id, result.matched_url_hash, len(merged),
|
||||
)
|
||||
logger.debug("命中重复: {}", result.short_summary())
|
||||
return result
|
||||
|
||||
def stats(self) -> DedupStats:
|
||||
|
||||
Reference in New Issue
Block a user