feat: 大模型使用场景化配置与去重多源记录
- 新增 configs/llm_models.yaml: 4 个场景(event_extraction/daily_report/stock_report/embedding)
可独立配置 provider/model/api_key_env/base_url_env/temperature 等,含用途与模型要求说明
- 新增 configs/loader.py: YAML 场景加载器(优先级: CLI 参数 > YAML > .env > 内置默认)
- llm/client.py: load_llm_config 支持 scene 参数,LLMConfig 增加 max_attempts
- embedding/factory+remote+local: provider/model/batch_limit 支持场景覆盖
- scheduler/reporter+stock_reporter: 日报/个股摘要接入场景配置
- dedup: Fingerprint.source_ids 多源记录 + 旧库自动迁移 + DedupResult 多源字段
- scripts/run_dedup: uniques JSON 的 sources 字段 + data/deduped/{day}/sources.json 汇总
- scripts/run_event_extraction: 接入 event_extraction 场景
- 补充测试: 场景优先级/零值、多源合并、旧库迁移、embedding 场景覆盖
This commit is contained in:
+29
-3
@@ -4,9 +4,9 @@ from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from enum import StrEnum
|
||||
from typing import Literal
|
||||
from typing import Literal, Self
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from pydantic import BaseModel, Field, model_validator
|
||||
|
||||
|
||||
class DedupLayer(StrEnum):
|
||||
@@ -18,7 +18,12 @@ class DedupLayer(StrEnum):
|
||||
|
||||
|
||||
class Fingerprint(BaseModel):
|
||||
"""单篇文章的指纹记录,持久化到 SQLite。"""
|
||||
"""单篇文章的指纹记录,持久化到 SQLite。
|
||||
|
||||
source_ids: 同一内容组(去重后视为同一篇新闻)的全部来源列表,
|
||||
第一位是主源(即本指纹的 source_id);重复文章命中时由
|
||||
Deduper.ingest 自动合并,实现「一条唯一新闻记录多个源」。
|
||||
"""
|
||||
|
||||
url_hash: str = Field(..., description="主键,与 Article.url_hash 一致")
|
||||
content_hash: str = Field(..., description="标准化 content 的 SHA1[:16]")
|
||||
@@ -28,6 +33,20 @@ class Fingerprint(BaseModel):
|
||||
title: str
|
||||
publish_date: str | None = Field(default=None, description="YYYY-MM-DD,用于时间窗口")
|
||||
ingested_at: datetime = Field(default_factory=datetime.now)
|
||||
source_ids: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="同内容组全部来源(去重合并),始终包含 source_id 且其居首",
|
||||
)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _ensure_source_ids(self) -> Self:
|
||||
"""保证 source_ids 非空、去重且以主源 source_id 开头。"""
|
||||
seen: list[str] = []
|
||||
for s in [self.source_id, *self.source_ids]:
|
||||
if s and s not in seen:
|
||||
seen.append(s)
|
||||
self.source_ids = seen
|
||||
return self
|
||||
|
||||
|
||||
class DedupResult(BaseModel):
|
||||
@@ -39,6 +58,13 @@ class DedupResult(BaseModel):
|
||||
matched_url_hash: str | None = None
|
||||
matched_url: str | None = None
|
||||
matched_title: str | None = None
|
||||
matched_source_id: str | None = Field(
|
||||
default=None, description="匹配指纹的主源 source_id"
|
||||
)
|
||||
all_source_ids: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="该内容组(唯一新闻)的全部来源;含匹配指纹自身的来源",
|
||||
)
|
||||
hamming_distance: int | None = Field(
|
||||
default=None, description="仅 SimHash 层有值"
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user