179 lines
5.8 KiB
Python
179 lines
5.8 KiB
Python
"""三层去重主流程。
|
|
|
|
调用顺序: check / ingest 内部按 L1 → L2 → L3 顺序判定,任意层命中即返回。
|
|
|
|
Deduper 不要求线程安全;批处理串行调用即可。
|
|
"""
|
|
|
|
import logging
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
|
|
import yaml
|
|
|
|
from dedup.hasher import DEFAULT_HAMMING_THRESHOLD, content_hash, hamming, simhash64
|
|
from dedup.models import DedupLayer, DedupResult, DedupStats, Fingerprint
|
|
from dedup.store import DEFAULT_DB_PATH, FingerprintStore
|
|
from extractor.models import ProcessedArticle
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# 默认时间窗口(±N 天)
|
|
DEFAULT_TIME_WINDOW_DAYS = 30
|
|
|
|
|
|
def _load_dedup_config() -> dict:
|
|
"""从 system.yaml 加载去重配置。"""
|
|
config_path = Path("configs/system.yaml")
|
|
if config_path.exists():
|
|
try:
|
|
with open(config_path, encoding="utf-8") as f:
|
|
raw = yaml.safe_load(f)
|
|
return raw.get("dedup", {})
|
|
except Exception:
|
|
logger.warning("加载 dedup 配置失败,使用默认值")
|
|
return {}
|
|
|
|
|
|
def _publish_date(article: ProcessedArticle) -> str | None:
|
|
"""从 ProcessedArticle.publish_time 取 YYYY-MM-DD 字符串。"""
|
|
if not article.publish_time:
|
|
return None
|
|
# publish_time 格式为 ISO 8601,如 "2026-06-21T10:30:00"
|
|
try:
|
|
return article.publish_time[:10]
|
|
except (IndexError, TypeError):
|
|
return None
|
|
|
|
|
|
def article_to_fingerprint(article: ProcessedArticle) -> Fingerprint:
|
|
"""构造 Fingerprint(用于 ingest 写入或对外只读)。"""
|
|
return Fingerprint(
|
|
url_hash=article.url_hash,
|
|
content_hash=content_hash(article.content),
|
|
simhash=simhash64(article.content),
|
|
source_id=article.source_id,
|
|
url=article.url,
|
|
title=article.title,
|
|
publish_date=_publish_date(article),
|
|
ingested_at=datetime.now(),
|
|
)
|
|
|
|
|
|
class Deduper:
|
|
"""三层去重器。
|
|
|
|
构造完毕后:
|
|
- check(article) 仅判断,不写入
|
|
- ingest(article) 判断,不重复则写入指纹库,返回结果
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
db_path: str | Path = DEFAULT_DB_PATH,
|
|
simhash_threshold: int | None = None,
|
|
time_window_days: int | None = None,
|
|
) -> None:
|
|
config = _load_dedup_config()
|
|
|
|
self.store = FingerprintStore(db_path)
|
|
self.simhash_threshold = (
|
|
simhash_threshold
|
|
if simhash_threshold is not None
|
|
else config.get("hamming_distance_threshold", DEFAULT_HAMMING_THRESHOLD)
|
|
)
|
|
self.time_window_days = (
|
|
time_window_days
|
|
if time_window_days is not None
|
|
else config.get("simhash_window_days", DEFAULT_TIME_WINDOW_DAYS)
|
|
)
|
|
|
|
def close(self) -> None:
|
|
self.store.close()
|
|
|
|
def __enter__(self) -> "Deduper":
|
|
return self
|
|
|
|
def __exit__(self, *_: object) -> None:
|
|
self.close()
|
|
|
|
# ------------------------------------------------------------------ #
|
|
# 公共 API
|
|
# ------------------------------------------------------------------ #
|
|
|
|
def check(self, article: ProcessedArticle) -> DedupResult:
|
|
"""三层判重(只读,不写入指纹库)。"""
|
|
fp = article_to_fingerprint(article)
|
|
|
|
# L1: URL hash
|
|
existing = self.store.get_by_url_hash(fp.url_hash)
|
|
if existing is not None:
|
|
return DedupResult(
|
|
url_hash=fp.url_hash,
|
|
is_duplicate=True,
|
|
matched_layer=DedupLayer.URL,
|
|
matched_url_hash=existing.url_hash,
|
|
matched_url=existing.url,
|
|
matched_title=existing.title,
|
|
)
|
|
|
|
# L2: 内容 hash
|
|
existing = self.store.find_by_content_hash(fp.content_hash)
|
|
if existing is not None:
|
|
return DedupResult(
|
|
url_hash=fp.url_hash,
|
|
is_duplicate=True,
|
|
matched_layer=DedupLayer.CONTENT,
|
|
matched_url_hash=existing.url_hash,
|
|
matched_url=existing.url,
|
|
matched_title=existing.title,
|
|
)
|
|
|
|
# L3: SimHash 模糊
|
|
candidates = self.store.candidates_for_simhash(
|
|
fp.publish_date, self.time_window_days
|
|
)
|
|
best_dist: int | None = None
|
|
best_match: Fingerprint | None = None
|
|
for c in candidates:
|
|
d = hamming(fp.simhash, c.simhash)
|
|
if d <= self.simhash_threshold and (best_dist is None or d < best_dist):
|
|
best_dist = d
|
|
best_match = c
|
|
if d == 0: # 不可能更近,提前结束
|
|
break
|
|
|
|
if best_match is not None:
|
|
return DedupResult(
|
|
url_hash=fp.url_hash,
|
|
is_duplicate=True,
|
|
matched_layer=DedupLayer.SIMHASH,
|
|
matched_url_hash=best_match.url_hash,
|
|
matched_url=best_match.url,
|
|
matched_title=best_match.title,
|
|
hamming_distance=best_dist,
|
|
)
|
|
|
|
return DedupResult(url_hash=fp.url_hash, is_duplicate=False)
|
|
|
|
def ingest(self, article: ProcessedArticle) -> DedupResult:
|
|
"""判重 + 不重复则入库。"""
|
|
result = self.check(article)
|
|
if not result.is_duplicate:
|
|
fp = article_to_fingerprint(article)
|
|
self.store.upsert(fp)
|
|
logger.debug("指纹入库: %s %s", fp.url_hash, fp.title[:40])
|
|
else:
|
|
logger.debug("命中重复: %s", result.short_summary())
|
|
return result
|
|
|
|
def stats(self) -> DedupStats:
|
|
"""指纹库统计信息。"""
|
|
lo, hi = self.store.date_range()
|
|
return DedupStats(
|
|
total=self.store.count(),
|
|
by_source=self.store.count_by_source(),
|
|
earliest=lo,
|
|
latest=hi,
|
|
)
|