docs: 文档清理与重构 — 统一为 3 个核心文档

- 删除 5 个过时/残留文档(project_plan/agent_prompt/optimization_plan/report_db_design/deploy/README)
- 新建 docs/architecture.md(项目架构:11 包职责+数据模型+配置+产物)
- 重写 docs/user-guide.md(CLI 全量+增量/断点续跑+MCP+FAQ)
- 重写 README.md(精简入口+文档索引)
- 更新 continuation.md(追加本次记录)
- 更新 .gitignore(排除 data/* 运行产物)
This commit is contained in:
2026-08-22 17:10:39 +08:00
commit 65ead54b4f
101 changed files with 21093 additions and 0 deletions
+44
View File
@@ -0,0 +1,44 @@
"""三层新闻去重模块 (M3)。
公共 API:
- Deduper: 主类(check / ingest / stats)
- FingerprintStore: SQLite 指纹库(底层,通常无需直接用)
- DedupResult / DedupLayer / DedupStats / Fingerprint: 数据模型
- simhash64 / hamming / content_hash / normalize_content: 指纹算法
"""
from .deduper import (
DEFAULT_TIME_WINDOW_DAYS,
Deduper,
article_to_fingerprint,
)
from .hasher import (
DEFAULT_HAMMING_THRESHOLD,
NGRAM_SIZE,
SIMHASH_BITS,
content_hash,
hamming,
normalize_content,
simhash64,
)
from .models import DedupLayer, DedupResult, DedupStats, Fingerprint
from .store import DEFAULT_DB_PATH, FingerprintStore
__all__ = [
"DEFAULT_DB_PATH",
"DEFAULT_HAMMING_THRESHOLD",
"DEFAULT_TIME_WINDOW_DAYS",
"NGRAM_SIZE",
"SIMHASH_BITS",
"DedupLayer",
"DedupResult",
"DedupStats",
"Deduper",
"Fingerprint",
"FingerprintStore",
"article_to_fingerprint",
"content_hash",
"hamming",
"normalize_content",
"simhash64",
]
+183
View File
@@ -0,0 +1,183 @@
"""三层去重主流程。
调用顺序:check / ingest 内部按 L1 -> L2 -> L3 顺序判定,任意层命中即返回。
Deduper 不要求线程安全;批处理串行调用即可。
"""
from __future__ import annotations
from datetime import datetime
from pathlib import Path
from loguru import logger
from extractor import Article
from .hasher import (
DEFAULT_HAMMING_THRESHOLD,
content_hash,
hamming,
simhash64,
)
from .models import DedupLayer, DedupResult, DedupStats, Fingerprint
from .store import DEFAULT_DB_PATH, FingerprintStore
# 默认时间窗口(±N 天)
DEFAULT_TIME_WINDOW_DAYS = 30
def _publish_date(article: Article) -> str | None:
"""从 Article.publish_time 取 YYYY-MM-DD 字符串。"""
if article.publish_time is None:
return None
return article.publish_time.strftime("%Y-%m-%d")
def article_to_fingerprint(article: Article) -> Fingerprint:
"""构造 Fingerprint(用于 ingest 写入或对外只读)。
source_ids 初始为 [article.source_id],后续重复文章命中时由 ingest 合并。
"""
return Fingerprint(
url_hash=article.url_hash,
content_hash=content_hash(article.content),
simhash=simhash64(article.content),
source_id=article.source_id,
url=article.url,
title=article.title,
publish_date=_publish_date(article),
ingested_at=datetime.now(),
source_ids=[article.source_id],
)
class Deduper:
"""三层去重器。
构造完毕后:
- check(article) 仅判断,不写入;
- ingest(article) 判断,不重复则写入指纹库,返回结果。
"""
def __init__(
self,
db_path: str | Path = DEFAULT_DB_PATH,
simhash_threshold: int = DEFAULT_HAMMING_THRESHOLD,
time_window_days: int = DEFAULT_TIME_WINDOW_DAYS,
) -> None:
self.store = FingerprintStore(db_path)
self.simhash_threshold = simhash_threshold
self.time_window_days = time_window_days
def close(self) -> None:
self.store.close()
def __enter__(self) -> Deduper:
return self
def __exit__(self, *_: object) -> None:
self.close()
# ------------------------------------------------------------------ #
# 公共 API
# ------------------------------------------------------------------ #
def check(self, article: Article) -> DedupResult:
"""三层判重(只读)。命中时附带匹配指纹的多源信息(all_source_ids)。"""
fp = article_to_fingerprint(article)
# L1: URL hash
existing = self.store.get_by_url_hash(fp.url_hash)
if existing is not None:
return DedupResult(
url_hash=fp.url_hash,
is_duplicate=True,
matched_layer=DedupLayer.URL,
matched_url_hash=existing.url_hash,
matched_url=existing.url,
matched_title=existing.title,
matched_source_id=existing.source_id,
all_source_ids=existing.source_ids,
)
# L2: 内容 hash
existing = self.store.find_by_content_hash(fp.content_hash)
if existing is not None:
return DedupResult(
url_hash=fp.url_hash,
is_duplicate=True,
matched_layer=DedupLayer.CONTENT,
matched_url_hash=existing.url_hash,
matched_url=existing.url,
matched_title=existing.title,
matched_source_id=existing.source_id,
all_source_ids=existing.source_ids,
)
# L3: SimHash 模糊
candidates = self.store.candidates_for_simhash(
fp.publish_date, self.time_window_days
)
best_dist: int | None = None
best_match: Fingerprint | None = None
for c in candidates:
d = hamming(fp.simhash, c.simhash)
if d <= self.simhash_threshold and (best_dist is None or d < best_dist):
best_dist = d
best_match = c
if d == 0: # 不可能更近,提前结束
break
if best_match is not None:
return DedupResult(
url_hash=fp.url_hash,
is_duplicate=True,
matched_layer=DedupLayer.SIMHASH,
matched_url_hash=best_match.url_hash,
matched_url=best_match.url,
matched_title=best_match.title,
matched_source_id=best_match.source_id,
all_source_ids=best_match.source_ids,
hamming_distance=best_dist,
)
return DedupResult(url_hash=fp.url_hash, is_duplicate=False)
def ingest(self, article: Article) -> DedupResult:
"""判重 + 不重复则入库。
命中重复时,把当前文章的 source_id 合并进匹配指纹的 source_ids
(记录同一内容组的全部来源),并更新 all_source_ids 后返回。
"""
result = self.check(article)
if not result.is_duplicate:
fp = article_to_fingerprint(article)
self.store.upsert(fp)
logger.debug("入库: {} {}", fp.url_hash, fp.title[:30])
return result
# 重复:合并来源到匹配指纹(主源保持首位,Fingerprint validator 负责去重)
if result.matched_url_hash and article.source_id not in result.all_source_ids:
matched = self.store.get_by_url_hash(result.matched_url_hash)
if matched is not None:
merged = [*matched.source_ids, article.source_id]
self.store.upsert(matched.model_copy(update={"source_ids": merged}))
result = result.model_copy(
update={"all_source_ids": merged}
)
logger.debug(
"合并来源 {} -> {} ({} 个源)",
article.source_id, result.matched_url_hash, len(merged),
)
logger.debug("命中重复: {}", result.short_summary())
return result
def stats(self) -> DedupStats:
lo, hi = self.store.date_range()
return DedupStats(
total=self.store.count(),
by_source=self.store.count_by_source(),
earliest=lo,
latest=hi,
)
+93
View File
@@ -0,0 +1,93 @@
"""三层去重的指纹算法。
核心:
- normalize_content:把 content 折叠成纯净文本,用于跨源比对;
- content_hash:normalize 后 SHA1[:16];
- simhash64:字符 3-gram + md5 加权累加,产出 64 位无符号整数;
- hamming:两个 SimHash 的汉明距离。
设计取舍:
SimHash 的"分词"用字符 3-gram 而非 jieba。理由:
1. 中文场景下字符 3-gram 与词级 SimHash 在重复识别上效果接近,
而前者无外部依赖、ARM/嵌入式友好;
2. M5 Embedding 后续不依赖 jieba,引入只为 M3 不划算;
3. 重复率验收门槛 ≤ 5%(project_plan.md 第七章),3-gram 经验上
足以分辨。
"""
from __future__ import annotations
import hashlib
import unicodedata
# 64 位 SimHash 位宽
SIMHASH_BITS = 64
SIMHASH_MASK = (1 << SIMHASH_BITS) - 1
# 默认 SimHash 汉明距离阈值(<= 此值视为重复)
DEFAULT_HAMMING_THRESHOLD = 3
# 字符 n-gram 长度
NGRAM_SIZE = 3
def normalize_content(text: str) -> str:
"""把 content 折叠成"无空白无标点"形式,用于 L2 内容 hash 与 SimHash 输入。
使用 Unicode 类别判断:
- P* Punctuation(所有中英文标点)
- Z* Separator(空格 / 行 / 段分隔符)
- C* Control(NUL / 换行控制等)
保留 L*(字母)、N*(数字)、S*(符号,如 +/-、% 等),以及 CJK 字符。
"""
if not text:
return ""
return "".join(
ch for ch in text if unicodedata.category(ch)[0] not in ("P", "Z", "C")
)
def content_hash(text: str) -> str:
"""对 normalize_content(text) 做 SHA1,取前 16 hex 字符。"""
norm = normalize_content(text)
return hashlib.sha1(norm.encode("utf-8")).hexdigest()[:16]
def _ngrams(text: str, n: int = NGRAM_SIZE) -> list[str]:
"""字符级 n-gram。文本短于 n 时,直接整体作为单个 token。"""
if len(text) < n:
return [text] if text else []
return [text[i : i + n] for i in range(len(text) - n + 1)]
def simhash64(text: str) -> int:
"""64 位 SimHash。返回无符号整数,空文本返回 0。"""
norm = normalize_content(text)
if not norm:
return 0
grams = _ngrams(norm)
if not grams:
return 0
v = [0] * SIMHASH_BITS
for gram in grams:
h = int(hashlib.md5(gram.encode("utf-8"), usedforsecurity=False).hexdigest(), 16)
# 取低 64 位
h64 = h & SIMHASH_MASK
for i in range(SIMHASH_BITS):
if (h64 >> i) & 1:
v[i] += 1
else:
v[i] -= 1
fp = 0
for i in range(SIMHASH_BITS):
if v[i] > 0:
fp |= 1 << i
return fp
def hamming(a: int, b: int) -> int:
"""两个 SimHash 的汉明距离。"""
return bin((a ^ b) & SIMHASH_MASK).count("1")
+90
View File
@@ -0,0 +1,90 @@
"""三层去重模块的数据模型。"""
from __future__ import annotations
from datetime import datetime
from enum import StrEnum
from typing import Literal, Self
from pydantic import BaseModel, Field, model_validator
class DedupLayer(StrEnum):
"""命中去重的层。"""
URL = "url" # L1: 完全相同 URL
CONTENT = "content" # L2: 标准化后 content 完全一致
SIMHASH = "simhash" # L3: SimHash 汉明距离 <= 阈值
class Fingerprint(BaseModel):
"""单篇文章的指纹记录,持久化到 SQLite。
source_ids: 同一内容组(去重后视为同一篇新闻)的全部来源列表,
第一位是主源(即本指纹的 source_id);重复文章命中时由
Deduper.ingest 自动合并,实现「一条唯一新闻记录多个源」。
"""
url_hash: str = Field(..., description="主键,与 Article.url_hash 一致")
content_hash: str = Field(..., description="标准化 content 的 SHA1[:16]")
simhash: int = Field(..., description="64 位 SimHash 整数(无符号)")
source_id: str
url: str
title: str
publish_date: str | None = Field(default=None, description="YYYY-MM-DD,用于时间窗口")
ingested_at: datetime = Field(default_factory=datetime.now)
source_ids: list[str] = Field(
default_factory=list,
description="同内容组全部来源(去重合并),始终包含 source_id 且其居首",
)
@model_validator(mode="after")
def _ensure_source_ids(self) -> Self:
"""保证 source_ids 非空、去重且以主源 source_id 开头。"""
seen: list[str] = []
for s in [self.source_id, *self.source_ids]:
if s and s not in seen:
seen.append(s)
self.source_ids = seen
return self
class DedupResult(BaseModel):
"""对单篇文章的判重结果。"""
url_hash: str
is_duplicate: bool
matched_layer: DedupLayer | None = None
matched_url_hash: str | None = None
matched_url: str | None = None
matched_title: str | None = None
matched_source_id: str | None = Field(
default=None, description="匹配指纹的主源 source_id"
)
all_source_ids: list[str] = Field(
default_factory=list,
description="该内容组(唯一新闻)的全部来源;含匹配指纹自身的来源",
)
hamming_distance: int | None = Field(
default=None, description="仅 SimHash 层有值"
)
def short_summary(self) -> str:
if not self.is_duplicate:
return f"[UNIQUE] {self.url_hash}"
layer = self.matched_layer.value if self.matched_layer else "?"
extra = f" hd={self.hamming_distance}" if self.hamming_distance is not None else ""
return f"[DUP/{layer}] {self.url_hash} ~ {self.matched_url_hash}{extra}"
class DedupStats(BaseModel):
"""指纹库统计。"""
total: int = 0
by_source: dict[str, int] = Field(default_factory=dict)
earliest: str | None = None
latest: str | None = None
# 类型别名,便于在批处理日志中归类
DedupVerdict = Literal["unique", "duplicate"]
+207
View File
@@ -0,0 +1,207 @@
"""SQLite 指纹存储。
注意:SimHash 是 64 位无符号整数,SQLite INTEGER 是 64 位有符号
(范围 [-2^63, 2^63-1])。直接存可能溢出/转负数,虽然 XOR 仍然
正确但语义混乱。这里统一存为 16 位 hex TEXT,避免符号问题。
source_ids 列存 JSON 数组文本(同一内容组全部来源);旧库无此列时
自动 ALTER TABLE 迁移,旧数据读取时回退为 [source_id]。
"""
from __future__ import annotations
import json
import sqlite3
from datetime import datetime, timedelta
from pathlib import Path
from typing import Any
from loguru import logger
from .models import Fingerprint
DEFAULT_DB_PATH = Path("data/dedup/fingerprints.sqlite3")
_SCHEMA_SQL = """
CREATE TABLE IF NOT EXISTS fingerprints (
url_hash TEXT PRIMARY KEY,
content_hash TEXT NOT NULL,
simhash_hex TEXT NOT NULL,
source_id TEXT NOT NULL,
url TEXT NOT NULL,
title TEXT NOT NULL,
publish_date TEXT,
ingested_at TEXT NOT NULL,
source_ids TEXT
);
CREATE INDEX IF NOT EXISTS idx_content_hash ON fingerprints(content_hash);
CREATE INDEX IF NOT EXISTS idx_publish_date ON fingerprints(publish_date);
CREATE INDEX IF NOT EXISTS idx_source_id ON fingerprints(source_id);
"""
# 兼容旧库:为已存在但缺少 source_ids 列的表补列
_ALTER_SQL = "ALTER TABLE fingerprints ADD COLUMN source_ids TEXT"
def _to_hex(simhash: int) -> str:
return f"{simhash:016x}"
def _from_hex(hex_str: str) -> int:
return int(hex_str, 16)
def _to_sources_json(source_ids: list[str]) -> str:
return json.dumps(source_ids, ensure_ascii=False)
def _from_sources_json(raw: str | None, fallback: str) -> list[str]:
"""解析 source_ids 列;NULL/损坏时回退 [主源]。"""
if not raw:
return [fallback]
try:
val = json.loads(raw)
except (TypeError, ValueError):
return [fallback]
if isinstance(val, list) and val:
# 保证主源在首位(兼容手改/旧数据)
cleaned = [s for s in val if s and s != fallback]
return [fallback, *cleaned]
return [fallback]
def _row_to_fp(row: sqlite3.Row) -> Fingerprint:
return Fingerprint(
url_hash=row["url_hash"],
content_hash=row["content_hash"],
simhash=_from_hex(row["simhash_hex"]),
source_id=row["source_id"],
url=row["url"],
title=row["title"],
publish_date=row["publish_date"],
ingested_at=datetime.fromisoformat(row["ingested_at"]),
source_ids=_from_sources_json(row["source_ids"], row["source_id"]),
)
class FingerprintStore:
"""SQLite 包装。线程不安全(每个线程请新建实例)。"""
def __init__(self, db_path: str | Path = DEFAULT_DB_PATH) -> None:
self.db_path = Path(db_path)
self.db_path.parent.mkdir(parents=True, exist_ok=True)
self._conn: sqlite3.Connection = sqlite3.connect(
self.db_path, isolation_level=None
)
self._conn.row_factory = sqlite3.Row
self._conn.executescript(_SCHEMA_SQL)
self._migrate_source_ids()
logger.debug("打开指纹库: {}", self.db_path)
def _migrate_source_ids(self) -> None:
"""旧库兼容:为缺少 source_ids 列的表补列(新库无需执行)。"""
try:
self._conn.execute(_ALTER_SQL)
logger.info("指纹库迁移:为 fingerprints 表新增 source_ids 列")
except sqlite3.OperationalError:
logger.debug("source_ids 列已存在,跳过迁移")
def close(self) -> None:
self._conn.close()
def __enter__(self) -> FingerprintStore:
return self
def __exit__(self, *_: Any) -> None:
self.close()
# ------------------------------------------------------------------ #
# 查询
# ------------------------------------------------------------------ #
def get_by_url_hash(self, url_hash: str) -> Fingerprint | None:
row = self._conn.execute(
"SELECT * FROM fingerprints WHERE url_hash = ?", (url_hash,)
).fetchone()
return _row_to_fp(row) if row else None
def find_by_content_hash(self, content_hash: str) -> Fingerprint | None:
"""返回任一匹配项。"""
row = self._conn.execute(
"SELECT * FROM fingerprints WHERE content_hash = ? LIMIT 1",
(content_hash,),
).fetchone()
return _row_to_fp(row) if row else None
def candidates_for_simhash(
self,
publish_date: str | None,
window_days: int,
) -> list[Fingerprint]:
"""返回 publish_date ± window_days 内的指纹候选。
publish_date 为 None 时,不限定窗口(返回全部,慎用)。
"""
if publish_date is None or window_days < 0:
rows = self._conn.execute("SELECT * FROM fingerprints").fetchall()
return [_row_to_fp(r) for r in rows]
try:
center = datetime.strptime(publish_date, "%Y-%m-%d")
except ValueError:
logger.debug("publish_date 不可解析: {!r},退化为全表扫描", publish_date)
rows = self._conn.execute("SELECT * FROM fingerprints").fetchall()
return [_row_to_fp(r) for r in rows]
lo = (center - timedelta(days=window_days)).strftime("%Y-%m-%d")
hi = (center + timedelta(days=window_days)).strftime("%Y-%m-%d")
rows = self._conn.execute(
"SELECT * FROM fingerprints "
"WHERE publish_date IS NULL OR (publish_date >= ? AND publish_date <= ?)",
(lo, hi),
).fetchall()
return [_row_to_fp(r) for r in rows]
def count(self) -> int:
return self._conn.execute("SELECT COUNT(*) FROM fingerprints").fetchone()[0]
def count_by_source(self) -> dict[str, int]:
rows = self._conn.execute(
"SELECT source_id, COUNT(*) AS n FROM fingerprints GROUP BY source_id"
).fetchall()
return {r["source_id"]: r["n"] for r in rows}
def date_range(self) -> tuple[str | None, str | None]:
row = self._conn.execute(
"SELECT MIN(publish_date) AS lo, MAX(publish_date) AS hi FROM fingerprints"
).fetchone()
return (row["lo"], row["hi"]) if row else (None, None)
# ------------------------------------------------------------------ #
# 写入
# ------------------------------------------------------------------ #
def upsert(self, fp: Fingerprint) -> None:
self._conn.execute(
"INSERT OR REPLACE INTO fingerprints "
"(url_hash, content_hash, simhash_hex, source_id, url, title, "
" publish_date, ingested_at, source_ids) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
(
fp.url_hash,
fp.content_hash,
_to_hex(fp.simhash),
fp.source_id,
fp.url,
fp.title,
fp.publish_date,
fp.ingested_at.isoformat(),
_to_sources_json(fp.source_ids),
),
)
def delete(self, url_hash: str) -> None:
self._conn.execute("DELETE FROM fingerprints WHERE url_hash = ?", (url_hash,))
def clear(self) -> None:
"""清空指纹库,主要用于测试。"""
self._conn.execute("DELETE FROM fingerprints")