Files
news/scripts/run_dedup.py
T
simon 65ead54b4f docs: 文档清理与重构 — 统一为 3 个核心文档
- 删除 5 个过时/残留文档(project_plan/agent_prompt/optimization_plan/report_db_design/deploy/README)
- 新建 docs/architecture.md(项目架构:11 包职责+数据模型+配置+产物)
- 重写 docs/user-guide.md(CLI 全量+增量/断点续跑+MCP+FAQ)
- 重写 README.md(精简入口+文档索引)
- 更新 continuation.md(追加本次记录)
- 更新 .gitignore(排除 data/* 运行产物)
2026-08-22 17:10:39 +08:00

276 lines
10 KiB
Python

"""M3 批量去重入口脚本。
输入: data/processed/{source}/{YYYYMMDD}/*.json (M2 产物)
输出:
- 指纹库:data/dedup/fingerprints.sqlite3 (source_ids 列记录多源)
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json (含 sources 多源字段)
- 多源记录:data/deduped/{YYYYMMDD}/sources.json
{url_hash: [source_id, ...]},一条唯一新闻的全部来源
- 重复记录:data/deduped/{YYYYMMDD}/duplicates.jsonl
(含 matched_source_id / matched_source_ids)
用法:
uv run python -m scripts.run_dedup # 处理今日全部源
uv run python -m scripts.run_dedup --date 20260616
uv run python -m scripts.run_dedup --source sina --date 20260616
uv run python -m scripts.run_dedup --reset # 清空指纹库重新建立
"""
from __future__ import annotations
import argparse
import json
import sys
from collections import Counter
from datetime import date
from pathlib import Path
from loguru import logger
from pydantic import ValidationError
from dedup import Deduper
from extractor import Article
def _setup_logger(level: str) -> None:
logger.remove()
logger.add(
sys.stderr,
level=level,
format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}",
)
log_path = Path("logs") / "dedup.log"
log_path.parent.mkdir(parents=True, exist_ok=True)
logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8")
def _load_article(json_path: Path) -> Article | None:
try:
data = json.loads(json_path.read_text(encoding="utf-8"))
return Article.model_validate(data)
except (json.JSONDecodeError, ValidationError) as e:
logger.warning("跳过无法解析的 article 文件 {}: {}", json_path, e)
return None
def _list_source_dirs(processed_root: Path) -> list[str]:
if not processed_root.is_dir():
return []
return sorted(p.name for p in processed_root.iterdir() if p.is_dir())
def _write_unique(
url_hash: str,
article: Article,
uniques_dir: Path,
sources_map: dict[str, list[str]],
) -> None:
"""写 uniques JSON,附加 sources 多源字段(向后兼容:下游 Pydantic 忽略多余字段)。"""
data = json.loads(article.model_dump_json())
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [article.source_id])))
out_path = uniques_dir / f"{url_hash}.json"
out_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
def _update_unique_sources(
url_hash: str,
uniques_dir: Path,
sources_map: dict[str, list[str]],
) -> None:
"""仅更新已存在 uniques 文件的 sources 字段(不覆盖原文内容)。
跨日命中时对应 uniques 文件在历史日期目录,不在本次处理范围,以指纹库为准。
"""
uniq_path = uniques_dir / f"{url_hash}.json"
if not uniq_path.is_file():
return
try:
data = json.loads(uniq_path.read_text(encoding="utf-8"))
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [])))
uniq_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
except (json.JSONDecodeError, OSError) as e:
logger.warning("更新 uniques 多源失败 {}: {}", uniq_path, e)
def _merge_sources(url_hash: str, new_source: str, sources_map: dict[str, list[str]]) -> None:
"""把新源并入 url_hash 的源列表(去重保序,主源居首)。"""
cur = sources_map.setdefault(url_hash, [])
if new_source not in cur:
cur.append(new_source)
def _process_source_day(
source_id: str,
day: str,
processed_root: Path,
out_root: Path,
deduper: Deduper,
sources_map: dict[str, list[str]],
) -> tuple[int, int, Counter]:
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。
sources_map: 本次去重涉及内容组的 url_hash -> 全部来源列表(跨源累积,
最终写入 data/deduped/{day}/sources.json,供「显示新闻源」使用;
跨日命中的历史内容组也会记录,权威多源以指纹库 source_ids 列为准)。
"""
src_dir = processed_root / source_id / day
if not src_dir.is_dir():
logger.info("源 {} 日期 {} 无 processed 目录,跳过", source_id, day)
return 0, 0, Counter()
files = sorted(src_dir.glob("*.json"))
if not files:
logger.info("源 {} 日期 {} 无文章,跳过", source_id, day)
return 0, 0, Counter()
uniques_dir = out_root / day / "uniques"
uniques_dir.mkdir(parents=True, exist_ok=True)
dup_log = out_root / day / "duplicates.jsonl"
uniq_cnt = 0
dup_cnt = 0
layer_cnt: Counter = Counter()
with dup_log.open("a", encoding="utf-8") as dup_f:
for fp in files:
article = _load_article(fp)
if article is None:
continue
result = deduper.ingest(article)
if result.is_duplicate:
dup_cnt += 1
if result.matched_layer is not None:
layer_cnt[result.matched_layer.value] += 1
# 记录多源:把被去重文章的源并入对应唯一新闻
if result.matched_url_hash:
_merge_sources(result.matched_url_hash, article.source_id, sources_map)
# 若该唯一新闻文件在当天目录,同步更新其 sources 字段
_update_unique_sources(result.matched_url_hash, uniques_dir, sources_map)
dup_f.write(
json.dumps(
{
"source_id": article.source_id,
"url": article.url,
"url_hash": article.url_hash,
"title": article.title,
"matched_layer": (
result.matched_layer.value
if result.matched_layer
else None
),
"matched_url": result.matched_url,
"matched_url_hash": result.matched_url_hash,
"matched_title": result.matched_title,
"matched_source_id": result.matched_source_id,
"matched_source_ids": result.all_source_ids,
"hamming_distance": result.hamming_distance,
},
ensure_ascii=False,
)
+ "\n"
)
else:
uniq_cnt += 1
sources_map[article.url_hash] = [article.source_id]
_write_unique(article.url_hash, article, uniques_dir, sources_map)
total = uniq_cnt + dup_cnt
rate = dup_cnt / max(total, 1)
logger.info(
"源 {} 日期 {}: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}",
source_id,
day,
uniq_cnt,
dup_cnt,
rate,
dict(layer_cnt),
)
return uniq_cnt, dup_cnt, layer_cnt
def main() -> int:
parser = argparse.ArgumentParser(description="A 股新闻三层去重 (M3)")
parser.add_argument("--processed-root", default="data/processed")
parser.add_argument("--out-root", default="data/deduped")
parser.add_argument("--db", default="data/dedup/fingerprints.sqlite3")
parser.add_argument("--source", default=None, help="只处理单源")
parser.add_argument(
"--date", default=date.today().strftime("%Y%m%d"), help="日期 YYYYMMDD"
)
parser.add_argument("--simhash-threshold", type=int, default=3)
parser.add_argument("--window-days", type=int, default=30)
parser.add_argument("--reset", action="store_true", help="处理前清空指纹库")
parser.add_argument("--log-level", default="INFO")
args = parser.parse_args()
_setup_logger(args.log_level)
processed_root = Path(args.processed_root)
out_root = Path(args.out_root)
sources = [args.source] if args.source else _list_source_dirs(processed_root)
if not sources:
logger.error("{} 下无源目录", processed_root)
return 2
with Deduper(
db_path=args.db,
simhash_threshold=args.simhash_threshold,
time_window_days=args.window_days,
) as deduper:
if args.reset:
logger.warning("--reset:清空指纹库 {}", args.db)
deduper.store.clear()
# 清掉同日 duplicates.jsonl 避免重复追加(uniques 用 url_hash 文件名,会自然覆盖)
dup_log = out_root / args.date / "duplicates.jsonl"
if dup_log.exists():
dup_log.unlink()
total_uniq = 0
total_dup = 0
total_layers: Counter = Counter()
# 当天唯一新闻 url_hash -> 全部来源列表(跨源累积,多源记录)
sources_map: dict[str, list[str]] = {}
for src in sources:
u, d, lc = _process_source_day(
src, args.date, processed_root, out_root, deduper, sources_map
)
total_uniq += u
total_dup += d
total_layers.update(lc)
# 多源记录汇总:data/deduped/{day}/sources.json
# {url_hash: [source_id, ...]},配合 uniques/{url_hash}.json 的 sources 字段
# 与指纹库 source_ids 列,提供「一条唯一新闻多个来源」的完整记录。
sources_path = out_root / args.date / "sources.json"
sources_path.write_text(
json.dumps(
{k: v for k, v in sources_map.items() if v},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
logger.info(
"多源记录已写入 {} ({} 条唯一新闻,{} 条含多源)",
sources_path,
len(sources_map),
sum(1 for v in sources_map.values() if len(v) > 1),
)
total = total_uniq + total_dup
rate = total_dup / max(total, 1)
logger.info(
"全部完成: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}",
total_uniq,
total_dup,
rate,
dict(total_layers),
)
# 验收门槛: ≤ 5%
return 0 if rate <= 0.05 or total == 0 else 1
if __name__ == "__main__":
raise SystemExit(main())