- 删除 5 个过时/残留文档(project_plan/agent_prompt/optimization_plan/report_db_design/deploy/README) - 新建 docs/architecture.md(项目架构:11 包职责+数据模型+配置+产物) - 重写 docs/user-guide.md(CLI 全量+增量/断点续跑+MCP+FAQ) - 重写 README.md(精简入口+文档索引) - 更新 continuation.md(追加本次记录) - 更新 .gitignore(排除 data/* 运行产物)
276 lines
10 KiB
Python
276 lines
10 KiB
Python
"""M3 批量去重入口脚本。
|
|
|
|
输入: data/processed/{source}/{YYYYMMDD}/*.json (M2 产物)
|
|
输出:
|
|
- 指纹库:data/dedup/fingerprints.sqlite3 (source_ids 列记录多源)
|
|
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json (含 sources 多源字段)
|
|
- 多源记录:data/deduped/{YYYYMMDD}/sources.json
|
|
{url_hash: [source_id, ...]},一条唯一新闻的全部来源
|
|
- 重复记录:data/deduped/{YYYYMMDD}/duplicates.jsonl
|
|
(含 matched_source_id / matched_source_ids)
|
|
|
|
用法:
|
|
uv run python -m scripts.run_dedup # 处理今日全部源
|
|
uv run python -m scripts.run_dedup --date 20260616
|
|
uv run python -m scripts.run_dedup --source sina --date 20260616
|
|
uv run python -m scripts.run_dedup --reset # 清空指纹库重新建立
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
from collections import Counter
|
|
from datetime import date
|
|
from pathlib import Path
|
|
|
|
from loguru import logger
|
|
from pydantic import ValidationError
|
|
|
|
from dedup import Deduper
|
|
from extractor import Article
|
|
|
|
|
|
def _setup_logger(level: str) -> None:
|
|
logger.remove()
|
|
logger.add(
|
|
sys.stderr,
|
|
level=level,
|
|
format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}",
|
|
)
|
|
log_path = Path("logs") / "dedup.log"
|
|
log_path.parent.mkdir(parents=True, exist_ok=True)
|
|
logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8")
|
|
|
|
|
|
def _load_article(json_path: Path) -> Article | None:
|
|
try:
|
|
data = json.loads(json_path.read_text(encoding="utf-8"))
|
|
return Article.model_validate(data)
|
|
except (json.JSONDecodeError, ValidationError) as e:
|
|
logger.warning("跳过无法解析的 article 文件 {}: {}", json_path, e)
|
|
return None
|
|
|
|
|
|
def _list_source_dirs(processed_root: Path) -> list[str]:
|
|
if not processed_root.is_dir():
|
|
return []
|
|
return sorted(p.name for p in processed_root.iterdir() if p.is_dir())
|
|
|
|
|
|
def _write_unique(
|
|
url_hash: str,
|
|
article: Article,
|
|
uniques_dir: Path,
|
|
sources_map: dict[str, list[str]],
|
|
) -> None:
|
|
"""写 uniques JSON,附加 sources 多源字段(向后兼容:下游 Pydantic 忽略多余字段)。"""
|
|
data = json.loads(article.model_dump_json())
|
|
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [article.source_id])))
|
|
out_path = uniques_dir / f"{url_hash}.json"
|
|
out_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
|
|
|
|
def _update_unique_sources(
|
|
url_hash: str,
|
|
uniques_dir: Path,
|
|
sources_map: dict[str, list[str]],
|
|
) -> None:
|
|
"""仅更新已存在 uniques 文件的 sources 字段(不覆盖原文内容)。
|
|
|
|
跨日命中时对应 uniques 文件在历史日期目录,不在本次处理范围,以指纹库为准。
|
|
"""
|
|
uniq_path = uniques_dir / f"{url_hash}.json"
|
|
if not uniq_path.is_file():
|
|
return
|
|
try:
|
|
data = json.loads(uniq_path.read_text(encoding="utf-8"))
|
|
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [])))
|
|
uniq_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
except (json.JSONDecodeError, OSError) as e:
|
|
logger.warning("更新 uniques 多源失败 {}: {}", uniq_path, e)
|
|
|
|
|
|
def _merge_sources(url_hash: str, new_source: str, sources_map: dict[str, list[str]]) -> None:
|
|
"""把新源并入 url_hash 的源列表(去重保序,主源居首)。"""
|
|
cur = sources_map.setdefault(url_hash, [])
|
|
if new_source not in cur:
|
|
cur.append(new_source)
|
|
|
|
|
|
def _process_source_day(
|
|
source_id: str,
|
|
day: str,
|
|
processed_root: Path,
|
|
out_root: Path,
|
|
deduper: Deduper,
|
|
sources_map: dict[str, list[str]],
|
|
) -> tuple[int, int, Counter]:
|
|
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。
|
|
|
|
sources_map: 本次去重涉及内容组的 url_hash -> 全部来源列表(跨源累积,
|
|
最终写入 data/deduped/{day}/sources.json,供「显示新闻源」使用;
|
|
跨日命中的历史内容组也会记录,权威多源以指纹库 source_ids 列为准)。
|
|
"""
|
|
src_dir = processed_root / source_id / day
|
|
if not src_dir.is_dir():
|
|
logger.info("源 {} 日期 {} 无 processed 目录,跳过", source_id, day)
|
|
return 0, 0, Counter()
|
|
|
|
files = sorted(src_dir.glob("*.json"))
|
|
if not files:
|
|
logger.info("源 {} 日期 {} 无文章,跳过", source_id, day)
|
|
return 0, 0, Counter()
|
|
|
|
uniques_dir = out_root / day / "uniques"
|
|
uniques_dir.mkdir(parents=True, exist_ok=True)
|
|
dup_log = out_root / day / "duplicates.jsonl"
|
|
|
|
uniq_cnt = 0
|
|
dup_cnt = 0
|
|
layer_cnt: Counter = Counter()
|
|
|
|
with dup_log.open("a", encoding="utf-8") as dup_f:
|
|
for fp in files:
|
|
article = _load_article(fp)
|
|
if article is None:
|
|
continue
|
|
result = deduper.ingest(article)
|
|
if result.is_duplicate:
|
|
dup_cnt += 1
|
|
if result.matched_layer is not None:
|
|
layer_cnt[result.matched_layer.value] += 1
|
|
# 记录多源:把被去重文章的源并入对应唯一新闻
|
|
if result.matched_url_hash:
|
|
_merge_sources(result.matched_url_hash, article.source_id, sources_map)
|
|
# 若该唯一新闻文件在当天目录,同步更新其 sources 字段
|
|
_update_unique_sources(result.matched_url_hash, uniques_dir, sources_map)
|
|
dup_f.write(
|
|
json.dumps(
|
|
{
|
|
"source_id": article.source_id,
|
|
"url": article.url,
|
|
"url_hash": article.url_hash,
|
|
"title": article.title,
|
|
"matched_layer": (
|
|
result.matched_layer.value
|
|
if result.matched_layer
|
|
else None
|
|
),
|
|
"matched_url": result.matched_url,
|
|
"matched_url_hash": result.matched_url_hash,
|
|
"matched_title": result.matched_title,
|
|
"matched_source_id": result.matched_source_id,
|
|
"matched_source_ids": result.all_source_ids,
|
|
"hamming_distance": result.hamming_distance,
|
|
},
|
|
ensure_ascii=False,
|
|
)
|
|
+ "\n"
|
|
)
|
|
else:
|
|
uniq_cnt += 1
|
|
sources_map[article.url_hash] = [article.source_id]
|
|
_write_unique(article.url_hash, article, uniques_dir, sources_map)
|
|
|
|
total = uniq_cnt + dup_cnt
|
|
rate = dup_cnt / max(total, 1)
|
|
logger.info(
|
|
"源 {} 日期 {}: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}",
|
|
source_id,
|
|
day,
|
|
uniq_cnt,
|
|
dup_cnt,
|
|
rate,
|
|
dict(layer_cnt),
|
|
)
|
|
return uniq_cnt, dup_cnt, layer_cnt
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(description="A 股新闻三层去重 (M3)")
|
|
parser.add_argument("--processed-root", default="data/processed")
|
|
parser.add_argument("--out-root", default="data/deduped")
|
|
parser.add_argument("--db", default="data/dedup/fingerprints.sqlite3")
|
|
parser.add_argument("--source", default=None, help="只处理单源")
|
|
parser.add_argument(
|
|
"--date", default=date.today().strftime("%Y%m%d"), help="日期 YYYYMMDD"
|
|
)
|
|
parser.add_argument("--simhash-threshold", type=int, default=3)
|
|
parser.add_argument("--window-days", type=int, default=30)
|
|
parser.add_argument("--reset", action="store_true", help="处理前清空指纹库")
|
|
parser.add_argument("--log-level", default="INFO")
|
|
args = parser.parse_args()
|
|
|
|
_setup_logger(args.log_level)
|
|
processed_root = Path(args.processed_root)
|
|
out_root = Path(args.out_root)
|
|
|
|
sources = [args.source] if args.source else _list_source_dirs(processed_root)
|
|
if not sources:
|
|
logger.error("{} 下无源目录", processed_root)
|
|
return 2
|
|
|
|
with Deduper(
|
|
db_path=args.db,
|
|
simhash_threshold=args.simhash_threshold,
|
|
time_window_days=args.window_days,
|
|
) as deduper:
|
|
if args.reset:
|
|
logger.warning("--reset:清空指纹库 {}", args.db)
|
|
deduper.store.clear()
|
|
|
|
# 清掉同日 duplicates.jsonl 避免重复追加(uniques 用 url_hash 文件名,会自然覆盖)
|
|
dup_log = out_root / args.date / "duplicates.jsonl"
|
|
if dup_log.exists():
|
|
dup_log.unlink()
|
|
|
|
total_uniq = 0
|
|
total_dup = 0
|
|
total_layers: Counter = Counter()
|
|
# 当天唯一新闻 url_hash -> 全部来源列表(跨源累积,多源记录)
|
|
sources_map: dict[str, list[str]] = {}
|
|
for src in sources:
|
|
u, d, lc = _process_source_day(
|
|
src, args.date, processed_root, out_root, deduper, sources_map
|
|
)
|
|
total_uniq += u
|
|
total_dup += d
|
|
total_layers.update(lc)
|
|
|
|
# 多源记录汇总:data/deduped/{day}/sources.json
|
|
# {url_hash: [source_id, ...]},配合 uniques/{url_hash}.json 的 sources 字段
|
|
# 与指纹库 source_ids 列,提供「一条唯一新闻多个来源」的完整记录。
|
|
sources_path = out_root / args.date / "sources.json"
|
|
sources_path.write_text(
|
|
json.dumps(
|
|
{k: v for k, v in sources_map.items() if v},
|
|
ensure_ascii=False,
|
|
indent=2,
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
logger.info(
|
|
"多源记录已写入 {} ({} 条唯一新闻,{} 条含多源)",
|
|
sources_path,
|
|
len(sources_map),
|
|
sum(1 for v in sources_map.values() if len(v) > 1),
|
|
)
|
|
|
|
total = total_uniq + total_dup
|
|
rate = total_dup / max(total, 1)
|
|
logger.info(
|
|
"全部完成: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}",
|
|
total_uniq,
|
|
total_dup,
|
|
rate,
|
|
dict(total_layers),
|
|
)
|
|
# 验收门槛: ≤ 5%
|
|
return 0 if rate <= 0.05 or total == 0 else 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|