docs: 文档清理与重构 — 统一为 3 个核心文档
- 删除 5 个过时/残留文档(project_plan/agent_prompt/optimization_plan/report_db_design/deploy/README) - 新建 docs/architecture.md(项目架构:11 包职责+数据模型+配置+产物) - 重写 docs/user-guide.md(CLI 全量+增量/断点续跑+MCP+FAQ) - 重写 README.md(精简入口+文档索引) - 更新 continuation.md(追加本次记录) - 更新 .gitignore(排除 data/* 运行产物)
This commit is contained in:
@@ -0,0 +1,275 @@
|
||||
"""M3 批量去重入口脚本。
|
||||
|
||||
输入: data/processed/{source}/{YYYYMMDD}/*.json (M2 产物)
|
||||
输出:
|
||||
- 指纹库:data/dedup/fingerprints.sqlite3 (source_ids 列记录多源)
|
||||
- 唯一文章:data/deduped/{YYYYMMDD}/uniques/{url_hash}.json (含 sources 多源字段)
|
||||
- 多源记录:data/deduped/{YYYYMMDD}/sources.json
|
||||
{url_hash: [source_id, ...]},一条唯一新闻的全部来源
|
||||
- 重复记录:data/deduped/{YYYYMMDD}/duplicates.jsonl
|
||||
(含 matched_source_id / matched_source_ids)
|
||||
|
||||
用法:
|
||||
uv run python -m scripts.run_dedup # 处理今日全部源
|
||||
uv run python -m scripts.run_dedup --date 20260616
|
||||
uv run python -m scripts.run_dedup --source sina --date 20260616
|
||||
uv run python -m scripts.run_dedup --reset # 清空指纹库重新建立
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from collections import Counter
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
from loguru import logger
|
||||
from pydantic import ValidationError
|
||||
|
||||
from dedup import Deduper
|
||||
from extractor import Article
|
||||
|
||||
|
||||
def _setup_logger(level: str) -> None:
|
||||
logger.remove()
|
||||
logger.add(
|
||||
sys.stderr,
|
||||
level=level,
|
||||
format="{time:YYYY-MM-DD HH:mm:ss} | {level} | {name} | {message}",
|
||||
)
|
||||
log_path = Path("logs") / "dedup.log"
|
||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
logger.add(log_path, level="DEBUG", rotation="10 MB", retention=5, encoding="utf-8")
|
||||
|
||||
|
||||
def _load_article(json_path: Path) -> Article | None:
|
||||
try:
|
||||
data = json.loads(json_path.read_text(encoding="utf-8"))
|
||||
return Article.model_validate(data)
|
||||
except (json.JSONDecodeError, ValidationError) as e:
|
||||
logger.warning("跳过无法解析的 article 文件 {}: {}", json_path, e)
|
||||
return None
|
||||
|
||||
|
||||
def _list_source_dirs(processed_root: Path) -> list[str]:
|
||||
if not processed_root.is_dir():
|
||||
return []
|
||||
return sorted(p.name for p in processed_root.iterdir() if p.is_dir())
|
||||
|
||||
|
||||
def _write_unique(
|
||||
url_hash: str,
|
||||
article: Article,
|
||||
uniques_dir: Path,
|
||||
sources_map: dict[str, list[str]],
|
||||
) -> None:
|
||||
"""写 uniques JSON,附加 sources 多源字段(向后兼容:下游 Pydantic 忽略多余字段)。"""
|
||||
data = json.loads(article.model_dump_json())
|
||||
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [article.source_id])))
|
||||
out_path = uniques_dir / f"{url_hash}.json"
|
||||
out_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _update_unique_sources(
|
||||
url_hash: str,
|
||||
uniques_dir: Path,
|
||||
sources_map: dict[str, list[str]],
|
||||
) -> None:
|
||||
"""仅更新已存在 uniques 文件的 sources 字段(不覆盖原文内容)。
|
||||
|
||||
跨日命中时对应 uniques 文件在历史日期目录,不在本次处理范围,以指纹库为准。
|
||||
"""
|
||||
uniq_path = uniques_dir / f"{url_hash}.json"
|
||||
if not uniq_path.is_file():
|
||||
return
|
||||
try:
|
||||
data = json.loads(uniq_path.read_text(encoding="utf-8"))
|
||||
data["sources"] = list(dict.fromkeys(sources_map.get(url_hash, [])))
|
||||
uniq_path.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
except (json.JSONDecodeError, OSError) as e:
|
||||
logger.warning("更新 uniques 多源失败 {}: {}", uniq_path, e)
|
||||
|
||||
|
||||
def _merge_sources(url_hash: str, new_source: str, sources_map: dict[str, list[str]]) -> None:
|
||||
"""把新源并入 url_hash 的源列表(去重保序,主源居首)。"""
|
||||
cur = sources_map.setdefault(url_hash, [])
|
||||
if new_source not in cur:
|
||||
cur.append(new_source)
|
||||
|
||||
|
||||
def _process_source_day(
|
||||
source_id: str,
|
||||
day: str,
|
||||
processed_root: Path,
|
||||
out_root: Path,
|
||||
deduper: Deduper,
|
||||
sources_map: dict[str, list[str]],
|
||||
) -> tuple[int, int, Counter]:
|
||||
"""处理单源单日。返回 (uniques, duplicates, layer_counter)。
|
||||
|
||||
sources_map: 本次去重涉及内容组的 url_hash -> 全部来源列表(跨源累积,
|
||||
最终写入 data/deduped/{day}/sources.json,供「显示新闻源」使用;
|
||||
跨日命中的历史内容组也会记录,权威多源以指纹库 source_ids 列为准)。
|
||||
"""
|
||||
src_dir = processed_root / source_id / day
|
||||
if not src_dir.is_dir():
|
||||
logger.info("源 {} 日期 {} 无 processed 目录,跳过", source_id, day)
|
||||
return 0, 0, Counter()
|
||||
|
||||
files = sorted(src_dir.glob("*.json"))
|
||||
if not files:
|
||||
logger.info("源 {} 日期 {} 无文章,跳过", source_id, day)
|
||||
return 0, 0, Counter()
|
||||
|
||||
uniques_dir = out_root / day / "uniques"
|
||||
uniques_dir.mkdir(parents=True, exist_ok=True)
|
||||
dup_log = out_root / day / "duplicates.jsonl"
|
||||
|
||||
uniq_cnt = 0
|
||||
dup_cnt = 0
|
||||
layer_cnt: Counter = Counter()
|
||||
|
||||
with dup_log.open("a", encoding="utf-8") as dup_f:
|
||||
for fp in files:
|
||||
article = _load_article(fp)
|
||||
if article is None:
|
||||
continue
|
||||
result = deduper.ingest(article)
|
||||
if result.is_duplicate:
|
||||
dup_cnt += 1
|
||||
if result.matched_layer is not None:
|
||||
layer_cnt[result.matched_layer.value] += 1
|
||||
# 记录多源:把被去重文章的源并入对应唯一新闻
|
||||
if result.matched_url_hash:
|
||||
_merge_sources(result.matched_url_hash, article.source_id, sources_map)
|
||||
# 若该唯一新闻文件在当天目录,同步更新其 sources 字段
|
||||
_update_unique_sources(result.matched_url_hash, uniques_dir, sources_map)
|
||||
dup_f.write(
|
||||
json.dumps(
|
||||
{
|
||||
"source_id": article.source_id,
|
||||
"url": article.url,
|
||||
"url_hash": article.url_hash,
|
||||
"title": article.title,
|
||||
"matched_layer": (
|
||||
result.matched_layer.value
|
||||
if result.matched_layer
|
||||
else None
|
||||
),
|
||||
"matched_url": result.matched_url,
|
||||
"matched_url_hash": result.matched_url_hash,
|
||||
"matched_title": result.matched_title,
|
||||
"matched_source_id": result.matched_source_id,
|
||||
"matched_source_ids": result.all_source_ids,
|
||||
"hamming_distance": result.hamming_distance,
|
||||
},
|
||||
ensure_ascii=False,
|
||||
)
|
||||
+ "\n"
|
||||
)
|
||||
else:
|
||||
uniq_cnt += 1
|
||||
sources_map[article.url_hash] = [article.source_id]
|
||||
_write_unique(article.url_hash, article, uniques_dir, sources_map)
|
||||
|
||||
total = uniq_cnt + dup_cnt
|
||||
rate = dup_cnt / max(total, 1)
|
||||
logger.info(
|
||||
"源 {} 日期 {}: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}",
|
||||
source_id,
|
||||
day,
|
||||
uniq_cnt,
|
||||
dup_cnt,
|
||||
rate,
|
||||
dict(layer_cnt),
|
||||
)
|
||||
return uniq_cnt, dup_cnt, layer_cnt
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="A 股新闻三层去重 (M3)")
|
||||
parser.add_argument("--processed-root", default="data/processed")
|
||||
parser.add_argument("--out-root", default="data/deduped")
|
||||
parser.add_argument("--db", default="data/dedup/fingerprints.sqlite3")
|
||||
parser.add_argument("--source", default=None, help="只处理单源")
|
||||
parser.add_argument(
|
||||
"--date", default=date.today().strftime("%Y%m%d"), help="日期 YYYYMMDD"
|
||||
)
|
||||
parser.add_argument("--simhash-threshold", type=int, default=3)
|
||||
parser.add_argument("--window-days", type=int, default=30)
|
||||
parser.add_argument("--reset", action="store_true", help="处理前清空指纹库")
|
||||
parser.add_argument("--log-level", default="INFO")
|
||||
args = parser.parse_args()
|
||||
|
||||
_setup_logger(args.log_level)
|
||||
processed_root = Path(args.processed_root)
|
||||
out_root = Path(args.out_root)
|
||||
|
||||
sources = [args.source] if args.source else _list_source_dirs(processed_root)
|
||||
if not sources:
|
||||
logger.error("{} 下无源目录", processed_root)
|
||||
return 2
|
||||
|
||||
with Deduper(
|
||||
db_path=args.db,
|
||||
simhash_threshold=args.simhash_threshold,
|
||||
time_window_days=args.window_days,
|
||||
) as deduper:
|
||||
if args.reset:
|
||||
logger.warning("--reset:清空指纹库 {}", args.db)
|
||||
deduper.store.clear()
|
||||
|
||||
# 清掉同日 duplicates.jsonl 避免重复追加(uniques 用 url_hash 文件名,会自然覆盖)
|
||||
dup_log = out_root / args.date / "duplicates.jsonl"
|
||||
if dup_log.exists():
|
||||
dup_log.unlink()
|
||||
|
||||
total_uniq = 0
|
||||
total_dup = 0
|
||||
total_layers: Counter = Counter()
|
||||
# 当天唯一新闻 url_hash -> 全部来源列表(跨源累积,多源记录)
|
||||
sources_map: dict[str, list[str]] = {}
|
||||
for src in sources:
|
||||
u, d, lc = _process_source_day(
|
||||
src, args.date, processed_root, out_root, deduper, sources_map
|
||||
)
|
||||
total_uniq += u
|
||||
total_dup += d
|
||||
total_layers.update(lc)
|
||||
|
||||
# 多源记录汇总:data/deduped/{day}/sources.json
|
||||
# {url_hash: [source_id, ...]},配合 uniques/{url_hash}.json 的 sources 字段
|
||||
# 与指纹库 source_ids 列,提供「一条唯一新闻多个来源」的完整记录。
|
||||
sources_path = out_root / args.date / "sources.json"
|
||||
sources_path.write_text(
|
||||
json.dumps(
|
||||
{k: v for k, v in sources_map.items() if v},
|
||||
ensure_ascii=False,
|
||||
indent=2,
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
logger.info(
|
||||
"多源记录已写入 {} ({} 条唯一新闻,{} 条含多源)",
|
||||
sources_path,
|
||||
len(sources_map),
|
||||
sum(1 for v in sources_map.values() if len(v) > 1),
|
||||
)
|
||||
|
||||
total = total_uniq + total_dup
|
||||
rate = total_dup / max(total, 1)
|
||||
logger.info(
|
||||
"全部完成: 唯一 {} / 重复 {} (重复率 {:.1%}) layers={}",
|
||||
total_uniq,
|
||||
total_dup,
|
||||
rate,
|
||||
dict(total_layers),
|
||||
)
|
||||
# 验收门槛: ≤ 5%
|
||||
return 0 if rate <= 0.05 or total == 0 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user