fix: 去重合并多来源后同步到 events/Qdrant/MySQL,新增 sync-sources 回填命令

- dedup/pipeline: 合并 source_ids 后自动同步已生成的 events JSON 和 Qdrant payload
- 新增 sync_sources_from_deduped / uv run en-news sync-sources,用于历史数据回填
- sync-sources 会扫描 deduped 多来源唯一篇,更新 events、Qdrant、MySQL news_event.sources
- docs: 补充 sync-sources 使用说明
This commit is contained in:
2026-08-23 10:55:23 +08:00
parent cf74e5e7f5
commit 9e4f5b4c75
4 changed files with 178 additions and 0 deletions
+150
View File
@@ -172,10 +172,160 @@ def _merge_duplicate_source(article: ProcessedArticle, result: DedupResult) -> N
)
logger.debug("来源合并: %s → %s (sources=%s)",
article.source_id, result.matched_url_hash, merged)
# 同步到已生成的 events / Qdrant,避免“deduped 有多来源但 DB 没有”
_sync_merged_sources_downstream(result.matched_url_hash, merged)
except Exception as e:
logger.exception("来源合并失败 %s: %s", target, e)
def _sync_event_source_ids(url_hash: str, source_ids: list[str]) -> int:
"""把合并后的 source_ids 同步到已生成的 events JSON。"""
updated = 0
for ev_path in sorted(Path("data/events").glob(f"*/{url_hash}.json")):
try:
data = json.loads(ev_path.read_text(encoding="utf-8"))
if data.get("source_ids") != source_ids:
data["source_ids"] = list(source_ids)
ev_path.write_text(
json.dumps(data, indent=2, ensure_ascii=False),
encoding="utf-8",
)
logger.info("来源合并同步 events: %s -> %s", ev_path, source_ids)
updated += 1
except Exception as e:
logger.exception("同步 events 来源失败 %s: %s", ev_path, e)
return updated
def _sync_qdrant_source_ids(url_hash: str, source_ids: list[str]) -> int:
"""把合并后的 source_ids 同步到 Qdrant 中已存在的向量点。"""
emb_paths = sorted(Path("data/embeddings").glob(f"*/{url_hash}.json"))
if not emb_paths:
return 0
event_path = next(Path("data/events").glob(f"*/{url_hash}.json"), None)
if event_path is None:
return 0
try:
article_data = json.loads(event_path.read_text(encoding="utf-8"))
except Exception as e:
logger.warning("读取 events 失败,跳过 Qdrant 来源同步 %s: %s", event_path, e)
return 0
try:
from vectorstore.client import VectorStore, make_qdrant_client
from vectorstore.pipeline import _build_payload
client = make_qdrant_client()
store = VectorStore(client)
try:
payload = _build_payload(article_data)
for emb_path in emb_paths:
emb_data = json.loads(emb_path.read_text(encoding="utf-8"))
store.upsert([{
"id": url_hash,
"vector": emb_data["vector"],
"payload": payload,
}])
finally:
store.close()
logger.info("来源合并同步 Qdrant: %s -> %s", url_hash, source_ids)
return len(emb_paths)
except Exception as e:
logger.warning("同步 Qdrant 来源失败 %s: %s", url_hash, e)
return 0
def _sync_merged_sources_downstream(url_hash: str, source_ids: list[str]) -> None:
"""去重合并来源后,同步更新 events 与 Qdrant 中已存在的下游数据。"""
_sync_event_source_ids(url_hash, source_ids)
_sync_qdrant_source_ids(url_hash, source_ids)
def sync_sources_from_deduped() -> dict:
"""历史数据回填:扫描所有 deduped 唯一篇,把多来源同步到 events/Qdrant/MySQL。
该函数用于修复“deduped 已合并多来源,但旧 events/数据库未更新”的历史数据。
"""
stats = {
"deduped_scanned": 0,
"multi_source": 0,
"events_updated": 0,
"qdrant_updated": 0,
"db_updated": 0,
}
targets = sorted(Path("data/deduped").glob("*/uniques/*.json"))
for target in targets:
try:
data = json.loads(target.read_text(encoding="utf-8"))
except Exception:
continue
stats["deduped_scanned"] += 1
raw_ids = data.get("source_ids") or [data.get("source_id")]
source_ids = list(dict.fromkeys(x for x in raw_ids if x))
if len(source_ids) <= 1:
continue
stats["multi_source"] += 1
url_hash = data.get("url_hash", target.stem)
stats["events_updated"] += _sync_event_source_ids(url_hash, source_ids)
stats["qdrant_updated"] += _sync_qdrant_source_ids(url_hash, source_ids)
stats["db_updated"] += _update_db_sources(url_hash, source_ids)
logger.info(
"来源回填完成: scanned=%d multi=%d events=%d qdrant=%d db=%d",
stats["deduped_scanned"], stats["multi_source"],
stats["events_updated"], stats["qdrant_updated"], stats["db_updated"],
)
return stats
def _update_db_sources(url_hash: str, source_ids: list[str]) -> int:
"""根据最新 events 来源,更新 MySQL news_event 中该文章的 source/sources。"""
event_path = next(Path("data/events").glob(f"*/{url_hash}.json"), None)
if event_path is None:
return 0
try:
article_data = json.loads(event_path.read_text(encoding="utf-8"))
except Exception:
return 0
try:
from scheduler.reporter import _article_source_label, _article_sources_full
source_label = _article_source_label(article_data)
source_list = _article_sources_full(article_data)
if not source_label and not source_list:
return 0
from report_db import connect
conn = connect()
try:
with conn.cursor() as cur:
cur.execute(
"""
UPDATE news_event
SET source = %s, sources = %s
WHERE section = 'intl' AND url = %s
""",
(
source_label,
json.dumps(source_list, ensure_ascii=False) if source_list else None,
article_data.get("url"),
),
)
updated = cur.rowcount
conn.commit()
return updated or 0
finally:
conn.close()
except Exception as e:
logger.warning("同步 MySQL 来源失败 %s: %s", url_hash, e)
return 0
def _layer_num(result: DedupResult) -> int:
"""DedupResult → 命中层编号。"""
if result.matched_layer is None: