fix: 去重合并多来源后同步到 events/Qdrant/MySQL,新增 sync-sources 回填命令
- dedup/pipeline: 合并 source_ids 后自动同步已生成的 events JSON 和 Qdrant payload - 新增 sync_sources_from_deduped / uv run en-news sync-sources,用于历史数据回填 - sync-sources 会扫描 deduped 多来源唯一篇,更新 events、Qdrant、MySQL news_event.sources - docs: 补充 sync-sources 使用说明
This commit is contained in:
@@ -172,10 +172,160 @@ def _merge_duplicate_source(article: ProcessedArticle, result: DedupResult) -> N
|
||||
)
|
||||
logger.debug("来源合并: %s → %s (sources=%s)",
|
||||
article.source_id, result.matched_url_hash, merged)
|
||||
# 同步到已生成的 events / Qdrant,避免“deduped 有多来源但 DB 没有”
|
||||
_sync_merged_sources_downstream(result.matched_url_hash, merged)
|
||||
except Exception as e:
|
||||
logger.exception("来源合并失败 %s: %s", target, e)
|
||||
|
||||
|
||||
def _sync_event_source_ids(url_hash: str, source_ids: list[str]) -> int:
|
||||
"""把合并后的 source_ids 同步到已生成的 events JSON。"""
|
||||
updated = 0
|
||||
for ev_path in sorted(Path("data/events").glob(f"*/{url_hash}.json")):
|
||||
try:
|
||||
data = json.loads(ev_path.read_text(encoding="utf-8"))
|
||||
if data.get("source_ids") != source_ids:
|
||||
data["source_ids"] = list(source_ids)
|
||||
ev_path.write_text(
|
||||
json.dumps(data, indent=2, ensure_ascii=False),
|
||||
encoding="utf-8",
|
||||
)
|
||||
logger.info("来源合并同步 events: %s -> %s", ev_path, source_ids)
|
||||
updated += 1
|
||||
except Exception as e:
|
||||
logger.exception("同步 events 来源失败 %s: %s", ev_path, e)
|
||||
return updated
|
||||
|
||||
|
||||
def _sync_qdrant_source_ids(url_hash: str, source_ids: list[str]) -> int:
|
||||
"""把合并后的 source_ids 同步到 Qdrant 中已存在的向量点。"""
|
||||
emb_paths = sorted(Path("data/embeddings").glob(f"*/{url_hash}.json"))
|
||||
if not emb_paths:
|
||||
return 0
|
||||
|
||||
event_path = next(Path("data/events").glob(f"*/{url_hash}.json"), None)
|
||||
if event_path is None:
|
||||
return 0
|
||||
|
||||
try:
|
||||
article_data = json.loads(event_path.read_text(encoding="utf-8"))
|
||||
except Exception as e:
|
||||
logger.warning("读取 events 失败,跳过 Qdrant 来源同步 %s: %s", event_path, e)
|
||||
return 0
|
||||
|
||||
try:
|
||||
from vectorstore.client import VectorStore, make_qdrant_client
|
||||
from vectorstore.pipeline import _build_payload
|
||||
|
||||
client = make_qdrant_client()
|
||||
store = VectorStore(client)
|
||||
try:
|
||||
payload = _build_payload(article_data)
|
||||
for emb_path in emb_paths:
|
||||
emb_data = json.loads(emb_path.read_text(encoding="utf-8"))
|
||||
store.upsert([{
|
||||
"id": url_hash,
|
||||
"vector": emb_data["vector"],
|
||||
"payload": payload,
|
||||
}])
|
||||
finally:
|
||||
store.close()
|
||||
logger.info("来源合并同步 Qdrant: %s -> %s", url_hash, source_ids)
|
||||
return len(emb_paths)
|
||||
except Exception as e:
|
||||
logger.warning("同步 Qdrant 来源失败 %s: %s", url_hash, e)
|
||||
return 0
|
||||
|
||||
|
||||
def _sync_merged_sources_downstream(url_hash: str, source_ids: list[str]) -> None:
|
||||
"""去重合并来源后,同步更新 events 与 Qdrant 中已存在的下游数据。"""
|
||||
_sync_event_source_ids(url_hash, source_ids)
|
||||
_sync_qdrant_source_ids(url_hash, source_ids)
|
||||
|
||||
|
||||
def sync_sources_from_deduped() -> dict:
|
||||
"""历史数据回填:扫描所有 deduped 唯一篇,把多来源同步到 events/Qdrant/MySQL。
|
||||
|
||||
该函数用于修复“deduped 已合并多来源,但旧 events/数据库未更新”的历史数据。
|
||||
"""
|
||||
stats = {
|
||||
"deduped_scanned": 0,
|
||||
"multi_source": 0,
|
||||
"events_updated": 0,
|
||||
"qdrant_updated": 0,
|
||||
"db_updated": 0,
|
||||
}
|
||||
|
||||
targets = sorted(Path("data/deduped").glob("*/uniques/*.json"))
|
||||
for target in targets:
|
||||
try:
|
||||
data = json.loads(target.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
continue
|
||||
stats["deduped_scanned"] += 1
|
||||
|
||||
raw_ids = data.get("source_ids") or [data.get("source_id")]
|
||||
source_ids = list(dict.fromkeys(x for x in raw_ids if x))
|
||||
if len(source_ids) <= 1:
|
||||
continue
|
||||
|
||||
stats["multi_source"] += 1
|
||||
url_hash = data.get("url_hash", target.stem)
|
||||
stats["events_updated"] += _sync_event_source_ids(url_hash, source_ids)
|
||||
stats["qdrant_updated"] += _sync_qdrant_source_ids(url_hash, source_ids)
|
||||
stats["db_updated"] += _update_db_sources(url_hash, source_ids)
|
||||
|
||||
logger.info(
|
||||
"来源回填完成: scanned=%d multi=%d events=%d qdrant=%d db=%d",
|
||||
stats["deduped_scanned"], stats["multi_source"],
|
||||
stats["events_updated"], stats["qdrant_updated"], stats["db_updated"],
|
||||
)
|
||||
return stats
|
||||
|
||||
|
||||
def _update_db_sources(url_hash: str, source_ids: list[str]) -> int:
|
||||
"""根据最新 events 来源,更新 MySQL news_event 中该文章的 source/sources。"""
|
||||
event_path = next(Path("data/events").glob(f"*/{url_hash}.json"), None)
|
||||
if event_path is None:
|
||||
return 0
|
||||
try:
|
||||
article_data = json.loads(event_path.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
try:
|
||||
from scheduler.reporter import _article_source_label, _article_sources_full
|
||||
source_label = _article_source_label(article_data)
|
||||
source_list = _article_sources_full(article_data)
|
||||
if not source_label and not source_list:
|
||||
return 0
|
||||
|
||||
from report_db import connect
|
||||
conn = connect()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
UPDATE news_event
|
||||
SET source = %s, sources = %s
|
||||
WHERE section = 'intl' AND url = %s
|
||||
""",
|
||||
(
|
||||
source_label,
|
||||
json.dumps(source_list, ensure_ascii=False) if source_list else None,
|
||||
article_data.get("url"),
|
||||
),
|
||||
)
|
||||
updated = cur.rowcount
|
||||
conn.commit()
|
||||
return updated or 0
|
||||
finally:
|
||||
conn.close()
|
||||
except Exception as e:
|
||||
logger.warning("同步 MySQL 来源失败 %s: %s", url_hash, e)
|
||||
return 0
|
||||
|
||||
|
||||
def _layer_num(result: DedupResult) -> int:
|
||||
"""DedupResult → 命中层编号。"""
|
||||
if result.matched_layer is None:
|
||||
|
||||
Reference in New Issue
Block a user