diff --git a/README.md b/README.md index 72949b8..7e427a6 100644 --- a/README.md +++ b/README.md @@ -165,7 +165,12 @@ ls -lt data/reports/ ### 去重多来源(M3) -去重时跨源重复的新闻,会把所有来源记录到保留的唯一篇 `source_ids` 字段(首个来源为 `source_id`),经翻译透传后在日报事件 `source` 展示(如 "Barron's, CNBC, Reuters",最多 3 个)。 +去重时跨源重复的新闻,会把所有来源记录到保留的唯一篇 `source_ids` 字段(首个来源为 `source_id`),经翻译透传后: + +- `news_event.source`:拼接展示名(如 "Barron's, CNBC, Reuters",最多前 3 个,向后兼容) +- `news_event.sources`(TEXT,JSON 数组):**全部来源展示名,主源居首**,如 `["Barron's","CNBC","Reuters","Financial Times"]`,前端直接渲染完整来源列表 + +表结构变更(`news_event` 新增 `sources` 列)由 `report_db.init_schema()` 幂等迁移(`ALTER TABLE ... ADD COLUMN IF NOT EXISTS`,MariaDB 10.0.2+)。 ### 中断恢复(--resume) diff --git a/continuation.md b/continuation.md index f7609b5..d440862 100644 --- a/continuation.md +++ b/continuation.md @@ -4,6 +4,28 @@ --- +## 2026-08-12 会话成果(五):日报多来源入库(news_event.sources 列) + +**背景:** 前端需要展示一条新闻的多个来源;`source` 拼接字段(≤3 个)不够。 + +**改动(用户已授权改表,列已由用户添加):** + +| 文件 | 改动内容 | +|------|---------| +| `report_db/models.py` | `EventRow` 新增 `sources: list[str]`(全部来源展示名,主源居首) | +| `report_db/schema.py` | DDL 加 `sources TEXT NULL` 列;`init_schema` 增加幂等 `ALTER TABLE ... ADD COLUMN IF NOT EXISTS sources`(旧表迁移) | +| `report_db/db.py` | `save_report` INSERT 写入 sources(JSON 数组) | +| `scheduler/reporter.py` | 新增 `_article_sources_full()`(完整来源列表,主源居首);`_article_source_label` 重构为基于它(source 保持 ≤3 拼接兼容);`_build_report_data` 事件写入 `sources` | +| `tests/test_report_db.py` | 多源完整列表断言(2 源 / 4 源截断对比 / 单源回退) | + +**DB 迁移:** `news_event.sources` TEXT 列(用户已加);`init_schema()` 幂等补列(MariaDB `ADD COLUMN IF NOT EXISTS`),pi5 实测幂等 OK。 + +**部署中踩坑记录:** `rsync -a report_db/ scheduler/reporter.py pi5:/home/pi/intlnews/` 的混合参数把文件散落到目标根目录(report_db/ 带斜杠 = 内容进根、reporter.py 也进了根),导致 pi5 上模块未更新、sources 写入 None。**教训:rsync 多源参数混用目录与文件时目标路径易错,部署后必须 grep 关键标识(如 source_list)验证。** 已清理根目录残留并正确同步。 + +**验证:** 单元测试 3 用例(多源完整列表);pi5 实盘:`init_schema` 幂等、日报 report_id=224 sources 列全部写入(`["InvestingLive"]` 单源数组)。当前真实数据无多源新闻入日报(跨源合并发生在历史日期文件,未重新翻译),多源数组待后续新数据自然出现(链路已通,前端可直接读 `sources` JSON 数组)。 + +--- + ## 2026-08-12 会话成果(四):脚本阶段/模型显性输出 **背景:** 执行全流程时终端未显性标注"当前阶段",AI 调用的供应商/模型只在客户端初始化时输出一次。 diff --git a/report_db/db.py b/report_db/db.py index 64a179b..509ee91 100644 --- a/report_db/db.py +++ b/report_db/db.py @@ -113,12 +113,13 @@ def save_report(conn: pymysql.Connection, report: ReportData) -> int: cur.execute("DELETE FROM news_event WHERE report_id=%s", (report_id,)) for ev in report.events: + sources_json = json.dumps(ev.sources, ensure_ascii=False) if ev.sources else None cur.execute( """ INSERT INTO news_event (report_id, section, rank, importance, event_type, title, - summary, sentiment, source, url) - VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s) + summary, sentiment, source, sources, url) + VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) """, ( report_id, @@ -130,6 +131,7 @@ def save_report(conn: pymysql.Connection, report: ReportData) -> int: ev.summary, ev.sentiment, ev.source, + sources_json, ev.url, ), ) diff --git a/report_db/models.py b/report_db/models.py index e4cf47b..a0edab3 100644 --- a/report_db/models.py +++ b/report_db/models.py @@ -22,6 +22,10 @@ class EventRow(BaseModel): summary: str | None = None sentiment: str | None = None # positive | negative | neutral | '' source: str | None = None + sources: list[str] = Field( + default_factory=list, + description="全部来源展示名(JSON 数组,主源居首;多源新闻记录完整列表)", + ) url: str | None = None diff --git a/report_db/schema.py b/report_db/schema.py index 0460f1c..9bbf20a 100644 --- a/report_db/schema.py +++ b/report_db/schema.py @@ -34,11 +34,18 @@ DDL_STATEMENTS: list[str] = [ title VARCHAR(512) NOT NULL COMMENT '标题', summary TEXT NULL COMMENT '摘要/正文', sentiment VARCHAR(8) NULL COMMENT 'positive/negative/neutral', - source VARCHAR(64) NULL COMMENT '来源', + source VARCHAR(64) NULL COMMENT '来源(多源时拼接展示名,最多前 3 个)', + sources TEXT NULL COMMENT '全部来源(JSON 数组,主源居首;多源新闻完整列表)', url VARCHAR(512) NULL COMMENT '原文链接', created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, KEY idx_report_section (report_id, section), KEY idx_title (title(255)) ) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_unicode_ci COMMENT='日报事件明细' """, + # 旧表补列(幂等;MariaDB 10.0.2+ 支持 ADD COLUMN IF NOT EXISTS) + """ + ALTER TABLE news_event + ADD COLUMN IF NOT EXISTS sources TEXT NULL + COMMENT '全部来源(JSON 数组,主源居首;多源新闻完整列表)' AFTER source + """, ] diff --git a/scheduler/reporter.py b/scheduler/reporter.py index 2fc9344..762d79b 100644 --- a/scheduler/reporter.py +++ b/scheduler/reporter.py @@ -96,21 +96,31 @@ def _url_source_label(url: str, source_id: str = "") -> str: return domain_map.get(domain, domain) -def _article_source_label(article: dict) -> str | None: - """文章来源展示:去重合并后多来源时拼接展示名,否则回退单源逻辑。 +def _article_sources_full(article: dict) -> list[str]: + """全部来源展示名列表(主源居首、去重)。 - 多来源(ProcessedArticle.source_ids 长度 > 1)时展示如 "Reuters, CNBC", - 最多取前 3 个来源,截断至 64 字符(news_event.source 为 VARCHAR(64))。 + 多来源(ProcessedArticle.source_ids)时返回完整列表供 news_event.sources + JSON 数组使用;无合并信息时回退单源逻辑。 """ src_ids = list(dict.fromkeys( s for s in (article.get("source_ids") or []) if s and s != "?" )) if len(src_ids) > 1: - names = list(dict.fromkeys( - (_source_name(s) or s) for s in src_ids[:3] - )) - return ", ".join(names)[:64] - return _url_source_label(article.get("url"), article.get("source_id", "")) + return list(dict.fromkeys((_source_name(s) or s) for s in src_ids)) + src = _url_source_label(article.get("url"), article.get("source_id", "")) + return [src] if src not in (None, "", "?") else [] + + +def _article_source_label(article: dict) -> str | None: + """文章来源展示(向后兼容 news_event.source):多来源时拼接前 3 个展示名。 + + 多来源(source_ids 长度 > 1)时展示如 "Reuters, CNBC",最多取前 3 个来源, + 截断至 64 字符(source 字段为 VARCHAR(64));完整列表见 _article_sources_full。 + """ + names = _article_sources_full(article) + if len(names) > 1: + return ", ".join(names[:3])[:64] + return names[0] if names else None # 日报覆盖时间窗口(小时) @@ -603,6 +613,8 @@ def _build_report_data( url = article.get("url") or None # 去重合并后的多来源(如 "Reuters, CNBC"),否则回退单源展示 src = _article_source_label(article) + # 全部来源列表(主源居首,JSON 数组,写入 news_event.sources) + source_list = _article_sources_full(article) # 归一化:"" / "?" 不入库,留 None(DB 仅存 positive/negative/neutral) sentiment = ev.get("sentiment") or None if sentiment in ("", "?"): @@ -619,7 +631,8 @@ def _build_report_data( title=title or "(无标题)", summary=ev.get("summary_zh") or None, sentiment=sentiment, - source=src if src not in (None, "", "?") else None, + source=src, + sources=source_list, url=url, ) ) diff --git a/tests/test_report_db.py b/tests/test_report_db.py index d3a169e..e1dbb05 100644 --- a/tests/test_report_db.py +++ b/tests/test_report_db.py @@ -179,17 +179,31 @@ class TestBuildReportData: assert r.events[0].sentiment is None def test_multi_source_label(self) -> None: - """去重合并后的多来源 → source 拼接展示(Reuters, CNBC)。""" + """去重合并后的多来源 → source 拼接展示(Reuters, CNBC)+ sources 完整列表。""" now = datetime(2026, 8, 4, 8, 0, 0) ev = _fake_high_event("多来源事件", 5, source_id="reuters", url="https://reuters.com/news/9") - # 模拟 M3 去重合并:source_ids 含两个来源 + # 模拟 M3 去重合并:source_ids 含两个来源(主源 reuters 居首) ev["article"]["source_ids"] = ["reuters", "cnbc"] r = _build_report_data(now, {}, [ev], Counter(), Counter(), Counter(), Counter(), "") assert r.events[0].source == "Reuters, CNBC" + assert r.events[0].sources == ["Reuters", "CNBC"] # 主源居首 assert r.events[0].url == "https://reuters.com/news/9" + def test_multi_source_full_list(self) -> None: + """超过 3 个来源:source 截前 3,sources 存完整列表。""" + now = datetime(2026, 8, 4, 8, 0, 0) + ev = _fake_high_event("四来源事件", 5, source_id="barrons", + url="https://barrons.com/news/1") + ev["article"]["source_ids"] = ["barrons", "cnbc", "reuters", "ft"] + r = _build_report_data(now, {}, [ev], Counter(), Counter(), Counter(), + Counter(), "") + assert r.events[0].source == "Barron's, CNBC, Reuters" # 前 3 拼接 + assert r.events[0].sources == [ # 完整 4 个 + "Barron's", "CNBC", "Reuters", "Financial Times", + ] + def test_single_source_falls_back(self) -> None: """source_ids 为空/单一时回退单源逻辑(不拼接)。""" now = datetime(2026, 8, 4, 8, 0, 0) @@ -199,3 +213,4 @@ class TestBuildReportData: r = _build_report_data(now, {}, [ev], Counter(), Counter(), Counter(), Counter(), "") assert r.events[0].source == "InvestingLive" + assert r.events[0].sources == ["InvestingLive"]