feat: 日报结构化入库(M10 前后端分离数据层)

- 新增 report_db 包: MySQL 连接/建表/幂等写入 (news_report/news_event, myquant 库)
- 新增 report_import 包: 历史 178 份日报 HTML 解析入库, 表头驱动列映射
- reporter.py 完全切换: generate_report 结构化入库, 不再生成/上传 HTML
- CLI: 新增 report-import 子命令
- 依赖: uv add pymysql; 配置: NEWS_DB_* / REPORT_HISTORY_DIR
- 文档: docs/report_db_design.md(实现逻辑), docs/db_schema.md(表结构供 API/前端)
- 测试: 24 个单测通过 (parser/builder/models/importer)
This commit is contained in:
2026-08-03 21:32:07 +08:00
parent f2c80c5a9c
commit 366e60e8a9
25 changed files with 1934 additions and 33 deletions
+32
View File
@@ -0,0 +1,32 @@
"""历史导入器单元测试(文件匹配/扫描逻辑,不依赖真实 DB)。"""
from __future__ import annotations
from pathlib import Path
from report_import.importer import _match_file
class TestMatchFile:
def test_finance(self) -> None:
p = Path("20260711/finance_news_daily_20260710_0720.html")
assert _match_file(p, None, None)
assert _match_file(p, "20260710", None)
assert _match_file(p, None, "finance")
assert not _match_file(p, "20260711", None) # 文件名日期不含 20260711
assert not _match_file(p, None, "intl")
def test_no_timestamp_suffix(self) -> None:
# 早期文件无时间戳后缀,也应匹配
p = Path("20260616/finance_news_daily_20260616.html")
assert _match_file(p, None, None)
assert _match_file(p, "20260616", "finance")
def test_intl(self) -> None:
p = Path("20260711/intl_news_daily_20260711_070304.html")
assert _match_file(p, "20260711", "intl")
assert not _match_file(p, None, "finance")
def test_non_report_ignored(self) -> None:
assert not _match_file(Path("20260711/002714.SZ_0724.html"), None, None)
assert not _match_file(Path("20260711/readme.md"), None, None)