feat: 日报结构化入库(M10 前后端分离数据层)
- 新增 report_db 包: MySQL 连接/建表/幂等写入 (news_report/news_event, myquant 库) - 新增 report_import 包: 历史 178 份日报 HTML 解析入库, 表头驱动列映射 - reporter.py 完全切换: generate_report 结构化入库, 不再生成/上传 HTML - CLI: 新增 report-import 子命令 - 依赖: uv add pymysql; 配置: NEWS_DB_* / REPORT_HISTORY_DIR - 文档: docs/report_db_design.md(实现逻辑), docs/db_schema.md(表结构供 API/前端) - 测试: 24 个单测通过 (parser/builder/models/importer)
This commit is contained in:
@@ -0,0 +1,32 @@
|
||||
"""历史导入器单元测试(文件匹配/扫描逻辑,不依赖真实 DB)。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from report_import.importer import _match_file
|
||||
|
||||
|
||||
class TestMatchFile:
|
||||
def test_finance(self) -> None:
|
||||
p = Path("20260711/finance_news_daily_20260710_0720.html")
|
||||
assert _match_file(p, None, None)
|
||||
assert _match_file(p, "20260710", None)
|
||||
assert _match_file(p, None, "finance")
|
||||
assert not _match_file(p, "20260711", None) # 文件名日期不含 20260711
|
||||
assert not _match_file(p, None, "intl")
|
||||
|
||||
def test_no_timestamp_suffix(self) -> None:
|
||||
# 早期文件无时间戳后缀,也应匹配
|
||||
p = Path("20260616/finance_news_daily_20260616.html")
|
||||
assert _match_file(p, None, None)
|
||||
assert _match_file(p, "20260616", "finance")
|
||||
|
||||
def test_intl(self) -> None:
|
||||
p = Path("20260711/intl_news_daily_20260711_070304.html")
|
||||
assert _match_file(p, "20260711", "intl")
|
||||
assert not _match_file(p, None, "finance")
|
||||
|
||||
def test_non_report_ignored(self) -> None:
|
||||
assert not _match_file(Path("20260711/002714.SZ_0724.html"), None, None)
|
||||
assert not _match_file(Path("20260711/readme.md"), None, None)
|
||||
Reference in New Issue
Block a user