Files
ggx/tests/test_schema.py
T
simon cf6d4d2c56 功能:每日动态股票池回测(--mode daily)+ 每日增量同步 + PIT 批量取数层
说明:本提交是工作区中此前的未提交工作(在 14ec0c6 之后产生),**非本次会话所写**,
按用户要求整理并推送。已做安全检查(无明文凭据、无大文件、.env/logs/output 仍被忽略),
并完成可执行范围内的测试验证(见「测试」一节)。

## 新增能力

1) `hdiv backtest --mode daily --start <日期>`
   - src/hdiv/backtest/daily.py:两趟式(先逐日选股,再复用既有引擎模拟)
   - 每个交易日按当日可见数据重建股票池(PIT),每个交易日判断买卖点
   - `pool_exit_action`:hold(只减不加、不因掉出池子而清仓)/ sell(掉出即清仓)
   - `profile_on_trade`:买卖决策发生时计算并留痕个股画像,**不区分是否在当日池内**
     (卖出/减仓同样留痕,否则「为什么卖」缺证据)
   - 与 walkforward 的分工:daily 是一条连续路径的推演,不是过拟合检验;
     因此不使用训练段、不冻结分布,阈值口径一律 rolling
   - 拒绝 `--universe-run`(daily 的定义就是逐日重筛,冻结池与之矛盾)

2) PIT 批量取数层 src/hdiv/universe/pit.py
   - PitRepo 继承 Repo,**只重写取数**(按区块批量预载 + 逐日内存切片),
     派生逻辑(最新一期财报合并、单位归一化、支付率口径等)一行不重写
     —— 以保证与逐日单点查询**结果等价**
   - 候选集预剪枝:用「不可能通过」的边界条件提前排除,文档论证为精确等价而非近似
   - src/hdiv/universe/daily.py:每日动态筛选器(仍然调用既有 selector 与四个 Filter)

3) 每日增量同步 `hdiv sync daily`
   - src/hdiv/data/sync/daily.py:只抓「库里还没有的那几天」,
     按「当日股票数 ≥ 当年规模阈值」判定缺口,不重拉历史、不覆盖既有行;
     支持 `--dry-run` 先看待抓清单
   - deploy/daily-sync.sh、deploy/install-sync-schedule.sh、
     deploy/com.hddiv.sync.plist.example(launchd 每天 17:00)
   - 新表 hd_daily_universe(逐日入选成员留痕)+ sql/hd_daily_universe.sql + schema.py
     (该表已存在于库中,`ddl plan` 返回 0 个待执行动作)

4) Web 与文档
   - 前端支持 daily 模式记录下钻(web/app.js、web/app.css、web/index.html、
     web/favicon.svg)
   - README / docs/user-guide.md / docs/implementation-status.md 同步更新:
     三种回测模式的取舍、daily 的成本说明(6.7 年约 1.5 小时)与调优手段

## 测试

tests/ 共 500 项(新增 tests/test_daily.py 43 项、tests/test_sync_daily.py 36 项)。

已验证通过:
- 排除上述两个新文件的 **421 项:全部通过(pytest 退出码 0)**
- 两个新文件的**非 DB 单元测试 60 项:全部通过**

未能在合理时间内跑完:
- 两个新文件中 **19 项 DB 标记的重型测试**。实测瓶颈是一条**无界全表扫描**:
  `SELECT ... FROM hd_cashflow WHERE ann_date <= :asof ORDER BY symbol, end_date, ann_date`
  (31 万行,无 symbol/报告期下限)。全量套件跑到 161 项时已耗时 20 分钟、
  0 失败,按该速率预计需 3 小时以上,因此改为分档验证。
- 旁证:库中存在 3 次成功的 daily 端到端运行(2026-10-05 10:05 / 10:32 / 11:03,
  区间 2024-03-01~03-15),说明该路径可正常完成。

## 已知待改进

- 上述 `hd_cashflow`(及同类「按 ann_date 上界取全历史」)的查询缺
  symbol / 报告期下限,是 daily 模式的主要性能瓶颈,建议下一轮优化。
2026-10-05 11:57:13 +08:00

204 lines
7.0 KiB
Python

"""表结构与幂等性测试(对应 development-plan.md §7.2 硬约束)。
需要真实数据库的用例标记为 ``db``;纯结构校验的用例无需数据库。
"""
from __future__ import annotations
import re
import pytest
from hdiv.core.config import load_config
from hdiv.core.errors import SafetyViolation
from hdiv.data.schema import ALL_TABLES, TABLE_NAMES
CFG = load_config("datasource")
# 需要库的表清单(无库时跳过)
_PLAIN_TESTS_NEED_DB = pytest.mark.db
# ---------------------------------------------------------------------------
# 纯结构校验(无需数据库)
# ---------------------------------------------------------------------------
def test_table_count() -> None:
assert len(ALL_TABLES) == 31
assert len(set(TABLE_NAMES)) == 31
def test_all_tables_use_own_prefix() -> None:
bad = [t.name for t in ALL_TABLES if not t.name.startswith(CFG.database.own_prefix)]
assert not bad, f"以下表未使用自有前缀:{bad}"
def test_ddl_never_contains_destructive_statements() -> None:
destructive = re.compile(r"\b(DROP|TRUNCATE|DELETE)\b", re.IGNORECASE)
for t in ALL_TABLES:
assert not destructive.search(t.ddl), f"{t.name} 的 DDL 含破坏性语句"
for m in t.migrations:
assert not re.search(r"\b(DROP\s+TABLE|TRUNCATE|DELETE)\b", m.sql, re.IGNORECASE), (
f"{t.name} 迁移 {m.name} 含破坏性语句:{m.sql}"
)
def test_create_statements_are_idempotent() -> None:
for t in ALL_TABLES:
assert "CREATE TABLE IF NOT EXISTS" in t.ddl, f"{t.name} 未使用 IF NOT EXISTS"
def test_every_table_has_primary_key() -> None:
for t in ALL_TABLES:
assert "PRIMARY KEY" in t.ddl, f"{t.name} 缺少主键"
def test_tables_have_comment() -> None:
for t in ALL_TABLES:
assert t.comment, f"{t.name} 缺少注释"
assert "COMMENT='" in t.ddl or 'COMMENT="' in t.ddl, f"{t.name} 的 DDL 缺少表注释"
def test_no_nullable_column_in_unique_key() -> None:
"""MySQL 唯一约束不约束 NULL —— 参与唯一键的列必须 NOT NULL。
唯一例外是显式构造的 dedup_key(本身 NOT NULL)。
"""
problems: list[str] = []
for t in ALL_TABLES:
for m in re.finditer(r"UNIQUE KEY `\w+` \(([^)]+)\)", t.ddl):
cols = [c.strip().strip("`") for c in m.group(1).split(",")]
for col in cols:
# 找到该列的定义行
pat = re.compile(rf"^\s*`{re.escape(col)}`\s+([A-Za-z]+(?:\([^)]*\))?)(.*)$", re.M)
mm = pat.search(t.ddl)
if not mm:
problems.append(f"{t.name}.{col} 未找到列定义")
continue
rest = mm.group(2)
if "NOT NULL" not in rest:
problems.append(
f"{t.name}.{col} 可空却参与唯一键 —— NULL 会绕过唯一约束造成重复"
)
assert not problems, "\n".join(problems)
def test_character_set_is_utf8mb4() -> None:
for t in ALL_TABLES:
assert "utf8mb4" in t.ddl, f"{t.name} 未使用 utf8mb4"
def test_engine_is_innodb() -> None:
for t in ALL_TABLES:
assert "InnoDB" in t.ddl, f"{t.name} 未使用 InnoDB"
def test_migration_names_are_unique_within_table() -> None:
for t in ALL_TABLES:
names = [m.name for m in t.migrations]
assert len(names) == len(set(names)), f"{t.name} 迁移名重复:{names}"
def test_migrations_do_not_add_dropped_tables() -> None:
"""迁移只允许 ADD/MODIFY,不允许改表名或删列。"""
for t in ALL_TABLES:
for m in t.migrations:
assert not re.search(r"\bDROP\s+COLUMN\b", m.sql, re.IGNORECASE), m.sql
assert not re.search(r"\bRENAME\b", m.sql, re.IGNORECASE), m.sql
def test_own_prefix_guard_rejects_foreign_table() -> None:
from hdiv.data.ddl import _assert_own_prefix
with pytest.raises(SafetyViolation):
_assert_own_prefix(CFG, "stock")
with pytest.raises(SafetyViolation):
_assert_own_prefix(CFG, "hd_ok") if False else _assert_own_prefix(CFG, "alembic_version")
def test_own_prefix_guard_accepts_own_table() -> None:
from hdiv.data.ddl import _assert_own_prefix
_assert_own_prefix(CFG, "hd_dividend") # 不应抛错
# ---------------------------------------------------------------------------
# 需要数据库
# ---------------------------------------------------------------------------
@_PLAIN_TESTS_NEED_DB
def test_ddl_plan_is_idempotent_on_live_db() -> None:
"""连续两次 plan 都应返回 0 个动作(结构已就绪且迁移已应用)。"""
pytest.importorskip("pymysql")
from hdiv.data import ddl, db
db.load_dotenv_once()
try:
db.list_tables(CFG)
except Exception as exc: # pragma: no cover - 环境相关
pytest.skip(f"数据库不可用:{exc}")
actions = ddl.plan(CFG)
assert actions == [], f"结构未收敛,仍待执行:{[(a.kind, a.table, a.detail) for a in actions]}"
@_PLAIN_TESTS_NEED_DB
def test_all_declared_tables_exist() -> None:
pytest.importorskip("pymysql")
from hdiv.data import db
db.load_dotenv_once()
try:
existing = set(db.list_tables(CFG))
except Exception as exc: # pragma: no cover
pytest.skip(f"数据库不可用:{exc}")
missing = [t for t in TABLE_NAMES if t not in existing]
assert not missing, f"库中缺表:{missing}"
@_PLAIN_TESTS_NEED_DB
def test_verify_reports_no_problems() -> None:
pytest.importorskip("pymysql")
from hdiv.data import db, ddl
db.load_dotenv_once()
try:
problems = ddl.verify(CFG)
except Exception as exc: # pragma: no cover
pytest.skip(f"数据库不可用:{exc}")
assert not problems, "\n".join(problems)
@_PLAIN_TESTS_NEED_DB
def test_alembic_version_untouched() -> None:
"""决策 D1:本项目不得修改 qlib 的 Alembic 版本链。"""
pytest.importorskip("pymysql")
from hdiv.data import db
db.load_dotenv_once()
try:
df = db.read_sql("SELECT version_num FROM alembic_version", cfg=CFG)
except Exception as exc: # pragma: no cover
pytest.skip(f"数据库不可用:{exc}")
# 只要存在且可读即可;本项目从未写入该表(由 SafetyViolation 保证)
assert len(df) >= 1
@_PLAIN_TESTS_NEED_DB
def test_project_created_tables_all_have_own_prefix() -> None:
"""本项目建的表必须 100% 匹配 hd_ 前缀。"""
pytest.importorskip("pymysql")
from hdiv.data import db
from hdiv.data.schema import TABLE_NAMES as declared
db.load_dotenv_once()
try:
existing = set(db.list_tables(CFG))
except Exception as exc: # pragma: no cover
pytest.skip(f"数据库不可用:{exc}")
hd_tables = {t for t in existing if t.startswith(CFG.database.own_prefix)}
assert hd_tables <= set(declared), f"库中存在未声明的 hd_ 表:{hd_tables - set(declared)}"
assert set(declared) <= existing, f"声明的表未全部创建:{set(declared) - existing}"