本轮会话的三项正确性改造(均为「不报错、只让结果静默错」的类型):
1) 修复 stock_daily 量价单位前后不一致
- 现象:2015-2019 存 Tushare 原始单位(手/千元),2020 起存(股/元),2019 同日混合;
而流动性阈值按「元」配置 → 早年门槛实际是「日均成交额 ≥ 200 亿元」,
把 2015-2019 的股票池整体清空(实测 2016/2017/2018 各选出 0 只)。
- 修复:写入端 sync/price.py 统一换算;读取端 units.normalize_ohlcv_units
按行判定并幂等换算(price_history / avg_amount 都走它);
审计新增 UNIT-OHLCV 防回归。
- 效果:2016/2017/2018 的股票池变为 7/11/13 只。
2) 未来函数守卫(单次回测)
- 股票池自带 asof:若晚于回测起点即**拒绝执行**(原先静默冻结套用),
与 walk-forward 已有的拒绝理由一致;确需复现加 --allow-lookahead-universe,
偏差写入 unimplemented_json。
3) 新增实时(PIT)个股画像闸门
- profile/pit.py:每个决策日按当时可见数据重算过去 5 年画像,
惰性(仅买入条件已触发的标的)、面板按 asof 缓存、
规则不含财务指标时不查财报表;被剔除时产出 REJECT + 逐规则留痕。
- 指标定义复用 ProfileBuilder._profile_one(与批量画像逐值等价的回归测试)。
- profile/coverage.py:窗口覆盖率(按交易日历的真实开市天数),
策略新增 entry.profile_gate.min_window_coverage(默认 0,不改变既有行为)。
- core/metrics.py:闸门可用指标的唯一定义(配置期即校验,避免写错指标名静默失效)。
4) 行情回补到 2005(使 5/8/10 年窗口真正完整)
- stock_daily / adjust_factor / daily_basic 补到 2005-01-04;
hd_suspend / hd_limit 补到 2010-01-04。
- 5 年窗口覆盖率:2018-05-18 由 67.0% → 99.1%,2016-12-30 由 39.8% → 99.0%;
残差经逐日与 hd_suspend 交叉核实为真实停牌(16/16 命中)。
- 审计 G2/G3 与断点续传原先用固定阈值(2000 / 1500 只),
会把 2005-2009 的正常数据误判为异常 —— 改为按「当年应有上市股票数」成比例判定。
- 节流修正:daily/adj_factor/daily_basic 限频 480 → 170(实测该 token 约 196/min 即被拒)。
5) 自我声明如实化
- 原先「约束未生效」由「过滤后集合为空」判定,会把「这批股票恰好没停牌」
误报成「hd_suspend 无数据」;改为按表级判定。
- 补齐此前静默的「配置承诺但未实现」项:suspended_rule/limit_up_down_rule 的 defer、
cash_mode=reinvest/reinvest_rule、handle_rights_issue、signal_to_execution、
max_volume_pct、liquidity_limit_pct_adv —— 全部写入 unimplemented_json。
6) 手册:新增 §0「全流程操作(选股 → 画像 → 回测)」置于最前
- 逐步说明「命令做了什么、数据从哪来、落了哪些库、有哪些坑」;
含实时画像闸门 9 问 9 答、未来函数守卫表、成交与成本口径、验证 SQL。
- 修正旧 §2.4 漏传 --universe-run(选了池子却没用于回测);
修正两处声称「停牌顺延」「分红再投资」已实现的相反表述。
测试:403 项全部通过(含新增 test_units.py、test_profile_pit.py、
未实现声明诚实性测试、行序无关性回归测试)。
注意:本提交中 docs/*、README.md、src/hdiv/web/service.py 除本轮修改外,
也含此前遗留的未提交改动(无法按文件切分)。
269 lines
11 KiB
Python
269 lines
11 KiB
Python
"""同步层测试:值转换、去重键、数据帧构造、限频器。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from datetime import date, datetime
|
||
|
||
import pandas as pd
|
||
import pytest
|
||
|
||
from hdiv.data.db import _frame_to_records
|
||
from hdiv.data.sync.base import nullify_zero, stable_id, to_date, to_float
|
||
from hdiv.data.tushare_client import RateLimiter
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 值转换
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"raw,expected",
|
||
[
|
||
("20240102", date(2024, 1, 2)),
|
||
("19991231", date(1999, 12, 31)),
|
||
(date(2020, 5, 6), date(2020, 5, 6)),
|
||
(datetime(2020, 5, 6, 1, 2, 3), date(2020, 5, 6)),
|
||
("2020-05-06", date(2020, 5, 6)),
|
||
(None, None),
|
||
("", None),
|
||
("nan", None),
|
||
(float("nan"), None),
|
||
],
|
||
)
|
||
def test_to_date(raw, expected) -> None:
|
||
assert to_date(raw) == expected
|
||
|
||
|
||
@pytest.mark.parametrize(
|
||
"raw,expected",
|
||
[(1, 1.0), ("2.5", 2.5), (None, None), ("", None), ("abc", None), (float("nan"), None)],
|
||
)
|
||
def test_to_float(raw, expected) -> None:
|
||
assert to_float(raw) == expected
|
||
|
||
|
||
def test_nullify_zero_treats_zero_as_missing() -> None:
|
||
"""Tushare 用 0 表示「未实施」,需转 None 以免污染统计。"""
|
||
assert nullify_zero(0) is None
|
||
assert nullify_zero("0") is None
|
||
assert nullify_zero(1.5) == 1.5
|
||
assert nullify_zero(None) is None
|
||
|
||
|
||
def test_stable_id_is_deterministic() -> None:
|
||
a = stable_id("dividend", "600036.SH", "2020-01-01")
|
||
b = stable_id("dividend", "600036.SH", "2020-01-01")
|
||
c = stable_id("dividend", "600036.SH", "2020-01-02")
|
||
assert a == b != c
|
||
assert len(a) == 32 # 默认 length=32
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# NaN → SQL NULL(曾导致 "nan can not be used with MySQL" 的真实 bug)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def test_frame_to_records_converts_nan_to_none() -> None:
|
||
df = pd.DataFrame(
|
||
{
|
||
"a": [1.0, float("nan"), 3.0],
|
||
"b": ["x", None, "z"],
|
||
"c": [pd.Timestamp("2024-01-01"), pd.NaT, pd.Timestamp("2024-01-03")],
|
||
"d": pd.Series([1, 2, 3], dtype="Int64"),
|
||
}
|
||
)
|
||
recs = _frame_to_records(df)
|
||
assert recs[0]["a"] == 1.0
|
||
assert recs[1]["a"] is None, "NaN 必须转成 None,否则 pymysql 报错"
|
||
assert recs[1]["b"] is None
|
||
assert recs[1]["c"] is None, "NaT 必须转成 None"
|
||
# object dtype 保证 None 不被强制回 NaN
|
||
assert all(not (isinstance(v, float) and v != v) for r in recs for v in r.values())
|
||
|
||
|
||
def test_frame_to_records_preserves_dates_and_strings() -> None:
|
||
df = pd.DataFrame({"d": [date(2024, 1, 2)], "s": ["银行"], "n": [1.5]})
|
||
rec = _frame_to_records(df)[0]
|
||
assert rec["d"] == date(2024, 1, 2)
|
||
assert rec["s"] == "银行"
|
||
assert rec["n"] == 1.5
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 分红去重键
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def test_dividend_dedup_key_handles_null_ann_date() -> None:
|
||
"""ann_date 为 NULL 时必须有稳定的显式键 —— MySQL 唯一约束不约束 NULL。"""
|
||
from hdiv.data.sync.dividend import build_dedup_key
|
||
|
||
k1 = build_dedup_key("600036.SH", "20260630", "股东提议", None)
|
||
k2 = build_dedup_key("600036.SH", "20260630", "股东提议", None)
|
||
assert k1 == k2, "同样的 NULL 输入必须产生同样的键,否则会重复入库"
|
||
assert "NONE" in k1
|
||
assert k1 != build_dedup_key("600036.SH", "20260630", "预案", None)
|
||
assert k1 != build_dedup_key("600036.SH", "20250630", "股东提议", None)
|
||
|
||
|
||
def test_dividend_dedup_key_distinguishes_ann_date() -> None:
|
||
from hdiv.data.sync.dividend import build_dedup_key
|
||
|
||
a = build_dedup_key("600036.SH", "20251231", "实施", "20260328")
|
||
b = build_dedup_key("600036.SH", "20251231", "实施", "20260626")
|
||
assert a != b, "同一报告期的多次公告必须区分(预案 vs 实施)"
|
||
|
||
|
||
def test_dividend_rows_to_frame_filters_incomplete() -> None:
|
||
from hdiv.data.sync.dividend import rows_to_frame
|
||
|
||
rows = [
|
||
{"ts_code": "600036.SH", "end_date": "20231231", "ann_date": "20240301",
|
||
"div_proc": "实施", "cash_div_tax": 2.0, "ex_date": "20240710"},
|
||
{"ts_code": "600036.SH", "end_date": None, "div_proc": "实施"}, # 缺报告期 → 丢
|
||
{"ts_code": "", "end_date": "20231231", "div_proc": "实施"}, # 缺代码 → 丢
|
||
{"ts_code": "600036.SH", "end_date": "20231231", "div_proc": ""}, # 缺状态 → 丢
|
||
{"ts_code": "600036.SH", "end_date": "20230630", "ann_date": None,
|
||
"div_proc": "股东提议", "cash_div_tax": None}, # 保留(状态完整)
|
||
]
|
||
df = rows_to_frame(rows)
|
||
assert len(df) == 2
|
||
assert set(df["div_proc"]) == {"实施", "股东提议"}
|
||
assert df["symbol"].eq("600036.SH").all()
|
||
# DataFrame 层缺失值表现为 NaN(pandas 的 float64 语义),
|
||
# 但**写入数据库前**必须变成 NULL —— 这才是真正的约束(见 _frame_to_records)
|
||
recs = _frame_to_records(df)
|
||
proposer = next(r for r in recs if r["div_proc"] == "股东提议")
|
||
assert proposer["cash_div_tax"] is None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 财报:ann_date 为空必须丢弃(PIT 纪律)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def test_financial_rows_drop_missing_ann_date() -> None:
|
||
from hdiv.data.sync.financial import SPECS, rows_to_frame
|
||
|
||
spec = SPECS["fina_indicator"]
|
||
rows = [
|
||
{"ts_code": "600036.SH", "end_date": "20231231", "ann_date": "20240301", "roe": 15.0},
|
||
{"ts_code": "600036.SH", "end_date": "20230630", "ann_date": None, "roe": 8.0},
|
||
{"ts_code": "600036.SH", "end_date": None, "ann_date": "20240101", "roe": 1.0},
|
||
]
|
||
df, dropped = rows_to_frame(rows, spec)
|
||
assert len(df) == 1, "缺公告日的记录无法用于 PIT,必须丢弃"
|
||
assert dropped == 2
|
||
assert df.iloc[0]["roe"] == 15.0
|
||
|
||
|
||
def test_financial_report_type_preserved() -> None:
|
||
from hdiv.data.sync.financial import SPECS, rows_to_frame
|
||
|
||
spec = SPECS["income"]
|
||
rows = [
|
||
{"ts_code": "600036.SH", "end_date": "20231231", "ann_date": "20240301",
|
||
"report_type": "1", "total_revenue": 100.0},
|
||
{"ts_code": "600036.SH", "end_date": "20231231", "ann_date": "20240301",
|
||
"report_type": "2", "total_revenue": 30.0},
|
||
{"ts_code": "600036.SH", "end_date": "20230331", "ann_date": "20240401",
|
||
"report_type": None, "total_revenue": 5.0},
|
||
]
|
||
df, _ = rows_to_frame(rows, spec)
|
||
assert len(df) == 3, "report_type 不同不得被折叠(否则合并报表与单季混为一谈)"
|
||
assert set(df["report_type"]) == {"1", "2", "1"}, "缺省应为 1(合并报表)"
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 行情帧构造
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def test_price_frames_map_tushare_columns() -> None:
|
||
from hdiv.data.sync.price import adj_frame, basic_frame, daily_frame
|
||
|
||
d = daily_frame([{"ts_code": "000001.SZ", "trade_date": "20150105", "open": 10,
|
||
"high": 11, "low": 9, "close": 10.5, "vol": 100, "amount": 1000}])
|
||
assert list(d.columns) == ["symbol", "trade_date", "open", "high", "low",
|
||
"close", "volume", "amount", "source", "adjust"]
|
||
assert d.iloc[0]["symbol"] == "000001.SZ"
|
||
assert d.iloc[0]["volume"] == 10000, "Tushare 的 vol(手)必须换算为股(×100)"
|
||
assert d.iloc[0]["amount"] == 1000000, "Tushare 的 amount(千元)必须换算为元(×1000)"
|
||
|
||
a = adj_frame([{"ts_code": "000001.SZ", "trade_date": "20150105", "adj_factor": 1.2}])
|
||
assert a.iloc[0]["factor"] == 1.2
|
||
|
||
b = basic_frame([{"ts_code": "000001.SZ", "trade_date": "20150105", "pe": 8.0,
|
||
"dv_ttm": 4.5, "total_mv": 1e11}])
|
||
assert b.iloc[0]["pe"] == 8.0
|
||
assert b.iloc[0]["dv_ttm"] == 4.5
|
||
assert b.iloc[0]["total_mv"] == 1e11
|
||
|
||
|
||
def test_price_frames_drop_invalid_rows() -> None:
|
||
from hdiv.data.sync.price import daily_frame
|
||
|
||
df = daily_frame([
|
||
{"ts_code": "000001.SZ", "trade_date": "20150105", "close": 10},
|
||
{"ts_code": "", "trade_date": "20150105", "close": 10},
|
||
{"ts_code": "000002.SZ", "trade_date": None, "close": 10},
|
||
])
|
||
assert len(df) == 1
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 限频器
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def test_rate_limiter_enforces_budget() -> None:
|
||
rl = RateLimiter(per_minute=3)
|
||
for _ in range(3):
|
||
rl.acquire()
|
||
assert len(rl._hits) == 3
|
||
rl.reset()
|
||
assert len(rl._hits) == 0
|
||
|
||
|
||
def test_rate_limiter_cooldown_clears_window() -> None:
|
||
rl = RateLimiter(per_minute=2)
|
||
rl.acquire()
|
||
rl.acquire()
|
||
rl.cooldown(0.01)
|
||
assert len(rl._hits) == 0, "冷却必须清空窗口,否则会持续撞限频"
|
||
|
||
|
||
def test_per_api_limits_are_independent() -> None:
|
||
"""限频按接口计算:dividend 与 daily 应有不同额度。"""
|
||
from hdiv.data.tushare_client import TushareClient
|
||
|
||
from hdiv.data import db
|
||
|
||
db.load_dotenv_once()
|
||
cfg = __import__("hdiv.core.config", fromlist=["load_config"]).load_config("datasource")
|
||
ts = cfg.tushare
|
||
assert ts.limit_for("dividend") == 180
|
||
# daily 系列的额度必须**低于实测上限**:以 ~196 次/分钟跑 daily 会被 Tushare
|
||
# 拒绝,触发限频后要冷却 62 秒,逐日回补时远慢于平滑配速(见 datasource.yml 注释)
|
||
assert 100 <= ts.limit_for("daily") <= 200, ts.limit_for("daily")
|
||
assert ts.limit_for("daily") == ts.limit_for("adj_factor") == ts.limit_for("daily_basic")
|
||
assert ts.limit_for("未知接口") == ts.rate_limit_default
|
||
# 构造客户端需要 token;此处仅验证配置层
|
||
assert ts.rate_limit_cooldown_sec >= 60, "冷却必须覆盖 Tushare 的 60 秒滑动窗口"
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# index_weight 的月度切分
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def test_month_starts_split() -> None:
|
||
from hdiv.data.sync.index import _month_starts
|
||
|
||
segs = _month_starts(date(2024, 1, 15), date(2024, 4, 10))
|
||
assert segs[0] == (date(2024, 1, 15), date(2024, 1, 31))
|
||
assert segs[1] == (date(2024, 2, 1), date(2024, 2, 29)) # 闰年
|
||
assert segs[-1] == (date(2024, 4, 1), date(2024, 4, 10))
|
||
assert len(segs) == 4
|