修复:量价单位 / 未来函数守卫 / 实时画像闸门;行情回补到 2005;手册补全流程

本轮会话的三项正确性改造(均为「不报错、只让结果静默错」的类型):

1) 修复 stock_daily 量价单位前后不一致
   - 现象:2015-2019 存 Tushare 原始单位(手/千元),2020 起存(股/元),2019 同日混合;
     而流动性阈值按「元」配置 → 早年门槛实际是「日均成交额 ≥ 200 亿元」,
     把 2015-2019 的股票池整体清空(实测 2016/2017/2018 各选出 0 只)。
   - 修复:写入端 sync/price.py 统一换算;读取端 units.normalize_ohlcv_units
     按行判定并幂等换算(price_history / avg_amount 都走它);
     审计新增 UNIT-OHLCV 防回归。
   - 效果:2016/2017/2018 的股票池变为 7/11/13 只。

2) 未来函数守卫(单次回测)
   - 股票池自带 asof:若晚于回测起点即**拒绝执行**(原先静默冻结套用),
     与 walk-forward 已有的拒绝理由一致;确需复现加 --allow-lookahead-universe,
     偏差写入 unimplemented_json。

3) 新增实时(PIT)个股画像闸门
   - profile/pit.py:每个决策日按当时可见数据重算过去 5 年画像,
     惰性(仅买入条件已触发的标的)、面板按 asof 缓存、
     规则不含财务指标时不查财报表;被剔除时产出 REJECT + 逐规则留痕。
   - 指标定义复用 ProfileBuilder._profile_one(与批量画像逐值等价的回归测试)。
   - profile/coverage.py:窗口覆盖率(按交易日历的真实开市天数),
     策略新增 entry.profile_gate.min_window_coverage(默认 0,不改变既有行为)。
   - core/metrics.py:闸门可用指标的唯一定义(配置期即校验,避免写错指标名静默失效)。

4) 行情回补到 2005(使 5/8/10 年窗口真正完整)
   - stock_daily / adjust_factor / daily_basic 补到 2005-01-04;
     hd_suspend / hd_limit 补到 2010-01-04。
   - 5 年窗口覆盖率:2018-05-18 由 67.0% → 99.1%,2016-12-30 由 39.8% → 99.0%;
     残差经逐日与 hd_suspend 交叉核实为真实停牌(16/16 命中)。
   - 审计 G2/G3 与断点续传原先用固定阈值(2000 / 1500 只),
     会把 2005-2009 的正常数据误判为异常 —— 改为按「当年应有上市股票数」成比例判定。
   - 节流修正:daily/adj_factor/daily_basic 限频 480 → 170(实测该 token 约 196/min 即被拒)。

5) 自我声明如实化
   - 原先「约束未生效」由「过滤后集合为空」判定,会把「这批股票恰好没停牌」
     误报成「hd_suspend 无数据」;改为按表级判定。
   - 补齐此前静默的「配置承诺但未实现」项:suspended_rule/limit_up_down_rule 的 defer、
     cash_mode=reinvest/reinvest_rule、handle_rights_issue、signal_to_execution、
     max_volume_pct、liquidity_limit_pct_adv —— 全部写入 unimplemented_json。

6) 手册:新增 §0「全流程操作(选股 → 画像 → 回测)」置于最前
   - 逐步说明「命令做了什么、数据从哪来、落了哪些库、有哪些坑」;
     含实时画像闸门 9 问 9 答、未来函数守卫表、成交与成本口径、验证 SQL。
   - 修正旧 §2.4 漏传 --universe-run(选了池子却没用于回测);
     修正两处声称「停牌顺延」「分红再投资」已实现的相反表述。

测试:403 项全部通过(含新增 test_units.py、test_profile_pit.py、
未实现声明诚实性测试、行序无关性回归测试)。

注意:本提交中 docs/*、README.md、src/hdiv/web/service.py 除本轮修改外,
也含此前遗留的未提交改动(无法按文件切分)。
This commit is contained in:
2026-10-04 12:47:17 +08:00
parent fb6608193b
commit 14ec0c6c86
25 changed files with 4543 additions and 209 deletions
+590
View File
@@ -0,0 +1,590 @@
"""实时(Point-in-Time)个股画像测试。
三条必须被锁定的性质:
1. **与批量画像同一定义** —— ``PitProfileService`` 在某个 asof 上算出的指标,
必须与 ``ProfileBuilder.run(asof=...)`` 逐值一致。否则「回测用的画像」和
「页面上看的画像」是两个东西,这正是最难发现的一类错误。
2. **PIT 纪律** —— 未公告的财报、未实施/未除权的分红一律不得影响当日画像。
用一个「公告日前一天 vs 公告日当天」的对照来证明,而不是靠注释。
3. **不猜** —— 指标缺失/样本不足必须报 ``MISSING`` / ``INSUFFICIENT``,
闸门据此判定为「无法验证」并按配置保守处理。
"""
from __future__ import annotations
from datetime import date, timedelta
import pytest
from hdiv.profile.builder import METRIC_META
from hdiv.profile.pit import (
ALL_METRICS,
FINANCIAL_METRICS,
PERCENTILE_METRICS,
ProfileSnapshot,
evaluate_gate,
metrics_needing_financials,
)
# ---------------------------------------------------------------------------
# 非 DB:指标集合与闸门语义
# ---------------------------------------------------------------------------
class TestMetricGroups:
def test_declared_metrics_cover_display_names(self) -> None:
"""有中文展示名的指标必须都能作为闸门条件(否则页面能看、回测不能用)。"""
assert frozenset(METRIC_META) <= ALL_METRICS
def test_groups_are_disjoint_and_complete(self) -> None:
from hdiv.profile.pit import (
DIVIDEND_METRICS,
LIQUIDITY_METRICS,
RETURN_METRICS,
VALUATION_METRICS,
)
groups = [VALUATION_METRICS, RETURN_METRICS, DIVIDEND_METRICS,
FINANCIAL_METRICS, LIQUIDITY_METRICS]
union: set[str] = set()
for g in groups:
assert not (union & g), f"指标分组重叠:{union & g}"
union |= g
assert frozenset(union) == ALL_METRICS
def test_percentile_metrics_are_subset_of_valuation(self) -> None:
"""只有按窗口输出分布统计的指标才有历史分位。"""
assert PERCENTILE_METRICS <= ALL_METRICS
assert "dv_vol_daily" not in PERCENTILE_METRICS, "波动率是标量,没有历史分位"
def test_financial_detection(self) -> None:
assert metrics_needing_financials({"roe_avg"}) is True
# 分红质量指标依赖「最近已公告年报」反推考核财年,因此也算需要财报
assert metrics_needing_financials({"dividend_continuity_years"}) is True
assert metrics_needing_financials({"dv_yield", "pe_ttm"}) is False
class TestGateEvaluation:
def _snap(self, **kw) -> ProfileSnapshot:
s = ProfileSnapshot(symbol="X", asof=date(2018, 5, 18), window_years=5)
s.values.update(kw.get("values", {}))
s.percentiles.update(kw.get("percentiles", {}))
s.status.update(kw.get("status", {}))
s.windows.update({
k: kw.get("window", 5)
# 窗口要对**值指标与分位指标**都设上,否则 window=-1 会让覆盖率检查失效
for k in list(kw.get("values", {})) + list(kw.get("percentiles", {}))
})
s.coverage.update(kw.get("coverage", {}))
return s
def test_short_window_is_unverifiable_when_coverage_enforced(self) -> None:
"""名义 5 年但实际只有 67% 数据时:默认放行,强制覆盖率则拦下。
这是「声称 5 年」与「真有 5 年」的分界。实测 600036.SH 在 2018-05-18
的 5 年窗口只有 817/1219 个交易日(67%),而画像仍报 status=OK。
"""
s = self._snap(
percentiles={"dv_yield": 90.0},
status={"dv_yield": "OK"},
coverage={"dv_yield": 0.67},
)
rules = [{"metric": "dv_yield", "stat": "current_percentile",
"op": ">=", "value": 75}]
assert evaluate_gate(rules, s)["verdict"] == "PASS", "默认不因覆盖率淘汰"
g = evaluate_gate(rules, s, min_window_coverage=1.0)
assert g["verdict"] == "REJECT"
assert g["unverifiable"] == ["dv_yield.window_coverage=67%"]
assert g["checks"][0]["window_coverage"] == pytest.approx(0.67)
def test_full_window_passes_coverage_check(self) -> None:
s = self._snap(
percentiles={"dv_yield": 90.0},
status={"dv_yield": "OK"},
coverage={"dv_yield": 1.0},
)
rules = [{"metric": "dv_yield", "stat": "current_percentile",
"op": ">=", "value": 75}]
g = evaluate_gate(rules, s, min_window_coverage=1.0)
assert g["verdict"] == "PASS" and not g["unverifiable"]
def test_full_history_window_is_exempt_from_coverage(self) -> None:
"""窗口 0(全历史)没有「应有天数」,不得因覆盖率被拦。"""
s = ProfileSnapshot(symbol="X", asof=date(2018, 5, 18), window_years=5)
s.values["roe"] = 0.12
s.status["roe"] = "OK"
s.windows["roe"] = 0
s.coverage["roe"] = 1.0
g = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], s,
min_window_coverage=1.0)
assert g["verdict"] == "PASS"
def test_all_rules_pass(self) -> None:
s = self._snap(
values={"payout_ratio": 0.4},
percentiles={"dv_yield": 80.0},
status={"payout_ratio": "OK", "dv_yield": "OK"},
)
r = evaluate_gate([
{"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75},
{"metric": "payout_ratio", "op": "<=", "value": 1.0},
], s)
assert r["verdict"] == "PASS" and not r["failed"]
def test_one_rule_fails_is_reject(self) -> None:
s = self._snap(
values={"payout_ratio": 1.4},
percentiles={"dv_yield": 90.0},
status={"payout_ratio": "OK", "dv_yield": "OK"},
)
r = evaluate_gate([
{"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75},
{"metric": "payout_ratio", "op": "<=", "value": 1.0},
], s)
assert r["verdict"] == "REJECT"
assert r["failed"] == ["payout_ratio.current_value<=1"]
def test_missing_metric_is_not_treated_as_zero(self) -> None:
"""缺失指标绝不能当作 0 —— 否则 `<= 1.0` 这类规则会永远通过。"""
s = self._snap(values={}, status={})
r = evaluate_gate([{"metric": "payout_ratio", "op": "<=", "value": 1.0}], s)
assert r["verdict"] == "REJECT", "默认必须保守(无法验证即不买)"
assert r["unverifiable"] == ["payout_ratio.current_value"]
assert r["checks"][0]["actual"] is None
def test_unverifiable_can_be_configured_to_pass(self) -> None:
s = self._snap(values={}, status={})
r = evaluate_gate(
[{"metric": "payout_ratio", "op": "<=", "value": 1.0}], s,
on_unverifiable="pass",
)
assert r["verdict"] == "PASS" and r["unverifiable"]
def test_insufficient_sample_is_unverifiable(self) -> None:
s = self._snap(values={"roe": 0.1}, status={"roe": "INSUFFICIENT"})
r = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], s)
assert r["verdict"] == "REJECT"
assert r["checks"][0]["status"] == "INSUFFICIENT"
def test_none_snapshot_is_unverifiable(self) -> None:
r = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], None)
assert r["verdict"] == "REJECT" and r["checks"][0]["status"] == "MISSING"
def test_all_comparison_operators(self) -> None:
s = self._snap(values={"x": 5.0}, status={"x": "OK"})
for op, thr, ok in ((">=", 5.0, True), (">", 5.0, False),
("<=", 5.0, True), ("<", 5.0, False)):
r = evaluate_gate([{"metric": "x", "op": op, "value": thr}], s)
assert (r["verdict"] == "PASS") is ok, f"{op} {thr}"
# ---------------------------------------------------------------------------
# DB:与批量画像等价 + PIT 纪律
# ---------------------------------------------------------------------------
def _db_ready():
from hdiv.data import db
db.load_dotenv_once()
return db
@pytest.mark.db
def test_pit_profile_matches_batch_builder() -> None:
"""实时画像必须与 `hdiv profile --asof` 的结果逐值一致(同一定义)。
这是「回测里用的画像」与「页面上看到的画像」不会分叉的唯一保证。
"""
db = _db_ready()
from hdiv.profile.builder import ProfileBuilder
from hdiv.profile.pit import PitProfileService
try:
cfg = db.read_sql(
"SELECT symbol FROM stock WHERE symbol IN ('600036.SH','601398.SH','000651.SZ') "
"ORDER BY symbol LIMIT 3", cfg=__import__("hdiv.core.config", fromlist=["load_config"]).load_config("datasource"),
)
if cfg.empty:
pytest.skip("数据库无样本股票")
syms = cfg["symbol"].tolist()
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
asof = date(2018, 5, 18)
batch = ProfileBuilder.from_config()
svc = PitProfileService(window_years=5)
svc.prepare(syms, date(2004, 1, 1), date(2026, 9, 30))
for sym in syms:
res = batch.run(symbols=[sym], asof=asof, persist=False, verbose=False,
return_rows=True)
want = {
(r["metric_code"], int(r["window_years"])): r["current_value"]
for r in res["stat_rows"] if r["current_value"] is not None
}
snap = svc.snapshot(sym, asof)
assert snap is not None, f"{sym} 应能算出画像"
# 反向守护:实时画像不得产出未声明的指标代码(否则闸门配置无从校验)
assert set(snap.status) <= ALL_METRICS, (
f"未声明的指标:{set(snap.status) - ALL_METRICS}"
)
for (code, wy), v in want.items():
if wy not in (0, 5):
continue
# 实时画像每个指标只保留一个窗口(配置窗口优先)
got, status, got_wy = snap.get(code)
if got is None:
continue
if got_wy != wy:
continue
assert got == pytest.approx(v, rel=1e-9), (
f"{sym} {code} window={wy}: 实时画像 {got} != 批量画像 {v}"
)
@pytest.mark.db
def test_pit_profile_excludes_unannounced_report() -> None:
"""PIT 纪律:公告日前一天不得看到该年报的 ROE。
反例证明:若实现漏了 `ann_date <= asof`,公告日前后两个快照
会给出同一个 ROE(都用了新财报),测试即失败。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
sql = (
"SELECT symbol, end_date, ann_date, roe FROM hd_fina_indicator "
"WHERE MONTH(end_date) = 12 AND roe IS NOT NULL AND ann_date >= end_date "
" AND ann_date >= '2016-01-01' AND ann_date <= '2022-12-31' "
"ORDER BY symbol, ann_date"
)
df = db.read_sql(sql, cfg=load_config("datasource"))
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
if df.empty:
pytest.skip("没有可用的年报样本")
import pandas as pd
df = df.sort_values(["symbol", "ann_date"])
sym = None
row = None
for s, g in df.groupby("symbol"):
g = g.sort_values("ann_date")
if len(g) >= 2:
sym, row = s, g.iloc[1]
break
if sym is None:
pytest.skip("没有「至少两期年报」的样本")
ann = pd.to_datetime(row["ann_date"]).date()
svc = PitProfileService(window_years=5)
svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"roe"})
before = svc.snapshot(sym, ann - timedelta(days=1))
after = svc.snapshot(sym, ann)
if before is None or after is None:
pytest.skip(f"{sym} 在 {ann} 前后无行情")
roe_before, st_before, _ = before.get("roe")
roe_after, st_after, _ = after.get("roe")
if roe_before is None or roe_after is None:
pytest.skip(f"{sym} 缺少 ROE 数据")
assert roe_after == pytest.approx(float(row["roe"]) / 100.0, rel=1e-6), (
"公告日当天应已能看到该年报"
)
assert roe_before != pytest.approx(roe_after, rel=1e-12), (
f"公告日({ann})前一天不得看到该年报的 ROE —— 否则是未来函数"
)
@pytest.mark.db
def test_pit_profile_excludes_future_dividend() -> None:
"""PIT 纪律:未除权的分红不得进入当日 TTM 股息率。"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
df = db.read_sql(
"SELECT symbol, ex_date, cash_div_tax FROM hd_dividend "
"WHERE div_proc='实施' AND cash_div_tax > 0.2 AND ex_date >= '2016-01-01' "
" AND ex_date <= '2022-12-31' ORDER BY cash_div_tax DESC LIMIT 5",
cfg=load_config("datasource"),
)
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
if df.empty:
pytest.skip("没有分红样本")
import pandas as pd
sym = df.iloc[0]["symbol"]
ex = pd.to_datetime(df.iloc[0]["ex_date"]).date()
svc = PitProfileService(window_years=5)
svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"ttm_dps"})
before = svc.snapshot(sym, ex - timedelta(days=1))
after = svc.snapshot(sym, ex)
if before is None or after is None:
pytest.skip(f"{sym} 在除权日 {ex} 前后无行情")
v_before, _, _ = before.get("ttm_dps")
v_after, _, _ = after.get("ttm_dps")
if v_before is None or v_after is None:
pytest.skip(f"{sym} 缺少 TTM DPS")
assert v_after >= v_before, "除权日当天 TTM 分红应把新分红计入"
assert v_after != pytest.approx(v_before, rel=1e-12), (
f"除权日({ex})前一天不得包含该笔分红 —— 否则是未来函数"
)
@pytest.mark.db
def test_pit_profile_rejects_unavailable_window() -> None:
"""请求 profile.yml 未定义的窗口必须报错,而不是悄悄退回全历史。"""
_db_ready()
from hdiv.core.errors import HdivError
from hdiv.profile.pit import PitProfileService
with pytest.raises(HdivError) as ei:
PitProfileService(window_years=3)
assert "windows_years" in str(ei.value)
@pytest.mark.db
def test_pit_profile_is_lazy_and_reuses_panels() -> None:
"""惰性 + 面板复用:这是「长周期数据沿用、触发时才计算」的落地证据。
- 只配估值类规则时,绝不触碰财报表(各约 30 万行);
- 同一 asof 上多只股票复用同一份时点面板(而不是每股查一次库);
- 同一 (股票, asof) 第二次调用直接命中缓存。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol IN "
"('600036.SH','601398.SH','000651.SZ','600519.SH') ORDER BY symbol",
cfg=load_config("datasource"),
)
if rows.empty:
pytest.skip("数据库无样本股票")
syms = rows["symbol"].tolist()
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
asof = date(2020, 6, 30)
svc = PitProfileService(window_years=5)
svc.prepare(syms, date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"dv_yield", "pe_ttm", "pb"}) # 纯估值:不需要任何财报
snaps = [svc.snapshot(s, asof) for s in syms]
assert all(x is not None for x in snaps)
st = svc.stats()
assert st["financial_loads"] == 0, "纯估值规则不得载入财报面板"
assert st["liquidity_loads"] == 0, "未用到成交额时不得查询流动性"
assert st["asof_contexts"] == 1, "同一 asof 只应构建一次时点面板"
assert st["snapshots_computed"] == len(syms)
before = st["snapshots_cached"]
svc.snapshot(syms[0], asof)
assert svc.stats()["snapshots_cached"] == before + 1, "重复调用必须命中缓存"
# 需要财报的指标才会付出那次查询(并且同一 asof 只付一次)
svc2 = PitProfileService(window_years=5)
svc2.prepare(syms, date(2004, 1, 1), date(2026, 9, 30))
svc2.configure({"roe_avg"})
for s in syms:
svc2.snapshot(s, asof)
assert svc2.stats()["financial_loads"] == 1, "同一 asof 的财报面板只应载入一次"
# ---------------------------------------------------------------------------
# 窗口覆盖率:分母必须与 window_slice 的左开右闭口径一致
# ---------------------------------------------------------------------------
class _StubRepo:
"""只提供 trading_days 的最小替身(纯函数测试,不碰数据库)。"""
def __init__(self, days: list[date]) -> None:
self._days = days
def trading_days(self, start: date, end: date) -> list[date]:
return [d for d in self._days if start <= d <= end]
class TestWindowCoverage:
def test_left_endpoint_is_excluded_like_window_slice(self) -> None:
"""交易日历取闭区间,而 window_slice 是 `> start` —— 左端点要减掉。
不减会让覆盖率永远差一天,`min_window_coverage=1.0` 就变成「永远拒绝」。
"""
from hdiv.profile.coverage import expected_trading_days, window_start
asof = date(2021, 1, 4)
lo = window_start(asof, 1) # 2020-01-03
repo = _StubRepo([lo, date(2020, 1, 6), date(2020, 12, 31), asof])
# 闭区间 4 天,去掉左端点 → 3 天
assert expected_trading_days(repo, asof, 1) == 3
def test_left_endpoint_not_a_trading_day(self) -> None:
from hdiv.profile.coverage import expected_trading_days
asof = date(2021, 1, 4)
repo = _StubRepo([date(2020, 1, 6), date(2020, 12, 31), asof])
assert expected_trading_days(repo, asof, 1) == 3
def test_full_history_has_no_denominator(self) -> None:
from hdiv.profile.coverage import expected_trading_days
repo = _StubRepo([date(2020, 1, 6)])
assert expected_trading_days(repo, date(2021, 1, 4), 0) == 0
def test_coverage_ratio_is_capped_at_one(self) -> None:
from hdiv.profile.coverage import coverage_ratio
assert coverage_ratio(100, 200) == pytest.approx(0.5)
assert coverage_ratio(250, 200) == 1.0, "多出来的观测不放大覆盖率"
assert coverage_ratio(10, 0) is None, "没有分母时返回 None(不适用)"
def test_format_warning_only_when_short(self) -> None:
"""只列**不足**的窗口,不要因为 10 年窗口不足就说成「5 年数据不足」。"""
from hdiv.profile.coverage import format_warning
assert format_warning({}) is None
assert format_warning({"windows": {5: {"min": 1.0, "n": 9}}}) is None
msg = format_warning({"windows": {5: {"min": 0.67, "n": 9}}})
assert msg is not None and "67.0%" in msg and "5 年窗口" in msg
mixed = format_warning({
"windows": {5: {"min": 1.0, "n": 9}, 10: {"min": 0.34, "n": 7}},
})
assert mixed is not None
assert "10 年窗口" in mixed
assert "5 年窗口" not in mixed, "已达标的窗口不该出现在警告里"
@pytest.mark.db
def test_real_window_coverage_grows_with_asof() -> None:
"""真实数据:5 年窗口的覆盖率随数据积累而上升,2020 起才满覆盖。
行情/每日指标自 2015-01-05 才有,因此任何早于 2020-01 的 asof,
其「5 年窗口」都是被截短的 —— 这是**数据事实**,不是代码问题,
但必须能被看见。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol = '600036.SH'", cfg=load_config("datasource")
)
if rows.empty:
pytest.skip("无样本股票")
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
svc = PitProfileService(window_years=5)
svc.prepare(["600036.SH"], date(2015, 1, 1), date(2026, 9, 30))
svc.configure({"dv_yield"})
vals = {}
for a in ("2016-12-30", "2018-05-18", "2021-06-30"):
s = svc.snapshot("600036.SH", date.fromisoformat(a))
if s is None:
pytest.skip(f"{a} 无行情")
vals[a] = s.coverage["dv_yield"]
assert vals["2016-12-30"] < 0.5, f"2016 年 5 年窗口应严重不足:{vals}"
assert vals["2018-05-18"] == pytest.approx(0.67, abs=0.02)
assert vals["2021-06-30"] >= 0.999, f"2021 年应已满覆盖:{vals}"
# n_obs 必须一并暴露 —— 它是判断「窗口是否被截断」的原始依据
s = svc.snapshot("600036.SH", date(2018, 5, 18))
assert s.n_obs["dv_yield"] == 817, "实测 2018-05-18 的 5 年窗口为 817 个观测"
@pytest.mark.db
def test_snapshot_ignores_input_row_order() -> None:
"""回归:画像的「当日值」必须由**日期**决定,不能受输入行序影响。
2026-10-04 回补 2005-2014 时,新行是**追加**进 ``daily_basic`` 的,
同一股票的物理行序变成「2015-2026 在前、2005-2014 在后」;
而当时 ``_load_daily_basic`` 没有 ``ORDER BY``、``_profile_one`` 用
``iloc[-1]`` 取当日值 —— 于是格力电器 2018-05-18 的 PE(TTM)
被取成 2014 年的 8.51(真值 12.12)。这个测试把该不变量钉死:
**随机打乱输入面板的行序,结果必须逐值不变。**
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol = '000651.SZ'",
cfg=load_config("datasource"),
)
if rows.empty:
pytest.skip("无样本股票")
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
asof = date(2018, 5, 18)
sym = "000651.SZ"
def _snapshot(shuffle_seed: int | None) -> ProfileSnapshot:
svc = PitProfileService(window_years=5)
svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"pe_ttm", "pb", "dv_yield"})
if shuffle_seed is not None:
# 直接打乱内部面板:模拟「行序不是日期序」
svc._basics = svc._basics.sample(frac=1.0, random_state=shuffle_seed)
svc._price = svc._price.sample(frac=1.0, random_state=shuffle_seed + 1)
s = svc.snapshot(sym, asof)
assert s is not None
return s
ref = _snapshot(None)
for seed in (1, 7, 42):
got = _snapshot(seed)
for metric in ("pe_ttm", "pb", "dv_yield"):
if metric in ref.values and metric in got.values:
assert got.values[metric] == pytest.approx(ref.values[metric], rel=1e-9), (
f"打乱输入行序后 {metric} 变了:{got.values[metric]} != {ref.values[metric]}"
)
# 顺带断言该日 PE(TTM) 就是 12.12 那个量级(防止排序修好后取到错窗口)
assert ref.values["pe_ttm"] == pytest.approx(12.115, rel=1e-3), ref.values["pe_ttm"]
@pytest.mark.db
def test_daily_basic_loader_is_date_sorted() -> None:
"""``_load_daily_basic`` 必须返回按 (symbol, trade_date) 有序的帧。
下游把「最后一个观测」当作当日值 —— 有序性是**语义前提**,不是可选优化。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.builder import ProfileBuilder
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol IN ('000651.SZ','600036.SH') ORDER BY symbol",
cfg=load_config("datasource"),
)
if rows.empty:
pytest.skip("无样本股票")
syms = rows["symbol"].tolist()
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
df = ProfileBuilder.from_config()._load_daily_basic(syms, date(2005, 1, 1), date(2026, 9, 30))
assert not df.empty
for sym, g in df.groupby("symbol", sort=False):
assert g["trade_date"].is_monotonic_increasing, f"{sym} 的行序不是日期序"
assert df["trade_date"].min().date() <= date(2005, 1, 10), "回补后应包含 2005 年数据"