Files
ggx/tests/test_profile_pit.py
simon 14ec0c6c86 修复:量价单位 / 未来函数守卫 / 实时画像闸门;行情回补到 2005;手册补全流程
本轮会话的三项正确性改造(均为「不报错、只让结果静默错」的类型):

1) 修复 stock_daily 量价单位前后不一致
   - 现象:2015-2019 存 Tushare 原始单位(手/千元),2020 起存(股/元),2019 同日混合;
     而流动性阈值按「元」配置 → 早年门槛实际是「日均成交额 ≥ 200 亿元」,
     把 2015-2019 的股票池整体清空(实测 2016/2017/2018 各选出 0 只)。
   - 修复:写入端 sync/price.py 统一换算;读取端 units.normalize_ohlcv_units
     按行判定并幂等换算(price_history / avg_amount 都走它);
     审计新增 UNIT-OHLCV 防回归。
   - 效果:2016/2017/2018 的股票池变为 7/11/13 只。

2) 未来函数守卫(单次回测)
   - 股票池自带 asof:若晚于回测起点即**拒绝执行**(原先静默冻结套用),
     与 walk-forward 已有的拒绝理由一致;确需复现加 --allow-lookahead-universe,
     偏差写入 unimplemented_json。

3) 新增实时(PIT)个股画像闸门
   - profile/pit.py:每个决策日按当时可见数据重算过去 5 年画像,
     惰性(仅买入条件已触发的标的)、面板按 asof 缓存、
     规则不含财务指标时不查财报表;被剔除时产出 REJECT + 逐规则留痕。
   - 指标定义复用 ProfileBuilder._profile_one(与批量画像逐值等价的回归测试)。
   - profile/coverage.py:窗口覆盖率(按交易日历的真实开市天数),
     策略新增 entry.profile_gate.min_window_coverage(默认 0,不改变既有行为)。
   - core/metrics.py:闸门可用指标的唯一定义(配置期即校验,避免写错指标名静默失效)。

4) 行情回补到 2005(使 5/8/10 年窗口真正完整)
   - stock_daily / adjust_factor / daily_basic 补到 2005-01-04;
     hd_suspend / hd_limit 补到 2010-01-04。
   - 5 年窗口覆盖率:2018-05-18 由 67.0% → 99.1%,2016-12-30 由 39.8% → 99.0%;
     残差经逐日与 hd_suspend 交叉核实为真实停牌(16/16 命中)。
   - 审计 G2/G3 与断点续传原先用固定阈值(2000 / 1500 只),
     会把 2005-2009 的正常数据误判为异常 —— 改为按「当年应有上市股票数」成比例判定。
   - 节流修正:daily/adj_factor/daily_basic 限频 480 → 170(实测该 token 约 196/min 即被拒)。

5) 自我声明如实化
   - 原先「约束未生效」由「过滤后集合为空」判定,会把「这批股票恰好没停牌」
     误报成「hd_suspend 无数据」;改为按表级判定。
   - 补齐此前静默的「配置承诺但未实现」项:suspended_rule/limit_up_down_rule 的 defer、
     cash_mode=reinvest/reinvest_rule、handle_rights_issue、signal_to_execution、
     max_volume_pct、liquidity_limit_pct_adv —— 全部写入 unimplemented_json。

6) 手册:新增 §0「全流程操作(选股 → 画像 → 回测)」置于最前
   - 逐步说明「命令做了什么、数据从哪来、落了哪些库、有哪些坑」;
     含实时画像闸门 9 问 9 答、未来函数守卫表、成交与成本口径、验证 SQL。
   - 修正旧 §2.4 漏传 --universe-run(选了池子却没用于回测);
     修正两处声称「停牌顺延」「分红再投资」已实现的相反表述。

测试:403 项全部通过(含新增 test_units.py、test_profile_pit.py、
未实现声明诚实性测试、行序无关性回归测试)。

注意:本提交中 docs/*、README.md、src/hdiv/web/service.py 除本轮修改外,
也含此前遗留的未提交改动(无法按文件切分)。
2026-10-04 12:47:17 +08:00

591 lines
24 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""实时(Point-in-Time)个股画像测试。
三条必须被锁定的性质:
1. **与批量画像同一定义** —— ``PitProfileService`` 在某个 asof 上算出的指标,
必须与 ``ProfileBuilder.run(asof=...)`` 逐值一致。否则「回测用的画像」和
「页面上看的画像」是两个东西,这正是最难发现的一类错误。
2. **PIT 纪律** —— 未公告的财报、未实施/未除权的分红一律不得影响当日画像。
用一个「公告日前一天 vs 公告日当天」的对照来证明,而不是靠注释。
3. **不猜** —— 指标缺失/样本不足必须报 ``MISSING`` / ``INSUFFICIENT``,
闸门据此判定为「无法验证」并按配置保守处理。
"""
from __future__ import annotations
from datetime import date, timedelta
import pytest
from hdiv.profile.builder import METRIC_META
from hdiv.profile.pit import (
ALL_METRICS,
FINANCIAL_METRICS,
PERCENTILE_METRICS,
ProfileSnapshot,
evaluate_gate,
metrics_needing_financials,
)
# ---------------------------------------------------------------------------
# 非 DB:指标集合与闸门语义
# ---------------------------------------------------------------------------
class TestMetricGroups:
def test_declared_metrics_cover_display_names(self) -> None:
"""有中文展示名的指标必须都能作为闸门条件(否则页面能看、回测不能用)。"""
assert frozenset(METRIC_META) <= ALL_METRICS
def test_groups_are_disjoint_and_complete(self) -> None:
from hdiv.profile.pit import (
DIVIDEND_METRICS,
LIQUIDITY_METRICS,
RETURN_METRICS,
VALUATION_METRICS,
)
groups = [VALUATION_METRICS, RETURN_METRICS, DIVIDEND_METRICS,
FINANCIAL_METRICS, LIQUIDITY_METRICS]
union: set[str] = set()
for g in groups:
assert not (union & g), f"指标分组重叠:{union & g}"
union |= g
assert frozenset(union) == ALL_METRICS
def test_percentile_metrics_are_subset_of_valuation(self) -> None:
"""只有按窗口输出分布统计的指标才有历史分位。"""
assert PERCENTILE_METRICS <= ALL_METRICS
assert "dv_vol_daily" not in PERCENTILE_METRICS, "波动率是标量,没有历史分位"
def test_financial_detection(self) -> None:
assert metrics_needing_financials({"roe_avg"}) is True
# 分红质量指标依赖「最近已公告年报」反推考核财年,因此也算需要财报
assert metrics_needing_financials({"dividend_continuity_years"}) is True
assert metrics_needing_financials({"dv_yield", "pe_ttm"}) is False
class TestGateEvaluation:
def _snap(self, **kw) -> ProfileSnapshot:
s = ProfileSnapshot(symbol="X", asof=date(2018, 5, 18), window_years=5)
s.values.update(kw.get("values", {}))
s.percentiles.update(kw.get("percentiles", {}))
s.status.update(kw.get("status", {}))
s.windows.update({
k: kw.get("window", 5)
# 窗口要对**值指标与分位指标**都设上,否则 window=-1 会让覆盖率检查失效
for k in list(kw.get("values", {})) + list(kw.get("percentiles", {}))
})
s.coverage.update(kw.get("coverage", {}))
return s
def test_short_window_is_unverifiable_when_coverage_enforced(self) -> None:
"""名义 5 年但实际只有 67% 数据时:默认放行,强制覆盖率则拦下。
这是「声称 5 年」与「真有 5 年」的分界。实测 600036.SH 在 2018-05-18
的 5 年窗口只有 817/1219 个交易日(67%),而画像仍报 status=OK。
"""
s = self._snap(
percentiles={"dv_yield": 90.0},
status={"dv_yield": "OK"},
coverage={"dv_yield": 0.67},
)
rules = [{"metric": "dv_yield", "stat": "current_percentile",
"op": ">=", "value": 75}]
assert evaluate_gate(rules, s)["verdict"] == "PASS", "默认不因覆盖率淘汰"
g = evaluate_gate(rules, s, min_window_coverage=1.0)
assert g["verdict"] == "REJECT"
assert g["unverifiable"] == ["dv_yield.window_coverage=67%"]
assert g["checks"][0]["window_coverage"] == pytest.approx(0.67)
def test_full_window_passes_coverage_check(self) -> None:
s = self._snap(
percentiles={"dv_yield": 90.0},
status={"dv_yield": "OK"},
coverage={"dv_yield": 1.0},
)
rules = [{"metric": "dv_yield", "stat": "current_percentile",
"op": ">=", "value": 75}]
g = evaluate_gate(rules, s, min_window_coverage=1.0)
assert g["verdict"] == "PASS" and not g["unverifiable"]
def test_full_history_window_is_exempt_from_coverage(self) -> None:
"""窗口 0(全历史)没有「应有天数」,不得因覆盖率被拦。"""
s = ProfileSnapshot(symbol="X", asof=date(2018, 5, 18), window_years=5)
s.values["roe"] = 0.12
s.status["roe"] = "OK"
s.windows["roe"] = 0
s.coverage["roe"] = 1.0
g = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], s,
min_window_coverage=1.0)
assert g["verdict"] == "PASS"
def test_all_rules_pass(self) -> None:
s = self._snap(
values={"payout_ratio": 0.4},
percentiles={"dv_yield": 80.0},
status={"payout_ratio": "OK", "dv_yield": "OK"},
)
r = evaluate_gate([
{"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75},
{"metric": "payout_ratio", "op": "<=", "value": 1.0},
], s)
assert r["verdict"] == "PASS" and not r["failed"]
def test_one_rule_fails_is_reject(self) -> None:
s = self._snap(
values={"payout_ratio": 1.4},
percentiles={"dv_yield": 90.0},
status={"payout_ratio": "OK", "dv_yield": "OK"},
)
r = evaluate_gate([
{"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75},
{"metric": "payout_ratio", "op": "<=", "value": 1.0},
], s)
assert r["verdict"] == "REJECT"
assert r["failed"] == ["payout_ratio.current_value<=1"]
def test_missing_metric_is_not_treated_as_zero(self) -> None:
"""缺失指标绝不能当作 0 —— 否则 `<= 1.0` 这类规则会永远通过。"""
s = self._snap(values={}, status={})
r = evaluate_gate([{"metric": "payout_ratio", "op": "<=", "value": 1.0}], s)
assert r["verdict"] == "REJECT", "默认必须保守(无法验证即不买)"
assert r["unverifiable"] == ["payout_ratio.current_value"]
assert r["checks"][0]["actual"] is None
def test_unverifiable_can_be_configured_to_pass(self) -> None:
s = self._snap(values={}, status={})
r = evaluate_gate(
[{"metric": "payout_ratio", "op": "<=", "value": 1.0}], s,
on_unverifiable="pass",
)
assert r["verdict"] == "PASS" and r["unverifiable"]
def test_insufficient_sample_is_unverifiable(self) -> None:
s = self._snap(values={"roe": 0.1}, status={"roe": "INSUFFICIENT"})
r = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], s)
assert r["verdict"] == "REJECT"
assert r["checks"][0]["status"] == "INSUFFICIENT"
def test_none_snapshot_is_unverifiable(self) -> None:
r = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], None)
assert r["verdict"] == "REJECT" and r["checks"][0]["status"] == "MISSING"
def test_all_comparison_operators(self) -> None:
s = self._snap(values={"x": 5.0}, status={"x": "OK"})
for op, thr, ok in ((">=", 5.0, True), (">", 5.0, False),
("<=", 5.0, True), ("<", 5.0, False)):
r = evaluate_gate([{"metric": "x", "op": op, "value": thr}], s)
assert (r["verdict"] == "PASS") is ok, f"{op} {thr}"
# ---------------------------------------------------------------------------
# DB:与批量画像等价 + PIT 纪律
# ---------------------------------------------------------------------------
def _db_ready():
from hdiv.data import db
db.load_dotenv_once()
return db
@pytest.mark.db
def test_pit_profile_matches_batch_builder() -> None:
"""实时画像必须与 `hdiv profile --asof` 的结果逐值一致(同一定义)。
这是「回测里用的画像」与「页面上看到的画像」不会分叉的唯一保证。
"""
db = _db_ready()
from hdiv.profile.builder import ProfileBuilder
from hdiv.profile.pit import PitProfileService
try:
cfg = db.read_sql(
"SELECT symbol FROM stock WHERE symbol IN ('600036.SH','601398.SH','000651.SZ') "
"ORDER BY symbol LIMIT 3", cfg=__import__("hdiv.core.config", fromlist=["load_config"]).load_config("datasource"),
)
if cfg.empty:
pytest.skip("数据库无样本股票")
syms = cfg["symbol"].tolist()
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
asof = date(2018, 5, 18)
batch = ProfileBuilder.from_config()
svc = PitProfileService(window_years=5)
svc.prepare(syms, date(2004, 1, 1), date(2026, 9, 30))
for sym in syms:
res = batch.run(symbols=[sym], asof=asof, persist=False, verbose=False,
return_rows=True)
want = {
(r["metric_code"], int(r["window_years"])): r["current_value"]
for r in res["stat_rows"] if r["current_value"] is not None
}
snap = svc.snapshot(sym, asof)
assert snap is not None, f"{sym} 应能算出画像"
# 反向守护:实时画像不得产出未声明的指标代码(否则闸门配置无从校验)
assert set(snap.status) <= ALL_METRICS, (
f"未声明的指标:{set(snap.status) - ALL_METRICS}"
)
for (code, wy), v in want.items():
if wy not in (0, 5):
continue
# 实时画像每个指标只保留一个窗口(配置窗口优先)
got, status, got_wy = snap.get(code)
if got is None:
continue
if got_wy != wy:
continue
assert got == pytest.approx(v, rel=1e-9), (
f"{sym} {code} window={wy}: 实时画像 {got} != 批量画像 {v}"
)
@pytest.mark.db
def test_pit_profile_excludes_unannounced_report() -> None:
"""PIT 纪律:公告日前一天不得看到该年报的 ROE。
反例证明:若实现漏了 `ann_date <= asof`,公告日前后两个快照
会给出同一个 ROE(都用了新财报),测试即失败。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
sql = (
"SELECT symbol, end_date, ann_date, roe FROM hd_fina_indicator "
"WHERE MONTH(end_date) = 12 AND roe IS NOT NULL AND ann_date >= end_date "
" AND ann_date >= '2016-01-01' AND ann_date <= '2022-12-31' "
"ORDER BY symbol, ann_date"
)
df = db.read_sql(sql, cfg=load_config("datasource"))
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
if df.empty:
pytest.skip("没有可用的年报样本")
import pandas as pd
df = df.sort_values(["symbol", "ann_date"])
sym = None
row = None
for s, g in df.groupby("symbol"):
g = g.sort_values("ann_date")
if len(g) >= 2:
sym, row = s, g.iloc[1]
break
if sym is None:
pytest.skip("没有「至少两期年报」的样本")
ann = pd.to_datetime(row["ann_date"]).date()
svc = PitProfileService(window_years=5)
svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"roe"})
before = svc.snapshot(sym, ann - timedelta(days=1))
after = svc.snapshot(sym, ann)
if before is None or after is None:
pytest.skip(f"{sym} 在 {ann} 前后无行情")
roe_before, st_before, _ = before.get("roe")
roe_after, st_after, _ = after.get("roe")
if roe_before is None or roe_after is None:
pytest.skip(f"{sym} 缺少 ROE 数据")
assert roe_after == pytest.approx(float(row["roe"]) / 100.0, rel=1e-6), (
"公告日当天应已能看到该年报"
)
assert roe_before != pytest.approx(roe_after, rel=1e-12), (
f"公告日({ann})前一天不得看到该年报的 ROE —— 否则是未来函数"
)
@pytest.mark.db
def test_pit_profile_excludes_future_dividend() -> None:
"""PIT 纪律:未除权的分红不得进入当日 TTM 股息率。"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
df = db.read_sql(
"SELECT symbol, ex_date, cash_div_tax FROM hd_dividend "
"WHERE div_proc='实施' AND cash_div_tax > 0.2 AND ex_date >= '2016-01-01' "
" AND ex_date <= '2022-12-31' ORDER BY cash_div_tax DESC LIMIT 5",
cfg=load_config("datasource"),
)
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
if df.empty:
pytest.skip("没有分红样本")
import pandas as pd
sym = df.iloc[0]["symbol"]
ex = pd.to_datetime(df.iloc[0]["ex_date"]).date()
svc = PitProfileService(window_years=5)
svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"ttm_dps"})
before = svc.snapshot(sym, ex - timedelta(days=1))
after = svc.snapshot(sym, ex)
if before is None or after is None:
pytest.skip(f"{sym} 在除权日 {ex} 前后无行情")
v_before, _, _ = before.get("ttm_dps")
v_after, _, _ = after.get("ttm_dps")
if v_before is None or v_after is None:
pytest.skip(f"{sym} 缺少 TTM DPS")
assert v_after >= v_before, "除权日当天 TTM 分红应把新分红计入"
assert v_after != pytest.approx(v_before, rel=1e-12), (
f"除权日({ex})前一天不得包含该笔分红 —— 否则是未来函数"
)
@pytest.mark.db
def test_pit_profile_rejects_unavailable_window() -> None:
"""请求 profile.yml 未定义的窗口必须报错,而不是悄悄退回全历史。"""
_db_ready()
from hdiv.core.errors import HdivError
from hdiv.profile.pit import PitProfileService
with pytest.raises(HdivError) as ei:
PitProfileService(window_years=3)
assert "windows_years" in str(ei.value)
@pytest.mark.db
def test_pit_profile_is_lazy_and_reuses_panels() -> None:
"""惰性 + 面板复用:这是「长周期数据沿用、触发时才计算」的落地证据。
- 只配估值类规则时,绝不触碰财报表(各约 30 万行);
- 同一 asof 上多只股票复用同一份时点面板(而不是每股查一次库);
- 同一 (股票, asof) 第二次调用直接命中缓存。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol IN "
"('600036.SH','601398.SH','000651.SZ','600519.SH') ORDER BY symbol",
cfg=load_config("datasource"),
)
if rows.empty:
pytest.skip("数据库无样本股票")
syms = rows["symbol"].tolist()
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
asof = date(2020, 6, 30)
svc = PitProfileService(window_years=5)
svc.prepare(syms, date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"dv_yield", "pe_ttm", "pb"}) # 纯估值:不需要任何财报
snaps = [svc.snapshot(s, asof) for s in syms]
assert all(x is not None for x in snaps)
st = svc.stats()
assert st["financial_loads"] == 0, "纯估值规则不得载入财报面板"
assert st["liquidity_loads"] == 0, "未用到成交额时不得查询流动性"
assert st["asof_contexts"] == 1, "同一 asof 只应构建一次时点面板"
assert st["snapshots_computed"] == len(syms)
before = st["snapshots_cached"]
svc.snapshot(syms[0], asof)
assert svc.stats()["snapshots_cached"] == before + 1, "重复调用必须命中缓存"
# 需要财报的指标才会付出那次查询(并且同一 asof 只付一次)
svc2 = PitProfileService(window_years=5)
svc2.prepare(syms, date(2004, 1, 1), date(2026, 9, 30))
svc2.configure({"roe_avg"})
for s in syms:
svc2.snapshot(s, asof)
assert svc2.stats()["financial_loads"] == 1, "同一 asof 的财报面板只应载入一次"
# ---------------------------------------------------------------------------
# 窗口覆盖率:分母必须与 window_slice 的左开右闭口径一致
# ---------------------------------------------------------------------------
class _StubRepo:
"""只提供 trading_days 的最小替身(纯函数测试,不碰数据库)。"""
def __init__(self, days: list[date]) -> None:
self._days = days
def trading_days(self, start: date, end: date) -> list[date]:
return [d for d in self._days if start <= d <= end]
class TestWindowCoverage:
def test_left_endpoint_is_excluded_like_window_slice(self) -> None:
"""交易日历取闭区间,而 window_slice 是 `> start` —— 左端点要减掉。
不减会让覆盖率永远差一天,`min_window_coverage=1.0` 就变成「永远拒绝」。
"""
from hdiv.profile.coverage import expected_trading_days, window_start
asof = date(2021, 1, 4)
lo = window_start(asof, 1) # 2020-01-03
repo = _StubRepo([lo, date(2020, 1, 6), date(2020, 12, 31), asof])
# 闭区间 4 天,去掉左端点 → 3 天
assert expected_trading_days(repo, asof, 1) == 3
def test_left_endpoint_not_a_trading_day(self) -> None:
from hdiv.profile.coverage import expected_trading_days
asof = date(2021, 1, 4)
repo = _StubRepo([date(2020, 1, 6), date(2020, 12, 31), asof])
assert expected_trading_days(repo, asof, 1) == 3
def test_full_history_has_no_denominator(self) -> None:
from hdiv.profile.coverage import expected_trading_days
repo = _StubRepo([date(2020, 1, 6)])
assert expected_trading_days(repo, date(2021, 1, 4), 0) == 0
def test_coverage_ratio_is_capped_at_one(self) -> None:
from hdiv.profile.coverage import coverage_ratio
assert coverage_ratio(100, 200) == pytest.approx(0.5)
assert coverage_ratio(250, 200) == 1.0, "多出来的观测不放大覆盖率"
assert coverage_ratio(10, 0) is None, "没有分母时返回 None(不适用)"
def test_format_warning_only_when_short(self) -> None:
"""只列**不足**的窗口,不要因为 10 年窗口不足就说成「5 年数据不足」。"""
from hdiv.profile.coverage import format_warning
assert format_warning({}) is None
assert format_warning({"windows": {5: {"min": 1.0, "n": 9}}}) is None
msg = format_warning({"windows": {5: {"min": 0.67, "n": 9}}})
assert msg is not None and "67.0%" in msg and "5 年窗口" in msg
mixed = format_warning({
"windows": {5: {"min": 1.0, "n": 9}, 10: {"min": 0.34, "n": 7}},
})
assert mixed is not None
assert "10 年窗口" in mixed
assert "5 年窗口" not in mixed, "已达标的窗口不该出现在警告里"
@pytest.mark.db
def test_real_window_coverage_grows_with_asof() -> None:
"""真实数据:5 年窗口的覆盖率随数据积累而上升,2020 起才满覆盖。
行情/每日指标自 2015-01-05 才有,因此任何早于 2020-01 的 asof,
其「5 年窗口」都是被截短的 —— 这是**数据事实**,不是代码问题,
但必须能被看见。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol = '600036.SH'", cfg=load_config("datasource")
)
if rows.empty:
pytest.skip("无样本股票")
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
svc = PitProfileService(window_years=5)
svc.prepare(["600036.SH"], date(2015, 1, 1), date(2026, 9, 30))
svc.configure({"dv_yield"})
vals = {}
for a in ("2016-12-30", "2018-05-18", "2021-06-30"):
s = svc.snapshot("600036.SH", date.fromisoformat(a))
if s is None:
pytest.skip(f"{a} 无行情")
vals[a] = s.coverage["dv_yield"]
assert vals["2016-12-30"] < 0.5, f"2016 年 5 年窗口应严重不足:{vals}"
assert vals["2018-05-18"] == pytest.approx(0.67, abs=0.02)
assert vals["2021-06-30"] >= 0.999, f"2021 年应已满覆盖:{vals}"
# n_obs 必须一并暴露 —— 它是判断「窗口是否被截断」的原始依据
s = svc.snapshot("600036.SH", date(2018, 5, 18))
assert s.n_obs["dv_yield"] == 817, "实测 2018-05-18 的 5 年窗口为 817 个观测"
@pytest.mark.db
def test_snapshot_ignores_input_row_order() -> None:
"""回归:画像的「当日值」必须由**日期**决定,不能受输入行序影响。
2026-10-04 回补 2005-2014 时,新行是**追加**进 ``daily_basic`` 的,
同一股票的物理行序变成「2015-2026 在前、2005-2014 在后」;
而当时 ``_load_daily_basic`` 没有 ``ORDER BY``、``_profile_one`` 用
``iloc[-1]`` 取当日值 —— 于是格力电器 2018-05-18 的 PE(TTM)
被取成 2014 年的 8.51(真值 12.12)。这个测试把该不变量钉死:
**随机打乱输入面板的行序,结果必须逐值不变。**
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.pit import PitProfileService
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol = '000651.SZ'",
cfg=load_config("datasource"),
)
if rows.empty:
pytest.skip("无样本股票")
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
asof = date(2018, 5, 18)
sym = "000651.SZ"
def _snapshot(shuffle_seed: int | None) -> ProfileSnapshot:
svc = PitProfileService(window_years=5)
svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30))
svc.configure({"pe_ttm", "pb", "dv_yield"})
if shuffle_seed is not None:
# 直接打乱内部面板:模拟「行序不是日期序」
svc._basics = svc._basics.sample(frac=1.0, random_state=shuffle_seed)
svc._price = svc._price.sample(frac=1.0, random_state=shuffle_seed + 1)
s = svc.snapshot(sym, asof)
assert s is not None
return s
ref = _snapshot(None)
for seed in (1, 7, 42):
got = _snapshot(seed)
for metric in ("pe_ttm", "pb", "dv_yield"):
if metric in ref.values and metric in got.values:
assert got.values[metric] == pytest.approx(ref.values[metric], rel=1e-9), (
f"打乱输入行序后 {metric} 变了:{got.values[metric]} != {ref.values[metric]}"
)
# 顺带断言该日 PE(TTM) 就是 12.12 那个量级(防止排序修好后取到错窗口)
assert ref.values["pe_ttm"] == pytest.approx(12.115, rel=1e-3), ref.values["pe_ttm"]
@pytest.mark.db
def test_daily_basic_loader_is_date_sorted() -> None:
"""``_load_daily_basic`` 必须返回按 (symbol, trade_date) 有序的帧。
下游把「最后一个观测」当作当日值 —— 有序性是**语义前提**,不是可选优化。
"""
db = _db_ready()
from hdiv.core.config import load_config
from hdiv.profile.builder import ProfileBuilder
try:
rows = db.read_sql(
"SELECT symbol FROM stock WHERE symbol IN ('000651.SZ','600036.SH') ORDER BY symbol",
cfg=load_config("datasource"),
)
if rows.empty:
pytest.skip("无样本股票")
syms = rows["symbol"].tolist()
except Exception as exc:
pytest.skip(f"数据库不可用:{exc}")
df = ProfileBuilder.from_config()._load_daily_basic(syms, date(2005, 1, 1), date(2026, 9, 30))
assert not df.empty
for sym, g in df.groupby("symbol", sort=False):
assert g["trade_date"].is_monotonic_increasing, f"{sym} 的行序不是日期序"
assert df["trade_date"].min().date() <= date(2005, 1, 10), "回补后应包含 2005 年数据"