"""实时(Point-in-Time)个股画像测试。 三条必须被锁定的性质: 1. **与批量画像同一定义** —— ``PitProfileService`` 在某个 asof 上算出的指标, 必须与 ``ProfileBuilder.run(asof=...)`` 逐值一致。否则「回测用的画像」和 「页面上看的画像」是两个东西,这正是最难发现的一类错误。 2. **PIT 纪律** —— 未公告的财报、未实施/未除权的分红一律不得影响当日画像。 用一个「公告日前一天 vs 公告日当天」的对照来证明,而不是靠注释。 3. **不猜** —— 指标缺失/样本不足必须报 ``MISSING`` / ``INSUFFICIENT``, 闸门据此判定为「无法验证」并按配置保守处理。 """ from __future__ import annotations from datetime import date, timedelta import pytest from hdiv.profile.builder import METRIC_META from hdiv.profile.pit import ( ALL_METRICS, FINANCIAL_METRICS, PERCENTILE_METRICS, ProfileSnapshot, evaluate_gate, metrics_needing_financials, ) # --------------------------------------------------------------------------- # 非 DB:指标集合与闸门语义 # --------------------------------------------------------------------------- class TestMetricGroups: def test_declared_metrics_cover_display_names(self) -> None: """有中文展示名的指标必须都能作为闸门条件(否则页面能看、回测不能用)。""" assert frozenset(METRIC_META) <= ALL_METRICS def test_groups_are_disjoint_and_complete(self) -> None: from hdiv.profile.pit import ( DIVIDEND_METRICS, LIQUIDITY_METRICS, RETURN_METRICS, VALUATION_METRICS, ) groups = [VALUATION_METRICS, RETURN_METRICS, DIVIDEND_METRICS, FINANCIAL_METRICS, LIQUIDITY_METRICS] union: set[str] = set() for g in groups: assert not (union & g), f"指标分组重叠:{union & g}" union |= g assert frozenset(union) == ALL_METRICS def test_percentile_metrics_are_subset_of_valuation(self) -> None: """只有按窗口输出分布统计的指标才有历史分位。""" assert PERCENTILE_METRICS <= ALL_METRICS assert "dv_vol_daily" not in PERCENTILE_METRICS, "波动率是标量,没有历史分位" def test_financial_detection(self) -> None: assert metrics_needing_financials({"roe_avg"}) is True # 分红质量指标依赖「最近已公告年报」反推考核财年,因此也算需要财报 assert metrics_needing_financials({"dividend_continuity_years"}) is True assert metrics_needing_financials({"dv_yield", "pe_ttm"}) is False class TestGateEvaluation: def _snap(self, **kw) -> ProfileSnapshot: s = ProfileSnapshot(symbol="X", asof=date(2018, 5, 18), window_years=5) s.values.update(kw.get("values", {})) s.percentiles.update(kw.get("percentiles", {})) s.status.update(kw.get("status", {})) s.windows.update({ k: kw.get("window", 5) # 窗口要对**值指标与分位指标**都设上,否则 window=-1 会让覆盖率检查失效 for k in list(kw.get("values", {})) + list(kw.get("percentiles", {})) }) s.coverage.update(kw.get("coverage", {})) return s def test_short_window_is_unverifiable_when_coverage_enforced(self) -> None: """名义 5 年但实际只有 67% 数据时:默认放行,强制覆盖率则拦下。 这是「声称 5 年」与「真有 5 年」的分界。实测 600036.SH 在 2018-05-18 的 5 年窗口只有 817/1219 个交易日(67%),而画像仍报 status=OK。 """ s = self._snap( percentiles={"dv_yield": 90.0}, status={"dv_yield": "OK"}, coverage={"dv_yield": 0.67}, ) rules = [{"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75}] assert evaluate_gate(rules, s)["verdict"] == "PASS", "默认不因覆盖率淘汰" g = evaluate_gate(rules, s, min_window_coverage=1.0) assert g["verdict"] == "REJECT" assert g["unverifiable"] == ["dv_yield.window_coverage=67%"] assert g["checks"][0]["window_coverage"] == pytest.approx(0.67) def test_full_window_passes_coverage_check(self) -> None: s = self._snap( percentiles={"dv_yield": 90.0}, status={"dv_yield": "OK"}, coverage={"dv_yield": 1.0}, ) rules = [{"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75}] g = evaluate_gate(rules, s, min_window_coverage=1.0) assert g["verdict"] == "PASS" and not g["unverifiable"] def test_full_history_window_is_exempt_from_coverage(self) -> None: """窗口 0(全历史)没有「应有天数」,不得因覆盖率被拦。""" s = ProfileSnapshot(symbol="X", asof=date(2018, 5, 18), window_years=5) s.values["roe"] = 0.12 s.status["roe"] = "OK" s.windows["roe"] = 0 s.coverage["roe"] = 1.0 g = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], s, min_window_coverage=1.0) assert g["verdict"] == "PASS" def test_all_rules_pass(self) -> None: s = self._snap( values={"payout_ratio": 0.4}, percentiles={"dv_yield": 80.0}, status={"payout_ratio": "OK", "dv_yield": "OK"}, ) r = evaluate_gate([ {"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75}, {"metric": "payout_ratio", "op": "<=", "value": 1.0}, ], s) assert r["verdict"] == "PASS" and not r["failed"] def test_one_rule_fails_is_reject(self) -> None: s = self._snap( values={"payout_ratio": 1.4}, percentiles={"dv_yield": 90.0}, status={"payout_ratio": "OK", "dv_yield": "OK"}, ) r = evaluate_gate([ {"metric": "dv_yield", "stat": "current_percentile", "op": ">=", "value": 75}, {"metric": "payout_ratio", "op": "<=", "value": 1.0}, ], s) assert r["verdict"] == "REJECT" assert r["failed"] == ["payout_ratio.current_value<=1"] def test_missing_metric_is_not_treated_as_zero(self) -> None: """缺失指标绝不能当作 0 —— 否则 `<= 1.0` 这类规则会永远通过。""" s = self._snap(values={}, status={}) r = evaluate_gate([{"metric": "payout_ratio", "op": "<=", "value": 1.0}], s) assert r["verdict"] == "REJECT", "默认必须保守(无法验证即不买)" assert r["unverifiable"] == ["payout_ratio.current_value"] assert r["checks"][0]["actual"] is None def test_unverifiable_can_be_configured_to_pass(self) -> None: s = self._snap(values={}, status={}) r = evaluate_gate( [{"metric": "payout_ratio", "op": "<=", "value": 1.0}], s, on_unverifiable="pass", ) assert r["verdict"] == "PASS" and r["unverifiable"] def test_insufficient_sample_is_unverifiable(self) -> None: s = self._snap(values={"roe": 0.1}, status={"roe": "INSUFFICIENT"}) r = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], s) assert r["verdict"] == "REJECT" assert r["checks"][0]["status"] == "INSUFFICIENT" def test_none_snapshot_is_unverifiable(self) -> None: r = evaluate_gate([{"metric": "roe", "op": ">=", "value": 0.08}], None) assert r["verdict"] == "REJECT" and r["checks"][0]["status"] == "MISSING" def test_all_comparison_operators(self) -> None: s = self._snap(values={"x": 5.0}, status={"x": "OK"}) for op, thr, ok in ((">=", 5.0, True), (">", 5.0, False), ("<=", 5.0, True), ("<", 5.0, False)): r = evaluate_gate([{"metric": "x", "op": op, "value": thr}], s) assert (r["verdict"] == "PASS") is ok, f"{op} {thr}" # --------------------------------------------------------------------------- # DB:与批量画像等价 + PIT 纪律 # --------------------------------------------------------------------------- def _db_ready(): from hdiv.data import db db.load_dotenv_once() return db @pytest.mark.db def test_pit_profile_matches_batch_builder() -> None: """实时画像必须与 `hdiv profile --asof` 的结果逐值一致(同一定义)。 这是「回测里用的画像」与「页面上看到的画像」不会分叉的唯一保证。 """ db = _db_ready() from hdiv.profile.builder import ProfileBuilder from hdiv.profile.pit import PitProfileService try: cfg = db.read_sql( "SELECT symbol FROM stock WHERE symbol IN ('600036.SH','601398.SH','000651.SZ') " "ORDER BY symbol LIMIT 3", cfg=__import__("hdiv.core.config", fromlist=["load_config"]).load_config("datasource"), ) if cfg.empty: pytest.skip("数据库无样本股票") syms = cfg["symbol"].tolist() except Exception as exc: pytest.skip(f"数据库不可用:{exc}") asof = date(2018, 5, 18) batch = ProfileBuilder.from_config() svc = PitProfileService(window_years=5) svc.prepare(syms, date(2004, 1, 1), date(2026, 9, 30)) for sym in syms: res = batch.run(symbols=[sym], asof=asof, persist=False, verbose=False, return_rows=True) want = { (r["metric_code"], int(r["window_years"])): r["current_value"] for r in res["stat_rows"] if r["current_value"] is not None } snap = svc.snapshot(sym, asof) assert snap is not None, f"{sym} 应能算出画像" # 反向守护:实时画像不得产出未声明的指标代码(否则闸门配置无从校验) assert set(snap.status) <= ALL_METRICS, ( f"未声明的指标:{set(snap.status) - ALL_METRICS}" ) for (code, wy), v in want.items(): if wy not in (0, 5): continue # 实时画像每个指标只保留一个窗口(配置窗口优先) got, status, got_wy = snap.get(code) if got is None: continue if got_wy != wy: continue assert got == pytest.approx(v, rel=1e-9), ( f"{sym} {code} window={wy}: 实时画像 {got} != 批量画像 {v}" ) @pytest.mark.db def test_pit_profile_excludes_unannounced_report() -> None: """PIT 纪律:公告日前一天不得看到该年报的 ROE。 反例证明:若实现漏了 `ann_date <= asof`,公告日前后两个快照 会给出同一个 ROE(都用了新财报),测试即失败。 """ db = _db_ready() from hdiv.core.config import load_config from hdiv.profile.pit import PitProfileService try: sql = ( "SELECT symbol, end_date, ann_date, roe FROM hd_fina_indicator " "WHERE MONTH(end_date) = 12 AND roe IS NOT NULL AND ann_date >= end_date " " AND ann_date >= '2016-01-01' AND ann_date <= '2022-12-31' " "ORDER BY symbol, ann_date" ) df = db.read_sql(sql, cfg=load_config("datasource")) except Exception as exc: pytest.skip(f"数据库不可用:{exc}") if df.empty: pytest.skip("没有可用的年报样本") import pandas as pd df = df.sort_values(["symbol", "ann_date"]) sym = None row = None for s, g in df.groupby("symbol"): g = g.sort_values("ann_date") if len(g) >= 2: sym, row = s, g.iloc[1] break if sym is None: pytest.skip("没有「至少两期年报」的样本") ann = pd.to_datetime(row["ann_date"]).date() svc = PitProfileService(window_years=5) svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30)) svc.configure({"roe"}) before = svc.snapshot(sym, ann - timedelta(days=1)) after = svc.snapshot(sym, ann) if before is None or after is None: pytest.skip(f"{sym} 在 {ann} 前后无行情") roe_before, st_before, _ = before.get("roe") roe_after, st_after, _ = after.get("roe") if roe_before is None or roe_after is None: pytest.skip(f"{sym} 缺少 ROE 数据") assert roe_after == pytest.approx(float(row["roe"]) / 100.0, rel=1e-6), ( "公告日当天应已能看到该年报" ) assert roe_before != pytest.approx(roe_after, rel=1e-12), ( f"公告日({ann})前一天不得看到该年报的 ROE —— 否则是未来函数" ) @pytest.mark.db def test_pit_profile_excludes_future_dividend() -> None: """PIT 纪律:未除权的分红不得进入当日 TTM 股息率。""" db = _db_ready() from hdiv.core.config import load_config from hdiv.profile.pit import PitProfileService try: df = db.read_sql( "SELECT symbol, ex_date, cash_div_tax FROM hd_dividend " "WHERE div_proc='实施' AND cash_div_tax > 0.2 AND ex_date >= '2016-01-01' " " AND ex_date <= '2022-12-31' ORDER BY cash_div_tax DESC LIMIT 5", cfg=load_config("datasource"), ) except Exception as exc: pytest.skip(f"数据库不可用:{exc}") if df.empty: pytest.skip("没有分红样本") import pandas as pd sym = df.iloc[0]["symbol"] ex = pd.to_datetime(df.iloc[0]["ex_date"]).date() svc = PitProfileService(window_years=5) svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30)) svc.configure({"ttm_dps"}) before = svc.snapshot(sym, ex - timedelta(days=1)) after = svc.snapshot(sym, ex) if before is None or after is None: pytest.skip(f"{sym} 在除权日 {ex} 前后无行情") v_before, _, _ = before.get("ttm_dps") v_after, _, _ = after.get("ttm_dps") if v_before is None or v_after is None: pytest.skip(f"{sym} 缺少 TTM DPS") assert v_after >= v_before, "除权日当天 TTM 分红应把新分红计入" assert v_after != pytest.approx(v_before, rel=1e-12), ( f"除权日({ex})前一天不得包含该笔分红 —— 否则是未来函数" ) @pytest.mark.db def test_pit_profile_rejects_unavailable_window() -> None: """请求 profile.yml 未定义的窗口必须报错,而不是悄悄退回全历史。""" _db_ready() from hdiv.core.errors import HdivError from hdiv.profile.pit import PitProfileService with pytest.raises(HdivError) as ei: PitProfileService(window_years=3) assert "windows_years" in str(ei.value) @pytest.mark.db def test_pit_profile_is_lazy_and_reuses_panels() -> None: """惰性 + 面板复用:这是「长周期数据沿用、触发时才计算」的落地证据。 - 只配估值类规则时,绝不触碰财报表(各约 30 万行); - 同一 asof 上多只股票复用同一份时点面板(而不是每股查一次库); - 同一 (股票, asof) 第二次调用直接命中缓存。 """ db = _db_ready() from hdiv.core.config import load_config from hdiv.profile.pit import PitProfileService try: rows = db.read_sql( "SELECT symbol FROM stock WHERE symbol IN " "('600036.SH','601398.SH','000651.SZ','600519.SH') ORDER BY symbol", cfg=load_config("datasource"), ) if rows.empty: pytest.skip("数据库无样本股票") syms = rows["symbol"].tolist() except Exception as exc: pytest.skip(f"数据库不可用:{exc}") asof = date(2020, 6, 30) svc = PitProfileService(window_years=5) svc.prepare(syms, date(2004, 1, 1), date(2026, 9, 30)) svc.configure({"dv_yield", "pe_ttm", "pb"}) # 纯估值:不需要任何财报 snaps = [svc.snapshot(s, asof) for s in syms] assert all(x is not None for x in snaps) st = svc.stats() assert st["financial_loads"] == 0, "纯估值规则不得载入财报面板" assert st["liquidity_loads"] == 0, "未用到成交额时不得查询流动性" assert st["asof_contexts"] == 1, "同一 asof 只应构建一次时点面板" assert st["snapshots_computed"] == len(syms) before = st["snapshots_cached"] svc.snapshot(syms[0], asof) assert svc.stats()["snapshots_cached"] == before + 1, "重复调用必须命中缓存" # 需要财报的指标才会付出那次查询(并且同一 asof 只付一次) svc2 = PitProfileService(window_years=5) svc2.prepare(syms, date(2004, 1, 1), date(2026, 9, 30)) svc2.configure({"roe_avg"}) for s in syms: svc2.snapshot(s, asof) assert svc2.stats()["financial_loads"] == 1, "同一 asof 的财报面板只应载入一次" # --------------------------------------------------------------------------- # 窗口覆盖率:分母必须与 window_slice 的左开右闭口径一致 # --------------------------------------------------------------------------- class _StubRepo: """只提供 trading_days 的最小替身(纯函数测试,不碰数据库)。""" def __init__(self, days: list[date]) -> None: self._days = days def trading_days(self, start: date, end: date) -> list[date]: return [d for d in self._days if start <= d <= end] class TestWindowCoverage: def test_left_endpoint_is_excluded_like_window_slice(self) -> None: """交易日历取闭区间,而 window_slice 是 `> start` —— 左端点要减掉。 不减会让覆盖率永远差一天,`min_window_coverage=1.0` 就变成「永远拒绝」。 """ from hdiv.profile.coverage import expected_trading_days, window_start asof = date(2021, 1, 4) lo = window_start(asof, 1) # 2020-01-03 repo = _StubRepo([lo, date(2020, 1, 6), date(2020, 12, 31), asof]) # 闭区间 4 天,去掉左端点 → 3 天 assert expected_trading_days(repo, asof, 1) == 3 def test_left_endpoint_not_a_trading_day(self) -> None: from hdiv.profile.coverage import expected_trading_days asof = date(2021, 1, 4) repo = _StubRepo([date(2020, 1, 6), date(2020, 12, 31), asof]) assert expected_trading_days(repo, asof, 1) == 3 def test_full_history_has_no_denominator(self) -> None: from hdiv.profile.coverage import expected_trading_days repo = _StubRepo([date(2020, 1, 6)]) assert expected_trading_days(repo, date(2021, 1, 4), 0) == 0 def test_coverage_ratio_is_capped_at_one(self) -> None: from hdiv.profile.coverage import coverage_ratio assert coverage_ratio(100, 200) == pytest.approx(0.5) assert coverage_ratio(250, 200) == 1.0, "多出来的观测不放大覆盖率" assert coverage_ratio(10, 0) is None, "没有分母时返回 None(不适用)" def test_format_warning_only_when_short(self) -> None: """只列**不足**的窗口,不要因为 10 年窗口不足就说成「5 年数据不足」。""" from hdiv.profile.coverage import format_warning assert format_warning({}) is None assert format_warning({"windows": {5: {"min": 1.0, "n": 9}}}) is None msg = format_warning({"windows": {5: {"min": 0.67, "n": 9}}}) assert msg is not None and "67.0%" in msg and "5 年窗口" in msg mixed = format_warning({ "windows": {5: {"min": 1.0, "n": 9}, 10: {"min": 0.34, "n": 7}}, }) assert mixed is not None assert "10 年窗口" in mixed assert "5 年窗口" not in mixed, "已达标的窗口不该出现在警告里" @pytest.mark.db def test_real_window_coverage_grows_with_asof() -> None: """真实数据:5 年窗口的覆盖率随数据积累而上升,2020 起才满覆盖。 行情/每日指标自 2015-01-05 才有,因此任何早于 2020-01 的 asof, 其「5 年窗口」都是被截短的 —— 这是**数据事实**,不是代码问题, 但必须能被看见。 """ db = _db_ready() from hdiv.core.config import load_config from hdiv.profile.pit import PitProfileService try: rows = db.read_sql( "SELECT symbol FROM stock WHERE symbol = '600036.SH'", cfg=load_config("datasource") ) if rows.empty: pytest.skip("无样本股票") except Exception as exc: pytest.skip(f"数据库不可用:{exc}") svc = PitProfileService(window_years=5) svc.prepare(["600036.SH"], date(2015, 1, 1), date(2026, 9, 30)) svc.configure({"dv_yield"}) vals = {} for a in ("2016-12-30", "2018-05-18", "2021-06-30"): s = svc.snapshot("600036.SH", date.fromisoformat(a)) if s is None: pytest.skip(f"{a} 无行情") vals[a] = s.coverage["dv_yield"] assert vals["2016-12-30"] < 0.5, f"2016 年 5 年窗口应严重不足:{vals}" assert vals["2018-05-18"] == pytest.approx(0.67, abs=0.02) assert vals["2021-06-30"] >= 0.999, f"2021 年应已满覆盖:{vals}" # n_obs 必须一并暴露 —— 它是判断「窗口是否被截断」的原始依据 s = svc.snapshot("600036.SH", date(2018, 5, 18)) assert s.n_obs["dv_yield"] == 817, "实测 2018-05-18 的 5 年窗口为 817 个观测" @pytest.mark.db def test_snapshot_ignores_input_row_order() -> None: """回归:画像的「当日值」必须由**日期**决定,不能受输入行序影响。 2026-10-04 回补 2005-2014 时,新行是**追加**进 ``daily_basic`` 的, 同一股票的物理行序变成「2015-2026 在前、2005-2014 在后」; 而当时 ``_load_daily_basic`` 没有 ``ORDER BY``、``_profile_one`` 用 ``iloc[-1]`` 取当日值 —— 于是格力电器 2018-05-18 的 PE(TTM) 被取成 2014 年的 8.51(真值 12.12)。这个测试把该不变量钉死: **随机打乱输入面板的行序,结果必须逐值不变。** """ db = _db_ready() from hdiv.core.config import load_config from hdiv.profile.pit import PitProfileService try: rows = db.read_sql( "SELECT symbol FROM stock WHERE symbol = '000651.SZ'", cfg=load_config("datasource"), ) if rows.empty: pytest.skip("无样本股票") except Exception as exc: pytest.skip(f"数据库不可用:{exc}") asof = date(2018, 5, 18) sym = "000651.SZ" def _snapshot(shuffle_seed: int | None) -> ProfileSnapshot: svc = PitProfileService(window_years=5) svc.prepare([sym], date(2004, 1, 1), date(2026, 9, 30)) svc.configure({"pe_ttm", "pb", "dv_yield"}) if shuffle_seed is not None: # 直接打乱内部面板:模拟「行序不是日期序」 svc._basics = svc._basics.sample(frac=1.0, random_state=shuffle_seed) svc._price = svc._price.sample(frac=1.0, random_state=shuffle_seed + 1) s = svc.snapshot(sym, asof) assert s is not None return s ref = _snapshot(None) for seed in (1, 7, 42): got = _snapshot(seed) for metric in ("pe_ttm", "pb", "dv_yield"): if metric in ref.values and metric in got.values: assert got.values[metric] == pytest.approx(ref.values[metric], rel=1e-9), ( f"打乱输入行序后 {metric} 变了:{got.values[metric]} != {ref.values[metric]}" ) # 顺带断言该日 PE(TTM) 就是 12.12 那个量级(防止排序修好后取到错窗口) assert ref.values["pe_ttm"] == pytest.approx(12.115, rel=1e-3), ref.values["pe_ttm"] @pytest.mark.db def test_daily_basic_loader_is_date_sorted() -> None: """``_load_daily_basic`` 必须返回按 (symbol, trade_date) 有序的帧。 下游把「最后一个观测」当作当日值 —— 有序性是**语义前提**,不是可选优化。 """ db = _db_ready() from hdiv.core.config import load_config from hdiv.profile.builder import ProfileBuilder try: rows = db.read_sql( "SELECT symbol FROM stock WHERE symbol IN ('000651.SZ','600036.SH') ORDER BY symbol", cfg=load_config("datasource"), ) if rows.empty: pytest.skip("无样本股票") syms = rows["symbol"].tolist() except Exception as exc: pytest.skip(f"数据库不可用:{exc}") df = ProfileBuilder.from_config()._load_daily_basic(syms, date(2005, 1, 1), date(2026, 9, 30)) assert not df.empty for sym, g in df.groupby("symbol", sort=False): assert g["trade_date"].is_monotonic_increasing, f"{sym} 的行序不是日期序" assert df["trade_date"].min().date() <= date(2005, 1, 10), "回补后应包含 2005 年数据"