"""因子评估:截面 IC / RankIC / ICIR / 分层收益(AGENT.md §23)。 输入均为 面板(index=trade_date, columns=symbol): - factor:因子值 - forward:未来 horizon 期收益(每行是「当日可见、未来实现」的收益,用于横截面相关) 任何消费侧必须保证 factor 行 t 只用 <= t 的信息,forward 是 t 之后的实现 —— 两者错位即未来函数,由数据构造方负责(本模块只做统计)。 """ from __future__ import annotations import math import pandas as pd from app.domain.entities.research import FactorCorrelationReport, FactorTestReport, QuantileReturn MIN_CROSS_SECTION = 5 # 少于该样本数的日期跳过(避免噪声 IC) def cross_sectional_ic( factor: pd.DataFrame, forward: pd.DataFrame, method: str = "pearson" ) -> pd.Series: """逐日横截面相关(pearson=IC;spearman=RankIC 用 rank+pearson 等价,免 scipy)。""" rows: dict[pd.Timestamp, float] = {} idx = factor.index.intersection(forward.index) for dt in idx: f = factor.loc[dt].dropna() r = forward.loc[dt].reindex(f.index) pair = pd.concat([f, r], axis=1).dropna() if len(pair) < MIN_CROSS_SECTION: continue a, b = pair.iloc[:, 0], pair.iloc[:, 1] if method == "spearman": a, b = a.rank(), b.rank() ic = a.corr(b) if math.isfinite(ic): rows[dt] = float(ic) return pd.Series(rows, dtype=float).sort_index() def _icir(series: pd.Series) -> float: if len(series) < 2: return 0.0 std = float(series.std(ddof=1)) if std == 0 or math.isnan(std): return 0.0 return float(series.mean() / std * math.sqrt(len(series))) def quantile_returns(factor: pd.DataFrame, forward: pd.DataFrame, quantiles: int = 5) -> pd.Series: """逐日按因子值升序分层,返回各层平均未来收益(跨日再平均)。""" acc = {q: [] for q in range(quantiles)} idx = factor.index.intersection(forward.index) for dt in idx: f = factor.loc[dt].dropna() r = forward.loc[dt].reindex(f.index) pair = pd.concat([f, r], axis=1).dropna() if len(pair) < quantiles * 2: continue try: labels = pd.qcut(pair.iloc[:, 0], quantiles, labels=False, duplicates="drop") except ValueError: continue grouped = pair.iloc[:, 1].groupby(labels).mean() for q, val in grouped.items(): acc[int(q)].append(float(val)) means = {q: (sum(v) / len(v) if v else float("nan")) for q, v in acc.items()} return pd.Series(means) def _cross_section_corr(pa: pd.Series, pb: pd.Series, min_n: int) -> float | None: """两股票列在某一日期截面值的 Spearman 相关(样本不足 → None)。""" df = pd.concat([pa, pb], axis=1).dropna() if len(df) < min_n or df.iloc[:, 0].nunique() < 2: return None return df.iloc[:, 0].corr(df.iloc[:, 1], method="spearman") def factor_correlation_report( panels: dict[str, pd.DataFrame], *, min_symbols: int = MIN_CROSS_SECTION, max_days: int = 10000, ) -> FactorCorrelationReport: """两两因子相关:共同日期 ∩ 后,逐日横截面 Spearman 相关取均值。 用于冗余剔除与因子池管理(v3 §12 流程:Correlation → Redundancy Removal)。 """ names = list(panels) common = None for panel in panels.values(): idx = set(panel.index) common = idx if common is None else (common & idx) if not common: empty = {a: {b: (1.0 if a == b else 0.0) for b in names} for a in names} return FactorCorrelationReport( factors=names, corr_matrix=empty, sample_days=0, sample_min_symbols=min_symbols, ) days = sorted(common)[:max_days] corr_matrix: dict[str, dict[str, float]] = {} for a in names: corr_matrix[a] = {} for b in names: if a == b: corr_matrix[a][b] = 1.0 continue acc: list[float] = [] for day in days: if day not in panels[a].index or day not in panels[b].index: continue c = _cross_section_corr(panels[a].loc[day], panels[b].loc[day], min_symbols) if c is not None: acc.append(c) corr_matrix[a][b] = round(float(pd.Series(acc).mean()), 4) if acc else 0.0 return FactorCorrelationReport( factors=names, corr_matrix=corr_matrix, sample_days=len(days), sample_min_symbols=min_symbols, config_snapshot={"min_symbols": min_symbols}, ) def run_factor_test( factor: pd.DataFrame, forward: pd.DataFrame, *, factor_name: str = "", quantiles: int = 5, ) -> FactorTestReport: ic = cross_sectional_ic(factor, forward, "pearson") rank_ic = cross_sectional_ic(factor, forward, "spearman") q_ret = quantile_returns(factor, forward, quantiles) spread: int | None = None valid = [q for q in range(quantiles) if q in q_ret.index and not math.isnan(q_ret[q])] if len(valid) >= 2 and q_ret[valid[-1]] > q_ret[valid[0]]: spread = int(valid[-1]) # 高分层 > 低分层时报告层号 report = FactorTestReport( factor_name=factor_name or "factor", ic_mean=float(ic.mean()) if len(ic) else 0.0, icir=_icir(ic), rank_ic_mean=float(rank_ic.mean()) if len(rank_ic) else 0.0, positive_ratio_pct=float((ic > 0).mean() * 100) if len(ic) else 0.0, quantile_returns=[ QuantileReturn(quantile=int(q), return_pct=round(float(v) * 100, 4)) for q, v in sorted(q_ret.items()) ], spread_quantile=spread, sample_days=len(ic), ) return report