- evaluation.factor_correlation_report:多因子共同日期 ∩ 后逐日横截面 Spearman 相关 取均值 → FactorCorrelationReport(冗余剔除前置,v3 §12 Correlation→Redundancy) - ResearchService.run_factor_correlation + POST /api/factor-correlations (universe/factors/period;与其它研究同装配口径) - tests/test_factor_correlation.py:矩阵对角=1/近线性±相关符号/对称/无共同日期补零、 API 冒烟;全量 pytest 通过
157 lines
5.7 KiB
Python
157 lines
5.7 KiB
Python
"""因子评估:截面 IC / RankIC / ICIR / 分层收益(AGENT.md §23)。
|
||
|
||
输入均为 面板(index=trade_date, columns=symbol):
|
||
- factor:因子值
|
||
- forward:未来 horizon 期收益(每行是「当日可见、未来实现」的收益,用于横截面相关)
|
||
|
||
任何消费侧必须保证 factor 行 t 只用 <= t 的信息,forward 是 t 之后的实现 ——
|
||
两者错位即未来函数,由数据构造方负责(本模块只做统计)。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import math
|
||
|
||
import pandas as pd
|
||
|
||
from app.domain.entities.research import FactorCorrelationReport, FactorTestReport, QuantileReturn
|
||
|
||
MIN_CROSS_SECTION = 5 # 少于该样本数的日期跳过(避免噪声 IC)
|
||
|
||
|
||
def cross_sectional_ic(
|
||
factor: pd.DataFrame, forward: pd.DataFrame, method: str = "pearson"
|
||
) -> pd.Series:
|
||
"""逐日横截面相关(pearson=IC;spearman=RankIC 用 rank+pearson 等价,免 scipy)。"""
|
||
rows: dict[pd.Timestamp, float] = {}
|
||
idx = factor.index.intersection(forward.index)
|
||
for dt in idx:
|
||
f = factor.loc[dt].dropna()
|
||
r = forward.loc[dt].reindex(f.index)
|
||
pair = pd.concat([f, r], axis=1).dropna()
|
||
if len(pair) < MIN_CROSS_SECTION:
|
||
continue
|
||
a, b = pair.iloc[:, 0], pair.iloc[:, 1]
|
||
if method == "spearman":
|
||
a, b = a.rank(), b.rank()
|
||
ic = a.corr(b)
|
||
if math.isfinite(ic):
|
||
rows[dt] = float(ic)
|
||
return pd.Series(rows, dtype=float).sort_index()
|
||
|
||
|
||
def _icir(series: pd.Series) -> float:
|
||
if len(series) < 2:
|
||
return 0.0
|
||
std = float(series.std(ddof=1))
|
||
if std == 0 or math.isnan(std):
|
||
return 0.0
|
||
return float(series.mean() / std * math.sqrt(len(series)))
|
||
|
||
|
||
def quantile_returns(factor: pd.DataFrame, forward: pd.DataFrame, quantiles: int = 5) -> pd.Series:
|
||
"""逐日按因子值升序分层,返回各层平均未来收益(跨日再平均)。"""
|
||
acc = {q: [] for q in range(quantiles)}
|
||
idx = factor.index.intersection(forward.index)
|
||
for dt in idx:
|
||
f = factor.loc[dt].dropna()
|
||
r = forward.loc[dt].reindex(f.index)
|
||
pair = pd.concat([f, r], axis=1).dropna()
|
||
if len(pair) < quantiles * 2:
|
||
continue
|
||
try:
|
||
labels = pd.qcut(pair.iloc[:, 0], quantiles, labels=False, duplicates="drop")
|
||
except ValueError:
|
||
continue
|
||
grouped = pair.iloc[:, 1].groupby(labels).mean()
|
||
for q, val in grouped.items():
|
||
acc[int(q)].append(float(val))
|
||
means = {q: (sum(v) / len(v) if v else float("nan")) for q, v in acc.items()}
|
||
return pd.Series(means)
|
||
|
||
|
||
def _cross_section_corr(pa: pd.Series, pb: pd.Series, min_n: int) -> float | None:
|
||
"""两股票列在某一日期截面值的 Spearman 相关(样本不足 → None)。"""
|
||
df = pd.concat([pa, pb], axis=1).dropna()
|
||
if len(df) < min_n or df.iloc[:, 0].nunique() < 2:
|
||
return None
|
||
return df.iloc[:, 0].corr(df.iloc[:, 1], method="spearman")
|
||
|
||
|
||
def factor_correlation_report(
|
||
panels: dict[str, pd.DataFrame],
|
||
*,
|
||
min_symbols: int = MIN_CROSS_SECTION,
|
||
max_days: int = 10000,
|
||
) -> FactorCorrelationReport:
|
||
"""两两因子相关:共同日期 ∩ 后,逐日横截面 Spearman 相关取均值。
|
||
|
||
用于冗余剔除与因子池管理(v3 §12 流程:Correlation → Redundancy Removal)。
|
||
"""
|
||
names = list(panels)
|
||
common = None
|
||
for panel in panels.values():
|
||
idx = set(panel.index)
|
||
common = idx if common is None else (common & idx)
|
||
if not common:
|
||
empty = {a: {b: (1.0 if a == b else 0.0) for b in names} for a in names}
|
||
return FactorCorrelationReport(
|
||
factors=names, corr_matrix=empty,
|
||
sample_days=0, sample_min_symbols=min_symbols,
|
||
)
|
||
days = sorted(common)[:max_days]
|
||
corr_matrix: dict[str, dict[str, float]] = {}
|
||
for a in names:
|
||
corr_matrix[a] = {}
|
||
for b in names:
|
||
if a == b:
|
||
corr_matrix[a][b] = 1.0
|
||
continue
|
||
acc: list[float] = []
|
||
for day in days:
|
||
if day not in panels[a].index or day not in panels[b].index:
|
||
continue
|
||
c = _cross_section_corr(panels[a].loc[day], panels[b].loc[day], min_symbols)
|
||
if c is not None:
|
||
acc.append(c)
|
||
corr_matrix[a][b] = round(float(pd.Series(acc).mean()), 4) if acc else 0.0
|
||
return FactorCorrelationReport(
|
||
factors=names,
|
||
corr_matrix=corr_matrix,
|
||
sample_days=len(days),
|
||
sample_min_symbols=min_symbols,
|
||
config_snapshot={"min_symbols": min_symbols},
|
||
)
|
||
|
||
|
||
def run_factor_test(
|
||
factor: pd.DataFrame,
|
||
forward: pd.DataFrame,
|
||
*,
|
||
factor_name: str = "",
|
||
quantiles: int = 5,
|
||
) -> FactorTestReport:
|
||
ic = cross_sectional_ic(factor, forward, "pearson")
|
||
rank_ic = cross_sectional_ic(factor, forward, "spearman")
|
||
q_ret = quantile_returns(factor, forward, quantiles)
|
||
|
||
spread: int | None = None
|
||
valid = [q for q in range(quantiles) if q in q_ret.index and not math.isnan(q_ret[q])]
|
||
if len(valid) >= 2 and q_ret[valid[-1]] > q_ret[valid[0]]:
|
||
spread = int(valid[-1]) # 高分层 > 低分层时报告层号
|
||
|
||
report = FactorTestReport(
|
||
factor_name=factor_name or "factor",
|
||
ic_mean=float(ic.mean()) if len(ic) else 0.0,
|
||
icir=_icir(ic),
|
||
rank_ic_mean=float(rank_ic.mean()) if len(rank_ic) else 0.0,
|
||
positive_ratio_pct=float((ic > 0).mean() * 100) if len(ic) else 0.0,
|
||
quantile_returns=[
|
||
QuantileReturn(quantile=int(q), return_pct=round(float(v) * 100, 4))
|
||
for q, v in sorted(q_ret.items())
|
||
],
|
||
spread_quantile=spread,
|
||
sample_days=len(ic),
|
||
)
|
||
return report
|