Files
qlib/backend/app/quant/evaluation.py
T
Simon 0d05bfd187 feat(factor): C1 因子相关性分析(横截面 Spearman 矩阵 + API)
- evaluation.factor_correlation_report:多因子共同日期 ∩ 后逐日横截面 Spearman 相关
  取均值 → FactorCorrelationReport(冗余剔除前置,v3 §12 Correlation→Redundancy)
- ResearchService.run_factor_correlation + POST /api/factor-correlations
  (universe/factors/period;与其它研究同装配口径)
- tests/test_factor_correlation.py:矩阵对角=1/近线性±相关符号/对称/无共同日期补零、
  API 冒烟;全量 pytest 通过
2026-09-09 07:31:49 +08:00

157 lines
5.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""因子评估:截面 IC / RankIC / ICIR / 分层收益(AGENT.md §23)。
输入均为 面板(index=trade_date, columns=symbol):
- factor:因子值
- forward:未来 horizon 期收益(每行是「当日可见、未来实现」的收益,用于横截面相关)
任何消费侧必须保证 factor 行 t 只用 <= t 的信息,forward 是 t 之后的实现 ——
两者错位即未来函数,由数据构造方负责(本模块只做统计)。
"""
from __future__ import annotations
import math
import pandas as pd
from app.domain.entities.research import FactorCorrelationReport, FactorTestReport, QuantileReturn
MIN_CROSS_SECTION = 5 # 少于该样本数的日期跳过(避免噪声 IC)
def cross_sectional_ic(
factor: pd.DataFrame, forward: pd.DataFrame, method: str = "pearson"
) -> pd.Series:
"""逐日横截面相关(pearson=IC;spearman=RankIC 用 rank+pearson 等价,免 scipy)。"""
rows: dict[pd.Timestamp, float] = {}
idx = factor.index.intersection(forward.index)
for dt in idx:
f = factor.loc[dt].dropna()
r = forward.loc[dt].reindex(f.index)
pair = pd.concat([f, r], axis=1).dropna()
if len(pair) < MIN_CROSS_SECTION:
continue
a, b = pair.iloc[:, 0], pair.iloc[:, 1]
if method == "spearman":
a, b = a.rank(), b.rank()
ic = a.corr(b)
if math.isfinite(ic):
rows[dt] = float(ic)
return pd.Series(rows, dtype=float).sort_index()
def _icir(series: pd.Series) -> float:
if len(series) < 2:
return 0.0
std = float(series.std(ddof=1))
if std == 0 or math.isnan(std):
return 0.0
return float(series.mean() / std * math.sqrt(len(series)))
def quantile_returns(factor: pd.DataFrame, forward: pd.DataFrame, quantiles: int = 5) -> pd.Series:
"""逐日按因子值升序分层,返回各层平均未来收益(跨日再平均)。"""
acc = {q: [] for q in range(quantiles)}
idx = factor.index.intersection(forward.index)
for dt in idx:
f = factor.loc[dt].dropna()
r = forward.loc[dt].reindex(f.index)
pair = pd.concat([f, r], axis=1).dropna()
if len(pair) < quantiles * 2:
continue
try:
labels = pd.qcut(pair.iloc[:, 0], quantiles, labels=False, duplicates="drop")
except ValueError:
continue
grouped = pair.iloc[:, 1].groupby(labels).mean()
for q, val in grouped.items():
acc[int(q)].append(float(val))
means = {q: (sum(v) / len(v) if v else float("nan")) for q, v in acc.items()}
return pd.Series(means)
def _cross_section_corr(pa: pd.Series, pb: pd.Series, min_n: int) -> float | None:
"""两股票列在某一日期截面值的 Spearman 相关(样本不足 → None)。"""
df = pd.concat([pa, pb], axis=1).dropna()
if len(df) < min_n or df.iloc[:, 0].nunique() < 2:
return None
return df.iloc[:, 0].corr(df.iloc[:, 1], method="spearman")
def factor_correlation_report(
panels: dict[str, pd.DataFrame],
*,
min_symbols: int = MIN_CROSS_SECTION,
max_days: int = 10000,
) -> FactorCorrelationReport:
"""两两因子相关:共同日期 ∩ 后,逐日横截面 Spearman 相关取均值。
用于冗余剔除与因子池管理(v3 §12 流程:Correlation → Redundancy Removal)。
"""
names = list(panels)
common = None
for panel in panels.values():
idx = set(panel.index)
common = idx if common is None else (common & idx)
if not common:
empty = {a: {b: (1.0 if a == b else 0.0) for b in names} for a in names}
return FactorCorrelationReport(
factors=names, corr_matrix=empty,
sample_days=0, sample_min_symbols=min_symbols,
)
days = sorted(common)[:max_days]
corr_matrix: dict[str, dict[str, float]] = {}
for a in names:
corr_matrix[a] = {}
for b in names:
if a == b:
corr_matrix[a][b] = 1.0
continue
acc: list[float] = []
for day in days:
if day not in panels[a].index or day not in panels[b].index:
continue
c = _cross_section_corr(panels[a].loc[day], panels[b].loc[day], min_symbols)
if c is not None:
acc.append(c)
corr_matrix[a][b] = round(float(pd.Series(acc).mean()), 4) if acc else 0.0
return FactorCorrelationReport(
factors=names,
corr_matrix=corr_matrix,
sample_days=len(days),
sample_min_symbols=min_symbols,
config_snapshot={"min_symbols": min_symbols},
)
def run_factor_test(
factor: pd.DataFrame,
forward: pd.DataFrame,
*,
factor_name: str = "",
quantiles: int = 5,
) -> FactorTestReport:
ic = cross_sectional_ic(factor, forward, "pearson")
rank_ic = cross_sectional_ic(factor, forward, "spearman")
q_ret = quantile_returns(factor, forward, quantiles)
spread: int | None = None
valid = [q for q in range(quantiles) if q in q_ret.index and not math.isnan(q_ret[q])]
if len(valid) >= 2 and q_ret[valid[-1]] > q_ret[valid[0]]:
spread = int(valid[-1]) # 高分层 > 低分层时报告层号
report = FactorTestReport(
factor_name=factor_name or "factor",
ic_mean=float(ic.mean()) if len(ic) else 0.0,
icir=_icir(ic),
rank_ic_mean=float(rank_ic.mean()) if len(rank_ic) else 0.0,
positive_ratio_pct=float((ic > 0).mean() * 100) if len(ic) else 0.0,
quantile_returns=[
QuantileReturn(quantile=int(q), return_pct=round(float(v) * 100, 4))
for q, v in sorted(q_ret.items())
],
spread_quantile=spread,
sample_days=len(ic),
)
return report