feat(factor): C1 因子相关性分析(横截面 Spearman 矩阵 + API)

- evaluation.factor_correlation_report:多因子共同日期 ∩ 后逐日横截面 Spearman 相关
  取均值 → FactorCorrelationReport(冗余剔除前置,v3 §12 Correlation→Redundancy)
- ResearchService.run_factor_correlation + POST /api/factor-correlations
  (universe/factors/period;与其它研究同装配口径)
- tests/test_factor_correlation.py:矩阵对角=1/近线性±相关符号/对称/无共同日期补零、
  API 冒烟;全量 pytest 通过
This commit is contained in:
Simon
2026-09-09 07:31:49 +08:00
parent 93e32f4e63
commit 0d05bfd187
5 changed files with 227 additions and 1 deletions
+55 -1
View File
@@ -14,7 +14,7 @@ import math
import pandas as pd
from app.domain.entities.research import FactorTestReport, QuantileReturn
from app.domain.entities.research import FactorCorrelationReport, FactorTestReport, QuantileReturn
MIN_CROSS_SECTION = 5 # 少于该样本数的日期跳过(避免噪声 IC)
@@ -70,6 +70,60 @@ def quantile_returns(factor: pd.DataFrame, forward: pd.DataFrame, quantiles: int
return pd.Series(means)
def _cross_section_corr(pa: pd.Series, pb: pd.Series, min_n: int) -> float | None:
"""两股票列在某一日期截面值的 Spearman 相关(样本不足 → None)。"""
df = pd.concat([pa, pb], axis=1).dropna()
if len(df) < min_n or df.iloc[:, 0].nunique() < 2:
return None
return df.iloc[:, 0].corr(df.iloc[:, 1], method="spearman")
def factor_correlation_report(
panels: dict[str, pd.DataFrame],
*,
min_symbols: int = MIN_CROSS_SECTION,
max_days: int = 10000,
) -> FactorCorrelationReport:
"""两两因子相关:共同日期 ∩ 后,逐日横截面 Spearman 相关取均值。
用于冗余剔除与因子池管理(v3 §12 流程:Correlation → Redundancy Removal)。
"""
names = list(panels)
common = None
for panel in panels.values():
idx = set(panel.index)
common = idx if common is None else (common & idx)
if not common:
empty = {a: {b: (1.0 if a == b else 0.0) for b in names} for a in names}
return FactorCorrelationReport(
factors=names, corr_matrix=empty,
sample_days=0, sample_min_symbols=min_symbols,
)
days = sorted(common)[:max_days]
corr_matrix: dict[str, dict[str, float]] = {}
for a in names:
corr_matrix[a] = {}
for b in names:
if a == b:
corr_matrix[a][b] = 1.0
continue
acc: list[float] = []
for day in days:
if day not in panels[a].index or day not in panels[b].index:
continue
c = _cross_section_corr(panels[a].loc[day], panels[b].loc[day], min_symbols)
if c is not None:
acc.append(c)
corr_matrix[a][b] = round(float(pd.Series(acc).mean()), 4) if acc else 0.0
return FactorCorrelationReport(
factors=names,
corr_matrix=corr_matrix,
sample_days=len(days),
sample_min_symbols=min_symbols,
config_snapshot={"min_symbols": min_symbols},
)
def run_factor_test(
factor: pd.DataFrame,
forward: pd.DataFrame,
+9
View File
@@ -17,6 +17,7 @@ import pandas as pd
from app.domain.entities.research import (
BacktestResult,
FactorCorrelationReport,
FactorTestReport,
ResearchSpec,
)
@@ -24,7 +25,9 @@ from app.domain.repositories.market import (
DailyBarRepository,
StockRepository,
)
from app.quant.composite import build_factor_panels
from app.quant.engine import QuantEngine
from app.quant.evaluation import factor_correlation_report
from app.quant.universe import filter_stocks, resolve_members # noqa: F401 —— 范围过滤
# 流式路径每攒多少行落一个 DataFrame 分片(控制 concat 峰值)
@@ -129,6 +132,12 @@ class ResearchService:
daily = self._load_daily(spec)
return self._engine.run_backtest(daily, spec)
def run_factor_correlation(self, spec: ResearchSpec) -> FactorCorrelationReport:
"""多因子两两相关(v3 §12):同 universe/period 装配 → 横截面相关矩阵。"""
daily = self._load_daily(spec)
panels = {fs.name: build_factor_panels(daily, [fs])[0][1] for fs in spec.factors}
return factor_correlation_report(panels)
# ---- 数据装配 ----
def _load_daily(self, spec: ResearchSpec) -> pd.DataFrame: