Files
qlib/backend/app/quant/evaluation.py
T
Simon e9f59d3cf8 feat(backend): Phase 2 研究引擎 — ResearchSpec / 因子 / 评估 / 低频回测 / 引擎抽象
- domain:ResearchSpec(universe/factors/selection/rebalance/costs 校验)+ 标准化 BacktestResult / FactorTestReport
- 因子引擎:注册表 + 元数据,内置 9 个行情因子(momentum/volatility/量比/乖离/反转),支持自定义注册;只用行情字段规避未来函数
- 评估:横截面 IC / RankIC(rank+pearson 免 scipy)/ ICIR / 分层收益
- 回测:TopK 等权低频,无未来函数记账(t 收盘成交、自 t+1 计收益),成本/涨跌停/停牌约束,未建模项显式写入 unimplemented(AGENT §24)
- 引擎抽象 QuantEngine + LocalEngine(pandas 默认实现);qlib_adapter 桥接占位 —— pyqlib 无 aarch64+cp312 wheel(ROADMAP 已备注)
- 真实链路冒烟:600519 2024 月度动量回测闭环产出标准结果
- 测试 60 passed / ruff clean
2026-09-06 17:08:00 +08:00

103 lines
3.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""因子评估:截面 IC / RankIC / ICIR / 分层收益(AGENT.md §23)。
输入均为 面板(index=trade_date, columns=symbol):
- factor:因子值
- forward:未来 horizon 期收益(每行是「当日可见、未来实现」的收益,用于横截面相关)
任何消费侧必须保证 factor 行 t 只用 <= t 的信息,forward 是 t 之后的实现 ——
两者错位即未来函数,由数据构造方负责(本模块只做统计)。
"""
from __future__ import annotations
import math
import pandas as pd
from app.domain.entities.research import FactorTestReport, QuantileReturn
MIN_CROSS_SECTION = 5 # 少于该样本数的日期跳过(避免噪声 IC)
def cross_sectional_ic(
factor: pd.DataFrame, forward: pd.DataFrame, method: str = "pearson"
) -> pd.Series:
"""逐日横截面相关(pearson=IC;spearman=RankIC 用 rank+pearson 等价,免 scipy)。"""
rows: dict[pd.Timestamp, float] = {}
idx = factor.index.intersection(forward.index)
for dt in idx:
f = factor.loc[dt].dropna()
r = forward.loc[dt].reindex(f.index)
pair = pd.concat([f, r], axis=1).dropna()
if len(pair) < MIN_CROSS_SECTION:
continue
a, b = pair.iloc[:, 0], pair.iloc[:, 1]
if method == "spearman":
a, b = a.rank(), b.rank()
ic = a.corr(b)
if math.isfinite(ic):
rows[dt] = float(ic)
return pd.Series(rows, dtype=float).sort_index()
def _icir(series: pd.Series) -> float:
if len(series) < 2:
return 0.0
std = float(series.std(ddof=1))
if std == 0 or math.isnan(std):
return 0.0
return float(series.mean() / std * math.sqrt(len(series)))
def quantile_returns(factor: pd.DataFrame, forward: pd.DataFrame, quantiles: int = 5) -> pd.Series:
"""逐日按因子值升序分层,返回各层平均未来收益(跨日再平均)。"""
acc = {q: [] for q in range(quantiles)}
idx = factor.index.intersection(forward.index)
for dt in idx:
f = factor.loc[dt].dropna()
r = forward.loc[dt].reindex(f.index)
pair = pd.concat([f, r], axis=1).dropna()
if len(pair) < quantiles * 2:
continue
try:
labels = pd.qcut(pair.iloc[:, 0], quantiles, labels=False, duplicates="drop")
except ValueError:
continue
grouped = pair.iloc[:, 1].groupby(labels).mean()
for q, val in grouped.items():
acc[int(q)].append(float(val))
means = {q: (sum(v) / len(v) if v else float("nan")) for q, v in acc.items()}
return pd.Series(means)
def run_factor_test(
factor: pd.DataFrame,
forward: pd.DataFrame,
*,
factor_name: str = "",
quantiles: int = 5,
) -> FactorTestReport:
ic = cross_sectional_ic(factor, forward, "pearson")
rank_ic = cross_sectional_ic(factor, forward, "spearman")
q_ret = quantile_returns(factor, forward, quantiles)
spread: int | None = None
valid = [q for q in range(quantiles) if q in q_ret.index and not math.isnan(q_ret[q])]
if len(valid) >= 2 and q_ret[valid[-1]] > q_ret[valid[0]]:
spread = int(valid[-1]) # 高分层 > 低分层时报告层号
report = FactorTestReport(
factor_name=factor_name or "factor",
ic_mean=float(ic.mean()) if len(ic) else 0.0,
icir=_icir(ic),
rank_ic_mean=float(rank_ic.mean()) if len(rank_ic) else 0.0,
positive_ratio_pct=float((ic > 0).mean() * 100) if len(ic) else 0.0,
quantile_returns=[
QuantileReturn(quantile=int(q), return_pct=round(float(v) * 100, 4))
for q, v in sorted(q_ret.items())
],
spread_quantile=spread,
sample_days=len(ic),
)
return report