feat(backend): Phase 2 研究引擎 — ResearchSpec / 因子 / 评估 / 低频回测 / 引擎抽象

- domain:ResearchSpec(universe/factors/selection/rebalance/costs 校验)+ 标准化 BacktestResult / FactorTestReport
- 因子引擎:注册表 + 元数据,内置 9 个行情因子(momentum/volatility/量比/乖离/反转),支持自定义注册;只用行情字段规避未来函数
- 评估:横截面 IC / RankIC(rank+pearson 免 scipy)/ ICIR / 分层收益
- 回测:TopK 等权低频,无未来函数记账(t 收盘成交、自 t+1 计收益),成本/涨跌停/停牌约束,未建模项显式写入 unimplemented(AGENT §24)
- 引擎抽象 QuantEngine + LocalEngine(pandas 默认实现);qlib_adapter 桥接占位 —— pyqlib 无 aarch64+cp312 wheel(ROADMAP 已备注)
- 真实链路冒烟:600519 2024 月度动量回测闭环产出标准结果
- 测试 60 passed / ruff clean
This commit is contained in:
Simon
2026-09-06 17:08:00 +08:00
parent 2da234220a
commit e9f59d3cf8
12 changed files with 1388 additions and 2 deletions
+102
View File
@@ -0,0 +1,102 @@
"""因子评估:截面 IC / RankIC / ICIR / 分层收益(AGENT.md §23)。
输入均为 面板(index=trade_date, columns=symbol):
- factor:因子值
- forward:未来 horizon 期收益(每行是「当日可见、未来实现」的收益,用于横截面相关)
任何消费侧必须保证 factor 行 t 只用 <= t 的信息,forward 是 t 之后的实现 ——
两者错位即未来函数,由数据构造方负责(本模块只做统计)。
"""
from __future__ import annotations
import math
import pandas as pd
from app.domain.entities.research import FactorTestReport, QuantileReturn
MIN_CROSS_SECTION = 5 # 少于该样本数的日期跳过(避免噪声 IC)
def cross_sectional_ic(
factor: pd.DataFrame, forward: pd.DataFrame, method: str = "pearson"
) -> pd.Series:
"""逐日横截面相关(pearson=IC;spearman=RankIC 用 rank+pearson 等价,免 scipy)。"""
rows: dict[pd.Timestamp, float] = {}
idx = factor.index.intersection(forward.index)
for dt in idx:
f = factor.loc[dt].dropna()
r = forward.loc[dt].reindex(f.index)
pair = pd.concat([f, r], axis=1).dropna()
if len(pair) < MIN_CROSS_SECTION:
continue
a, b = pair.iloc[:, 0], pair.iloc[:, 1]
if method == "spearman":
a, b = a.rank(), b.rank()
ic = a.corr(b)
if math.isfinite(ic):
rows[dt] = float(ic)
return pd.Series(rows, dtype=float).sort_index()
def _icir(series: pd.Series) -> float:
if len(series) < 2:
return 0.0
std = float(series.std(ddof=1))
if std == 0 or math.isnan(std):
return 0.0
return float(series.mean() / std * math.sqrt(len(series)))
def quantile_returns(factor: pd.DataFrame, forward: pd.DataFrame, quantiles: int = 5) -> pd.Series:
"""逐日按因子值升序分层,返回各层平均未来收益(跨日再平均)。"""
acc = {q: [] for q in range(quantiles)}
idx = factor.index.intersection(forward.index)
for dt in idx:
f = factor.loc[dt].dropna()
r = forward.loc[dt].reindex(f.index)
pair = pd.concat([f, r], axis=1).dropna()
if len(pair) < quantiles * 2:
continue
try:
labels = pd.qcut(pair.iloc[:, 0], quantiles, labels=False, duplicates="drop")
except ValueError:
continue
grouped = pair.iloc[:, 1].groupby(labels).mean()
for q, val in grouped.items():
acc[int(q)].append(float(val))
means = {q: (sum(v) / len(v) if v else float("nan")) for q, v in acc.items()}
return pd.Series(means)
def run_factor_test(
factor: pd.DataFrame,
forward: pd.DataFrame,
*,
factor_name: str = "",
quantiles: int = 5,
) -> FactorTestReport:
ic = cross_sectional_ic(factor, forward, "pearson")
rank_ic = cross_sectional_ic(factor, forward, "spearman")
q_ret = quantile_returns(factor, forward, quantiles)
spread: int | None = None
valid = [q for q in range(quantiles) if q in q_ret.index and not math.isnan(q_ret[q])]
if len(valid) >= 2 and q_ret[valid[-1]] > q_ret[valid[0]]:
spread = int(valid[-1]) # 高分层 > 低分层时报告层号
report = FactorTestReport(
factor_name=factor_name or "factor",
ic_mean=float(ic.mean()) if len(ic) else 0.0,
icir=_icir(ic),
rank_ic_mean=float(rank_ic.mean()) if len(rank_ic) else 0.0,
positive_ratio_pct=float((ic > 0).mean() * 100) if len(ic) else 0.0,
quantile_returns=[
QuantileReturn(quantile=int(q), return_pct=round(float(v) * 100, 4))
for q, v in sorted(q_ret.items())
],
spread_quantile=spread,
sample_days=len(ic),
)
return report