refactor(quant): M7.2a Composite Engine 模块化(quant/composite.py)

- cross_sectional_zscore / composite_score / build_factor_panels 从 local_engine 迁入
  quant/composite.py;新增统一入口 build_score_panel(daily, factor_specs)
- local_engine re-export 保持旧引用兼容;selection/engine 的评分面板构建均指向
  composite —— 选股与回测的复合分实现收敛于一处
- 回归:quant/eval/research/selection 一致性/qlib 引擎测试全过;全量 pytest 通过
This commit is contained in:
Simon
2026-09-09 00:29:23 +08:00
parent 8f47b5b603
commit 273aee2772
4 changed files with 83 additions and 57 deletions
+73
View File
@@ -0,0 +1,73 @@
"""Composite Factor Engine(M7.2,ARCHITECTURE_v2 §13)。
把「多因子 → 加权复合分面板」独立成模块,供:
- 选股(SelectionEngine.run_score_selection)
- 回测(LocalEngine.run_backtest)
共用同一实现(v2 §25 一致性)。
method 先落地 fixed(截面 zscore × 方向 × 权重 求和);Rank/Z/IC 加权留接口。
"""
from __future__ import annotations
import math
import pandas as pd
from app.quant.factors import FactorDef, compute_factor
def cross_sectional_zscore(panel: pd.DataFrame) -> pd.DataFrame:
"""截面 z-score。
候选不足 2 只(如单股票池)时退化为 0:无比较基准,但保留为可候选值;
全列缺失才为 NaN(该日不可选股)。
"""
def _row_z(row: pd.Series) -> pd.Series:
valid = row.dropna()
if len(valid) == 0:
return pd.Series(float("nan"), index=row.index)
if len(valid) == 1:
return pd.Series(0.0, index=row.index)
mu, sd = valid.mean(), valid.std()
if sd == 0 or math.isnan(sd):
return pd.Series(0.0, index=row.index)
return (row - mu) / sd
return panel.apply(_row_z, axis=1)
def composite_score(panels: list[tuple[str, pd.DataFrame, float, str]]) -> pd.DataFrame:
"""按 (name, panel, weight, direction) 计算加权复合 zscore。
direction="lower_is_better" 的因子取负号后相加(统一为「得分高者优先」)。
"""
total = None
for _name, panel, weight, direction in panels:
z = cross_sectional_zscore(panel)
if direction == "lower_is_better":
z = -z
contribution = z * weight
total = contribution if total is None else total.add(contribution, fill_value=0)
assert total is not None
return total
def build_factor_panels(
daily: pd.DataFrame, factor_specs
) -> list[tuple[str, pd.DataFrame, float, str]]:
"""按 spec.factors 计算面板与权重(因子不存在即报错)。"""
panels: list[tuple[str, pd.DataFrame, float, str]] = []
for fs in factor_specs:
defn: FactorDef
defn, panel = compute_factor(fs.name, daily)
panels.append((fs.name, panel, fs.weight, defn.direction))
return panels
def build_score_panel(daily: pd.DataFrame, factor_specs) -> pd.DataFrame:
"""因子加权复合分面板(index=trade_date, columns=symbol)。
回测与选股共用的统一入口 —— 保证 v2 §25/§27 一致性。
"""
return composite_score(build_factor_panels(daily, factor_specs))
+5 -52
View File
@@ -26,8 +26,12 @@ from app.domain.entities.research import (
Trade,
YearlyReturn,
)
from app.quant.composite import ( # noqa: F401 —— re-export(模块化后旧引用仍可用)
build_factor_panels,
composite_score,
cross_sectional_zscore,
)
from app.quant.evaluation import run_factor_test
from app.quant.factors import FactorDef, compute_factor
TRADING_DAYS = 252
_DEFAULT_UNIMPLEMENTED = [
@@ -46,45 +50,6 @@ def _limit_up_ratio(symbol: str) -> float:
return 1.099
def cross_sectional_zscore(panel: pd.DataFrame) -> pd.DataFrame:
"""截面 z-score。
候选不足 2 只(如单股票池)时退化为 0:无比较基准,但保留为可候选值;
全列缺失才为 NaN(该日不可选股)。
"""
def _row_z(row: pd.Series) -> pd.Series:
valid = row.dropna()
if len(valid) == 0:
return pd.Series(float("nan"), index=row.index)
if len(valid) == 1:
return pd.Series(0.0, index=row.index)
mu, sd = valid.mean(), valid.std()
if sd == 0 or math.isnan(sd):
return pd.Series(0.0, index=row.index)
return (row - mu) / sd
return panel.apply(_row_z, axis=1)
def composite_score(
panels: list[tuple[str, pd.DataFrame, float, str]],
) -> pd.DataFrame:
"""按 (name, panel, weight, direction) 计算加权复合 zscore。
direction="lower_is_better" 的因子取负号后相加(统一为「得分高者优先」)。
"""
total = None
for _name, panel, weight, direction in panels:
z = cross_sectional_zscore(panel)
if direction == "lower_is_better":
z = -z
contribution = z * weight
total = contribution if total is None else total.add(contribution, fill_value=0)
assert total is not None
return total
def rebalance_dates(index: pd.Index, rebalance: str, start: date) -> list[pd.Timestamp]:
"""按频率取首个交易日(>= start)。"""
periods = index.to_period("M" if rebalance == "monthly" else "W")
@@ -303,18 +268,6 @@ class TopKBacktestRunner:
)
def build_factor_panels(
daily: pd.DataFrame, factor_specs
) -> list[tuple[str, pd.DataFrame, float, str]]:
"""按 spec.factors 计算面板与权重(因子不存在即报错)。"""
panels: list[tuple[str, pd.DataFrame, float, str]] = []
for fs in factor_specs:
defn: FactorDef
defn, panel = compute_factor(fs.name, daily)
panels.append((fs.name, panel, fs.weight, defn.direction))
return panels
def run_spec_factor_test(
daily: pd.DataFrame,
spec: ResearchSpec,
+2 -3
View File
@@ -21,8 +21,8 @@ from app.domain.entities.selection import (
SelectionResult,
SelectionStatistics,
)
from app.quant.composite import build_score_panel
from app.quant.factors import FactorError, compute_factor, get_factor
from app.quant.local_engine import build_factor_panels, composite_score
_UNIMPLEMENTED_DEFAULT = [
"exclude_suspended 依赖停牌数据,当前未建模(结果可能包含停牌股)",
@@ -35,8 +35,7 @@ def score_panel_for_factors(daily: pd.DataFrame, factor_specs) -> pd.DataFrame:
回测(LocalEngine)与选股(run_score_selection)共用同一构建 ——
保证 v2 §25/§27「历史回测与当前选股使用同一套引擎」的一致性。
"""
panels = build_factor_panels(daily, factor_specs) # 未知因子在此抛 FactorError
return composite_score(panels)
return build_score_panel(daily, factor_specs) # 未知因子在此抛 FactorError
def resolve_observation_date(daily: pd.DataFrame, as_of: date | None) -> pd.Timestamp | None:
+3 -2
View File
@@ -14,9 +14,10 @@ from app.domain.entities.research import (
SelectionSpec,
UniverseSpec,
)
from app.quant.composite import composite_score, cross_sectional_zscore
from app.quant.evaluation import run_factor_test
from app.quant.factors import compute_factor
from app.quant.local_engine import composite_score, cross_sectional_zscore, rebalance_dates
from app.quant.local_engine import rebalance_dates
from pydantic import ValidationError
from conftest_quant import synthetic_daily
@@ -115,7 +116,7 @@ class TestEvaluation:
class TestSingleStockDegradation:
def test_zscore_single_stock_keeps_candidate(self) -> None:
from app.quant.local_engine import cross_sectional_zscore
from app.quant.composite import cross_sectional_zscore
daily = synthetic_daily({"ONLY": 0.001}, n=80)
_d, panel = compute_factor("momentum_20", daily)