refactor(quant): M7.2a Composite Engine 模块化(quant/composite.py)
- cross_sectional_zscore / composite_score / build_factor_panels 从 local_engine 迁入 quant/composite.py;新增统一入口 build_score_panel(daily, factor_specs) - local_engine re-export 保持旧引用兼容;selection/engine 的评分面板构建均指向 composite —— 选股与回测的复合分实现收敛于一处 - 回归:quant/eval/research/selection 一致性/qlib 引擎测试全过;全量 pytest 通过
This commit is contained in:
@@ -0,0 +1,73 @@
|
|||||||
|
"""Composite Factor Engine(M7.2,ARCHITECTURE_v2 §13)。
|
||||||
|
|
||||||
|
把「多因子 → 加权复合分面板」独立成模块,供:
|
||||||
|
- 选股(SelectionEngine.run_score_selection)
|
||||||
|
- 回测(LocalEngine.run_backtest)
|
||||||
|
共用同一实现(v2 §25 一致性)。
|
||||||
|
method 先落地 fixed(截面 zscore × 方向 × 权重 求和);Rank/Z/IC 加权留接口。
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
from app.quant.factors import FactorDef, compute_factor
|
||||||
|
|
||||||
|
|
||||||
|
def cross_sectional_zscore(panel: pd.DataFrame) -> pd.DataFrame:
|
||||||
|
"""截面 z-score。
|
||||||
|
|
||||||
|
候选不足 2 只(如单股票池)时退化为 0:无比较基准,但保留为可候选值;
|
||||||
|
全列缺失才为 NaN(该日不可选股)。
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _row_z(row: pd.Series) -> pd.Series:
|
||||||
|
valid = row.dropna()
|
||||||
|
if len(valid) == 0:
|
||||||
|
return pd.Series(float("nan"), index=row.index)
|
||||||
|
if len(valid) == 1:
|
||||||
|
return pd.Series(0.0, index=row.index)
|
||||||
|
mu, sd = valid.mean(), valid.std()
|
||||||
|
if sd == 0 or math.isnan(sd):
|
||||||
|
return pd.Series(0.0, index=row.index)
|
||||||
|
return (row - mu) / sd
|
||||||
|
|
||||||
|
return panel.apply(_row_z, axis=1)
|
||||||
|
|
||||||
|
|
||||||
|
def composite_score(panels: list[tuple[str, pd.DataFrame, float, str]]) -> pd.DataFrame:
|
||||||
|
"""按 (name, panel, weight, direction) 计算加权复合 zscore。
|
||||||
|
|
||||||
|
direction="lower_is_better" 的因子取负号后相加(统一为「得分高者优先」)。
|
||||||
|
"""
|
||||||
|
total = None
|
||||||
|
for _name, panel, weight, direction in panels:
|
||||||
|
z = cross_sectional_zscore(panel)
|
||||||
|
if direction == "lower_is_better":
|
||||||
|
z = -z
|
||||||
|
contribution = z * weight
|
||||||
|
total = contribution if total is None else total.add(contribution, fill_value=0)
|
||||||
|
assert total is not None
|
||||||
|
return total
|
||||||
|
|
||||||
|
|
||||||
|
def build_factor_panels(
|
||||||
|
daily: pd.DataFrame, factor_specs
|
||||||
|
) -> list[tuple[str, pd.DataFrame, float, str]]:
|
||||||
|
"""按 spec.factors 计算面板与权重(因子不存在即报错)。"""
|
||||||
|
panels: list[tuple[str, pd.DataFrame, float, str]] = []
|
||||||
|
for fs in factor_specs:
|
||||||
|
defn: FactorDef
|
||||||
|
defn, panel = compute_factor(fs.name, daily)
|
||||||
|
panels.append((fs.name, panel, fs.weight, defn.direction))
|
||||||
|
return panels
|
||||||
|
|
||||||
|
|
||||||
|
def build_score_panel(daily: pd.DataFrame, factor_specs) -> pd.DataFrame:
|
||||||
|
"""因子加权复合分面板(index=trade_date, columns=symbol)。
|
||||||
|
|
||||||
|
回测与选股共用的统一入口 —— 保证 v2 §25/§27 一致性。
|
||||||
|
"""
|
||||||
|
return composite_score(build_factor_panels(daily, factor_specs))
|
||||||
@@ -26,8 +26,12 @@ from app.domain.entities.research import (
|
|||||||
Trade,
|
Trade,
|
||||||
YearlyReturn,
|
YearlyReturn,
|
||||||
)
|
)
|
||||||
|
from app.quant.composite import ( # noqa: F401 —— re-export(模块化后旧引用仍可用)
|
||||||
|
build_factor_panels,
|
||||||
|
composite_score,
|
||||||
|
cross_sectional_zscore,
|
||||||
|
)
|
||||||
from app.quant.evaluation import run_factor_test
|
from app.quant.evaluation import run_factor_test
|
||||||
from app.quant.factors import FactorDef, compute_factor
|
|
||||||
|
|
||||||
TRADING_DAYS = 252
|
TRADING_DAYS = 252
|
||||||
_DEFAULT_UNIMPLEMENTED = [
|
_DEFAULT_UNIMPLEMENTED = [
|
||||||
@@ -46,45 +50,6 @@ def _limit_up_ratio(symbol: str) -> float:
|
|||||||
return 1.099
|
return 1.099
|
||||||
|
|
||||||
|
|
||||||
def cross_sectional_zscore(panel: pd.DataFrame) -> pd.DataFrame:
|
|
||||||
"""截面 z-score。
|
|
||||||
|
|
||||||
候选不足 2 只(如单股票池)时退化为 0:无比较基准,但保留为可候选值;
|
|
||||||
全列缺失才为 NaN(该日不可选股)。
|
|
||||||
"""
|
|
||||||
|
|
||||||
def _row_z(row: pd.Series) -> pd.Series:
|
|
||||||
valid = row.dropna()
|
|
||||||
if len(valid) == 0:
|
|
||||||
return pd.Series(float("nan"), index=row.index)
|
|
||||||
if len(valid) == 1:
|
|
||||||
return pd.Series(0.0, index=row.index)
|
|
||||||
mu, sd = valid.mean(), valid.std()
|
|
||||||
if sd == 0 or math.isnan(sd):
|
|
||||||
return pd.Series(0.0, index=row.index)
|
|
||||||
return (row - mu) / sd
|
|
||||||
|
|
||||||
return panel.apply(_row_z, axis=1)
|
|
||||||
|
|
||||||
|
|
||||||
def composite_score(
|
|
||||||
panels: list[tuple[str, pd.DataFrame, float, str]],
|
|
||||||
) -> pd.DataFrame:
|
|
||||||
"""按 (name, panel, weight, direction) 计算加权复合 zscore。
|
|
||||||
|
|
||||||
direction="lower_is_better" 的因子取负号后相加(统一为「得分高者优先」)。
|
|
||||||
"""
|
|
||||||
total = None
|
|
||||||
for _name, panel, weight, direction in panels:
|
|
||||||
z = cross_sectional_zscore(panel)
|
|
||||||
if direction == "lower_is_better":
|
|
||||||
z = -z
|
|
||||||
contribution = z * weight
|
|
||||||
total = contribution if total is None else total.add(contribution, fill_value=0)
|
|
||||||
assert total is not None
|
|
||||||
return total
|
|
||||||
|
|
||||||
|
|
||||||
def rebalance_dates(index: pd.Index, rebalance: str, start: date) -> list[pd.Timestamp]:
|
def rebalance_dates(index: pd.Index, rebalance: str, start: date) -> list[pd.Timestamp]:
|
||||||
"""按频率取首个交易日(>= start)。"""
|
"""按频率取首个交易日(>= start)。"""
|
||||||
periods = index.to_period("M" if rebalance == "monthly" else "W")
|
periods = index.to_period("M" if rebalance == "monthly" else "W")
|
||||||
@@ -303,18 +268,6 @@ class TopKBacktestRunner:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def build_factor_panels(
|
|
||||||
daily: pd.DataFrame, factor_specs
|
|
||||||
) -> list[tuple[str, pd.DataFrame, float, str]]:
|
|
||||||
"""按 spec.factors 计算面板与权重(因子不存在即报错)。"""
|
|
||||||
panels: list[tuple[str, pd.DataFrame, float, str]] = []
|
|
||||||
for fs in factor_specs:
|
|
||||||
defn: FactorDef
|
|
||||||
defn, panel = compute_factor(fs.name, daily)
|
|
||||||
panels.append((fs.name, panel, fs.weight, defn.direction))
|
|
||||||
return panels
|
|
||||||
|
|
||||||
|
|
||||||
def run_spec_factor_test(
|
def run_spec_factor_test(
|
||||||
daily: pd.DataFrame,
|
daily: pd.DataFrame,
|
||||||
spec: ResearchSpec,
|
spec: ResearchSpec,
|
||||||
|
|||||||
@@ -21,8 +21,8 @@ from app.domain.entities.selection import (
|
|||||||
SelectionResult,
|
SelectionResult,
|
||||||
SelectionStatistics,
|
SelectionStatistics,
|
||||||
)
|
)
|
||||||
|
from app.quant.composite import build_score_panel
|
||||||
from app.quant.factors import FactorError, compute_factor, get_factor
|
from app.quant.factors import FactorError, compute_factor, get_factor
|
||||||
from app.quant.local_engine import build_factor_panels, composite_score
|
|
||||||
|
|
||||||
_UNIMPLEMENTED_DEFAULT = [
|
_UNIMPLEMENTED_DEFAULT = [
|
||||||
"exclude_suspended 依赖停牌数据,当前未建模(结果可能包含停牌股)",
|
"exclude_suspended 依赖停牌数据,当前未建模(结果可能包含停牌股)",
|
||||||
@@ -35,8 +35,7 @@ def score_panel_for_factors(daily: pd.DataFrame, factor_specs) -> pd.DataFrame:
|
|||||||
回测(LocalEngine)与选股(run_score_selection)共用同一构建 ——
|
回测(LocalEngine)与选股(run_score_selection)共用同一构建 ——
|
||||||
保证 v2 §25/§27「历史回测与当前选股使用同一套引擎」的一致性。
|
保证 v2 §25/§27「历史回测与当前选股使用同一套引擎」的一致性。
|
||||||
"""
|
"""
|
||||||
panels = build_factor_panels(daily, factor_specs) # 未知因子在此抛 FactorError
|
return build_score_panel(daily, factor_specs) # 未知因子在此抛 FactorError
|
||||||
return composite_score(panels)
|
|
||||||
|
|
||||||
|
|
||||||
def resolve_observation_date(daily: pd.DataFrame, as_of: date | None) -> pd.Timestamp | None:
|
def resolve_observation_date(daily: pd.DataFrame, as_of: date | None) -> pd.Timestamp | None:
|
||||||
|
|||||||
@@ -14,9 +14,10 @@ from app.domain.entities.research import (
|
|||||||
SelectionSpec,
|
SelectionSpec,
|
||||||
UniverseSpec,
|
UniverseSpec,
|
||||||
)
|
)
|
||||||
|
from app.quant.composite import composite_score, cross_sectional_zscore
|
||||||
from app.quant.evaluation import run_factor_test
|
from app.quant.evaluation import run_factor_test
|
||||||
from app.quant.factors import compute_factor
|
from app.quant.factors import compute_factor
|
||||||
from app.quant.local_engine import composite_score, cross_sectional_zscore, rebalance_dates
|
from app.quant.local_engine import rebalance_dates
|
||||||
from pydantic import ValidationError
|
from pydantic import ValidationError
|
||||||
|
|
||||||
from conftest_quant import synthetic_daily
|
from conftest_quant import synthetic_daily
|
||||||
@@ -115,7 +116,7 @@ class TestEvaluation:
|
|||||||
|
|
||||||
class TestSingleStockDegradation:
|
class TestSingleStockDegradation:
|
||||||
def test_zscore_single_stock_keeps_candidate(self) -> None:
|
def test_zscore_single_stock_keeps_candidate(self) -> None:
|
||||||
from app.quant.local_engine import cross_sectional_zscore
|
from app.quant.composite import cross_sectional_zscore
|
||||||
|
|
||||||
daily = synthetic_daily({"ONLY": 0.001}, n=80)
|
daily = synthetic_daily({"ONLY": 0.001}, n=80)
|
||||||
_d, panel = compute_factor("momentum_20", daily)
|
_d, panel = compute_factor("momentum_20", daily)
|
||||||
|
|||||||
Reference in New Issue
Block a user