- 新增 finance/tests/ 6 个测试套件(agents/backtest/dao_upsert/factors/features/fundamental_lookahead) - 数据层: data_manager / dao 优化,新增 upsert 逻辑 - 因子层: 基本面因子抽象定位 _mapping、ROE/PE/PB 重构 - 回测层: vectorbt/engine 大改动(251 行),report 增强 - ML 层: features/backtest_integration 特征工程与回测优化 - CLI: agent_cli 重构 - config/settings 扩充配置项
239 lines
9.0 KiB
Python
239 lines
9.0 KiB
Python
"""
|
||
特征工程:因子 → 特征矩阵 + 目标标签。
|
||
|
||
严禁使用未来数据。所有变换的统计量(去极值边界、NaN 填充中位数、缩放器)
|
||
只在 fit(训练)阶段从训练样本估算并缓存在 self 上,predict 阶段复用这些
|
||
训练统计,避免训练/推理分布不一致(泄漏)和跨股票重复 refit scaler。
|
||
"""
|
||
|
||
import numpy as np
|
||
import pandas as pd
|
||
from sklearn.preprocessing import RobustScaler
|
||
|
||
|
||
class FeatureEngine:
|
||
"""
|
||
特征工程引擎。
|
||
|
||
参数:
|
||
lookahead: 预测未来 N 个交易日
|
||
label_type: 'regression' | 'classification'
|
||
winsorize_pct: 去极值的分位数边界 (0.01, 0.99)
|
||
nan_threshold: NaN 占比超过此值的因子直接剔除
|
||
"""
|
||
|
||
def __init__(
|
||
self,
|
||
lookahead: int = 5,
|
||
label_type: str = "regression",
|
||
winsorize_pct: tuple[float, float] = (0.01, 0.99),
|
||
nan_threshold: float = 0.3,
|
||
):
|
||
self.lookahead = lookahead
|
||
self.label_type = label_type
|
||
self.winsorize_pct = winsorize_pct
|
||
self.nan_threshold = nan_threshold
|
||
self._scaler = RobustScaler()
|
||
self._scaler_fitted = False
|
||
self._valid_features: list[str] = []
|
||
# 训练阶段缓存的统计量,predict 阶段复用
|
||
self._winsor_lower: pd.Series = pd.Series(dtype=float)
|
||
self._winsor_upper: pd.Series = pd.Series(dtype=float)
|
||
self._fill_medians: pd.Series = pd.Series(dtype=float)
|
||
|
||
def reset(self):
|
||
"""清空训练状态,便于重新 fit 新的训练集。"""
|
||
self._scaler = RobustScaler()
|
||
self._scaler_fitted = False
|
||
self._valid_features = []
|
||
self._winsor_lower = pd.Series(dtype=float)
|
||
self._winsor_upper = pd.Series(dtype=float)
|
||
self._fill_medians = pd.Series(dtype=float)
|
||
return self
|
||
|
||
# ── 标签构建 ──────────────────────────────────────────
|
||
|
||
def build_labels(self, price_df: pd.DataFrame) -> pd.Series:
|
||
"""
|
||
构建目标标签。
|
||
|
||
regression: (close_{t+N} - close_t) / close_t * 100
|
||
classification: 1 if return > 0 else 0
|
||
"""
|
||
close = price_df["close"]
|
||
future = close.shift(-self.lookahead)
|
||
ret = (future - close) / close * 100
|
||
|
||
if self.label_type == "classification":
|
||
# 末尾 lookahead 行无法构建标签,用 NaN 标记而非强制判负(避免标签偏差)
|
||
cls = (ret > 0).astype(float)
|
||
cls = cls.where(~ret.isna(), np.nan)
|
||
return cls.rename(f"y_fwd_{self.lookahead}")
|
||
|
||
return ret.rename(f"y_fwd_{self.lookahead}")
|
||
|
||
# ── 特征变换(fit 估算统计 / predict 复用统计) ────────
|
||
|
||
def _winsorize_bounds(self, X: pd.DataFrame):
|
||
lo, hi = self.winsorize_pct
|
||
q = X.quantile([lo, hi])
|
||
self._winsor_lower = q.loc[lo]
|
||
self._winsor_upper = q.loc[hi]
|
||
|
||
def _apply_transform(self, X: pd.DataFrame, fit: bool) -> pd.DataFrame:
|
||
"""去极值 + NaN 填充 + 缩放。fit 时估算并缓存统计,否则复用。"""
|
||
X = X.copy()
|
||
|
||
# NaN 填充:前值填充,缺失再按记录的中位数填充
|
||
X = X.ffill()
|
||
if fit:
|
||
# 用有效特征(非全 NaN 列)做列中位数
|
||
self._fill_medians = X.median()
|
||
for col in X.columns:
|
||
if col in self._fill_medians:
|
||
X[col] = X[col].fillna(self._fill_medians[col])
|
||
|
||
# 去极值
|
||
if fit:
|
||
self._winsorize_bounds(X)
|
||
for col in X.columns:
|
||
if col in self._winsor_lower.index and col in self._winsor_upper.index:
|
||
X[col] = X[col].clip(self._winsor_lower[col], self._winsor_upper[col])
|
||
|
||
# 缩放:fit 时 fit_transform,predict 时 transform(复用训练统计)
|
||
if fit:
|
||
X_scaled = self._scaler.fit_transform(X)
|
||
self._scaler_fitted = True
|
||
else:
|
||
if not self._scaler_fitted:
|
||
raise RuntimeError(
|
||
"FeatureEngine 尚未 fit,无法在 predict 模式下 transform。"
|
||
"必须先用 fit=True 调用 build 训练缩放统计。"
|
||
)
|
||
X_scaled = self._scaler.transform(X)
|
||
|
||
return pd.DataFrame(X_scaled, index=X.index, columns=X.columns)
|
||
|
||
# ── 特征构建 ──────────────────────────────────────────
|
||
|
||
def build(
|
||
self,
|
||
factor_df: pd.DataFrame,
|
||
price_df: pd.DataFrame,
|
||
fit: bool = True,
|
||
) -> tuple[pd.DataFrame, pd.Series]:
|
||
"""
|
||
构建特征矩阵 X 和标签 y。
|
||
|
||
参数:
|
||
factor_df: 因子 DataFrame, index=trade_date, columns=因子名
|
||
price_df: 价格 DataFrame, 需有 'close'
|
||
fit: True=训练模式(fit scaler + 记录有效特征),False=预测模式
|
||
|
||
返回:
|
||
X, y(y 在 predict 模式下为 None)
|
||
"""
|
||
X = factor_df.copy()
|
||
|
||
# 1. 剔除 NaN 率过高的列(只在 fit 时决定,predict 沿用同一列集)
|
||
if fit:
|
||
nan_ratio = X.isna().mean()
|
||
excluded = ("close", "open", "high", "low", "volume")
|
||
self._valid_features = [
|
||
c for c in X.columns
|
||
if nan_ratio[c] <= self.nan_threshold and c not in excluded
|
||
]
|
||
if not self._valid_features:
|
||
# 没有有效特征 → 空矩阵
|
||
return pd.DataFrame(index=X.index), None
|
||
if not self._valid_features:
|
||
# predict 且从未 fit → 无有效特征
|
||
return pd.DataFrame(index=X.index), None
|
||
X = X[self._valid_features].copy()
|
||
|
||
X = self._apply_transform(X, fit=fit)
|
||
|
||
# 2. 构建标签
|
||
y = self.build_labels(price_df) if fit else None
|
||
|
||
# 3. 对齐(删掉无法构建标签的行)
|
||
if fit:
|
||
valid_idx = X.index.intersection(y.dropna().index)
|
||
X = X.loc[valid_idx]
|
||
y = y.loc[valid_idx]
|
||
|
||
return X, y
|
||
|
||
# ── 多股票构建(一次性 fit,消除跨股票 refit 泄漏) ────
|
||
|
||
def build_universe(
|
||
self,
|
||
factor_universe: dict[str, pd.DataFrame],
|
||
price_universe: dict[str, pd.DataFrame],
|
||
) -> tuple[pd.DataFrame, pd.Series]:
|
||
"""
|
||
多股票拼接特征矩阵。
|
||
|
||
相比旧版(每只股票独立 fit=True 反复 refit scaler),现在:
|
||
- 先拼所有股票的因子值为一张横截面表,统一一次性 fit 缩放统计,
|
||
保证跨股票同分布;
|
||
- 标签按每只股票自身的前向收益构建,避免未来的跨股票串档。
|
||
"""
|
||
# 1. 收集每只股票有效期间内的特征行(保留 _ts_code 以区分)
|
||
parts: list[pd.DataFrame] = []
|
||
key_order: list[str] = []
|
||
for ts_code in factor_universe:
|
||
f_df = factor_universe[ts_code]
|
||
p_df = price_universe.get(ts_code)
|
||
if p_df is None or f_df.empty or p_df.empty or "close" not in p_df.columns:
|
||
continue
|
||
common = f_df.index.intersection(p_df.index)
|
||
if len(common) == 0:
|
||
continue
|
||
f_df = f_df.loc[common]
|
||
f_df = f_df.copy()
|
||
f_df["_ts_code"] = ts_code
|
||
parts.append(f_df)
|
||
key_order.append(ts_code)
|
||
|
||
if not parts:
|
||
return pd.DataFrame(), pd.Series()
|
||
|
||
X_all = pd.concat(parts)
|
||
|
||
# 2. 剔除 NaN 率过高的列(基于全横截面 fit)
|
||
nan_ratio = X_all.isna().mean()
|
||
excluded = ("close", "open", "high", "low", "volume", "_ts_code")
|
||
self._valid_features = [
|
||
c for c in X_all.columns
|
||
if nan_ratio[c] <= self.nan_threshold and c not in excluded
|
||
]
|
||
if not self._valid_features:
|
||
return pd.DataFrame(), pd.Series()
|
||
|
||
# 3. 一次性 fit 变换统计并应用(单次跨股票)
|
||
feat = X_all[self._valid_features].copy()
|
||
feat_scaled = self._apply_transform(feat, fit=True)
|
||
|
||
# 4. 每只股票构建自身标签并对齐(不把标签跨股票串起来)
|
||
y_parts = []
|
||
rows = []
|
||
for ts_code in key_order:
|
||
rows_mask = X_all["_ts_code"] == ts_code
|
||
f_local = feat_scaled[rows_mask]
|
||
p_local = price_universe[ts_code].loc[f_local.index]
|
||
y_local = self.build_labels(p_local).dropna()
|
||
keep = f_local.index.intersection(y_local.index)
|
||
if len(keep) == 0:
|
||
continue
|
||
rows.append(f_local.loc[keep])
|
||
y_parts.append(y_local.loc[keep])
|
||
|
||
if not rows:
|
||
return pd.DataFrame(), pd.Series()
|
||
|
||
X_out = pd.concat(rows)
|
||
y_out = pd.concat(y_parts)
|
||
if "_ts_code" in X_out.columns:
|
||
X_out = X_out.drop(columns=["_ts_code"])
|
||
return X_out, y_out |