feat: 量化引擎加固 — 新增测试 + 数据/因子/回测层优化
- 新增 finance/tests/ 6 个测试套件(agents/backtest/dao_upsert/factors/features/fundamental_lookahead) - 数据层: data_manager / dao 优化,新增 upsert 逻辑 - 因子层: 基本面因子抽象定位 _mapping、ROE/PE/PB 重构 - 回测层: vectorbt/engine 大改动(251 行),report 增强 - ML 层: features/backtest_integration 特征工程与回测优化 - CLI: agent_cli 重构 - config/settings 扩充配置项
This commit is contained in:
+140
-43
@@ -1,7 +1,9 @@
|
||||
"""
|
||||
特征工程:因子 → 特征矩阵 + 目标标签。
|
||||
|
||||
严禁使用未来数据。所有变换基于 expanding window 或训练集统计。
|
||||
严禁使用未来数据。所有变换的统计量(去极值边界、NaN 填充中位数、缩放器)
|
||||
只在 fit(训练)阶段从训练样本估算并缓存在 self 上,predict 阶段复用这些
|
||||
训练统计,避免训练/推理分布不一致(泄漏)和跨股票重复 refit scaler。
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
@@ -34,6 +36,20 @@ class FeatureEngine:
|
||||
self._scaler = RobustScaler()
|
||||
self._scaler_fitted = False
|
||||
self._valid_features: list[str] = []
|
||||
# 训练阶段缓存的统计量,predict 阶段复用
|
||||
self._winsor_lower: pd.Series = pd.Series(dtype=float)
|
||||
self._winsor_upper: pd.Series = pd.Series(dtype=float)
|
||||
self._fill_medians: pd.Series = pd.Series(dtype=float)
|
||||
|
||||
def reset(self):
|
||||
"""清空训练状态,便于重新 fit 新的训练集。"""
|
||||
self._scaler = RobustScaler()
|
||||
self._scaler_fitted = False
|
||||
self._valid_features = []
|
||||
self._winsor_lower = pd.Series(dtype=float)
|
||||
self._winsor_upper = pd.Series(dtype=float)
|
||||
self._fill_medians = pd.Series(dtype=float)
|
||||
return self
|
||||
|
||||
# ── 标签构建 ──────────────────────────────────────────
|
||||
|
||||
@@ -49,10 +65,55 @@ class FeatureEngine:
|
||||
ret = (future - close) / close * 100
|
||||
|
||||
if self.label_type == "classification":
|
||||
return (ret > 0).astype(int)
|
||||
# 末尾 lookahead 行无法构建标签,用 NaN 标记而非强制判负(避免标签偏差)
|
||||
cls = (ret > 0).astype(float)
|
||||
cls = cls.where(~ret.isna(), np.nan)
|
||||
return cls.rename(f"y_fwd_{self.lookahead}")
|
||||
|
||||
return ret.rename(f"y_fwd_{self.lookahead}")
|
||||
|
||||
# ── 特征变换(fit 估算统计 / predict 复用统计) ────────
|
||||
|
||||
def _winsorize_bounds(self, X: pd.DataFrame):
|
||||
lo, hi = self.winsorize_pct
|
||||
q = X.quantile([lo, hi])
|
||||
self._winsor_lower = q.loc[lo]
|
||||
self._winsor_upper = q.loc[hi]
|
||||
|
||||
def _apply_transform(self, X: pd.DataFrame, fit: bool) -> pd.DataFrame:
|
||||
"""去极值 + NaN 填充 + 缩放。fit 时估算并缓存统计,否则复用。"""
|
||||
X = X.copy()
|
||||
|
||||
# NaN 填充:前值填充,缺失再按记录的中位数填充
|
||||
X = X.ffill()
|
||||
if fit:
|
||||
# 用有效特征(非全 NaN 列)做列中位数
|
||||
self._fill_medians = X.median()
|
||||
for col in X.columns:
|
||||
if col in self._fill_medians:
|
||||
X[col] = X[col].fillna(self._fill_medians[col])
|
||||
|
||||
# 去极值
|
||||
if fit:
|
||||
self._winsorize_bounds(X)
|
||||
for col in X.columns:
|
||||
if col in self._winsor_lower.index and col in self._winsor_upper.index:
|
||||
X[col] = X[col].clip(self._winsor_lower[col], self._winsor_upper[col])
|
||||
|
||||
# 缩放:fit 时 fit_transform,predict 时 transform(复用训练统计)
|
||||
if fit:
|
||||
X_scaled = self._scaler.fit_transform(X)
|
||||
self._scaler_fitted = True
|
||||
else:
|
||||
if not self._scaler_fitted:
|
||||
raise RuntimeError(
|
||||
"FeatureEngine 尚未 fit,无法在 predict 模式下 transform。"
|
||||
"必须先用 fit=True 调用 build 训练缩放统计。"
|
||||
)
|
||||
X_scaled = self._scaler.transform(X)
|
||||
|
||||
return pd.DataFrame(X_scaled, index=X.index, columns=X.columns)
|
||||
|
||||
# ── 特征构建 ──────────────────────────────────────────
|
||||
|
||||
def build(
|
||||
@@ -74,40 +135,28 @@ class FeatureEngine:
|
||||
"""
|
||||
X = factor_df.copy()
|
||||
|
||||
# 1. 剔除 NaN 率过高的列
|
||||
# 1. 剔除 NaN 率过高的列(只在 fit 时决定,predict 沿用同一列集)
|
||||
if fit:
|
||||
nan_ratio = X.isna().mean()
|
||||
self._valid_features = list(nan_ratio[nan_ratio <= self.nan_threshold].index)
|
||||
# 排除非因子列
|
||||
self._valid_features = [c for c in self._valid_features
|
||||
if c not in ("close", "open", "high", "low", "volume")]
|
||||
X = X[self._valid_features].copy() if self._valid_features else X
|
||||
excluded = ("close", "open", "high", "low", "volume")
|
||||
self._valid_features = [
|
||||
c for c in X.columns
|
||||
if nan_ratio[c] <= self.nan_threshold and c not in excluded
|
||||
]
|
||||
if not self._valid_features:
|
||||
# 没有有效特征 → 空矩阵
|
||||
return pd.DataFrame(index=X.index), None
|
||||
if not self._valid_features:
|
||||
# predict 且从未 fit → 无有效特征
|
||||
return pd.DataFrame(index=X.index), None
|
||||
X = X[self._valid_features].copy()
|
||||
|
||||
# 2. 缺失值填充:前值填充 → 截面中位数
|
||||
X = X.ffill().fillna(X.median())
|
||||
X = self._apply_transform(X, fit=fit)
|
||||
|
||||
# 3. 去极值(Winsorize)
|
||||
if fit:
|
||||
lo, hi = self.winsorize_pct
|
||||
self._winsor_lower = X.quantile(lo)
|
||||
self._winsor_upper = X.quantile(hi)
|
||||
for col in X.columns:
|
||||
if col in getattr(self, "_winsor_lower", pd.Series()):
|
||||
X[col] = X[col].clip(self._winsor_lower[col], self._winsor_upper[col])
|
||||
|
||||
# 4. 标准化(训练时 fit,预测时 transform)
|
||||
if fit:
|
||||
X_scaled = self._scaler.fit_transform(X)
|
||||
self._scaler_fitted = True
|
||||
else:
|
||||
X_scaled = self._scaler.transform(X)
|
||||
|
||||
X = pd.DataFrame(X_scaled, index=X.index, columns=X.columns)
|
||||
|
||||
# 5. 构建标签
|
||||
# 2. 构建标签
|
||||
y = self.build_labels(price_df) if fit else None
|
||||
|
||||
# 6. 对齐(删掉无法构建标签的行)
|
||||
# 3. 对齐(删掉无法构建标签的行)
|
||||
if fit:
|
||||
valid_idx = X.index.intersection(y.dropna().index)
|
||||
X = X.loc[valid_idx]
|
||||
@@ -115,28 +164,76 @@ class FeatureEngine:
|
||||
|
||||
return X, y
|
||||
|
||||
# ── 多股票构建 ────────────────────────────────────────
|
||||
# ── 多股票构建(一次性 fit,消除跨股票 refit 泄漏) ────
|
||||
|
||||
def build_universe(
|
||||
self,
|
||||
factor_universe: dict[str, pd.DataFrame],
|
||||
price_universe: dict[str, pd.DataFrame],
|
||||
) -> tuple[pd.DataFrame, pd.Series]:
|
||||
"""多股票拼接特征矩阵(每只股票独立处理再拼接)。"""
|
||||
X_parts, y_parts = [], []
|
||||
"""
|
||||
多股票拼接特征矩阵。
|
||||
|
||||
相比旧版(每只股票独立 fit=True 反复 refit scaler),现在:
|
||||
- 先拼所有股票的因子值为一张横截面表,统一一次性 fit 缩放统计,
|
||||
保证跨股票同分布;
|
||||
- 标签按每只股票自身的前向收益构建,避免未来的跨股票串档。
|
||||
"""
|
||||
# 1. 收集每只股票有效期间内的特征行(保留 _ts_code 以区分)
|
||||
parts: list[pd.DataFrame] = []
|
||||
key_order: list[str] = []
|
||||
for ts_code in factor_universe:
|
||||
f_df = factor_universe[ts_code]
|
||||
p_df = price_universe.get(ts_code)
|
||||
if p_df is None or f_df.empty or p_df.empty:
|
||||
if p_df is None or f_df.empty or p_df.empty or "close" not in p_df.columns:
|
||||
continue
|
||||
X, y = self.build(f_df, p_df, fit=True)
|
||||
if X.empty:
|
||||
common = f_df.index.intersection(p_df.index)
|
||||
if len(common) == 0:
|
||||
continue
|
||||
X["_ts_code"] = ts_code
|
||||
X_parts.append(X)
|
||||
y_parts.append(y)
|
||||
if not X_parts:
|
||||
f_df = f_df.loc[common]
|
||||
f_df = f_df.copy()
|
||||
f_df["_ts_code"] = ts_code
|
||||
parts.append(f_df)
|
||||
key_order.append(ts_code)
|
||||
|
||||
if not parts:
|
||||
return pd.DataFrame(), pd.Series()
|
||||
X_all = pd.concat(X_parts)
|
||||
y_all = pd.concat(y_parts)
|
||||
return X_all.drop(columns=["_ts_code"]), y_all
|
||||
|
||||
X_all = pd.concat(parts)
|
||||
|
||||
# 2. 剔除 NaN 率过高的列(基于全横截面 fit)
|
||||
nan_ratio = X_all.isna().mean()
|
||||
excluded = ("close", "open", "high", "low", "volume", "_ts_code")
|
||||
self._valid_features = [
|
||||
c for c in X_all.columns
|
||||
if nan_ratio[c] <= self.nan_threshold and c not in excluded
|
||||
]
|
||||
if not self._valid_features:
|
||||
return pd.DataFrame(), pd.Series()
|
||||
|
||||
# 3. 一次性 fit 变换统计并应用(单次跨股票)
|
||||
feat = X_all[self._valid_features].copy()
|
||||
feat_scaled = self._apply_transform(feat, fit=True)
|
||||
|
||||
# 4. 每只股票构建自身标签并对齐(不把标签跨股票串起来)
|
||||
y_parts = []
|
||||
rows = []
|
||||
for ts_code in key_order:
|
||||
rows_mask = X_all["_ts_code"] == ts_code
|
||||
f_local = feat_scaled[rows_mask]
|
||||
p_local = price_universe[ts_code].loc[f_local.index]
|
||||
y_local = self.build_labels(p_local).dropna()
|
||||
keep = f_local.index.intersection(y_local.index)
|
||||
if len(keep) == 0:
|
||||
continue
|
||||
rows.append(f_local.loc[keep])
|
||||
y_parts.append(y_local.loc[keep])
|
||||
|
||||
if not rows:
|
||||
return pd.DataFrame(), pd.Series()
|
||||
|
||||
X_out = pd.concat(rows)
|
||||
y_out = pd.concat(y_parts)
|
||||
if "_ts_code" in X_out.columns:
|
||||
X_out = X_out.drop(columns=["_ts_code"])
|
||||
return X_out, y_out
|
||||
Reference in New Issue
Block a user