Files
simonandClaude Opus 4.7 271a9343a5 Initial commit: cc-cursor 全链路量化研究平台
7 Sprints 全部完成:
  Sprint 0: 基础设施 (DataManager + MariaDB)
  Sprint 1: 因子引擎 (34因子/12分类)
  Sprint 2: VectorBT 回测 (5策略+截面)
  Sprint 3: Optuna 优化 (+Walk-Forward)
  Sprint 4: ML 模型 (LightGBM+CatBoost)
  Sprint 5: Qwen 情绪因子 (三源新闻+日期对齐)
  Sprint 6: Agent 系统 (4Agent+日报.md/.html)

生产加固 (15项): Tushare双源fallback, SSH自动恢复, pool_pre_ping,
  save_daily先删后插, load_dotenv绝对路径, 日报5d/20d修复,
  RiskAgent改上证指数, 昨日对比+数据截止, mac_report utf8mb4,
  CLAUDE-*.md 9条已知Bug, demo全参数化, djapi数据源归一化,
  indexDatas API修正

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-06-07 15:59:05 +08:00

143 lines
4.9 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
特征工程:因子 → 特征矩阵 + 目标标签。
严禁使用未来数据。所有变换基于 expanding window 或训练集统计。
"""
import numpy as np
import pandas as pd
from sklearn.preprocessing import RobustScaler
class FeatureEngine:
"""
特征工程引擎。
参数:
lookahead: 预测未来 N 个交易日
label_type: 'regression' | 'classification'
winsorize_pct: 去极值的分位数边界 (0.01, 0.99)
nan_threshold: NaN 占比超过此值的因子直接剔除
"""
def __init__(
self,
lookahead: int = 5,
label_type: str = "regression",
winsorize_pct: tuple[float, float] = (0.01, 0.99),
nan_threshold: float = 0.3,
):
self.lookahead = lookahead
self.label_type = label_type
self.winsorize_pct = winsorize_pct
self.nan_threshold = nan_threshold
self._scaler = RobustScaler()
self._scaler_fitted = False
self._valid_features: list[str] = []
# ── 标签构建 ──────────────────────────────────────────
def build_labels(self, price_df: pd.DataFrame) -> pd.Series:
"""
构建目标标签。
regression: (close_{t+N} - close_t) / close_t * 100
classification: 1 if return > 0 else 0
"""
close = price_df["close"]
future = close.shift(-self.lookahead)
ret = (future - close) / close * 100
if self.label_type == "classification":
return (ret > 0).astype(int)
return ret.rename(f"y_fwd_{self.lookahead}")
# ── 特征构建 ──────────────────────────────────────────
def build(
self,
factor_df: pd.DataFrame,
price_df: pd.DataFrame,
fit: bool = True,
) -> tuple[pd.DataFrame, pd.Series]:
"""
构建特征矩阵 X 和标签 y。
参数:
factor_df: 因子 DataFrame, index=trade_date, columns=因子名
price_df: 价格 DataFrame, 需有 'close'
fit: True=训练模式(fit scaler + 记录有效特征),False=预测模式
返回:
X, yy 在 predict 模式下为 None
"""
X = factor_df.copy()
# 1. 剔除 NaN 率过高的列
if fit:
nan_ratio = X.isna().mean()
self._valid_features = list(nan_ratio[nan_ratio <= self.nan_threshold].index)
# 排除非因子列
self._valid_features = [c for c in self._valid_features
if c not in ("close", "open", "high", "low", "volume")]
X = X[self._valid_features].copy() if self._valid_features else X
# 2. 缺失值填充:前值填充 → 截面中位数
X = X.ffill().fillna(X.median())
# 3. 去极值(Winsorize
if fit:
lo, hi = self.winsorize_pct
self._winsor_lower = X.quantile(lo)
self._winsor_upper = X.quantile(hi)
for col in X.columns:
if col in getattr(self, "_winsor_lower", pd.Series()):
X[col] = X[col].clip(self._winsor_lower[col], self._winsor_upper[col])
# 4. 标准化(训练时 fit,预测时 transform
if fit:
X_scaled = self._scaler.fit_transform(X)
self._scaler_fitted = True
else:
X_scaled = self._scaler.transform(X)
X = pd.DataFrame(X_scaled, index=X.index, columns=X.columns)
# 5. 构建标签
y = self.build_labels(price_df) if fit else None
# 6. 对齐(删掉无法构建标签的行)
if fit:
valid_idx = X.index.intersection(y.dropna().index)
X = X.loc[valid_idx]
y = y.loc[valid_idx]
return X, y
# ── 多股票构建 ────────────────────────────────────────
def build_universe(
self,
factor_universe: dict[str, pd.DataFrame],
price_universe: dict[str, pd.DataFrame],
) -> tuple[pd.DataFrame, pd.Series]:
"""多股票拼接特征矩阵(每只股票独立处理再拼接)。"""
X_parts, y_parts = [], []
for ts_code in factor_universe:
f_df = factor_universe[ts_code]
p_df = price_universe.get(ts_code)
if p_df is None or f_df.empty or p_df.empty:
continue
X, y = self.build(f_df, p_df, fit=True)
if X.empty:
continue
X["_ts_code"] = ts_code
X_parts.append(X)
y_parts.append(y)
if not X_parts:
return pd.DataFrame(), pd.Series()
X_all = pd.concat(X_parts)
y_all = pd.concat(y_parts)
return X_all.drop(columns=["_ts_code"]), y_all