feat: 量化引擎加固 — 新增测试 + 数据/因子/回测层优化
- 新增 finance/tests/ 6 个测试套件(agents/backtest/dao_upsert/factors/features/fundamental_lookahead) - 数据层: data_manager / dao 优化,新增 upsert 逻辑 - 因子层: 基本面因子抽象定位 _mapping、ROE/PE/PB 重构 - 回测层: vectorbt/engine 大改动(251 行),report 增强 - ML 层: features/backtest_integration 特征工程与回测优化 - CLI: agent_cli 重构 - config/settings 扩充配置项
This commit is contained in:
@@ -8,6 +8,10 @@ import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from agents.base import BaseAgent
|
||||
from config.settings import (
|
||||
SELECTION_CORE_FACTORS, SELECTION_UNIVERSE_SIZE,
|
||||
SELECTION_SCORE_LIMIT, SELECTION_WINSORIZE_ZSCORE,
|
||||
)
|
||||
|
||||
|
||||
class SelectionAgent(BaseAgent):
|
||||
@@ -43,18 +47,13 @@ class SelectionAgent(BaseAgent):
|
||||
ts_codes = self.sent.get_scope_stocks()
|
||||
else:
|
||||
stocks = self.dm.get_stock_list()
|
||||
ts_codes = list(stocks.index[:100])
|
||||
ts_codes = list(stocks.index[:SELECTION_UNIVERSE_SIZE])
|
||||
|
||||
if not ts_codes:
|
||||
return {"date": date or self._today(), "top_picks": [], "score_df": pd.DataFrame()}
|
||||
|
||||
# 选择核心因子(覆盖多个维度,减少计算量)
|
||||
core_factors = [
|
||||
"momentum_20", "momentum_60",
|
||||
"rsi_14", "volatility_20",
|
||||
"vol_ratio_5", "ma_dev_20",
|
||||
"turnover_5", "amplitude_5",
|
||||
]
|
||||
# 选择核心因子(覆盖多个维度,减少计算量;参数来自配置中心)
|
||||
core_factors = list(SELECTION_CORE_FACTORS)
|
||||
factor_objects = [get_factor(n) for n in core_factors]
|
||||
|
||||
self.log("打分 {} 只股票 (权重={})".format(len(ts_codes), weighting))
|
||||
@@ -70,7 +69,7 @@ class SelectionAgent(BaseAgent):
|
||||
len(available) / len(ts_codes) * 100 if ts_codes else 0))
|
||||
|
||||
# 2. 打分(限制上限防止单次太慢)
|
||||
score_limit = min(len(available), 300)
|
||||
score_limit = min(len(available), SELECTION_SCORE_LIMIT)
|
||||
scores = {}
|
||||
valid_count = 0
|
||||
for i, ts_code in enumerate(available[:score_limit]):
|
||||
@@ -165,23 +164,38 @@ class SelectionAgent(BaseAgent):
|
||||
if len(row_clean) < 3:
|
||||
return None
|
||||
|
||||
# z-score 标准化
|
||||
z = (row_clean - factor_df[row_clean.index].mean()) / factor_df[row_clean.index].std().replace(0, 1)
|
||||
# z-score 标准化(用历史均值/标准差),可选的去极值避免单股离群主导 Top
|
||||
cols = row_clean.index
|
||||
mu = factor_df[cols].mean()
|
||||
std = factor_df[cols].std().replace(0, 1)
|
||||
z = (row_clean - mu) / std
|
||||
if SELECTION_WINSORIZE_ZSCORE:
|
||||
z = z.clip(-3, 3)
|
||||
return float(z.mean())
|
||||
|
||||
def _score_ml(self, ts_code: str, factor_df: pd.DataFrame, date: str | None) -> float | None:
|
||||
"""ML 模型打分。"""
|
||||
from models.features import FeatureEngine
|
||||
fe = FeatureEngine(lookahead=5)
|
||||
"""ML 模型打分。
|
||||
|
||||
daily = self.dm.get_daily(ts_code)
|
||||
要求注入名为 feature_engine 的、已用训练集 fit 过的 FeatureEngine,
|
||||
以及 ml_models(已训练模型)。两者缺一时明确退出,而不是沿用旧的
|
||||
未 fit 引擎静默失败。
|
||||
"""
|
||||
fe = self.feature_engine
|
||||
if fe is None:
|
||||
self.log("ML 打分需要注入 feature_engine(已 fit),当前未提供,跳过 ML 打分")
|
||||
return None
|
||||
if not self.ml_models:
|
||||
self.log("ML 打分需要 ml_models(已训练),当前为空,跳过 ML 打分")
|
||||
return None
|
||||
|
||||
daily = self.dm.get_daily(ts_code) if self.dm else None
|
||||
if daily is None or daily.empty:
|
||||
return None
|
||||
daily = daily.set_index("trade_date")
|
||||
|
||||
try:
|
||||
X, _ = fe.build(factor_df, daily, fit=False)
|
||||
if X.empty:
|
||||
if X is None or X.empty:
|
||||
return None
|
||||
if date and date in X.index:
|
||||
X = X.loc[[date]]
|
||||
@@ -191,5 +205,7 @@ class SelectionAgent(BaseAgent):
|
||||
model = self.ml_models.get("lightgbm") or list(self.ml_models.values())[0]
|
||||
pred = model.predict(X)
|
||||
return float(pred.iloc[0]) if len(pred) > 0 else None
|
||||
except Exception:
|
||||
except Exception as e:
|
||||
# 不再静默返回 None:记录原因,便于定位预测路径问题
|
||||
self.log("[WARN] ML 打分失败 ({}): {}".format(ts_code, e))
|
||||
return None
|
||||
|
||||
Reference in New Issue
Block a user