feat: 股息率案例口径 + 策略库与图表统一 + 回测存档完整化

汇总三轮未提交的开发(每轮均在本机 MariaDB + 真实浏览器上验证):

1) 股息率案例(全市场股息率最高 n 只,默认 20,每 m 月择股)
   - 新增日频估值表 daily_basic + 迁移;股息率因子(dv_ratio / dividend_yield / TTM)
   - 名称历史表 stock_name_history:剔除 ST 按**择股日当时名称**判定,消除
     「曾高股息后 ST」的股息陷阱(实测 3.70pp 偏差)
   - 区间择股/调仓双周期(m 择股 / y 调仓)、指数成分与白名单、停牌近似剔除
   - 复权因子口径核对(4,164,742 行、缺失 0.0%)、收盘价成交与涨跌停拦单
   - 案例实测:2020-01-01~2026-09-04 总收益 +24.86%(年化 3.52%、回撤 -28.58%)

2) 策略库与前端统一
   - strategy 表 + CRUD/PUT 原地更新 + `describe_strategy` 按 spec 真实推导
     「一句话说明 + 计算公式 + 执行步骤 + 注意事项」(与引擎实执行规则同源)
   - 任何出现股票代码处都成对显示名称且可点击进个股页
   - 全站图表基座统一 TradingView Lightweight Charts(ECharts 依赖、
     锁文件、组件与文档标注一并清除),买卖点标记只落在真实交易日上

3) 回测存档完整化(可往复查看)
   - 同步端点(POST /api/backtests、/api/factor-tests)此前完全不落库 → 现在同样归档,
     归档 id 经响应头 X-Experiment-Id 返回(不破坏 response_model)
   - data_version 首次真实写入(数据快照指纹:最新交易日 + 各表规模)
   - 个股收益曲线默认**全量保存**(此前硬截断 60 只);超出体积预算才裁剪,
     并写 archive_meta(机器可读)+ unimplemented(人可读)如实标注
   - 列表 kind/q 过滤 + X-Total-Count(此前 limit=50 静默截断)、DELETE 归档
   - 只读归档页 /experiments/{id}(Server Component,SSR 直出**选股条件**与
     **交易执行依据**);结果视图按 kind 分发(backtest/factor_test/selection),
     非回测归档不套用回测口径
   - 新增 CLI:prune_experiments(保留策略,默认 dry-run)、
     restore_experiment_from_job(从 Job 副本按原 id 重建被删的历史归档,默认 dry-run)

门禁:pytest 388 passed、ruff All checks passed、tsc 0 错误、图表单测 7 passed、
next build 成功、契约脚本 verify_strategy_workspace 59/59(含按 kind 逐类验证归档页)。
This commit is contained in:
Simon
2026-09-20 07:31:04 +08:00
parent 7e15b7251e
commit 23972e7063
112 changed files with 17908 additions and 3893 deletions
+114 -51
View File
@@ -14,7 +14,7 @@ from datetime import date
import pandas as pd
from app.domain.entities.market import FinancialIndicator
from app.domain.entities.market import DAILY_BASIC_NUMERIC_FIELDS, FinancialIndicator
from app.domain.entities.selection import (
SelectionCandidate,
SelectionQuery,
@@ -182,8 +182,12 @@ _STATIC_PREFIX = "static."
_FUNDAMENTAL_PREFIX = "fundamental."
def condition_needed_columns(query: SelectionQuery) -> set[str]:
"""条件引用的行情列(fundamental/static 走元数据与财务表,不需要行情列)。"""
def condition_needed_columns(query) -> set[str]:
"""条件引用的行情/指标列(fundamental/static 走元数据与财务表,不需要面板列)。
query 可以是 SelectionQuery,也可以是任何带 `conditions` 的对象
(ResearchSpec 亦然)—— 回测与选股共用本函数,保证列裁剪一致。
"""
needed = {"close"}
names = [c.field for c in query.conditions] + [
c.ref for c in query.conditions if c.ref and not c.ref.startswith(_FUNDAMENTAL_PREFIX)
@@ -195,16 +199,110 @@ def condition_needed_columns(query: SelectionQuery) -> set[str]:
if f not in _TECH_DERIVED:
needed.add(f)
continue
if f in DAILY_BASIC_NUMERIC_FIELDS:
needed.add(f) # 每日指标列(dv_ratio / pe / pb / total_mv …)
continue
try: # 其余按已注册因子处理
defn, _fn = get_factor(f)
except FactorError:
raise ValueError(
f"条件字段未知:{f}(可用: 行情列/ma20/ma60/已注册因子/static.*/fundamental.*)"
f"条件字段未知:{f}(可用: 行情列/ma20/ma60/已注册因子/"
"每日指标列(dv_ratio 等)/static.*/fundamental.*)"
) from None
needed.update(defn.requires)
return needed
def build_condition_fields(
daily: pd.DataFrame,
conditions,
obs: pd.Timestamp,
) -> dict[str, pd.Series]:
"""条件各字段在 obs(<= as_of 的最近交易日)的截面值。
返回 {字段名: Series(index=symbol)},覆盖:
- 行情原列:close / open / high / low / volume / amount
- 技术派生:ma20 / ma60
- 每日指标列:dv_ratio / dv_ttm / pe / pb / total_mv …(由 Service 并入 daily)
- 已注册因子:momentum_60 等(在 <=obs 的截断数据上计算,无未来函数)
回测与选股共用本函数(v2 §25 一致性)。
"""
view = daily[pd.to_datetime(daily["trade_date"]) <= obs]
if view.empty:
return {}
close = view.pivot(index="trade_date", columns="symbol", values="close").sort_index()
close.index = pd.to_datetime(close.index)
fields: dict[str, pd.Series] = {}
for col in ("open", "high", "low", "volume", "amount"):
if col in view.columns:
panel = view.pivot(index="trade_date", columns="symbol", values=col).sort_index()
panel.index = pd.to_datetime(panel.index)
if obs in panel.index:
fields[col] = panel.loc[obs]
if obs in close.index:
fields["close"] = close.loc[obs]
ma20 = close.rolling(20).mean()
ma60 = close.rolling(60).mean()
if obs in ma20.index:
fields["ma20"] = ma20.loc[obs]
if obs in ma60.index:
fields["ma60"] = ma60.loc[obs]
wanted: set[str] = set()
for cond in conditions:
for f in (cond.field, cond.ref):
if not f or f.startswith((_STATIC_PREFIX, _FUNDAMENTAL_PREFIX)):
continue
if f in _TECH_DERIVED or f in fields:
continue
wanted.add(f)
for f in sorted(wanted):
if f in DAILY_BASIC_NUMERIC_FIELDS:
if f in view.columns:
panel = view.pivot(index="trade_date", columns="symbol", values=f).sort_index()
panel.index = pd.to_datetime(panel.index)
if obs in panel.index:
fields[f] = panel.loc[obs]
continue
try:
_defn, panel = compute_factor(f, view)
except FactorError:
continue # 已在 condition_needed_columns 报错;此处防御
if obs in panel.index:
fields[f] = panel.loc[obs]
return fields
def eligible_symbols(
candidates,
conditions,
statics: dict[str, dict],
fields: dict[str, pd.Series],
financial: dict[str, FinancialIndicator],
) -> dict[str, list[str]]:
"""逐股求值全部条件(AND),返回 {symbol: [各条件通过情况文案]}(仅通过者)。
`candidates` 限定参与求值的股票(通常 = universe 过滤后的 symbol 列表)。
回测(ResearchSpec.conditions)与选股(SelectionQuery.conditions)共用,
确保「历史某日 Selection == 回测当日 Selection」(v2 §25 / v3 §28)。
"""
passed: dict[str, list[str]] = {}
for sym in candidates:
statuses: list[str] = []
all_ok = True
for cond in conditions:
ok = _eval_condition(cond, sym, statics, fields, financial)
statuses.append(
f"{cond.field} {cond.op} {cond.ref or cond.value}: {'通过' if ok else '未通过'}"
)
all_ok = all_ok and ok
if all_ok:
passed[sym] = statuses
return passed
def run_condition_selection(
daily: pd.DataFrame,
stocks: list,
@@ -229,57 +327,22 @@ def run_condition_selection(
config_snapshot=query.model_dump(mode="json"),
)
view = daily[pd.to_datetime(daily["trade_date"]) <= obs]
close = view.pivot(index="trade_date", columns="symbol", values="close").sort_index()
close.index = pd.to_datetime(close.index)
# 技术字段面板(obs 行)
tech: dict[str, pd.Series] = {}
for col in ("close", "open", "high", "low", "volume", "amount"):
if col in view.columns and col != "close":
panel = view.pivot(index="trade_date", columns="symbol", values=col).sort_index()
panel.index = pd.to_datetime(panel.index)
tech[col] = panel.loc[obs]
tech["close"] = close.loc[obs]
tech["ma20"] = close.rolling(20).mean().loc[obs]
tech["ma60"] = close.rolling(60).mean().loc[obs]
# 因子字段按需计算
for cond in query.conditions:
for f in (cond.field, cond.ref):
if f is None or f.startswith((_STATIC_PREFIX, _FUNDAMENTAL_PREFIX)) or f in tech:
continue
if f in _TECH_DERIVED or f in ("close", "open", "high", "low", "volume", "amount"):
continue
try:
_defn, panel = compute_factor(f, view)
except FactorError:
continue # 已在 condition_needed_columns 报错;此处防御
if obs in panel.index:
tech[f] = panel.loc[obs]
# 共享字段面板 + 求值器(回测与选股同一实现,v2 §25 一致性)
tech = build_condition_fields(daily, query.conditions, obs)
statics = {s.symbol: s.model_dump() for s in stocks}
passed = eligible_symbols(sorted(statics), query.conditions, statics, tech, financial or {})
candidates: list[SelectionCandidate] = []
passed_symbols: list[str] = []
for sym in sorted(statics):
statuses: list[str] = []
all_ok = True
for cond in query.conditions:
ok = _eval_condition(cond, sym, statics, tech, financial or {})
statuses.append(f"{cond.field} {cond.op} {cond.ref or cond.value}: {'通过' if ok else '未通过'}")
all_ok = all_ok and ok
if all_ok:
passed_symbols.append(sym)
candidates.append(
SelectionCandidate(
symbol=sym,
rank=0, # 占位,末尾统一编号
score=1.0,
filter_status=statuses,
selection_reason=[f"通过全部 {len(query.conditions)} 条条件"],
)
for rank, sym in enumerate(sorted(passed), start=1):
candidates.append(
SelectionCandidate(
symbol=sym,
rank=rank,
score=1.0,
filter_status=passed[sym],
selection_reason=[f"通过全部 {len(query.conditions)} 条条件"],
)
for rank, c in enumerate(candidates, start=1):
c.rank = rank
)
return SelectionResult(
as_of_date=resolved,