Files
qlib/backend/app/quant/trade_reasons.py
T
Simon 48a97c2a12 feat(backtest): 买卖点理由(用数据说话)+ 因子曲线 + 曲线新页面放大
用户要求:「所有买卖点详细说明买卖理由,用数据说话」「回测图上增加因子相关曲线
(买卖依据是股息率,就加股息率曲线)」「所有曲线能弹出新页面放大」。

一、买卖理由(后端产出结构化数据,前端只展示)
- 新增 `quant/trade_reasons.py`:封闭词表 + 文案构造器,组合引擎与单策略引擎共用,
  避免两个引擎对同一件事写出两种说法。理由里带**引擎当时的真实数字**:
  综合分名次/候选数/综合分/各因子原始值/持有交易日/预算与最低佣金/涨停比值等。
- 买入:按名次建仓、顺延成交、涨停未买、停牌未买、现金不足、不足最低佣金;
  卖出:跌出 TopN(含第几名掉出)、被股票池过滤(与「跌出 TopN」分开写)、
  超 Tmax 强制了结、Tmin 保护暂留、停牌/跌停顺延。
- `ActionRecord.reason` 覆盖**成交与未成交**全部买卖点(原 `reject_reason` 保留不动,
  老归档仍可读);`Trade.entry_reason / exit_reason` 跟着成交记录走。
- 名次来自调仓日完整排名(新增 `_ranked_by_day`),拿不到名次时如实写「未给出名次」,
  绝不编造一个名次填进去。
- 未成交明细不再只写执行层原因:把「为什么选中它、当时各因子多少」一并给出。

二、因子曲线
- `FactorCurve`:每个策略因子一条曲线,值为**当日持仓按市值加权平均的原始值**
  (不做 z-score、不按方向取反,空仓日不落点、不插值、不用 0 填充),并带
  label/direction/unit 供界面说明口径;`FactorDef/FactorTemplate` 新增 `unit`
  (股息率 %、量比/接近新高 倍数、动量等 小数),11 个内置因子实例已逐一核对。
- 归档体积预算照旧按整包计量,无需改迁移。

三、界面
- 结果页新增「买卖说明」区块:全部买卖点 + 理由 + 数字标签,支持方向/成交状态/关键字
  筛选与日期排序;成交明细表加「为什么买 / 为什么卖」两列;新增「因子曲线」区块,
  每条曲线标出组合成交日,直接对照「买卖发生在什么水平」。
- 「新页面放大」:每条曲线(净值/回撤/因子/个股/月度)都能开 `/charts/{归档id}?s=...`
  整页看大图;放大页是 Server Component,数据从归档直出,URL 可分享且与归档一致。
  未归档的结果如实说明「未归档,无法放大」,不给坏链接。
- 数字格式与后端 `f"{v:.4f}"` 同规则(四舍六入五成双):修掉 0.03125 在理由原文里
  显示 0.0312、旁边标签显示 0.0313 的不一致(17 组边界值与 Python 逐一比对一致)。
- `/factors/compose` 结果区改用同一个 `BacktestResultView`,两处口径不会再漂移。

验证:
- 新增 `tests/test_trade_reasons.py` 8 条(买入数字、跌出 TopN 名次、不在候选池、
  Tmax、Tmin 暂留、涨停未成交、因子曲线加权值、空仓不落点);后端 510 条全过,ruff clean。
- 真实数据端到端:`/api/combos/run` 6 个月高股息组合(EXP-8EA2819B)13 个买卖点
  100% 带理由与数字,因子曲线 dividend_yield 117 点、单位 %;
  `scripts/verify_backtest_page_contract.py`(4 年、301 个买卖点、140 笔成交)扩展断言
  理由词表/名次/因子值/曲线单调性后通过。
- 浏览器实测:归档详情页与放大页 `/charts/...?s=factor:dividend_yield` 等 5 种曲线
  全部 200 渲染,截图确认表格与曲线数值正确。
2026-10-01 17:57:00 +08:00

404 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""买卖理由(数据化)与因子曲线的共享构造器。
**为什么单独一个模块**:两套回测引擎(`combo_engine` 组合回测、`local_engine`
单策略回测)都要回答同一个问题 ——「这一买一卖,当时的数字是多少?」。如果各写一份,
措辞、口径、字段名迟早分叉,用户在两处看到的「理由」会互相矛盾。
因此这里只放两件事:
1. **封闭的原因词表 + 构造器**:每个 `code` 对应一类可判定的原因,`text` 里带关键数字,
`data` 里放当时的原始数值(排名 / 候选数 / 综合分 / 因子原始值 / 持有交易日 / 预算 …)。
引擎只允许用词表里的 code(`REASON_CODES`),避免出现「文案随手写」的漂移。
2. **因子曲线的口径实现**:持仓股票的**权重加权平均原始值**,空仓日不落点、
不插值、不按方向取负(低为好的因子也原样画,方向由 `FactorCurve.direction` 说明)。
数值一律「引擎当时算出来的」:前端只展示,不推算 —— 界面上不该出现看起来像真的数字。
"""
from __future__ import annotations
import math
from collections.abc import Mapping
import pandas as pd
from app.domain.entities.research import CurvePoint, FactorCurve, TradeReason
from app.quant.factors import FactorDef
# ---------- 封闭原因词表 ----------
# 买入(成交)
BUY_ENTER = "buy_enter_topn"
BUY_DEFER_FILLED = "buy_defer_filled"
# 买入(未成交 / 未执行)
BUY_SKIP_LIMIT_UP = "buy_skip_limit_up"
BUY_SKIP_HALTED = "buy_skip_halted"
BUY_SKIP_NO_CASH = "buy_skip_no_cash"
BUY_SKIP_MIN_COMMISSION = "buy_skip_min_commission"
# 卖出(成交)
SELL_DROP_TOPN = "sell_drop_topn"
SELL_FORCE_TMAX = "sell_force_tmax"
# 卖出(顺延 / 未成交)
SELL_DEFER_TMIN = "sell_defer_tmin"
SELL_DEFER_HALTED = "sell_defer_halted"
SELL_DEFER_LIMIT_DOWN = "sell_defer_limit_down"
REASON_CODES = frozenset(
{
BUY_ENTER,
BUY_DEFER_FILLED,
BUY_SKIP_LIMIT_UP,
BUY_SKIP_HALTED,
BUY_SKIP_NO_CASH,
BUY_SKIP_MIN_COMMISSION,
SELL_DROP_TOPN,
SELL_FORCE_TMAX,
SELL_DEFER_TMIN,
SELL_DEFER_HALTED,
SELL_DEFER_LIMIT_DOWN,
}
)
# 原因分类的中文短标签(前端筛选 / 表格上色用;改文案只改这里)
REASON_LABELS: dict[str, str] = {
BUY_ENTER: "按名次建仓",
BUY_DEFER_FILLED: "顺延后成交",
BUY_SKIP_LIMIT_UP: "涨停未买",
BUY_SKIP_HALTED: "停牌未买",
BUY_SKIP_NO_CASH: "现金不足",
BUY_SKIP_MIN_COMMISSION: "不足最低佣金",
SELL_DROP_TOPN: "跌出 TopN",
SELL_FORCE_TMAX: "持有超 Tmax",
SELL_DEFER_TMIN: "Tmin 保护暂留",
SELL_DEFER_HALTED: "停牌未卖",
SELL_DEFER_LIMIT_DOWN: "跌停未卖",
}
def _num(v, digits: int = 6):
"""把 numpy/pandas 数值安全地压成原生 float(NaN/inf 一律不带进理由里)。"""
if v is None:
return None
try:
f = float(v)
except (TypeError, ValueError):
return None
if math.isnan(f) or math.isinf(f):
return None
return round(f, digits)
def factor_values(
factor_panels: Mapping[str, tuple[FactorDef, pd.DataFrame]],
day,
symbol: str,
) -> dict[str, float]:
"""该个股在 `day` 的各因子**原始值**(缺失因子不写进 data,不用 0 冒充)。
用交易日精确匹配:调仓日的打分与理由是同一份面板,所以这里取不到值就意味着
「该股当日无该因子值」,如实缺失比填 0 更可信。
"""
out: dict[str, float] = {}
for name, (_defn, panel) in factor_panels.items():
if day not in panel.index or symbol not in panel.columns:
continue
v = _num(panel.at[day, symbol])
if v is not None:
out[name] = v
return out
def _rank_data(
*,
rank: int | None,
total: int | None,
top_n: int | None,
score: float | None,
factors: dict[str, float] | None,
) -> dict:
data: dict = {}
if rank is not None:
data["rank"] = int(rank)
if total is not None:
data["total"] = int(total)
if top_n is not None:
data["top_n"] = int(top_n)
if score is not None:
data["score"] = _num(score)
if factors:
data["factors"] = factors
return data
def _rank_text(rank: int | None, total: int | None, top_n: int | None, score: float | None) -> str:
if rank is None:
return "调仓日综合分未给出名次"
parts = [f"综合分第 {rank}"]
if total:
parts.append(f"/{total}")
parts.append(" 名")
if top_n is not None:
parts.append(f"(TopN={top_n})")
if score is not None:
parts.append(f",综合分 {score:.4f}")
return "".join(parts)
def buy_filled(
*,
rank: int | None,
total: int | None,
top_n: int | None,
score: float | None,
factors: dict[str, float] | None,
price: float | None,
budget: float | None = None,
deferred: bool = False,
) -> TradeReason:
"""买入成交的理由:名次 + 综合分 + 各因子当时的原始值 + 成交价。"""
head = "顺延买入成交" if deferred else "调仓日选中并建仓"
text = f"{head}:{_rank_text(rank, total, top_n, score)}"
if price is not None:
text += f";成交价 {price:.2f} 元"
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
if price is not None:
data["price"] = _num(price, 4)
if budget is not None:
data["budget"] = _num(budget, 2)
return TradeReason(code=BUY_DEFER_FILLED if deferred else BUY_ENTER, text=text, data=data)
def buy_skipped(
code: str,
*,
rank: int | None = None,
total: int | None = None,
top_n: int | None = None,
score: float | None = None,
factors: dict[str, float] | None = None,
close: float | None = None,
prev_close: float | None = None,
limit_ratio: float | None = None,
budget: float | None = None,
min_commission: float | None = None,
not_in_pool: bool = False,
) -> TradeReason:
"""买入未成交 / 未执行的理由(涨停、停牌、现金不足、佣金门槛)。"""
if code not in REASON_CODES:
raise ValueError(f"未知的买入未成交原因:{code}")
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
base = _rank_text(rank, total, top_n, score)
if code == BUY_SKIP_LIMIT_UP:
ratio = (
_num(close / prev_close, 4)
if close is not None and prev_close not in (None, 0)
else None
)
text = f"{base},但当日涨停"
if ratio is not None:
text += f"(收盘 {close:.2f} / 前收 {prev_close:.2f} = {ratio:.3f}"
text += f" ≥ 涨停阈值 {limit_ratio:.3f})" if limit_ratio else ")"
text += ",无法追买"
if ratio is not None:
data["close_prev_ratio"] = ratio
if limit_ratio is not None:
data["limit_ratio"] = _num(limit_ratio, 4)
elif code == BUY_SKIP_HALTED:
text = f"{base},但当日无行情(停牌),无法买入"
elif code == BUY_SKIP_NO_CASH:
text = f"{base},但可用现金不足,未成交"
if budget is not None:
text += f"(可用预算 {budget:.2f} 元)"
elif code == BUY_SKIP_MIN_COMMISSION:
text = f"{base},但预算不足以覆盖最低佣金,未成交"
if budget is not None and min_commission is not None:
text += f"(预算 {budget:.2f} 元 < 最低佣金 {min_commission:.2f} 元)"
else: # pragma: no cover - 上面的分支已覆盖全部代码
text = base
if close is not None:
data["close"] = _num(close, 4)
if prev_close is not None:
data["prev_close"] = _num(prev_close, 4)
if budget is not None:
data["budget"] = _num(budget, 2)
if min_commission is not None:
data["min_commission"] = _num(min_commission, 2)
if not_in_pool:
data["in_pool"] = False
return TradeReason(code=code, text=text, data=data)
def sell_filled(
*,
code: str,
rank: int | None,
total: int | None,
top_n: int | None,
score: float | None,
factors: dict[str, float] | None,
hold_days: int,
tmin: int | None = None,
tmax: int | None = None,
price: float | None = None,
return_pct: float | None = None,
not_in_pool: bool = False,
) -> TradeReason:
"""卖出成交的理由(跌出 TopN / 持有超 Tmax),带持有交易日与当时名次。
`not_in_pool=True` 表示该股已**不在候选池**(被股票池/条件过滤,如转为 ST),
与「在池内但排名掉出去」是两回事,文案与 data 都分开写。
"""
if code == SELL_FORCE_TMAX:
text = f"持有 {hold_days} 个交易日 > Tmax={tmax},强制了结(与排名无关)"
else:
code = SELL_DROP_TOPN
if not_in_pool:
text = f"调仓日已不在候选池(被股票池/条件过滤);持有 {hold_days} 个交易日"
else:
text = f"调仓日跌出 TopN:{_rank_text(rank, total, top_n, score)};持有 {hold_days} 个交易日"
if tmin is not None:
text += f" ≥ Tmin={tmin}"
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
data["hold_days"] = int(hold_days)
if not_in_pool:
data["in_pool"] = False
if tmin is not None:
data["tmin"] = int(tmin)
if tmax is not None:
data["tmax"] = int(tmax)
if price is not None:
text += f";卖出价 {price:.2f} 元"
data["price"] = _num(price, 4)
if return_pct is not None:
data["return_pct"] = _num(return_pct, 4)
return TradeReason(code=code, text=text, data=data)
def sell_deferred(
code: str,
*,
cause: str,
rank: int | None = None,
total: int | None = None,
top_n: int | None = None,
score: float | None = None,
factors: dict[str, float] | None = None,
hold_days: int | None = None,
tmin: int | None = None,
tmax: int | None = None,
close: float | None = None,
prev_close: float | None = None,
limit_ratio: float | None = None,
not_in_pool: bool = False,
) -> TradeReason:
"""卖出未成交(顺延 / 暂留)的理由:Tmin 保护 / 停牌 / 跌停。
`tmax` 有值时说明是 Tmax 强制了结被卡住,文案据此区分 —— 两者后续行为不同
(Tmin 保护等到满 Tmin,Tmax 每天重试且不认排名)。
"""
if code not in REASON_CODES:
raise ValueError(f"未知的卖出顺延原因:{code}")
if tmax is not None:
head = f"持有 {hold_days} 个交易日超 Tmax={tmax},本应强制了结"
elif code == SELL_DEFER_TMIN:
why = (
"调仓日已不在候选池(被股票池/条件过滤)"
if not_in_pool
else f"掉出 TopN({_rank_text(rank, total, top_n, score)})"
)
head = f"{why}但仅持 {hold_days} 个交易日 < Tmin={tmin},按 Tmin 保护暂留"
else:
head = (
"调仓日已不在候选池(被股票池/条件过滤)"
if not_in_pool
else f"调仓日跌出 TopN({_rank_text(rank, total, top_n, score)})"
)
if cause == "tmin":
text = f"{head},暂留至满 Tmin"
elif cause == "halted":
text = f"{head},但当日无行情(停牌),顺延"
elif cause == "limit_down":
ratio = _num(close / prev_close, 4) if close is not None and prev_close not in (None, 0) else None
text = f"{head},但当日跌停"
if ratio is not None:
text += f"(收盘 {close:.2f} / 前收 {prev_close:.2f} = {ratio:.3f}"
text += f" ≤ 跌停阈值 {limit_ratio:.3f})" if limit_ratio else ")"
text += ",无法卖出,顺延"
if ratio is not None:
data_ratio = ratio
close_v, prev_v = _num(close, 4), _num(prev_close, 4)
else:
data_ratio, close_v, prev_v = None, None, None
else: # pragma: no cover - 调用方只传 halted / limit_down
text = f"{head},顺延"
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
if not_in_pool:
data["in_pool"] = False
if hold_days is not None:
data["hold_days"] = int(hold_days)
if tmin is not None:
data["tmin"] = int(tmin)
if tmax is not None:
data["tmax"] = int(tmax)
if cause == "limit_down":
if data_ratio is not None:
data["close_prev_ratio"] = data_ratio
data["close"] = close_v
data["prev_close"] = prev_v
if limit_ratio is not None:
data["limit_ratio"] = _num(limit_ratio, 4)
if close is not None and "close" not in data and cause == "halted":
data["close"] = _num(close, 4)
return TradeReason(code=code, text=text, data=data)
# ---------- 因子曲线(持仓加权平均原始值) ----------
def weighted_average(values: Mapping[str, float], weights: Mapping[str, float]) -> float | None:
"""权重加权平均;没有任何有效样本时返回 None(调用方据此不落点)。"""
total_w = 0.0
acc = 0.0
for symbol, w in weights.items():
v = values.get(symbol)
if v is None or w <= 0:
continue
acc += float(v) * float(w)
total_w += float(w)
if total_w <= 0:
return None
return acc / total_w
def build_factor_curves(
factor_panels: Mapping[str, tuple[FactorDef, pd.DataFrame]],
weights_by_day: Mapping[object, dict[str, float]],
) -> list[FactorCurve]:
"""按「每日持仓权重」聚合出每个因子的曲线。
- `weights_by_day`:`{交易日: {symbol: 该股市值}}`,空仓日给空字典(不落点);
- 值 = 该日持仓上该因子的权重加权平均**原始值**(不做 z-score、不按方向取负);
- 曲线按因子键排序,保证同一份数据每次归档的顺序一致(便于 diff)。
"""
out: list[FactorCurve] = []
for name in sorted(factor_panels):
defn, panel = factor_panels[name]
points: list[CurvePoint] = []
for day, weights in weights_by_day.items():
if not weights or day not in panel.index:
continue
row = panel.loc[day]
values = {s: _num(row.get(s)) for s in weights}
avg = weighted_average(values, weights)
if avg is None:
continue
points.append(CurvePoint(date=day.date() if hasattr(day, "date") else day, value=round(avg, 6)))
out.append(
FactorCurve(
name=name,
label=defn.display,
direction=defn.direction,
unit=defn.unit,
points=points,
)
)
return out