Files
qlib/backend/app/quant/trade_reasons.py
T
Simon 7e369d9680 feat(quant): 单策略引擎也给出买卖理由与因子曲线(口径与组合引擎一致)
「因子组合」/`POST /api/research/backtests` 走的是 `LocalEngine/TopKBacktestRunner`,
上一版只把理由接进了组合引擎,同一件事在两个引擎上就会有两种说法。这次补齐:

- `engine.py` / `qlib_adapter/engine.py`:因子面板**只算一次**
  (`build_factor_panels_full`)→ 复合分与「理由里引用的因子原始值」同源同张面板;
  复合分口径逐字未变(与 `selection.score_panel_for_factors` 相同)。
- `local_engine.py`:调仓日保留完整排名与合格集,各站点写入结构化理由 ——
  买入(按名次建仓 / 顺延成交 / 涨停 / 停牌 / 现金不足 / 不足最低佣金)、
  卖出(全量换仓 / 跌出 TopN / 不在候选池 / 停牌顺延 / 跌停顺延);
  `Trade.entry_reason/exit_reason` 两端齐全;每个交易日记录持仓市值,
  结果填 `factor_curves`(持仓市值加权平均的因子原始值,空仓日不落点)。
- 新增 `SELL_REBALANCE_FULL`(「调仓换仓卖出」):单策略调仓是「先全清再建仓」,
  被卖出的股票**可能仍排在 TopN 内**(如 rank=1),这时写「跌出 TopN」就是假解释;
  按事实分 code(仍在 TopN 内 → 全量换仓;否则 → 跌出 TopN / 不在候选池)。
- 顺延成交不拿挂单日的旧名次冒充当日名次(rank/total/score=None,因子值/成交价/预算
  取成交当日真实值);「候选池不足」的提示记录保持 reason=None(词表里没有对应语义,
  硬套就是编理由)。

验证:
- 新增 `tests/test_local_engine_reasons.py` 14 条:理由数字对回面板、涨停比值对回行情与
  板块规则、停牌/跌停/现金不足/最低佣金、顺延成交、全量换仓 vs 不在池两个分支、
  Trade 两端理由、因子曲线市值加权(手算加权值断言 + 等权平均对不上)、空仓不落点。
  后端 524 条全过(510 + 14),ruff clean。
- 强回归:用改前引擎并排跑 9 个场景,`signal_history`(日期/方向/成交/原因文案/价格)、
  `trades`、`positions`、`summary`、净值/回撤、`unimplemented` 逐条一致 —— 理由与曲线
  是纯新增字段,成交行为零变化。
2026-10-01 18:08:42 +08:00

422 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""买卖理由(数据化)与因子曲线的共享构造器。
**为什么单独一个模块**:两套回测引擎(`combo_engine` 组合回测、`local_engine`
单策略回测)都要回答同一个问题 ——「这一买一卖,当时的数字是多少?」。如果各写一份,
措辞、口径、字段名迟早分叉,用户在两处看到的「理由」会互相矛盾。
因此这里只放两件事:
1. **封闭的原因词表 + 构造器**:每个 `code` 对应一类可判定的原因,`text` 里带关键数字,
`data` 里放当时的原始数值(排名 / 候选数 / 综合分 / 因子原始值 / 持有交易日 / 预算 …)。
引擎只允许用词表里的 code(`REASON_CODES`),避免出现「文案随手写」的漂移。
2. **因子曲线的口径实现**:持仓股票的**权重加权平均原始值**,空仓日不落点、
不插值、不按方向取负(低为好的因子也原样画,方向由 `FactorCurve.direction` 说明)。
数值一律「引擎当时算出来的」:前端只展示,不推算 —— 界面上不该出现看起来像真的数字。
"""
from __future__ import annotations
import math
from collections.abc import Mapping
import pandas as pd
from app.domain.entities.research import CurvePoint, FactorCurve, TradeReason
from app.quant.factors import FactorDef
# ---------- 封闭原因词表 ----------
# 买入(成交)
BUY_ENTER = "buy_enter_topn"
BUY_DEFER_FILLED = "buy_defer_filled"
# 买入(未成交 / 未执行)
BUY_SKIP_LIMIT_UP = "buy_skip_limit_up"
BUY_SKIP_HALTED = "buy_skip_halted"
BUY_SKIP_NO_CASH = "buy_skip_no_cash"
BUY_SKIP_MIN_COMMISSION = "buy_skip_min_commission"
# 卖出(成交)
SELL_DROP_TOPN = "sell_drop_topn"
SELL_FORCE_TMAX = "sell_force_tmax"
# 卖出(成交):策略在调仓日**全量换仓**(先清仓再建仓),该股当时仍在 TopN 内。
# 为什么单列一个 code:单策略回测(TopK runner)的调仓语义就是「全清再买」,
# 被卖出的股票很可能仍然排在前列 —— 这时说「跌出 TopN」与 data 里的 rank=1 自相矛盾,
# 等于给用户一个假的解释。分开写才是如实描述。
SELL_REBALANCE_FULL = "sell_rebalance_full"
# 卖出(顺延 / 未成交)
SELL_DEFER_TMIN = "sell_defer_tmin"
SELL_DEFER_HALTED = "sell_defer_halted"
SELL_DEFER_LIMIT_DOWN = "sell_defer_limit_down"
REASON_CODES = frozenset(
{
BUY_ENTER,
BUY_DEFER_FILLED,
BUY_SKIP_LIMIT_UP,
BUY_SKIP_HALTED,
BUY_SKIP_NO_CASH,
BUY_SKIP_MIN_COMMISSION,
SELL_DROP_TOPN,
SELL_FORCE_TMAX,
SELL_REBALANCE_FULL,
SELL_DEFER_TMIN,
SELL_DEFER_HALTED,
SELL_DEFER_LIMIT_DOWN,
}
)
# 原因分类的中文短标签(前端筛选 / 表格上色用;改文案只改这里)
REASON_LABELS: dict[str, str] = {
BUY_ENTER: "按名次建仓",
BUY_DEFER_FILLED: "顺延后成交",
BUY_SKIP_LIMIT_UP: "涨停未买",
BUY_SKIP_HALTED: "停牌未买",
BUY_SKIP_NO_CASH: "现金不足",
BUY_SKIP_MIN_COMMISSION: "不足最低佣金",
SELL_DROP_TOPN: "跌出 TopN",
SELL_FORCE_TMAX: "持有超 Tmax",
SELL_REBALANCE_FULL: "调仓换仓卖出",
SELL_DEFER_TMIN: "Tmin 保护暂留",
SELL_DEFER_HALTED: "停牌未卖",
SELL_DEFER_LIMIT_DOWN: "跌停未卖",
}
def _num(v, digits: int = 6):
"""把 numpy/pandas 数值安全地压成原生 float(NaN/inf 一律不带进理由里)。"""
if v is None:
return None
try:
f = float(v)
except (TypeError, ValueError):
return None
if math.isnan(f) or math.isinf(f):
return None
return round(f, digits)
def factor_values(
factor_panels: Mapping[str, tuple[FactorDef, pd.DataFrame]],
day,
symbol: str,
) -> dict[str, float]:
"""该个股在 `day` 的各因子**原始值**(缺失因子不写进 data,不用 0 冒充)。
用交易日精确匹配:调仓日的打分与理由是同一份面板,所以这里取不到值就意味着
「该股当日无该因子值」,如实缺失比填 0 更可信。
"""
out: dict[str, float] = {}
for name, (_defn, panel) in factor_panels.items():
if day not in panel.index or symbol not in panel.columns:
continue
v = _num(panel.at[day, symbol])
if v is not None:
out[name] = v
return out
def _rank_data(
*,
rank: int | None,
total: int | None,
top_n: int | None,
score: float | None,
factors: dict[str, float] | None,
) -> dict:
data: dict = {}
if rank is not None:
data["rank"] = int(rank)
if total is not None:
data["total"] = int(total)
if top_n is not None:
data["top_n"] = int(top_n)
if score is not None:
data["score"] = _num(score)
if factors:
data["factors"] = factors
return data
def _rank_text(rank: int | None, total: int | None, top_n: int | None, score: float | None) -> str:
if rank is None:
return "调仓日综合分未给出名次"
parts = [f"综合分第 {rank}"]
if total:
parts.append(f"/{total}")
parts.append(" 名")
if top_n is not None:
parts.append(f"(TopN={top_n})")
if score is not None:
parts.append(f",综合分 {score:.4f}")
return "".join(parts)
def buy_filled(
*,
rank: int | None,
total: int | None,
top_n: int | None,
score: float | None,
factors: dict[str, float] | None,
price: float | None,
budget: float | None = None,
deferred: bool = False,
) -> TradeReason:
"""买入成交的理由:名次 + 综合分 + 各因子当时的原始值 + 成交价。"""
head = "顺延买入成交" if deferred else "调仓日选中并建仓"
text = f"{head}:{_rank_text(rank, total, top_n, score)}"
if price is not None:
text += f";成交价 {price:.2f} 元"
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
if price is not None:
data["price"] = _num(price, 4)
if budget is not None:
data["budget"] = _num(budget, 2)
return TradeReason(code=BUY_DEFER_FILLED if deferred else BUY_ENTER, text=text, data=data)
def buy_skipped(
code: str,
*,
rank: int | None = None,
total: int | None = None,
top_n: int | None = None,
score: float | None = None,
factors: dict[str, float] | None = None,
close: float | None = None,
prev_close: float | None = None,
limit_ratio: float | None = None,
budget: float | None = None,
min_commission: float | None = None,
not_in_pool: bool = False,
) -> TradeReason:
"""买入未成交 / 未执行的理由(涨停、停牌、现金不足、佣金门槛)。"""
if code not in REASON_CODES:
raise ValueError(f"未知的买入未成交原因:{code}")
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
base = _rank_text(rank, total, top_n, score)
if code == BUY_SKIP_LIMIT_UP:
ratio = (
_num(close / prev_close, 4)
if close is not None and prev_close not in (None, 0)
else None
)
text = f"{base},但当日涨停"
if ratio is not None:
text += f"(收盘 {close:.2f} / 前收 {prev_close:.2f} = {ratio:.3f}"
text += f" ≥ 涨停阈值 {limit_ratio:.3f})" if limit_ratio else ")"
text += ",无法追买"
if ratio is not None:
data["close_prev_ratio"] = ratio
if limit_ratio is not None:
data["limit_ratio"] = _num(limit_ratio, 4)
elif code == BUY_SKIP_HALTED:
text = f"{base},但当日无行情(停牌),无法买入"
elif code == BUY_SKIP_NO_CASH:
text = f"{base},但可用现金不足,未成交"
if budget is not None:
text += f"(可用预算 {budget:.2f} 元)"
elif code == BUY_SKIP_MIN_COMMISSION:
text = f"{base},但预算不足以覆盖最低佣金,未成交"
if budget is not None and min_commission is not None:
text += f"(预算 {budget:.2f} 元 < 最低佣金 {min_commission:.2f} 元)"
else: # pragma: no cover - 上面的分支已覆盖全部代码
text = base
if close is not None:
data["close"] = _num(close, 4)
if prev_close is not None:
data["prev_close"] = _num(prev_close, 4)
if budget is not None:
data["budget"] = _num(budget, 2)
if min_commission is not None:
data["min_commission"] = _num(min_commission, 2)
if not_in_pool:
data["in_pool"] = False
return TradeReason(code=code, text=text, data=data)
def sell_filled(
*,
code: str,
rank: int | None,
total: int | None,
top_n: int | None,
score: float | None,
factors: dict[str, float] | None,
hold_days: int,
tmin: int | None = None,
tmax: int | None = None,
price: float | None = None,
return_pct: float | None = None,
not_in_pool: bool = False,
) -> TradeReason:
"""卖出成交的理由(跌出 TopN / 持有超 Tmax / 全量换仓),带持有交易日与当时名次。
`not_in_pool=True` 表示该股已**不在候选池**(被股票池/条件过滤,如转为 ST),
与「在池内但排名掉出去」是两回事,文案与 data 都分开写。
`SELL_REBALANCE_FULL` 用于「策略每次调仓都先全清再建仓」的引擎:该股当时仍在前列,
卖它不是因为掉出 TopN,而是策略本身的调仓方式 —— 不能套用跌出 TopN 的说法。
"""
if code == SELL_FORCE_TMAX:
text = f"持有 {hold_days} 个交易日 > Tmax={tmax},强制了结(与排名无关)"
elif code == SELL_REBALANCE_FULL:
text = (
f"调仓日全量换仓:该策略每次调仓先清仓再按新名单建仓"
f"(该股当时仍在 TopN 内:{_rank_text(rank, total, top_n, score)});"
f"持有 {hold_days} 个交易日"
)
if tmin is not None:
text += f" ≥ Tmin={tmin}"
else:
code = SELL_DROP_TOPN
if not_in_pool:
text = f"调仓日已不在候选池(被股票池/条件过滤);持有 {hold_days} 个交易日"
else:
text = f"调仓日跌出 TopN:{_rank_text(rank, total, top_n, score)};持有 {hold_days} 个交易日"
if tmin is not None:
text += f" ≥ Tmin={tmin}"
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
data["hold_days"] = int(hold_days)
if not_in_pool:
data["in_pool"] = False
if tmin is not None:
data["tmin"] = int(tmin)
if tmax is not None:
data["tmax"] = int(tmax)
if price is not None:
text += f";卖出价 {price:.2f} 元"
data["price"] = _num(price, 4)
if return_pct is not None:
data["return_pct"] = _num(return_pct, 4)
return TradeReason(code=code, text=text, data=data)
def sell_deferred(
code: str,
*,
cause: str,
rank: int | None = None,
total: int | None = None,
top_n: int | None = None,
score: float | None = None,
factors: dict[str, float] | None = None,
hold_days: int | None = None,
tmin: int | None = None,
tmax: int | None = None,
close: float | None = None,
prev_close: float | None = None,
limit_ratio: float | None = None,
not_in_pool: bool = False,
) -> TradeReason:
"""卖出未成交(顺延 / 暂留)的理由:Tmin 保护 / 停牌 / 跌停。
`tmax` 有值时说明是 Tmax 强制了结被卡住,文案据此区分 —— 两者后续行为不同
(Tmin 保护等到满 Tmin,Tmax 每天重试且不认排名)。
"""
if code not in REASON_CODES:
raise ValueError(f"未知的卖出顺延原因:{code}")
if tmax is not None:
head = f"持有 {hold_days} 个交易日超 Tmax={tmax},本应强制了结"
elif code == SELL_DEFER_TMIN:
why = (
"调仓日已不在候选池(被股票池/条件过滤)"
if not_in_pool
else f"掉出 TopN({_rank_text(rank, total, top_n, score)})"
)
head = f"{why}但仅持 {hold_days} 个交易日 < Tmin={tmin},按 Tmin 保护暂留"
else:
head = (
"调仓日已不在候选池(被股票池/条件过滤)"
if not_in_pool
else f"调仓日跌出 TopN({_rank_text(rank, total, top_n, score)})"
)
if cause == "tmin":
text = f"{head},暂留至满 Tmin"
elif cause == "halted":
text = f"{head},但当日无行情(停牌),顺延"
elif cause == "limit_down":
ratio = _num(close / prev_close, 4) if close is not None and prev_close not in (None, 0) else None
text = f"{head},但当日跌停"
if ratio is not None:
text += f"(收盘 {close:.2f} / 前收 {prev_close:.2f} = {ratio:.3f}"
text += f" ≤ 跌停阈值 {limit_ratio:.3f})" if limit_ratio else ")"
text += ",无法卖出,顺延"
if ratio is not None:
data_ratio = ratio
close_v, prev_v = _num(close, 4), _num(prev_close, 4)
else:
data_ratio, close_v, prev_v = None, None, None
else: # pragma: no cover - 调用方只传 halted / limit_down
text = f"{head},顺延"
data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors)
if not_in_pool:
data["in_pool"] = False
if hold_days is not None:
data["hold_days"] = int(hold_days)
if tmin is not None:
data["tmin"] = int(tmin)
if tmax is not None:
data["tmax"] = int(tmax)
if cause == "limit_down":
if data_ratio is not None:
data["close_prev_ratio"] = data_ratio
data["close"] = close_v
data["prev_close"] = prev_v
if limit_ratio is not None:
data["limit_ratio"] = _num(limit_ratio, 4)
if close is not None and "close" not in data and cause == "halted":
data["close"] = _num(close, 4)
return TradeReason(code=code, text=text, data=data)
# ---------- 因子曲线(持仓加权平均原始值) ----------
def weighted_average(values: Mapping[str, float], weights: Mapping[str, float]) -> float | None:
"""权重加权平均;没有任何有效样本时返回 None(调用方据此不落点)。"""
total_w = 0.0
acc = 0.0
for symbol, w in weights.items():
v = values.get(symbol)
if v is None or w <= 0:
continue
acc += float(v) * float(w)
total_w += float(w)
if total_w <= 0:
return None
return acc / total_w
def build_factor_curves(
factor_panels: Mapping[str, tuple[FactorDef, pd.DataFrame]],
weights_by_day: Mapping[object, dict[str, float]],
) -> list[FactorCurve]:
"""按「每日持仓权重」聚合出每个因子的曲线。
- `weights_by_day`:`{交易日: {symbol: 该股市值}}`,空仓日给空字典(不落点);
- 值 = 该日持仓上该因子的权重加权平均**原始值**(不做 z-score、不按方向取负);
- 曲线按因子键排序,保证同一份数据每次归档的顺序一致(便于 diff)。
"""
out: list[FactorCurve] = []
for name in sorted(factor_panels):
defn, panel = factor_panels[name]
points: list[CurvePoint] = []
for day, weights in weights_by_day.items():
if not weights or day not in panel.index:
continue
row = panel.loc[day]
values = {s: _num(row.get(s)) for s in weights}
avg = weighted_average(values, weights)
if avg is None:
continue
points.append(CurvePoint(date=day.date() if hasattr(day, "date") else day, value=round(avg, 6)))
out.append(
FactorCurve(
name=name,
label=defn.display,
direction=defn.direction,
unit=defn.unit,
points=points,
)
)
return out