"""买卖理由(数据化)与因子曲线的共享构造器。 **为什么单独一个模块**:两套回测引擎(`combo_engine` 组合回测、`local_engine` 单策略回测)都要回答同一个问题 ——「这一买一卖,当时的数字是多少?」。如果各写一份, 措辞、口径、字段名迟早分叉,用户在两处看到的「理由」会互相矛盾。 因此这里只放两件事: 1. **封闭的原因词表 + 构造器**:每个 `code` 对应一类可判定的原因,`text` 里带关键数字, `data` 里放当时的原始数值(排名 / 候选数 / 综合分 / 因子原始值 / 持有交易日 / 预算 …)。 引擎只允许用词表里的 code(`REASON_CODES`),避免出现「文案随手写」的漂移。 2. **因子曲线的口径实现**:持仓股票的**权重加权平均原始值**,空仓日不落点、 不插值、不按方向取负(低为好的因子也原样画,方向由 `FactorCurve.direction` 说明)。 数值一律「引擎当时算出来的」:前端只展示,不推算 —— 界面上不该出现看起来像真的数字。 """ from __future__ import annotations import math from collections.abc import Mapping import pandas as pd from app.domain.entities.research import CurvePoint, FactorCurve, TradeReason from app.quant.factors import FactorDef # ---------- 封闭原因词表 ---------- # 买入(成交) BUY_ENTER = "buy_enter_topn" BUY_DEFER_FILLED = "buy_defer_filled" # 买入(未成交 / 未执行) BUY_SKIP_LIMIT_UP = "buy_skip_limit_up" BUY_SKIP_HALTED = "buy_skip_halted" BUY_SKIP_NO_CASH = "buy_skip_no_cash" BUY_SKIP_MIN_COMMISSION = "buy_skip_min_commission" # 卖出(成交) SELL_DROP_TOPN = "sell_drop_topn" SELL_FORCE_TMAX = "sell_force_tmax" # 卖出(成交):策略在调仓日**全量换仓**(先清仓再建仓),该股当时仍在 TopN 内。 # 为什么单列一个 code:单策略回测(TopK runner)的调仓语义就是「全清再买」, # 被卖出的股票很可能仍然排在前列 —— 这时说「跌出 TopN」与 data 里的 rank=1 自相矛盾, # 等于给用户一个假的解释。分开写才是如实描述。 SELL_REBALANCE_FULL = "sell_rebalance_full" # 卖出(顺延 / 未成交) SELL_DEFER_TMIN = "sell_defer_tmin" SELL_DEFER_HALTED = "sell_defer_halted" SELL_DEFER_LIMIT_DOWN = "sell_defer_limit_down" REASON_CODES = frozenset( { BUY_ENTER, BUY_DEFER_FILLED, BUY_SKIP_LIMIT_UP, BUY_SKIP_HALTED, BUY_SKIP_NO_CASH, BUY_SKIP_MIN_COMMISSION, SELL_DROP_TOPN, SELL_FORCE_TMAX, SELL_REBALANCE_FULL, SELL_DEFER_TMIN, SELL_DEFER_HALTED, SELL_DEFER_LIMIT_DOWN, } ) # 原因分类的中文短标签(前端筛选 / 表格上色用;改文案只改这里) REASON_LABELS: dict[str, str] = { BUY_ENTER: "按名次建仓", BUY_DEFER_FILLED: "顺延后成交", BUY_SKIP_LIMIT_UP: "涨停未买", BUY_SKIP_HALTED: "停牌未买", BUY_SKIP_NO_CASH: "现金不足", BUY_SKIP_MIN_COMMISSION: "不足最低佣金", SELL_DROP_TOPN: "跌出 TopN", SELL_FORCE_TMAX: "持有超 Tmax", SELL_REBALANCE_FULL: "调仓换仓卖出", SELL_DEFER_TMIN: "Tmin 保护暂留", SELL_DEFER_HALTED: "停牌未卖", SELL_DEFER_LIMIT_DOWN: "跌停未卖", } def _num(v, digits: int = 6): """把 numpy/pandas 数值安全地压成原生 float(NaN/inf 一律不带进理由里)。""" if v is None: return None try: f = float(v) except (TypeError, ValueError): return None if math.isnan(f) or math.isinf(f): return None return round(f, digits) def factor_values( factor_panels: Mapping[str, tuple[FactorDef, pd.DataFrame]], day, symbol: str, ) -> dict[str, float]: """该个股在 `day` 的各因子**原始值**(缺失因子不写进 data,不用 0 冒充)。 用交易日精确匹配:调仓日的打分与理由是同一份面板,所以这里取不到值就意味着 「该股当日无该因子值」,如实缺失比填 0 更可信。 """ out: dict[str, float] = {} for name, (_defn, panel) in factor_panels.items(): if day not in panel.index or symbol not in panel.columns: continue v = _num(panel.at[day, symbol]) if v is not None: out[name] = v return out def _rank_data( *, rank: int | None, total: int | None, top_n: int | None, score: float | None, factors: dict[str, float] | None, ) -> dict: data: dict = {} if rank is not None: data["rank"] = int(rank) if total is not None: data["total"] = int(total) if top_n is not None: data["top_n"] = int(top_n) if score is not None: data["score"] = _num(score) if factors: data["factors"] = factors return data def _rank_text(rank: int | None, total: int | None, top_n: int | None, score: float | None) -> str: if rank is None: return "调仓日综合分未给出名次" parts = [f"综合分第 {rank}"] if total: parts.append(f"/{total}") parts.append(" 名") if top_n is not None: parts.append(f"(TopN={top_n})") if score is not None: parts.append(f",综合分 {score:.4f}") return "".join(parts) def buy_filled( *, rank: int | None, total: int | None, top_n: int | None, score: float | None, factors: dict[str, float] | None, price: float | None, budget: float | None = None, deferred: bool = False, ) -> TradeReason: """买入成交的理由:名次 + 综合分 + 各因子当时的原始值 + 成交价。""" head = "顺延买入成交" if deferred else "调仓日选中并建仓" text = f"{head}:{_rank_text(rank, total, top_n, score)}" if price is not None: text += f";成交价 {price:.2f} 元" data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors) if price is not None: data["price"] = _num(price, 4) if budget is not None: data["budget"] = _num(budget, 2) return TradeReason(code=BUY_DEFER_FILLED if deferred else BUY_ENTER, text=text, data=data) def buy_skipped( code: str, *, rank: int | None = None, total: int | None = None, top_n: int | None = None, score: float | None = None, factors: dict[str, float] | None = None, close: float | None = None, prev_close: float | None = None, limit_ratio: float | None = None, budget: float | None = None, min_commission: float | None = None, not_in_pool: bool = False, ) -> TradeReason: """买入未成交 / 未执行的理由(涨停、停牌、现金不足、佣金门槛)。""" if code not in REASON_CODES: raise ValueError(f"未知的买入未成交原因:{code}") data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors) base = _rank_text(rank, total, top_n, score) if code == BUY_SKIP_LIMIT_UP: ratio = ( _num(close / prev_close, 4) if close is not None and prev_close not in (None, 0) else None ) text = f"{base},但当日涨停" if ratio is not None: text += f"(收盘 {close:.2f} / 前收 {prev_close:.2f} = {ratio:.3f}" text += f" ≥ 涨停阈值 {limit_ratio:.3f})" if limit_ratio else ")" text += ",无法追买" if ratio is not None: data["close_prev_ratio"] = ratio if limit_ratio is not None: data["limit_ratio"] = _num(limit_ratio, 4) elif code == BUY_SKIP_HALTED: text = f"{base},但当日无行情(停牌),无法买入" elif code == BUY_SKIP_NO_CASH: text = f"{base},但可用现金不足,未成交" if budget is not None: text += f"(可用预算 {budget:.2f} 元)" elif code == BUY_SKIP_MIN_COMMISSION: text = f"{base},但预算不足以覆盖最低佣金,未成交" if budget is not None and min_commission is not None: text += f"(预算 {budget:.2f} 元 < 最低佣金 {min_commission:.2f} 元)" else: # pragma: no cover - 上面的分支已覆盖全部代码 text = base if close is not None: data["close"] = _num(close, 4) if prev_close is not None: data["prev_close"] = _num(prev_close, 4) if budget is not None: data["budget"] = _num(budget, 2) if min_commission is not None: data["min_commission"] = _num(min_commission, 2) if not_in_pool: data["in_pool"] = False return TradeReason(code=code, text=text, data=data) def sell_filled( *, code: str, rank: int | None, total: int | None, top_n: int | None, score: float | None, factors: dict[str, float] | None, hold_days: int, tmin: int | None = None, tmax: int | None = None, price: float | None = None, return_pct: float | None = None, not_in_pool: bool = False, ) -> TradeReason: """卖出成交的理由(跌出 TopN / 持有超 Tmax / 全量换仓),带持有交易日与当时名次。 `not_in_pool=True` 表示该股已**不在候选池**(被股票池/条件过滤,如转为 ST), 与「在池内但排名掉出去」是两回事,文案与 data 都分开写。 `SELL_REBALANCE_FULL` 用于「策略每次调仓都先全清再建仓」的引擎:该股当时仍在前列, 卖它不是因为掉出 TopN,而是策略本身的调仓方式 —— 不能套用跌出 TopN 的说法。 """ if code == SELL_FORCE_TMAX: text = f"持有 {hold_days} 个交易日 > Tmax={tmax},强制了结(与排名无关)" elif code == SELL_REBALANCE_FULL: text = ( f"调仓日全量换仓:该策略每次调仓先清仓再按新名单建仓" f"(该股当时仍在 TopN 内:{_rank_text(rank, total, top_n, score)});" f"持有 {hold_days} 个交易日" ) if tmin is not None: text += f" ≥ Tmin={tmin}" else: code = SELL_DROP_TOPN if not_in_pool: text = f"调仓日已不在候选池(被股票池/条件过滤);持有 {hold_days} 个交易日" else: text = f"调仓日跌出 TopN:{_rank_text(rank, total, top_n, score)};持有 {hold_days} 个交易日" if tmin is not None: text += f" ≥ Tmin={tmin}" data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors) data["hold_days"] = int(hold_days) if not_in_pool: data["in_pool"] = False if tmin is not None: data["tmin"] = int(tmin) if tmax is not None: data["tmax"] = int(tmax) if price is not None: text += f";卖出价 {price:.2f} 元" data["price"] = _num(price, 4) if return_pct is not None: data["return_pct"] = _num(return_pct, 4) return TradeReason(code=code, text=text, data=data) def sell_deferred( code: str, *, cause: str, rank: int | None = None, total: int | None = None, top_n: int | None = None, score: float | None = None, factors: dict[str, float] | None = None, hold_days: int | None = None, tmin: int | None = None, tmax: int | None = None, close: float | None = None, prev_close: float | None = None, limit_ratio: float | None = None, not_in_pool: bool = False, ) -> TradeReason: """卖出未成交(顺延 / 暂留)的理由:Tmin 保护 / 停牌 / 跌停。 `tmax` 有值时说明是 Tmax 强制了结被卡住,文案据此区分 —— 两者后续行为不同 (Tmin 保护等到满 Tmin,Tmax 每天重试且不认排名)。 """ if code not in REASON_CODES: raise ValueError(f"未知的卖出顺延原因:{code}") if tmax is not None: head = f"持有 {hold_days} 个交易日超 Tmax={tmax},本应强制了结" elif code == SELL_DEFER_TMIN: why = ( "调仓日已不在候选池(被股票池/条件过滤)" if not_in_pool else f"掉出 TopN({_rank_text(rank, total, top_n, score)})" ) head = f"{why}但仅持 {hold_days} 个交易日 < Tmin={tmin},按 Tmin 保护暂留" else: head = ( "调仓日已不在候选池(被股票池/条件过滤)" if not_in_pool else f"调仓日跌出 TopN({_rank_text(rank, total, top_n, score)})" ) if cause == "tmin": text = f"{head},暂留至满 Tmin" elif cause == "halted": text = f"{head},但当日无行情(停牌),顺延" elif cause == "limit_down": ratio = _num(close / prev_close, 4) if close is not None and prev_close not in (None, 0) else None text = f"{head},但当日跌停" if ratio is not None: text += f"(收盘 {close:.2f} / 前收 {prev_close:.2f} = {ratio:.3f}" text += f" ≤ 跌停阈值 {limit_ratio:.3f})" if limit_ratio else ")" text += ",无法卖出,顺延" if ratio is not None: data_ratio = ratio close_v, prev_v = _num(close, 4), _num(prev_close, 4) else: data_ratio, close_v, prev_v = None, None, None else: # pragma: no cover - 调用方只传 halted / limit_down text = f"{head},顺延" data = _rank_data(rank=rank, total=total, top_n=top_n, score=score, factors=factors) if not_in_pool: data["in_pool"] = False if hold_days is not None: data["hold_days"] = int(hold_days) if tmin is not None: data["tmin"] = int(tmin) if tmax is not None: data["tmax"] = int(tmax) if cause == "limit_down": if data_ratio is not None: data["close_prev_ratio"] = data_ratio data["close"] = close_v data["prev_close"] = prev_v if limit_ratio is not None: data["limit_ratio"] = _num(limit_ratio, 4) if close is not None and "close" not in data and cause == "halted": data["close"] = _num(close, 4) return TradeReason(code=code, text=text, data=data) # ---------- 因子曲线(持仓加权平均原始值) ---------- def weighted_average(values: Mapping[str, float], weights: Mapping[str, float]) -> float | None: """权重加权平均;没有任何有效样本时返回 None(调用方据此不落点)。""" total_w = 0.0 acc = 0.0 for symbol, w in weights.items(): v = values.get(symbol) if v is None or w <= 0: continue acc += float(v) * float(w) total_w += float(w) if total_w <= 0: return None return acc / total_w def build_factor_curves( factor_panels: Mapping[str, tuple[FactorDef, pd.DataFrame]], weights_by_day: Mapping[object, dict[str, float]], ) -> list[FactorCurve]: """按「每日持仓权重」聚合出每个因子的曲线。 - `weights_by_day`:`{交易日: {symbol: 该股市值}}`,空仓日给空字典(不落点); - 值 = 该日持仓上该因子的权重加权平均**原始值**(不做 z-score、不按方向取负); - 曲线按因子键排序,保证同一份数据每次归档的顺序一致(便于 diff)。 """ out: list[FactorCurve] = [] for name in sorted(factor_panels): defn, panel = factor_panels[name] points: list[CurvePoint] = [] for day, weights in weights_by_day.items(): if not weights or day not in panel.index: continue row = panel.loc[day] values = {s: _num(row.get(s)) for s in weights} avg = weighted_average(values, weights) if avg is None: continue points.append(CurvePoint(date=day.date() if hasattr(day, "date") else day, value=round(avg, 6))) out.append( FactorCurve( name=name, label=defn.display, direction=defn.direction, unit=defn.unit, points=points, ) ) return out