feat(backend): 字段库(condition_field)+ 因子参数化(模板/受控参数)+ 单位换算底座
字段库(本次新增的表与接口): - `condition_field` 表 + `/api/condition-fields`:中文名/说明可编辑、可停用; `kind`/单位阶梯/`base_unit` 由代码注册表收敛(改类型 422,伪字段 422, 越界单位 422),停用的字段不再进条件下拉,但既有策略仍按名字解析。 - 说明书里的数值条件按字段注册表补**基准单位**后缀(字段间比较不加,不猜单位)。 因子参数化(键即身份,冻结口径): - 模板 + 参数注册表(`quant/factors.py`):`ParamSpec`(类型/范围/枚举/默认值/说明)+ `FactorTemplate`(公式/依赖列/参数);规范键把**全部**参数写进名字,如 `momentum(window=90,direction=lower_is_better)`,所以改参数 = 新建一个身份, 旧因子/既有策略/已归档实验都不变义;`momentum(window=90)`(缺参数)明确拒绝 —— 缺项要靠模板默认值补齐,而默认值是可改的代码细节,一旦改动会追溯性改义。 - 参数只在受控范围内取值(窗口 2~500、方向二选一),越界/未知模板/多给参数一律 422 并列出允许范围,不静默截断、不悄悄取默认值;内置实例的启用开关由代码决定(422)。 - `/api/factors` 暴露 `template`/`params`/`param_specs`/`label`/`source`/`enabled`/ `resolvable`;新增 `/api/factors/templates`、`POST /api/factors`、`PATCH /api/factors`; `get_factor = resolve_factor` 兼容全部旧调用点,参数化键也是一等条件字段。 - 迁移链:c5d6(存量策略陈旧说明重算)→ d6e7(condition_field)→ a7c1 (factor_definition.enabled + name varchar(128))。 测试:新增 test_condition_fields.py / test_factor_params.py;全量 pytest 500 passed。
This commit is contained in:
@@ -0,0 +1,375 @@
|
||||
"""条件字段注册表 —— 「字段库」的唯一事实来源(2026-10)。
|
||||
|
||||
要解决的问题
|
||||
------------
|
||||
策略库的「过滤条件」此前是**手填字段名**的输入框:用户必须知道 dv_ratio /
|
||||
static.industry / fundamental.roe 这类内部标识,既看不到含义,写错了也不报错 ——
|
||||
引擎对未知字段求值一律返回 None,条件**永远不通过**,策略会安静地选出 0 只股票。
|
||||
这正是 AGENT.md 禁止的「静默失败 / 假装支持」。
|
||||
|
||||
本模块的职责
|
||||
------------
|
||||
把**引擎真正支持的字段域**集中声明一次(中文名 + 含义 + 单位 + 分组 + 类型 + 排序),
|
||||
供三方共用:
|
||||
|
||||
1. ``/api/condition-fields`` 据此 seed 目录、据此校验用户新增的自定义字段;
|
||||
2. 前端据此渲染分组下拉、含义提示、并按类型收窄可选比较符;
|
||||
3. :func:`is_supported_field` 直接查 ``Stock`` / ``FinancialIndicator`` 的字段定义与
|
||||
因子注册表(``quant.factors``),**不另写一套近似规则** —— 避免注册表与引擎漂移。
|
||||
|
||||
诚实性约束(AGENT.md §24)
|
||||
--------------------------
|
||||
只登记真能算的字段。日期字段(如 ``static.list_date``)无法比较大小,:func:`reason_unsupported`
|
||||
会明确拒绝并说明理由,而不是放行让用户建出一条「永远选不出股票」的条件。
|
||||
单位一律照抄数据源落库口径(见 ``data_sources/tushare.py`` 的换算注释),不凭印象写。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from app.domain.entities.condition_field import OPS_BY_KIND
|
||||
from app.domain.entities.market import (
|
||||
DAILY_BAR_NUMERIC_FIELDS,
|
||||
DAILY_BASIC_NUMERIC_FIELDS,
|
||||
FinancialIndicator,
|
||||
Stock,
|
||||
)
|
||||
from app.quant.factors import FactorError, get_factor, list_factors, resolve_factor
|
||||
|
||||
# ---------- 分组(下拉的 optgroup 顺序即此顺序) ----------
|
||||
|
||||
GROUP_STOCK = "股票基础"
|
||||
GROUP_QUOTE = "行情"
|
||||
GROUP_TECH = "技术指标"
|
||||
GROUP_DAILY = "每日指标"
|
||||
GROUP_FUNDAMENTAL = "财务指标"
|
||||
GROUP_FACTOR = "因子"
|
||||
|
||||
GROUP_ORDER: tuple[str, ...] = (
|
||||
GROUP_STOCK,
|
||||
GROUP_QUOTE,
|
||||
GROUP_TECH,
|
||||
GROUP_DAILY,
|
||||
GROUP_FUNDAMENTAL,
|
||||
GROUP_FACTOR,
|
||||
)
|
||||
|
||||
# 比较符规则(哪种类型能比大小)定义在领域实体 domain/entities/condition_field.py,
|
||||
# 此处转发 —— 保证「字段库 API 返回的 ops」与「注册表里 FieldDef.ops」出自同一处。
|
||||
|
||||
# ---------- 单位阶梯(2026-10) ----------
|
||||
#
|
||||
# 单位分两层,避免「改个显示单位把历史策略的数值偷偷换义」:
|
||||
# · **基准单位**(FieldDef.unit):引擎存储与比较用的单位,写死在数据源落库口径里,
|
||||
# 不可改。归档里的 ConditionSpec 存的永远是基准单位值 —— 复现不受界面设置影响。
|
||||
# · **界面单位**(本阶梯里的备选项):只在「输入/显示」这一层做换算,factor 表示
|
||||
# 「该单位 → 基准单位的系数」(即提交前 ×factor,回显时 ÷factor)。
|
||||
# 所以用户在字段库把总市值选成「亿元」,输入 5 会存成 50000(万元)——引擎比较的仍是
|
||||
# 基准单位,而界面上看到的始终是 5 亿元。改单位不会让任何历史策略变义。
|
||||
#
|
||||
# 只登记换算无歧义、且实际会用到的单位:金额(元/万元/亿元)、股数(股/手/万手)、
|
||||
# 股本(万股/亿股)。百分数(%)与倍数(倍)不提供备选 —— 换成小数只会制造误读。
|
||||
|
||||
U_MONEY_YUAN: tuple[tuple[str, float], ...] = (("元", 1.0), ("万元", 1e4), ("亿元", 1e8))
|
||||
U_MONEY_WAN: tuple[tuple[str, float], ...] = (("万元", 1.0), ("亿元", 1e4))
|
||||
U_SHARE_GU: tuple[tuple[str, float], ...] = (("股", 1.0), ("手", 100.0), ("万手", 1e6))
|
||||
U_SHARE_WAN: tuple[tuple[str, float], ...] = (("万股", 1.0), ("亿股", 1e4))
|
||||
|
||||
# 哪些字段提供备选界面单位(键 = 引擎字段名;不在表里的字段只能用基准单位)
|
||||
_UNIT_LADDERS: dict[str, tuple[tuple[str, float], ...]] = {
|
||||
# 行情原列:volume 入库为股(源为手 ×100),amount 入库为元(源为千元 ×1000)
|
||||
"volume": U_SHARE_GU,
|
||||
"amount": U_MONEY_YUAN,
|
||||
# 每日指标:市值为万元,股本为万股
|
||||
"total_mv": U_MONEY_WAN,
|
||||
"circ_mv": U_MONEY_WAN,
|
||||
"total_share": U_SHARE_WAN,
|
||||
"float_share": U_SHARE_WAN,
|
||||
"free_share": U_SHARE_WAN,
|
||||
# 财务指标:金额入库为元
|
||||
"fundamental.net_profit": U_MONEY_YUAN,
|
||||
"fundamental.total_revenue": U_MONEY_YUAN,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FieldDef:
|
||||
"""一个条件字段的登记项(引擎真能算的字段)。"""
|
||||
|
||||
name: str # 引擎字段名(写进 condition.field)
|
||||
label: str # 中文名(下拉里给人看的)
|
||||
description: str # 含义 / 口径(含单位),必须可核对
|
||||
kind: str # num | str
|
||||
group_name: str
|
||||
unit: str = "" # **基准单位**(引擎存储/比较用),不可由界面更改
|
||||
curated: bool = True # True=默认进字段库;False=仅登记为「可新增」(少用字段)
|
||||
units: tuple[tuple[str, float], ...] = () # 可选界面单位;(单位, →基准单位系数),首项须是基准单位
|
||||
|
||||
@property
|
||||
def ops(self) -> tuple[str, ...]:
|
||||
return OPS_BY_KIND.get(self.kind, ())
|
||||
|
||||
@property
|
||||
def unit_options(self) -> tuple[tuple[str, float], ...]:
|
||||
"""可选界面单位;未登记阶梯的字段只有基准单位一项(界面不给选择)。"""
|
||||
return self.units or ((self.unit, 1.0),)
|
||||
|
||||
|
||||
# ---------- 股票基础(static.*):来自 Stock 实体,只有字符串字段可比较 ----------
|
||||
|
||||
# (attr, 中文名, 含义, curated)
|
||||
_STATIC_FIELDS: tuple[tuple[str, str, str, bool], ...] = (
|
||||
("industry", "所属行业", "股票基础信息里的行业名称(字符串),如「银行」「白酒」。等值用「=」,多值用「属于」。", True),
|
||||
("market", "上市板块", "主板 / 创业板 / 科创板 / 北交所(字符串)。", True),
|
||||
("area", "注册地域", "公司注册地省份或地区(字符串),如「广东」「北京」。", True),
|
||||
("exchange", "交易所", "SH 上交所 / SZ 深交所 / BJ 北交所(字符串)。", False),
|
||||
("status", "上市状态", "L 上市 / D 退市 / P 暂停上市(字符串)。研究池已默认剔除退市股。", False),
|
||||
("name", "股票名称", "证券简称(字符串)。一般用于核对,不建议拿来做条件。", False),
|
||||
("symbol", "股票代码", "Tushare 风格代码,如 600519.SH(字符串)。多值用「属于」。", False),
|
||||
)
|
||||
|
||||
# ---------- 行情原列(open/high/low/close/volume/amount) ----------
|
||||
# 换算口径见 data_sources/tushare.py: volume=vol(手)*100 → 股;amount=amount(千元)*1000 → 元。
|
||||
|
||||
_BAR_FIELDS: tuple[tuple[str, str, str, str, bool], ...] = (
|
||||
("close", "收盘价", "当日收盘价(元)。复权口径由公共配置的 price_adjustment 决定(默认 hfq)。", "元", True),
|
||||
("open", "开盘价", "当日开盘价(元),复权口径同上。", "元", True),
|
||||
("high", "最高价", "当日最高价(元),复权口径同上。", "元", True),
|
||||
("low", "最低价", "当日最低价(元),复权口径同上。", "元", True),
|
||||
("volume", "成交量", "当日成交股数(股)。数据源原始单位为「手」,入库时已 ×100 换算。", "股", True),
|
||||
("amount", "成交额", "当日成交金额(元)。数据源原始单位为「千元」,入库时已 ×1000 换算。", "元", True),
|
||||
)
|
||||
|
||||
# ---------- 技术派生(滚动窗口计算,非行情原列) ----------
|
||||
|
||||
_TECH_FIELDS: tuple[tuple[str, str, str, str, bool], ...] = (
|
||||
("ma20", "20 日均线", "收盘价的 20 个交易日简单移动平均(元),在选股日当日取值。", "元", True),
|
||||
("ma60", "60 日均线", "收盘价的 60 个交易日简单移动平均(元),在选股日当日取值。", "元", True),
|
||||
)
|
||||
|
||||
# ---------- 每日指标(daily_basic) ----------
|
||||
# 单位照抄 data_sources/tushare.py: 百分数/倍数为原样,股本为万股,市值为万元。
|
||||
|
||||
_DAILY_FIELDS: tuple[tuple[str, str, str, str, bool], ...] = (
|
||||
("dv_ratio", "股息率", "近 12 个月现金分红 / 总市值 × 100(%),逐日时点值 —— 即因子 dividend_yield 的口径。", "%", True),
|
||||
("dv_ttm", "股息率 TTM", "近 12 个月滚动现金分红 / 总市值 × 100(%),即因子 dividend_yield_ttm 的口径。", "%", True),
|
||||
("pe", "市盈率 PE", "总市值 / 最新年报净利润(倍,静态口径)。", "倍", True),
|
||||
("pe_ttm", "市盈率 PE(TTM)", "总市值 / 最近 12 个月净利润(倍)。", "倍", True),
|
||||
("pb", "市净率 PB", "总市值 / 最新报告期净资产(倍)。", "倍", True),
|
||||
("turnover_rate", "换手率", "当日成交股数 / 流通股本 × 100(%)。", "%", True),
|
||||
("volume_ratio", "量比", "当日成交量 / 过去 5 日平均成交量(倍)。", "倍", True),
|
||||
("total_mv", "总市值", "总股本 × 当日收盘价(万元)。", "万元", True),
|
||||
("circ_mv", "流通市值", "流通股本 × 当日收盘价(万元)。", "万元", True),
|
||||
("ps", "市销率 PS", "总市值 / 最新年报营业收入(倍)。", "倍", False),
|
||||
("ps_ttm", "市销率 PS(TTM)", "总市值 / 最近 12 个月营业收入(倍)。", "倍", False),
|
||||
("total_share", "总股本", "总股本(万股)。", "万股", False),
|
||||
("float_share", "流通股本", "流通股本(万股)。", "万股", False),
|
||||
("free_share", "自由流通股本", "自由流通股本(万股)。", "万股", False),
|
||||
)
|
||||
|
||||
# ---------- 财务指标(fundamental.*) ----------
|
||||
# 可见性口径:只取 announce_date <= 选股日的最新已公告值(防未来函数,见 selection.run_condition_selection)。
|
||||
|
||||
_FUNDAMENTAL_FIELDS: tuple[tuple[str, str, str, str, bool], ...] = (
|
||||
("roe", "净资产收益率 ROE", "最新已公告报告期的净资产收益率(%)。", "%", True),
|
||||
("eps", "每股收益 EPS", "最新已公告报告期的每股收益(元)。", "元", True),
|
||||
("gross_margin", "毛利率", "最新已公告报告期的毛利率(%)。", "%", True),
|
||||
("net_profit", "归母净利润", "最新已公告报告期的归母净利润(元)。", "元", False),
|
||||
("total_revenue", "营业总收入", "最新已公告报告期的营业总收入(元)。注意:目前只有新浪兜底源提供该字段,多数行可能为空 —— 缺失时条件视为不通过。", "元", False),
|
||||
)
|
||||
|
||||
# static.* 里明确不支持的字段(存在但不可比较)
|
||||
_STATIC_UNSUPPORTED: dict[str, str] = {
|
||||
"list_date": "上市日期是日期,不是可比较的数值/字符串;请改用股票池的「上市天数」设置",
|
||||
"delist_date": "退市日期是日期,不是可比较的数值/字符串",
|
||||
}
|
||||
|
||||
|
||||
def _defs() -> list[FieldDef]:
|
||||
"""构造全部内置字段定义(每次调用重新构造,保证与代码注册表实时一致)。"""
|
||||
out: list[FieldDef] = []
|
||||
for attr, label, desc, curated in _STATIC_FIELDS:
|
||||
out.append(
|
||||
FieldDef(
|
||||
name=f"static.{attr}",
|
||||
label=label,
|
||||
description=desc,
|
||||
kind="str",
|
||||
group_name=GROUP_STOCK,
|
||||
curated=curated,
|
||||
)
|
||||
)
|
||||
for name, label, desc, unit, curated in _BAR_FIELDS:
|
||||
out.append(
|
||||
FieldDef(
|
||||
name, label, desc, "num", GROUP_QUOTE, unit, curated,
|
||||
_UNIT_LADDERS.get(name, ()),
|
||||
)
|
||||
)
|
||||
for name, label, desc, unit, curated in _TECH_FIELDS:
|
||||
out.append(
|
||||
FieldDef(name, label, desc, "num", GROUP_TECH, unit, curated, _UNIT_LADDERS.get(name, ()))
|
||||
)
|
||||
for name, label, desc, unit, curated in _DAILY_FIELDS:
|
||||
out.append(
|
||||
FieldDef(name, label, desc, "num", GROUP_DAILY, unit, curated, _UNIT_LADDERS.get(name, ()))
|
||||
)
|
||||
for name, label, desc, unit, curated in _FUNDAMENTAL_FIELDS:
|
||||
full = f"fundamental.{name}"
|
||||
out.append(
|
||||
FieldDef(full, label, desc, "num", GROUP_FUNDAMENTAL, unit, curated, _UNIT_LADDERS.get(full, ()))
|
||||
)
|
||||
for d in list_factors():
|
||||
# 因子既能当「打分因子」也能当「过滤条件」:这里复用因子注册表的元数据,
|
||||
# 不另写描述,避免两处文案漂移。因子的 description 里已写明公式与口径;
|
||||
# 标签用中文名(含参数),如「动量(窗口 60,越高越好)」——
|
||||
# 参数化实例不在这里(它们在 /factors 目录里,条件下拉按名并入)。
|
||||
out.append(
|
||||
FieldDef(
|
||||
name=d.name,
|
||||
label=f"{d.display}(因子)",
|
||||
description=d.description + (f" 用法:{d.brief}" if d.brief else ""),
|
||||
kind="num",
|
||||
group_name=GROUP_FACTOR,
|
||||
curated=True,
|
||||
)
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
def builtin_fields() -> list[FieldDef]:
|
||||
"""全部引擎支持的字段(含 curated=False 的「可新增但不默认展示」项)。
|
||||
|
||||
名字唯一性由因子名与各分组前缀保证;一旦重复说明注册表写错,直接抛错而非静默覆盖。
|
||||
单位阶梯同样自检:首项必须是基准单位且系数为 1.0,系数必须为正 —— 写错了会让
|
||||
「界面显示 5 亿元、引擎按 5 万元比」这种错静默溜进生产。
|
||||
"""
|
||||
defs = _defs()
|
||||
names = [d.name for d in defs]
|
||||
dup = {n for n in names if names.count(n) > 1}
|
||||
if dup:
|
||||
raise ValueError(f"条件字段注册表存在重名:{sorted(dup)}")
|
||||
for d in defs:
|
||||
if not d.units:
|
||||
continue
|
||||
base, factor = d.units[0]
|
||||
if base != d.unit or factor != 1.0:
|
||||
raise ValueError(
|
||||
f"字段 {d.name} 的单位阶梯首项必须是基准单位 {d.unit!r}(系数 1.0),实际 {d.units[0]!r}"
|
||||
)
|
||||
if any(f <= 0 for _, f in d.units):
|
||||
raise ValueError(f"字段 {d.name} 的单位换算系数必须为正:{d.units}")
|
||||
if len({u for u, _ in d.units}) != len(d.units):
|
||||
raise ValueError(f"字段 {d.name} 的单位阶梯有重复单位:{d.units}")
|
||||
return defs
|
||||
|
||||
|
||||
def curated_fields() -> list[FieldDef]:
|
||||
"""默认进「字段库」的字段(下拉里开箱可见的那批)。"""
|
||||
return [d for d in builtin_fields() if d.curated]
|
||||
|
||||
|
||||
def get_field(name: str) -> FieldDef | None:
|
||||
"""按字段名取定义:注册表字段,或**参数化因子键**(如 momentum(window=90,direction=…))。
|
||||
|
||||
为什么参数化因子也要能取到:它是引擎真认的条件字段(`momentum_60 > 0` 一直合法),
|
||||
而字段库/说明书的单位后缀、类型判断都走这里。取不到会让人误以为「引擎不支持」,
|
||||
甚至让「把参数化因子加进字段库」这一步半路 assert 崩掉(500 而不是 422)。
|
||||
"""
|
||||
for d in builtin_fields():
|
||||
if d.name == name:
|
||||
return d
|
||||
return _factor_field(name)
|
||||
|
||||
|
||||
def _factor_field(name: str) -> FieldDef | None:
|
||||
"""参数化因子键 → 字段定义(因子是无量纲量,不带单位)。"""
|
||||
try:
|
||||
defn, _fn = resolve_factor(name)
|
||||
except FactorError:
|
||||
return None
|
||||
if defn.name in {d.name for d in builtin_fields()}:
|
||||
return None # 注册表字段已在上一步返回;这里只处理新增的参数化实例
|
||||
return FieldDef(
|
||||
name=defn.name,
|
||||
label=f"{defn.display}(因子)",
|
||||
description=defn.description + (f" 用法:{defn.brief}" if defn.brief else ""),
|
||||
kind="num",
|
||||
group_name=GROUP_FACTOR,
|
||||
curated=False, # 不自动进字段库:目录在 /factors 管,条件里按名并进来
|
||||
)
|
||||
|
||||
|
||||
def _stock_attrs() -> set[str]:
|
||||
return set(Stock.model_fields)
|
||||
|
||||
|
||||
def _fundamental_attrs() -> set[str]:
|
||||
"""FinancialIndicator 里可比较的数值字段(排除 symbol/报告期/来源等元数据)。"""
|
||||
skip = {"symbol", "report_date", "announce_date", "source"}
|
||||
return {n for n in FinancialIndicator.model_fields if n not in skip}
|
||||
|
||||
|
||||
def reason_unsupported(name: str) -> str:
|
||||
"""字段不可用的理由(用于 422 文案;可用字段返回空串)。"""
|
||||
if not name or not name.strip():
|
||||
return "字段名为空"
|
||||
if name in DAILY_BAR_NUMERIC_FIELDS:
|
||||
return ""
|
||||
if name in ("ma20", "ma60"):
|
||||
return ""
|
||||
if name in DAILY_BASIC_NUMERIC_FIELDS:
|
||||
return ""
|
||||
if name.startswith("static."):
|
||||
attr = name[len("static.") :]
|
||||
if attr in _STATIC_UNSUPPORTED:
|
||||
return f"{name} 不可用作条件:{_STATIC_UNSUPPORTED[attr]}"
|
||||
if attr in _stock_attrs():
|
||||
return ""
|
||||
return f"{name} 不存在:股票基础信息里没有 {attr} 字段"
|
||||
if name.startswith("fundamental."):
|
||||
attr = name[len("fundamental.") :]
|
||||
if attr in _fundamental_attrs():
|
||||
return ""
|
||||
return f"{name} 不存在:财务指标里没有 {attr} 字段"
|
||||
try:
|
||||
get_factor(name)
|
||||
except FactorError:
|
||||
return (
|
||||
f"{name} 不是引擎支持的字段。可用:行情列({', '.join(DAILY_BAR_NUMERIC_FIELDS)})、"
|
||||
"ma20/ma60、每日指标列、static.<股票基础字段>、fundamental.<财务字段>、"
|
||||
"已注册因子名,或参数化因子键(如 momentum(window=90,direction=higher_is_better))"
|
||||
)
|
||||
return ""
|
||||
|
||||
|
||||
def is_supported_field(name: str) -> bool:
|
||||
"""引擎是否真能算这个字段(False = 条件永远不通过,必须拒绝)。"""
|
||||
return reason_unsupported(name) == ""
|
||||
|
||||
|
||||
def available_fields(existing: set[str]) -> list[FieldDef]:
|
||||
"""引擎支持但**尚未进目录**的字段(用户「新增字段」时可选项)。
|
||||
|
||||
只从注册表里挑,用户因此不可能加进一个引擎算不出来的字段(§24 不假装支持)。
|
||||
"""
|
||||
return [d for d in builtin_fields() if not d.curated and d.name not in existing]
|
||||
|
||||
|
||||
# ---------- 单位换算(界面单位 ⇄ 基准单位) ----------
|
||||
|
||||
|
||||
def unit_options(name: str) -> list[tuple[str, float]]:
|
||||
"""该字段可选的界面单位(首项为基准单位,系数 = 该单位 → 基准单位)。字段不存在 → 空列表。
|
||||
|
||||
**换算在界面层做**(前端按系数换算输入/回显),存储与引擎一律用基准单位:
|
||||
这样归档里的 ConditionSpec 永远不随界面设置改变含义。
|
||||
"""
|
||||
d = get_field(name)
|
||||
return list(d.unit_options) if d else []
|
||||
|
||||
|
||||
def unit_allowed(name: str, unit: str) -> bool:
|
||||
"""该单位是否在字段允许的阶梯里(API 据此 422 拒绝自由文本单位)。"""
|
||||
return unit in {u for u, _ in unit_options(name)}
|
||||
+553
-155
@@ -1,4 +1,34 @@
|
||||
"""因子引擎:因子注册表、元数据与计算(Phase 2,低频选股因子)。
|
||||
"""因子引擎:因子**模板**(含可编辑参数)、实例注册表与计算(Phase 2 起,2026-10 参数化)。
|
||||
|
||||
## 三层概念(这是本模块的核心约定)
|
||||
|
||||
1. **模板(FactorTemplate)**:算法的家族,如 `momentum`(动量)、`volatility`(波动率)。
|
||||
模板声明「哪些参数可编辑、允许范围、默认值」以及计算函数 `fn(fields, params)`。
|
||||
2. **参数(params)**:模板的可编辑取值,如 `window=90`、`direction=lower_is_better`。
|
||||
参数约束是**受控范围**(整数区间 / 枚举),不允许自由值 —— 见 AGENT.md §24:
|
||||
写不进去就报错,绝不静默接受一个引擎其实不支持的设置。
|
||||
3. **因子实例(FactorDef)**:`(模板, 参数)` 的具体因子,**名字里带着全部参数**:
|
||||
|
||||
momentum_60 ← 内置实例(代码里登记的历史名)
|
||||
momentum(window=90,direction=higher_is_better) ← 参数化实例(目录里创建)
|
||||
|
||||
实例名即身份:参数写进名字,任何地方(策略 JSON、归档 spec、组合组件、条件字段)
|
||||
存下这个名字,就同时冻结了「用哪个模板 + 哪些参数」——**历史归档不会因为之后
|
||||
改了什么参数而改变含义**。这也是为什么不把参数放在另一个字段里:那需要改动
|
||||
所有已经存了因子名的地方(策略/归档/回放/信号/Agent 工具),而且容易漏。
|
||||
|
||||
## 参数与默认值:为什么键里总是写全 direction
|
||||
|
||||
键里**不省略任何可编辑参数**(哪怕等于模板默认值)。若省略,`momentum(window=90)`
|
||||
的含义就取决于「模板默认方向」这一代码事实:将来代码把默认方向一改,用户已经存下的
|
||||
策略/归档会跟着变义 —— 与「单位」那次拒绝的做法同理。写全参数后,键自解释、
|
||||
不依赖任何默认值,改默认值只影响新建实例。
|
||||
|
||||
## 兼容性
|
||||
|
||||
`get_factor()` 现在能吃两种名字:注册表里的历史名(内置实例)与参数化键;
|
||||
`compute_factor()`、`list_factors()` 的行为保持不变,因此下游(选股/回测/组合/
|
||||
说明书/条件字段/Agent 工具)无需感知参数化的存在,也**不会绕过参数校验**。
|
||||
|
||||
数据形态:行情长表 DataFrame(列 symbol/trade_date/close/high/low/volume/amount,
|
||||
以及经 ResearchService 并入的每日指标列如 dv_ratio/dv_ttm),
|
||||
@@ -10,15 +40,68 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
import inspect
|
||||
import re
|
||||
from collections.abc import Callable, Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
|
||||
DIRECTION_HIGHER = "higher_is_better"
|
||||
DIRECTION_LOWER = "lower_is_better"
|
||||
DIRECTIONS = (DIRECTION_HIGHER, DIRECTION_LOWER)
|
||||
|
||||
# 参数名常量(键里的字面量,改它等于改所有已存键的含义,别改)
|
||||
P_WINDOW = "window"
|
||||
P_FAST = "fast"
|
||||
P_SLOW = "slow"
|
||||
P_DIRECTION = "direction"
|
||||
|
||||
# 窗口参数的允许范围:受控区间而非固定档(任意整数都能算,但要有边界)。
|
||||
# 上限 500 个交易日 ≈ 两年,够长;下限 2 是因为 shift(1)/rolling(1) 的波动率无意义。
|
||||
WINDOW_MIN = 2
|
||||
WINDOW_MAX = 500
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ParamSpec:
|
||||
"""一个可编辑参数的约束(受控范围,越界一律报错而不是截断/静默忽略)。"""
|
||||
|
||||
name: str
|
||||
label: str
|
||||
kind: str # "int" | "enum"
|
||||
default: Any
|
||||
minimum: int | None = None
|
||||
maximum: int | None = None
|
||||
choices: tuple[str, ...] = ()
|
||||
note: str = ""
|
||||
|
||||
def describe(self) -> str:
|
||||
"""人类可读的约束说明(用于错误文案与目录展示)。"""
|
||||
if self.kind == "enum":
|
||||
return "、".join(self.choices)
|
||||
if self.minimum is not None and self.maximum is not None:
|
||||
return f"{self.minimum} ~ {self.maximum} 的整数"
|
||||
return "整数"
|
||||
|
||||
|
||||
DIRECTION_SPEC = ParamSpec(
|
||||
name=P_DIRECTION,
|
||||
label="方向",
|
||||
kind="enum",
|
||||
default=DIRECTION_HIGHER,
|
||||
choices=DIRECTIONS,
|
||||
note="越高越好 / 越低越好:决定复合分里的排序方向(低为好自动取负)。",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FactorDef:
|
||||
"""因子元数据(AGENT.md §22 要求逐项明确)。"""
|
||||
"""因子元数据(AGENT.md §22 要求逐项明确)。
|
||||
|
||||
新增字段都带默认值:历史代码用位置参数构造 FactorDef 的地方不受影响。
|
||||
"""
|
||||
|
||||
name: str
|
||||
description: str
|
||||
@@ -28,9 +111,55 @@ class FactorDef:
|
||||
lookback: int = 20
|
||||
direction: str = "higher_is_better" # | lower_is_better
|
||||
requires: tuple[str, ...] = ("close",)
|
||||
# ---- 参数化(2026-10)----
|
||||
template: str = "" # 模板名,如 "momentum";空串 = 手工登记的老式因子
|
||||
params: Mapping[str, Any] = field(default_factory=dict) # 冻结的参数取值
|
||||
param_specs: tuple[ParamSpec, ...] = () # 可编辑参数与约束(供目录/界面)
|
||||
source: str = "builtin" # builtin(代码注册表)| custom(目录里创建的参数化实例)
|
||||
label: str = "" # 中文显示名(含参数),如「动量(窗口 90,越高越好)」
|
||||
|
||||
@property
|
||||
def display(self) -> str:
|
||||
"""界面用显示名:没有 label 时退回 name(老因子/自定义登记行)。"""
|
||||
return self.label or self.name
|
||||
|
||||
|
||||
FactorFn = Callable[[dict[str, pd.DataFrame]], pd.DataFrame]
|
||||
FactorFn = Callable[[dict[str, pd.DataFrame], Mapping[str, Any]], pd.DataFrame]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FactorTemplate:
|
||||
"""算法家族 + 可编辑参数声明 + 内置实例(历史名)。"""
|
||||
|
||||
name: str
|
||||
label: str # 中文家族名,如「动量」
|
||||
description: str # 可含 {window} / {fast} / {slow} 占位
|
||||
formula: str
|
||||
brief: str
|
||||
fn: FactorFn
|
||||
param_specs: tuple[ParamSpec, ...] = ()
|
||||
requires: tuple[str, ...] = ("close",)
|
||||
frequency: str = "daily"
|
||||
direction_default: str = DIRECTION_HIGHER
|
||||
lookback_of: Callable[[Mapping[str, Any]], int] | None = None
|
||||
check: Callable[[Mapping[str, Any]], str | None] | None = None # 跨参数约束
|
||||
instances: tuple[tuple[str, Mapping[str, Any]], ...] = () # ((历史名, 参数), ...)
|
||||
label_of: Callable[[Mapping[str, Any]], str] | None = None
|
||||
|
||||
def specs(self) -> tuple[ParamSpec, ...]:
|
||||
"""全部可编辑参数(模板自己的参数 + 方向,方向恒在最后)。"""
|
||||
direction = ParamSpec(
|
||||
name=DIRECTION_SPEC.name,
|
||||
label=DIRECTION_SPEC.label,
|
||||
kind=DIRECTION_SPEC.kind,
|
||||
default=self.direction_default,
|
||||
choices=DIRECTION_SPEC.choices,
|
||||
note=DIRECTION_SPEC.note,
|
||||
)
|
||||
return (*self.param_specs, direction)
|
||||
|
||||
def defaults(self) -> dict[str, Any]:
|
||||
return {s.name: s.default for s in self.specs()}
|
||||
|
||||
|
||||
class FactorError(ValueError):
|
||||
@@ -38,42 +167,318 @@ class FactorError(ValueError):
|
||||
|
||||
|
||||
_REGISTRY: dict[str, tuple[FactorDef, FactorFn]] = {}
|
||||
_TEMPLATES: dict[str, FactorTemplate] = {}
|
||||
|
||||
# 参数化键:template(k=v,k=v)。模板名与参数名限定为标识符,值限定为标识符/数字,
|
||||
# 避免出现靠运气才能解析的名字(宁可在创建时就被拒)。
|
||||
_KEY_RE = re.compile(r"^(?P<template>[A-Za-z_][A-Za-z0-9_]*)\((?P<args>[^()]*)\)$")
|
||||
_ARG_RE = re.compile(r"^(?P<key>[A-Za-z_][A-Za-z0-9_]*)=(?P<value>[A-Za-z_][A-Za-z0-9_]*|-?\d+)$")
|
||||
|
||||
|
||||
def _accepts_params(fn: Callable) -> bool:
|
||||
"""判断计算函数是否声明了 params 形参(兼容老的一参数写法)。"""
|
||||
try:
|
||||
params = inspect.signature(fn).parameters
|
||||
except (TypeError, ValueError): # 内建/C 实现,按老写法处理
|
||||
return False
|
||||
if any(p.kind is inspect.Parameter.VAR_POSITIONAL for p in params.values()):
|
||||
return True
|
||||
positional = [
|
||||
p
|
||||
for p in params.values()
|
||||
if p.kind in (inspect.Parameter.POSITIONAL_ONLY, inspect.Parameter.POSITIONAL_OR_KEYWORD)
|
||||
]
|
||||
return len(positional) >= 2
|
||||
|
||||
|
||||
def _bind(fn: Callable) -> FactorFn:
|
||||
"""把计算函数统一成 (fields, params) 两参数调用。"""
|
||||
if _accepts_params(fn):
|
||||
|
||||
def _bound(fields, params, _fn=fn):
|
||||
return _fn(fields, params)
|
||||
|
||||
return _bound
|
||||
|
||||
def _legacy(fields, params, _fn=fn):
|
||||
return _fn(fields)
|
||||
|
||||
return _legacy
|
||||
|
||||
|
||||
def register_template(template: FactorTemplate) -> FactorTemplate:
|
||||
"""注册模板,并把它的内置实例(历史名)登记进注册表。"""
|
||||
if template.name in _TEMPLATES:
|
||||
raise FactorError(f"模板 {template.name} 已注册")
|
||||
_TEMPLATES[template.name] = template
|
||||
for name, params in template.instances:
|
||||
defn = build_factor_def(template, params, name=name, source="builtin")
|
||||
_REGISTRY[name] = (defn, _bind(template.fn))
|
||||
return template
|
||||
|
||||
|
||||
def register(defn: FactorDef) -> Callable[[FactorFn], FactorFn]:
|
||||
"""装饰器:注册自定义因子。"""
|
||||
"""装饰器:注册**自定义因子**(老式登记,无参数化;测试与扩展用)。"""
|
||||
|
||||
def deco(fn: FactorFn) -> FactorFn:
|
||||
if defn.name in _REGISTRY:
|
||||
raise FactorError(f"因子 {defn.name} 已注册")
|
||||
_REGISTRY[defn.name] = (defn, fn)
|
||||
_REGISTRY[defn.name] = (defn, _bind(fn))
|
||||
return fn
|
||||
|
||||
return deco
|
||||
|
||||
|
||||
def get_factor(name: str) -> tuple[FactorDef, FactorFn]:
|
||||
if name not in _REGISTRY:
|
||||
raise FactorError(f"未知因子:{name}(可用:{', '.join(sorted(_REGISTRY))})")
|
||||
return _REGISTRY[name]
|
||||
def list_templates() -> list[FactorTemplate]:
|
||||
return [t for _n, t in sorted(_TEMPLATES.items())]
|
||||
|
||||
|
||||
def get_template(name: str) -> FactorTemplate:
|
||||
if name not in _TEMPLATES:
|
||||
raise FactorError(f"未知因子模板:{name}(可用:{', '.join(sorted(_TEMPLATES))})")
|
||||
return _TEMPLATES[name]
|
||||
|
||||
|
||||
def _render(text: str, params: Mapping[str, Any]) -> str:
|
||||
"""渲染带 {param} 占位的文案;没有占位就原样返回(不做 format,避免误伤花括号)。"""
|
||||
if "{" not in text:
|
||||
return text
|
||||
try:
|
||||
return text.format(**params)
|
||||
except KeyError as exc: # 模板写错占位名 —— 宁可当场炸,也不要漏出半成品文案
|
||||
raise FactorError(f"因子文案占位符缺少参数 {exc}:{text}") from None
|
||||
|
||||
|
||||
def _param_summary(template: FactorTemplate, params: Mapping[str, Any]) -> str:
|
||||
"""非方向参数的摘要,如「窗口 90」「快线 5、慢线 60」。"""
|
||||
return "、".join(f"{spec.label} {params[spec.name]}" for spec in template.param_specs)
|
||||
|
||||
|
||||
def _default_label(template: FactorTemplate, params: Mapping[str, Any]) -> str:
|
||||
summary = _param_summary(template, params)
|
||||
direction = "越高越好" if params[P_DIRECTION] == DIRECTION_HIGHER else "越低越好"
|
||||
inner = ",".join([p for p in (summary, direction) if p])
|
||||
return f"{template.label}({inner})"
|
||||
|
||||
|
||||
def fill_params(template: FactorTemplate, params: Mapping[str, Any]) -> dict[str, Any]:
|
||||
"""校验参数并补全缺省值(**创建路径**用:内置实例、目录里新建参数化因子)。
|
||||
|
||||
只认声明过的参数,类型/范围/枚举全部受控,跨参数约束另查 —— 缺的参数取模板默认值。
|
||||
"""
|
||||
specs = {s.name: s for s in template.specs()}
|
||||
unknown = sorted(set(params) - set(specs))
|
||||
if unknown:
|
||||
raise FactorError(
|
||||
f"因子模板 {template.name} 不支持的参数:{', '.join(unknown)};"
|
||||
f"可编辑参数只有 {', '.join(specs)}。参数不能随便加 —— 引擎算不了的要当场拒绝。"
|
||||
)
|
||||
merged: dict[str, Any] = {**template.defaults(), **params}
|
||||
return _check_all(template, merged)
|
||||
|
||||
|
||||
def validate_params(template: FactorTemplate, params: Mapping[str, Any]) -> dict[str, Any]:
|
||||
"""严格校验:**每个参数都必须显式给出**(解析已存的因子键用)。
|
||||
|
||||
为什么解析时不许省:省掉的参数只能靠「模板默认值」补,而默认值是会随代码改的
|
||||
事实 —— 一旦改了,用户早就存下的策略/归档就会跟着变义(模块头详述)。
|
||||
"""
|
||||
specs = {s.name: s for s in template.specs()}
|
||||
unknown = sorted(set(params) - set(specs))
|
||||
if unknown:
|
||||
raise FactorError(
|
||||
f"因子模板 {template.name} 不支持的参数:{', '.join(unknown)};"
|
||||
f"可编辑参数只有 {', '.join(specs)}。"
|
||||
)
|
||||
missing = sorted(set(specs) - set(params))
|
||||
if missing:
|
||||
canonical = canonical_key(template, params)
|
||||
raise FactorError(
|
||||
f"因子键缺少参数:{', '.join(missing)}。键里必须写全所有参数"
|
||||
f"(否则含义会取决于模板默认值);规范写法:{canonical}"
|
||||
)
|
||||
return _check_all(template, params)
|
||||
|
||||
|
||||
def _check_all(template: FactorTemplate, params: Mapping[str, Any]) -> dict[str, Any]:
|
||||
out = {spec.name: _check_param(template, spec, params[spec.name]) for spec in template.specs()}
|
||||
if template.check is not None:
|
||||
problem = template.check(out)
|
||||
if problem:
|
||||
raise FactorError(f"因子模板 {template.name} 参数不合法:{problem}")
|
||||
return out
|
||||
|
||||
|
||||
def _check_param(template: FactorTemplate, spec: ParamSpec, value: Any) -> Any:
|
||||
if spec.kind == "enum":
|
||||
if not isinstance(value, str) or value not in spec.choices:
|
||||
raise FactorError(
|
||||
f"因子「{template.name}」的参数 {spec.name}={value!r} 不合法:"
|
||||
f"只能是 {spec.describe()}。"
|
||||
)
|
||||
return value
|
||||
if isinstance(value, str): # 键里解析出来的是字符串,数字要能转
|
||||
try:
|
||||
value = int(value)
|
||||
except ValueError:
|
||||
raise FactorError(
|
||||
f"因子「{template.name}」的参数 {spec.name}={value!r} 不是整数:"
|
||||
f"应为 {spec.describe()}。"
|
||||
) from None
|
||||
if not isinstance(value, int) or isinstance(value, bool):
|
||||
raise FactorError(
|
||||
f"因子「{template.name}」的参数 {spec.name}={value!r} 不是整数:"
|
||||
f"应为 {spec.describe()}。"
|
||||
)
|
||||
if spec.minimum is not None and value < spec.minimum:
|
||||
raise FactorError(
|
||||
f"因子「{template.name}」的参数 {spec.name}={value} 太小:应为 {spec.describe()}。"
|
||||
)
|
||||
if spec.maximum is not None and value > spec.maximum:
|
||||
raise FactorError(
|
||||
f"因子「{template.name}」的参数 {spec.name}={value} 太大:应为 {spec.describe()}。"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
def build_factor_def(
|
||||
template: FactorTemplate,
|
||||
params: Mapping[str, Any],
|
||||
*,
|
||||
name: str,
|
||||
source: str = "custom",
|
||||
) -> FactorDef:
|
||||
"""由模板 + 参数构造因子实例(未知/越界参数在此处被拒)。"""
|
||||
checked = fill_params(template, params)
|
||||
lookback = template.lookback_of(checked) if template.lookback_of else 0
|
||||
label_of = template.label_of or (lambda p: _default_label(template, p))
|
||||
return FactorDef(
|
||||
name=name,
|
||||
description=_render(template.description, checked),
|
||||
formula=_render(template.formula, checked),
|
||||
brief=_render(template.brief, checked),
|
||||
frequency=template.frequency,
|
||||
lookback=lookback,
|
||||
direction=checked[P_DIRECTION],
|
||||
requires=template.requires,
|
||||
template=template.name,
|
||||
params=dict(checked),
|
||||
param_specs=template.specs(),
|
||||
source=source,
|
||||
label=label_of(checked),
|
||||
)
|
||||
|
||||
|
||||
def canonical_key(template: FactorTemplate | str, params: Mapping[str, Any]) -> str:
|
||||
"""参数化实例的规范名:参数按模板声明顺序写全(含方向),如
|
||||
|
||||
momentum(window=90,direction=higher_is_better)
|
||||
|
||||
写全的好处见模块头:键自解释,不依赖任何默认值。
|
||||
"""
|
||||
tpl = get_template(template) if isinstance(template, str) else template
|
||||
checked = fill_params(tpl, params)
|
||||
args = ",".join(f"{spec.name}={checked[spec.name]}" for spec in tpl.specs())
|
||||
return f"{tpl.name}({args})"
|
||||
|
||||
|
||||
def parse_factor_key(name: str) -> tuple[FactorTemplate, dict[str, Any]]:
|
||||
"""解析参数化键 → (模板, 参数);不是参数化键或参数不合法都抛 FactorError。"""
|
||||
m = _KEY_RE.match(name.strip())
|
||||
if not m:
|
||||
raise FactorError(
|
||||
f"因子键格式不对:{name};内置因子用注册表名(如 momentum_60),"
|
||||
"参数化因子用 template(k=v,...)(如 momentum(window=90,direction=higher_is_better))"
|
||||
)
|
||||
template_name = m.group("template")
|
||||
try:
|
||||
template = get_template(template_name)
|
||||
except FactorError as exc:
|
||||
if template_name in _REGISTRY:
|
||||
raise FactorError(
|
||||
f"{template_name} 是内置因子实例名,不能在它上面再带参数;"
|
||||
"要参数化请用模板名,例如 momentum(window=20,direction=higher_is_better)"
|
||||
) from None
|
||||
raise exc
|
||||
raw: dict[str, Any] = {}
|
||||
args = m.group("args").strip()
|
||||
if args:
|
||||
for part in args.split(","):
|
||||
am = _ARG_RE.match(part.strip())
|
||||
if not am:
|
||||
raise FactorError(
|
||||
f"因子键里的参数写法不对:{part.strip()};应为 名=值(值只能是整数或标识符)"
|
||||
)
|
||||
key = am.group("key")
|
||||
if key in raw:
|
||||
raise FactorError(f"因子键里参数重复:{key}")
|
||||
raw[key] = am.group("value")
|
||||
return template, validate_params(template, raw)
|
||||
|
||||
|
||||
def _derived(name: str) -> tuple[FactorDef, FactorFn]:
|
||||
template, params = parse_factor_key(name)
|
||||
key = canonical_key(template, params)
|
||||
if key != name.strip():
|
||||
raise FactorError(
|
||||
f"因子键 {name} 不是规范写法:同样参数请写成 {key}"
|
||||
"(参数顺序固定、值要写全,避免同一个因子出现多个名字)"
|
||||
)
|
||||
defn = build_factor_def(template, params, name=key, source="custom")
|
||||
return defn, _bind(template.fn)
|
||||
|
||||
|
||||
def resolve_factor(name: str) -> tuple[FactorDef, FactorFn]:
|
||||
"""按名字取因子(注册表历史名 / 参数化键都行),取不到就抛可读错误。"""
|
||||
if name in _REGISTRY:
|
||||
return _REGISTRY[name]
|
||||
return _derived(name)
|
||||
|
||||
|
||||
def is_resolvable(name: str) -> bool:
|
||||
try:
|
||||
resolve_factor(name)
|
||||
except FactorError:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
# 兼容旧名:所有既有调用点(选股/回测/组合/说明书/条件字段/Agent 工具)自动支持参数化键。
|
||||
get_factor = resolve_factor
|
||||
|
||||
|
||||
def list_factors() -> list[FactorDef]:
|
||||
"""注册表里的**内置实例**(历史名),按名字排序(目录 seed 用)。"""
|
||||
return [d for d, _fn in sorted(_REGISTRY.values(), key=lambda x: x[0].name)]
|
||||
|
||||
|
||||
def compute_factor(name: str, daily: pd.DataFrame) -> tuple[FactorDef, pd.DataFrame]:
|
||||
"""计算因子:从行情长表提取所需字段的面板后调用因子函数。"""
|
||||
defn, fn = get_factor(name)
|
||||
defn, fn = resolve_factor(name)
|
||||
fields: dict[str, pd.DataFrame] = {}
|
||||
for col in defn.requires:
|
||||
panel = daily.pivot(index="trade_date", columns="symbol", values=col).sort_index()
|
||||
panel.index = pd.to_datetime(panel.index)
|
||||
fields[col] = panel
|
||||
return defn, fn(fields)
|
||||
return defn, fn(fields, defn.params)
|
||||
|
||||
|
||||
# ---------- 内置因子 ----------
|
||||
# ---------- 内置因子模板 ----------
|
||||
# 每个模板的 instances 是历史名 + 它的参数:这些名字已经存在于策略/归档/文档/测试里,
|
||||
# 必须继续可解析,所以它们不是「参数化键」,而是代码登记的实例。
|
||||
|
||||
|
||||
def _window_spec(label: str = "窗口") -> ParamSpec:
|
||||
return ParamSpec(
|
||||
name=P_WINDOW,
|
||||
label=label,
|
||||
kind="int",
|
||||
default=20,
|
||||
minimum=WINDOW_MIN,
|
||||
maximum=WINDOW_MAX,
|
||||
note=f"{WINDOW_MIN}~{WINDOW_MAX} 个交易日;改窗口 = 换一个因子身份(新键),"
|
||||
"旧键仍按旧参数计算。",
|
||||
)
|
||||
|
||||
|
||||
def _rolling_return(prices: pd.DataFrame, lookback: int) -> pd.DataFrame:
|
||||
@@ -84,181 +489,174 @@ def _rolling_vol(prices: pd.DataFrame, lookback: int) -> pd.DataFrame:
|
||||
return prices.pct_change().rolling(lookback).std()
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"momentum_20",
|
||||
"过去 20 个交易日收益率",
|
||||
"close / close.shift(20) - 1",
|
||||
brief="短期动量:近一个月强势股延续性较强,适合趋势延续环境(牛市中段);震荡市易追高。",
|
||||
lookback=20,
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="momentum",
|
||||
label="动量",
|
||||
description="过去 {window} 个交易日收益率",
|
||||
formula="close / close.shift({window}) - 1",
|
||||
brief="动量:强者延续,适合趋势延续环境;窗口越短越敏感、越长越稳。",
|
||||
fn=lambda fields, params: _rolling_return(fields["close"], params[P_WINDOW]),
|
||||
param_specs=(_window_spec(),),
|
||||
lookback_of=lambda params: params[P_WINDOW],
|
||||
instances=(
|
||||
("momentum_20", {P_WINDOW: 20}),
|
||||
("momentum_60", {P_WINDOW: 60}),
|
||||
("momentum_120", {P_WINDOW: 120}),
|
||||
),
|
||||
)
|
||||
)
|
||||
def _momentum_20(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return _rolling_return(fields["close"], 20)
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"momentum_60",
|
||||
"过去 60 个交易日收益率",
|
||||
"close / close.shift(60) - 1",
|
||||
brief="中期动量:A 股常见有效时段(约 1~3 个月),趋势行情首选;需结合市场阶段判断方向。",
|
||||
lookback=60,
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="volatility",
|
||||
label="波动率",
|
||||
description="过去 {window} 个交易日收益率波动率",
|
||||
formula="std(pct_change, {window})",
|
||||
brief="低波动防御:近段波动小的股票抗跌,弱市/熊市阶段相对占优(方向越低越好)。",
|
||||
fn=lambda fields, params: _rolling_vol(fields["close"], params[P_WINDOW]),
|
||||
param_specs=(_window_spec(),),
|
||||
direction_default=DIRECTION_LOWER,
|
||||
lookback_of=lambda params: params[P_WINDOW],
|
||||
instances=(
|
||||
("volatility_20", {P_WINDOW: 20}),
|
||||
("volatility_60", {P_WINDOW: 60}),
|
||||
),
|
||||
)
|
||||
)
|
||||
def _momentum_60(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return _rolling_return(fields["close"], 60)
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"momentum_120",
|
||||
"过去 120 个交易日收益率",
|
||||
"close / close.shift(120) - 1",
|
||||
brief="长期动量:反映近半年强势,适合大级别趋势;换手慢、回撤修复慢,弱市慎用。",
|
||||
lookback=120,
|
||||
)
|
||||
)
|
||||
def _momentum_120(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return _rolling_return(fields["close"], 120)
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"volatility_20",
|
||||
"过去 20 个交易日收益率波动率",
|
||||
"std(pct_change, 20)",
|
||||
brief="低波防御(方向 lower_is_better):近月波动小的股票抗跌,弱市/熊市阶段相对占优。",
|
||||
lookback=20,
|
||||
direction="lower_is_better",
|
||||
)
|
||||
)
|
||||
def _volatility_20(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return _rolling_vol(fields["close"], 20)
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"volatility_60",
|
||||
"过去 60 个交易日收益率波动率",
|
||||
"std(pct_change, 60)",
|
||||
brief="低波动(方向 lower_is_better):近一季低波组合长期回测常有超额,是防御型核心因子。",
|
||||
lookback=60,
|
||||
direction="lower_is_better",
|
||||
)
|
||||
)
|
||||
def _volatility_60(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return _rolling_vol(fields["close"], 60)
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"close_to_high_60",
|
||||
"收盘价相对 60 日最高价的接近程度",
|
||||
"close / rolling_max(high, 60)",
|
||||
brief="贴近 60 日高点(接近新高):趋势确认型强势股,常与动量互补;需配合市场热度判断。",
|
||||
lookback=60,
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="close_to_high",
|
||||
label="接近新高",
|
||||
description="收盘价相对 {window} 日最高价的接近程度",
|
||||
formula="close / rolling_max(high, {window})",
|
||||
brief="贴近 n 日高点(接近新高):趋势确认型强势股,常与动量互补;需配合市场热度判断。",
|
||||
fn=lambda fields, params: fields["close"] / fields["high"].rolling(params[P_WINDOW]).max(),
|
||||
param_specs=(_window_spec(),),
|
||||
requires=("close", "high"),
|
||||
lookback_of=lambda params: params[P_WINDOW],
|
||||
instances=(("close_to_high_60", {P_WINDOW: 60}),),
|
||||
)
|
||||
)
|
||||
def _close_to_high_60(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
high = fields["high"]
|
||||
return fields["close"] / high.rolling(60).max()
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"volume_ratio_5_60",
|
||||
"量比:5 日均量 / 60 日均量",
|
||||
"mean(volume, 5) / mean(volume, 60)",
|
||||
def _check_fast_slow(params: Mapping[str, Any]) -> str | None:
|
||||
if params[P_FAST] >= params[P_SLOW]:
|
||||
return f"快线窗口({params[P_FAST]}) 必须小于慢线窗口({params[P_SLOW]})"
|
||||
return None
|
||||
|
||||
|
||||
def _volume_ratio_label(params: Mapping[str, Any]) -> str:
|
||||
direction = "越高越好" if params[P_DIRECTION] == DIRECTION_HIGHER else "越低越好"
|
||||
return f"量比({params[P_FAST]}/{params[P_SLOW]} 日,{direction})"
|
||||
|
||||
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="volume_ratio",
|
||||
label="量比",
|
||||
description="量比:{fast} 日均量 / {slow} 日均量",
|
||||
formula="mean(volume, {fast}) / mean(volume, {slow})",
|
||||
brief="量比放大提示资金关注(短线活跃型);高换手也伴随更高波动,注意与波动因子搭配。",
|
||||
lookback=60,
|
||||
fn=lambda fields, params: (
|
||||
fields["volume"].rolling(params[P_FAST]).mean()
|
||||
/ fields["volume"].rolling(params[P_SLOW]).mean()
|
||||
),
|
||||
param_specs=(
|
||||
ParamSpec(
|
||||
name=P_FAST,
|
||||
label="快线",
|
||||
kind="int",
|
||||
default=5,
|
||||
minimum=WINDOW_MIN,
|
||||
maximum=WINDOW_MAX,
|
||||
note="短窗口天数,必须小于慢线。",
|
||||
),
|
||||
ParamSpec(
|
||||
name=P_SLOW,
|
||||
label="慢线",
|
||||
kind="int",
|
||||
default=60,
|
||||
minimum=WINDOW_MIN,
|
||||
maximum=WINDOW_MAX,
|
||||
note="长窗口天数,决定回看长度。",
|
||||
),
|
||||
),
|
||||
requires=("volume",),
|
||||
lookback_of=lambda params: params[P_SLOW],
|
||||
check=_check_fast_slow,
|
||||
instances=(("volume_ratio_5_60", {P_FAST: 5, P_SLOW: 60}),),
|
||||
label_of=_volume_ratio_label,
|
||||
)
|
||||
)
|
||||
def _volume_ratio_5_60(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
vol = fields["volume"]
|
||||
return vol.rolling(5).mean() / vol.rolling(60).mean()
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"ma_bias_20",
|
||||
"20 日均线乖离率",
|
||||
"(close - ma(close, 20)) / ma(close, 20)",
|
||||
brief="20 日均线乖离:上行趋势中正乖离偏强;乖离过大易回落,需警惕过热。",
|
||||
lookback=20,
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="ma_bias",
|
||||
label="均线乖离",
|
||||
description="{window} 日均线乖离率",
|
||||
formula="(close - ma(close, {window})) / ma(close, {window})",
|
||||
brief="均线乖离:上行趋势中正乖离偏强;乖离过大易回落,需警惕过热。",
|
||||
fn=lambda fields, params: (
|
||||
(fields["close"] - fields["close"].rolling(params[P_WINDOW]).mean())
|
||||
/ fields["close"].rolling(params[P_WINDOW]).mean()
|
||||
),
|
||||
param_specs=(_window_spec(),),
|
||||
lookback_of=lambda params: params[P_WINDOW],
|
||||
instances=(("ma_bias_20", {P_WINDOW: 20}),),
|
||||
)
|
||||
)
|
||||
def _ma_bias_20(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
close = fields["close"]
|
||||
ma = close.rolling(20).mean()
|
||||
return (close - ma) / ma
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"reversal_5",
|
||||
"短期反转:过去 5 日收益率取负(越低越接近超跌)",
|
||||
"-1 * (close / close.shift(5) - 1)",
|
||||
brief="短期反转(方向 higher_is_better):前期跌幅大的超跌反弹机会,适合震荡/修复行情。",
|
||||
lookback=5,
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="reversal",
|
||||
label="短期反转",
|
||||
description="短期反转:过去 {window} 日收益率取负",
|
||||
formula="-1 * (close / close.shift({window}) - 1)",
|
||||
brief="短期反转:前期跌幅大的超跌反弹机会,适合震荡/修复行情。",
|
||||
fn=lambda fields, params: -1.0 * _rolling_return(fields["close"], params[P_WINDOW]),
|
||||
param_specs=(_window_spec(),),
|
||||
lookback_of=lambda params: params[P_WINDOW],
|
||||
instances=(("reversal_5", {P_WINDOW: 5}),),
|
||||
)
|
||||
)
|
||||
def _reversal_5(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return -1.0 * _rolling_return(fields["close"], 5)
|
||||
|
||||
|
||||
# ---------- 每日指标(daily_basic)因子 ----------
|
||||
# 数据来源:daily_basic 表(Tushare daily_basic 接口),由 ResearchService / SelectionService
|
||||
# 装配后并入 daily 长表(见 quant/service.load_basic_df)。requires 里的列名即
|
||||
# domain.entities.market.DAILY_BASIC_NUMERIC_FIELDS 中的列。
|
||||
|
||||
# 特别分红导致的股息率畸高阈值(%):dv_ratio 会因一次性特别分红冲到 30%+,
|
||||
# 直接用「最高股息率」排序会被这类非经常性事件占满头部(实测 600738 在 2020-01-02
|
||||
# 为 37.2%)。本因子不隐式截断(截断属选股条件,应由用户在 conditions 里显式配置),
|
||||
# 但把阈值作为常量暴露,供前端/条件模板引用。
|
||||
# 为 37.2%)。本因子不隐式截断(截断属选股条件,应由用户在 conditions 里显式配置)。
|
||||
# 注意:该常量目前**没有**任何代码引用(曾计划供条件模板引用);要按此上限过滤,
|
||||
# 请在策略条件里显式配置 dv_ratio <= 30,而不是指望因子内部截断。
|
||||
DIVIDEND_YIELD_SPECIAL_CAP_PCT = 30.0
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"dividend_yield",
|
||||
"股息率(近 12 个月现金分红 / 总市值 × 100,%)",
|
||||
"dv_ratio(Tushare daily_basic,逐日时点值)",
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="dividend_yield",
|
||||
label="股息率",
|
||||
description="股息率(近 12 个月现金分红 / 总市值 × 100,%)",
|
||||
formula="dv_ratio(Tushare daily_basic,逐日时点值)",
|
||||
brief=(
|
||||
"高股息:熊市/震荡市防御性较强,分红提供现金回报底;"
|
||||
"需警惕「高股息陷阱」——股息率高常因股价下跌或一次性特别分红,"
|
||||
"建议配合 dv_ratio 上限过滤与盈利质量条件使用。"
|
||||
),
|
||||
frequency="daily",
|
||||
lookback=0, # 时点截面值,无滚动窗口
|
||||
direction="higher_is_better",
|
||||
fn=lambda fields, params: fields["dv_ratio"],
|
||||
requires=("dv_ratio",),
|
||||
lookback_of=lambda params: 0, # 时点截面值,无滚动窗口
|
||||
instances=(("dividend_yield", {P_DIRECTION: DIRECTION_HIGHER}),),
|
||||
)
|
||||
)
|
||||
def _dividend_yield(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
"""股息率面板(index=trade_date, columns=symbol)。
|
||||
|
||||
直接取当日 dv_ratio 时点值:该值由数据源按「过去 12 个月现金分红 / 当日总市值」
|
||||
逐日重算,只含已发生事件,按 trade_date <= as_of 取值即无未来函数。
|
||||
缺失值保持 NaN(由复合分/排序统一 dropna 处理),不做 0 填充 —— 0 会被误读成
|
||||
「股息率为 0 的合格标的」,从而污染横截面排序。
|
||||
"""
|
||||
return fields["dv_ratio"]
|
||||
|
||||
|
||||
@register(
|
||||
FactorDef(
|
||||
"dividend_yield_ttm",
|
||||
"股息率 TTM(近 12 个月滚动现金分红 / 总市值 × 100,%)",
|
||||
"dv_ttm(Tushare daily_basic,逐日时点值)",
|
||||
brief="同 dividend_yield,但口径为 TTM;与 dv_ratio 多数日期取值一致,可作交叉验证。",
|
||||
frequency="daily",
|
||||
lookback=0,
|
||||
direction="higher_is_better",
|
||||
register_template(
|
||||
FactorTemplate(
|
||||
name="dividend_yield_ttm",
|
||||
label="股息率 TTM",
|
||||
description="股息率 TTM(近 12 个月滚动现金分红 / 总市值 × 100,%)",
|
||||
formula="dv_ttm(Tushare daily_basic,逐日时点值)",
|
||||
brief="同股息率,但口径为 TTM;与 dv_ratio 多数日期取值一致,可作交叉验证。",
|
||||
fn=lambda fields, params: fields["dv_ttm"],
|
||||
requires=("dv_ttm",),
|
||||
lookback_of=lambda params: 0,
|
||||
instances=(("dividend_yield_ttm", {P_DIRECTION: DIRECTION_HIGHER}),),
|
||||
)
|
||||
)
|
||||
def _dividend_yield_ttm(fields: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
||||
return fields["dv_ttm"]
|
||||
)
|
||||
@@ -207,6 +207,7 @@ def condition_needed_columns(query) -> set[str]:
|
||||
except FactorError:
|
||||
raise ValueError(
|
||||
f"条件字段未知:{f}(可用: 行情列/ma20/ma60/已注册因子/"
|
||||
"参数化因子键(如 momentum(window=90,direction=higher_is_better))/"
|
||||
"每日指标列(dv_ratio 等)/static.*/fundamental.*)"
|
||||
) from None
|
||||
needed.update(defn.requires)
|
||||
@@ -390,7 +391,13 @@ def _field_value(field, sym, statics, tech, financial):
|
||||
|
||||
|
||||
def _compare(left, right, op: str) -> bool:
|
||||
"""混合比较:None 视为不可用 → 除 ne 外不通过;数值/字符串分别处理。"""
|
||||
"""混合比较:None 视为不可用 → 除 ne 外不通过;数值/字符串分别处理。
|
||||
|
||||
任何类型不匹配(拿日期字段去比大小、in 的右侧不是列表…)一律返回 False,
|
||||
**绝不抛异常**:条件是用户可以随手改的输入,一个手滑的字段名不该把选股/回测
|
||||
打成 500。字段库(quant.condition_fields)会在源头上拒绝不可比较的字段,
|
||||
这里只是最后一道防线。
|
||||
"""
|
||||
if op == "ne":
|
||||
return left != right
|
||||
if left is None or right is None:
|
||||
@@ -403,12 +410,16 @@ def _compare(left, right, op: str) -> bool:
|
||||
# 字符串/其它:支持 eq/ne/in/not_in
|
||||
if op == "eq":
|
||||
return left == right
|
||||
if op == "in":
|
||||
return left in right
|
||||
if op == "not_in":
|
||||
return left not in right
|
||||
if op in ("in", "not_in"):
|
||||
try:
|
||||
return left in right if op == "in" else left not in right
|
||||
except TypeError: # 右侧不是容器 → 该条件无法求值
|
||||
return False
|
||||
if op in ("gt", "gte", "lt", "lte"):
|
||||
return _num_cmp(left, right, op) # 尝试数值,字符串会 ValueError → False
|
||||
try:
|
||||
return _num_cmp(left, right, op) # 尝试数值,字符串会 ValueError → False
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
return False
|
||||
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@ from app.domain.entities.market import (
|
||||
)
|
||||
from app.domain.entities.research import ResearchSpec
|
||||
from app.domain.entities.strategy import SelectionStrategy, StrategyDefinition
|
||||
from app.quant.condition_fields import get_field
|
||||
from app.quant.factors import FactorDef, FactorError, get_factor
|
||||
|
||||
# 选股策略(SelectionStrategy)不含回测参数,走 _describe_selection_only 专用分支;
|
||||
@@ -137,10 +138,12 @@ def _describe_selection_only(st, factor_meta) -> StrategyDoc:
|
||||
" ⚠ 持仓数量 / 持仓天数区间 / 调仓时机 / 起始资金 / 费率 / 复权口径 / 回测区间"
|
||||
"均不在本策略内 —— 它们在「回测组合」中指定,运行时与公共配置合并。"
|
||||
)
|
||||
# 步骤文案**不带序号**:前端把它渲染进 <ol>(StrategyDocCard),编号由列表提供;
|
||||
# 后端再写一遍「1. 2. 3.」会渲染成「1. 1. …」双重编号(2026-10 修正)。
|
||||
steps = [
|
||||
"1. 按股票池口径筛出候选 universe(市场 / 剔 ST / 上市天数 / 指数成分)。",
|
||||
"2." + (" 逐条求值过滤条件(AND),剔除不满足者。" if st.conditions else " (未设过滤条件,候选 = universe。)"),
|
||||
"3. 对剩余股票按上述因子打分并降序排列 → 得到候选排名(TopN 在回测组合里截取)。",
|
||||
"按股票池口径筛出候选 universe(市场 / 剔 ST / 上市天数 / 指数成分)。",
|
||||
"逐条求值过滤条件(AND),剔除不满足者。" if st.conditions else "未设过滤条件,候选 = universe。",
|
||||
"对剩余股票按上述因子打分并降序排列 → 得到候选排名(TopN 在回测组合里截取)。",
|
||||
]
|
||||
warnings.append(
|
||||
"本说明只覆盖选股口径;回测的资金/持仓/调仓/成本/区间由「回测组合」+「公共配置」决定,"
|
||||
@@ -187,13 +190,25 @@ def _describe_factors_from_specs(factor_specs, factor_meta, warnings) -> list[di
|
||||
return out
|
||||
|
||||
|
||||
def _unit_suffix(field: str, cond) -> str:
|
||||
"""字面量条件的**基准单位**后缀(引擎就是按它比较的)。
|
||||
|
||||
字段间比较(ref)不加:两侧同一单位,写出来只会误导。字段不在注册表里 → 不加,
|
||||
宁可少写也不猜单位(猜错就是静默的口径错误)。
|
||||
"""
|
||||
if getattr(cond, "ref", None) is not None:
|
||||
return ""
|
||||
d = get_field(field)
|
||||
return f" {d.unit}" if d is not None and d.unit else ""
|
||||
|
||||
|
||||
def _condition_lines(conditions, warnings) -> list[str]:
|
||||
"""把 ConditionSpec 列表渲染成可读行(复用既有字段域校验逻辑)。"""
|
||||
lines: list[str] = []
|
||||
for c in conditions:
|
||||
op = _OP_TEXT.get(c.op, c.op)
|
||||
right = f"字段 {c.ref}" if c.ref else f"{c.value}"
|
||||
lines.append(f"{c.field} {op} {right}")
|
||||
lines.append(f"{c.field} {op} {right}{_unit_suffix(c.field, c)}")
|
||||
return lines
|
||||
|
||||
|
||||
@@ -583,12 +598,15 @@ def _render_condition(cond) -> str:
|
||||
op = _OP_TEXT.get(cond.op, cond.op)
|
||||
if cond.ref is not None:
|
||||
right = cond.ref
|
||||
suffix = ""
|
||||
elif cond.op in ("in", "not_in"):
|
||||
items = cond.value if isinstance(cond.value, Sequence) else [cond.value]
|
||||
right = "[" + ", ".join(_fmt_value(v) for v in items) + "]"
|
||||
suffix = _unit_suffix(cond.field, cond)
|
||||
else:
|
||||
right = _fmt_value(cond.value)
|
||||
return f"{cond.field} {op} {right}"
|
||||
suffix = _unit_suffix(cond.field, cond)
|
||||
return f"{cond.field} {op} {right}{suffix}"
|
||||
|
||||
|
||||
def _is_known_field(field: str) -> bool:
|
||||
|
||||
Reference in New Issue
Block a user