Files
qlib/backend/app/api/jobs.py
T
Simon 23972e7063 feat: 股息率案例口径 + 策略库与图表统一 + 回测存档完整化
汇总三轮未提交的开发(每轮均在本机 MariaDB + 真实浏览器上验证):

1) 股息率案例(全市场股息率最高 n 只,默认 20,每 m 月择股)
   - 新增日频估值表 daily_basic + 迁移;股息率因子(dv_ratio / dividend_yield / TTM)
   - 名称历史表 stock_name_history:剔除 ST 按**择股日当时名称**判定,消除
     「曾高股息后 ST」的股息陷阱(实测 3.70pp 偏差)
   - 区间择股/调仓双周期(m 择股 / y 调仓)、指数成分与白名单、停牌近似剔除
   - 复权因子口径核对(4,164,742 行、缺失 0.0%)、收盘价成交与涨跌停拦单
   - 案例实测:2020-01-01~2026-09-04 总收益 +24.86%(年化 3.52%、回撤 -28.58%)

2) 策略库与前端统一
   - strategy 表 + CRUD/PUT 原地更新 + `describe_strategy` 按 spec 真实推导
     「一句话说明 + 计算公式 + 执行步骤 + 注意事项」(与引擎实执行规则同源)
   - 任何出现股票代码处都成对显示名称且可点击进个股页
   - 全站图表基座统一 TradingView Lightweight Charts(ECharts 依赖、
     锁文件、组件与文档标注一并清除),买卖点标记只落在真实交易日上

3) 回测存档完整化(可往复查看)
   - 同步端点(POST /api/backtests、/api/factor-tests)此前完全不落库 → 现在同样归档,
     归档 id 经响应头 X-Experiment-Id 返回(不破坏 response_model)
   - data_version 首次真实写入(数据快照指纹:最新交易日 + 各表规模)
   - 个股收益曲线默认**全量保存**(此前硬截断 60 只);超出体积预算才裁剪,
     并写 archive_meta(机器可读)+ unimplemented(人可读)如实标注
   - 列表 kind/q 过滤 + X-Total-Count(此前 limit=50 静默截断)、DELETE 归档
   - 只读归档页 /experiments/{id}(Server Component,SSR 直出**选股条件**与
     **交易执行依据**);结果视图按 kind 分发(backtest/factor_test/selection),
     非回测归档不套用回测口径
   - 新增 CLI:prune_experiments(保留策略,默认 dry-run)、
     restore_experiment_from_job(从 Job 副本按原 id 重建被删的历史归档,默认 dry-run)

门禁:pytest 388 passed、ruff All checks passed、tsc 0 错误、图表单测 7 passed、
next build 成功、契约脚本 verify_strategy_workspace 59/59(含按 kind 逐类验证归档页)。
2026-09-20 07:31:04 +08:00

155 lines
5.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""异步研究 Job API(Phase 4):提交 / 查询 / SSE 进度。
POST /api/jobs 创建 Job(BackgroundTasks 后台执行),立即返回 job_id
GET /api/jobs/{id} 状态 + 结果(成功时内嵌 result)
GET /api/jobs/{id}/events SSE 进度(queued→running→success|failed)
执行模式见 config job.mode:subprocess 时研究任务在独立子进程跑(内存隔离),
API worker 不被重任务拖垮(内存优化专项)。
"""
from __future__ import annotations
import asyncio
import json
from datetime import datetime
from typing import Annotated
from fastapi import APIRouter, BackgroundTasks, HTTPException, Query
from fastapi.responses import StreamingResponse
from app.api.deps import DbSession, ExperimentRepoDep, JobRepoDep
from app.application.services.job_executor import new_id, run_job_background, terminate_active
from app.domain.entities.research import (
BacktestResult,
FactorTestReport,
JobRecord,
JobStatus,
ResearchSpec,
)
from app.infrastructure.persistence.sqlalchemy.repositories.jobs_impl import (
SqlAlchemyJobRepository,
)
from app.infrastructure.persistence.sqlalchemy.session import SessionLocal
router = APIRouter(prefix="/jobs", tags=["jobs"])
def _decode_result(kind: str, result_json: str | None):
from app.domain.entities.selection import SelectionResult
if result_json is None:
return None
if kind == "selection":
return SelectionResult.model_validate_json(result_json)
model = BacktestResult if kind == "backtest" else FactorTestReport
return model.model_validate_json(result_json)
def _job_view(job: JobRecord, experiment_repo=None) -> dict:
"""Job 视图:`result` 契约不变(成功时内嵌**完整**结果)。
结果来源(2026-09 起完整结果只在 experiment 存一份,job.result_json 不再重复写):
1. `job.experiment_id` 有值且能读到归档 → 解码 experiment.result_json;
2. 否则回退解码 `job.result_json`(老记录 / 归档被删除前的历史数据);
3. 归档被删除且 job 侧无副本 → `result=None`,并给出
`result_unavailable_reason` 如实说明原因(AGENT §7:不静默给空结果)。
"""
view = job.model_dump()
view.pop("result_json", None)
view["spec"] = json.loads(job.spec_json)
result = None
source = None
if job.experiment_id and experiment_repo is not None:
exp = experiment_repo.get(job.experiment_id)
if exp is not None:
result = _decode_result(job.kind, exp.result_json)
source = "experiment"
else:
view["result_unavailable_reason"] = (
f"归档 {job.experiment_id} 已不存在(可能已被删除);"
"完整结果仅存于归档,Job 记录本身不再保存结果副本"
)
if result is None and job.result_json:
result = _decode_result(job.kind, job.result_json)
source = "job"
view["result"] = result
view["result_source"] = source
return view
@router.post("", summary="创建异步研究 Job")
def create_job(
spec: ResearchSpec,
background: BackgroundTasks,
session: DbSession,
job_repo: JobRepoDep,
) -> dict:
job = JobRecord(
id=new_id("JOB"),
kind=spec.type,
spec_json=spec.model_dump_json(),
status=JobStatus.QUEUED,
created_at=datetime.now(),
)
job_repo.create(job)
session.commit()
background.add_task(run_job_background, job.id)
return {"job_id": job.id, "status": job.status}
@router.get("", summary="Job 列表")
def list_jobs(
job_repo: JobRepoDep,
experiment_repo: ExperimentRepoDep,
kind: Annotated[str | None, Query(description="按类型过滤(backtest/factor_test)")] = None,
limit: Annotated[int, Query(ge=1, le=200)] = 20,
) -> list[dict]:
return [
_job_view(j, experiment_repo) for j in job_repo.list_recent(kind=kind, limit=limit)
]
@router.post("/{job_id}/cancel", summary="取消 Job(queued/running)")
def cancel_job(job_id: str, session: DbSession, job_repo: JobRepoDep) -> dict:
job = job_repo.get(job_id)
if job is None:
raise HTTPException(status_code=404, detail=f"Job {job_id} 不存在")
if job.status not in (JobStatus.QUEUED, JobStatus.RUNNING):
return {"job_id": job_id, "status": job.status, "cancelled": False}
job.status = JobStatus.CANCELLED
job.stage = None
job_repo.update(job)
session.commit()
terminate_active(job_id) # 终止研究子进程(若有);父进程兜底已跳过 CANCELLED
return {"job_id": job_id, "status": JobStatus.CANCELLED, "cancelled": True}
@router.get("/{job_id}", summary="查询 Job 状态与结果")
def get_job(job_id: str, job_repo: JobRepoDep, experiment_repo: ExperimentRepoDep) -> dict:
job = job_repo.get(job_id)
if job is None:
raise HTTPException(status_code=404, detail=f"Job {job_id} 不存在")
return _job_view(job, experiment_repo)
@router.get("/{job_id}/events", summary="Job 进度 SSE")
async def job_events(job_id: str) -> StreamingResponse:
async def gen():
while True:
with SessionLocal() as session:
job = SqlAlchemyJobRepository(session).get(job_id)
if job is None:
yield "event: error\ndata: job not found\n\n"
return
payload = json.dumps(
{"job_id": job.id, "status": job.status, "stage": job.stage}, ensure_ascii=False
)
yield f"data: {payload}\n\n"
if job.status in (JobStatus.SUCCESS, JobStatus.FAILED, JobStatus.CANCELLED):
return
await asyncio.sleep(0.4)
return StreamingResponse(gen(), media_type="text/event-stream")