perf(backend): 内存优化三项——全市场研究不再占满 8G
1) 数据装配流式+列裁剪:Repository 新增 stream_range_many_columns(只 SELECT 所需列、SQL 侧转 REAL、yield_per 分批),引擎按 required_columns 取数 (LocalEngine 仅 close+因子字段),消除 ORM/Decimal 全量物化; 2) 研究 Job 独立子进程执行(job.mode=subprocess):python -m app.cli.run_job 在子进程内设 RLIMIT_AS 上限,OOM 归档 failed 而非拖垮 API worker; 子进程异常退出由父进程补记 failed;并发上限 2; 3) 服务启动清理:残留 queued/running Job 标记 failed(防永久 running)。 实测同款全市场回测:uvicorn worker RSS 稳定 ~220MB,任务峰值内存由 4.1GB+ 降至 ~470MB,24s 完成并归档(此前 43s 未完成即 OOM)。 新增/更新测试 96 passed,ruff 干净。
This commit is contained in:
@@ -42,6 +42,15 @@ class SqlAlchemyJobRepository:
|
||||
for r in self._session.scalars(stmt).all()
|
||||
]
|
||||
|
||||
def list_by_status(self, status: str, limit: int = 100) -> list[JobRecord]:
|
||||
rows = self._session.scalars(
|
||||
select(JobModel)
|
||||
.where(JobModel.status == status)
|
||||
.order_by(JobModel.created_at)
|
||||
.limit(limit)
|
||||
).all()
|
||||
return [JobRecord.model_validate(r, from_attributes=True) for r in rows]
|
||||
|
||||
|
||||
class SqlAlchemyExperimentRepository:
|
||||
def __init__(self, session: Session) -> None:
|
||||
|
||||
@@ -8,11 +8,11 @@ Repository 以 domain.entities 类型进出(AGENT.md §10)。
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Sequence
|
||||
from collections.abc import Iterator, Sequence
|
||||
from datetime import date
|
||||
from typing import Any
|
||||
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy import Float, String, cast, select
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.domain.entities.market import (
|
||||
@@ -32,6 +32,9 @@ from app.infrastructure.persistence.sqlalchemy.models.market import (
|
||||
TradingCalendarModel,
|
||||
)
|
||||
|
||||
# 日线数值列白名单(研究面板只需这些;symbol/trade_date 恒返回)
|
||||
BAR_FLOAT_COLUMNS = ("open", "high", "low", "close", "volume", "amount")
|
||||
|
||||
# 实体类型 → (ORM Model, 幂等键列)
|
||||
_TABLE = {
|
||||
Stock: (StockModel, ["symbol"]),
|
||||
@@ -167,6 +170,41 @@ class SqlAlchemyDailyBarRepository:
|
||||
).all()
|
||||
return [DailyBar.model_validate(r, from_attributes=True) for r in rows]
|
||||
|
||||
def stream_range_many_columns(
|
||||
self,
|
||||
symbols: Sequence[str],
|
||||
start: date,
|
||||
end: date,
|
||||
columns: Sequence[str],
|
||||
) -> Iterator[tuple]:
|
||||
"""流式返回 (symbol, trade_date_iso, *float_cols) 元组,分批拉取。
|
||||
|
||||
内存优化:与 get_range_many 不同,不实例化 ORM 对象 / Decimal,
|
||||
只 SELECT 所需列并在 SQL 侧 CAST 为 REAL,适合一次装配几十万~几百万行面板。
|
||||
"""
|
||||
cols = list(columns)
|
||||
unknown = [c for c in cols if c not in BAR_FLOAT_COLUMNS]
|
||||
if unknown:
|
||||
raise ValueError(f"不支持的行情列: {unknown}(可用: {BAR_FLOAT_COLUMNS})")
|
||||
numeric_expr = [cast(getattr(StockDailyModel, c), Float) for c in cols]
|
||||
stmt = (
|
||||
select(StockDailyModel.symbol, cast(StockDailyModel.trade_date, String), *numeric_expr)
|
||||
.where(
|
||||
StockDailyModel.symbol.in_(list(symbols)),
|
||||
StockDailyModel.trade_date >= start,
|
||||
StockDailyModel.trade_date <= end,
|
||||
)
|
||||
.order_by(StockDailyModel.symbol, StockDailyModel.trade_date)
|
||||
.execution_options(yield_per=20000)
|
||||
)
|
||||
result = self._session.execute(stmt)
|
||||
while True:
|
||||
chunk = result.fetchmany(20000)
|
||||
if not chunk:
|
||||
break
|
||||
for row in chunk:
|
||||
yield tuple(row)
|
||||
|
||||
def latest_date(self, symbol: str) -> date | None:
|
||||
return self._session.scalar(
|
||||
select(StockDailyModel.trade_date)
|
||||
|
||||
Reference in New Issue
Block a user