Files
simon 366e60e8a9 feat: 日报结构化入库(M10 前后端分离数据层)
- 新增 report_db 包: MySQL 连接/建表/幂等写入 (news_report/news_event, myquant 库)
- 新增 report_import 包: 历史 178 份日报 HTML 解析入库, 表头驱动列映射
- reporter.py 完全切换: generate_report 结构化入库, 不再生成/上传 HTML
- CLI: 新增 report-import 子命令
- 依赖: uv add pymysql; 配置: NEWS_DB_* / REPORT_HISTORY_DIR
- 文档: docs/report_db_design.md(实现逻辑), docs/db_schema.md(表结构供 API/前端)
- 测试: 24 个单测通过 (parser/builder/models/importer)
2026-08-03 21:32:07 +08:00

299 lines
10 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""历史日报 HTML 解析器(finance / intl)。
策略:表头驱动列映射,不依赖列位置;板块按 h2 标题识别;
数据总览按 h3 标题归类为 stats JSON 快照(前端自行解析)。
"""
from __future__ import annotations
import re
from datetime import date, datetime
from bs4 import BeautifulSoup, Tag
from report_db.models import EventRow, ReportData
# 情绪图标 → sentiment 值
_SENTIMENT_ICON: dict[str, str] = {"⚪": "neutral", "🔴": "negative", "🟢": "positive"}
# 数据总览 h3 标题关键词 → stats key
_STATS_SECTION_KEYS: list[tuple[str, str]] = [
("管道", "pipeline"),
("各源", "sources"),
("情绪", "sentiment"),
("重要度", "importance"),
("事件类型", "event_types"),
("来源", "source_dist"),
]
class ReportParseError(Exception):
"""整份文件解析失败。"""
# --------------------------------------------------------------------------- #
# 文件名 / 时间解析
# --------------------------------------------------------------------------- #
def _parse_datetime_from_filename(file_name: str) -> tuple[date, datetime] | None:
"""从文件名解析日报日期与生成时间。
支持 `{type}_news_daily_{YYYYMMDD}.html` 与带时间戳的
`{type}_news_daily_{YYYYMMDD}_{HHMMSS}.html` / `..._{HHMM}.html`。
"""
m = re.search(r"_daily_(\d{8})(?:_(\d{4})(\d{2})?)?", file_name)
if not m:
return None
day = date(int(m.group(1)[:4]), int(m.group(1)[4:6]), int(m.group(1)[6:8]))
hh = mm = ss = 0
if m.group(2):
hh, mm = int(m.group(2)[:2]), int(m.group(2)[2:4])
ss = int(m.group(3) or 0)
return day, datetime(day.year, day.month, day.day, hh, mm, ss)
def _parse_header_generated_at(soup: BeautifulSoup, fallback: datetime) -> datetime:
"""从 <header> 中"生成于 YYYY-MM-DD HH:MM:SS"解析生成时间。"""
p = soup.select_one("header p")
if p:
m = re.search(r"生成于 (\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2})", p.get_text())
if m:
return datetime.strptime(m.group(1), "%Y-%m-%d %H:%M:%S")
return fallback
# --------------------------------------------------------------------------- #
# AI 摘要
# --------------------------------------------------------------------------- #
def _parse_ai_summary(soup: BeautifulSoup) -> str | None:
"""AI 摘要:<div class="ai-summary">li 逐行输出。"""
div = soup.select_one("div.ai-summary")
if div is None:
return None
items = [li.get_text(strip=True) for li in div.find_all("li") if li.get_text(strip=True)]
if items:
return "\n".join(items)
text = re.sub(r"\s+", " ", div.get_text(strip=True))
return text or None
# --------------------------------------------------------------------------- #
# 事件表解析
# --------------------------------------------------------------------------- #
def _to_int(text: str) -> int | None:
m = re.search(r"\d+", text or "")
return int(m.group(0)) if m else None
def _parse_summary(cell: Tag) -> tuple[str | None, str | None]:
"""摘要列:剥除 <small>[来源]</small>,返回 (摘要, 来源)。"""
source = None
small = cell.find("small")
if small:
m = re.search(r"\[([^\]]+)\]", small.get_text())
if m:
source = m.group(1)
small.decompose()
text = re.sub(r"\s+", " ", cell.get_text(strip=True))
return (text or None, source)
def _clean_title(cell: Tag) -> str:
"""标题列:剥除 <small> 股票代码标注等,返回纯标题。"""
for small in cell.find_all("small"):
small.decompose()
return re.sub(r"\s+", " ", cell.get_text(strip=True))
def _parse_event_table(table: Tag, section: str) -> list[EventRow]:
"""事件表 → EventRow 列表。表头驱动列映射,兼容 xwlb/news/cninfo/intl 四类表。"""
rows = table.find_all("tr")
if len(rows) < 2:
return []
header_cells = rows[0].find_all(["th", "td"])
col_index: dict[str, int] = {}
icon_col: int | None = None
for i, cell in enumerate(header_cells):
text = cell.get_text(strip=True)
if text:
col_index[text] = i
elif icon_col is None:
icon_col = i
out: list[EventRow] = []
for row in rows[1:]:
tds = row.find_all("td")
if not tds:
continue
def col(name: str, tds: list[Tag] = tds) -> Tag | None: # noqa: B008 - 绑定循环变量
idx = col_index.get(name)
return tds[idx] if idx is not None and idx < len(tds) else None
title_cell = col("标题")
if title_cell is None:
continue
a = title_cell.find("a")
url = a.get("href") if a else None
summary_cell = col("摘要")
summary: str | None = None
source: str | None = None
if summary_cell is not None:
summary, source = _parse_summary(summary_cell)
if source is None:
src_cell = col("源")
if src_cell is not None and src_cell.get_text(strip=True):
source = src_cell.get_text(strip=True)
sentiment = None
if icon_col is not None and icon_col < len(tds):
sentiment = _SENTIMENT_ICON.get(tds[icon_col].get_text(strip=True))
imp_cell = col("重要度")
type_cell = col("事件类型")
out.append(
EventRow(
section=section,
rank=_to_int(tds[0].get_text(strip=True)) or len(out) + 1,
importance=_to_int(imp_cell.get_text(strip=True)) if imp_cell else None,
event_type=type_cell.get_text(strip=True) if type_cell else None,
title=_clean_title(title_cell),
summary=summary,
sentiment=sentiment,
source=source,
url=url,
)
)
return out
def _collect_events(soup: BeautifulSoup, report_type: str) -> list[EventRow]:
"""按 h2 板块标题收集各事件表。"""
events: list[EventRow] = []
for h2 in soup.find_all("h2"):
title = h2.get_text()
table = h2.find_next_sibling("table")
if table is None:
continue
section: str | None = None
if "新闻联播" in title:
section = "xwlb"
elif "公告" in title or "调研" in title:
section = "cninfo"
elif "重要事件" in title:
section = "intl" if report_type == "intl" else "news"
if section is not None:
events.extend(_parse_event_table(table, section))
return events
# --------------------------------------------------------------------------- #
# 数据总览 stats
# --------------------------------------------------------------------------- #
def _table_to_rows(table: Tag) -> list[dict[str, str]]:
"""表格 → [{表头: 值, ...}, ...](首行为表头)。"""
rows: list[list[str]] = []
for tr in table.find_all("tr"):
cells = [re.sub(r"\s+", " ", c.get_text(strip=True)) for c in tr.find_all(["th", "td"])]
if cells:
rows.append(cells)
if not rows:
return []
header = rows[0]
return [dict(zip(header, r, strict=False)) for r in rows[1:]]
def _parse_stats_block(h3: Tag) -> tuple[str, object] | None:
"""h3 数据总览区块 → (stats_key, value)。缺失/未知板块返回 None。"""
title = h3.get_text()
key = next((k for kw, k in _STATS_SECTION_KEYS if kw in title), None)
if key is None:
return None
block = h3.find_next_sibling()
if block is None:
return key, {}
if block.name == "table":
return key, _table_to_rows(block)
classes = block.get("class", []) if isinstance(block.get("class"), list) else []
if "stats-grid" in classes:
cards: dict[str, str | int] = {}
for card in block.find_all("div", class_="stat-card"):
label = card.select_one(".label")
num = card.select_one(".num")
if label is not None:
num_text = num.get_text(strip=True) if num else ""
cards[label.get_text(strip=True)] = _to_int(num_text) if _to_int(num_text) is not None else num_text
return key, cards
if "source-grid" in classes:
items: dict[str, str | int] = {}
for item in block.find_all("div", class_="source-item"):
name = item.select_one(".s-name")
count = item.select_one(".s-count")
if name is not None:
count_text = count.get_text(strip=True) if count else ""
items[name.get_text(strip=True)] = _to_int(count_text) or count_text
return key, items
if "sentiment-bar" in classes:
legend = block.find_next_sibling("div", class_="sentiment-legend")
spans = legend.find_all("span") if legend else []
return key, [re.sub(r"\s+", " ", s.get_text(strip=True)) for s in spans]
text = re.sub(r"\s+", " ", block.get_text(strip=True))
return key, text[:500]
def _collect_stats(soup: BeautifulSoup) -> dict[str, object]:
stats: dict[str, object] = {}
for h3 in soup.find_all("h3"):
parsed = _parse_stats_block(h3)
if parsed is not None:
stats[parsed[0]] = parsed[1]
return stats
# --------------------------------------------------------------------------- #
# 主入口
# --------------------------------------------------------------------------- #
def _parse(html: str, file_name: str, report_type: str) -> ReportData:
parsed = _parse_datetime_from_filename(file_name)
if parsed is None:
raise ReportParseError(f"无法从文件名解析日报日期: {file_name}")
day, gen_from_file = parsed
soup = BeautifulSoup(html, "html.parser")
generated_at = _parse_header_generated_at(soup, gen_from_file)
return ReportData(
report_date=day,
report_type=report_type,
file_name=file_name,
generated_at=generated_at,
ai_summary=_parse_ai_summary(soup),
stats=_collect_stats(soup),
events=_collect_events(soup, report_type),
)
def parse_finance_report(html: str, file_name: str) -> ReportData:
"""解析 A 股日报 finance_news_daily_*.html。"""
return _parse(html, file_name, "finance")
def parse_intl_report(html: str, file_name: str) -> ReportData:
"""解析国际财经日报 intl_news_daily_*.html。"""
return _parse(html, file_name, "intl")
def parse_report(html: str, file_name: str) -> ReportData:
"""按文件名前缀自动分流 finance / intl。"""
if "intl_news_daily" in file_name:
return parse_intl_report(html, file_name)
return parse_finance_report(html, file_name)