- 新增 report_db 包: MySQL 连接/建表/幂等写入 (news_report/news_event, myquant 库) - 新增 report_import 包: 历史 178 份日报 HTML 解析入库, 表头驱动列映射 - reporter.py 完全切换: generate_report 结构化入库, 不再生成/上传 HTML - CLI: 新增 report-import 子命令 - 依赖: uv add pymysql; 配置: NEWS_DB_* / REPORT_HISTORY_DIR - 文档: docs/report_db_design.md(实现逻辑), docs/db_schema.md(表结构供 API/前端) - 测试: 24 个单测通过 (parser/builder/models/importer)
299 lines
10 KiB
Python
299 lines
10 KiB
Python
"""历史日报 HTML 解析器(finance / intl)。
|
||
|
||
策略:表头驱动列映射,不依赖列位置;板块按 h2 标题识别;
|
||
数据总览按 h3 标题归类为 stats JSON 快照(前端自行解析)。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
from datetime import date, datetime
|
||
|
||
from bs4 import BeautifulSoup, Tag
|
||
|
||
from report_db.models import EventRow, ReportData
|
||
|
||
# 情绪图标 → sentiment 值
|
||
_SENTIMENT_ICON: dict[str, str] = {"⚪": "neutral", "🔴": "negative", "🟢": "positive"}
|
||
|
||
# 数据总览 h3 标题关键词 → stats key
|
||
_STATS_SECTION_KEYS: list[tuple[str, str]] = [
|
||
("管道", "pipeline"),
|
||
("各源", "sources"),
|
||
("情绪", "sentiment"),
|
||
("重要度", "importance"),
|
||
("事件类型", "event_types"),
|
||
("来源", "source_dist"),
|
||
]
|
||
|
||
|
||
class ReportParseError(Exception):
|
||
"""整份文件解析失败。"""
|
||
|
||
|
||
# --------------------------------------------------------------------------- #
|
||
# 文件名 / 时间解析
|
||
# --------------------------------------------------------------------------- #
|
||
|
||
def _parse_datetime_from_filename(file_name: str) -> tuple[date, datetime] | None:
|
||
"""从文件名解析日报日期与生成时间。
|
||
|
||
支持 `{type}_news_daily_{YYYYMMDD}.html` 与带时间戳的
|
||
`{type}_news_daily_{YYYYMMDD}_{HHMMSS}.html` / `..._{HHMM}.html`。
|
||
"""
|
||
m = re.search(r"_daily_(\d{8})(?:_(\d{4})(\d{2})?)?", file_name)
|
||
if not m:
|
||
return None
|
||
day = date(int(m.group(1)[:4]), int(m.group(1)[4:6]), int(m.group(1)[6:8]))
|
||
hh = mm = ss = 0
|
||
if m.group(2):
|
||
hh, mm = int(m.group(2)[:2]), int(m.group(2)[2:4])
|
||
ss = int(m.group(3) or 0)
|
||
return day, datetime(day.year, day.month, day.day, hh, mm, ss)
|
||
|
||
|
||
def _parse_header_generated_at(soup: BeautifulSoup, fallback: datetime) -> datetime:
|
||
"""从 <header> 中"生成于 YYYY-MM-DD HH:MM:SS"解析生成时间。"""
|
||
p = soup.select_one("header p")
|
||
if p:
|
||
m = re.search(r"生成于 (\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2})", p.get_text())
|
||
if m:
|
||
return datetime.strptime(m.group(1), "%Y-%m-%d %H:%M:%S")
|
||
return fallback
|
||
|
||
|
||
# --------------------------------------------------------------------------- #
|
||
# AI 摘要
|
||
# --------------------------------------------------------------------------- #
|
||
|
||
def _parse_ai_summary(soup: BeautifulSoup) -> str | None:
|
||
"""AI 摘要:<div class="ai-summary">,li 逐行输出。"""
|
||
div = soup.select_one("div.ai-summary")
|
||
if div is None:
|
||
return None
|
||
items = [li.get_text(strip=True) for li in div.find_all("li") if li.get_text(strip=True)]
|
||
if items:
|
||
return "\n".join(items)
|
||
text = re.sub(r"\s+", " ", div.get_text(strip=True))
|
||
return text or None
|
||
|
||
|
||
# --------------------------------------------------------------------------- #
|
||
# 事件表解析
|
||
# --------------------------------------------------------------------------- #
|
||
|
||
def _to_int(text: str) -> int | None:
|
||
m = re.search(r"\d+", text or "")
|
||
return int(m.group(0)) if m else None
|
||
|
||
|
||
def _parse_summary(cell: Tag) -> tuple[str | None, str | None]:
|
||
"""摘要列:剥除 <small>[来源]</small>,返回 (摘要, 来源)。"""
|
||
source = None
|
||
small = cell.find("small")
|
||
if small:
|
||
m = re.search(r"\[([^\]]+)\]", small.get_text())
|
||
if m:
|
||
source = m.group(1)
|
||
small.decompose()
|
||
text = re.sub(r"\s+", " ", cell.get_text(strip=True))
|
||
return (text or None, source)
|
||
|
||
|
||
def _clean_title(cell: Tag) -> str:
|
||
"""标题列:剥除 <small> 股票代码标注等,返回纯标题。"""
|
||
for small in cell.find_all("small"):
|
||
small.decompose()
|
||
return re.sub(r"\s+", " ", cell.get_text(strip=True))
|
||
|
||
|
||
def _parse_event_table(table: Tag, section: str) -> list[EventRow]:
|
||
"""事件表 → EventRow 列表。表头驱动列映射,兼容 xwlb/news/cninfo/intl 四类表。"""
|
||
rows = table.find_all("tr")
|
||
if len(rows) < 2:
|
||
return []
|
||
header_cells = rows[0].find_all(["th", "td"])
|
||
col_index: dict[str, int] = {}
|
||
icon_col: int | None = None
|
||
for i, cell in enumerate(header_cells):
|
||
text = cell.get_text(strip=True)
|
||
if text:
|
||
col_index[text] = i
|
||
elif icon_col is None:
|
||
icon_col = i
|
||
|
||
out: list[EventRow] = []
|
||
for row in rows[1:]:
|
||
tds = row.find_all("td")
|
||
if not tds:
|
||
continue
|
||
|
||
def col(name: str, tds: list[Tag] = tds) -> Tag | None: # noqa: B008 - 绑定循环变量
|
||
idx = col_index.get(name)
|
||
return tds[idx] if idx is not None and idx < len(tds) else None
|
||
|
||
title_cell = col("标题")
|
||
if title_cell is None:
|
||
continue
|
||
a = title_cell.find("a")
|
||
url = a.get("href") if a else None
|
||
|
||
summary_cell = col("摘要")
|
||
summary: str | None = None
|
||
source: str | None = None
|
||
if summary_cell is not None:
|
||
summary, source = _parse_summary(summary_cell)
|
||
if source is None:
|
||
src_cell = col("源")
|
||
if src_cell is not None and src_cell.get_text(strip=True):
|
||
source = src_cell.get_text(strip=True)
|
||
|
||
sentiment = None
|
||
if icon_col is not None and icon_col < len(tds):
|
||
sentiment = _SENTIMENT_ICON.get(tds[icon_col].get_text(strip=True))
|
||
|
||
imp_cell = col("重要度")
|
||
type_cell = col("事件类型")
|
||
out.append(
|
||
EventRow(
|
||
section=section,
|
||
rank=_to_int(tds[0].get_text(strip=True)) or len(out) + 1,
|
||
importance=_to_int(imp_cell.get_text(strip=True)) if imp_cell else None,
|
||
event_type=type_cell.get_text(strip=True) if type_cell else None,
|
||
title=_clean_title(title_cell),
|
||
summary=summary,
|
||
sentiment=sentiment,
|
||
source=source,
|
||
url=url,
|
||
)
|
||
)
|
||
return out
|
||
|
||
|
||
def _collect_events(soup: BeautifulSoup, report_type: str) -> list[EventRow]:
|
||
"""按 h2 板块标题收集各事件表。"""
|
||
events: list[EventRow] = []
|
||
for h2 in soup.find_all("h2"):
|
||
title = h2.get_text()
|
||
table = h2.find_next_sibling("table")
|
||
if table is None:
|
||
continue
|
||
section: str | None = None
|
||
if "新闻联播" in title:
|
||
section = "xwlb"
|
||
elif "公告" in title or "调研" in title:
|
||
section = "cninfo"
|
||
elif "重要事件" in title:
|
||
section = "intl" if report_type == "intl" else "news"
|
||
if section is not None:
|
||
events.extend(_parse_event_table(table, section))
|
||
return events
|
||
|
||
|
||
# --------------------------------------------------------------------------- #
|
||
# 数据总览 stats
|
||
# --------------------------------------------------------------------------- #
|
||
|
||
def _table_to_rows(table: Tag) -> list[dict[str, str]]:
|
||
"""表格 → [{表头: 值, ...}, ...](首行为表头)。"""
|
||
rows: list[list[str]] = []
|
||
for tr in table.find_all("tr"):
|
||
cells = [re.sub(r"\s+", " ", c.get_text(strip=True)) for c in tr.find_all(["th", "td"])]
|
||
if cells:
|
||
rows.append(cells)
|
||
if not rows:
|
||
return []
|
||
header = rows[0]
|
||
return [dict(zip(header, r, strict=False)) for r in rows[1:]]
|
||
|
||
|
||
def _parse_stats_block(h3: Tag) -> tuple[str, object] | None:
|
||
"""h3 数据总览区块 → (stats_key, value)。缺失/未知板块返回 None。"""
|
||
title = h3.get_text()
|
||
key = next((k for kw, k in _STATS_SECTION_KEYS if kw in title), None)
|
||
if key is None:
|
||
return None
|
||
block = h3.find_next_sibling()
|
||
if block is None:
|
||
return key, {}
|
||
|
||
if block.name == "table":
|
||
return key, _table_to_rows(block)
|
||
|
||
classes = block.get("class", []) if isinstance(block.get("class"), list) else []
|
||
if "stats-grid" in classes:
|
||
cards: dict[str, str | int] = {}
|
||
for card in block.find_all("div", class_="stat-card"):
|
||
label = card.select_one(".label")
|
||
num = card.select_one(".num")
|
||
if label is not None:
|
||
num_text = num.get_text(strip=True) if num else ""
|
||
cards[label.get_text(strip=True)] = _to_int(num_text) if _to_int(num_text) is not None else num_text
|
||
return key, cards
|
||
if "source-grid" in classes:
|
||
items: dict[str, str | int] = {}
|
||
for item in block.find_all("div", class_="source-item"):
|
||
name = item.select_one(".s-name")
|
||
count = item.select_one(".s-count")
|
||
if name is not None:
|
||
count_text = count.get_text(strip=True) if count else ""
|
||
items[name.get_text(strip=True)] = _to_int(count_text) or count_text
|
||
return key, items
|
||
if "sentiment-bar" in classes:
|
||
legend = block.find_next_sibling("div", class_="sentiment-legend")
|
||
spans = legend.find_all("span") if legend else []
|
||
return key, [re.sub(r"\s+", " ", s.get_text(strip=True)) for s in spans]
|
||
|
||
text = re.sub(r"\s+", " ", block.get_text(strip=True))
|
||
return key, text[:500]
|
||
|
||
|
||
def _collect_stats(soup: BeautifulSoup) -> dict[str, object]:
|
||
stats: dict[str, object] = {}
|
||
for h3 in soup.find_all("h3"):
|
||
parsed = _parse_stats_block(h3)
|
||
if parsed is not None:
|
||
stats[parsed[0]] = parsed[1]
|
||
return stats
|
||
|
||
|
||
# --------------------------------------------------------------------------- #
|
||
# 主入口
|
||
# --------------------------------------------------------------------------- #
|
||
|
||
def _parse(html: str, file_name: str, report_type: str) -> ReportData:
|
||
parsed = _parse_datetime_from_filename(file_name)
|
||
if parsed is None:
|
||
raise ReportParseError(f"无法从文件名解析日报日期: {file_name}")
|
||
day, gen_from_file = parsed
|
||
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
generated_at = _parse_header_generated_at(soup, gen_from_file)
|
||
|
||
return ReportData(
|
||
report_date=day,
|
||
report_type=report_type,
|
||
file_name=file_name,
|
||
generated_at=generated_at,
|
||
ai_summary=_parse_ai_summary(soup),
|
||
stats=_collect_stats(soup),
|
||
events=_collect_events(soup, report_type),
|
||
)
|
||
|
||
|
||
def parse_finance_report(html: str, file_name: str) -> ReportData:
|
||
"""解析 A 股日报 finance_news_daily_*.html。"""
|
||
return _parse(html, file_name, "finance")
|
||
|
||
|
||
def parse_intl_report(html: str, file_name: str) -> ReportData:
|
||
"""解析国际财经日报 intl_news_daily_*.html。"""
|
||
return _parse(html, file_name, "intl")
|
||
|
||
|
||
def parse_report(html: str, file_name: str) -> ReportData:
|
||
"""按文件名前缀自动分流 finance / intl。"""
|
||
if "intl_news_daily" in file_name:
|
||
return parse_intl_report(html, file_name)
|
||
return parse_finance_report(html, file_name)
|