"""历史日报 HTML 解析器(finance / intl)。 策略:表头驱动列映射,不依赖列位置;板块按 h2 标题识别; 数据总览按 h3 标题归类为 stats JSON 快照(前端自行解析)。 """ from __future__ import annotations import re from datetime import date, datetime from bs4 import BeautifulSoup, Tag from report_db.models import EventRow, ReportData # 情绪图标 → sentiment 值 _SENTIMENT_ICON: dict[str, str] = {"⚪": "neutral", "🔴": "negative", "🟢": "positive"} # 数据总览 h3 标题关键词 → stats key _STATS_SECTION_KEYS: list[tuple[str, str]] = [ ("管道", "pipeline"), ("各源", "sources"), ("情绪", "sentiment"), ("重要度", "importance"), ("事件类型", "event_types"), ("来源", "source_dist"), ] class ReportParseError(Exception): """整份文件解析失败。""" # --------------------------------------------------------------------------- # # 文件名 / 时间解析 # --------------------------------------------------------------------------- # def _parse_datetime_from_filename(file_name: str) -> tuple[date, datetime] | None: """从文件名解析日报日期与生成时间。 支持 `{type}_news_daily_{YYYYMMDD}.html` 与带时间戳的 `{type}_news_daily_{YYYYMMDD}_{HHMMSS}.html` / `..._{HHMM}.html`。 """ m = re.search(r"_daily_(\d{8})(?:_(\d{4})(\d{2})?)?", file_name) if not m: return None day = date(int(m.group(1)[:4]), int(m.group(1)[4:6]), int(m.group(1)[6:8])) hh = mm = ss = 0 if m.group(2): hh, mm = int(m.group(2)[:2]), int(m.group(2)[2:4]) ss = int(m.group(3) or 0) return day, datetime(day.year, day.month, day.day, hh, mm, ss) def _parse_header_generated_at(soup: BeautifulSoup, fallback: datetime) -> datetime: """从
中"生成于 YYYY-MM-DD HH:MM:SS"解析生成时间。""" p = soup.select_one("header p") if p: m = re.search(r"生成于 (\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2})", p.get_text()) if m: return datetime.strptime(m.group(1), "%Y-%m-%d %H:%M:%S") return fallback # --------------------------------------------------------------------------- # # AI 摘要 # --------------------------------------------------------------------------- # def _parse_ai_summary(soup: BeautifulSoup) -> str | None: """AI 摘要:
,li 逐行输出。""" div = soup.select_one("div.ai-summary") if div is None: return None items = [li.get_text(strip=True) for li in div.find_all("li") if li.get_text(strip=True)] if items: return "\n".join(items) text = re.sub(r"\s+", " ", div.get_text(strip=True)) return text or None # --------------------------------------------------------------------------- # # 事件表解析 # --------------------------------------------------------------------------- # def _to_int(text: str) -> int | None: m = re.search(r"\d+", text or "") return int(m.group(0)) if m else None def _parse_summary(cell: Tag) -> tuple[str | None, str | None]: """摘要列:剥除 [来源],返回 (摘要, 来源)。""" source = None small = cell.find("small") if small: m = re.search(r"\[([^\]]+)\]", small.get_text()) if m: source = m.group(1) small.decompose() text = re.sub(r"\s+", " ", cell.get_text(strip=True)) return (text or None, source) def _clean_title(cell: Tag) -> str: """标题列:剥除 股票代码标注等,返回纯标题。""" for small in cell.find_all("small"): small.decompose() return re.sub(r"\s+", " ", cell.get_text(strip=True)) def _parse_event_table(table: Tag, section: str) -> list[EventRow]: """事件表 → EventRow 列表。表头驱动列映射,兼容 xwlb/news/cninfo/intl 四类表。""" rows = table.find_all("tr") if len(rows) < 2: return [] header_cells = rows[0].find_all(["th", "td"]) col_index: dict[str, int] = {} icon_col: int | None = None for i, cell in enumerate(header_cells): text = cell.get_text(strip=True) if text: col_index[text] = i elif icon_col is None: icon_col = i out: list[EventRow] = [] for row in rows[1:]: tds = row.find_all("td") if not tds: continue def col(name: str, tds: list[Tag] = tds) -> Tag | None: # noqa: B008 - 绑定循环变量 idx = col_index.get(name) return tds[idx] if idx is not None and idx < len(tds) else None title_cell = col("标题") if title_cell is None: continue a = title_cell.find("a") url = a.get("href") if a else None summary_cell = col("摘要") summary: str | None = None source: str | None = None if summary_cell is not None: summary, source = _parse_summary(summary_cell) if source is None: src_cell = col("源") if src_cell is not None and src_cell.get_text(strip=True): source = src_cell.get_text(strip=True) sentiment = None if icon_col is not None and icon_col < len(tds): sentiment = _SENTIMENT_ICON.get(tds[icon_col].get_text(strip=True)) imp_cell = col("重要度") type_cell = col("事件类型") out.append( EventRow( section=section, rank=_to_int(tds[0].get_text(strip=True)) or len(out) + 1, importance=_to_int(imp_cell.get_text(strip=True)) if imp_cell else None, event_type=type_cell.get_text(strip=True) if type_cell else None, title=_clean_title(title_cell), summary=summary, sentiment=sentiment, source=source, url=url, ) ) return out def _collect_events(soup: BeautifulSoup, report_type: str) -> list[EventRow]: """按 h2 板块标题收集各事件表。""" events: list[EventRow] = [] for h2 in soup.find_all("h2"): title = h2.get_text() table = h2.find_next_sibling("table") if table is None: continue section: str | None = None if "新闻联播" in title: section = "xwlb" elif "公告" in title or "调研" in title: section = "cninfo" elif "重要事件" in title: section = "intl" if report_type == "intl" else "news" if section is not None: events.extend(_parse_event_table(table, section)) return events # --------------------------------------------------------------------------- # # 数据总览 stats # --------------------------------------------------------------------------- # def _table_to_rows(table: Tag) -> list[dict[str, str]]: """表格 → [{表头: 值, ...}, ...](首行为表头)。""" rows: list[list[str]] = [] for tr in table.find_all("tr"): cells = [re.sub(r"\s+", " ", c.get_text(strip=True)) for c in tr.find_all(["th", "td"])] if cells: rows.append(cells) if not rows: return [] header = rows[0] return [dict(zip(header, r, strict=False)) for r in rows[1:]] def _parse_stats_block(h3: Tag) -> tuple[str, object] | None: """h3 数据总览区块 → (stats_key, value)。缺失/未知板块返回 None。""" title = h3.get_text() key = next((k for kw, k in _STATS_SECTION_KEYS if kw in title), None) if key is None: return None block = h3.find_next_sibling() if block is None: return key, {} if block.name == "table": return key, _table_to_rows(block) classes = block.get("class", []) if isinstance(block.get("class"), list) else [] if "stats-grid" in classes: cards: dict[str, str | int] = {} for card in block.find_all("div", class_="stat-card"): label = card.select_one(".label") num = card.select_one(".num") if label is not None: num_text = num.get_text(strip=True) if num else "" cards[label.get_text(strip=True)] = _to_int(num_text) if _to_int(num_text) is not None else num_text return key, cards if "source-grid" in classes: items: dict[str, str | int] = {} for item in block.find_all("div", class_="source-item"): name = item.select_one(".s-name") count = item.select_one(".s-count") if name is not None: count_text = count.get_text(strip=True) if count else "" items[name.get_text(strip=True)] = _to_int(count_text) or count_text return key, items if "sentiment-bar" in classes: legend = block.find_next_sibling("div", class_="sentiment-legend") spans = legend.find_all("span") if legend else [] return key, [re.sub(r"\s+", " ", s.get_text(strip=True)) for s in spans] text = re.sub(r"\s+", " ", block.get_text(strip=True)) return key, text[:500] def _collect_stats(soup: BeautifulSoup) -> dict[str, object]: stats: dict[str, object] = {} for h3 in soup.find_all("h3"): parsed = _parse_stats_block(h3) if parsed is not None: stats[parsed[0]] = parsed[1] return stats # --------------------------------------------------------------------------- # # 主入口 # --------------------------------------------------------------------------- # def _parse(html: str, file_name: str, report_type: str) -> ReportData: parsed = _parse_datetime_from_filename(file_name) if parsed is None: raise ReportParseError(f"无法从文件名解析日报日期: {file_name}") day, gen_from_file = parsed soup = BeautifulSoup(html, "html.parser") generated_at = _parse_header_generated_at(soup, gen_from_file) return ReportData( report_date=day, report_type=report_type, file_name=file_name, generated_at=generated_at, ai_summary=_parse_ai_summary(soup), stats=_collect_stats(soup), events=_collect_events(soup, report_type), ) def parse_finance_report(html: str, file_name: str) -> ReportData: """解析 A 股日报 finance_news_daily_*.html。""" return _parse(html, file_name, "finance") def parse_intl_report(html: str, file_name: str) -> ReportData: """解析国际财经日报 intl_news_daily_*.html。""" return _parse(html, file_name, "intl") def parse_report(html: str, file_name: str) -> ReportData: """按文件名前缀自动分流 finance / intl。""" if "intl_news_daily" in file_name: return parse_intl_report(html, file_name) return parse_finance_report(html, file_name)