中"生成于 YYYY-MM-DD HH:MM:SS"解析生成时间。"""
p = soup.select_one("header p")
if p:
m = re.search(r"生成于 (\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2})", p.get_text())
if m:
return datetime.strptime(m.group(1), "%Y-%m-%d %H:%M:%S")
return fallback
# --------------------------------------------------------------------------- #
# AI 摘要
# --------------------------------------------------------------------------- #
def _parse_ai_summary(soup: BeautifulSoup) -> str | None:
"""AI 摘要:,li 逐行输出。"""
div = soup.select_one("div.ai-summary")
if div is None:
return None
items = [li.get_text(strip=True) for li in div.find_all("li") if li.get_text(strip=True)]
if items:
return "\n".join(items)
text = re.sub(r"\s+", " ", div.get_text(strip=True))
return text or None
# --------------------------------------------------------------------------- #
# 事件表解析
# --------------------------------------------------------------------------- #
def _to_int(text: str) -> int | None:
m = re.search(r"\d+", text or "")
return int(m.group(0)) if m else None
def _parse_summary(cell: Tag) -> tuple[str | None, str | None]:
"""摘要列:剥除 [来源],返回 (摘要, 来源)。"""
source = None
small = cell.find("small")
if small:
m = re.search(r"\[([^\]]+)\]", small.get_text())
if m:
source = m.group(1)
small.decompose()
text = re.sub(r"\s+", " ", cell.get_text(strip=True))
return (text or None, source)
def _clean_title(cell: Tag) -> str:
"""标题列:剥除 股票代码标注等,返回纯标题。"""
for small in cell.find_all("small"):
small.decompose()
return re.sub(r"\s+", " ", cell.get_text(strip=True))
def _parse_event_table(table: Tag, section: str) -> list[EventRow]:
"""事件表 → EventRow 列表。表头驱动列映射,兼容 xwlb/news/cninfo/intl 四类表。"""
rows = table.find_all("tr")
if len(rows) < 2:
return []
header_cells = rows[0].find_all(["th", "td"])
col_index: dict[str, int] = {}
icon_col: int | None = None
for i, cell in enumerate(header_cells):
text = cell.get_text(strip=True)
if text:
col_index[text] = i
elif icon_col is None:
icon_col = i
out: list[EventRow] = []
for row in rows[1:]:
tds = row.find_all("td")
if not tds:
continue
def col(name: str, tds: list[Tag] = tds) -> Tag | None: # noqa: B008 - 绑定循环变量
idx = col_index.get(name)
return tds[idx] if idx is not None and idx < len(tds) else None
title_cell = col("标题")
if title_cell is None:
continue
a = title_cell.find("a")
url = a.get("href") if a else None
summary_cell = col("摘要")
summary: str | None = None
source: str | None = None
if summary_cell is not None:
summary, source = _parse_summary(summary_cell)
if source is None:
src_cell = col("源")
if src_cell is not None and src_cell.get_text(strip=True):
source = src_cell.get_text(strip=True)
sentiment = None
if icon_col is not None and icon_col < len(tds):
sentiment = _SENTIMENT_ICON.get(tds[icon_col].get_text(strip=True))
imp_cell = col("重要度")
type_cell = col("事件类型")
out.append(
EventRow(
section=section,
rank=_to_int(tds[0].get_text(strip=True)) or len(out) + 1,
importance=_to_int(imp_cell.get_text(strip=True)) if imp_cell else None,
event_type=type_cell.get_text(strip=True) if type_cell else None,
title=_clean_title(title_cell),
summary=summary,
sentiment=sentiment,
source=source,
url=url,
)
)
return out
def _collect_events(soup: BeautifulSoup, report_type: str) -> list[EventRow]:
"""按 h2 板块标题收集各事件表。"""
events: list[EventRow] = []
for h2 in soup.find_all("h2"):
title = h2.get_text()
table = h2.find_next_sibling("table")
if table is None:
continue
section: str | None = None
if "新闻联播" in title:
section = "xwlb"
elif "公告" in title or "调研" in title:
section = "cninfo"
elif "重要事件" in title:
section = "intl" if report_type == "intl" else "news"
if section is not None:
events.extend(_parse_event_table(table, section))
return events
# --------------------------------------------------------------------------- #
# 数据总览 stats
# --------------------------------------------------------------------------- #
def _table_to_rows(table: Tag) -> list[dict[str, str]]:
"""表格 → [{表头: 值, ...}, ...](首行为表头)。"""
rows: list[list[str]] = []
for tr in table.find_all("tr"):
cells = [re.sub(r"\s+", " ", c.get_text(strip=True)) for c in tr.find_all(["th", "td"])]
if cells:
rows.append(cells)
if not rows:
return []
header = rows[0]
return [dict(zip(header, r, strict=False)) for r in rows[1:]]
def _parse_stats_block(h3: Tag) -> tuple[str, object] | None:
"""h3 数据总览区块 → (stats_key, value)。缺失/未知板块返回 None。"""
title = h3.get_text()
key = next((k for kw, k in _STATS_SECTION_KEYS if kw in title), None)
if key is None:
return None
block = h3.find_next_sibling()
if block is None:
return key, {}
if block.name == "table":
return key, _table_to_rows(block)
classes = block.get("class", []) if isinstance(block.get("class"), list) else []
if "stats-grid" in classes:
cards: dict[str, str | int] = {}
for card in block.find_all("div", class_="stat-card"):
label = card.select_one(".label")
num = card.select_one(".num")
if label is not None:
num_text = num.get_text(strip=True) if num else ""
cards[label.get_text(strip=True)] = _to_int(num_text) if _to_int(num_text) is not None else num_text
return key, cards
if "source-grid" in classes:
items: dict[str, str | int] = {}
for item in block.find_all("div", class_="source-item"):
name = item.select_one(".s-name")
count = item.select_one(".s-count")
if name is not None:
count_text = count.get_text(strip=True) if count else ""
items[name.get_text(strip=True)] = _to_int(count_text) or count_text
return key, items
if "sentiment-bar" in classes:
legend = block.find_next_sibling("div", class_="sentiment-legend")
spans = legend.find_all("span") if legend else []
return key, [re.sub(r"\s+", " ", s.get_text(strip=True)) for s in spans]
text = re.sub(r"\s+", " ", block.get_text(strip=True))
return key, text[:500]
def _collect_stats(soup: BeautifulSoup) -> dict[str, object]:
stats: dict[str, object] = {}
for h3 in soup.find_all("h3"):
parsed = _parse_stats_block(h3)
if parsed is not None:
stats[parsed[0]] = parsed[1]
return stats
# --------------------------------------------------------------------------- #
# 主入口
# --------------------------------------------------------------------------- #
def _parse(html: str, file_name: str, report_type: str) -> ReportData:
parsed = _parse_datetime_from_filename(file_name)
if parsed is None:
raise ReportParseError(f"无法从文件名解析日报日期: {file_name}")
day, gen_from_file = parsed
soup = BeautifulSoup(html, "html.parser")
generated_at = _parse_header_generated_at(soup, gen_from_file)
return ReportData(
report_date=day,
report_type=report_type,
file_name=file_name,
generated_at=generated_at,
ai_summary=_parse_ai_summary(soup),
stats=_collect_stats(soup),
events=_collect_events(soup, report_type),
)
def parse_finance_report(html: str, file_name: str) -> ReportData:
"""解析 A 股日报 finance_news_daily_*.html。"""
return _parse(html, file_name, "finance")
def parse_intl_report(html: str, file_name: str) -> ReportData:
"""解析国际财经日报 intl_news_daily_*.html。"""
return _parse(html, file_name, "intl")
def parse_report(html: str, file_name: str) -> ReportData:
"""按文件名前缀自动分流 finance / intl。"""
if "intl_news_daily" in file_name:
return parse_intl_report(html, file_name)
return parse_finance_report(html, file_name)