"""每日摘要报告生成器 v2.0。 输出 HTML 日报,包含: 一、AI 摘要 (最新新闻 + {cninfo_days}d 公告/调研/互动) 二、重要事件: 新闻 (最新抓取, importance≥4, 最多20篇) 三、重要事件: 公告/互动 ({cninfo_days}d, cninfo, 最多20篇) 四、数据总览 (重要度/各源/M1-M6/情绪/事件类型分布) """ from __future__ import annotations import json import os as _os import re as _re import subprocess import time from collections import Counter from datetime import date, datetime, timedelta from pathlib import Path from typing import Any from dotenv import load_dotenv from loguru import logger from report_db.models import EventRow, ReportData # noqa: F401 - 供 _build_report_data 注解使用 # 确保 .env 已加载(模块级常量依赖环境变量) load_dotenv() # --------------------------------------------------------------------------- # # 配置 # --------------------------------------------------------------------------- # UPLOAD_HOST = "simon@doorcome.cn" UPLOAD_BASE = "/var/www/html/echart/research" CNINFO_DAYS_BACK = int(_os.environ.get("STOCK_REPORT_DAYS", "15")) # 与个股日报共用参数, 默认值保持一致 NEWS_DAYS_BACK = 1 # 新闻回溯天数 _MAX_HIGH_EVENTS = 20 # LLM 摘要调用重试参数(环境变量可覆盖) _LLM_RETRY_TIMES = int(_os.environ.get("LLM_RETRY_TIMES", "3")) _LLM_RETRY_BACKOFF_SEC = float(_os.environ.get("LLM_RETRY_BACKOFF_SEC", "2.0")) # 日报新闻回溯窗口(小时):07:00 生成当日日报时覆盖昨日全天至今晨的新闻 _NEWS_LOOKBACK_HOURS = 30 def _load_source_names() -> dict[str, str]: import yaml try: with open("configs/sources.yaml", encoding="utf-8") as f: data = yaml.safe_load(f) return {s["id"]: s["name"] for s in (data.get("sources") or []) if s.get("id")} except Exception: return {} def _source_name(src_id: str) -> str: return _load_source_names().get(src_id, src_id) def _load_watchlist_codes() -> set[str]: import yaml try: with open("configs/watchlist.yaml", encoding="utf-8") as f: data = yaml.safe_load(f) or {} return {it["code"] for it in (data.get("watchlist") or []) if it.get("code")} except Exception: return set() # --------------------------------------------------------------------------- # # 数据收集 # --------------------------------------------------------------------------- # def _count_jsonl(path: Path) -> int: if not path.is_file(): return 0 return sum(1 for _ in open(path, encoding="utf-8")) def _count_json(pattern: str) -> int: return len(list(Path().glob(pattern))) def _load_events_from_dir(day_str: str) -> list[dict]: """从 data/events/{day_str}/ 加载所有事件。""" events: list[dict] = [] ev_dir = Path(f"data/events/{day_str}") if not ev_dir.is_dir(): return events for fp in sorted(ev_dir.glob("*.json")): try: obj = json.loads(fp.read_text(encoding="utf-8")) ev = obj.get("event", {}) events.append({ "title": obj.get("title", ""), "url": obj.get("url", ""), "source_id": obj.get("source_id", ""), "publish_time": obj.get("publish_time"), "event": ev, }) except (json.JSONDecodeError, OSError): pass return events def _collect_news_events(day_str: str) -> dict[str, Any]: """收集新闻事件(排除 cninfo)。 读取 `day_str` 与前一天两个事件目录,按 publish_time 过滤最近 `_NEWS_LOOKBACK_HOURS`(默认 30)小时内的新闻——07:00 生成当日日报时 可覆盖昨日全天至今晨的新闻。无 publish_time 的事件保留(容错)。 """ day = datetime.strptime(day_str, "%Y%m%d").date() prev_day = (day - timedelta(days=1)).strftime("%Y%m%d") all_ev = _load_events_from_dir(day_str) + _load_events_from_dir(prev_day) # publish_time 过滤: 最近 30 小时(时间缺失/格式异常的事件保留) cutoff = (datetime.now() - timedelta(hours=_NEWS_LOOKBACK_HOURS)).astimezone() news_ev: list[dict] = [] for e in all_ev: if e["source_id"] == "cninfo": continue pt = e.get("publish_time") if pt: try: # naive 时间假定为本地时区, 与带时区(aware)的 cutoff 统一比较 t = datetime.fromisoformat(pt) if t.tzinfo is None: t = t.astimezone() if t < cutoff: continue except (ValueError, TypeError): pass # 时间格式异常时保留 news_ev.append(e) sentiments: Counter = Counter() importances: Counter = Counter() event_types: Counter = Counter() for e in news_ev: ev = e["event"] sentiments[ev.get("sentiment", "?")] += 1 importances[ev.get("importance", 0)] += 1 event_types[ev.get("event_type", "?")] += 1 # 高重要度: 优先 ≥4, 不足时逐级回退(≥3 → ≥2 → 全部按 importance 排序) def _get_high(evs, threshold): return sorted( [e for e in evs if e["event"].get("importance", 0) >= threshold], key=lambda e: -e["event"].get("importance", 0), ) min_show = 3 high = _get_high(news_ev, 4) hi_threshold = 4 if len(high) < min_show: high = _get_high(news_ev, 3) hi_threshold = 3 if len(high) < min_show: high = _get_high(news_ev, 2) hi_threshold = 2 if len(high) < min_show: high = sorted(news_ev, key=lambda e: -e["event"].get("importance", 0)) hi_threshold = 0 high = high[:_MAX_HIGH_EVENTS] return { "total": len(news_ev), "high": high, "hi_threshold": hi_threshold, "sentiments": dict(sentiments), "importances": dict(sorted(importances.items())), "event_types": dict(event_types.most_common(10)), } def _collect_cninfo_events(today_str: str, days_back: int = CNINFO_DAYS_BACK) -> dict[str, Any]: """收集近 N 日 cninfo 公告/调研/互动(直接从 processed 数据读取,不依赖 M4 事件抽取)。 cninfo 公告/调研数据已结构化(stock_code/name/title/time/type), 无需经过 LLM 事件抽取即可直接用于日报。 """ today = datetime.strptime(today_str, "%Y%m%d") since_str = (today - timedelta(days=days_back)).strftime("%Y-%m-%d") wl_codes = _load_watchlist_codes() items: list[dict] = [] seen_urls: set[str] = set() proc_root = Path("data/processed/cninfo") if not proc_root.is_dir(): return {"total": 0, "high": [], "hi_threshold": 4, "by_day": {}, "announcement": 0, "research": 0, "irm": 0} for day_dir in sorted(proc_root.glob("*"), reverse=True): if not day_dir.is_dir(): continue for fp in sorted(day_dir.glob("*.json"), reverse=True): if fp.name == "index.jsonl": continue try: obj = json.loads(fp.read_text(encoding="utf-8")) except (json.JSONDecodeError, OSError): continue url = obj.get("url") or "" if url in seen_urls: continue seen_urls.add(url) pt = (obj.get("publish_time") or "").strip() # 按 publish_time 过滤 if pt and pt[:10] < since_str: continue item_type = obj.get("item_type") or "announcement" # 互动易数据跳过(当前无法获取真实数据) if item_type == "irm": continue # 过滤旧数据的 IRM 假阳性(标题为通用占位符或 URL 为 irm 搜索页) if "互动问答" in (obj.get("title") or ""): continue if "irm.cninfo.com.cn" in (obj.get("url") or ""): continue title = obj.get("title") or "" url = obj.get("url") or "" stock_name = obj.get("author") or "" content = obj.get("content") or "" # 计算重要度(基于是否在 watchlist 中 + 内容长度) code_in_title = "" for c in wl_codes: if c in title: code_in_title = c break importance = 3 if code_in_title else 2 if item_type == "research": importance = 3 # 调研通常更重要 items.append({ "title": title, "url": url, "source_id": "cninfo", "publish_time": pt, "event": { "stock_codes": [code_in_title] if code_in_title else [], "company_names": [stock_name] if stock_name else [], "industries": [], "sentiment": "neutral", "importance": importance, "event_type": { "announcement": "公司公告", "research": "投资者调研", }.get(item_type, "公告"), "summary": content[:120] if content else title[:120], }, }) # 按发布时间排序(最新在前) items.sort(key=lambda e: e.get("publish_time") or "", reverse=True) # 按发布时间的日期分组统计 by_day: Counter = Counter() for e in items: pt = (e.get("publish_time") or "")[:10] if pt: by_day[pt] += 1 # 高重要度: 优先 ≥4(调研),逐级回退 def _get_high(evs, threshold): return sorted( [e for e in evs if e["event"].get("importance", 0) >= threshold], key=lambda e: -e["event"].get("importance", 0), ) min_show = 3 high = _get_high(items, 4) hi_threshold = 4 if len(high) < min_show: high = _get_high(items, 2) hi_threshold = 2 if len(high) < min_show: high = sorted(items, key=lambda e: -e["event"].get("importance", 0)) hi_threshold = 0 high = high[:_MAX_HIGH_EVENTS] return { "total": len(items), "high": high, "hi_threshold": hi_threshold, "by_day": dict(by_day.most_common(7)), "announcement": sum(1 for e in items if "公告" in (e["event"].get("event_type", "") or "")), "research": sum(1 for e in items if "调研" in (e["event"].get("event_type", "") or "")), "irm": sum(1 for e in items if "互动" in (e["event"].get("event_type", "") or "")), } def _score_xwlb_importance(title: str, content: str = "") -> int: """新闻联播条目启发式重要度评分 (1-5)。 基于标题+正文关键词匹配,优先匹配高等级: 5: 直接涉及股市/金融/货币政策 4: 重大经济/产业政策/能源 3: 领导人活动/外交/区域发展/外资 2: 一般国内要闻/农业/生态 1: 文化/体育/社会/国际简讯 """ text = title + content L5 = ["降准", "降息", "印花税", "IPO", "注册制", "退市", "并购重组", "增持", "回购", "证券", "股市", "上市"] L4 = ["经济", "财政", "税收", "国债", "专项债", "碳", "产业", "制造业", "新能源", "芯片", "半导体", "人工智能", "算力", "平台经济", "房地产", "外贸", "消费", "投资", "供应链", "能源", "电力"] L3 = ["习近平", "李强", "总理", "主席", "会谈", "访问", "自贸区", "长三角", "粤港澳", "一带一路", "央企", "国企", "营商环境", "外资", "达沃斯"] L2 = ["会议", "改革", "立法", "监管", "粮食", "农业", "水利", "铁路", "公路", "港口", "生态", "救灾"] if any(kw in text for kw in L5): return 5 if any(kw in text for kw in L4): return 4 if any(kw in text for kw in L3): return 3 if any(kw in text for kw in L2): return 2 return 1 def _collect_xwlb(day_str: str) -> dict[str, Any]: """收集新闻联播要闻(从 doorcome API /api/xwlbFine/ 获取)。 《新闻联播》每天 19:00 播出:日报在早上生成时当日联播尚未播出, 因此固定取 `day_str` 前一日(最近一期已播出)的联播数据。 API 返回 AI 精编后的独立新闻条目(含标题+正文), 跳过第 1 条"内容提要"(仅为节目开场白)。 返回: {"items": [event_dict, ...], "date": "MM月DD日", "source_date": "前一日"} """ import urllib.request # 取前一晚(已播出)的联播:day_str 前一天 prev_day = (datetime.strptime(day_str, "%Y%m%d") - timedelta(days=1)).strftime("%Y%m%d") result: dict[str, Any] = {"items": [], "date": "", "source_date": prev_day} api_url = f"https://api.doorcome.cn/api/xwlbFine/?start_date={prev_day}&end_date={prev_day}" try: req = urllib.request.Request(api_url) with urllib.request.urlopen(req, timeout=15) as resp: body = json.loads(resp.read().decode("utf-8")) except Exception as e: logger.warning("新闻联播 API 请求失败: {}", e) return result raw_news = body.get("data", {}).get("news", []) if not raw_news: return result # 提取日期 dates = {n.get("news_days", "") for n in raw_news if n.get("news_days")} if dates: d = min(dates) result["date"] = f"{d[5:7]}月{d[8:10]}日" result["source_date"] = prev_day # 转换为事件格式,跳过第 1 条(内容提要/开场白) events: list[dict] = [] for n in raw_news: sid = n.get("daily_sub_id", 0) if sid <= 1: # 跳过"内容提要" continue title = n.get("news_title", "") content = n.get("news_improve", "") importance = _score_xwlb_importance(title, content) events.append({ "title": title[:100], "url": "", # 新闻联播无独立文章链接 "source_id": "xwlb", "publish_time": n.get("news_days", ""), "event": { "stock_codes": [], "company_names": [], "industries": [], "sentiment": "neutral", "importance": importance, "event_type": "新闻联播", "summary": content[:80] if content else title[:80], }, }) # 按重要度降序 events.sort(key=lambda e: (-e["event"]["importance"], e["title"])) result["items"] = events return result def _load_article_urls_from_index(index_path: Path) -> list[dict]: """从 index.jsonl 中加载 stage=article 的条目(排除列表页)。""" if not index_path.is_file(): return [] articles: list[dict] = [] for line in open(index_path, encoding="utf-8"): try: obj = json.loads(line) if obj.get("stage") == "article" and obj.get("success"): articles.append(obj) except (json.JSONDecodeError, KeyError): pass return articles def _extract_date_from_url(url: str) -> datetime | None: """从 URL 中提取发布日期(用于估算 24h 新鲜度)。""" import re as _re2 patterns = [ _re2.compile(r'/(\d{4})[-/](\d{2})[-/](\d{2})/'), _re2.compile(r'/(\d{4})(\d{2})(\d{2})/'), _re2.compile(r'(\d{4})(\d{2})(\d{2})\.(?:s?html|pdf)'), _re2.compile(r'/t(\d{4})(\d{2})(\d{2})_'), ] for pat in patterns: m = pat.search(url) if m: try: return datetime(int(m.group(1)), int(m.group(2)), int(m.group(3))) except ValueError: pass return None def _collect_pipeline_stats(day_str: str) -> dict[str, Any]: """收集管道统计数据(仅计文章级条目 + 24h 新鲜度)。""" now = datetime.now() cutoff_24h = now - timedelta(hours=24) raw_by_source: dict[str, int] = {} raw_by_source_24h: dict[str, int] = {} raw_total = 0 raw_total_24h = 0 for idx in Path("data/raw").glob(f"*/{day_str}/index.jsonl"): src = idx.parent.parent.name articles = _load_article_urls_from_index(idx) n = len(articles) name = _source_name(src) raw_by_source[name] = n raw_total += n # 统计 24h 内文章 n_24h = 0 for art in articles: dt = _extract_date_from_url(art.get("url", "")) if dt and dt >= cutoff_24h: n_24h += 1 raw_by_source_24h[name] = n_24h raw_total_24h += n_24h # cninfo raw(仅计文章级条目) cninfo_raw = 0 for idx in Path("data/raw/cninfo").glob("*/index.jsonl"): cninfo_raw += len(_load_article_urls_from_index(idx)) proc = _count_json(f"data/processed/*/{day_str}/*.json") deduped = _count_json(f"data/deduped/{day_str}/uniques/*.json") dup_path = Path(f"data/deduped/{day_str}/duplicates.jsonl") dups = _count_jsonl(dup_path) emb_count = _count_json(f"data/embeddings/{day_str}/*.json") qdrant_count = 0 try: from vectorstore import VectorStore, make_qdrant_client c = make_qdrant_client() s = VectorStore(c) qdrant_count = s.count() s.close() except Exception: pass return { "raw_total": raw_total, "raw_total_24h": raw_total_24h, "raw_by_source": raw_by_source, "raw_by_source_24h": raw_by_source_24h, "cninfo_raw": cninfo_raw, "proc": proc, "deduped": deduped, "dups": dups, "emb_count": emb_count, "qdrant_count": qdrant_count, } # --------------------------------------------------------------------------- # # AI 摘要 # --------------------------------------------------------------------------- # def _generate_ai_summary(news: dict, cninfo: dict, day_str: str, xwlb: dict | None = None) -> str: """LLM 生成 500 字以内日报摘要,囊括全部新闻、公告及新闻联播。""" lines: list[str] = [] # 新闻联播(如有) if xwlb and xwlb.get("items"): items = xwlb["items"] lines.append(f"## 新闻联播要闻 ({xwlb.get('date', '')}, {len(items)} 条, 按重要度排序)") for e in items[:10]: ev = e["event"] lines.append(f"- [重要度{ev.get('importance', 0)}] {e['title']}") # 新闻(已在 _collect_news_events 中过滤) if news["high"]: lines.append(f"## 过去 24 小时高重要度新闻 ({len(news['high'])} 条)") for e in news["high"][:12]: ev = e["event"] sentiment = ev.get("sentiment", "") s_icon = {"positive": "利好", "negative": "利空", "neutral": "中性"}.get(sentiment, "") lines.append(f"- [{s_icon}][{ev.get('event_type', '')}] {e['title']}。{ev.get('summary', '')}") # 公告/调研 if cninfo["high"]: lines.append(f"## 近 {CNINFO_DAYS_BACK} 日重要公告/调研 ({len(cninfo['high'])} 条)") for e in cninfo["high"][:8]: ev = e.get("event", {}) lines.append(f"- [{ev.get('event_type', '公司公告')}] {e['title']}") if not lines: return "" try: from llm.client import load_llm_config, make_sync_client config = load_llm_config() client = make_sync_client(config) return _llm_summarize(client, config.model, lines, day_str) except Exception as e: logger.warning("AI 摘要生成失败: {}", e) return "" def _split_lines_into_chunks(lines: list[str], max_chars: int = 3000) -> list[list[str]]: """将 lines 按 max_chars 分块,保证每条新闻(line)不被截断。""" chunks: list[list[str]] = [] current: list[str] = [] current_len = 0 for line in lines: line_len = len(line) + 1 # +1 for newline if current and current_len + line_len > max_chars: chunks.append(current) current = [] current_len = 0 current.append(line) current_len += line_len if current: chunks.append(current) return chunks def _llm_summarize(client, model: str, lines: list[str], day_str: str) -> str: """LLM 摘要:单块直接总结,多块先分段总结再合并。""" chunks = _split_lines_into_chunks(lines) if len(chunks) == 1: return _llm_call(client, model, _build_prompt(chunks[0], day_str)) # 多块:每块独立总结 partials: list[str] = [] for i, chunk in enumerate(chunks, 1): prompt = f"""以下是今日日报素材的第 {i}/{len(chunks)} 部分 (共 {len(lines)} 条, 本批 {len(chunk)} 条),请用要点总结,每条一行,以 "- " 开头: {chr(10).join(chunk)} 直接输出要点列表:""" result = _llm_call(client, model, prompt, max_tokens=800) if result: partials.append(result) logger.info("AI 摘要: 分块 {}/{} 完成 ({} 字)", i, len(chunks), len(result)) if not partials: logger.warning("AI 摘要: 所有分块均返回空") return "" if len(partials) < len(chunks): logger.warning("AI 摘要: {}/{} 分块返回空, 仅合并成功部分", len(chunks) - len(partials), len(chunks)) # 合并:将各块摘要合成最终日报摘要 merge_prompt = f"""以下是 {len(partials)} 组分段摘要,请合并为一份完整的日报摘要 ({day_str}): {chr(10).join(f'--- 第{i+1}组 ---{chr(10)}{p}' for i, p in enumerate(partials))} 请合并为要点总结,每条一行以 "- " 开头,要求: 1. 前 3 条为影响最大的事件,说明为什么重要 2. 汇总近 {CNINFO_DAYS_BACK} 日公司公告/调研核心信息 3. 市场情绪基调(利好/利空/中性) 4. 值得持续关注的行业或主题 5. 纯要点,不要开场白/结束语 6. 总字数 500 字以内 直接输出要点列表:""" return _llm_call(client, model, merge_prompt, max_tokens=1500) def _build_prompt(lines: list[str], day_str: str) -> str: """构建标准日报摘要 prompt。""" return f"""以下是今日需要总结的全部内容(含新闻联播、财经新闻、公司公告),请据此生成日报摘要 ({day_str}): {chr(10).join(lines)} 请用要点总结,每条一行,以 "- " 开头,要求: 1. 前 3 条为过去 24 小时影响最大的事件(优先参考新闻联播中的重大政策信号),说明为什么重要 2. 汇总近 {CNINFO_DAYS_BACK} 日重要公司公告/调研的核心信息 3. 市场情绪基调(利好/利空/中性) 4. 值得持续关注的行业或主题 5. 纯要点,不要开场白/结束语/标题 6. 总字数控制在 500 字以内 直接输出要点列表:""" def _llm_call(client, model: str, prompt: str, max_tokens: int = 1500) -> str: """单次 LLM 调用(带重试),返回 strip 后的文本。 失败按指数退避重试 `_LLM_RETRY_TIMES` 次(默认 3),全部失败则抛出最后一次异常。 若 finish_reason 为 'length' 则说明达到 max_tokens 上限被截断。 """ last_exc: Exception | None = None for attempt in range(_LLM_RETRY_TIMES): try: resp = client.chat.completions.create( model=model, messages=[ {"role": "system", "content": "你是 A 股日报撰写助手,输出简洁、有洞察的新闻摘要。"}, {"role": "user", "content": prompt}, ], temperature=0.3, max_tokens=max_tokens, ) content = (resp.choices[0].message.content or "").strip() finish = getattr(resp.choices[0], "finish_reason", None) if finish == "length": logger.warning( "AI 摘要可能被截断: max_tokens={} finish_reason=length 实际输出 {} 字符", max_tokens, len(content), ) return content except Exception as e: last_exc = e if attempt < _LLM_RETRY_TIMES - 1: wait = _LLM_RETRY_BACKOFF_SEC * (2 ** attempt) logger.warning( "AI 摘要 LLM 调用失败(第 {}/{} 次): {}; {} 秒后重试", attempt + 1, _LLM_RETRY_TIMES, e, round(wait, 2), ) time.sleep(wait) logger.error("AI 摘要 LLM 调用重试 {} 次仍失败: {}", _LLM_RETRY_TIMES, last_exc) assert last_exc is not None raise last_exc # --------------------------------------------------------------------------- # # HTML 渲染 # --------------------------------------------------------------------------- # _HTML_TEMPLATE = """ A 股 Deep Research 日报 — {date}_{time}

📊 A 股 Deep Research 日报

{date} · 生成于 {generated_at}

一、AI 摘要

{ai_summary}
{xwlb_section}

三、🔥 重要事件:新闻 ({raw_total_24h}/{raw_total} 篇 24h 内, importance ≥ {news_threshold}, 共 {news_high_count} 篇)

{news_table}

四、📋 重要事件:公告 / 调研 / 互动 (近 {cninfo_days} 日, importance ≥ {cninfo_threshold}, 共 {cninfo_high_count} 篇)

{cninfo_table}

五、数据总览

5.1 M1 → M6 管道

{raw_total} ({raw_total_24h} 24h)
M1 原始文章
{cninfo_raw}
M1 cninfo
{proc}
M2 正文提取
{deduped}
M3 去重唯一
{emb_count}
M5 向量
{qdrant_count}
M6 Qdrant

5.2 各源数据 ({raw_total_24h}/{raw_total} 篇 24h 内)

{source_cards}

5.3 情绪分布 (最新抓取)

{sentiment_section}

5.4 重要度分布

{importance_headers}{importance_counts}
重要度
数量

5.5 事件类型分布

{event_type_rows}
事件类型数量
""" def _render_event_table(events: list[dict], show_source: bool = True, show_summary: bool = True) -> str: """渲染事件表格。""" if not events: return "

暂无符合条件的数据

" rows: list[str] = [] wl_codes = _load_watchlist_codes() for i, e in enumerate(events, 1): ev = e["event"] sentiment = ev.get("sentiment", "") icon = {"positive": "🟢", "negative": "🔴", "neutral": "⚪"}.get(sentiment, "") badge_cls = {"positive": "badge-pos", "negative": "badge-neg"}.get(sentiment, "badge-neu") imp = ev.get("importance", 0) imp_cls = f"imp-{imp}" if imp >= 4 else "" title = e["title"][:70] src = _source_name(e.get("source_id", "")) codes_in_event = {c.strip().split(".")[0] for c in (ev.get("stock_codes") or [])} star = "⭐ " if codes_in_event & wl_codes else "" code_str = f" ({','.join(list(codes_in_event)[:3])})" if codes_in_event else "" url = e.get("url", "") title_cell = f'{star}{title}{code_str}' if url else f"{star}{title}{code_str}" cols = [ f"{i}", f'{icon}', f"{title_cell}", ] if show_source: cols.append(f"{src}") cols.append(f'{imp}') cols.append(f"{ev.get('event_type', '')}") if show_summary: cols.append(f"{(ev.get('summary', '') or '')[:60]}") rows.append(f'{"".join(cols)}') headers = ["#", "", "标题"] if show_source: headers.append("源") headers += ["重要度", "事件类型"] if show_summary: headers.append("摘要") header_row = "".join(f"{h}" for h in headers) return f"{header_row}{''.join(rows)}
" def _render_source_cards(raw_by_source: dict[str, int], raw_by_source_24h: dict[str, int], cninfo_raw: int) -> str: """渲染源数据卡片,每行 6 个,显示总量和 24h 新鲜数。""" cards: list[str] = [] # 新闻源 for name, count in sorted(raw_by_source.items()): fresh = raw_by_source_24h.get(name, 0) cards.append( f'
' f'
{count} ({fresh} 24h)
' f'
{name}
' f'
' ) # cninfo cards.append( f'
' f'
{cninfo_raw}
' f'
📋 cninfo
' f'
' ) return "\n".join(cards) def _render_xwlb_section(xwlb: dict | None) -> str: """渲染新闻联播要闻 HTML 板块(重要事件格式,按重要度排序)。""" if not xwlb or not xwlb.get("items"): return "" items = xwlb["items"] date_label = xwlb.get("date", "") source_date = xwlb.get("source_date", "") table_html = _render_event_table(items, show_source=False, show_summary=False) return f"""

二、📺 新闻联播 ({date_label}, 共 {len(items)} 条, 按重要度排序)

{table_html}

来源: 央视《新闻联播》· 数据取自 doorcome API (xwlbFine) · {source_date}

""" def _render_html(news: dict, cninfo: dict, pipeline: dict, ai_summary: str, day_str: str, xwlb: dict | None = None) -> str: """组装完整 HTML(M10 起废弃:日报已改为结构化入库,此函数不再被调用,保留以便回退)。""" # AI 摘要 → HTML summary_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", ai_summary) summary_html = _re.sub(r"\*(.+?)\*", r"\1", summary_html) summary_html = _re.sub(r"`(.+?)`", r"\1", summary_html) if summary_html.strip(): lines = summary_html.strip().splitlines() if any(ln.strip().startswith("- ") for ln in lines): items = [] for ln in lines: s = ln.strip() if s.startswith("- "): items.append(f"
  • {s[2:]}
  • ") elif s: items.append(f"
  • {s}
  • ") summary_html = f"" else: summary_html = summary_html.replace("\n", "
    ") else: summary_html = "

    AI 摘要暂不可用

    " # 新闻表格 news_table = _render_event_table(news["high"]) # cninfo 表格 cninfo_table = _render_event_table(cninfo["high"], show_source=False, show_summary=False) # 源数据卡片 source_cards = _render_source_cards( pipeline["raw_by_source"], pipeline.get("raw_by_source_24h", {}), pipeline["cninfo_raw"], ) # 情绪 s = news["sentiments"] pos = s.get("positive", 0) neg = s.get("negative", 0) neu = s.get("neutral", 0) total_s = max(pos + neg + neu, 1) sentiment_section = ( f'
    ' f'
    ' f'
    ' f'
    ' f'
    ' f'
    ' f'🟢 利好 {pos} ({pos/total_s:.0%})' f'🔴 利空 {neg} ({neg/total_s:.0%})' f'⚪ 中性 {neu} ({neu/total_s:.0%})' f'
    ' ) # 重要度 imps = news["importances"] imp_keys = sorted(imps.keys()) importance_headers = "".join(f"等级 {k}" for k in imp_keys) importance_counts = "".join(f"{imps[k]}" for k in imp_keys) # 事件类型 et = news["event_types"] event_type_rows = "\n".join( f"{k}{v}" for k, v in et.items() ) return _HTML_TEMPLATE.format( date=day_str, time=datetime.now().strftime("%H%M"), generated_at=datetime.now().strftime("%Y-%m-%d %H:%M:%S"), ai_summary=summary_html, xwlb_section=_render_xwlb_section(xwlb) if xwlb else "", news_high_count=len(news["high"]), news_threshold=news.get("hi_threshold", 4), news_table=news_table, cninfo_high_count=len(cninfo["high"]), cninfo_threshold=cninfo.get("hi_threshold", 4), cninfo_days=CNINFO_DAYS_BACK, cninfo_table=cninfo_table, raw_total=pipeline["raw_total"], raw_total_24h=pipeline.get("raw_total_24h", pipeline["raw_total"]), cninfo_raw=pipeline["cninfo_raw"], proc=pipeline["proc"], deduped=pipeline["deduped"], emb_count=pipeline["emb_count"], qdrant_count=pipeline["qdrant_count"], source_cards=source_cards, sentiment_section=sentiment_section, importance_headers=importance_headers, importance_counts=importance_counts, event_type_rows=event_type_rows, ) # --------------------------------------------------------------------------- # # 生成 + 上传 # --------------------------------------------------------------------------- # def _build_report_data(news: dict, cninfo: dict, pipeline: dict, ai_summary: str, day_str: str, xwlb: dict | None = None) -> ReportData: """组装结构化日报数据(M10:写入 MySQL 的前置步骤)。 事件板块映射:news["high"]→news / cninfo["high"]→cninfo / xwlb["items"]→xwlb。 数据总览统计以 JSON 快照存入 stats(前端自行解析)。 """ events: list[EventRow] = [] def _rows(items: list[dict], section: str) -> None: for i, e in enumerate(items, 1): ev = e.get("event", {}) events.append( EventRow( section=section, rank=i, importance=ev.get("importance"), event_type=ev.get("event_type"), title=str(e.get("title", ""))[:512], summary=(ev.get("summary") or None), sentiment=ev.get("sentiment") or None, source=e.get("source_id") or None, url=e.get("url") or None, ) ) _rows(news.get("high", []), "news") _rows(cninfo.get("high", []), "cninfo") if xwlb: _rows(xwlb.get("items", []), "xwlb") stats: dict[str, Any] = { "pipeline": pipeline, "news": { "total": news.get("total", 0), "hi_threshold": news.get("hi_threshold"), "sentiments": news.get("sentiments", {}), "importances": news.get("importances", {}), "event_types": news.get("event_types", {}), }, "cninfo": { "total": cninfo.get("total", 0), "hi_threshold": cninfo.get("hi_threshold"), "by_day": cninfo.get("by_day", {}), "announcement": cninfo.get("announcement", 0), "research": cninfo.get("research", 0), "irm": cninfo.get("irm", 0), }, } if xwlb: stats["xwlb"] = {"total": len(xwlb.get("items", [])), "date": xwlb.get("date", "")} return ReportData( report_date=datetime.strptime(day_str, "%Y%m%d").date(), report_type="finance", file_name="", # 新生成日报唯一键退化为 (report_date, finance, "") generated_at=datetime.now(), ai_summary=ai_summary or None, stats=stats, events=events, ) def generate_report(day_str: str | None = None, *, upload: bool = True) -> int | None: """生成每日日报并结构化入库(M10 完全切换,不再生成 HTML)。 `upload` 参数保留以兼容 scheduler/pipeline.py 调用,已无实际作用。 返回 report_id(成功)或 None(无数据/失败)。 """ day_str = day_str or date.today().strftime("%Y%m%d") logger.info("生成日报: {}", day_str) # 收集数据 try: news = _collect_news_events(day_str) cninfo = _collect_cninfo_events(day_str, days_back=CNINFO_DAYS_BACK) pipeline = _collect_pipeline_stats(day_str) xwlb = _collect_xwlb(day_str) except Exception as e: logger.exception("收集日报数据失败: {}", e) return None if news["total"] == 0 and cninfo["total"] == 0 and not xwlb.get("items"): logger.warning("{} 无数据,跳过日报生成", day_str) return None # AI 摘要(新闻联播 + 新闻 + cninfo) ai_summary = _generate_ai_summary(news, cninfo, day_str, xwlb=xwlb) # 结构化入库(替代原 HTML 渲染 + 上传) report = _build_report_data(news, cninfo, pipeline, ai_summary, day_str, xwlb=xwlb) try: from report_db import connect, save_report conn = connect() try: report_id = save_report(conn, report) finally: conn.close() except Exception as e: logger.exception("日报入库失败: {}", e) return None logger.info("日报已入库: report_id={}", report_id) return report_id def _upload(html_path: Path, file_tag: str) -> bool: """上传 HTML 报告到 Web 服务器(M10 起废弃:不再被调用,保留以便回退)。""" today_str = date.today().strftime("%Y%m%d") remote_dir = f"{UPLOAD_BASE}/{today_str}/" logger.info("上传日报到 {}:{}", UPLOAD_HOST, remote_dir) try: r1 = subprocess.run( ["ssh", UPLOAD_HOST, f"mkdir -p {remote_dir}"], timeout=15, capture_output=True, text=True, ) if r1.returncode != 0: logger.warning("创建远程目录失败: {}", r1.stderr.strip()) return False r2 = subprocess.run( ["scp", str(html_path), f"{UPLOAD_HOST}:{remote_dir}finance_news_daily_{file_tag}.html"], timeout=30, capture_output=True, text=True, ) if r2.returncode != 0: logger.warning("上传日报失败: {}", r2.stderr.strip()) return False logger.info("上传完成: http://doorcome.cn/echart/research/{}/", today_str) return True except Exception as e: logger.warning("上传日报异常(不阻塞): {}", e) return False