394 lines
13 KiB
Python
394 lines
13 KiB
Python
"""RSS/Atom Feed 抓取模块。
|
||
|
||
用于有 RSS 服务的新闻源(如 MarketWatch),绕过网页反爬。
|
||
支持 RSS 2.0、Atom、Google News RSS 三种格式。
|
||
|
||
Google News RSS: 处理标题去来源后缀、链接为 Google 跳转 URL。
|
||
"""
|
||
|
||
import hashlib
|
||
import logging
|
||
import re
|
||
from datetime import datetime, timezone
|
||
from html import unescape
|
||
from pathlib import Path
|
||
from typing import Any
|
||
from urllib.parse import urljoin
|
||
from xml.etree import ElementTree
|
||
|
||
import httpx
|
||
|
||
from crawler.models import ArticleItem, CrawlResult, SourceConfig
|
||
from crawler.utils import get_news_day
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
# ── HTML 标签清洗 ──────────────────────────────────────
|
||
|
||
|
||
def _strip_html(text: str) -> str:
|
||
"""去除 HTML 标签,保留纯文本。"""
|
||
if not text:
|
||
return ""
|
||
text = unescape(text)
|
||
return re.sub(r"<[^>]+>", "", text).strip()
|
||
|
||
|
||
# ── Namespace ──────────────────────────────────────────
|
||
|
||
|
||
def _ns(tag: str) -> str:
|
||
"""Atom namespace helper。"""
|
||
return f"{{http://www.w3.org/2005/Atom}}{tag}"
|
||
|
||
|
||
def _compute_url_hash(url: str) -> str:
|
||
return hashlib.sha256(url.encode("utf-8")).hexdigest()[:16]
|
||
|
||
|
||
# 非内容类扩展名 — RSS 项也可能包含静态资源
|
||
_SKIP_EXT = re.compile(
|
||
r"\.(png|ico|gif|jpg|jpeg|svg|webp|css|js|xml|json|rss|pdf|zip|woff2?|ttf|eot)"
|
||
r"([?#]|$)",
|
||
re.IGNORECASE,
|
||
)
|
||
|
||
# 明显非文章路径
|
||
_SKIP_PATH = re.compile(
|
||
r"/(manifest|robots|sitemap|_next/static|__nextjs_|favicon)"
|
||
r"[/.]",
|
||
re.IGNORECASE,
|
||
)
|
||
|
||
|
||
def _is_valid_article_url(url: str, source: SourceConfig) -> bool:
|
||
"""校验 RSS 条目 URL 是否为有效文章链接。
|
||
|
||
过滤规则:
|
||
1. 静态资源(JS/CSS/图片/字体/数据文件)
|
||
2. 技术路径(manifest/_next/static/robots)
|
||
3. 源 article_url_pattern 不匹配
|
||
4. 域名不一致(RSS 跨站污染,如 Yahoo Finance RSS 混入 sports/tech)
|
||
"""
|
||
if _SKIP_EXT.search(url):
|
||
logger.debug("RSS 跳过静态资源: %s", url[:80])
|
||
return False
|
||
if _SKIP_PATH.search(url):
|
||
logger.debug("RSS 跳过技术路径: %s", url[:80])
|
||
return False
|
||
|
||
# 源级 article_url_pattern 校验
|
||
pattern = getattr(source, "article_url_pattern", "")
|
||
if pattern:
|
||
try:
|
||
if not re.search(pattern, url, re.IGNORECASE):
|
||
logger.debug("RSS 跳过不匹配 article_url_pattern: %s", url[:80])
|
||
return False
|
||
except re.error:
|
||
pass # 正则异常不阻塞
|
||
|
||
# 域名一致性校验(RSS 跨站污染防护)
|
||
# 例外: Google News RSS 的链接是 news.google.com 跳转 URL,跳过域名校验
|
||
rss_url = getattr(source, "rss_url", "")
|
||
if rss_url and "news.google.com" not in rss_url:
|
||
from urllib.parse import urlparse
|
||
item_domain = (urlparse(url).hostname or "").removeprefix("www.")
|
||
hp_domain = (urlparse(source.homepage).hostname or "").removeprefix("www.")
|
||
|
||
if hp_domain and item_domain:
|
||
if item_domain != hp_domain and not item_domain.endswith("." + hp_domain):
|
||
logger.debug(
|
||
"RSS 跳过跨站 URL: %s (domain=%s, expected=%s)",
|
||
url[:80], item_domain, hp_domain,
|
||
)
|
||
return False
|
||
|
||
return True
|
||
|
||
|
||
def _parse_date(date_str: str | None) -> str:
|
||
"""尝试解析常见日期格式,返回 ISO 8601 字符串。"""
|
||
if not date_str:
|
||
return datetime.now().isoformat()
|
||
|
||
formats = [
|
||
"%a, %d %b %Y %H:%M:%S %z", # RFC 2822
|
||
"%a, %d %b %Y %H:%M:%S %Z",
|
||
"%Y-%m-%dT%H:%M:%S%z", # ISO 8601
|
||
"%Y-%m-%dT%H:%M:%SZ",
|
||
"%Y-%m-%d %H:%M:%S",
|
||
"%Y-%m-%d",
|
||
]
|
||
for fmt in formats:
|
||
try:
|
||
return datetime.strptime(date_str, fmt).isoformat()
|
||
except ValueError:
|
||
continue
|
||
return date_str
|
||
|
||
|
||
# ── Google News RSS 标题清洗 ────────────────────────────
|
||
|
||
_KNOWN_SOURCES = [
|
||
"Reuters", "WSJ", "Bloomberg", "CNBC", "Financial Times",
|
||
"MarketWatch", "Barron's", "Yahoo Finance", "Investing.com",
|
||
"Seeking Alpha", "The Economist", "ForexLive", "ZeroHedge",
|
||
]
|
||
|
||
|
||
def _clean_google_news_title(title: str) -> str:
|
||
"""Google News RSS 标题格式: 'Article Title - SourceName' → 去掉来源后缀。"""
|
||
for src in _KNOWN_SOURCES:
|
||
suffix = f" - {src}"
|
||
if title.endswith(suffix):
|
||
return title[: -len(suffix)].strip()
|
||
# 通用回退:最后一个 " - " 之后可能是来源
|
||
last_dash = title.rfind(" - ")
|
||
if last_dash > 0:
|
||
suffix = title[last_dash + 3:]
|
||
if len(suffix) < 30 and not suffix.startswith("http"):
|
||
return title[:last_dash].strip()
|
||
return title
|
||
|
||
|
||
# ── RSS/Atom 解析 ─────────────────────────────────────
|
||
|
||
|
||
def _extract_rss_items(xml_text: str) -> list[dict[str, Any]]:
|
||
"""从 RSS/Atom XML 中提取文章条目。
|
||
|
||
自动识别:RSS 2.0、Google News RSS、Atom。
|
||
"""
|
||
# 清理可能的 BOM
|
||
if xml_text.startswith(""):
|
||
xml_text = xml_text[1:]
|
||
|
||
root = ElementTree.fromstring(xml_text)
|
||
items: list[dict[str, Any]] = []
|
||
|
||
# 检测是否为 Google News RSS
|
||
first_item = root.find(".//item")
|
||
is_google_news = False
|
||
if first_item is not None:
|
||
source_el = first_item.find("source")
|
||
if source_el is not None and source_el.text:
|
||
is_google_news = True
|
||
|
||
# RSS 2.0 / Google News RSS: channel/item
|
||
for item in root.iter("item"):
|
||
title = _strip_html(_text(item, "title"))
|
||
link = _text(item, "link")
|
||
description = _strip_html(_text(item, "description"))
|
||
pub_date = _text(item, "pubDate")
|
||
|
||
if is_google_news and title:
|
||
title = _clean_google_news_title(title)
|
||
# 描述通常比标题更长,含摘要
|
||
if description and len(description) > len(title):
|
||
summary = description
|
||
else:
|
||
summary = title
|
||
else:
|
||
summary = description or title
|
||
|
||
if link and title:
|
||
items.append({
|
||
"title": title,
|
||
"url": link.strip(),
|
||
"summary": summary,
|
||
"publish_time": _parse_date(pub_date),
|
||
})
|
||
|
||
if items:
|
||
tag = "Google News RSS" if is_google_news else "RSS 2.0"
|
||
logger.debug("%s: 提取 %d 条", tag, len(items))
|
||
return items
|
||
|
||
# Atom: feed/entry
|
||
for entry in root.iter(_ns("entry")):
|
||
title = _strip_html(_text(entry, _ns("title")))
|
||
link = _attr(entry, _ns("link"), "href")
|
||
summary = _strip_html(_text(entry, _ns("summary")))
|
||
updated = _text(entry, _ns("updated"))
|
||
if link and title:
|
||
items.append({
|
||
"title": title,
|
||
"url": link.strip(),
|
||
"summary": summary or title,
|
||
"publish_time": _parse_date(updated),
|
||
})
|
||
|
||
logger.debug("Atom: 提取 %d 条", len(items))
|
||
return items
|
||
|
||
|
||
def _text(element: ElementTree.Element, tag: str) -> str:
|
||
"""安全获取子元素文本。"""
|
||
child = element.find(tag)
|
||
return (child.text or "").strip() if child is not None and child.text else ""
|
||
|
||
|
||
def _attr(element: ElementTree.Element, tag: str, attr: str) -> str:
|
||
"""安全获取子元素属性。"""
|
||
child = element.find(tag)
|
||
return (child.get(attr) or "").strip() if child is not None else ""
|
||
|
||
|
||
# ── 主抓取接口 ───────────────────────────────────────
|
||
|
||
|
||
def crawl_rss_source(source: SourceConfig) -> CrawlResult:
|
||
"""通过 RSS/Atom Feed 抓取单个新闻源。
|
||
|
||
支持 Google News RSS 代理模式:
|
||
- rss_url 为 news.google.com 时自动识别
|
||
- 标题自动去来源后缀
|
||
- 不抓取原文(跳过 Crawl4AI),直接用 RSS 摘要入库
|
||
|
||
Args:
|
||
source: 源配置(需含 rss_url 字段)
|
||
|
||
Returns:
|
||
CrawlResult
|
||
"""
|
||
rss_url = getattr(source, "rss_url", None)
|
||
if not rss_url:
|
||
return CrawlResult(
|
||
source_id=source.id,
|
||
source_name=source.name,
|
||
error="源未配置 rss_url",
|
||
)
|
||
|
||
today = get_news_day()
|
||
out_dir = Path(f"data/raw/{source.id}/{today}")
|
||
out_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
start_time = datetime.now()
|
||
articles: list[ArticleItem] = []
|
||
crawl_time = start_time.isoformat()
|
||
|
||
try:
|
||
resp = httpx.get(
|
||
rss_url,
|
||
follow_redirects=True,
|
||
timeout=30.0,
|
||
headers={
|
||
"User-Agent": (
|
||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||
"AppleWebKit/537.36 Chrome/131.0.0.0 Safari/537.36"
|
||
),
|
||
"Accept": "application/rss+xml, application/xml, text/xml, */*",
|
||
},
|
||
)
|
||
resp.raise_for_status()
|
||
xml_text = resp.text
|
||
except Exception as e:
|
||
logger.error("[%s] RSS 抓取失败: %s", source.id, e)
|
||
return CrawlResult(
|
||
source_id=source.id,
|
||
source_name=source.name,
|
||
start_time=crawl_time,
|
||
end_time=datetime.now().isoformat(),
|
||
error=str(e),
|
||
)
|
||
|
||
# 解析 RSS
|
||
try:
|
||
items = _extract_rss_items(xml_text)
|
||
except Exception as e:
|
||
logger.error("[%s] RSS 解析失败: %s", source.id, e)
|
||
return CrawlResult(
|
||
source_id=source.id,
|
||
source_name=source.name,
|
||
start_time=crawl_time,
|
||
end_time=datetime.now().isoformat(),
|
||
error=f"RSS 解析失败: {e}",
|
||
)
|
||
|
||
if not items:
|
||
logger.warning("[%s] RSS 返回 0 条", source.id)
|
||
return CrawlResult(
|
||
source_id=source.id,
|
||
source_name=source.name,
|
||
total_found=0,
|
||
start_time=crawl_time,
|
||
end_time=datetime.now().isoformat(),
|
||
)
|
||
|
||
# 过滤已存在的 URL(增量)
|
||
index_path = out_dir / "index.jsonl"
|
||
existing_hashes: set[str] = _load_existing_hashes(index_path)
|
||
|
||
# 构造 ArticleItem
|
||
success = 0
|
||
for item in items[:source.max_articles_per_run]:
|
||
url_hash = _compute_url_hash(item["url"])
|
||
if url_hash in existing_hashes:
|
||
logger.debug("[%s] 跳过 RSS 已抓取: %s", source.id, item["url"][:80])
|
||
continue
|
||
|
||
# URL 有效性校验(静态资源 / 跨站污染 / 不匹配 article_url_pattern)
|
||
if not _is_valid_article_url(item["url"], source):
|
||
continue
|
||
|
||
# 保存文章摘要为 Markdown
|
||
md_content = f"# {item['title']}\n\n"
|
||
md_content += f"**来源**: {source.name}\n\n"
|
||
md_content += f"**发布时间**: {item['publish_time']}\n\n"
|
||
md_content += f"**原文链接**: {item['url']}\n\n"
|
||
md_content += f"{item['summary']}\n"
|
||
|
||
md_path = out_dir / f"{url_hash}.md"
|
||
md_path.write_text(md_content, encoding="utf-8")
|
||
html_path = out_dir / f"{url_hash}.html"
|
||
html_content = f"<html><head><title>{item['title']}</title></head><body>{item['summary']}</body></html>"
|
||
html_path.write_text(html_content, encoding="utf-8")
|
||
|
||
article = ArticleItem(
|
||
source_id=source.id,
|
||
source_name=source.name,
|
||
url=item["url"],
|
||
url_hash=url_hash,
|
||
title=item["title"],
|
||
crawl_time=crawl_time,
|
||
publish_time=item["publish_time"],
|
||
html_path=str(html_path),
|
||
md_path=str(md_path),
|
||
word_count=len(item["summary"].split()),
|
||
status="success",
|
||
)
|
||
articles.append(article)
|
||
existing_hashes.add(url_hash)
|
||
success += 1
|
||
|
||
logger.info("[%s] RSS 抓取: %d 条新文章(总共 %d 条)",
|
||
source.id, success, len(items))
|
||
|
||
return CrawlResult(
|
||
source_id=source.id,
|
||
source_name=source.name,
|
||
total_found=len(items),
|
||
total_success=success,
|
||
articles=articles,
|
||
start_time=crawl_time,
|
||
end_time=datetime.now().isoformat(),
|
||
)
|
||
|
||
|
||
def _load_existing_hashes(index_path: Path) -> set[str]:
|
||
"""读取 index.jsonl 中已有 url_hash。"""
|
||
hashes: set[str] = set()
|
||
if not index_path.exists():
|
||
return hashes
|
||
import json
|
||
with open(index_path, encoding="utf-8") as f:
|
||
for line in f:
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
try:
|
||
hashes.add(json.loads(line).get("url_hash", ""))
|
||
except json.JSONDecodeError:
|
||
continue
|
||
return hashes
|