Files
simonandClaude 5c106120bf feat: 增强反爬策略 + forexlive 迁移 investinglive
- crawler.py: 启用 magic/stealth 全自动反检测, --disable-webrtc,
  delay_before_return_html=3s, remove_consent_popups, 过滤列表页URL
- rss_crawler.py: 新增 _make_rss_client() 支持 httpx 代理,
  Google News RSS 跳过 article_url_pattern 域名过滤
- sources.yaml: investing.com 改用 Google News RSS 代理绕过Cloudflare;
  forexlive -> investinglive (域名已301重定向)
- domestic_crawl_8g.sh: HTTP_PROXY/HTTPS_PROXY 在 .env 加载前设置

Co-Authored-By: Claude <noreply@anthropic.com>
2026-07-23 09:25:23 +08:00

412 lines
13 KiB
Python
Raw Permalink Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""RSS/Atom Feed 抓取模块。
用于有 RSS 服务的新闻源(如 MarketWatch),绕过网页反爬。
支持 RSS 2.0、Atom、Google News RSS 三种格式。
Google News RSS: 处理标题去来源后缀、链接为 Google 跳转 URL。
"""
import hashlib
import logging
import re
from datetime import datetime, timezone
from html import unescape
from pathlib import Path
from typing import Any
from urllib.parse import urljoin
from xml.etree import ElementTree
import httpx
from crawler.models import ArticleItem, CrawlResult, SourceConfig
from crawler.utils import get_news_day
logger = logging.getLogger(__name__)
# ── HTML 标签清洗 ──────────────────────────────────────
def _strip_html(text: str) -> str:
"""去除 HTML 标签,保留纯文本。"""
if not text:
return ""
text = unescape(text)
return re.sub(r"<[^>]+>", "", text).strip()
# ── Namespace ──────────────────────────────────────────
def _ns(tag: str) -> str:
"""Atom namespace helper。"""
return f"{{http://www.w3.org/2005/Atom}}{tag}"
def _compute_url_hash(url: str) -> str:
return hashlib.sha256(url.encode("utf-8")).hexdigest()[:16]
# 非内容类扩展名 — RSS 项也可能包含静态资源
_SKIP_EXT = re.compile(
r"\.(png|ico|gif|jpg|jpeg|svg|webp|css|js|xml|json|rss|pdf|zip|woff2?|ttf|eot)"
r"([?#]|$)",
re.IGNORECASE,
)
# 明显非文章路径
_SKIP_PATH = re.compile(
r"/(manifest|robots|sitemap|_next/static|__nextjs_|favicon)"
r"[/.]",
re.IGNORECASE,
)
def _is_valid_article_url(url: str, source: SourceConfig) -> bool:
"""校验 RSS 条目 URL 是否为有效文章链接。
过滤规则:
1. 静态资源(JS/CSS/图片/字体/数据文件)
2. 技术路径(manifest/_next/static/robots
3. 源 article_url_pattern 不匹配
4. 域名不一致(RSS 跨站污染,如 Yahoo Finance RSS 混入 sports/tech
"""
if _SKIP_EXT.search(url):
logger.debug("RSS 跳过静态资源: %s", url[:80])
return False
if _SKIP_PATH.search(url):
logger.debug("RSS 跳过技术路径: %s", url[:80])
return False
# 源级 article_url_pattern 校验
# Google News RSS 跳过(链接为 news.google.com 格式,不匹配源 pattern
rss_url = getattr(source, "rss_url", "")
is_google_news = "news.google.com" in rss_url
if not is_google_news:
pattern = getattr(source, "article_url_pattern", "")
if pattern:
try:
if not re.search(pattern, url, re.IGNORECASE):
logger.debug("RSS 跳过不匹配 article_url_pattern: %s", url[:80])
return False
except re.error:
pass # 正则异常不阻塞
# 域名一致性校验(RSS 跨站污染防护)
# 例外: Google News RSS 的链接是 news.google.com 跳转 URL,跳过域名校验
rss_url = getattr(source, "rss_url", "")
if rss_url and "news.google.com" not in rss_url:
from urllib.parse import urlparse
item_domain = (urlparse(url).hostname or "").removeprefix("www.")
hp_domain = (urlparse(source.homepage).hostname or "").removeprefix("www.")
if hp_domain and item_domain:
if item_domain != hp_domain and not item_domain.endswith("." + hp_domain):
logger.debug(
"RSS 跳过跨站 URL: %s (domain=%s, expected=%s)",
url[:80], item_domain, hp_domain,
)
return False
return True
def _parse_date(date_str: str | None) -> str:
"""尝试解析常见日期格式,返回 ISO 8601 字符串。"""
if not date_str:
return datetime.now().isoformat()
formats = [
"%a, %d %b %Y %H:%M:%S %z", # RFC 2822
"%a, %d %b %Y %H:%M:%S %Z",
"%Y-%m-%dT%H:%M:%S%z", # ISO 8601
"%Y-%m-%dT%H:%M:%SZ",
"%Y-%m-%d %H:%M:%S",
"%Y-%m-%d",
]
for fmt in formats:
try:
return datetime.strptime(date_str, fmt).isoformat()
except ValueError:
continue
return date_str
# ── Google News RSS 标题清洗 ────────────────────────────
_KNOWN_SOURCES = [
"Reuters", "WSJ", "Bloomberg", "CNBC", "Financial Times",
"MarketWatch", "Barron's", "Yahoo Finance", "Investing.com",
"Seeking Alpha", "The Economist", "ForexLive", "ZeroHedge",
]
def _clean_google_news_title(title: str) -> str:
"""Google News RSS 标题格式: 'Article Title - SourceName' → 去掉来源后缀。"""
for src in _KNOWN_SOURCES:
suffix = f" - {src}"
if title.endswith(suffix):
return title[: -len(suffix)].strip()
# 通用回退:最后一个 " - " 之后可能是来源
last_dash = title.rfind(" - ")
if last_dash > 0:
suffix = title[last_dash + 3:]
if len(suffix) < 30 and not suffix.startswith("http"):
return title[:last_dash].strip()
return title
# ── RSS/Atom 解析 ─────────────────────────────────────
def _extract_rss_items(xml_text: str) -> list[dict[str, Any]]:
"""从 RSS/Atom XML 中提取文章条目。
自动识别:RSS 2.0、Google News RSS、Atom。
"""
# 清理可能的 BOM
if xml_text.startswith(""):
xml_text = xml_text[1:]
root = ElementTree.fromstring(xml_text)
items: list[dict[str, Any]] = []
# 检测是否为 Google News RSS
first_item = root.find(".//item")
is_google_news = False
if first_item is not None:
source_el = first_item.find("source")
if source_el is not None and source_el.text:
is_google_news = True
# RSS 2.0 / Google News RSS: channel/item
for item in root.iter("item"):
title = _strip_html(_text(item, "title"))
link = _text(item, "link")
description = _strip_html(_text(item, "description"))
pub_date = _text(item, "pubDate")
if is_google_news and title:
title = _clean_google_news_title(title)
# 描述通常比标题更长,含摘要
if description and len(description) > len(title):
summary = description
else:
summary = title
else:
summary = description or title
if link and title:
items.append({
"title": title,
"url": link.strip(),
"summary": summary,
"publish_time": _parse_date(pub_date),
})
if items:
tag = "Google News RSS" if is_google_news else "RSS 2.0"
logger.debug("%s: 提取 %d 条", tag, len(items))
return items
# Atom: feed/entry
for entry in root.iter(_ns("entry")):
title = _strip_html(_text(entry, _ns("title")))
link = _attr(entry, _ns("link"), "href")
summary = _strip_html(_text(entry, _ns("summary")))
updated = _text(entry, _ns("updated"))
if link and title:
items.append({
"title": title,
"url": link.strip(),
"summary": summary or title,
"publish_time": _parse_date(updated),
})
logger.debug("Atom: 提取 %d 条", len(items))
return items
def _text(element: ElementTree.Element, tag: str) -> str:
"""安全获取子元素文本。"""
child = element.find(tag)
return (child.text or "").strip() if child is not None and child.text else ""
def _attr(element: ElementTree.Element, tag: str, attr: str) -> str:
"""安全获取子元素属性。"""
child = element.find(tag)
return (child.get(attr) or "").strip() if child is not None else ""
# ── 主抓取接口 ───────────────────────────────────────
def _make_rss_client() -> httpx.Client:
"""创建 httpx 客户端,自动配置代理"""
try:
from crawler.config import load_system_config
cfg = load_system_config()
p = cfg.get("proxy", {})
if p.get("enabled") and p.get("url"):
return httpx.Client(proxy=p["url"], follow_redirects=True, timeout=30.0)
except Exception:
pass
return httpx.Client(follow_redirects=True, timeout=30.0)
def crawl_rss_source(source: SourceConfig) -> CrawlResult:
"""通过 RSS/Atom Feed 抓取单个新闻源。
支持 Google News RSS 代理模式:
- rss_url 为 news.google.com 时自动识别
- 标题自动去来源后缀
- 不抓取原文(跳过 Crawl4AI),直接用 RSS 摘要入库
Args:
source: 源配置(需含 rss_url 字段)
Returns:
CrawlResult
"""
rss_url = getattr(source, "rss_url", None)
if not rss_url:
return CrawlResult(
source_id=source.id,
source_name=source.name,
error="源未配置 rss_url",
)
today = get_news_day()
out_dir = Path(f"data/raw/{source.id}/{today}")
out_dir.mkdir(parents=True, exist_ok=True)
start_time = datetime.now()
articles: list[ArticleItem] = []
crawl_time = start_time.isoformat()
try:
client = _make_rss_client()
resp = client.get(
rss_url,
timeout=30.0,
headers={
"User-Agent": (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 Chrome/131.0.0.0 Safari/537.36"
),
"Accept": "application/rss+xml, application/xml, text/xml, */*",
},
)
resp.raise_for_status()
xml_text = resp.text
except Exception as e:
logger.error("[%s] RSS 抓取失败: %s", source.id, e)
return CrawlResult(
source_id=source.id,
source_name=source.name,
start_time=crawl_time,
end_time=datetime.now().isoformat(),
error=str(e),
)
# 解析 RSS
try:
items = _extract_rss_items(xml_text)
except Exception as e:
logger.error("[%s] RSS 解析失败: %s", source.id, e)
return CrawlResult(
source_id=source.id,
source_name=source.name,
start_time=crawl_time,
end_time=datetime.now().isoformat(),
error=f"RSS 解析失败: {e}",
)
if not items:
logger.warning("[%s] RSS 返回 0 条", source.id)
return CrawlResult(
source_id=source.id,
source_name=source.name,
total_found=0,
start_time=crawl_time,
end_time=datetime.now().isoformat(),
)
# 过滤已存在的 URL(增量)
index_path = out_dir / "index.jsonl"
existing_hashes: set[str] = _load_existing_hashes(index_path)
# 构造 ArticleItem
success = 0
for item in items[:source.max_articles_per_run]:
url_hash = _compute_url_hash(item["url"])
if url_hash in existing_hashes:
logger.debug("[%s] 跳过 RSS 已抓取: %s", source.id, item["url"][:80])
continue
# URL 有效性校验(静态资源 / 跨站污染 / 不匹配 article_url_pattern
if not _is_valid_article_url(item["url"], source):
continue
# 保存文章摘要为 Markdown
md_content = f"# {item['title']}\n\n"
md_content += f"**来源**: {source.name}\n\n"
md_content += f"**发布时间**: {item['publish_time']}\n\n"
md_content += f"**原文链接**: {item['url']}\n\n"
md_content += f"{item['summary']}\n"
md_path = out_dir / f"{url_hash}.md"
md_path.write_text(md_content, encoding="utf-8")
html_path = out_dir / f"{url_hash}.html"
html_content = f"<html><head><title>{item['title']}</title></head><body>{item['summary']}</body></html>"
html_path.write_text(html_content, encoding="utf-8")
article = ArticleItem(
source_id=source.id,
source_name=source.name,
url=item["url"],
url_hash=url_hash,
title=item["title"],
crawl_time=crawl_time,
publish_time=item["publish_time"],
html_path=str(html_path),
md_path=str(md_path),
word_count=len(item["summary"].split()),
status="success",
)
articles.append(article)
existing_hashes.add(url_hash)
success += 1
logger.info("[%s] RSS 抓取: %d 条新文章(总共 %d 条)",
source.id, success, len(items))
return CrawlResult(
source_id=source.id,
source_name=source.name,
total_found=len(items),
total_success=success,
articles=articles,
start_time=crawl_time,
end_time=datetime.now().isoformat(),
)
def _load_existing_hashes(index_path: Path) -> set[str]:
"""读取 index.jsonl 中已有 url_hash。"""
hashes: set[str] = set()
if not index_path.exists():
return hashes
import json
with open(index_path, encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line:
continue
try:
hashes.add(json.loads(line).get("url_hash", ""))
except json.JSONDecodeError:
continue
return hashes