初始化

This commit is contained in:
2026-07-18 16:13:52 +08:00
parent c0070f0a5c
commit fe8b417ab6
75 changed files with 12898 additions and 1 deletions
+68
View File
@@ -0,0 +1,68 @@
"""爬虫数据模型"""
from datetime import datetime
from pathlib import Path
from pydantic import BaseModel, Field
class SourceConfig(BaseModel):
"""单个新闻源配置"""
id: str
name: str
enabled: bool = True
homepage: str
article_url_pattern: str
js_render: bool = False
max_articles_per_run: int = 30
rss_url: str | None = None # RSS/Atom feed URL(优先使用,绕过反爬)
anti_bot_mode: str | None = None # "stealth" | "headful" | None(反爬策略)
@property
def output_dir(self) -> Path:
"""按日期组织的输出目录"""
today = datetime.now().strftime("%Y%m%d")
return Path(f"data/raw/{self.id}/{today}")
class ArticleItem(BaseModel):
"""单篇已抓取的文章元数据"""
source_id: str
source_name: str
url: str
url_hash: str
title: str
crawl_time: str # ISO 8601
publish_time: str = "" # ISO 8601RSS 源可提取,网页源由 extractor 补充
html_path: str # 相对路径,如 data/raw/reuters/20260621/abc123.html
md_path: str = "" # Crawl4AI 生成的 Markdown 路径
word_count: int = 0
status: str = "success" # success | failed
error: str = ""
class CrawlResult(BaseModel):
"""单次抓取结果统计"""
source_id: str
source_name: str
total_found: int = 0
total_success: int = 0
total_failed: int = 0
articles: list[ArticleItem] = Field(default_factory=list)
start_time: str = ""
end_time: str = ""
error: str = ""
class PipelineStats(BaseModel):
"""一次完整抓取管道的统计"""
start_time: str = ""
end_time: str = ""
sources_crawled: int = 0
sources_failed: int = 0
total_articles: int = 0
results: list[CrawlResult] = Field(default_factory=list)