xferrr/Documentprocessing
0
1# -*- coding: utf-8 -*-2"""3优化版番茄小说下载器 - 整合两个版本的优点,支持完整下载4"""5 6import os7import time8import json9import re10import random11import threading12import hashlib13import zipfile14from pathlib import Path15from typing import Dict, List, Tuple, Optional, Set16from datetime import datetime17import requests18from requests.adapters import HTTPAdapter19from urllib3.util.retry import Retry20import urllib321 22# 禁用SSL警告23urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)24 25# ===================== 配置 =====================26class Config:27 """配置类"""28 # API配置 - 使用能下载完整内容的API29 API_BASE_URL = "https://fq.shusan.cn"30 ENDPOINTS = {31 'detail': '/api/v1/book/detail',32 'chapter_list': '/api/v1/book/chapter-list',33 'chapter_content': '/api/v1/chapter/content',34 'full_content': '/api/v1/chapter/content' # 整书下载35 }36 37 # 请求配置38 REQUEST_TIMEOUT = 3039 MAX_RETRIES = 540 RETRY_DELAY = 241 DOWNLOAD_DELAY = (1.0, 3.0) # 随机延迟范围42 43 # 连接池配置44 CONNECTION_POOL_SIZE = 1045 46 # 下载配置47 MAX_WORKERS = 548 CHUNK_SIZE = 819249 50 @classmethod51 def get_headers(cls) -> Dict[str, str]:52 """获取请求头"""53 return {54 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',55 'Accept': 'application/json, text/plain, */*',56 'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',57 'Accept-Encoding': 'gzip, deflate, br',58 'Referer': 'https://fanqienovel.com/',59 'Origin': 'https://fanqienovel.com',60 'Connection': 'keep-alive',61 'Sec-Fetch-Dest': 'empty',62 'Sec-Fetch-Mode': 'cors',63 'Sec-Fetch-Site': 'same-site',64 'DNT': '1',65 'Sec-GPC': '1'66 }67 68 69# ===================== API管理器 =====================70class APIManager:71 """API管理器 - 处理所有API请求"""72 73 def __init__(self):74 self.base_url = Config.API_BASE_URL75 self.endpoints = Config.ENDPOINTS76 self._session = None77 self._lock = threading.Lock()78 self._init_time = time.time()79 80 @property81 def session(self) -> requests.Session:82 """获取或创建会话(线程安全)"""83 with self._lock:84 if self._session is None:85 self._session = requests.Session()86 87 # 配置重试策略88 retry_strategy = Retry(89 total=Config.MAX_RETRIES,90 backoff_factor=Config.RETRY_DELAY,91 status_forcelist=[429, 500, 502, 503, 504],92 allowed_methods=["GET", "POST"],93 raise_on_status=False94 )95 96 # 配置适配器97 adapter = HTTPAdapter(98 pool_connections=Config.CONNECTION_POOL_SIZE,99 pool_maxsize=Config.CONNECTION_POOL_SIZE,100 max_retries=retry_strategy,101 pool_block=False102 )103 104 self._session.mount("https://", adapter)105 self._session.mount("http://", adapter)106 107 # 设置请求头108 self._session.headers.update(Config.get_headers())109 self._session.headers.update({110 'Accept': 'application/json, text/plain, */*',111 'Accept-Language': 'zh-CN,zh;q=0.9',112 'Connection': 'keep-alive',113 'Cache-Control': 'no-cache',114 'Pragma': 'no-cache'115 })116 117 return self._session118 119 def _make_request(self, endpoint: str, params: Dict = None, method: str = 'GET') -> Optional[Dict]:120 """发送HTTP请求"""121 url = f"{self.base_url}{endpoint}"122 123 for attempt in range(Config.MAX_RETRIES):124 try:125 # 请求延迟126 if attempt > 0:127 delay = Config.RETRY_DELAY * (2 ** attempt)128 time.sleep(min(delay, 10))129 130 # 发送请求131 if method.upper() == 'GET':132 response = self.session.get(133 url,134 params=params,135 timeout=Config.REQUEST_TIMEOUT,136 verify=False # 忽略SSL验证137 )138 else:139 response = self.session.post(140 url,141 json=params,142 timeout=Config.REQUEST_TIMEOUT,143 verify=False144 )145 146 response.raise_for_status()147 148 # 解析JSON响应149 data = response.json()150 151 # 检查API返回码152 if data.get('code') == 200:153 return data.get('data', {})154 else:155 print(f"⚠️ API返回错误: {data.get('msg', '未知错误')}")156 if attempt == Config.MAX_RETRIES - 1:157 return None158 159 except requests.exceptions.Timeout:160 print(f"⏱️ 请求超时 ({attempt + 1}/{Config.MAX_RETRIES}): {endpoint}")161 if attempt == Config.MAX_RETRIES - 1:162 return None163 164 except requests.exceptions.RequestException as e:165 print(f"⚠️ 请求失败 ({attempt + 1}/{Config.MAX_RETRIES}): {e}")166 if attempt == Config.MAX_RETRIES - 1:167 return None168 169 except json.JSONDecodeError as e:170 print(f"❌ JSON解析失败: {e}")171 if attempt == Config.MAX_RETRIES - 1:172 return None173 174 return None175 176 def get_book_detail(self, book_id: str) -> Optional[Dict]:177 """获取书籍详情"""178 params = {"book_id": book_id}179 data = self._make_request(self.endpoints['detail'], params)180 181 if data and isinstance(data, dict):182 # 处理嵌套的数据结构183 if 'data' in data:184 return data['data']185 return data186 187 return None188 189 def get_chapter_list(self, book_id: str) -> Optional[List[Dict]]:190 """获取章节列表 - 支持完整章节"""191 params = {"book_id": book_id}192 data = self._make_request(self.endpoints['chapter_list'], params)193 194 if not data:195 return None196 197 # 解析章节列表 - 处理不同的数据结构198 chapters = []199 200 if isinstance(data, dict):201 # 尝试不同字段获取章节202 chapter_sources = [203 data.get('chapterListWithVolume', []),204 data.get('chapterList', []),205 data.get('chapters', []),206 data.get('data', []),207 data if isinstance(data, list) else []208 ]209 210 for source in chapter_sources:211 if isinstance(source, list) and source:212 # 处理卷结构213 for item in source:214 if isinstance(item, list):215 # 如果是列表,表示一卷内的章节216 for ch in item:217 if isinstance(ch, dict):218 self._extract_chapter_info(ch, chapters)219 elif isinstance(item, dict):220 self._extract_chapter_info(item, chapters)221 break222 elif isinstance(data, list):223 for item in data:224 if isinstance(item, dict):225 self._extract_chapter_info(item, chapters)226 227 return chapters if chapters else None228 229 def _extract_chapter_info(self, chapter_data: Dict, chapters_list: List):230 """从章节数据中提取信息"""231 item_id = chapter_data.get('itemId') or chapter_data.get('item_id') or chapter_data.get('id')232 title = chapter_data.get('title') or chapter_data.get('chapter_title', '未知章节')233 234 if item_id and title:235 chapters_list.append({236 'id': str(item_id),237 'title': title,238 'index': len(chapters_list)239 })240 241 def get_chapter_content(self, item_id: str) -> Optional[Dict]:242 """获取章节内容"""243 params = {"item_id": item_id, "tab": "小说"}244 data = self._make_request(self.endpoints['chapter_content'], params)245 246 if data and isinstance(data, dict):247 return {248 'title': data.get('title', ''),249 'content': data.get('content', ''),250 'success': True251 }252 253 return None254 255 def get_full_book_content(self, book_id: str) -> Optional[str]:256 """尝试获取整本书内容(备用方案)"""257 try:258 # 使用不同的参数尝试获取完整内容259 params_list = [260 {"book_id": book_id, "tab": "下载"},261 {"book_id": book_id, "tab": "novel", "full": "1"},262 {"book_id": book_id, "format": "txt"}263 ]264 265 for params in params_list:266 data = self._make_request(self.endpoints['full_content'], params)267 if data and isinstance(data, str) and len(data) > 1000:268 return data269 270 return None271 272 except Exception as e:273 print(f"获取整书内容失败: {e}")274 return None275 276 277# ===================== 字体解码器 =====================278class FontDecoder:279 """字体解码器 - 处理番茄小说的自定义字体"""280 281 # 字体映射表(简化版,实际使用时需要根据实际情况更新)282 FONT_MAP = {283 58344: 'd', 58345: '在', 58346: '主', 58347: '特', 58348: '家', 58349: '军', 58350: '然', 58351: '表', 58352: '场', 58353: '4', 58354: '要', 58355: '只', 58357: '和', 58359: '6', 58360: '别', 58361: '还', 58362: 'g', 58363: '现', 58364: '儿L', 58365: '岁', 58368: '此', 58369: '象', 58370: '月', 58371: '3', 58372: '出', 58373: '战', 58374: '工', 58375: '相', 58376: '。', 58377: '男', 58378: '直', 58379: '失', 58380: '世', 58381: 'f', 58382: '都', 58383: '平', 58384: '文', 58385: '什', 58386: 'v', 58387: 'o', 58388: '将', 58389: '真', 58390: '工', 58391: '那', 58392: '当', 58394: '会', 58395: '立', 58396: '些', 58397: '山', 58398: '是', 58399: '十', 58400: '张', 58401: '学', 58402: '气', 58403: '大', 58404: '爱', 58405: '两', 58406: '命', 58407: '全', 58408: '后', 58409: '东', 58410: '性', 58411: '通', 58412: '被', 58413: '1', 58414: '它', 58415: '乐', 58416: '接', 58417: '而', 58418: '感', 58419: '车', 58420: '山', 58421: '公', 58422: '了', 58423: '常', 58424: '以', 58425: '何', 58426: '可', 58427: '话', 58428: '先', 58429: 'p', 58430: 'j', 58431: '叫', 58432: '轻', 58433: 'm', 58434: '土', 58435: 'w', 58436: '着', 58437: '变', 58438: '尔', 58439: '快', 58440: '上', 58441: '个', 58442: '说', 58443: '少', 58444: '色', 58445: '里', 58446: '安', 58447: '花', 58448: '远', 58449: '7', 58450: '难', 58451: '师', 58452: '放', 58453: '代', 58454: '报', 58455: '认', 58456: '面', 58457: '道', 58458: 's', 58460: '克', 58461: '地', 58462: '度', 58463: '上', 58464: '好', 58465: '机', 58466: 'u', 58467: '民', 58468: '写', 58469: '把', 58470: '万', 58471: '同', 58472: '水', 58473: '新', 58474: '没', 58475: '书', 58476: '电', 58477: '吃', 58478: '像', 58479: '斯', 58480: '5', 58481: '为', 58482: 'v', 58483: '白', 58484: '几l', 58485: '日', 58486: '教', 58487: '看', 58488: '但', 58489: '第', 58490: '加', 58491: '候', 58492: '作', 58493: '上', 58494: '拉', 58495: '住', 58496: '有', 58497: '法', 58498: 'r', 58499: '事', 58500: '应', 58501: '位', 58502: '利', 58503: '你', 58504: '声', 58505: '身', 58506: '国', 58507: '问', 58508: '马', 58509: '女', 58510: '他', 58511: 'y', 58512: '比', 58513: '父', 58515: 'a', 58516: 'h', 58517: 'n', 58518: 's', 58519: 'x', 58520: '边', 58521: '美', 58522: '对', 58523: '所', 58524: '金', 58525: '活', 58526: '回', 58527: '意', 58528: '到', 58529: '之', 58530: '从', 58531: 'j', 58532: '知', 58533: '又', 58534: '内', 58535: '因', 58536: '点', 58537: 'o', 58538: '三', 58539: '定', 58540: '8', 58541: 'r', 58542: 'b', 58543: '正', 58544: '或', 58545: '夫', 58546: '向', 58547: '德', 58548: '听', 58549: '更', 58551: '得', 58552: '告', 58553: '并', 58554: '本', 58555: 'q', 58556: '过', 58557: '记', 58558: '上', 58559: '让', 58560: '打', 58561: 'f', 58562: '人', 58563: '就', 58564: '者', 58565: '去', 58566: '原', 58567: '满', 58568: '体', 58569: '做', 58570: '经', 58571: 'k', 58572: '走', 58573: '如', 58574: '孩', 58575: 'c', 58576: 'g', 58577: '给', 58578: '使', 58579: '物', 58581: '最', 58582: '笑', 58583: '部', 58585: '员', 58586: '等', 58587: '受', 58588: 'k', 58589: '行', 58591: '条', 58592: '果', 58593: '动', 58594: '光', 58595: '门', 58596: '头', 58597: '见', 58598: '往', 58599: '自', 58600: '解', 58601: '成', 58602: '处', 58603: '天', 58604: '能', 58605: '干', 58606: '名', 58607: '其', 58608: '发', 58609: '总', 58610: '母', 58611: '的', 58612: '死', 58613: '手', 58614: '入', 58615: '路', 58616: '进', 58617: '心', 58618: '来', 58619: 'h', 58620: '时', 58621: '力', 58622: '多', 58623: '开', 58624: '已', 58625: '许', 58626: 'd', 58627: '至', 58628: '由', 58629: '很', 58630: '界', 58631: 'n', 58632: '小', 58633: '与', 58634: 'z', 58635: '想', 58636: '代', 58637: '么', 58638: '分', 58639: '生', 58640: '口', 58641: '再', 58642: '妈', 58643: '望', 58644: '次', 58645: '西', 58646: '风', 58647: '种', 58648: '带', 58649: 'J', 58651: '实', 58652: '情', 58653: '才', 58654: '这', 58656: 'e', 58657: '我', 58658: '神', 58659: '格', 58660: '长', 58661: '觉', 58662: '间', 58663: '年', 58664: '眼', 58665: '无', 58666: '不', 58667: '亲', 58668: '关', 58669: '结', 58670: 'o', 58671: '友', 58672: '信', 58673: '下', 58674: '却', 58675: '重', 58676: '己', 58677: '老', 58678: '2', 58679: '音', 58680: '字', 58681: 'm', 58682: '呢', 58683: '明', 58684: '之', 58685: '前', 58686: '高', 58687: 'p', 58688: 'b', 58689: '目', 58690: '太', 58691: 'e', 58692: '9', 58693: '起', 58694: '棱', 58695: '她', 58696: '也', 58697: 'w', 58698: '用', 58699: '方', 58700: '子', 58701: '英', 58702: '每', 58703: '理', 58704: '便', 58705: '四', 58706: '数', 58707: '期', 58708: '中', 58709: 'c', 58710: '外', 58711: '样', 58712: 'a', 58713: '海', 58714: '们', 58715: '任', 58356: 'v', 58514: 'x',284 65292: ',', 58590: '一', 65311: '?', 65281: '!', 65288: '(', 65289: ')'285 }286 287 @classmethod288 def decode(cls, text: str) -> str:289 """解码文本"""290 if not text:291 return text292 293 result = []294 for char in text:295 char_code = ord(char)296 decoded_char = cls.FONT_MAP.get(char_code, char)297 result.append(decoded_char)298 299 return ''.join(result)300 301 @classmethod302 def process_content(cls, content: str) -> str:303 """处理并清理内容"""304 if not content:305 return ""306 307 # 解码308 content = cls.decode(content)309 310 # 清理HTML标签311 content = re.sub(r'<br\s*/?>', '\n', content, flags=re.IGNORECASE)312 content = re.sub(r'<p[^>]*>', '\n', content, flags=re.IGNORECASE)313 content = re.sub(r'</p>', '\n', content, flags=re.IGNORECASE)314 content = re.sub(r'<[^>]+>', '', content)315 316 # 清理空白字符317 content = re.sub(r'[ \t]+', ' ', content)318 content = re.sub(r'\n[ \t]+', '\n', content)319 content = re.sub(r'[ \t]+\n', '\n', content)320 321 # 规范化换行322 content = re.sub(r'\n{3,}', '\n\n', content)323 324 # 分段325 paragraphs = [p.strip() for p in content.split('\n') if p.strip()]326 return '\n\n'.join(paragraphs)327 328 329# ===================== 状态管理器 =====================330class StatusManager:331 """下载状态管理器"""332 333 def __init__(self, data_dir: Path):334 self.data_dir = data_dir335 self.status_dir = data_dir / ".status"336 self.status_dir.mkdir(exist_ok=True)337 338 def get_status_file(self, source_url: str) -> Path:339 """获取状态文件路径"""340 url_hash = hashlib.md5(source_url.encode()).hexdigest()[:12]341 return self.status_dir / f"status_{url_hash}.json"342 343 def load_downloaded_ids(self, source_url: str) -> Set[str]:344 """加载已下载的章节ID"""345 status_file = self.get_status_file(source_url)346 347 if status_file.exists():348 try:349 with open(status_file, 'r', encoding='utf-8') as f:350 return set(json.load(f))351 except (json.JSONDecodeError, UnicodeDecodeError):352 pass353 354 return set()355 356 def save_downloaded_id(self, source_url: str, chapter_id: str):357 """保存已下载的章节ID"""358 status_file = self.get_status_file(source_url)359 downloaded = self.load_downloaded_ids(source_url)360 downloaded.add(chapter_id)361 362 try:363 with open(status_file, 'w', encoding='utf-8') as f:364 json.dump(list(downloaded), f, ensure_ascii=False, indent=2)365 except Exception as e:366 print(f"保存状态失败: {e}")367 368 def clear_status(self, source_url: str):369 """清除下载状态"""370 status_file = self.get_status_file(source_url)371 if status_file.exists():372 try:373 status_file.unlink()374 except Exception as e:375 print(f"清除状态失败: {e}")376 377 378# ===================== 内容处理器 =====================379class ContentProcessor:380 """优化版内容处理器 - 整合两个版本的优点"""381 382 def __init__(self):383 self.api_manager = APIManager()384 self.font_decoder = FontDecoder()385 self.status_manager = None386 self.data_dir = Path("./novel_cache")387 self.data_dir.mkdir(exist_ok=True)388 self.status_manager = StatusManager(self.data_dir)389 390 def extract_book_id(self, url: str) -> Optional[str]:391 """从URL中提取书籍ID"""392 patterns = [393 r'page/(\d+)',394 r'book_id=(\d+)',395 r'book/(\d+)',396 r'bid=(\d+)',397 r'novel/(\d+)'398 ]399 400 for pattern in patterns:401 match = re.search(pattern, url)402 if match:403 return match.group(1)404 405 # 尝试直接提取数字406 match = re.search(r'(\d{10,})', url)407 if match:408 return match.group(1)409 410 return None411 412 def fetch_document_info(self, source_url: str) -> Tuple[str, List[Dict]]:413 """414 获取文档信息(书籍详情和章节列表)415 返回: (书名, 章节列表)416 """417 print(f"📖 正在处理: {source_url}")418 419 # 提取书籍ID420 book_id = self.extract_book_id(source_url)421 if not book_id:422 raise ValueError(f"无法从URL中提取书籍ID: {source_url}")423 424 print(f"🔍 提取到书籍ID: {book_id}")425 426 # 获取书籍详情427 book_detail = self.api_manager.get_book_detail(book_id)428 if not book_detail:429 raise ValueError(f"获取书籍详情失败: {book_id}")430 431 # 提取书名432 book_name = book_detail.get('book_name') or book_detail.get('title') or f"未知书籍_{book_id}"433 safe_book_name = re.sub(r'[\\/:*?"<>|]', '_', book_name)434 print(f"📚 书籍: {safe_book_name}")435 436 # 获取章节列表437 chapters = self.api_manager.get_chapter_list(book_id)438 if not chapters:439 raise ValueError(f"获取章节列表失败: {book_id}")440 441 print(f"📜 找到 {len(chapters)} 个章节")442 443 # 验证章节完整性444 if len(chapters) < 10:445 print(f"⚠️ 章节数较少 ({len(chapters)}),可能未获取到完整列表")446 447 return safe_book_name, chapters448 449 def fetch_section_content(self, section_info: Dict) -> Tuple[bool, str, str]:450 """获取章节内容"""451 item_id = section_info.get('id')452 title = section_info.get('title', f'章节_{item_id}')453 454 if not item_id:455 return False, title, "缺少章节ID"456 457 # 随机延迟,避免请求过快458 time.sleep(random.uniform(*Config.DOWNLOAD_DELAY))459 460 # 获取章节内容461 content_data = self.api_manager.get_chapter_content(item_id)462 if not content_data or not content_data.get('success'):463 return False, title, "获取章节内容失败"464 465 # 处理内容466 raw_content = content_data.get('content', '')467 processed_content = self.font_decoder.process_content(raw_content)468 469 # 构建完整内容470 full_content = f"{title}\n\n{processed_content}"471 472 return True, title, full_content473 474 def download_chapters_batch(self, chapters: List[Dict], source_url: str, 475 base_dir: Path, progress_callback=None) -> Tuple[int, int]:476 """批量下载章节"""477 downloaded_ids = self.status_manager.load_downloaded_ids(source_url)478 479 # 过滤已下载章节480 chapters_to_download = []481 for chapter in chapters:482 if chapter.get('id') and chapter['id'] not in downloaded_ids:483 chapters_to_download.append(chapter)484 485 if not chapters_to_download:486 return 0, 0487 488 print(f"📥 需要下载 {len(chapters_to_download)} 个新章节")489 490 success_count = 0491 failed_count = 0492 493 # 批量下载494 for idx, chapter in enumerate(chapters_to_download, 1):495 if progress_callback:496 progress_callback(idx, len(chapters_to_download), chapter.get('title', ''))497 498 try:499 ok, title, content = self.fetch_section_content(chapter)500 501 if ok and content:502 # 保存文件503 file_path = base_dir / f"{title}.txt"504 file_path.write_text(content, encoding='utf-8')505 506 # 更新状态507 self.status_manager.save_downloaded_id(source_url, chapter['id'])508 success_count += 1509 510 if idx % 5 == 0 or idx == len(chapters_to_download):511 print(f"✅ {idx}/{len(chapters_to_download)}: {title[:30]}...")512 else:513 failed_count += 1514 print(f"❌ 章节下载失败: {chapter.get('title', '未知')}")515 516 except Exception as e:517 failed_count += 1518 print(f"❌ 下载异常: {e}")519 520 # 章节间延迟521 if idx < len(chapters_to_download):522 time.sleep(random.uniform(0.5, 1.5))523 524 return success_count, failed_count525 526 def create_archive(self, base_dir: Path) -> str:527 """创建ZIP归档"""528 archive_path = base_dir / f"{base_dir.name}.zip"529 530 with zipfile.ZipFile(archive_path, 'w', zipfile.ZIP_DEFLATED) as zipf:531 for txt_file in base_dir.glob("*.txt"):532 # 只添加.txt文件533 if txt_file.is_file() and txt_file.suffix == '.txt':534 zipf.write(txt_file, txt_file.name)535 536 return str(archive_path)537 538 def get_file_count(self, base_dir: Path) -> int:539 """获取文件数量"""540 return len(list(base_dir.glob("*.txt")))541 542 def get_existing_chapters(self, base_dir: Path) -> Set[str]:543 """获取已存在的章节标题"""544 return {f.stem for f in base_dir.glob("*.txt")}545 546 547# 全局实例548_processor_instance = None549 550def get_content_processor() -> ContentProcessor:551 """获取内容处理器实例(单例模式)"""552 global _processor_instance553 if _processor_instance is None:554 _processor_instance = ContentProcessor()555 return _processor_instance