Team Ai
Apppublic

xferrr/Documentprocessing

sourceHugging Faceupdated 10mo agoView on Hugging Face
0likes
content_processor.py555 linesDownload Raw Back to root
1# -*- coding: utf-8 -*-2"""3优化版番茄小说下载器 - 整合两个版本的优点,支持完整下载4"""5 6import os7import time8import json9import re10import random11import threading12import hashlib13import zipfile14from pathlib import Path15from typing import Dict, List, Tuple, Optional, Set16from datetime import datetime17import requests18from requests.adapters import HTTPAdapter19from urllib3.util.retry import Retry20import urllib321 22# 禁用SSL警告23urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)24 25# ===================== 配置 =====================26class Config:27    """配置类"""28    # API配置 - 使用能下载完整内容的API29    API_BASE_URL = "https://fq.shusan.cn"30    ENDPOINTS = {31        'detail': '/api/v1/book/detail',32        'chapter_list': '/api/v1/book/chapter-list',33        'chapter_content': '/api/v1/chapter/content',34        'full_content': '/api/v1/chapter/content'  # 整书下载35    }36    37    # 请求配置38    REQUEST_TIMEOUT = 3039    MAX_RETRIES = 540    RETRY_DELAY = 241    DOWNLOAD_DELAY = (1.0, 3.0)  # 随机延迟范围42    43    # 连接池配置44    CONNECTION_POOL_SIZE = 1045    46    # 下载配置47    MAX_WORKERS = 548    CHUNK_SIZE = 819249    50    @classmethod51    def get_headers(cls) -> Dict[str, str]:52        """获取请求头"""53        return {54            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',55            'Accept': 'application/json, text/plain, */*',56            'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',57            'Accept-Encoding': 'gzip, deflate, br',58            'Referer': 'https://fanqienovel.com/',59            'Origin': 'https://fanqienovel.com',60            'Connection': 'keep-alive',61            'Sec-Fetch-Dest': 'empty',62            'Sec-Fetch-Mode': 'cors',63            'Sec-Fetch-Site': 'same-site',64            'DNT': '1',65            'Sec-GPC': '1'66        }67 68 69# ===================== API管理器 =====================70class APIManager:71    """API管理器 - 处理所有API请求"""72    73    def __init__(self):74        self.base_url = Config.API_BASE_URL75        self.endpoints = Config.ENDPOINTS76        self._session = None77        self._lock = threading.Lock()78        self._init_time = time.time()79    80    @property81    def session(self) -> requests.Session:82        """获取或创建会话(线程安全)"""83        with self._lock:84            if self._session is None:85                self._session = requests.Session()86                87                # 配置重试策略88                retry_strategy = Retry(89                    total=Config.MAX_RETRIES,90                    backoff_factor=Config.RETRY_DELAY,91                    status_forcelist=[429, 500, 502, 503, 504],92                    allowed_methods=["GET", "POST"],93                    raise_on_status=False94                )95                96                # 配置适配器97                adapter = HTTPAdapter(98                    pool_connections=Config.CONNECTION_POOL_SIZE,99                    pool_maxsize=Config.CONNECTION_POOL_SIZE,100                    max_retries=retry_strategy,101                    pool_block=False102                )103                104                self._session.mount("https://", adapter)105                self._session.mount("http://", adapter)106                107                # 设置请求头108                self._session.headers.update(Config.get_headers())109                self._session.headers.update({110                    'Accept': 'application/json, text/plain, */*',111                    'Accept-Language': 'zh-CN,zh;q=0.9',112                    'Connection': 'keep-alive',113                    'Cache-Control': 'no-cache',114                    'Pragma': 'no-cache'115                })116            117            return self._session118    119    def _make_request(self, endpoint: str, params: Dict = None, method: str = 'GET') -> Optional[Dict]:120        """发送HTTP请求"""121        url = f"{self.base_url}{endpoint}"122        123        for attempt in range(Config.MAX_RETRIES):124            try:125                # 请求延迟126                if attempt > 0:127                    delay = Config.RETRY_DELAY * (2 ** attempt)128                    time.sleep(min(delay, 10))129                130                # 发送请求131                if method.upper() == 'GET':132                    response = self.session.get(133                        url,134                        params=params,135                        timeout=Config.REQUEST_TIMEOUT,136                        verify=False  # 忽略SSL验证137                    )138                else:139                    response = self.session.post(140                        url,141                        json=params,142                        timeout=Config.REQUEST_TIMEOUT,143                        verify=False144                    )145                146                response.raise_for_status()147                148                # 解析JSON响应149                data = response.json()150                151                # 检查API返回码152                if data.get('code') == 200:153                    return data.get('data', {})154                else:155                    print(f"⚠️ API返回错误: {data.get('msg', '未知错误')}")156                    if attempt == Config.MAX_RETRIES - 1:157                        return None158                    159            except requests.exceptions.Timeout:160                print(f"⏱️ 请求超时 ({attempt + 1}/{Config.MAX_RETRIES}): {endpoint}")161                if attempt == Config.MAX_RETRIES - 1:162                    return None163                    164            except requests.exceptions.RequestException as e:165                print(f"⚠️ 请求失败 ({attempt + 1}/{Config.MAX_RETRIES}): {e}")166                if attempt == Config.MAX_RETRIES - 1:167                    return None168                    169            except json.JSONDecodeError as e:170                print(f"❌ JSON解析失败: {e}")171                if attempt == Config.MAX_RETRIES - 1:172                    return None173        174        return None175    176    def get_book_detail(self, book_id: str) -> Optional[Dict]:177        """获取书籍详情"""178        params = {"book_id": book_id}179        data = self._make_request(self.endpoints['detail'], params)180        181        if data and isinstance(data, dict):182            # 处理嵌套的数据结构183            if 'data' in data:184                return data['data']185            return data186        187        return None188    189    def get_chapter_list(self, book_id: str) -> Optional[List[Dict]]:190        """获取章节列表 - 支持完整章节"""191        params = {"book_id": book_id}192        data = self._make_request(self.endpoints['chapter_list'], params)193        194        if not data:195            return None196        197        # 解析章节列表 - 处理不同的数据结构198        chapters = []199        200        if isinstance(data, dict):201            # 尝试不同字段获取章节202            chapter_sources = [203                data.get('chapterListWithVolume', []),204                data.get('chapterList', []),205                data.get('chapters', []),206                data.get('data', []),207                data if isinstance(data, list) else []208            ]209            210            for source in chapter_sources:211                if isinstance(source, list) and source:212                    # 处理卷结构213                    for item in source:214                        if isinstance(item, list):215                            # 如果是列表,表示一卷内的章节216                            for ch in item:217                                if isinstance(ch, dict):218                                    self._extract_chapter_info(ch, chapters)219                        elif isinstance(item, dict):220                            self._extract_chapter_info(item, chapters)221                    break222        elif isinstance(data, list):223            for item in data:224                if isinstance(item, dict):225                    self._extract_chapter_info(item, chapters)226        227        return chapters if chapters else None228    229    def _extract_chapter_info(self, chapter_data: Dict, chapters_list: List):230        """从章节数据中提取信息"""231        item_id = chapter_data.get('itemId') or chapter_data.get('item_id') or chapter_data.get('id')232        title = chapter_data.get('title') or chapter_data.get('chapter_title', '未知章节')233        234        if item_id and title:235            chapters_list.append({236                'id': str(item_id),237                'title': title,238                'index': len(chapters_list)239            })240    241    def get_chapter_content(self, item_id: str) -> Optional[Dict]:242        """获取章节内容"""243        params = {"item_id": item_id, "tab": "小说"}244        data = self._make_request(self.endpoints['chapter_content'], params)245        246        if data and isinstance(data, dict):247            return {248                'title': data.get('title', ''),249                'content': data.get('content', ''),250                'success': True251            }252        253        return None254    255    def get_full_book_content(self, book_id: str) -> Optional[str]:256        """尝试获取整本书内容(备用方案)"""257        try:258            # 使用不同的参数尝试获取完整内容259            params_list = [260                {"book_id": book_id, "tab": "下载"},261                {"book_id": book_id, "tab": "novel", "full": "1"},262                {"book_id": book_id, "format": "txt"}263            ]264            265            for params in params_list:266                data = self._make_request(self.endpoints['full_content'], params)267                if data and isinstance(data, str) and len(data) > 1000:268                    return data269            270            return None271            272        except Exception as e:273            print(f"获取整书内容失败: {e}")274            return None275 276 277# ===================== 字体解码器 =====================278class FontDecoder:279    """字体解码器 - 处理番茄小说的自定义字体"""280    281    # 字体映射表(简化版,实际使用时需要根据实际情况更新)282    FONT_MAP = {283            58344: 'd', 58345: '在', 58346: '主', 58347: '特', 58348: '家', 58349: '军', 58350: '然', 58351: '表', 58352: '场', 58353: '4', 58354: '要', 58355: '只', 58357: '和', 58359: '6', 58360: '别', 58361: '还', 58362: 'g', 58363: '现', 58364: '儿L', 58365: '岁', 58368: '此', 58369: '象', 58370: '月', 58371: '3', 58372: '出', 58373: '战', 58374: '工', 58375: '相', 58376: '。', 58377: '男', 58378: '直', 58379: '失', 58380: '世', 58381: 'f', 58382: '都', 58383: '平', 58384: '文', 58385: '什', 58386: 'v', 58387: 'o', 58388: '将', 58389: '真', 58390: '工', 58391: '那', 58392: '当', 58394: '会', 58395: '立', 58396: '些', 58397: '山', 58398: '是', 58399: '十', 58400: '张', 58401: '学', 58402: '气', 58403: '大', 58404: '爱', 58405: '两', 58406: '命', 58407: '全', 58408: '后', 58409: '东', 58410: '性', 58411: '通', 58412: '被', 58413: '1', 58414: '它', 58415: '乐', 58416: '接', 58417: '而', 58418: '感', 58419: '车', 58420: '山', 58421: '公', 58422: '了', 58423: '常', 58424: '以', 58425: '何', 58426: '可', 58427: '话', 58428: '先', 58429: 'p', 58430: 'j', 58431: '叫', 58432: '轻', 58433: 'm', 58434: '土', 58435: 'w', 58436: '着', 58437: '变', 58438: '尔', 58439: '快', 58440: '上', 58441: '个', 58442: '说', 58443: '少', 58444: '色', 58445: '里', 58446: '安', 58447: '花', 58448: '远', 58449: '7', 58450: '难', 58451: '师', 58452: '放', 58453: '代', 58454: '报', 58455: '认', 58456: '面', 58457: '道', 58458: 's', 58460: '克', 58461: '地', 58462: '度', 58463: '上', 58464: '好', 58465: '机', 58466: 'u', 58467: '民', 58468: '写', 58469: '把', 58470: '万', 58471: '同', 58472: '水', 58473: '新', 58474: '没', 58475: '书', 58476: '电', 58477: '吃', 58478: '像', 58479: '斯', 58480: '5', 58481: '为', 58482: 'v', 58483: '白', 58484: '几l', 58485: '日', 58486: '教', 58487: '看', 58488: '但', 58489: '第', 58490: '加', 58491: '候', 58492: '作', 58493: '上', 58494: '拉', 58495: '住', 58496: '有', 58497: '法', 58498: 'r', 58499: '事', 58500: '应', 58501: '位', 58502: '利', 58503: '你', 58504: '声', 58505: '身', 58506: '国', 58507: '问', 58508: '马', 58509: '女', 58510: '他', 58511: 'y', 58512: '比', 58513: '父', 58515: 'a', 58516: 'h', 58517: 'n', 58518: 's', 58519: 'x', 58520: '边', 58521: '美', 58522: '对', 58523: '所', 58524: '金', 58525: '活', 58526: '回', 58527: '意', 58528: '到', 58529: '之', 58530: '从', 58531: 'j', 58532: '知', 58533: '又', 58534: '内', 58535: '因', 58536: '点', 58537: 'o', 58538: '三', 58539: '定', 58540: '8', 58541: 'r', 58542: 'b', 58543: '正', 58544: '或', 58545: '夫', 58546: '向', 58547: '德', 58548: '听', 58549: '更', 58551: '得', 58552: '告', 58553: '并', 58554: '本', 58555: 'q', 58556: '过', 58557: '记', 58558: '上', 58559: '让', 58560: '打', 58561: 'f', 58562: '人', 58563: '就', 58564: '者', 58565: '去', 58566: '原', 58567: '满', 58568: '体', 58569: '做', 58570: '经', 58571: 'k', 58572: '走', 58573: '如', 58574: '孩', 58575: 'c', 58576: 'g', 58577: '给', 58578: '使', 58579: '物', 58581: '最', 58582: '笑', 58583: '部', 58585: '员', 58586: '等', 58587: '受', 58588: 'k', 58589: '行', 58591: '条', 58592: '果', 58593: '动', 58594: '光', 58595: '门', 58596: '头', 58597: '见', 58598: '往', 58599: '自', 58600: '解', 58601: '成', 58602: '处', 58603: '天', 58604: '能', 58605: '干', 58606: '名', 58607: '其', 58608: '发', 58609: '总', 58610: '母', 58611: '的', 58612: '死', 58613: '手', 58614: '入', 58615: '路', 58616: '进', 58617: '心', 58618: '来', 58619: 'h', 58620: '时', 58621: '力', 58622: '多', 58623: '开', 58624: '已', 58625: '许', 58626: 'd', 58627: '至', 58628: '由', 58629: '很', 58630: '界', 58631: 'n', 58632: '小', 58633: '与', 58634: 'z', 58635: '想', 58636: '代', 58637: '么', 58638: '分', 58639: '生', 58640: '口', 58641: '再', 58642: '妈', 58643: '望', 58644: '次', 58645: '西', 58646: '风', 58647: '种', 58648: '带', 58649: 'J', 58651: '实', 58652: '情', 58653: '才', 58654: '这', 58656: 'e', 58657: '我', 58658: '神', 58659: '格', 58660: '长', 58661: '觉', 58662: '间', 58663: '年', 58664: '眼', 58665: '无', 58666: '不', 58667: '亲', 58668: '关', 58669: '结', 58670: 'o', 58671: '友', 58672: '信', 58673: '下', 58674: '却', 58675: '重', 58676: '己', 58677: '老', 58678: '2', 58679: '音', 58680: '字', 58681: 'm', 58682: '呢', 58683: '明', 58684: '之', 58685: '前', 58686: '高', 58687: 'p', 58688: 'b', 58689: '目', 58690: '太', 58691: 'e', 58692: '9', 58693: '起', 58694: '棱', 58695: '她', 58696: '也', 58697: 'w', 58698: '用', 58699: '方', 58700: '子', 58701: '英', 58702: '每', 58703: '理', 58704: '便', 58705: '四', 58706: '数', 58707: '期', 58708: '中', 58709: 'c', 58710: '外', 58711: '样', 58712: 'a', 58713: '海', 58714: '们', 58715: '任', 58356: 'v', 58514: 'x',284            65292: ',', 58590: '一', 65311: '?', 65281: '!', 65288: '(', 65289: ')'285    }286    287    @classmethod288    def decode(cls, text: str) -> str:289        """解码文本"""290        if not text:291            return text292        293        result = []294        for char in text:295            char_code = ord(char)296            decoded_char = cls.FONT_MAP.get(char_code, char)297            result.append(decoded_char)298        299        return ''.join(result)300    301    @classmethod302    def process_content(cls, content: str) -> str:303        """处理并清理内容"""304        if not content:305            return ""306        307        # 解码308        content = cls.decode(content)309        310        # 清理HTML标签311        content = re.sub(r'<br\s*/?>', '\n', content, flags=re.IGNORECASE)312        content = re.sub(r'<p[^>]*>', '\n', content, flags=re.IGNORECASE)313        content = re.sub(r'</p>', '\n', content, flags=re.IGNORECASE)314        content = re.sub(r'<[^>]+>', '', content)315        316        # 清理空白字符317        content = re.sub(r'[ \t]+', ' ', content)318        content = re.sub(r'\n[ \t]+', '\n', content)319        content = re.sub(r'[ \t]+\n', '\n', content)320        321        # 规范化换行322        content = re.sub(r'\n{3,}', '\n\n', content)323        324        # 分段325        paragraphs = [p.strip() for p in content.split('\n') if p.strip()]326        return '\n\n'.join(paragraphs)327 328 329# ===================== 状态管理器 =====================330class StatusManager:331    """下载状态管理器"""332    333    def __init__(self, data_dir: Path):334        self.data_dir = data_dir335        self.status_dir = data_dir / ".status"336        self.status_dir.mkdir(exist_ok=True)337    338    def get_status_file(self, source_url: str) -> Path:339        """获取状态文件路径"""340        url_hash = hashlib.md5(source_url.encode()).hexdigest()[:12]341        return self.status_dir / f"status_{url_hash}.json"342    343    def load_downloaded_ids(self, source_url: str) -> Set[str]:344        """加载已下载的章节ID"""345        status_file = self.get_status_file(source_url)346        347        if status_file.exists():348            try:349                with open(status_file, 'r', encoding='utf-8') as f:350                    return set(json.load(f))351            except (json.JSONDecodeError, UnicodeDecodeError):352                pass353        354        return set()355    356    def save_downloaded_id(self, source_url: str, chapter_id: str):357        """保存已下载的章节ID"""358        status_file = self.get_status_file(source_url)359        downloaded = self.load_downloaded_ids(source_url)360        downloaded.add(chapter_id)361        362        try:363            with open(status_file, 'w', encoding='utf-8') as f:364                json.dump(list(downloaded), f, ensure_ascii=False, indent=2)365        except Exception as e:366            print(f"保存状态失败: {e}")367    368    def clear_status(self, source_url: str):369        """清除下载状态"""370        status_file = self.get_status_file(source_url)371        if status_file.exists():372            try:373                status_file.unlink()374            except Exception as e:375                print(f"清除状态失败: {e}")376 377 378# ===================== 内容处理器 =====================379class ContentProcessor:380    """优化版内容处理器 - 整合两个版本的优点"""381    382    def __init__(self):383        self.api_manager = APIManager()384        self.font_decoder = FontDecoder()385        self.status_manager = None386        self.data_dir = Path("./novel_cache")387        self.data_dir.mkdir(exist_ok=True)388        self.status_manager = StatusManager(self.data_dir)389    390    def extract_book_id(self, url: str) -> Optional[str]:391        """从URL中提取书籍ID"""392        patterns = [393            r'page/(\d+)',394            r'book_id=(\d+)',395            r'book/(\d+)',396            r'bid=(\d+)',397            r'novel/(\d+)'398        ]399        400        for pattern in patterns:401            match = re.search(pattern, url)402            if match:403                return match.group(1)404        405        # 尝试直接提取数字406        match = re.search(r'(\d{10,})', url)407        if match:408            return match.group(1)409        410        return None411    412    def fetch_document_info(self, source_url: str) -> Tuple[str, List[Dict]]:413        """414        获取文档信息(书籍详情和章节列表)415        返回: (书名, 章节列表)416        """417        print(f"📖 正在处理: {source_url}")418        419        # 提取书籍ID420        book_id = self.extract_book_id(source_url)421        if not book_id:422            raise ValueError(f"无法从URL中提取书籍ID: {source_url}")423        424        print(f"🔍 提取到书籍ID: {book_id}")425        426        # 获取书籍详情427        book_detail = self.api_manager.get_book_detail(book_id)428        if not book_detail:429            raise ValueError(f"获取书籍详情失败: {book_id}")430        431        # 提取书名432        book_name = book_detail.get('book_name') or book_detail.get('title') or f"未知书籍_{book_id}"433        safe_book_name = re.sub(r'[\\/:*?"<>|]', '_', book_name)434        print(f"📚 书籍: {safe_book_name}")435        436        # 获取章节列表437        chapters = self.api_manager.get_chapter_list(book_id)438        if not chapters:439            raise ValueError(f"获取章节列表失败: {book_id}")440        441        print(f"📜 找到 {len(chapters)} 个章节")442        443        # 验证章节完整性444        if len(chapters) < 10:445            print(f"⚠️ 章节数较少 ({len(chapters)}),可能未获取到完整列表")446        447        return safe_book_name, chapters448    449    def fetch_section_content(self, section_info: Dict) -> Tuple[bool, str, str]:450        """获取章节内容"""451        item_id = section_info.get('id')452        title = section_info.get('title', f'章节_{item_id}')453        454        if not item_id:455            return False, title, "缺少章节ID"456        457        # 随机延迟,避免请求过快458        time.sleep(random.uniform(*Config.DOWNLOAD_DELAY))459        460        # 获取章节内容461        content_data = self.api_manager.get_chapter_content(item_id)462        if not content_data or not content_data.get('success'):463            return False, title, "获取章节内容失败"464        465        # 处理内容466        raw_content = content_data.get('content', '')467        processed_content = self.font_decoder.process_content(raw_content)468        469        # 构建完整内容470        full_content = f"{title}\n\n{processed_content}"471        472        return True, title, full_content473    474    def download_chapters_batch(self, chapters: List[Dict], source_url: str, 475                              base_dir: Path, progress_callback=None) -> Tuple[int, int]:476        """批量下载章节"""477        downloaded_ids = self.status_manager.load_downloaded_ids(source_url)478        479        # 过滤已下载章节480        chapters_to_download = []481        for chapter in chapters:482            if chapter.get('id') and chapter['id'] not in downloaded_ids:483                chapters_to_download.append(chapter)484        485        if not chapters_to_download:486            return 0, 0487        488        print(f"📥 需要下载 {len(chapters_to_download)} 个新章节")489        490        success_count = 0491        failed_count = 0492        493        # 批量下载494        for idx, chapter in enumerate(chapters_to_download, 1):495            if progress_callback:496                progress_callback(idx, len(chapters_to_download), chapter.get('title', ''))497            498            try:499                ok, title, content = self.fetch_section_content(chapter)500                501                if ok and content:502                    # 保存文件503                    file_path = base_dir / f"{title}.txt"504                    file_path.write_text(content, encoding='utf-8')505                    506                    # 更新状态507                    self.status_manager.save_downloaded_id(source_url, chapter['id'])508                    success_count += 1509                    510                    if idx % 5 == 0 or idx == len(chapters_to_download):511                        print(f"✅ {idx}/{len(chapters_to_download)}: {title[:30]}...")512                else:513                    failed_count += 1514                    print(f"❌ 章节下载失败: {chapter.get('title', '未知')}")515                    516            except Exception as e:517                failed_count += 1518                print(f"❌ 下载异常: {e}")519            520            # 章节间延迟521            if idx < len(chapters_to_download):522                time.sleep(random.uniform(0.5, 1.5))523        524        return success_count, failed_count525    526    def create_archive(self, base_dir: Path) -> str:527        """创建ZIP归档"""528        archive_path = base_dir / f"{base_dir.name}.zip"529        530        with zipfile.ZipFile(archive_path, 'w', zipfile.ZIP_DEFLATED) as zipf:531            for txt_file in base_dir.glob("*.txt"):532                # 只添加.txt文件533                if txt_file.is_file() and txt_file.suffix == '.txt':534                    zipf.write(txt_file, txt_file.name)535        536        return str(archive_path)537    538    def get_file_count(self, base_dir: Path) -> int:539        """获取文件数量"""540        return len(list(base_dir.glob("*.txt")))541    542    def get_existing_chapters(self, base_dir: Path) -> Set[str]:543        """获取已存在的章节标题"""544        return {f.stem for f in base_dir.glob("*.txt")}545 546 547# 全局实例548_processor_instance = None549 550def get_content_processor() -> ContentProcessor:551    """获取内容处理器实例(单例模式)"""552    global _processor_instance553    if _processor_instance is None:554        _processor_instance = ContentProcessor()555    return _processor_instance