Team Ai
Apppublic

xferrr/Documentprocessing

sourceHugging Faceupdated 10mo agoView on Hugging Face
0likes
app.py1129 linesDownload Raw Back to root
1import os2import time3import json4import random5import threading6import hashlib7import zipfile8import re9import traceback10from pathlib import Path11from datetime import datetime, timedelta12from contextlib import asynccontextmanager13from typing import Dict, List, Tuple, Optional, Set14import requests15from requests.adapters import HTTPAdapter16from urllib3.util.retry import Retry17from fastapi import FastAPI, HTTPException18from fastapi.staticfiles import StaticFiles19from fastapi.responses import FileResponse, HTMLResponse, JSONResponse20import uvicorn21import asyncio22 23# ===================== 配置 =====================24SOURCE_URL = os.getenv("SOURCE_URL", "https://fanqienovel.com/page/7569914141335374910").strip()25CHECK_INTERVAL = int(os.getenv("CHECK_INTERVAL", "3600"))26DATA_DIR = Path("./novel_cache")27DATA_DIR.mkdir(exist_ok=True)28 29# ===================== API配置 =====================30API_BASE_URL = "https://fq.shusan.cn"31ENDPOINTS = {32    'detail': '/api/v1/book/detail',33    'chapter_list': '/api/v1/book/chapter-list',34    'chapter_content': '/api/v1/chapter/content',35}36 37# ===================== 增强的请求头(模拟真实浏览器) =====================38USER_AGENTS = [39    'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',40    'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/119.0.0.0 Safari/537.36',41    'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',42    'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'43]44 45def get_random_headers() -> Dict[str, str]:46    """获取随机请求头,模拟真实浏览器"""47    return {48        'User-Agent': random.choice(USER_AGENTS),49        'Accept': 'application/json, text/plain, */*',50        'Accept-Language': 'zh-CN,zh;q=0.9,en-US;q=0.8,en;q=0.7',51        'Accept-Encoding': 'gzip, deflate, br',52        'Referer': 'https://fanqienovel.com/',53        'Origin': 'https://fanqienovel.com',54        'Connection': 'keep-alive',55        'Cache-Control': 'no-cache',56        'Pragma': 'no-cache',57        'Sec-Ch-Ua': '"Not_A Brand";v="8", "Chromium";v="120", "Google Chrome";v="120"',58        'Sec-Ch-Ua-Mobile': '?0',59        'Sec-Ch-Ua-Platform': '"Windows"',60        'Sec-Fetch-Dest': 'empty',61        'Sec-Fetch-Mode': 'cors',62        'Sec-Fetch-Site': 'cross-site',63        'Dnt': '1',64    }65 66# ===================== 字体解码映射 =====================67FONT_MAP = {68    58344: 'd', 58345: '在', 58346: '主', 58347: '特', 58348: '家', 58349: '军', 58350: '然', 58351: '表', 58352: '场', 58353: '4',69    58354: '要', 58355: '只', 58357: '和', 58359: '6', 58360: '别', 58361: '还', 58362: 'g', 58363: '现', 58364: '儿L', 58365: '岁',70    58368: '此', 58369: '象', 58370: '月', 58371: '3', 58372: '出', 58373: '战', 58374: '工', 58375: '相', 58376: '。', 58377: '男',71    58378: '直', 58379: '失', 58380: '世', 58381: 'f', 58382: '都', 58383: '平', 58384: '文', 58385: '什', 58386: 'v', 58387: 'o',72    58388: '将', 58389: '真', 58390: '工', 58391: '那', 58392: '当', 58394: '会', 58395: '立', 58396: '些', 58397: '山', 58398: '是',73    58399: '十', 58400: '张', 58401: '学', 58402: '气', 58403: '大', 58404: '爱', 58405: '两', 58406: '命', 58407: '全', 58408: '后',74    58409: '东', 58410: '性', 58411: '通', 58412: '被', 58413: '1', 58414: '它', 58415: '乐', 58416: '接', 58417: '而', 58418: '感',75    58419: '车', 58420: '山', 58421: '公', 58422: '了', 58423: '常', 58424: '以', 58425: '何', 58426: '可', 58427: '话', 58428: '先',76    58429: 'p', 58430: 'j', 58431: '叫', 58432: '轻', 58433: 'm', 58434: '土', 58435: 'w', 58436: '着', 58437: '变', 58438: '尔',77    58439: '快', 58440: '上', 58441: '个', 58442: '说', 58443: '少', 58444: '色', 58445: '里', 58446: '安', 58447: '花', 58448: '远',78    58449: '7', 58450: '难', 58451: '师', 58452: '放', 58453: '代', 58454: '报', 58455: '认', 58456: '面', 58457: '道', 58458: 's',79    58460: '克', 58461: '地', 58462: '度', 58463: '上', 58464: '好', 58465: '机', 58466: 'u', 58467: '民', 58468: '写', 58469: '把',80    58470: '万', 58471: '同', 58472: '水', 58473: '新', 58474: '没', 58475: '书', 58476: '电', 58477: '吃', 58478: '像', 58479: '斯',81    58480: '5', 58481: '为', 58482: 'v', 58483: '白', 58484: '几l', 58485: '日', 58486: '教', 58487: '看', 58488: '但', 58489: '第',82    58490: '加', 58491: '候', 58492: '作', 58493: '上', 58494: '拉', 58495: '住', 58496: '有', 58497: '法', 58498: 'r', 58499: '事',83    58500: '应', 58501: '位', 58502: '利', 58503: '你', 58504: '声', 58505: '身', 58506: '国', 58507: '问', 58508: '马', 58509: '女',84    58510: '他', 58511: 'y', 58512: '比', 58513: '父', 58515: 'a', 58516: 'h', 58517: 'n', 58518: 's', 58519: 'x', 58520: '边',85    58521: '美', 58522: '对', 58523: '所', 58524: '金', 58525: '活', 58526: '回', 58527: '意', 58528: '到', 58529: '之', 58530: '从',86    58531: 'j', 58532: '知', 58533: '又', 58534: '内', 58535: '因', 58536: '点', 58537: 'o', 58538: '三', 58539: '定', 58540: '8',87    58541: 'r', 58542: 'b', 58543: '正', 58544: '或', 58545: '夫', 58546: '向', 58547: '德', 58548: '听', 58549: '更', 58551: '得',88    58552: '告', 58553: '并', 58554: '本', 58555: 'q', 58556: '过', 58557: '记', 58558: '上', 58559: '让', 58560: '打', 58561: 'f',89    58562: '人', 58563: '就', 58564: '者', 58565: '去', 58566: '原', 58567: '满', 58568: '体', 58569: '做', 58570: '经', 58571: 'k',90    58572: '走', 58573: '如', 58574: '孩', 58575: 'c', 58576: 'g', 58577: '给', 58578: '使', 58579: '物', 58581: '最', 58582: '笑',91    58583: '部', 58585: '员', 58586: '等', 58587: '受', 58588: 'k', 58589: '行', 58591: '条', 58592: '果', 58593: '动', 58594: '光',92    58595: '门', 58596: '头', 58597: '见', 58598: '往', 58599: '自', 58600: '解', 58601: '成', 58602: '处', 58603: '天', 58604: '能',93    58605: '干', 58606: '名', 58607: '其', 58608: '发', 58609: '总', 58610: '母', 58611: '的', 58612: '死', 58613: '手', 58614: '入',94    58615: '路', 58616: '进', 58617: '心', 58618: '来', 58619: 'h', 58620: '时', 58621: '力', 58622: '多', 58623: '开', 58624: '已',95    58625: '许', 58626: 'd', 58627: '至', 58628: '由', 58629: '很', 58630: '界', 58631: 'n', 58632: '小', 58633: '与', 58634: 'z',96    58635: '想', 58636: '代', 58637: '么', 58638: '分', 58639: '生', 58640: '口', 58641: '再', 58642: '妈', 58643: '望', 58644: '次',97    58645: '西', 58646: '风', 58647: '种', 58648: '带', 58649: 'J', 58651: '实', 58652: '情', 58653: '才', 58654: '这', 58656: 'e',98    58657: '我', 58658: '神', 58659: '格', 58660: '长', 58661: '觉', 58662: '间', 58663: '年', 58664: '眼', 58665: '无', 58666: '不',99    58667: '亲', 58668: '关', 58669: '结', 58670: 'o', 58671: '友', 58672: '信', 58673: '下', 58674: '却', 58675: '重', 58676: '己',100    58677: '老', 58678: '2', 58679: '音', 58680: '字', 58681: 'm', 58682: '呢', 58683: '明', 58684: '之', 58685: '前', 58686: '高',101    58687: 'p', 58688: 'b', 58689: '目', 58690: '太', 58691: 'e', 58692: '9', 58693: '起', 58694: '棱', 58695: '她', 58696: '也',102    58697: 'w', 58698: '用', 58699: '方', 58700: '子', 58701: '英', 58702: '每', 58703: '理', 58704: '便', 58705: '四', 58706: '数',103    58707: '期', 58708: '中', 58709: 'c', 58710: '外', 58711: '样', 58712: 'a', 58713: '海', 58714: '们', 58715: '任', 58356: 'v',104    58514: 'x', 65292: ',', 58590: '一', 65311: '?', 65281: '!', 65288: '(', 65289: ')', 12290: '。', 8212: '—',105    8216: '‘', 8217: '’', 8220: '「', 8221: '」', 8230: '…', 12289: '、'106}107 108# ===================== 状态管理器 =====================109class AppStatus:110    """应用状态管理器"""111    112    def __init__(self):113        self.lock = threading.RLock()114        self.status = {115            "message": "系统初始化中...",116            "last_update": None,117            "file_count": 0,118            "processing": False,119            "scheduler_started": False,120            "start_time": datetime.now().isoformat(),121            "last_check_update": None,122            "total_chapters": 0,123            "downloaded_chapters": 0,124            "archive_ready": False,125            "archive_size_mb": 0,126            "check_update_cooldown": 0,127            "current_progress": 0,128            "current_task": None,129            "book_name": "未知书籍",130            "book_id": "unknown"131        }132    133    def update(self, **kwargs):134        """更新状态"""135        with self.lock:136            for key, value in kwargs.items():137                if key in self.status:138                    self.status[key] = value139            self.status["last_update"] = datetime.now().strftime("%H:%M:%S")140    141    def get(self, key=None):142        """获取状态"""143        with self.lock:144            if key:145                return self.status.get(key)146            return self.status.copy()147    148    def set_processing(self, value: bool):149        """设置处理状态"""150        with self.lock:151            self.status["processing"] = value152    153    def is_processing(self) -> bool:154        """检查是否正在处理"""155        with self.lock:156            return self.status["processing"]157    158    def update_progress(self, current: int, total: int, message: str = ""):159        """更新进度"""160        with self.lock:161            if total > 0:162                self.status["current_progress"] = int((current / total) * 100)163            if message:164                self.status["current_task"] = message165 166# ===================== API管理器(增强版) =====================167class APIManager:168    """API管理器 - 包含重试机制和增强的请求头"""169    170    def __init__(self):171        self.session = requests.Session()172        173        # 配置重试策略174        retry_strategy = Retry(175            total=3,176            status_forcelist=[429, 500, 502, 503, 504],177            backoff_factor=1,178            allowed_methods=["HEAD", "GET", "OPTIONS"]179        )180        181        adapter = HTTPAdapter(max_retries=retry_strategy)182        self.session.mount("https://", adapter)183        self.session.mount("http://", adapter)184        185        self.session.headers.update(get_random_headers())186    187    def _update_session_headers(self):188        """更新Session的请求头(轮换User-Agent等)"""189        self.session.headers.update(get_random_headers())190    191    def extract_book_id(self, url: str) -> Optional[str]:192        """从URL中提取书籍ID"""193        patterns = [194            r'page/(\d+)',195            r'book_id=(\d+)',196            r'book/(\d+)',197            r'bid=(\d+)',198            r'novel/(\d+)'199        ]200        201        for pattern in patterns:202            match = re.search(pattern, url)203            if match:204                return match.group(1)205        206        match = re.search(r'(\d{10,})', url)207        if match:208            return match.group(1)209        210        return None211    212    def get_book_detail(self, book_id: str) -> Optional[Dict]:213        """获取书籍详情,带重试机制"""214        for attempt in range(3):215            try:216                self._update_session_headers()217                218                url = f"{API_BASE_URL}{ENDPOINTS['detail']}"219                params = {"book_id": book_id}220                221                response = self.session.get(url, params=params, timeout=30)222                response.raise_for_status()223                224                data = response.json()225                if data.get('code') == 200:226                    book_data = data.get('data', {})227                    if isinstance(book_data, dict) and 'data' in book_data:228                        return book_data['data']229                    return book_data230                231                print(f"获取书籍详情失败,返回码: {data.get('code')}")232                return None233                234            except Exception as e:235                print(f"获取书籍详情失败 (尝试 {attempt + 1}/3): {e}")236                if attempt < 2:237                    time.sleep(random.uniform(2, 5))238                else:239                    return None240    241    def get_chapter_list(self, book_id: str) -> Optional[List[Dict]]:242        """获取章节列表,带重试机制"""243        for attempt in range(3):244            try:245                self._update_session_headers()246                247                url = f"{API_BASE_URL}{ENDPOINTS['chapter_list']}"248                params = {"book_id": book_id}249                250                response = self.session.get(url, params=params, timeout=30)251                response.raise_for_status()252                253                data = response.json()254                if data.get('code') != 200:255                    print(f"获取章节列表失败,返回码: {data.get('code')}")256                    return None257                258                chapter_data = data.get('data', {})259                chapters = []260                261                if isinstance(chapter_data, dict):262                    chapter_sources = [263                        chapter_data.get('chapterListWithVolume', []),264                        chapter_data.get('chapterList', []),265                        chapter_data.get('chapters', []),266                        chapter_data.get('data', []),267                        chapter_data if isinstance(chapter_data, list) else []268                    ]269                    270                    for source in chapter_sources:271                        if isinstance(source, list) and source:272                            self._parse_chapter_source(source, chapters)273                            break274                275                if chapters:276                    return chapters277                    278            except Exception as e:279                print(f"获取章节列表失败 (尝试 {attempt + 1}/3): {e}")280                if attempt < 2:281                    time.sleep(random.uniform(2, 5))282                else:283                    return None284        285        return None286    287    def _parse_chapter_source(self, source: List, chapters: List):288        """解析章节源数据"""289        for item in source:290            if isinstance(item, list):291                for ch in item:292                    self._extract_chapter_info(ch, chapters)293            elif isinstance(item, dict):294                self._extract_chapter_info(item, chapters)295    296    def _extract_chapter_info(self, chapter_data: Dict, chapters: List):297        """从章节数据中提取信息"""298        item_id = chapter_data.get('itemId') or chapter_data.get('item_id') or chapter_data.get('id')299        title = chapter_data.get('title') or chapter_data.get('chapter_title', '未知章节')300        301        if item_id and title:302            chapters.append({303                'id': str(item_id),304                'title': title,305                'index': len(chapters)306            })307    308    def get_chapter_content(self, item_id: str) -> Optional[Dict]:309        """获取章节内容,带重试机制和更长延迟"""310        for attempt in range(3):311            try:312                self._update_session_headers()313                314                time.sleep(random.uniform(3.0, 6.0))315                316                url = f"{API_BASE_URL}{ENDPOINTS['chapter_content']}"317                params = {"item_id": item_id, "tab": "小说"}318                319                response = self.session.get(url, params=params, timeout=30)320                response.raise_for_status()321                322                data = response.json()323                if data.get('code') == 200:324                    content_data = data.get('data', {})325                    return {326                        'title': content_data.get('title', ''),327                        'content': content_data.get('content', ''),328                        'success': True329                    }330                331                print(f"获取章节内容失败,返回码: {data.get('code')} (尝试 {attempt + 1}/3)")332                333            except Exception as e:334                print(f"获取章节内容失败 {item_id} (尝试 {attempt + 1}/3): {e}")335                if attempt < 2:336                    time.sleep(random.uniform(5, 10))337                else:338                    return None339        340        return None341 342# ===================== 内容处理器 =====================343class ContentProcessor:344    """内容处理器"""345    346    def __init__(self):347        self.api = APIManager()348        self.status_manager = StatusManager(DATA_DIR)349    350    def decode_text(self, text: str) -> str:351        """解码文本"""352        if not text:353            return text354        355        result = []356        for char in text:357            char_code = ord(char)358            decoded_char = FONT_MAP.get(char_code, char)359            result.append(decoded_char)360        361        return ''.join(result)362    363    def process_content(self, content: str) -> str:364        """处理内容"""365        if not content:366            return ""367        368        content = self.decode_text(content)369        370        content = re.sub(r'<br\s*/?>', '\n', content, flags=re.IGNORECASE)371        content = re.sub(r'<p[^>]*>', '\n', content, flags=re.IGNORECASE)372        content = re.sub(r'</p>', '\n', content, flags=re.IGNORECASE)373        content = re.sub(r'<[^>]+>', '', content)374        375        content = re.sub(r'[ \t]+', ' ', content)376        content = re.sub(r'\n[ \t]+', '\n', content)377        content = re.sub(r'[ \t]+\n', '\n', content)378        content = re.sub(r'\n{3,}', '\n\n', content)379        380        paragraphs = [p.strip() for p in content.split('\n') if p.strip()]381        return '\n\n'.join(paragraphs)382    383    def fetch_document_info(self, source_url: str) -> Tuple[str, List[Dict]]:384        """获取文档信息"""385        print(f"📖 正在处理: {source_url}")386        387        book_id = self.api.extract_book_id(source_url)388        if not book_id:389            raise ValueError(f"无法从URL中提取书籍ID: {source_url}")390        391        print(f"🔍 提取到书籍ID: {book_id}")392        393        book_detail = self.api.get_book_detail(book_id)394        if not book_detail:395            raise ValueError(f"获取书籍详情失败: {book_id}")396        397        book_name = book_detail.get('book_name') or book_detail.get('title') or f"未知书籍_{book_id}"398        safe_book_name = re.sub(r'[\\/:*?"<>|]', '_', book_name)399        print(f"📚 书籍: {safe_book_name}")400        401        chapters = self.api.get_chapter_list(book_id)402        if not chapters:403            raise ValueError(f"获取章节列表失败: {book_id}")404        405        print(f"📜 找到 {len(chapters)} 个章节")406        407        app_status.update(book_name=safe_book_name, book_id=book_id)408        409        return safe_book_name, chapters410    411    def get_base_dir(self, book_name: str) -> Path:412        """获取基础目录"""413        base_dir = DATA_DIR / book_name414        base_dir.mkdir(exist_ok=True)415        return base_dir416    417    def download_chapters(self, chapters: List[Dict], source_url: str, 418                         base_dir: Path, progress_callback=None) -> Tuple[int, int]:419        """下载章节"""420        downloaded_ids = self.status_manager.load_downloaded_ids(source_url)421        422        chapters_to_download = []423        for chapter in chapters:424            if chapter.get('id') and chapter['id'] not in downloaded_ids:425                chapters_to_download.append(chapter)426        427        if not chapters_to_download:428            print("✅ 所有章节均已下载")429            return 0, 0430        431        print(f"📥 需要下载 {len(chapters_to_download)} 个新章节")432        433        success_count = 0434        failed_count = 0435        436        for idx, chapter in enumerate(chapters_to_download, 1):437            if progress_callback:438                progress_callback(idx, len(chapters_to_download), chapter.get('title', ''))439            440            try:441                content_data = self.api.get_chapter_content(chapter['id'])442                443                if content_data and content_data.get('success'):444                    title = content_data.get('title', chapter['title'])445                    raw_content = content_data.get('content', '')446                    processed_content = self.process_content(raw_content)447                    448                    if processed_content:449                        file_path = base_dir / f"{title}.txt"450                        full_content = f"{title}\n\n{processed_content}"451                        file_path.write_text(full_content, encoding='utf-8')452                        453                        self.status_manager.save_downloaded_id(source_url, chapter['id'])454                        success_count += 1455                        456                        if idx % 5 == 0 or idx == len(chapters_to_download):457                            print(f"✅ {idx}/{len(chapters_to_download)}: {title[:30]}...")458                    else:459                        failed_count += 1460                        print(f"❌ 章节内容为空: {chapter['title']}")461                else:462                    failed_count += 1463                    print(f"❌ 章节下载失败: {chapter['title']}")464                    465            except Exception as e:466                failed_count += 1467                print(f"❌ 下载异常: {e}")468            469            if idx < len(chapters_to_download):470                time.sleep(random.uniform(2, 4))471        472        return success_count, failed_count473    474    def create_archive(self, base_dir: Path) -> str:475        """创建ZIP归档"""476        if not base_dir.exists():477            raise ValueError(f"目录不存在: {base_dir}")478        479        archive_path = base_dir / f"{base_dir.name}.zip"480        481        with zipfile.ZipFile(archive_path, 'w', zipfile.ZIP_DEFLATED) as zipf:482            for txt_file in base_dir.glob("*.txt"):483                if txt_file.is_file() and txt_file.suffix == '.txt':484                    zipf.write(txt_file, txt_file.name)485        486        return str(archive_path)487    488    def get_file_count(self, base_dir: Path) -> int:489        """获取文件数量"""490        return len(list(base_dir.glob("*.txt")))491 492# ===================== 状态管理器 =====================493class StatusManager:494    """下载状态管理器"""495    496    def __init__(self, data_dir: Path):497        self.data_dir = data_dir498        self.status_dir = data_dir / ".status"499        self.status_dir.mkdir(exist_ok=True)500    501    def get_status_file(self, source_url: str) -> Path:502        """获取状态文件路径"""503        url_hash = hashlib.md5(source_url.encode()).hexdigest()[:12]504        return self.status_dir / f"status_{url_hash}.json"505    506    def load_downloaded_ids(self, source_url: str) -> Set[str]:507        """加载已下载的章节ID"""508        status_file = self.get_status_file(source_url)509        510        if status_file.exists():511            try:512                with open(status_file, 'r', encoding='utf-8') as f:513                    return set(json.load(f))514            except (json.JSONDecodeError, UnicodeDecodeError):515                pass516        517        return set()518    519    def save_downloaded_id(self, source_url: str, chapter_id: str):520        """保存已下载的章节ID"""521        status_file = self.get_status_file(source_url)522        downloaded = self.load_downloaded_ids(source_url)523        downloaded.add(chapter_id)524        525        try:526            with open(status_file, 'w', encoding='utf-8') as f:527                json.dump(list(downloaded), f, ensure_ascii=False, indent=2)528        except Exception as e:529            print(f"保存状态失败: {e}")530    531    def clear_status(self, source_url: str):532        """清除下载状态"""533        status_file = self.get_status_file(source_url)534        if status_file.exists():535            try:536                status_file.unlink()537            except Exception as e:538                print(f"清除状态失败: {e}")539 540# ===================== 全局实例 =====================541app_status = AppStatus()542processor = ContentProcessor()543 544# ===================== 工具函数 =====================545def update_status(message: str):546    """更新状态并打印日志"""547    app_status.update(message=message)548    print(f"📌 {message}")549 550def can_check_update() -> bool:551    """检查是否可以执行更新检查"""552    last_check = app_status.get("last_check_update")553    if not last_check:554        return True555    556    try:557        last_time = datetime.fromisoformat(last_check)558        return datetime.now() - last_time >= timedelta(minutes=10)559    except:560        return True561 562def get_cooldown_remaining() -> int:563    """获取冷却剩余时间(秒)"""564    last_check = app_status.get("last_check_update")565    if not last_check:566        return 0567    568    try:569        last_time = datetime.fromisoformat(last_check)570        next_time = last_time + timedelta(minutes=10)571        remaining = (next_time - datetime.now()).total_seconds()572        return max(0, int(remaining))573    except:574        return 0575 576def process_all(is_manual: bool = False):577    """处理所有章节下载"""578    if app_status.is_processing():579        update_status("⏳ 已有任务在运行,请稍后...")580        return581    582    if is_manual:583        if not can_check_update():584            remaining = get_cooldown_remaining()585            minutes = remaining // 60586            seconds = remaining % 60587            update_status(f"⏰ 检查太频繁了,请等待 {minutes}分{seconds}秒 后再试")588            return589        app_status.update(last_check_update=datetime.now().isoformat())590    591    app_status.set_processing(True)592    593    try:594        update_status("📖 正在获取书籍信息...")595        book_name, chapters = processor.fetch_document_info(SOURCE_URL)596        597        if not chapters:598            update_status("❌ 未找到任何章节")599            return600        601        base_dir = processor.get_base_dir(book_name)602        app_status.update(total_chapters=len(chapters))603        604        existing_count = processor.get_file_count(base_dir)605        app_status.update(downloaded_chapters=existing_count)606        607        update_status(f"📚 书籍: {book_name}")608        update_status(f"📜 总共 {len(chapters)} 个章节,已有 {existing_count} 个")609        610        def progress_callback(current, total, title):611            app_status.update_progress(current, total, f"正在下载: {title}")612            if current % 5 == 0 or current == total:613                update_status(f"⏬ {current}/{total}: {title[:30]}...")614        615        success, failed = processor.download_chapters(616            chapters, SOURCE_URL, base_dir, progress_callback617        )618        619        new_count = processor.get_file_count(base_dir)620        app_status.update(file_count=new_count, downloaded_chapters=new_count)621        622        if success > 0:623            update_status("🗜️ 正在创建ZIP归档...")624            try:625                archive_path = processor.create_archive(base_dir)626                627                if Path(archive_path).exists():628                    size_mb = Path(archive_path).stat().st_size / (1024 * 1024)629                    app_status.update(archive_ready=True, archive_size_mb=round(size_mb, 2))630                    update_status(f"✅ 归档完成 ({size_mb:.2f} MB)")631            except Exception as e:632                update_status(f"❌ 创建归档失败: {e}")633        634        action_type = "手动检查" if is_manual else "定时检查"635        final_msg = f"🎉 {action_type}完成!成功:{success} | 失败:{failed} | 总计:{new_count}"636        update_status(final_msg)637        638        app_status.update_progress(0, 100, "空闲")639        640    except Exception as e:641        error_msg = f"❌ 处理失败: {str(e)}"642        update_status(error_msg)643        print(f"详细错误: {e}")644        traceback.print_exc()645        646    finally:647        app_status.set_processing(False)648 649# ===================== 定时任务 =====================650def scheduler_thread():651    """定时任务线程"""652    time.sleep(10)653    update_status("⏰ 定时任务已激活,将每小时检查一次")654    655    while True:656        try:657            if not app_status.is_processing():658                process_all(is_manual=False)659                update_status(f"⏰ 下次检查将在 {CHECK_INTERVAL//60} 分钟后执行")660            661            time.sleep(CHECK_INTERVAL)662        except Exception as e:663            update_status(f"❌ 定时任务异常: {e},30分钟后重试")664            time.sleep(1800)665 666# ===================== FastAPI应用 =====================667@asynccontextmanager668async def lifespan(app: FastAPI):669    """应用生命周期管理"""670    print("🚀 正在启动应用...")671    672    try:673        thread = threading.Thread(target=scheduler_thread, daemon=True)674        thread.start()675        app_status.update(scheduler_started=True)676        print("✅ 定时任务线程已启动")677    except Exception as e:678        print(f"❌ 启动定时任务失败: {e}")679    680    update_status("✅ 应用启动完成")681    print("✅ 应用启动完成,开始服务")682    yield683    684    print("⏹️ 应用正在关闭...")685 686app = FastAPI(687    title="番茄小说下载器",688    description="支持完整章节下载的小说下载器",689    version="2.0.0",690    docs_url="/docs",691    redoc_url="/redoc",692    lifespan=lifespan693)694 695# ===================== 重要的健康检查路由 =====================696@app.get("/health")697async def health():698    """健康检查"""699    return JSONResponse(700        content={701            "status": "healthy",702            "timestamp": datetime.now().isoformat(),703            "service": "fanqie-novel-downloader",704            "version": "2.0.0",705            "uptime": str(datetime.now() - datetime.fromisoformat(app_status.get("start_time")))706        },707        status_code=200708    )709 710@app.head("/health")711async def health_head():712    """HEAD方法健康检查"""713    return JSONResponse(714        content={"status": "healthy"},715        status_code=200,716        headers={"Content-Type": "application/json"}717    )718 719@app.get("/keepalive")720async def keepalive():721    """保活接口"""722    return JSONResponse(723        content={724            "status": "alive", 725            "timestamp": datetime.now().isoformat(),726            "message": "服务运行正常"727        },728        status_code=200729    )730 731@app.head("/keepalive")732async def keepalive_head():733    """HEAD方法保活接口"""734    return JSONResponse(735        content={"status": "alive"},736        status_code=200,737        headers={"Content-Type": "application/json"}738    )739 740@app.get("/api/health")741async def api_health():742    """API健康检查"""743    return await health()744 745# ===================== Web界面路由 =====================746@app.get("/", response_class=HTMLResponse)747async def root():748    """根路径 - 返回Web界面"""749    return """750    <!DOCTYPE html>751    <html lang="zh-CN">752    <head>753        <meta charset="UTF-8">754        <meta name="viewport" content="width=device-width, initial-scale=1.0">755        <title>番茄小说下载器</title>756        <style>757            * { margin: 0; padding: 0; box-sizing: border-box; }758            body { 759                font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, 'Helvetica Neue', Arial, sans-serif;760                line-height: 1.6; color: #333; background: linear-gradient(135deg, #667eea 0%, #764ba2 100%);761                min-height: 100vh; padding: 20px;762            }763            .container { 764                max-width: 1400px; margin: 0 auto; background: white; border-radius: 15px;765                padding: 30px; box-shadow: 0 20px 40px rgba(0,0,0,0.1); min-height: 600px;766            }767            .header { text-align: center; margin-bottom: 30px; }768            .header h1 { font-size: 2.5rem; margin-bottom: 10px; color: #333; }769            .header p { font-size: 1.2rem; color: #666; }770            .content { display: grid; grid-template-columns: 1fr 2fr; gap: 30px; }771            .control-panel { padding-right: 20px; border-right: 1px solid #eee; }772            .control-panel h2, .info-panel h2 { margin-bottom: 20px; color: #333; font-size: 1.5rem; }773            .button-group { display: flex; flex-direction: column; gap: 10px; margin-bottom: 20px; }774            .btn-primary, .btn-secondary, .btn-success {775                padding: 12px 20px; border: none; border-radius: 8px; font-size: 16px;776                font-weight: 600; cursor: pointer; transition: all 0.3s ease; text-align: center;777            }778            .btn-primary { background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); color: white; }779            .btn-secondary { background: #f8f9fa; color: #495057; border: 1px solid #dee2e6; }780            .btn-success { background: linear-gradient(135deg, #5cb85c 0%, #449d44 100%); color: white; }781            .status-box { background: #f8f9fa; border: 1px solid #dee2e6; border-radius: 8px; padding: 15px; margin-bottom: 20px; }782            .cooldown-info { background: #fff3cd; border: 1px solid #ffeaa7; border-radius: 8px; padding: 10px; margin-top: 10px; }783            .progress-container { margin: 20px 0; }784            .progress-bar { height: 10px; background: #e9ecef; border-radius: 5px; overflow: hidden; }785            .progress-fill { height: 100%; background: linear-gradient(90deg, #667eea 0%, #764ba2 100%); border-radius: 5px; transition: width 0.3s ease; }786            .file-list { max-height: 300px; overflow-y: auto; }787            @media (max-width: 1024px) { .content { grid-template-columns: 1fr; } .control-panel { border-right: none; border-bottom: 1px solid #eee; padding-bottom: 20px; } }788        </style>789    </head>790    <body>791        <div class="container">792            <div class="header">793                <h1>📚 番茄小说下载器</h1>794                <p>支持完整章节下载,解决11章后无法下载的问题</p>795            </div>796            797            <div class="content">798                <div class="control-panel">799                    <h2>🎮 控制面板</h2>800                    <div class="button-group">801                        <button id="btn-check-update" class="btn-primary">🔍 检查更新</button>802                        <button id="btn-refresh" class="btn-secondary">🔄 刷新状态</button>803                        <button id="btn-download" class="btn-success">📥 下载归档</button>804                    </div>805                    806                    <div class="cooldown-info" id="cooldown-info" style="display: none;">807                        <small>⏰ 冷却中:<span id="cooldown-time"></span></small>808                    </div>809                    810                    <div class="progress-container" id="progress-container" style="display: none;">811                        <div class="progress-label" id="progress-label">正在处理...</div>812                        <div class="progress-bar">813                            <div class="progress-fill" id="progress-fill" style="width: 0%"></div>814                        </div>815                        <div class="progress-text" id="progress-text">0%</div>816                    </div>817                    818                    <h2>📥 归档下载</h2>819                    <div id="archive-status" class="status-box">820                        <div>暂无归档文件</div>821                    </div>822                </div>823                824                <div class="info-panel">825                    <h2>📊 系统状态</h2>826                    <div class="status-box">827                        <strong>状态:</strong><span id="status-message">加载中...</span><br>828                        <strong>最后更新:</strong><span id="last-update">-</span><br>829                        <strong>书籍:</strong><span id="book-name">未知</span><br>830                        <strong>章节总数:</strong><span id="total-chapters">0</span><br>831                        <strong>已下载:</strong><span id="downloaded-chapters">0</span><br>832                        <strong>文件数量:</strong><span id="file-count">0</span>833                    </div>834                    835                    <h2>📄 已下载章节</h2>836                    <div id="file-list" class="file-list status-box">加载中...</div>837                    838                    <h2>📋 系统信息</h2>839                    <div class="status-box">840                        <strong>服务状态:</strong><span id="service-status">运行中 ✅</span><br>841                        <strong>启动时间:</strong><span id="start-time">-</span><br>842                        <strong>当前时间:</strong><span id="current-time">-</span><br>843                        <strong>定时任务:</strong><span id="scheduler-status">运行中</span>844                    </div>845                </div>846            </div>847        </div>848        849        <script>850        class AppState {851            constructor() {852                this.updateInterval = null;853                this.cooldownInterval = null;854            }855            856            async updateStatus() {857                try {858                    const response = await fetch('/api/status');859                    const data = await response.json();860                    861                    document.getElementById('status-message').textContent = data.message || '未知';862                    document.getElementById('last-update').textContent = data.last_update || '-';863                    document.getElementById('book-name').textContent = data.book_name || '未知';864                    document.getElementById('total-chapters').textContent = data.total_chapters || 0;865                    document.getElementById('downloaded-chapters').textContent = data.downloaded_chapters || 0;866                    document.getElementById('file-count').textContent = data.file_count || 0;867                    document.getElementById('start-time').textContent = new Date(data.start_time).toLocaleString();868                    document.getElementById('scheduler-status').textContent = data.scheduler_started ? '运行中 ✅' : '停止 ❌';869                    870                    const progressContainer = document.getElementById('progress-container');871                    const progressFill = document.getElementById('progress-fill');872                    const progressText = document.getElementById('progress-text');873                    const progressLabel = document.getElementById('progress-label');874                    875                    if (data.processing) {876                        progressContainer.style.display = 'block';877                        progressFill.style.width = `${data.current_progress || 0}%`;878                        progressText.textContent = `${data.current_progress || 0}%`;879                        progressLabel.textContent = data.current_task || '正在处理...';880                    } else {881                        progressContainer.style.display = 'none';882                    }883                    884                    const archiveStatus = document.getElementById('archive-status');885                    if (data.archive_ready) {886                        archiveStatus.innerHTML = `887                            <strong>归档文件:</strong> ${(data.archive_size_mb || 0).toFixed(2)} MB<br>888                            <small>点击下载按钮获取完整小说</small>889                        `;890                    } else {891                        archiveStatus.innerHTML = '<div>暂无归档文件,请先检查更新</div>';892                    }893                    894                    const checkBtn = document.getElementById('btn-check-update');895                    if (checkBtn) {896                        if (data.processing) {897                            checkBtn.disabled = true;898                            checkBtn.innerHTML = '⏳ 处理中...';899                        } else if (data.check_update_cooldown > 0) {900                            checkBtn.disabled = true;901                            const minutes = Math.floor(data.check_update_cooldown / 60);902                            const seconds = data.check_update_cooldown % 60;903                            checkBtn.innerHTML = `⏰ 冷却中 (${minutes}:${seconds.toString().padStart(2, '0')})`;904                        } else {905                            checkBtn.disabled = false;906                            checkBtn.innerHTML = '🔍 检查更新';907                        }908                    }909                    910                    const downloadBtn = document.getElementById('btn-download');911                    if (downloadBtn) {912                        downloadBtn.disabled = !data.archive_ready;913                    }914                    915                } catch (error) {916                    console.error('获取状态失败:', error);917                }918            }919            920            async updateFileList() {921                try {922                    const response = await fetch('/api/files');923                    const data = await response.json();924                    925                    const fileList = document.getElementById('file-list');926                    if (data.status === 'success' && data.files && data.files.length > 0) {927                        const fileItems = data.files.map(file => `928                            <div style="padding: 5px 0; border-bottom: 1px solid #eee;">929                                ${file.name} (${Math.round(file.size/1024)}KB)930                            </div>931                        `).join('');932                        fileList.innerHTML = fileItems;933                    } else {934                        fileList.innerHTML = '<div>暂无文件</div>';935                    }936                } catch (error) {937                    console.error('获取文件列表失败:', error);938                }939            }940            941            startPolling() {942                this.updateStatus();943                this.updateFileList();944                945                this.updateInterval = setInterval(() => {946                    this.updateStatus();947                }, 3000);948                949                setInterval(() => {950                    this.updateFileList();951                }, 30000);952            }953            954            stopPolling() {955                if (this.updateInterval) {956                    clearInterval(this.updateInterval);957                }958            }959        }960        961        const appState = new AppState();962        963        async function checkUpdate() {964            const button = document.getElementById('btn-check-update');965            button.disabled = true;966            button.innerHTML = '⏳ 请求中...';967            968            try {969                const response = await fetch('/api/check-update', { method: 'POST' });970                const result = await response.json();971                972                if (response.ok) {973                    alert(result.message);974                } else {975                    alert('错误: ' + (result.detail || '未知错误'));976                }977            } catch (error) {978                alert('请求失败: ' + error.message);979            }980        }981        982        async function downloadArchive() {983            window.location.href = '/api/download';984        }985        986        function updateCurrentTime() {987            const now = new Date();988            document.getElementById('current-time').textContent = now.toLocaleString();989        }990        991        document.addEventListener('DOMContentLoaded', () => {992            document.getElementById('btn-check-update').addEventListener('click', checkUpdate);993            document.getElementById('btn-refresh').addEventListener('click', () => {994                appState.updateStatus();995                appState.updateFileList();996            });997            document.getElementById('btn-download').addEventListener('click', downloadArchive);998            999            appState.startPolling();1000            updateCurrentTime();1001            setInterval(updateCurrentTime, 1000);1002        });1003        </script>1004    </body>1005    </html>1006    """1007 1008# ===================== API路由 =====================1009@app.get("/api/status")1010async def api_status():1011    """获取系统状态"""1012    status = app_status.get()1013    status["timestamp"] = datetime.now().isoformat()1014    status["check_update_cooldown"] = get_cooldown_remaining()1015    1016    try:1017        book_name = app_status.get("book_name")1018        if book_name and book_name != "未知书籍":1019            base_dir = processor.get_base_dir(book_name)1020            archive_path = base_dir / f"{book_name}.zip"1021            1022            if archive_path.exists():1023                status["archive_ready"] = True1024                status["archive_size_mb"] = round(archive_path.stat().st_size / (1024 * 1024), 2)1025    except Exception as e:1026        print(f"检查归档文件失败: {e}")1027    1028    return status1029 1030@app.post("/api/check-update")1031async def check_update():1032    """手动触发更新检查"""1033    if app_status.is_processing():1034        raise HTTPException(status_code=429, detail="已有任务在运行中,请稍后再试")1035    1036    if not can_check_update():1037        remaining = get_cooldown_remaining()1038        minutes = remaining // 601039        seconds = remaining % 601040        raise HTTPException(1041            status_code=429, 1042            detail=f"检查太频繁了,请等待 {minutes}分{seconds}秒 后再试"1043        )1044    1045    threading.Thread(target=process_all, args=(True,), daemon=True).start()1046    1047    return {1048        "status": "success", 1049        "message": "检查更新任务已启动!",1050        "next_check_allowed_in": 6001051    }1052 1053@app.get("/api/files")1054async def get_files():1055    """获取文件列表"""1056    try:1057        book_name = app_status.get("book_name")1058        if not book_name or book_name == "未知书籍":1059            return {"status": "success", "files": [], "count": 0}1060        1061        base_dir = processor.get_base_dir(book_name)1062        1063        files = []1064        for txt_file in base_dir.glob("*.txt"):1065            files.append({1066                "name": txt_file.name,1067                "size": txt_file.stat().st_size,1068                "modified": datetime.fromtimestamp(txt_file.stat().st_mtime).isoformat()1069            })1070        1071        return {1072            "status": "success", 1073            "files": sorted(files, key=lambda x: x["name"]), 1074            "count": len(files)1075        }1076    except Exception as e:1077        return {"status": "error", "message": str(e)}1078 1079@app.get("/api/download")1080async def download_archive():1081    """下载归档文件"""1082    try:1083        book_name = app_status.get("book_name")1084        if not book_name or book_name == "未知书籍":1085            raise HTTPException(status_code=404, detail="请先执行检查更新以获取书籍信息")1086        1087        base_dir = processor.get_base_dir(book_name)1088        archive_path = base_dir / f"{book_name}.zip"1089        1090        if not archive_path.exists():1091            try:1092                processor.create_archive(base_dir)1093            except Exception as e:1094                raise HTTPException(status_code=404, detail=f"归档文件不存在且创建失败: {str(e)}")1095        1096        if not archive_path.exists():1097            raise HTTPException(status_code=404, detail="归档文件不存在,请先执行检查更新任务")1098        1099        return FileResponse(1100            path=archive_path,1101            filename=f"{book_name}.zip",1102            media_type="application/zip"1103        )1104    except HTTPException:1105        raise1106    except Exception as e:1107        raise HTTPException(status_code=500, detail=str(e))1108 1109# ===================== 静态文件服务 =====================1110static_dir = Path("static")1111static_dir.mkdir(exist_ok=True)1112 1113app.mount("/static", StaticFiles(directory="static"), name="static")1114 1115# ===================== 启动应用(修复版) =====================1116if __name__ == "__main__":1117    port = int(os.getenv("PORT", 7860))1118    1119    # 关键修复:使用 server.run() 而不是 asyncio.run()1120    # 这样可以保持主线程运行,防止应用立即退出1121    uvicorn.run(1122        app,1123        host="0.0.0.0",1124        port=port,1125        access_log=True,1126        log_level="info",1127        loop="asyncio",1128        lifespan="on"1129    )