xferrr/Documentprocessing
0
1import os2import time3import json4import random5import threading6import hashlib7import zipfile8import re9import traceback10from pathlib import Path11from datetime import datetime, timedelta12from contextlib import asynccontextmanager13from typing import Dict, List, Tuple, Optional, Set14import requests15from requests.adapters import HTTPAdapter16from urllib3.util.retry import Retry17from fastapi import FastAPI, HTTPException18from fastapi.staticfiles import StaticFiles19from fastapi.responses import FileResponse, HTMLResponse, JSONResponse20import uvicorn21import asyncio22 23# ===================== 配置 =====================24SOURCE_URL = os.getenv("SOURCE_URL", "https://fanqienovel.com/page/7569914141335374910").strip()25CHECK_INTERVAL = int(os.getenv("CHECK_INTERVAL", "3600"))26DATA_DIR = Path("./novel_cache")27DATA_DIR.mkdir(exist_ok=True)28 29# ===================== API配置 =====================30API_BASE_URL = "https://fq.shusan.cn"31ENDPOINTS = {32 'detail': '/api/v1/book/detail',33 'chapter_list': '/api/v1/book/chapter-list',34 'chapter_content': '/api/v1/chapter/content',35}36 37# ===================== 增强的请求头(模拟真实浏览器) =====================38USER_AGENTS = [39 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',40 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/119.0.0.0 Safari/537.36',41 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',42 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'43]44 45def get_random_headers() -> Dict[str, str]:46 """获取随机请求头,模拟真实浏览器"""47 return {48 'User-Agent': random.choice(USER_AGENTS),49 'Accept': 'application/json, text/plain, */*',50 'Accept-Language': 'zh-CN,zh;q=0.9,en-US;q=0.8,en;q=0.7',51 'Accept-Encoding': 'gzip, deflate, br',52 'Referer': 'https://fanqienovel.com/',53 'Origin': 'https://fanqienovel.com',54 'Connection': 'keep-alive',55 'Cache-Control': 'no-cache',56 'Pragma': 'no-cache',57 'Sec-Ch-Ua': '"Not_A Brand";v="8", "Chromium";v="120", "Google Chrome";v="120"',58 'Sec-Ch-Ua-Mobile': '?0',59 'Sec-Ch-Ua-Platform': '"Windows"',60 'Sec-Fetch-Dest': 'empty',61 'Sec-Fetch-Mode': 'cors',62 'Sec-Fetch-Site': 'cross-site',63 'Dnt': '1',64 }65 66# ===================== 字体解码映射 =====================67FONT_MAP = {68 58344: 'd', 58345: '在', 58346: '主', 58347: '特', 58348: '家', 58349: '军', 58350: '然', 58351: '表', 58352: '场', 58353: '4',69 58354: '要', 58355: '只', 58357: '和', 58359: '6', 58360: '别', 58361: '还', 58362: 'g', 58363: '现', 58364: '儿L', 58365: '岁',70 58368: '此', 58369: '象', 58370: '月', 58371: '3', 58372: '出', 58373: '战', 58374: '工', 58375: '相', 58376: '。', 58377: '男',71 58378: '直', 58379: '失', 58380: '世', 58381: 'f', 58382: '都', 58383: '平', 58384: '文', 58385: '什', 58386: 'v', 58387: 'o',72 58388: '将', 58389: '真', 58390: '工', 58391: '那', 58392: '当', 58394: '会', 58395: '立', 58396: '些', 58397: '山', 58398: '是',73 58399: '十', 58400: '张', 58401: '学', 58402: '气', 58403: '大', 58404: '爱', 58405: '两', 58406: '命', 58407: '全', 58408: '后',74 58409: '东', 58410: '性', 58411: '通', 58412: '被', 58413: '1', 58414: '它', 58415: '乐', 58416: '接', 58417: '而', 58418: '感',75 58419: '车', 58420: '山', 58421: '公', 58422: '了', 58423: '常', 58424: '以', 58425: '何', 58426: '可', 58427: '话', 58428: '先',76 58429: 'p', 58430: 'j', 58431: '叫', 58432: '轻', 58433: 'm', 58434: '土', 58435: 'w', 58436: '着', 58437: '变', 58438: '尔',77 58439: '快', 58440: '上', 58441: '个', 58442: '说', 58443: '少', 58444: '色', 58445: '里', 58446: '安', 58447: '花', 58448: '远',78 58449: '7', 58450: '难', 58451: '师', 58452: '放', 58453: '代', 58454: '报', 58455: '认', 58456: '面', 58457: '道', 58458: 's',79 58460: '克', 58461: '地', 58462: '度', 58463: '上', 58464: '好', 58465: '机', 58466: 'u', 58467: '民', 58468: '写', 58469: '把',80 58470: '万', 58471: '同', 58472: '水', 58473: '新', 58474: '没', 58475: '书', 58476: '电', 58477: '吃', 58478: '像', 58479: '斯',81 58480: '5', 58481: '为', 58482: 'v', 58483: '白', 58484: '几l', 58485: '日', 58486: '教', 58487: '看', 58488: '但', 58489: '第',82 58490: '加', 58491: '候', 58492: '作', 58493: '上', 58494: '拉', 58495: '住', 58496: '有', 58497: '法', 58498: 'r', 58499: '事',83 58500: '应', 58501: '位', 58502: '利', 58503: '你', 58504: '声', 58505: '身', 58506: '国', 58507: '问', 58508: '马', 58509: '女',84 58510: '他', 58511: 'y', 58512: '比', 58513: '父', 58515: 'a', 58516: 'h', 58517: 'n', 58518: 's', 58519: 'x', 58520: '边',85 58521: '美', 58522: '对', 58523: '所', 58524: '金', 58525: '活', 58526: '回', 58527: '意', 58528: '到', 58529: '之', 58530: '从',86 58531: 'j', 58532: '知', 58533: '又', 58534: '内', 58535: '因', 58536: '点', 58537: 'o', 58538: '三', 58539: '定', 58540: '8',87 58541: 'r', 58542: 'b', 58543: '正', 58544: '或', 58545: '夫', 58546: '向', 58547: '德', 58548: '听', 58549: '更', 58551: '得',88 58552: '告', 58553: '并', 58554: '本', 58555: 'q', 58556: '过', 58557: '记', 58558: '上', 58559: '让', 58560: '打', 58561: 'f',89 58562: '人', 58563: '就', 58564: '者', 58565: '去', 58566: '原', 58567: '满', 58568: '体', 58569: '做', 58570: '经', 58571: 'k',90 58572: '走', 58573: '如', 58574: '孩', 58575: 'c', 58576: 'g', 58577: '给', 58578: '使', 58579: '物', 58581: '最', 58582: '笑',91 58583: '部', 58585: '员', 58586: '等', 58587: '受', 58588: 'k', 58589: '行', 58591: '条', 58592: '果', 58593: '动', 58594: '光',92 58595: '门', 58596: '头', 58597: '见', 58598: '往', 58599: '自', 58600: '解', 58601: '成', 58602: '处', 58603: '天', 58604: '能',93 58605: '干', 58606: '名', 58607: '其', 58608: '发', 58609: '总', 58610: '母', 58611: '的', 58612: '死', 58613: '手', 58614: '入',94 58615: '路', 58616: '进', 58617: '心', 58618: '来', 58619: 'h', 58620: '时', 58621: '力', 58622: '多', 58623: '开', 58624: '已',95 58625: '许', 58626: 'd', 58627: '至', 58628: '由', 58629: '很', 58630: '界', 58631: 'n', 58632: '小', 58633: '与', 58634: 'z',96 58635: '想', 58636: '代', 58637: '么', 58638: '分', 58639: '生', 58640: '口', 58641: '再', 58642: '妈', 58643: '望', 58644: '次',97 58645: '西', 58646: '风', 58647: '种', 58648: '带', 58649: 'J', 58651: '实', 58652: '情', 58653: '才', 58654: '这', 58656: 'e',98 58657: '我', 58658: '神', 58659: '格', 58660: '长', 58661: '觉', 58662: '间', 58663: '年', 58664: '眼', 58665: '无', 58666: '不',99 58667: '亲', 58668: '关', 58669: '结', 58670: 'o', 58671: '友', 58672: '信', 58673: '下', 58674: '却', 58675: '重', 58676: '己',100 58677: '老', 58678: '2', 58679: '音', 58680: '字', 58681: 'm', 58682: '呢', 58683: '明', 58684: '之', 58685: '前', 58686: '高',101 58687: 'p', 58688: 'b', 58689: '目', 58690: '太', 58691: 'e', 58692: '9', 58693: '起', 58694: '棱', 58695: '她', 58696: '也',102 58697: 'w', 58698: '用', 58699: '方', 58700: '子', 58701: '英', 58702: '每', 58703: '理', 58704: '便', 58705: '四', 58706: '数',103 58707: '期', 58708: '中', 58709: 'c', 58710: '外', 58711: '样', 58712: 'a', 58713: '海', 58714: '们', 58715: '任', 58356: 'v',104 58514: 'x', 65292: ',', 58590: '一', 65311: '?', 65281: '!', 65288: '(', 65289: ')', 12290: '。', 8212: '—',105 8216: '‘', 8217: '’', 8220: '「', 8221: '」', 8230: '…', 12289: '、'106}107 108# ===================== 状态管理器 =====================109class AppStatus:110 """应用状态管理器"""111 112 def __init__(self):113 self.lock = threading.RLock()114 self.status = {115 "message": "系统初始化中...",116 "last_update": None,117 "file_count": 0,118 "processing": False,119 "scheduler_started": False,120 "start_time": datetime.now().isoformat(),121 "last_check_update": None,122 "total_chapters": 0,123 "downloaded_chapters": 0,124 "archive_ready": False,125 "archive_size_mb": 0,126 "check_update_cooldown": 0,127 "current_progress": 0,128 "current_task": None,129 "book_name": "未知书籍",130 "book_id": "unknown"131 }132 133 def update(self, **kwargs):134 """更新状态"""135 with self.lock:136 for key, value in kwargs.items():137 if key in self.status:138 self.status[key] = value139 self.status["last_update"] = datetime.now().strftime("%H:%M:%S")140 141 def get(self, key=None):142 """获取状态"""143 with self.lock:144 if key:145 return self.status.get(key)146 return self.status.copy()147 148 def set_processing(self, value: bool):149 """设置处理状态"""150 with self.lock:151 self.status["processing"] = value152 153 def is_processing(self) -> bool:154 """检查是否正在处理"""155 with self.lock:156 return self.status["processing"]157 158 def update_progress(self, current: int, total: int, message: str = ""):159 """更新进度"""160 with self.lock:161 if total > 0:162 self.status["current_progress"] = int((current / total) * 100)163 if message:164 self.status["current_task"] = message165 166# ===================== API管理器(增强版) =====================167class APIManager:168 """API管理器 - 包含重试机制和增强的请求头"""169 170 def __init__(self):171 self.session = requests.Session()172 173 # 配置重试策略174 retry_strategy = Retry(175 total=3,176 status_forcelist=[429, 500, 502, 503, 504],177 backoff_factor=1,178 allowed_methods=["HEAD", "GET", "OPTIONS"]179 )180 181 adapter = HTTPAdapter(max_retries=retry_strategy)182 self.session.mount("https://", adapter)183 self.session.mount("http://", adapter)184 185 self.session.headers.update(get_random_headers())186 187 def _update_session_headers(self):188 """更新Session的请求头(轮换User-Agent等)"""189 self.session.headers.update(get_random_headers())190 191 def extract_book_id(self, url: str) -> Optional[str]:192 """从URL中提取书籍ID"""193 patterns = [194 r'page/(\d+)',195 r'book_id=(\d+)',196 r'book/(\d+)',197 r'bid=(\d+)',198 r'novel/(\d+)'199 ]200 201 for pattern in patterns:202 match = re.search(pattern, url)203 if match:204 return match.group(1)205 206 match = re.search(r'(\d{10,})', url)207 if match:208 return match.group(1)209 210 return None211 212 def get_book_detail(self, book_id: str) -> Optional[Dict]:213 """获取书籍详情,带重试机制"""214 for attempt in range(3):215 try:216 self._update_session_headers()217 218 url = f"{API_BASE_URL}{ENDPOINTS['detail']}"219 params = {"book_id": book_id}220 221 response = self.session.get(url, params=params, timeout=30)222 response.raise_for_status()223 224 data = response.json()225 if data.get('code') == 200:226 book_data = data.get('data', {})227 if isinstance(book_data, dict) and 'data' in book_data:228 return book_data['data']229 return book_data230 231 print(f"获取书籍详情失败,返回码: {data.get('code')}")232 return None233 234 except Exception as e:235 print(f"获取书籍详情失败 (尝试 {attempt + 1}/3): {e}")236 if attempt < 2:237 time.sleep(random.uniform(2, 5))238 else:239 return None240 241 def get_chapter_list(self, book_id: str) -> Optional[List[Dict]]:242 """获取章节列表,带重试机制"""243 for attempt in range(3):244 try:245 self._update_session_headers()246 247 url = f"{API_BASE_URL}{ENDPOINTS['chapter_list']}"248 params = {"book_id": book_id}249 250 response = self.session.get(url, params=params, timeout=30)251 response.raise_for_status()252 253 data = response.json()254 if data.get('code') != 200:255 print(f"获取章节列表失败,返回码: {data.get('code')}")256 return None257 258 chapter_data = data.get('data', {})259 chapters = []260 261 if isinstance(chapter_data, dict):262 chapter_sources = [263 chapter_data.get('chapterListWithVolume', []),264 chapter_data.get('chapterList', []),265 chapter_data.get('chapters', []),266 chapter_data.get('data', []),267 chapter_data if isinstance(chapter_data, list) else []268 ]269 270 for source in chapter_sources:271 if isinstance(source, list) and source:272 self._parse_chapter_source(source, chapters)273 break274 275 if chapters:276 return chapters277 278 except Exception as e:279 print(f"获取章节列表失败 (尝试 {attempt + 1}/3): {e}")280 if attempt < 2:281 time.sleep(random.uniform(2, 5))282 else:283 return None284 285 return None286 287 def _parse_chapter_source(self, source: List, chapters: List):288 """解析章节源数据"""289 for item in source:290 if isinstance(item, list):291 for ch in item:292 self._extract_chapter_info(ch, chapters)293 elif isinstance(item, dict):294 self._extract_chapter_info(item, chapters)295 296 def _extract_chapter_info(self, chapter_data: Dict, chapters: List):297 """从章节数据中提取信息"""298 item_id = chapter_data.get('itemId') or chapter_data.get('item_id') or chapter_data.get('id')299 title = chapter_data.get('title') or chapter_data.get('chapter_title', '未知章节')300 301 if item_id and title:302 chapters.append({303 'id': str(item_id),304 'title': title,305 'index': len(chapters)306 })307 308 def get_chapter_content(self, item_id: str) -> Optional[Dict]:309 """获取章节内容,带重试机制和更长延迟"""310 for attempt in range(3):311 try:312 self._update_session_headers()313 314 time.sleep(random.uniform(3.0, 6.0))315 316 url = f"{API_BASE_URL}{ENDPOINTS['chapter_content']}"317 params = {"item_id": item_id, "tab": "小说"}318 319 response = self.session.get(url, params=params, timeout=30)320 response.raise_for_status()321 322 data = response.json()323 if data.get('code') == 200:324 content_data = data.get('data', {})325 return {326 'title': content_data.get('title', ''),327 'content': content_data.get('content', ''),328 'success': True329 }330 331 print(f"获取章节内容失败,返回码: {data.get('code')} (尝试 {attempt + 1}/3)")332 333 except Exception as e:334 print(f"获取章节内容失败 {item_id} (尝试 {attempt + 1}/3): {e}")335 if attempt < 2:336 time.sleep(random.uniform(5, 10))337 else:338 return None339 340 return None341 342# ===================== 内容处理器 =====================343class ContentProcessor:344 """内容处理器"""345 346 def __init__(self):347 self.api = APIManager()348 self.status_manager = StatusManager(DATA_DIR)349 350 def decode_text(self, text: str) -> str:351 """解码文本"""352 if not text:353 return text354 355 result = []356 for char in text:357 char_code = ord(char)358 decoded_char = FONT_MAP.get(char_code, char)359 result.append(decoded_char)360 361 return ''.join(result)362 363 def process_content(self, content: str) -> str:364 """处理内容"""365 if not content:366 return ""367 368 content = self.decode_text(content)369 370 content = re.sub(r'<br\s*/?>', '\n', content, flags=re.IGNORECASE)371 content = re.sub(r'<p[^>]*>', '\n', content, flags=re.IGNORECASE)372 content = re.sub(r'</p>', '\n', content, flags=re.IGNORECASE)373 content = re.sub(r'<[^>]+>', '', content)374 375 content = re.sub(r'[ \t]+', ' ', content)376 content = re.sub(r'\n[ \t]+', '\n', content)377 content = re.sub(r'[ \t]+\n', '\n', content)378 content = re.sub(r'\n{3,}', '\n\n', content)379 380 paragraphs = [p.strip() for p in content.split('\n') if p.strip()]381 return '\n\n'.join(paragraphs)382 383 def fetch_document_info(self, source_url: str) -> Tuple[str, List[Dict]]:384 """获取文档信息"""385 print(f"📖 正在处理: {source_url}")386 387 book_id = self.api.extract_book_id(source_url)388 if not book_id:389 raise ValueError(f"无法从URL中提取书籍ID: {source_url}")390 391 print(f"🔍 提取到书籍ID: {book_id}")392 393 book_detail = self.api.get_book_detail(book_id)394 if not book_detail:395 raise ValueError(f"获取书籍详情失败: {book_id}")396 397 book_name = book_detail.get('book_name') or book_detail.get('title') or f"未知书籍_{book_id}"398 safe_book_name = re.sub(r'[\\/:*?"<>|]', '_', book_name)399 print(f"📚 书籍: {safe_book_name}")400 401 chapters = self.api.get_chapter_list(book_id)402 if not chapters:403 raise ValueError(f"获取章节列表失败: {book_id}")404 405 print(f"📜 找到 {len(chapters)} 个章节")406 407 app_status.update(book_name=safe_book_name, book_id=book_id)408 409 return safe_book_name, chapters410 411 def get_base_dir(self, book_name: str) -> Path:412 """获取基础目录"""413 base_dir = DATA_DIR / book_name414 base_dir.mkdir(exist_ok=True)415 return base_dir416 417 def download_chapters(self, chapters: List[Dict], source_url: str, 418 base_dir: Path, progress_callback=None) -> Tuple[int, int]:419 """下载章节"""420 downloaded_ids = self.status_manager.load_downloaded_ids(source_url)421 422 chapters_to_download = []423 for chapter in chapters:424 if chapter.get('id') and chapter['id'] not in downloaded_ids:425 chapters_to_download.append(chapter)426 427 if not chapters_to_download:428 print("✅ 所有章节均已下载")429 return 0, 0430 431 print(f"📥 需要下载 {len(chapters_to_download)} 个新章节")432 433 success_count = 0434 failed_count = 0435 436 for idx, chapter in enumerate(chapters_to_download, 1):437 if progress_callback:438 progress_callback(idx, len(chapters_to_download), chapter.get('title', ''))439 440 try:441 content_data = self.api.get_chapter_content(chapter['id'])442 443 if content_data and content_data.get('success'):444 title = content_data.get('title', chapter['title'])445 raw_content = content_data.get('content', '')446 processed_content = self.process_content(raw_content)447 448 if processed_content:449 file_path = base_dir / f"{title}.txt"450 full_content = f"{title}\n\n{processed_content}"451 file_path.write_text(full_content, encoding='utf-8')452 453 self.status_manager.save_downloaded_id(source_url, chapter['id'])454 success_count += 1455 456 if idx % 5 == 0 or idx == len(chapters_to_download):457 print(f"✅ {idx}/{len(chapters_to_download)}: {title[:30]}...")458 else:459 failed_count += 1460 print(f"❌ 章节内容为空: {chapter['title']}")461 else:462 failed_count += 1463 print(f"❌ 章节下载失败: {chapter['title']}")464 465 except Exception as e:466 failed_count += 1467 print(f"❌ 下载异常: {e}")468 469 if idx < len(chapters_to_download):470 time.sleep(random.uniform(2, 4))471 472 return success_count, failed_count473 474 def create_archive(self, base_dir: Path) -> str:475 """创建ZIP归档"""476 if not base_dir.exists():477 raise ValueError(f"目录不存在: {base_dir}")478 479 archive_path = base_dir / f"{base_dir.name}.zip"480 481 with zipfile.ZipFile(archive_path, 'w', zipfile.ZIP_DEFLATED) as zipf:482 for txt_file in base_dir.glob("*.txt"):483 if txt_file.is_file() and txt_file.suffix == '.txt':484 zipf.write(txt_file, txt_file.name)485 486 return str(archive_path)487 488 def get_file_count(self, base_dir: Path) -> int:489 """获取文件数量"""490 return len(list(base_dir.glob("*.txt")))491 492# ===================== 状态管理器 =====================493class StatusManager:494 """下载状态管理器"""495 496 def __init__(self, data_dir: Path):497 self.data_dir = data_dir498 self.status_dir = data_dir / ".status"499 self.status_dir.mkdir(exist_ok=True)500 501 def get_status_file(self, source_url: str) -> Path:502 """获取状态文件路径"""503 url_hash = hashlib.md5(source_url.encode()).hexdigest()[:12]504 return self.status_dir / f"status_{url_hash}.json"505 506 def load_downloaded_ids(self, source_url: str) -> Set[str]:507 """加载已下载的章节ID"""508 status_file = self.get_status_file(source_url)509 510 if status_file.exists():511 try:512 with open(status_file, 'r', encoding='utf-8') as f:513 return set(json.load(f))514 except (json.JSONDecodeError, UnicodeDecodeError):515 pass516 517 return set()518 519 def save_downloaded_id(self, source_url: str, chapter_id: str):520 """保存已下载的章节ID"""521 status_file = self.get_status_file(source_url)522 downloaded = self.load_downloaded_ids(source_url)523 downloaded.add(chapter_id)524 525 try:526 with open(status_file, 'w', encoding='utf-8') as f:527 json.dump(list(downloaded), f, ensure_ascii=False, indent=2)528 except Exception as e:529 print(f"保存状态失败: {e}")530 531 def clear_status(self, source_url: str):532 """清除下载状态"""533 status_file = self.get_status_file(source_url)534 if status_file.exists():535 try:536 status_file.unlink()537 except Exception as e:538 print(f"清除状态失败: {e}")539 540# ===================== 全局实例 =====================541app_status = AppStatus()542processor = ContentProcessor()543 544# ===================== 工具函数 =====================545def update_status(message: str):546 """更新状态并打印日志"""547 app_status.update(message=message)548 print(f"📌 {message}")549 550def can_check_update() -> bool:551 """检查是否可以执行更新检查"""552 last_check = app_status.get("last_check_update")553 if not last_check:554 return True555 556 try:557 last_time = datetime.fromisoformat(last_check)558 return datetime.now() - last_time >= timedelta(minutes=10)559 except:560 return True561 562def get_cooldown_remaining() -> int:563 """获取冷却剩余时间(秒)"""564 last_check = app_status.get("last_check_update")565 if not last_check:566 return 0567 568 try:569 last_time = datetime.fromisoformat(last_check)570 next_time = last_time + timedelta(minutes=10)571 remaining = (next_time - datetime.now()).total_seconds()572 return max(0, int(remaining))573 except:574 return 0575 576def process_all(is_manual: bool = False):577 """处理所有章节下载"""578 if app_status.is_processing():579 update_status("⏳ 已有任务在运行,请稍后...")580 return581 582 if is_manual:583 if not can_check_update():584 remaining = get_cooldown_remaining()585 minutes = remaining // 60586 seconds = remaining % 60587 update_status(f"⏰ 检查太频繁了,请等待 {minutes}分{seconds}秒 后再试")588 return589 app_status.update(last_check_update=datetime.now().isoformat())590 591 app_status.set_processing(True)592 593 try:594 update_status("📖 正在获取书籍信息...")595 book_name, chapters = processor.fetch_document_info(SOURCE_URL)596 597 if not chapters:598 update_status("❌ 未找到任何章节")599 return600 601 base_dir = processor.get_base_dir(book_name)602 app_status.update(total_chapters=len(chapters))603 604 existing_count = processor.get_file_count(base_dir)605 app_status.update(downloaded_chapters=existing_count)606 607 update_status(f"📚 书籍: {book_name}")608 update_status(f"📜 总共 {len(chapters)} 个章节,已有 {existing_count} 个")609 610 def progress_callback(current, total, title):611 app_status.update_progress(current, total, f"正在下载: {title}")612 if current % 5 == 0 or current == total:613 update_status(f"⏬ {current}/{total}: {title[:30]}...")614 615 success, failed = processor.download_chapters(616 chapters, SOURCE_URL, base_dir, progress_callback617 )618 619 new_count = processor.get_file_count(base_dir)620 app_status.update(file_count=new_count, downloaded_chapters=new_count)621 622 if success > 0:623 update_status("🗜️ 正在创建ZIP归档...")624 try:625 archive_path = processor.create_archive(base_dir)626 627 if Path(archive_path).exists():628 size_mb = Path(archive_path).stat().st_size / (1024 * 1024)629 app_status.update(archive_ready=True, archive_size_mb=round(size_mb, 2))630 update_status(f"✅ 归档完成 ({size_mb:.2f} MB)")631 except Exception as e:632 update_status(f"❌ 创建归档失败: {e}")633 634 action_type = "手动检查" if is_manual else "定时检查"635 final_msg = f"🎉 {action_type}完成!成功:{success} | 失败:{failed} | 总计:{new_count}"636 update_status(final_msg)637 638 app_status.update_progress(0, 100, "空闲")639 640 except Exception as e:641 error_msg = f"❌ 处理失败: {str(e)}"642 update_status(error_msg)643 print(f"详细错误: {e}")644 traceback.print_exc()645 646 finally:647 app_status.set_processing(False)648 649# ===================== 定时任务 =====================650def scheduler_thread():651 """定时任务线程"""652 time.sleep(10)653 update_status("⏰ 定时任务已激活,将每小时检查一次")654 655 while True:656 try:657 if not app_status.is_processing():658 process_all(is_manual=False)659 update_status(f"⏰ 下次检查将在 {CHECK_INTERVAL//60} 分钟后执行")660 661 time.sleep(CHECK_INTERVAL)662 except Exception as e:663 update_status(f"❌ 定时任务异常: {e},30分钟后重试")664 time.sleep(1800)665 666# ===================== FastAPI应用 =====================667@asynccontextmanager668async def lifespan(app: FastAPI):669 """应用生命周期管理"""670 print("🚀 正在启动应用...")671 672 try:673 thread = threading.Thread(target=scheduler_thread, daemon=True)674 thread.start()675 app_status.update(scheduler_started=True)676 print("✅ 定时任务线程已启动")677 except Exception as e:678 print(f"❌ 启动定时任务失败: {e}")679 680 update_status("✅ 应用启动完成")681 print("✅ 应用启动完成,开始服务")682 yield683 684 print("⏹️ 应用正在关闭...")685 686app = FastAPI(687 title="番茄小说下载器",688 description="支持完整章节下载的小说下载器",689 version="2.0.0",690 docs_url="/docs",691 redoc_url="/redoc",692 lifespan=lifespan693)694 695# ===================== 重要的健康检查路由 =====================696@app.get("/health")697async def health():698 """健康检查"""699 return JSONResponse(700 content={701 "status": "healthy",702 "timestamp": datetime.now().isoformat(),703 "service": "fanqie-novel-downloader",704 "version": "2.0.0",705 "uptime": str(datetime.now() - datetime.fromisoformat(app_status.get("start_time")))706 },707 status_code=200708 )709 710@app.head("/health")711async def health_head():712 """HEAD方法健康检查"""713 return JSONResponse(714 content={"status": "healthy"},715 status_code=200,716 headers={"Content-Type": "application/json"}717 )718 719@app.get("/keepalive")720async def keepalive():721 """保活接口"""722 return JSONResponse(723 content={724 "status": "alive", 725 "timestamp": datetime.now().isoformat(),726 "message": "服务运行正常"727 },728 status_code=200729 )730 731@app.head("/keepalive")732async def keepalive_head():733 """HEAD方法保活接口"""734 return JSONResponse(735 content={"status": "alive"},736 status_code=200,737 headers={"Content-Type": "application/json"}738 )739 740@app.get("/api/health")741async def api_health():742 """API健康检查"""743 return await health()744 745# ===================== Web界面路由 =====================746@app.get("/", response_class=HTMLResponse)747async def root():748 """根路径 - 返回Web界面"""749 return """750 <!DOCTYPE html>751 <html lang="zh-CN">752 <head>753 <meta charset="UTF-8">754 <meta name="viewport" content="width=device-width, initial-scale=1.0">755 <title>番茄小说下载器</title>756 <style>757 * { margin: 0; padding: 0; box-sizing: border-box; }758 body { 759 font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, 'Helvetica Neue', Arial, sans-serif;760 line-height: 1.6; color: #333; background: linear-gradient(135deg, #667eea 0%, #764ba2 100%);761 min-height: 100vh; padding: 20px;762 }763 .container { 764 max-width: 1400px; margin: 0 auto; background: white; border-radius: 15px;765 padding: 30px; box-shadow: 0 20px 40px rgba(0,0,0,0.1); min-height: 600px;766 }767 .header { text-align: center; margin-bottom: 30px; }768 .header h1 { font-size: 2.5rem; margin-bottom: 10px; color: #333; }769 .header p { font-size: 1.2rem; color: #666; }770 .content { display: grid; grid-template-columns: 1fr 2fr; gap: 30px; }771 .control-panel { padding-right: 20px; border-right: 1px solid #eee; }772 .control-panel h2, .info-panel h2 { margin-bottom: 20px; color: #333; font-size: 1.5rem; }773 .button-group { display: flex; flex-direction: column; gap: 10px; margin-bottom: 20px; }774 .btn-primary, .btn-secondary, .btn-success {775 padding: 12px 20px; border: none; border-radius: 8px; font-size: 16px;776 font-weight: 600; cursor: pointer; transition: all 0.3s ease; text-align: center;777 }778 .btn-primary { background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); color: white; }779 .btn-secondary { background: #f8f9fa; color: #495057; border: 1px solid #dee2e6; }780 .btn-success { background: linear-gradient(135deg, #5cb85c 0%, #449d44 100%); color: white; }781 .status-box { background: #f8f9fa; border: 1px solid #dee2e6; border-radius: 8px; padding: 15px; margin-bottom: 20px; }782 .cooldown-info { background: #fff3cd; border: 1px solid #ffeaa7; border-radius: 8px; padding: 10px; margin-top: 10px; }783 .progress-container { margin: 20px 0; }784 .progress-bar { height: 10px; background: #e9ecef; border-radius: 5px; overflow: hidden; }785 .progress-fill { height: 100%; background: linear-gradient(90deg, #667eea 0%, #764ba2 100%); border-radius: 5px; transition: width 0.3s ease; }786 .file-list { max-height: 300px; overflow-y: auto; }787 @media (max-width: 1024px) { .content { grid-template-columns: 1fr; } .control-panel { border-right: none; border-bottom: 1px solid #eee; padding-bottom: 20px; } }788 </style>789 </head>790 <body>791 <div class="container">792 <div class="header">793 <h1>📚 番茄小说下载器</h1>794 <p>支持完整章节下载,解决11章后无法下载的问题</p>795 </div>796 797 <div class="content">798 <div class="control-panel">799 <h2>🎮 控制面板</h2>800 <div class="button-group">801 <button id="btn-check-update" class="btn-primary">🔍 检查更新</button>802 <button id="btn-refresh" class="btn-secondary">🔄 刷新状态</button>803 <button id="btn-download" class="btn-success">📥 下载归档</button>804 </div>805 806 <div class="cooldown-info" id="cooldown-info" style="display: none;">807 <small>⏰ 冷却中:<span id="cooldown-time"></span></small>808 </div>809 810 <div class="progress-container" id="progress-container" style="display: none;">811 <div class="progress-label" id="progress-label">正在处理...</div>812 <div class="progress-bar">813 <div class="progress-fill" id="progress-fill" style="width: 0%"></div>814 </div>815 <div class="progress-text" id="progress-text">0%</div>816 </div>817 818 <h2>📥 归档下载</h2>819 <div id="archive-status" class="status-box">820 <div>暂无归档文件</div>821 </div>822 </div>823 824 <div class="info-panel">825 <h2>📊 系统状态</h2>826 <div class="status-box">827 <strong>状态:</strong><span id="status-message">加载中...</span><br>828 <strong>最后更新:</strong><span id="last-update">-</span><br>829 <strong>书籍:</strong><span id="book-name">未知</span><br>830 <strong>章节总数:</strong><span id="total-chapters">0</span><br>831 <strong>已下载:</strong><span id="downloaded-chapters">0</span><br>832 <strong>文件数量:</strong><span id="file-count">0</span>833 </div>834 835 <h2>📄 已下载章节</h2>836 <div id="file-list" class="file-list status-box">加载中...</div>837 838 <h2>📋 系统信息</h2>839 <div class="status-box">840 <strong>服务状态:</strong><span id="service-status">运行中 ✅</span><br>841 <strong>启动时间:</strong><span id="start-time">-</span><br>842 <strong>当前时间:</strong><span id="current-time">-</span><br>843 <strong>定时任务:</strong><span id="scheduler-status">运行中</span>844 </div>845 </div>846 </div>847 </div>848 849 <script>850 class AppState {851 constructor() {852 this.updateInterval = null;853 this.cooldownInterval = null;854 }855 856 async updateStatus() {857 try {858 const response = await fetch('/api/status');859 const data = await response.json();860 861 document.getElementById('status-message').textContent = data.message || '未知';862 document.getElementById('last-update').textContent = data.last_update || '-';863 document.getElementById('book-name').textContent = data.book_name || '未知';864 document.getElementById('total-chapters').textContent = data.total_chapters || 0;865 document.getElementById('downloaded-chapters').textContent = data.downloaded_chapters || 0;866 document.getElementById('file-count').textContent = data.file_count || 0;867 document.getElementById('start-time').textContent = new Date(data.start_time).toLocaleString();868 document.getElementById('scheduler-status').textContent = data.scheduler_started ? '运行中 ✅' : '停止 ❌';869 870 const progressContainer = document.getElementById('progress-container');871 const progressFill = document.getElementById('progress-fill');872 const progressText = document.getElementById('progress-text');873 const progressLabel = document.getElementById('progress-label');874 875 if (data.processing) {876 progressContainer.style.display = 'block';877 progressFill.style.width = `${data.current_progress || 0}%`;878 progressText.textContent = `${data.current_progress || 0}%`;879 progressLabel.textContent = data.current_task || '正在处理...';880 } else {881 progressContainer.style.display = 'none';882 }883 884 const archiveStatus = document.getElementById('archive-status');885 if (data.archive_ready) {886 archiveStatus.innerHTML = `887 <strong>归档文件:</strong> ${(data.archive_size_mb || 0).toFixed(2)} MB<br>888 <small>点击下载按钮获取完整小说</small>889 `;890 } else {891 archiveStatus.innerHTML = '<div>暂无归档文件,请先检查更新</div>';892 }893 894 const checkBtn = document.getElementById('btn-check-update');895 if (checkBtn) {896 if (data.processing) {897 checkBtn.disabled = true;898 checkBtn.innerHTML = '⏳ 处理中...';899 } else if (data.check_update_cooldown > 0) {900 checkBtn.disabled = true;901 const minutes = Math.floor(data.check_update_cooldown / 60);902 const seconds = data.check_update_cooldown % 60;903 checkBtn.innerHTML = `⏰ 冷却中 (${minutes}:${seconds.toString().padStart(2, '0')})`;904 } else {905 checkBtn.disabled = false;906 checkBtn.innerHTML = '🔍 检查更新';907 }908 }909 910 const downloadBtn = document.getElementById('btn-download');911 if (downloadBtn) {912 downloadBtn.disabled = !data.archive_ready;913 }914 915 } catch (error) {916 console.error('获取状态失败:', error);917 }918 }919 920 async updateFileList() {921 try {922 const response = await fetch('/api/files');923 const data = await response.json();924 925 const fileList = document.getElementById('file-list');926 if (data.status === 'success' && data.files && data.files.length > 0) {927 const fileItems = data.files.map(file => `928 <div style="padding: 5px 0; border-bottom: 1px solid #eee;">929 ${file.name} (${Math.round(file.size/1024)}KB)930 </div>931 `).join('');932 fileList.innerHTML = fileItems;933 } else {934 fileList.innerHTML = '<div>暂无文件</div>';935 }936 } catch (error) {937 console.error('获取文件列表失败:', error);938 }939 }940 941 startPolling() {942 this.updateStatus();943 this.updateFileList();944 945 this.updateInterval = setInterval(() => {946 this.updateStatus();947 }, 3000);948 949 setInterval(() => {950 this.updateFileList();951 }, 30000);952 }953 954 stopPolling() {955 if (this.updateInterval) {956 clearInterval(this.updateInterval);957 }958 }959 }960 961 const appState = new AppState();962 963 async function checkUpdate() {964 const button = document.getElementById('btn-check-update');965 button.disabled = true;966 button.innerHTML = '⏳ 请求中...';967 968 try {969 const response = await fetch('/api/check-update', { method: 'POST' });970 const result = await response.json();971 972 if (response.ok) {973 alert(result.message);974 } else {975 alert('错误: ' + (result.detail || '未知错误'));976 }977 } catch (error) {978 alert('请求失败: ' + error.message);979 }980 }981 982 async function downloadArchive() {983 window.location.href = '/api/download';984 }985 986 function updateCurrentTime() {987 const now = new Date();988 document.getElementById('current-time').textContent = now.toLocaleString();989 }990 991 document.addEventListener('DOMContentLoaded', () => {992 document.getElementById('btn-check-update').addEventListener('click', checkUpdate);993 document.getElementById('btn-refresh').addEventListener('click', () => {994 appState.updateStatus();995 appState.updateFileList();996 });997 document.getElementById('btn-download').addEventListener('click', downloadArchive);998 999 appState.startPolling();1000 updateCurrentTime();1001 setInterval(updateCurrentTime, 1000);1002 });1003 </script>1004 </body>1005 </html>1006 """1007 1008# ===================== API路由 =====================1009@app.get("/api/status")1010async def api_status():1011 """获取系统状态"""1012 status = app_status.get()1013 status["timestamp"] = datetime.now().isoformat()1014 status["check_update_cooldown"] = get_cooldown_remaining()1015 1016 try:1017 book_name = app_status.get("book_name")1018 if book_name and book_name != "未知书籍":1019 base_dir = processor.get_base_dir(book_name)1020 archive_path = base_dir / f"{book_name}.zip"1021 1022 if archive_path.exists():1023 status["archive_ready"] = True1024 status["archive_size_mb"] = round(archive_path.stat().st_size / (1024 * 1024), 2)1025 except Exception as e:1026 print(f"检查归档文件失败: {e}")1027 1028 return status1029 1030@app.post("/api/check-update")1031async def check_update():1032 """手动触发更新检查"""1033 if app_status.is_processing():1034 raise HTTPException(status_code=429, detail="已有任务在运行中,请稍后再试")1035 1036 if not can_check_update():1037 remaining = get_cooldown_remaining()1038 minutes = remaining // 601039 seconds = remaining % 601040 raise HTTPException(1041 status_code=429, 1042 detail=f"检查太频繁了,请等待 {minutes}分{seconds}秒 后再试"1043 )1044 1045 threading.Thread(target=process_all, args=(True,), daemon=True).start()1046 1047 return {1048 "status": "success", 1049 "message": "检查更新任务已启动!",1050 "next_check_allowed_in": 6001051 }1052 1053@app.get("/api/files")1054async def get_files():1055 """获取文件列表"""1056 try:1057 book_name = app_status.get("book_name")1058 if not book_name or book_name == "未知书籍":1059 return {"status": "success", "files": [], "count": 0}1060 1061 base_dir = processor.get_base_dir(book_name)1062 1063 files = []1064 for txt_file in base_dir.glob("*.txt"):1065 files.append({1066 "name": txt_file.name,1067 "size": txt_file.stat().st_size,1068 "modified": datetime.fromtimestamp(txt_file.stat().st_mtime).isoformat()1069 })1070 1071 return {1072 "status": "success", 1073 "files": sorted(files, key=lambda x: x["name"]), 1074 "count": len(files)1075 }1076 except Exception as e:1077 return {"status": "error", "message": str(e)}1078 1079@app.get("/api/download")1080async def download_archive():1081 """下载归档文件"""1082 try:1083 book_name = app_status.get("book_name")1084 if not book_name or book_name == "未知书籍":1085 raise HTTPException(status_code=404, detail="请先执行检查更新以获取书籍信息")1086 1087 base_dir = processor.get_base_dir(book_name)1088 archive_path = base_dir / f"{book_name}.zip"1089 1090 if not archive_path.exists():1091 try:1092 processor.create_archive(base_dir)1093 except Exception as e:1094 raise HTTPException(status_code=404, detail=f"归档文件不存在且创建失败: {str(e)}")1095 1096 if not archive_path.exists():1097 raise HTTPException(status_code=404, detail="归档文件不存在,请先执行检查更新任务")1098 1099 return FileResponse(1100 path=archive_path,1101 filename=f"{book_name}.zip",1102 media_type="application/zip"1103 )1104 except HTTPException:1105 raise1106 except Exception as e:1107 raise HTTPException(status_code=500, detail=str(e))1108 1109# ===================== 静态文件服务 =====================1110static_dir = Path("static")1111static_dir.mkdir(exist_ok=True)1112 1113app.mount("/static", StaticFiles(directory="static"), name="static")1114 1115# ===================== 启动应用(修复版) =====================1116if __name__ == "__main__":1117 port = int(os.getenv("PORT", 7860))1118 1119 # 关键修复:使用 server.run() 而不是 asyncio.run()1120 # 这样可以保持主线程运行,防止应用立即退出1121 uvicorn.run(1122 app,1123 host="0.0.0.0",1124 port=port,1125 access_log=True,1126 log_level="info",1127 loop="asyncio",1128 lifespan="on"1129 )