jasy0/databaseTester
0
1import requests2import json3import os4import time5 6# Global base URLs7base_url = "https://api.mangadex.org"8author_url = f"{base_url}/author"9cover_url = f"{base_url}/cover"10 11# Define endpoints for database operations12login_url = "https://jasy0-mangadb.hf.space/login/"13query_url = "https://jasy0-mangadb.hf.space/query/"14 15# Retrieve the password from the environment variable16password = os.getenv("SQLITE_WEB_PASSWORD")17 18if not password:19 raise ValueError("Environment variable SQLITE_WEB_PASSWORD is not set or empty!")20 21# Step 1: Log in to get the session cookies22session = requests.Session()23try:24 login_response = session.post(25 login_url,26 headers={27 "Content-Type": "application/x-www-form-urlencoded",28 "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9",29 "User-Agent": "Mozilla/5.0"30 },31 data={"password": password}32 )33 if login_response.status_code != 200:34 raise ConnectionError("Login failed!")35except requests.exceptions.RequestException as e:36 raise ConnectionError(f"An error occurred during login: {e}")37 38# Step 2: Create the database table with required columns39create_table_query = """40CREATE TABLE IF NOT EXISTS mangadb (41 id INTEGER PRIMARY KEY AUTOINCREMENT,42 title TEXT NOT NULL,43 author_name TEXT NOT NULL,44 cover TEXT,45 volume_collection TEXT,46 chapters TEXT,47 manga_id TEXT NOT NULL,48 first_chapter TEXT,49 last_chapter TEXT,50 UNIQUE (title, author_name, volume_collection)51);52"""53session.post(query_url, data={"sql": create_table_query})54 55# Function to ensure all variables are strings56def ensure_string(value):57 if isinstance(value, dict):58 return json.dumps(value) # Convert dictionaries to JSON strings59 elif value is None:60 return "N/A"61 else:62 return str(value)63 64# Function to fetch combined titles65def get_combined_titles(manga_id):66 """67 Fetches and combines the main English title and alternative titles from the MangaDex API.68 Always includes the English title (`en`) and at least one alternative title, limiting the total to 5 titles.69 """70 api_url = f"{base_url}/manga/{manga_id}"71 try:72 response = requests.get(api_url)73 response.raise_for_status()74 data = response.json()75 76 # Extract main and alternative titles77 title_data = data.get("data", {}).get("attributes", {})78 main_title = title_data.get("title", {})79 alt_titles = title_data.get("altTitles", [])80 81 # Start with the English title (`en`), if available82 combined_titles = []83 if "en" in main_title:84 combined_titles.append(main_title["en"])85 else:86 combined_titles.extend(main_title.values()) # Fallback to any main title if `en` is missing87 88 # Add alternative titles, ensuring no duplicates89 for alt_title in alt_titles:90 combined_titles.extend(alt_title.values())91 92 # Remove duplicates while preserving order and limit to 5 titles93 combined_titles = list(dict.fromkeys(combined_titles))94 while len(combined_titles) > 5:95 combined_titles.pop()96 97 # Ensure at least one alternative title is present98 if len(combined_titles) == 1 and alt_titles:99 for alt_title in alt_titles:100 for title in alt_title.values():101 if title not in combined_titles:102 combined_titles.append(title)103 break104 105 # Return as a comma-separated string106 return ", ".join(combined_titles) if combined_titles else None107 except requests.exceptions.RequestException as e:108 print(f"Error fetching combined titles: {e}")109 return None110 111# Function to fetch the author's name112def get_author_by_manga_id(manga_id):113 api_url = f"{base_url}/manga/{manga_id}"114 try:115 response = requests.get(api_url)116 response.raise_for_status()117 response_json = response.json()118 author_relationships = response_json["data"]["relationships"]119 authors = [rel for rel in author_relationships if rel["type"] == "author"]120 if authors:121 author_id = authors[0]["id"]122 return get_author_name_by_id(author_id)123 except requests.exceptions.RequestException:124 return None125 126def get_author_name_by_id(author_id):127 api_url = f"{author_url}/{author_id}"128 try:129 response = requests.get(api_url)130 response.raise_for_status()131 response_json = response.json()132 return response_json["data"]["attributes"]["name"]133 except requests.exceptions.RequestException:134 return None135 136# Function to fetch chapters by manga ID137def get_chapters_by_manga_id(manga_id):138 chapters = []139 limit = 100140 offset = 0141 142 while True:143 try:144 response = requests.get(145 f"{base_url}/chapter",146 params={"manga": manga_id, "limit": limit, "offset": offset}147 )148 response.raise_for_status()149 response_json = response.json()150 if "data" in response_json and response_json["data"]:151 chapters.extend(response_json["data"])152 if len(response_json["data"]) < limit:153 break154 offset += limit155 else:156 break157 except requests.exceptions.RequestException as e:158 print(f"An error occurred while fetching chapters: {e}")159 break160 return chapters161 162# Function to extract volume numbers and collections163def extract_volume_numbers(chapters):164 volumes = {}165 collections = set()166 for chapter in chapters:167 volume = chapter["attributes"].get("volume", "N/A")168 chapter_number = chapter["attributes"].get("chapter", None)169 if chapter_number:170 if volume != "N/A":171 volumes.setdefault(volume, []).append(chapter_number)172 else:173 collections.add(chapter_number)174 return volumes, collections175 176# Function to fetch volumes with covers177def get_volumes_with_covers(manga_id):178 covers = {}179 limit = 10180 offset = 0181 url_template = (182 f"{cover_url}?limit={{limit}}&offset={{offset}}&manga[]={manga_id}"183 "&order%5BcreatedAt%5D=asc&order%5BupdatedAt%5D=asc&order%5Bvolume%5D=asc"184 )185 while True:186 try:187 url = url_template.format(limit=limit, offset=offset)188 response = requests.get(url)189 response.raise_for_status()190 data = response.json()191 if "data" in data:192 results = data["data"]193 if not results:194 break195 for result in results:196 volume = result["attributes"].get("volume", "N/A")197 filename = result["attributes"].get("fileName", "N/A")198 covers[volume] = filename199 offset += limit200 else:201 break202 except requests.exceptions.RequestException as e:203 print(f"An error occurred while fetching covers: {e}")204 break205 return covers206 207# Function to insert chapters with decimal adjustments208def insert_chapters_with_decimals(data_to_insert):209 for entry in data_to_insert:210 chapters_set = set(entry["chapters"])211 new_chapters = []212 for chapter in entry["chapters"]:213 new_chapters.append(chapter)214 if '.' not in chapter:215 base_chapter = int(float(chapter))216 for i in range(1, 10):217 decimal_chapter = f"{base_chapter}.{i}"218 if decimal_chapter not in chapters_set:219 new_chapters.append(decimal_chapter)220 chapters_set.add(decimal_chapter)221 entry["chapters"] = sorted(new_chapters, key=lambda x: (int(float(x)), x))222 223# Function to add first and last chapters224def add_first_last_chapters(data):225 for entry in data:226 chapters_without_decimals = [ch for ch in entry["chapters"] if '.' not in ch]227 if chapters_without_decimals:228 first_chapter = min([int(float(ch)) for ch in chapters_without_decimals])229 last_chapter = max([int(float(ch)) for ch in chapters_without_decimals])230 entry["first_chapter"] = first_chapter231 entry["last_chapter"] = last_chapter232 else:233 entry["first_chapter"] = None234 entry["last_chapter"] = None235 236# Function to insert data into the database237def insert_data_to_db(data_to_insert):238 for entry in data_to_insert:239 upsert_query = """240 INSERT INTO mangadb (title, author_name, cover, volume_collection, chapters, manga_id, first_chapter, last_chapter)241 VALUES 242 ('{title}', '{author_name}', '{cover}', '{volume_collection}', '{chapters}', '{manga_id}', '{first_chapter}', '{last_chapter}')243 ON CONFLICT (title, author_name)244 DO UPDATE SET 245 cover = EXCLUDED.cover,246 chapters = EXCLUDED.chapters,247 volume_collection = EXCLUDED.volume_collection,248 first_chapter = EXCLUDED.first_chapter,249 last_chapter = EXCLUDED.last_chapter;250 251 """252 upsert_query = upsert_query.format(253 title=ensure_string(entry["title"]),254 author_name=ensure_string(entry["author"]),255 cover=ensure_string(entry["cover"]),256 volume_collection=ensure_string(entry["volume_collection"]),257 chapters=ensure_string(entry["chapters"]),258 manga_id=ensure_string(entry["manga_id"]),259 first_chapter=ensure_string(entry.get("first_chapter")),260 last_chapter=ensure_string(entry.get("last_chapter"))261 )262 try:263 insert_response = session.post(query_url, data={"sql": upsert_query})264 if insert_response.status_code == 200:265 print(f"Data inserted/updated successfully for volume/collection: {entry['volume_collection']}!")266 else:267 print(f"Failed to insert/update data for volume/collection: {entry['volume_collection']}!")268 except requests.exceptions.RequestException as e:269 print(f"An error occurred during data insertion/update: {e}")270 271def filter_volumes(volumes, collections, cover_volumes):272 """273 Filters and processes the volumes and collections based on specific rules274 and criteria, ensuring cleaned and structured data for further processing.275 When chapters are moved from collections to volumes, the collections are updated accordingly.276 Adds .1, .2, ..., .9 to both the volume list and collection list.277 """278 merged_volumes = {}279 280 # Step 1: Merge volumes by combining chapters and removing duplicates281 for volume, chapters in volumes.items():282 if volume in merged_volumes:283 merged_volumes[volume].extend(chapters)284 else:285 merged_volumes[volume] = chapters286 merged_volumes[volume] = list(set(merged_volumes[volume]))287 288 # Handle collections if None exists in merged_volumes289 if None in merged_volumes:290 collections.update(merged_volumes[None])291 del merged_volumes[None]292 293 # Step 2: Move special target chapters into volume 0294 target_chapters = {'0', '0.1', '0.2', '0.3', '0.4', '0.5', '0.6', '0.7', '0.8', '0.9'}295 default_volume_0 = merged_volumes.get('0', [])296 if any(ch in collections for ch in target_chapters):297 default_volume_0.extend([ch for ch in collections if ch in target_chapters])298 default_volume_0 = list(set(default_volume_0))299 collections.difference_update(target_chapters)300 301 # Step 3: Remove volumes that do not have covers in cover_volumes302 merged_volumes = {vol: chap for vol, chap in merged_volumes.items() if vol in cover_volumes}303 304 # Step 4: Add .1, .2, ..., .9 to chapters in volumes305 for volume, chapters in merged_volumes.items():306 updated_chapters = []307 for chapter in chapters:308 updated_chapters.append(chapter)309 if '.' not in chapter: # Only for non-decimal chapters310 base_chapter = int(float(chapter))311 for i in range(1, 10):312 updated_chapters.append(f"{base_chapter}.{i}")313 merged_volumes[volume] = sorted(set(updated_chapters), key=lambda x: float(x))314 315 # Step 5: Deduplicate collections by removing chapters already assigned to volumes316 all_volume_chapters = set(ch for chapters in merged_volumes.values() for ch in chapters)317 collections = list(set(collections) - all_volume_chapters)318 319 # Step 6: Add .1, .2, ..., .9 to collections320 updated_collections = []321 for chapter in collections:322 updated_collections.append(chapter)323 if '.' not in chapter: # Only for non-decimal chapters324 base_chapter = int(float(chapter))325 for i in range(1, 10):326 updated_collections.append(f"{base_chapter}.{i}")327 collections = sorted(set(updated_collections), key=lambda x: float(x))328 329 # Step 7: Dynamically move chapters from collections to volumes330 for chapter in collections[:]: # Copy the list to avoid modification during iteration331 for volume, volume_chapters in merged_volumes.items():332 if chapter in volume_chapters:333 # Move chapter to volume and remove from collection334 collections.remove(chapter)335 merged_volumes[volume].append(chapter)336 337 # Step 8: Create collections from leftover chapters338 collection_lists = []339 limit = 80340 for i in range(0, len(collections), limit):341 collection_lists.append(collections[i:i + limit])342 343 collections = []344 for idx, collection in enumerate(collection_lists):345 collection_name = f'Collection {idx + 1}'346 collections.append({collection_name: collection})347 348 # Remove empty collections349 collections = [collection for collection in collections if any(collection.values())]350 351 # Step 9: Sort chapters in merged_volumes and collections in numerical order352 for volume in merged_volumes:353 merged_volumes[volume] = sorted(merged_volumes[volume], key=lambda x: (int(float(x)), x))354 355 for collection in collections:356 for collection_name in collection:357 collection[collection_name] = sorted(collection[collection_name], key=lambda x: (int(float(x)), x))358 359 return merged_volumes, collections360 361 362def process_manga_folder(manga_id):363 """364 Processes a manga by its ID to retrieve chapters, volumes, and covers,365 and organizes the data for storage in a database.366 Dynamically updates collections when chapters are moved to new volumes.367 """368 print(f"Processing Manga with ID: {manga_id}")369 370 # Step 1: Fetch manga data371 author_name = get_author_by_manga_id(manga_id)372 combined_titles = ensure_string(get_combined_titles(manga_id)) # Ensure title is a string373 chapters = get_chapters_by_manga_id(manga_id)374 375 # Step 2: Validate fetched data and handle missing chapters376 if not isinstance(combined_titles, str):377 print(f"Error: Combined Titles is not a string! Type: {type(combined_titles)}")378 if not isinstance(author_name, str):379 print(f"Error: Author Name is not a string! Type: {type(author_name)}")380 if not chapters:381 print(f"No chapters found for the given manga ID: {manga_id}")382 return383 384 # Step 3: Extract volumes and collections385 volumes, collections = extract_volume_numbers(chapters)386 387 # Handle chapter 0 (move to volume 1, if it exists)388 if "0" in volumes:389 if "1" in volumes:390 volumes["1"].extend(volumes["0"])391 else:392 volumes["1"] = volumes["0"]393 del volumes["0"]394 395 # Step 4: Fetch cover volumes396 cover_volumes = get_volumes_with_covers(manga_id)397 398 # Step 5: Filter volumes and collections399 merged_volumes, collections = filter_volumes(volumes, collections, cover_volumes)400 401 # Step 6: Prepare data for insertion into the database402 data_to_insert = []403 404 for volume, chapter_numbers in merged_volumes.items():405 data_to_insert.append({406 "title": combined_titles,407 "volume_collection": f"Volume {volume}",408 "cover": None, # Default cover, updated later409 "chapters": chapter_numbers,410 "manga_id": manga_id,411 "author_name": author_name412 })413 414 # Fetch cover images and assign them to respective volumes415 covers = get_volumes_with_covers(manga_id)416 last_volume_cover = None417 valid_volumes = [v for v in merged_volumes.keys() if v is not None]418 last_volume = str(max(valid_volumes, key=int)) if valid_volumes else None419 420 for volume, filename in sorted(covers.items()):421 if str(volume) == last_volume:422 last_volume_cover = filename423 for entry in data_to_insert:424 if entry["volume_collection"] == f"Volume {volume}":425 entry["cover"] = filename426 427 # Recheck collections to ensure they're updated after volumes are finalized428 final_collections = []429 for idx, collection in enumerate(collections):430 collection_name = f"Collection {idx + 1}"431 final_collections.append({432 "title": combined_titles,433 "volume_collection": collection_name,434 "cover": last_volume_cover,435 "chapters": list(collection.values())[0] if collection.values() else [],436 "manga_id": manga_id,437 "author_name": author_name438 })439 440 # Add updated collections to data_to_insert441 data_to_insert.extend(final_collections)442 443 # Step 7: Add metadata for first and last chapters444 add_first_last_chapters(data_to_insert)445 446 # Step 8: Debug and validate data before inserting447 for entry in data_to_insert:448 if not isinstance(entry["title"], str):449 print(f"Error: Title is not a string! Type: {type(entry['title'])}")450 if not isinstance(entry["author_name"], str):451 print(f"Error: Author Name is not a string! Type: {type(entry['author_name'])}")452 if isinstance(entry["cover"], dict):453 entry["cover"] = json.dumps(entry["cover"]) # Serialize dictionary to string454 print("Cover data serialized to string.")455 456 # Print debug information for each entry457 print(f"Title: {entry['title']}")458 print(f"Author Name: {entry['author_name']}")459 print(f"Cover: {entry['cover']}")460 print(f"Chapters: {entry['chapters']}")461 print(f"First Chapter: {entry.get('first_chapter', 'N/A')}")462 print(f"Last Chapter: {entry.get('last_chapter', 'N/A')}")463 464 # Step 9: Insert the prepared data into the database with UPSERT logic465 for entry in data_to_insert:466 upsert_query = """467 INSERT INTO mangadb (title, author_name, cover, volume_collection, chapters, manga_id, first_chapter, last_chapter)468 VALUES 469 ('{title}', '{author_name}', '{cover}', '{volume_collection}', '{chapters}', '{manga_id}', '{first_chapter}', '{last_chapter}')470 ON CONFLICT (title, author_name, volume_collection)471 DO UPDATE SET 472 cover = EXCLUDED.cover,473 chapters = EXCLUDED.chapters,474 first_chapter = EXCLUDED.first_chapter,475 last_chapter = EXCLUDED.last_chapter;476 """477 upsert_query = upsert_query.format(478 title=entry["title"] or "N/A",479 author_name=entry["author_name"] or "Unknown",480 cover=entry["cover"] or "N/A",481 volume_collection=entry["volume_collection"] or "N/A",482 chapters=json.dumps(entry["chapters"]) if entry["chapters"] else "[]",483 manga_id=entry["manga_id"] or "N/A",484 first_chapter=entry.get("first_chapter", "N/A"),485 last_chapter=entry.get("last_chapter", "N/A")486 )487 488 # Execute the query489 try:490 insert_response = session.post(query_url, data={"sql": upsert_query})491 if insert_response.status_code == 200:492 print(f"Data inserted/updated successfully for volume/collection: {entry['volume_collection']}!")493 else:494 print(f"Failed to insert/update data for volume/collection: {entry['volume_collection']}!")495 print("Status Code:", insert_response.status_code)496 print("Response:", insert_response.text)497 except requests.exceptions.RequestException as e:498 print(f"An error occurred during data insertion/update: {e}")499 500 print("Manga processing completed successfully!")501 502manga_id = "9d9b04ad-9a83-49f4-8ae4-a9a3780fe9c0"503process_manga_folder(manga_id)