Team Ai
Apppublic

jasy0/databaseTester

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
script.py503 linesDownload Raw Back to root
1import requests2import json3import os4import time5 6# Global base URLs7base_url = "https://api.mangadex.org"8author_url = f"{base_url}/author"9cover_url = f"{base_url}/cover"10 11# Define endpoints for database operations12login_url = "https://jasy0-mangadb.hf.space/login/"13query_url = "https://jasy0-mangadb.hf.space/query/"14 15# Retrieve the password from the environment variable16password = os.getenv("SQLITE_WEB_PASSWORD")17 18if not password:19    raise ValueError("Environment variable SQLITE_WEB_PASSWORD is not set or empty!")20 21# Step 1: Log in to get the session cookies22session = requests.Session()23try:24    login_response = session.post(25        login_url,26        headers={27            "Content-Type": "application/x-www-form-urlencoded",28            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9",29            "User-Agent": "Mozilla/5.0"30        },31        data={"password": password}32    )33    if login_response.status_code != 200:34        raise ConnectionError("Login failed!")35except requests.exceptions.RequestException as e:36    raise ConnectionError(f"An error occurred during login: {e}")37 38# Step 2: Create the database table with required columns39create_table_query = """40CREATE TABLE IF NOT EXISTS mangadb (41    id INTEGER PRIMARY KEY AUTOINCREMENT,42    title TEXT NOT NULL,43    author_name TEXT NOT NULL,44    cover TEXT,45    volume_collection TEXT,46    chapters TEXT,47    manga_id TEXT NOT NULL,48    first_chapter TEXT,49    last_chapter TEXT,50    UNIQUE (title, author_name, volume_collection)51);52"""53session.post(query_url, data={"sql": create_table_query})54 55# Function to ensure all variables are strings56def ensure_string(value):57    if isinstance(value, dict):58        return json.dumps(value)  # Convert dictionaries to JSON strings59    elif value is None:60        return "N/A"61    else:62        return str(value)63 64# Function to fetch combined titles65def get_combined_titles(manga_id):66    """67    Fetches and combines the main English title and alternative titles from the MangaDex API.68    Always includes the English title (`en`) and at least one alternative title, limiting the total to 5 titles.69    """70    api_url = f"{base_url}/manga/{manga_id}"71    try:72        response = requests.get(api_url)73        response.raise_for_status()74        data = response.json()75        76        # Extract main and alternative titles77        title_data = data.get("data", {}).get("attributes", {})78        main_title = title_data.get("title", {})79        alt_titles = title_data.get("altTitles", [])80        81        # Start with the English title (`en`), if available82        combined_titles = []83        if "en" in main_title:84            combined_titles.append(main_title["en"])85        else:86            combined_titles.extend(main_title.values())  # Fallback to any main title if `en` is missing87        88        # Add alternative titles, ensuring no duplicates89        for alt_title in alt_titles:90            combined_titles.extend(alt_title.values())91        92        # Remove duplicates while preserving order and limit to 5 titles93        combined_titles = list(dict.fromkeys(combined_titles))94        while len(combined_titles) > 5:95            combined_titles.pop()96 97        # Ensure at least one alternative title is present98        if len(combined_titles) == 1 and alt_titles:99            for alt_title in alt_titles:100                for title in alt_title.values():101                    if title not in combined_titles:102                        combined_titles.append(title)103                        break104 105        # Return as a comma-separated string106        return ", ".join(combined_titles) if combined_titles else None107    except requests.exceptions.RequestException as e:108        print(f"Error fetching combined titles: {e}")109        return None110 111# Function to fetch the author's name112def get_author_by_manga_id(manga_id):113    api_url = f"{base_url}/manga/{manga_id}"114    try:115        response = requests.get(api_url)116        response.raise_for_status()117        response_json = response.json()118        author_relationships = response_json["data"]["relationships"]119        authors = [rel for rel in author_relationships if rel["type"] == "author"]120        if authors:121            author_id = authors[0]["id"]122            return get_author_name_by_id(author_id)123    except requests.exceptions.RequestException:124        return None125 126def get_author_name_by_id(author_id):127    api_url = f"{author_url}/{author_id}"128    try:129        response = requests.get(api_url)130        response.raise_for_status()131        response_json = response.json()132        return response_json["data"]["attributes"]["name"]133    except requests.exceptions.RequestException:134        return None135 136# Function to fetch chapters by manga ID137def get_chapters_by_manga_id(manga_id):138    chapters = []139    limit = 100140    offset = 0141 142    while True:143        try:144            response = requests.get(145                f"{base_url}/chapter",146                params={"manga": manga_id, "limit": limit, "offset": offset}147            )148            response.raise_for_status()149            response_json = response.json()150            if "data" in response_json and response_json["data"]:151                chapters.extend(response_json["data"])152                if len(response_json["data"]) < limit:153                    break154                offset += limit155            else:156                break157        except requests.exceptions.RequestException as e:158            print(f"An error occurred while fetching chapters: {e}")159            break160    return chapters161 162# Function to extract volume numbers and collections163def extract_volume_numbers(chapters):164    volumes = {}165    collections = set()166    for chapter in chapters:167        volume = chapter["attributes"].get("volume", "N/A")168        chapter_number = chapter["attributes"].get("chapter", None)169        if chapter_number:170            if volume != "N/A":171                volumes.setdefault(volume, []).append(chapter_number)172            else:173                collections.add(chapter_number)174    return volumes, collections175 176# Function to fetch volumes with covers177def get_volumes_with_covers(manga_id):178    covers = {}179    limit = 10180    offset = 0181    url_template = (182        f"{cover_url}?limit={{limit}}&offset={{offset}}&manga[]={manga_id}"183        "&order%5BcreatedAt%5D=asc&order%5BupdatedAt%5D=asc&order%5Bvolume%5D=asc"184    )185    while True:186        try:187            url = url_template.format(limit=limit, offset=offset)188            response = requests.get(url)189            response.raise_for_status()190            data = response.json()191            if "data" in data:192                results = data["data"]193                if not results:194                    break195                for result in results:196                    volume = result["attributes"].get("volume", "N/A")197                    filename = result["attributes"].get("fileName", "N/A")198                    covers[volume] = filename199                offset += limit200            else:201                break202        except requests.exceptions.RequestException as e:203            print(f"An error occurred while fetching covers: {e}")204            break205    return covers206 207# Function to insert chapters with decimal adjustments208def insert_chapters_with_decimals(data_to_insert):209    for entry in data_to_insert:210        chapters_set = set(entry["chapters"])211        new_chapters = []212        for chapter in entry["chapters"]:213            new_chapters.append(chapter)214            if '.' not in chapter:215                base_chapter = int(float(chapter))216                for i in range(1, 10):217                    decimal_chapter = f"{base_chapter}.{i}"218                    if decimal_chapter not in chapters_set:219                        new_chapters.append(decimal_chapter)220                        chapters_set.add(decimal_chapter)221        entry["chapters"] = sorted(new_chapters, key=lambda x: (int(float(x)), x))222 223# Function to add first and last chapters224def add_first_last_chapters(data):225    for entry in data:226        chapters_without_decimals = [ch for ch in entry["chapters"] if '.' not in ch]227        if chapters_without_decimals:228            first_chapter = min([int(float(ch)) for ch in chapters_without_decimals])229            last_chapter = max([int(float(ch)) for ch in chapters_without_decimals])230            entry["first_chapter"] = first_chapter231            entry["last_chapter"] = last_chapter232        else:233            entry["first_chapter"] = None234            entry["last_chapter"] = None235 236# Function to insert data into the database237def insert_data_to_db(data_to_insert):238    for entry in data_to_insert:239        upsert_query = """240        INSERT INTO mangadb (title, author_name, cover, volume_collection, chapters, manga_id, first_chapter, last_chapter)241        VALUES 242            ('{title}', '{author_name}', '{cover}', '{volume_collection}', '{chapters}', '{manga_id}', '{first_chapter}', '{last_chapter}')243        ON CONFLICT (title, author_name)244        DO UPDATE SET 245            cover = EXCLUDED.cover,246            chapters = EXCLUDED.chapters,247            volume_collection = EXCLUDED.volume_collection,248            first_chapter = EXCLUDED.first_chapter,249            last_chapter = EXCLUDED.last_chapter;250            251        """252        upsert_query = upsert_query.format(253            title=ensure_string(entry["title"]),254            author_name=ensure_string(entry["author"]),255            cover=ensure_string(entry["cover"]),256            volume_collection=ensure_string(entry["volume_collection"]),257            chapters=ensure_string(entry["chapters"]),258            manga_id=ensure_string(entry["manga_id"]),259            first_chapter=ensure_string(entry.get("first_chapter")),260            last_chapter=ensure_string(entry.get("last_chapter"))261        )262        try:263            insert_response = session.post(query_url, data={"sql": upsert_query})264            if insert_response.status_code == 200:265                print(f"Data inserted/updated successfully for volume/collection: {entry['volume_collection']}!")266            else:267                print(f"Failed to insert/update data for volume/collection: {entry['volume_collection']}!")268        except requests.exceptions.RequestException as e:269            print(f"An error occurred during data insertion/update: {e}")270 271def filter_volumes(volumes, collections, cover_volumes):272    """273    Filters and processes the volumes and collections based on specific rules274    and criteria, ensuring cleaned and structured data for further processing.275    When chapters are moved from collections to volumes, the collections are updated accordingly.276    Adds .1, .2, ..., .9 to both the volume list and collection list.277    """278    merged_volumes = {}279 280    # Step 1: Merge volumes by combining chapters and removing duplicates281    for volume, chapters in volumes.items():282        if volume in merged_volumes:283            merged_volumes[volume].extend(chapters)284        else:285            merged_volumes[volume] = chapters286        merged_volumes[volume] = list(set(merged_volumes[volume]))287 288    # Handle collections if None exists in merged_volumes289    if None in merged_volumes:290        collections.update(merged_volumes[None])291        del merged_volumes[None]292 293    # Step 2: Move special target chapters into volume 0294    target_chapters = {'0', '0.1', '0.2', '0.3', '0.4', '0.5', '0.6', '0.7', '0.8', '0.9'}295    default_volume_0 = merged_volumes.get('0', [])296    if any(ch in collections for ch in target_chapters):297        default_volume_0.extend([ch for ch in collections if ch in target_chapters])298        default_volume_0 = list(set(default_volume_0))299        collections.difference_update(target_chapters)300 301    # Step 3: Remove volumes that do not have covers in cover_volumes302    merged_volumes = {vol: chap for vol, chap in merged_volumes.items() if vol in cover_volumes}303 304    # Step 4: Add .1, .2, ..., .9 to chapters in volumes305    for volume, chapters in merged_volumes.items():306        updated_chapters = []307        for chapter in chapters:308            updated_chapters.append(chapter)309            if '.' not in chapter:  # Only for non-decimal chapters310                base_chapter = int(float(chapter))311                for i in range(1, 10):312                    updated_chapters.append(f"{base_chapter}.{i}")313        merged_volumes[volume] = sorted(set(updated_chapters), key=lambda x: float(x))314 315    # Step 5: Deduplicate collections by removing chapters already assigned to volumes316    all_volume_chapters = set(ch for chapters in merged_volumes.values() for ch in chapters)317    collections = list(set(collections) - all_volume_chapters)318 319    # Step 6: Add .1, .2, ..., .9 to collections320    updated_collections = []321    for chapter in collections:322        updated_collections.append(chapter)323        if '.' not in chapter:  # Only for non-decimal chapters324            base_chapter = int(float(chapter))325            for i in range(1, 10):326                updated_collections.append(f"{base_chapter}.{i}")327    collections = sorted(set(updated_collections), key=lambda x: float(x))328 329    # Step 7: Dynamically move chapters from collections to volumes330    for chapter in collections[:]:  # Copy the list to avoid modification during iteration331        for volume, volume_chapters in merged_volumes.items():332            if chapter in volume_chapters:333                # Move chapter to volume and remove from collection334                collections.remove(chapter)335                merged_volumes[volume].append(chapter)336 337    # Step 8: Create collections from leftover chapters338    collection_lists = []339    limit = 80340    for i in range(0, len(collections), limit):341        collection_lists.append(collections[i:i + limit])342 343    collections = []344    for idx, collection in enumerate(collection_lists):345        collection_name = f'Collection {idx + 1}'346        collections.append({collection_name: collection})347 348    # Remove empty collections349    collections = [collection for collection in collections if any(collection.values())]350 351    # Step 9: Sort chapters in merged_volumes and collections in numerical order352    for volume in merged_volumes:353        merged_volumes[volume] = sorted(merged_volumes[volume], key=lambda x: (int(float(x)), x))354 355    for collection in collections:356        for collection_name in collection:357            collection[collection_name] = sorted(collection[collection_name], key=lambda x: (int(float(x)), x))358 359    return merged_volumes, collections360 361 362def process_manga_folder(manga_id):363    """364    Processes a manga by its ID to retrieve chapters, volumes, and covers,365    and organizes the data for storage in a database.366    Dynamically updates collections when chapters are moved to new volumes.367    """368    print(f"Processing Manga with ID: {manga_id}")369 370    # Step 1: Fetch manga data371    author_name = get_author_by_manga_id(manga_id)372    combined_titles = ensure_string(get_combined_titles(manga_id))  # Ensure title is a string373    chapters = get_chapters_by_manga_id(manga_id)374 375    # Step 2: Validate fetched data and handle missing chapters376    if not isinstance(combined_titles, str):377        print(f"Error: Combined Titles is not a string! Type: {type(combined_titles)}")378    if not isinstance(author_name, str):379        print(f"Error: Author Name is not a string! Type: {type(author_name)}")380    if not chapters:381        print(f"No chapters found for the given manga ID: {manga_id}")382        return383 384    # Step 3: Extract volumes and collections385    volumes, collections = extract_volume_numbers(chapters)386 387    # Handle chapter 0 (move to volume 1, if it exists)388    if "0" in volumes:389        if "1" in volumes:390            volumes["1"].extend(volumes["0"])391        else:392            volumes["1"] = volumes["0"]393        del volumes["0"]394 395    # Step 4: Fetch cover volumes396    cover_volumes = get_volumes_with_covers(manga_id)397 398    # Step 5: Filter volumes and collections399    merged_volumes, collections = filter_volumes(volumes, collections, cover_volumes)400 401    # Step 6: Prepare data for insertion into the database402    data_to_insert = []403 404    for volume, chapter_numbers in merged_volumes.items():405        data_to_insert.append({406            "title": combined_titles,407            "volume_collection": f"Volume {volume}",408            "cover": None,  # Default cover, updated later409            "chapters": chapter_numbers,410            "manga_id": manga_id,411            "author_name": author_name412        })413 414    # Fetch cover images and assign them to respective volumes415    covers = get_volumes_with_covers(manga_id)416    last_volume_cover = None417    valid_volumes = [v for v in merged_volumes.keys() if v is not None]418    last_volume = str(max(valid_volumes, key=int)) if valid_volumes else None419 420    for volume, filename in sorted(covers.items()):421        if str(volume) == last_volume:422            last_volume_cover = filename423        for entry in data_to_insert:424            if entry["volume_collection"] == f"Volume {volume}":425                entry["cover"] = filename426 427    # Recheck collections to ensure they're updated after volumes are finalized428    final_collections = []429    for idx, collection in enumerate(collections):430        collection_name = f"Collection {idx + 1}"431        final_collections.append({432            "title": combined_titles,433            "volume_collection": collection_name,434            "cover": last_volume_cover,435            "chapters": list(collection.values())[0] if collection.values() else [],436            "manga_id": manga_id,437            "author_name": author_name438        })439 440    # Add updated collections to data_to_insert441    data_to_insert.extend(final_collections)442 443    # Step 7: Add metadata for first and last chapters444    add_first_last_chapters(data_to_insert)445 446    # Step 8: Debug and validate data before inserting447    for entry in data_to_insert:448        if not isinstance(entry["title"], str):449            print(f"Error: Title is not a string! Type: {type(entry['title'])}")450        if not isinstance(entry["author_name"], str):451            print(f"Error: Author Name is not a string! Type: {type(entry['author_name'])}")452        if isinstance(entry["cover"], dict):453            entry["cover"] = json.dumps(entry["cover"])  # Serialize dictionary to string454            print("Cover data serialized to string.")455 456        # Print debug information for each entry457        print(f"Title: {entry['title']}")458        print(f"Author Name: {entry['author_name']}")459        print(f"Cover: {entry['cover']}")460        print(f"Chapters: {entry['chapters']}")461        print(f"First Chapter: {entry.get('first_chapter', 'N/A')}")462        print(f"Last Chapter: {entry.get('last_chapter', 'N/A')}")463 464    # Step 9: Insert the prepared data into the database with UPSERT logic465    for entry in data_to_insert:466        upsert_query = """467        INSERT INTO mangadb (title, author_name, cover, volume_collection, chapters, manga_id, first_chapter, last_chapter)468        VALUES 469            ('{title}', '{author_name}', '{cover}', '{volume_collection}', '{chapters}', '{manga_id}', '{first_chapter}', '{last_chapter}')470        ON CONFLICT (title, author_name, volume_collection)471        DO UPDATE SET 472            cover = EXCLUDED.cover,473            chapters = EXCLUDED.chapters,474            first_chapter = EXCLUDED.first_chapter,475            last_chapter = EXCLUDED.last_chapter;476        """477        upsert_query = upsert_query.format(478            title=entry["title"] or "N/A",479            author_name=entry["author_name"] or "Unknown",480            cover=entry["cover"] or "N/A",481            volume_collection=entry["volume_collection"] or "N/A",482            chapters=json.dumps(entry["chapters"]) if entry["chapters"] else "[]",483            manga_id=entry["manga_id"] or "N/A",484            first_chapter=entry.get("first_chapter", "N/A"),485            last_chapter=entry.get("last_chapter", "N/A")486        )487 488        # Execute the query489        try:490            insert_response = session.post(query_url, data={"sql": upsert_query})491            if insert_response.status_code == 200:492                print(f"Data inserted/updated successfully for volume/collection: {entry['volume_collection']}!")493            else:494                print(f"Failed to insert/update data for volume/collection: {entry['volume_collection']}!")495                print("Status Code:", insert_response.status_code)496                print("Response:", insert_response.text)497        except requests.exceptions.RequestException as e:498            print(f"An error occurred during data insertion/update: {e}")499 500    print("Manga processing completed successfully!")501 502manga_id = "9d9b04ad-9a83-49f4-8ae4-a9a3780fe9c0"503process_manga_folder(manga_id)