Team Ai
Apppublic

Multimedika/Bot_Development

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
aws_loader.py108 linesDownload Raw Back to service
1import os2import boto33import tempfile4import fitz5from io import BytesIO6 7from fastapi.responses import JSONResponse8import logging9 10 11class Loader:12    def __init__(self):13        # Create S3 and Transcribe clients with credentials14        self.bucket_name = "multimedika"15        self.s3_client = boto3.client(16            "s3",17            aws_access_key_id=os.getenv("AWS_ACCESS_KEY_ID"),18            aws_secret_access_key=os.getenv("AWS_SECRET_ACCESS_KEY"),19            region_name="us-west-2",20        )21        22    def upload_to_s3(self, file_stream: BytesIO, title, folder_name="summarizer"):23        try:24            # If folder_name is provided, prepend it to the title25            if folder_name:26                object_name = f"{folder_name}/{title}"27 28            # Open the PDF with PyMuPDF (fitz)29            pdf_document = fitz.open(stream=file_stream.getvalue(), filetype="pdf")30            print("Jumlah halaman : ", pdf_document.page_count)31            32                        # Create a stream for the full PDF33            full_pdf_stream = BytesIO()34            pdf_document.save(full_pdf_stream)  # Save the full document to a stream35            full_pdf_stream.seek(0)  # Reset the stream position to the start36 37            # Define the S3 object name for the full PDF38            full_pdf_object_name = f"{folder_name}/full_book/{title}.pdf"39 40            # Upload the full PDF to S341            self.s3_client.upload_fileobj(full_pdf_stream, self.bucket_name, full_pdf_object_name)42            print(f"Full PDF '{title}.pdf' successfully uploaded as '{full_pdf_object_name}' to bucket '{self.bucket_name}'.")43            44            # Loop through each page of the PDF45            for page_num in range(pdf_document.page_count):46                try:47                    # Convert the page to bytes (as a separate PDF)48                    page_stream = BytesIO()49                    single_page_pdf = fitz.open()  # Create a new PDF50                    single_page_pdf.insert_pdf(pdf_document, from_page=page_num, to_page=page_num)51                    single_page_pdf.save(page_stream)52                    single_page_pdf.close()53 54                    # Reset the stream position to the start55                    page_stream.seek(0)56 57                    # Define the object name for each page (e.g., 'summarizer/object_name/page_1.pdf')58                    page_object_name = f"{object_name}/{page_num + 1}.pdf"59 60                    # Upload each page to S361                    self.s3_client.upload_fileobj(page_stream, self.bucket_name, page_object_name)62 63                    print(f"Page {page_num + 1} of '{object_name}' successfully uploaded as '{page_object_name}' to bucket '{self.bucket_name}'.")64 65                except Exception as page_error:66                    # Log the error but continue with the next page67                    logging.error(f"Error uploading page {page_num + 1}: {page_error}")68                    continue69 70        except Exception as e:71            return JSONResponse(status_code=500, content=f"Error uploading to AWS: {e}")72    73    def change_name_of_book(self, current_object_name, new_object_name, folder_name="summarizer"):74        try:75            if folder_name:76                current_object_name = f"{folder_name}/{current_object_name}"77                new_object_name = f"{folder_name}/{new_object_name}"78 79            # Copy the current object to a new object with the new name80            copy_source = {'Bucket': self.bucket_name, 'Key': current_object_name}81            self.s3_client.copy(copy_source, self.bucket_name, new_object_name)82 83            # Delete the old object84            self.s3_client.delete_object(Bucket=self.bucket_name, Key=current_object_name)85 86            print(f"Renamed '{current_object_name}' to '{new_object_name}'.")87 88        except Exception as e:89            return JSONResponse(status_code=500, content=f"Error renaming book: {e}")90    91    def change_name_of_image(self, current_object_name, new_object_name, folder_name="summarizer"):92        try:93            if folder_name:94                current_object_name = f"{folder_name}/{current_object_name}"95                new_object_name = f"{folder_name}/{new_object_name}"96 97            # Copy the current object to a new object with the new name98            copy_source = {'Bucket': self.bucket_name, 'Key': current_object_name}99            self.s3_client.copy(copy_source, self.bucket_name, new_object_name)100 101            # Delete the old object102            self.s3_client.delete_object(Bucket=self.bucket_name, Key=current_object_name)103 104            print(f"Renamed image '{current_object_name}' to '{new_object_name}' in bucket '{self.bucket_name}'.")105 106        except Exception as e:107            return JSONResponse(status_code=500, content=f"Error renaming image: {e}")108