Team Ai
Apppublic

hprasath/image-processing

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
app.py363 linesDownload Raw Back to root
1import moviepy.editor as mp2from flask import Flask, request, jsonify3from flask_cors import CORS4import requests5from io import BytesIO6import speech_recognition as sr7import io8import fitz  # PyMuPDF for working with PDFs9import numpy as np10import cv211from flask_caching import Cache12 13from utils.audioEmbedding.index import extract_audio_embeddings14from utils.videoEmbedding.index import get_video_embedding15from utils.imageToText.index import extract_text16from utils.sentanceEmbedding.index import get_text_vector , get_text_discription_vector17from utils.imageEmbedding.index import get_image_embedding18from utils.similarityScore import get_all_similarities19from utils.objectDetection.index import detect_objects20 21 22 23app = Flask(__name__)24cache = Cache(app, config={'CACHE_TYPE': 'simple'})  # You can choose a caching type based on your requirements25CORS(app)26import moviepy.editor as mp27import tempfile28 29def get_face_locations(binary_data):30    # Convert binary image data to numpy array31    print(1)32    nparr = np.frombuffer(binary_data, np.uint8)33    image = cv2.imdecode(nparr, cv2.IMREAD_COLOR)34    35    # Load the pre-trained face detection model36    face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_frontalface_default.xml')37 38    # Convert the image to grayscale39    gray_image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)40 41    # Detect faces in the image42    faces = face_cascade.detectMultiScale(gray_image, scaleFactor=1.1, minNeighbors=5, minSize=(30, 30))43 44    # Extract face locations45    print(2)46    face_locations = []47    for (x, y, w, h) in faces:48        face_locations.append({"top": y, "right": x + w, "bottom": y + h, "left": x})49    print(3)50    return face_locations51 52def seperate_image_text_from_pdf(pdf_url):53    # List to store page information54    try:55        pages_info = []56 57        # Fetch the PDF from the URL58        response = requests.get(pdf_url)59 60        if response.status_code == 200:61            # Create a temporary file to save the PDF data62            with tempfile.NamedTemporaryFile(delete=False) as tmp_file:63                tmp_file.write(response.content)64                tmp_file_path = tmp_file.name65 66            # Open the PDF67            pdf = fitz.open(tmp_file_path)68 69            # Iterate through each page70            for page_num in range(len(pdf)):71                page = pdf.load_page(page_num)72 73                # Extract text74                text = page.get_text()75 76                # Count images77                image_list = page.get_images(full=True)78 79                # Convert images to BytesIO and store in a list80                images_bytes = []81                for img_index, img_info in enumerate(image_list):82                    xref = img_info[0]83                    base_image = pdf.extract_image(xref)84                    image_bytes = base_image["image"]85                    images_bytes.append(image_bytes)86 87                # Store page information in a dictionary88                page_info = {89                    "pgno": page_num + 1,90                    "images": images_bytes,91                    "text": text92                }93 94                # Append page information to the list95                pages_info.append(page_info)96 97            # Close the PDF98            pdf.close()99 100            # Clean up the temporary file101            import os102            os.unlink(tmp_file_path)103        else:104            print("Failed to fetch the PDF from the URL.")105    except Exception as e:106        print("An error occurred:", e)107        return "Error"108 109    return pages_info110 111def pdf_image_text_embedding_and_text_embedding(pages_info):112    try:113        page_embeddings = []114 115        # Iterate through each page116        for page in pages_info:117            # Extract text from the page118            text = page.get("text", "")119            images = page.get("images", [])120 121            image_embeddings = []122            for image in images:123                try:124                    image_embedding = get_image_embedding(image)125                    extracted_text = extract_text(image)126                    image_embeddings.append({"image_embedding": image_embedding.tolist(), "extracted_text": extracted_text})127                except Exception as image_error:128                    print(f"Error processing image: {image_error}")129                    # Log the error or handle it as needed130 131            # Store the page embeddings in a dictionary132            page_embedding = {133                "images": image_embeddings,134                "text": text,135            }136 137            page_embeddings.append(page_embedding)138 139        return page_embeddings140    except Exception as e:141        print(f"An error occurred: {e}")142        return "Error"143    144 145def separate_audio_from_video(video_url):146    try:147        # Load the video file148        video = mp.VideoFileClip(video_url)149 150        # Extract audio151        audio = video.audio152 153        # Create a temporary file to write the audio data154        try :155            with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as temp_audio_file:156                temp_audio_filename = temp_audio_file.name157 158                # Write the audio data to the temporary file159                audio.write_audiofile(temp_audio_filename)160 161                # Read the audio data from the temporary file as bytes162                with open(temp_audio_filename, "rb") as f:163                    audio_bytes = f.read()164        except Exception as e:165            return "Error"166 167        return audio_bytes168 169    except Exception as e:170        print("An error occurred:", e)171        return "Error"172 173@cache.cached(timeout=300)174@app.route('/get_text_embedding', methods=['POST'])175def get_text_embedding_route():176    try:177        text = request.json.get("text")178        text_embedding = get_text_vector(text)179        return jsonify({"text_embedding": text_embedding}), 200180 181    except Exception as e:182        return jsonify({"error": str(e)}), 500183 184 185@cache.cached(timeout=300)186@app.route('/extract_audio_text_and_embedding', methods=['POST'])187def get_audio_embedding_route():188    audio_url = request.json.get('audio_url')189    print(audio_url)190    response = requests.get(audio_url)191    audio_data = response.content192    audio_embedding = extract_audio_embeddings(audio_data)193    audio_embedding_list = audio_embedding194    audio_file = BytesIO(audio_data)195    r = sr.Recognizer()196    with sr.AudioFile(audio_file) as source:197        audio_data = r.record(source)198    extracted_text = ""199    try:200        text = r.recognize_google(audio_data)201        extracted_text = text202    except Exception as e:203        print(e)204    return jsonify({"extracted_text": extracted_text, "audio_embedding": audio_embedding_list}), 200205 206# Route to get image embeddings207@cache.cached(timeout=300)208@app.route('/extract_image_text_and_embedding', methods=['POST'])209def get_image_embedding_route():210    try:211        image_url = request.json.get("imageUrl")212        print(image_url)213        response = requests.get(image_url)214        if response.status_code != 200:215            return jsonify({"error": "Failed to download image"}), 500216        binary_data = response.content217        extracted_text = extract_text(binary_data)218        image_embedding = get_image_embedding(binary_data)219        image_embedding_list = image_embedding.tolist()220        return jsonify({"image_embedding": image_embedding_list,"extracted_text":extracted_text}), 200221 222    except Exception as e:223        return jsonify({"error": str(e)}), 500224 225# Route to get video embeddings226@cache.cached(timeout=300)227@app.route('/extract_video_text_and_embedding', methods=['POST'])228def get_video_embedding_route():229    try:230        video_url = request.json.get("videoUrl")231        try:232            audio_data = separate_audio_from_video(video_url)233        except Exception as e:234            return jsonify({"error": "Failed to extract audio from video 1"}), 500235        try:236            audio_embedding = extract_audio_embeddings(audio_data)237        except Exception as e:238            return jsonify({"error": "Failed to extract audio embeddings 2 "+e}), 500239        audio_embedding_list = audio_embedding240        try :241            audio_file = io.BytesIO(audio_data)242        except Exception as e:243            return jsonify({"error": "Failed to extract audio embeddings 3"}), 500244        try :245            r = sr.Recognizer()246            with sr.AudioFile(audio_file) as source:247                audio_data = r.record(source)248        except Exception as e:249            return jsonify({"error": "Failed to extract audio embeddings 4"}), 500250        extracted_text = ""251        try:252            text = r.recognize_google(audio_data)253            extracted_text = text254        except Exception as e:255            print(e)256        video_embedding = get_video_embedding(video_url)257        return jsonify({"video_embedding": video_embedding,"extracted_audio_text": extracted_text, "audio_embedding": audio_embedding_list}), 200258 259    except Exception as e:260        print(e)261        return jsonify({"error": str(e)}), 500262 263@cache.cached(timeout=300)264@app.route('/extract_pdf_text_and_embedding', methods=['POST'])265def extract_pdf_text_and_embedding():266    list = []267    try:268        list.append(1)269        pdf_url = request.json.get("pdfUrl")270        list.append(2)271        print(1)272        pages_info = "Error"273        try :274            pages_info = seperate_image_text_from_pdf(pdf_url)275        except Exception as e:276            print(e)277            return jsonify({"error": "Failed to fetch the PDF from the URL"}), 500278        list.append(3)279        if(pages_info == "Error"):280            return jsonify({"error": "Failed to fetch the PDF from the URL seperate_image_text_from_pdf "}), 500281        list.append(4)282        print(pages_info)283        try:284            content = pdf_image_text_embedding_and_text_embedding(pages_info)285        except Exception as e1:286            print(e1)287            return jsonify({"error": "An error occurred 1 while processing the PDF"}), 500288        if content == "Error":289            return jsonify({"error": "An error occurred 2 while processing the PDF"}), 500290        list.append(5)291        print(content)292        return jsonify({"content": content}), 200293 294    except Exception as e:295        print(e)296        return jsonify({"error": str(list)}), 500297    finally:298        print("kasi",list)299 300# Route to get text description embeddings301@cache.cached(timeout=300)302@app.route('/getTextDescriptionEmbedding', methods=['POST'])303def get_text_description_embedding_route():304    try:305        text = request.json.get("text")306        text_description_embedding = get_text_discription_vector(text)307        return jsonify({"text_description_embedding": text_description_embedding.tolist()}), 200308 309    except Exception as e:310        return jsonify({"error": str(e)}), 500311 312 313 314# Route to get object detection results315@cache.cached(timeout=300)316@app.route('/detectObjects', methods=['POST'])317def detect_objects_route():318    try:319        image_url = request.json.get("imageUrl")320        response = requests.get(image_url)321        if response.status_code != 200:322            return jsonify({"error": "Failed to download image"}), 500323        binary_data = response.content324        object_detection_results = detect_objects(binary_data)325        return jsonify({"object_detection_results": object_detection_results}), 200326 327    except Exception as e:328        return jsonify({"error": str(e)}), 500329 330# Route to get face locations331@cache.cached(timeout=300)332@app.route('/getFaceLocations', methods=['POST'])333def get_face_locations_route():334    try:335        image_url = request.json.get("imageUrl")336        response = requests.get(image_url)337        print(11)338        if response.status_code != 200:339            return jsonify({"error": "Failed to download image"}), 500340        print(22)341        binary_data = response.content342        face_locations = get_face_locations(binary_data)343        print(33)344        print("ok",face_locations)345        return jsonify({"face_locations": str(face_locations)}), 200346 347    except Exception as e:348        print(e)349        return jsonify({"error": str(e)}), 500350 351# Route to get similarity score352@cache.cached(timeout=300)353@app.route('/getSimilarityScore', methods=['POST'])354def get_similarity_score_route():355    try:356        embedding1 = request.json.get("embedding1")357        embedding2 = request.json.get("embedding2")358        # Assuming embeddings are provided as lists359        similarity_score = get_all_similarities(embedding1, embedding2)360        return jsonify({"similarity_score": similarity_score}), 200361 362    except Exception as e:363        return jsonify({"error": str(e)}), 500