hprasath/image-processing
0
1import moviepy.editor as mp2from flask import Flask, request, jsonify3from flask_cors import CORS4import requests5from io import BytesIO6import speech_recognition as sr7import io8import fitz # PyMuPDF for working with PDFs9import numpy as np10import cv211from flask_caching import Cache12 13from utils.audioEmbedding.index import extract_audio_embeddings14from utils.videoEmbedding.index import get_video_embedding15from utils.imageToText.index import extract_text16from utils.sentanceEmbedding.index import get_text_vector , get_text_discription_vector17from utils.imageEmbedding.index import get_image_embedding18from utils.similarityScore import get_all_similarities19from utils.objectDetection.index import detect_objects20 21 22 23app = Flask(__name__)24cache = Cache(app, config={'CACHE_TYPE': 'simple'}) # You can choose a caching type based on your requirements25CORS(app)26import moviepy.editor as mp27import tempfile28 29def get_face_locations(binary_data):30 # Convert binary image data to numpy array31 print(1)32 nparr = np.frombuffer(binary_data, np.uint8)33 image = cv2.imdecode(nparr, cv2.IMREAD_COLOR)34 35 # Load the pre-trained face detection model36 face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_frontalface_default.xml')37 38 # Convert the image to grayscale39 gray_image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)40 41 # Detect faces in the image42 faces = face_cascade.detectMultiScale(gray_image, scaleFactor=1.1, minNeighbors=5, minSize=(30, 30))43 44 # Extract face locations45 print(2)46 face_locations = []47 for (x, y, w, h) in faces:48 face_locations.append({"top": y, "right": x + w, "bottom": y + h, "left": x})49 print(3)50 return face_locations51 52def seperate_image_text_from_pdf(pdf_url):53 # List to store page information54 try:55 pages_info = []56 57 # Fetch the PDF from the URL58 response = requests.get(pdf_url)59 60 if response.status_code == 200:61 # Create a temporary file to save the PDF data62 with tempfile.NamedTemporaryFile(delete=False) as tmp_file:63 tmp_file.write(response.content)64 tmp_file_path = tmp_file.name65 66 # Open the PDF67 pdf = fitz.open(tmp_file_path)68 69 # Iterate through each page70 for page_num in range(len(pdf)):71 page = pdf.load_page(page_num)72 73 # Extract text74 text = page.get_text()75 76 # Count images77 image_list = page.get_images(full=True)78 79 # Convert images to BytesIO and store in a list80 images_bytes = []81 for img_index, img_info in enumerate(image_list):82 xref = img_info[0]83 base_image = pdf.extract_image(xref)84 image_bytes = base_image["image"]85 images_bytes.append(image_bytes)86 87 # Store page information in a dictionary88 page_info = {89 "pgno": page_num + 1,90 "images": images_bytes,91 "text": text92 }93 94 # Append page information to the list95 pages_info.append(page_info)96 97 # Close the PDF98 pdf.close()99 100 # Clean up the temporary file101 import os102 os.unlink(tmp_file_path)103 else:104 print("Failed to fetch the PDF from the URL.")105 except Exception as e:106 print("An error occurred:", e)107 return "Error"108 109 return pages_info110 111def pdf_image_text_embedding_and_text_embedding(pages_info):112 try:113 page_embeddings = []114 115 # Iterate through each page116 for page in pages_info:117 # Extract text from the page118 text = page.get("text", "")119 images = page.get("images", [])120 121 image_embeddings = []122 for image in images:123 try:124 image_embedding = get_image_embedding(image)125 extracted_text = extract_text(image)126 image_embeddings.append({"image_embedding": image_embedding.tolist(), "extracted_text": extracted_text})127 except Exception as image_error:128 print(f"Error processing image: {image_error}")129 # Log the error or handle it as needed130 131 # Store the page embeddings in a dictionary132 page_embedding = {133 "images": image_embeddings,134 "text": text,135 }136 137 page_embeddings.append(page_embedding)138 139 return page_embeddings140 except Exception as e:141 print(f"An error occurred: {e}")142 return "Error"143 144 145def separate_audio_from_video(video_url):146 try:147 # Load the video file148 video = mp.VideoFileClip(video_url)149 150 # Extract audio151 audio = video.audio152 153 # Create a temporary file to write the audio data154 try :155 with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as temp_audio_file:156 temp_audio_filename = temp_audio_file.name157 158 # Write the audio data to the temporary file159 audio.write_audiofile(temp_audio_filename)160 161 # Read the audio data from the temporary file as bytes162 with open(temp_audio_filename, "rb") as f:163 audio_bytes = f.read()164 except Exception as e:165 return "Error"166 167 return audio_bytes168 169 except Exception as e:170 print("An error occurred:", e)171 return "Error"172 173@cache.cached(timeout=300)174@app.route('/get_text_embedding', methods=['POST'])175def get_text_embedding_route():176 try:177 text = request.json.get("text")178 text_embedding = get_text_vector(text)179 return jsonify({"text_embedding": text_embedding}), 200180 181 except Exception as e:182 return jsonify({"error": str(e)}), 500183 184 185@cache.cached(timeout=300)186@app.route('/extract_audio_text_and_embedding', methods=['POST'])187def get_audio_embedding_route():188 audio_url = request.json.get('audio_url')189 print(audio_url)190 response = requests.get(audio_url)191 audio_data = response.content192 audio_embedding = extract_audio_embeddings(audio_data)193 audio_embedding_list = audio_embedding194 audio_file = BytesIO(audio_data)195 r = sr.Recognizer()196 with sr.AudioFile(audio_file) as source:197 audio_data = r.record(source)198 extracted_text = ""199 try:200 text = r.recognize_google(audio_data)201 extracted_text = text202 except Exception as e:203 print(e)204 return jsonify({"extracted_text": extracted_text, "audio_embedding": audio_embedding_list}), 200205 206# Route to get image embeddings207@cache.cached(timeout=300)208@app.route('/extract_image_text_and_embedding', methods=['POST'])209def get_image_embedding_route():210 try:211 image_url = request.json.get("imageUrl")212 print(image_url)213 response = requests.get(image_url)214 if response.status_code != 200:215 return jsonify({"error": "Failed to download image"}), 500216 binary_data = response.content217 extracted_text = extract_text(binary_data)218 image_embedding = get_image_embedding(binary_data)219 image_embedding_list = image_embedding.tolist()220 return jsonify({"image_embedding": image_embedding_list,"extracted_text":extracted_text}), 200221 222 except Exception as e:223 return jsonify({"error": str(e)}), 500224 225# Route to get video embeddings226@cache.cached(timeout=300)227@app.route('/extract_video_text_and_embedding', methods=['POST'])228def get_video_embedding_route():229 try:230 video_url = request.json.get("videoUrl")231 try:232 audio_data = separate_audio_from_video(video_url)233 except Exception as e:234 return jsonify({"error": "Failed to extract audio from video 1"}), 500235 try:236 audio_embedding = extract_audio_embeddings(audio_data)237 except Exception as e:238 return jsonify({"error": "Failed to extract audio embeddings 2 "+e}), 500239 audio_embedding_list = audio_embedding240 try :241 audio_file = io.BytesIO(audio_data)242 except Exception as e:243 return jsonify({"error": "Failed to extract audio embeddings 3"}), 500244 try :245 r = sr.Recognizer()246 with sr.AudioFile(audio_file) as source:247 audio_data = r.record(source)248 except Exception as e:249 return jsonify({"error": "Failed to extract audio embeddings 4"}), 500250 extracted_text = ""251 try:252 text = r.recognize_google(audio_data)253 extracted_text = text254 except Exception as e:255 print(e)256 video_embedding = get_video_embedding(video_url)257 return jsonify({"video_embedding": video_embedding,"extracted_audio_text": extracted_text, "audio_embedding": audio_embedding_list}), 200258 259 except Exception as e:260 print(e)261 return jsonify({"error": str(e)}), 500262 263@cache.cached(timeout=300)264@app.route('/extract_pdf_text_and_embedding', methods=['POST'])265def extract_pdf_text_and_embedding():266 list = []267 try:268 list.append(1)269 pdf_url = request.json.get("pdfUrl")270 list.append(2)271 print(1)272 pages_info = "Error"273 try :274 pages_info = seperate_image_text_from_pdf(pdf_url)275 except Exception as e:276 print(e)277 return jsonify({"error": "Failed to fetch the PDF from the URL"}), 500278 list.append(3)279 if(pages_info == "Error"):280 return jsonify({"error": "Failed to fetch the PDF from the URL seperate_image_text_from_pdf "}), 500281 list.append(4)282 print(pages_info)283 try:284 content = pdf_image_text_embedding_and_text_embedding(pages_info)285 except Exception as e1:286 print(e1)287 return jsonify({"error": "An error occurred 1 while processing the PDF"}), 500288 if content == "Error":289 return jsonify({"error": "An error occurred 2 while processing the PDF"}), 500290 list.append(5)291 print(content)292 return jsonify({"content": content}), 200293 294 except Exception as e:295 print(e)296 return jsonify({"error": str(list)}), 500297 finally:298 print("kasi",list)299 300# Route to get text description embeddings301@cache.cached(timeout=300)302@app.route('/getTextDescriptionEmbedding', methods=['POST'])303def get_text_description_embedding_route():304 try:305 text = request.json.get("text")306 text_description_embedding = get_text_discription_vector(text)307 return jsonify({"text_description_embedding": text_description_embedding.tolist()}), 200308 309 except Exception as e:310 return jsonify({"error": str(e)}), 500311 312 313 314# Route to get object detection results315@cache.cached(timeout=300)316@app.route('/detectObjects', methods=['POST'])317def detect_objects_route():318 try:319 image_url = request.json.get("imageUrl")320 response = requests.get(image_url)321 if response.status_code != 200:322 return jsonify({"error": "Failed to download image"}), 500323 binary_data = response.content324 object_detection_results = detect_objects(binary_data)325 return jsonify({"object_detection_results": object_detection_results}), 200326 327 except Exception as e:328 return jsonify({"error": str(e)}), 500329 330# Route to get face locations331@cache.cached(timeout=300)332@app.route('/getFaceLocations', methods=['POST'])333def get_face_locations_route():334 try:335 image_url = request.json.get("imageUrl")336 response = requests.get(image_url)337 print(11)338 if response.status_code != 200:339 return jsonify({"error": "Failed to download image"}), 500340 print(22)341 binary_data = response.content342 face_locations = get_face_locations(binary_data)343 print(33)344 print("ok",face_locations)345 return jsonify({"face_locations": str(face_locations)}), 200346 347 except Exception as e:348 print(e)349 return jsonify({"error": str(e)}), 500350 351# Route to get similarity score352@cache.cached(timeout=300)353@app.route('/getSimilarityScore', methods=['POST'])354def get_similarity_score_route():355 try:356 embedding1 = request.json.get("embedding1")357 embedding2 = request.json.get("embedding2")358 # Assuming embeddings are provided as lists359 similarity_score = get_all_similarities(embedding1, embedding2)360 return jsonify({"similarity_score": similarity_score}), 200361 362 except Exception as e:363 return jsonify({"error": str(e)}), 500