sagarsahoo220887/ocr_image_processing
0
1from PIL import Image2import pytesseract3import os4import fitz # PyMuPDF5 6# Specify the path to the Tesseract executable if necessary7# pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'8 9 10def load_ocr_image(image_filename):11 img = Image.open(image_filename)12 return img13 14def write_to_file(file_name_path, text_content):15 print('Writing to file - ' + file_name_path)16 f = open(file_name_path, 'w')17 f.write(text_content)18 f.close()19 20def delete_file(file_path):21 os.remove(file_path)22 23def create_temp_file(loaded_file):24 # save the file temporarily25 temp_file = f"./tmp_{loaded_file.name}"26 with open(temp_file, "wb") as file:27 file.write(loaded_file.getvalue())28 return temp_file29 30## TODO: write ur own custom method31def extract_text_from_image(image_path):32 # Open the image file33 try:34 with Image.open(image_path) as img:35 # Use pytesseract to do OCR on the image36 text = pytesseract.image_to_string(img)37 # Write the extracted text to a file38 # write_to_file('ocr_text.txt', text)39 return text40 except Exception as e:41 print(f"Error processing the image: {e}")42 return None43 44def extarct_text_from_ocr_pdf(pdf_file):45 print('Inside extarct_text_from_ocr_pdf ===> ', pdf_file)46 try:47 tmp_pdf_file_path = create_temp_file(pdf_file)48 print('tmp_pdf_file_path - ', tmp_pdf_file_path)49 # Open the PDF file50 doc = fitz.open(tmp_pdf_file_path)51 extracted_text = ""52 # Iterate through each page53 for page in doc:54 extracted_text += page.get_text() + "\n"55 # Close the PDF file56 doc.close()57 # Write the extracted text to a file58 # write_to_file('ocr_text.txt', extracted_text)59 delete_file(tmp_pdf_file_path)60 return extracted_text61 except Exception as e:62 print('Exception in extarct_text_from_ocr_pdf - ', e)63 64 65def extract_ocr_data(uploaded_file):66 print('Inside extract_ocr_data')67 try:68 if uploaded_file.type in ['image/jpg', 'image/jpeg', 'image/png']:69 ocr_extracted_data = extract_text_from_image(uploaded_file)70 return ocr_extracted_data71 72 elif uploaded_file.type in ['application/pdf']:73 # For PDF files74 # pdf_extracted_data = read_pdf(uploaded_file)75 # st.write(pdf_extracted_data)76 77 # For OCR pdf files78 ocr_extracted_data = extarct_text_from_ocr_pdf(uploaded_file)79 return ocr_extracted_data80 except Exception as e:81 print('Exception inside extract_ocr_data - ', e)82 83def display_ocr_file(uploaded_file):84 img = load_ocr_image(uploaded_file)85 return img86 