Team Ai
Apppublic

sagarsahoo220887/ocr_image_processing

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
ocr_process.py86 linesDownload Raw Back to root
1from PIL import Image2import pytesseract3import os4import fitz  # PyMuPDF5 6# Specify the path to the Tesseract executable if necessary7# pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'8 9 10def load_ocr_image(image_filename):11    img = Image.open(image_filename)12    return img13 14def write_to_file(file_name_path, text_content):15    print('Writing to file - ' + file_name_path)16    f = open(file_name_path, 'w')17    f.write(text_content)18    f.close()19 20def delete_file(file_path):21    os.remove(file_path)22 23def create_temp_file(loaded_file):24    # save the file temporarily25    temp_file = f"./tmp_{loaded_file.name}"26    with open(temp_file, "wb") as file:27        file.write(loaded_file.getvalue())28    return temp_file29 30## TODO: write ur own custom method31def extract_text_from_image(image_path):32    # Open the image file33    try:34        with Image.open(image_path) as img:35            # Use pytesseract to do OCR on the image36            text = pytesseract.image_to_string(img)37            # Write the extracted text to a file38            # write_to_file('ocr_text.txt', text)39            return text40    except Exception as e:41        print(f"Error processing the image: {e}")42        return None43    44def extarct_text_from_ocr_pdf(pdf_file):45    print('Inside extarct_text_from_ocr_pdf ===> ', pdf_file)46    try:47        tmp_pdf_file_path = create_temp_file(pdf_file)48        print('tmp_pdf_file_path - ', tmp_pdf_file_path)49        # Open the PDF file50        doc = fitz.open(tmp_pdf_file_path)51        extracted_text = ""52        # Iterate through each page53        for page in doc:54            extracted_text += page.get_text() + "\n"55        # Close the PDF file56        doc.close()57        # Write the extracted text to a file58        # write_to_file('ocr_text.txt', extracted_text)59        delete_file(tmp_pdf_file_path)60        return extracted_text61    except Exception as e:62        print('Exception in extarct_text_from_ocr_pdf - ', e)63 64 65def extract_ocr_data(uploaded_file):66    print('Inside extract_ocr_data')67    try:68        if uploaded_file.type in ['image/jpg', 'image/jpeg', 'image/png']:69            ocr_extracted_data = extract_text_from_image(uploaded_file)70            return ocr_extracted_data71            72        elif uploaded_file.type in ['application/pdf']:73            # For PDF files74            # pdf_extracted_data = read_pdf(uploaded_file)75            # st.write(pdf_extracted_data)76 77            # For OCR pdf files78            ocr_extracted_data =  extarct_text_from_ocr_pdf(uploaded_file)79            return ocr_extracted_data80    except Exception as e:81        print('Exception inside extract_ocr_data - ', e)82 83def display_ocr_file(uploaded_file):84    img = load_ocr_image(uploaded_file)85    return img86