ImageProcessing/backend
0
1import pickle
2import re
3from PIL import Image
4from transformers import pipeline
5import io
6
7def clean_text(text):
8 clean_text = re.sub(r'<[^>]+>', '', text)
9 clean_text = clean_text.strip()
10 clean_text = re.sub(r'\s+', ' ', clean_text)
11 return clean_text
12
13pipe = pipeline("image-to-text", model="jinhybr/OCR-Donut-CORD")
14
15def extract_text(binary_image):
16 image = Image.open(io.BytesIO(binary_image))
17 result = pipe(image)
18 text = result[0]['generated_text']
19 cleaned_text = clean_text(text)
20 return cleaned_text
21
22# print(extract_text(open("pictures/users/2.jpg", "rb").read()))
23
24print("OCR pipeline loaded successfully!")