salverz/llm-document-parser
0
1import os2from pathlib import Path3from docling.datamodel.document import ConversionResult4from huggingface_hub import snapshot_download5 6from docling.datamodel.base_models import InputFormat7from docling.datamodel.pipeline_options import EasyOcrOptions, OcrMacOptions, PdfPipeline, PdfPipelineOptions, PipelineOptions, RapidOcrOptions, TesseractOcrOptions8from docling.document_converter import DocumentConverter, ImageFormatOption, PdfFormatOption9from docling.backend.pypdfium2_backend import PyPdfiumDocumentBackend, PyPdfiumPageBackend10from docling.pipeline.standard_pdf_pipeline import StandardPdfPipeline11from docling.pipeline.simple_pipeline import SimplePipeline12 13 14# TODO: REFACTOR LOAD OCR MODEL TO JUST EITHER USE SERVER MODELS OR MOBILE MODELS15def load_rapid_ocr_model(det_model: str, rec_model: str, cls_model: str) -> DocumentConverter:16 """17 Load the RapidOCR model from Hugging Face Hub.18 Args:19 det_model (str): Path to the detection model.20 rec_model (str): Path to the recognition model.21 cls_model (str): Path to the classification model.22 Returns:23 DocumentConverter: The loaded RapidOCR model.24 """25 print("Downloading RapidOCR models")26 download_path = snapshot_download(repo_id="SWHL/RapidOCR")27 28 det_model_path = os.path.join(29 download_path, det_model30 )31 rec_model_path = os.path.join(32 download_path, rec_model33 )34 cls_model_path = os.path.join(35 download_path, cls_model36 )37 38 ocr_options = RapidOcrOptions(39 det_model_path=det_model_path,40 rec_model_path=rec_model_path,41 cls_model_path=cls_model_path42 )43 44 pipeline_options = PdfPipelineOptions(45 ocr_options=ocr_options46 )47 48 doc_converter = DocumentConverter(49 format_options={50 InputFormat.IMAGE: ImageFormatOption(51 pipeline_options=pipeline_options52 )53 }54 )55 56 return doc_converter57 58def load_ocr_mac_model() -> DocumentConverter:59 """60 Load the OCR Mac model.61 Returns:62 DocumentConverter: The loaded OCR Mac model.63 """64 ocr_options = OcrMacOptions(65 framework='vision'66 )67 68 pipeline_options = PdfPipelineOptions(69 ocr_options=ocr_options70 )71 72 doc_converter = DocumentConverter(73 allowed_formats=[74 InputFormat.PDF,75 InputFormat.IMAGE,76 ],77 format_options={78 InputFormat.PDF: PdfFormatOption(79 pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options80 ),81 InputFormat.IMAGE: PdfFormatOption(82 pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options83 )84 }85 )86 87 return doc_converter88 89def load_tesseract_model(tessdata_path: str) -> DocumentConverter:90 """91 Load the Tesseract OCR model. 92 Args:93 tessdata_path (str): Path to the Tesseract data directory.94 Returns:95 DocumentConverter: The loaded Tesseract OCR model.96 """97 os.environ["TESSDATA_PREFIX"] = tessdata_path98 99 ocr_options = TesseractOcrOptions()100 101 pipeline_options = PdfPipelineOptions(102 ocr_options=ocr_options103 )104 105 doc_converter = DocumentConverter(106 allowed_formats=[107 InputFormat.PDF,108 InputFormat.IMAGE109 ],110 format_options={111 InputFormat.PDF: PdfFormatOption(112 pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options113 ),114 InputFormat.IMAGE: PdfFormatOption(115 pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options116 )117 }118 )119 120 return doc_converter121 122def load_easy_ocr_model() -> DocumentConverter:123 """124 Load the EasyOCR model.125 Returns:126 DocumentConverter: The loaded EasyOCR model.127 """128 ocr_options = EasyOcrOptions()129 130 pipeline_options = PdfPipelineOptions(131 ocr_options=ocr_options132 )133 134 doc_converter = DocumentConverter(135 allowed_formats=[136 InputFormat.PDF,137 InputFormat.IMAGE138 ],139 format_options={140 InputFormat.PDF: PdfFormatOption(141 pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options142 ),143 InputFormat.IMAGE: PdfFormatOption(144 pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options145 )146 }147 )148 return doc_converter149 150def image_to_text(document_converter: DocumentConverter, file_path: Path) -> ConversionResult:151 """152 Convert an image to text using the specified document converter.153 Args:154 document_converter (DocumentConverter): The document converter to use.155 file_path (Path): Path to the image file.156 Returns:157 ConversionResult: The result of the conversion.158 """159 conv_results = document_converter.convert(file_path)160 return conv_results161 