Team Ai
Apppublic

salverz/llm-document-parser

sourceHugging Faceapache-2.0updated 1y agoView on Hugging Face
0likes
convert_doc_docling.py161 linesDownload Raw Back to llm_document_parser
1import os2from pathlib import Path3from docling.datamodel.document import ConversionResult4from huggingface_hub import snapshot_download5 6from docling.datamodel.base_models import InputFormat7from docling.datamodel.pipeline_options import EasyOcrOptions, OcrMacOptions, PdfPipeline, PdfPipelineOptions, PipelineOptions, RapidOcrOptions, TesseractOcrOptions8from docling.document_converter import DocumentConverter, ImageFormatOption, PdfFormatOption9from docling.backend.pypdfium2_backend import PyPdfiumDocumentBackend, PyPdfiumPageBackend10from docling.pipeline.standard_pdf_pipeline import StandardPdfPipeline11from docling.pipeline.simple_pipeline import SimplePipeline12 13 14# TODO: REFACTOR LOAD OCR MODEL TO JUST EITHER USE SERVER MODELS OR MOBILE MODELS15def load_rapid_ocr_model(det_model: str, rec_model: str, cls_model: str) -> DocumentConverter:16    """17    Load the RapidOCR model from Hugging Face Hub.18    Args:19        det_model (str): Path to the detection model.20        rec_model (str): Path to the recognition model.21        cls_model (str): Path to the classification model.22    Returns:23        DocumentConverter: The loaded RapidOCR model.24    """25    print("Downloading RapidOCR models")26    download_path = snapshot_download(repo_id="SWHL/RapidOCR")27 28    det_model_path = os.path.join(29        download_path, det_model30    )31    rec_model_path = os.path.join(32        download_path, rec_model33    )34    cls_model_path = os.path.join(35        download_path, cls_model36    )37 38    ocr_options = RapidOcrOptions(39        det_model_path=det_model_path,40        rec_model_path=rec_model_path,41        cls_model_path=cls_model_path42    )43 44    pipeline_options = PdfPipelineOptions(45        ocr_options=ocr_options46    )47 48    doc_converter = DocumentConverter(49        format_options={50            InputFormat.IMAGE: ImageFormatOption(51                pipeline_options=pipeline_options52            )53        }54    )55 56    return doc_converter57 58def load_ocr_mac_model() -> DocumentConverter:59    """60    Load the OCR Mac model.61    Returns:62        DocumentConverter: The loaded OCR Mac model.63    """64    ocr_options = OcrMacOptions(65        framework='vision'66    )67 68    pipeline_options = PdfPipelineOptions(69        ocr_options=ocr_options70    )71 72    doc_converter = DocumentConverter(73        allowed_formats=[74            InputFormat.PDF,75            InputFormat.IMAGE,76        ],77        format_options={78            InputFormat.PDF: PdfFormatOption(79                pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options80            ),81            InputFormat.IMAGE: PdfFormatOption(82                pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options83            )84        }85    )86    87    return doc_converter88 89def load_tesseract_model(tessdata_path: str) -> DocumentConverter:90    """91    Load the Tesseract OCR model. 92    Args:93        tessdata_path (str): Path to the Tesseract data directory.94    Returns:95        DocumentConverter: The loaded Tesseract OCR model.96    """97    os.environ["TESSDATA_PREFIX"] = tessdata_path98 99    ocr_options = TesseractOcrOptions()100 101    pipeline_options = PdfPipelineOptions(102        ocr_options=ocr_options103    )104 105    doc_converter = DocumentConverter(106        allowed_formats=[107            InputFormat.PDF,108            InputFormat.IMAGE109        ],110        format_options={111            InputFormat.PDF: PdfFormatOption(112                pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options113            ),114            InputFormat.IMAGE: PdfFormatOption(115                pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options116            )117        }118    )119 120    return doc_converter121 122def load_easy_ocr_model() -> DocumentConverter:123    """124    Load the EasyOCR model.125    Returns:126        DocumentConverter: The loaded EasyOCR model.127    """128    ocr_options = EasyOcrOptions()129 130    pipeline_options = PdfPipelineOptions(131        ocr_options=ocr_options132    )133 134    doc_converter = DocumentConverter(135        allowed_formats=[136            InputFormat.PDF,137            InputFormat.IMAGE138        ],139        format_options={140            InputFormat.PDF: PdfFormatOption(141                pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options142            ),143            InputFormat.IMAGE: PdfFormatOption(144                pipeline_cls=StandardPdfPipeline, backend=PyPdfiumDocumentBackend, pipeline_options=pipeline_options145            )146        }147    )148    return doc_converter149 150def image_to_text(document_converter: DocumentConverter, file_path: Path) -> ConversionResult:151    """152    Convert an image to text using the specified document converter.153    Args:154        document_converter (DocumentConverter): The document converter to use.155        file_path (Path): Path to the image file.156    Returns:157        ConversionResult: The result of the conversion.158    """159    conv_results = document_converter.convert(file_path)160    return conv_results161