Team Ai
Apppublic

huggingchat/document-parser

sourceHugging Facegpl-2.0updated 2y agoView on Hugging Face
32likes
app.py114 linesDownload Raw Back to root
1import gradio as gr2import spaces3import subprocess4import os5import shutil6import string7import random8from pypdf import PdfReader9import ocrmypdf10 11 12def random_word(length):13    letters = string.ascii_lowercase14    return "".join(random.choice(letters) for _ in range(length))15 16 17def convert_pdf(input_file):18    reader = PdfReader(input_file)19    metadata = extract_metadata_from_pdf(reader)20    text = extract_text_from_pdf(reader)21 22    # Check if there are any images23    image_count = 024    for page in reader.pages:25        image_count += len(page.images)26 27    # If there are images and not much content, perform OCR on the document28    if image_count > 0 and len(text) < 1000:29        out_pdf_file = input_file.replace(".pdf", "_ocr.pdf")30        ocrmypdf.ocr(input_file, out_pdf_file, force_ocr=True)31 32        # Re-extract text33        text = extract_text_from_pdf(PdfReader(input_file))34 35        # Delete the OCR file36        os.remove(out_pdf_file)37 38    return text, metadata39 40 41def extract_text_from_pdf(reader):42    full_text = ""43    for idx, page in enumerate(reader.pages):44        text = page.extract_text()45        if len(text) > 0:46            full_text += f"---- Page {idx} ----\n" + page.extract_text() + "\n\n"47 48    return full_text.strip()49 50 51def extract_metadata_from_pdf(reader):52    return {53        "author": reader.metadata.author,54        "creator": reader.metadata.creator,55        "producer": reader.metadata.producer,56        "subject": reader.metadata.subject,57        "title": reader.metadata.title,58    }59 60 61def convert_pandoc(input_file, filename):62    # Temporarily copy the file63    shutil.copyfile(input_file, filename)64 65    # Convert the file to markdown with pandoc66    output_file = f"{random_word(16)}.md"67    result = subprocess.call(["pandoc", filename, "-t", "markdown", "-o", output_file])68    if result != 0:69        raise ValueError("Error converting file to markdown with pandoc")70 71    # Read the file and delete temporary files72    with open(output_file, "r") as f:73        markdown = f.read()74    os.remove(output_file)75    os.remove(filename)76 77    return markdown78 79 80@spaces.GPU81def convert(input_file, filename):82    plain_text_filetypes = [83        ".txt",84        ".csv",85        ".tsv",86        ".md",87        ".yaml",88        ".toml",89        ".json",90        ".json5",91        ".jsonc",92    ]93    # Already a plain text file that wouldn't benefit from pandoc so return the content94    if any(filename.endswith(ft) for ft in plain_text_filetypes):95        with open(input_file, "r") as f:96            return f.read(), {}97 98    if filename.endswith(".pdf"):99        return convert_pdf(input_file)100 101    return convert_pandoc(input_file, filename), {}102 103 104# We accept a filename because the gradio JS interface removes this information105# and it's critical for choosing the correct processing pipeline106gr.Interface(107    convert,108    inputs=[gr.File(label="Upload File", type="filepath"), gr.Text(label="Filename")],109    outputs=[110        gr.Text(label="Markdown"),111        gr.JSON(label="Metadata"),112    ],113).launch()114