huggingchat/document-parser
32
1import gradio as gr2import spaces3import subprocess4import os5import shutil6import string7import random8from pypdf import PdfReader9import ocrmypdf10 11 12def random_word(length):13 letters = string.ascii_lowercase14 return "".join(random.choice(letters) for _ in range(length))15 16 17def convert_pdf(input_file):18 reader = PdfReader(input_file)19 metadata = extract_metadata_from_pdf(reader)20 text = extract_text_from_pdf(reader)21 22 # Check if there are any images23 image_count = 024 for page in reader.pages:25 image_count += len(page.images)26 27 # If there are images and not much content, perform OCR on the document28 if image_count > 0 and len(text) < 1000:29 out_pdf_file = input_file.replace(".pdf", "_ocr.pdf")30 ocrmypdf.ocr(input_file, out_pdf_file, force_ocr=True)31 32 # Re-extract text33 text = extract_text_from_pdf(PdfReader(input_file))34 35 # Delete the OCR file36 os.remove(out_pdf_file)37 38 return text, metadata39 40 41def extract_text_from_pdf(reader):42 full_text = ""43 for idx, page in enumerate(reader.pages):44 text = page.extract_text()45 if len(text) > 0:46 full_text += f"---- Page {idx} ----\n" + page.extract_text() + "\n\n"47 48 return full_text.strip()49 50 51def extract_metadata_from_pdf(reader):52 return {53 "author": reader.metadata.author,54 "creator": reader.metadata.creator,55 "producer": reader.metadata.producer,56 "subject": reader.metadata.subject,57 "title": reader.metadata.title,58 }59 60 61def convert_pandoc(input_file, filename):62 # Temporarily copy the file63 shutil.copyfile(input_file, filename)64 65 # Convert the file to markdown with pandoc66 output_file = f"{random_word(16)}.md"67 result = subprocess.call(["pandoc", filename, "-t", "markdown", "-o", output_file])68 if result != 0:69 raise ValueError("Error converting file to markdown with pandoc")70 71 # Read the file and delete temporary files72 with open(output_file, "r") as f:73 markdown = f.read()74 os.remove(output_file)75 os.remove(filename)76 77 return markdown78 79 80@spaces.GPU81def convert(input_file, filename):82 plain_text_filetypes = [83 ".txt",84 ".csv",85 ".tsv",86 ".md",87 ".yaml",88 ".toml",89 ".json",90 ".json5",91 ".jsonc",92 ]93 # Already a plain text file that wouldn't benefit from pandoc so return the content94 if any(filename.endswith(ft) for ft in plain_text_filetypes):95 with open(input_file, "r") as f:96 return f.read(), {}97 98 if filename.endswith(".pdf"):99 return convert_pdf(input_file)100 101 return convert_pandoc(input_file, filename), {}102 103 104# We accept a filename because the gradio JS interface removes this information105# and it's critical for choosing the correct processing pipeline106gr.Interface(107 convert,108 inputs=[gr.File(label="Upload File", type="filepath"), gr.Text(label="Filename")],109 outputs=[110 gr.Text(label="Markdown"),111 gr.JSON(label="Metadata"),112 ],113).launch()114 