Team Ai
Apppublic

Amrrs/pdf-table-extractor

sourceHugging Faceupdated 5y agoView on Hugging Face
6likes
app.py63 linesDownload Raw Back to root
1import streamlit as st # data app development2import subprocess # process in the os3from subprocess import STDOUT, check_call #os process manipuation4import os #os process manipuation5import base64 # byte object into a pdf file 6import camelot as cam # extracting tables from PDFs 7 8# to run this only once and it's cached9@st.cache10def gh():11    """install ghostscript on the linux machine"""12    proc = subprocess.Popen('apt-get install -y ghostscript', shell=True, stdin=None, stdout=open(os.devnull,"wb"), stderr=STDOUT, executable="/bin/bash")13    proc.wait()14 15gh()16 17 18 19st.title("PDF Table Extractor")20st.subheader("with `Camelot` Python library")21 22st.image("https://raw.githubusercontent.com/camelot-dev/camelot/master/docs/_static/camelot.png", width=200)23 24 25# file uploader on streamlit 26 27input_pdf = st.file_uploader(label = "upload your pdf here", type = 'pdf')28 29st.markdown("### Page Number")30 31page_number = st.text_input("Enter the page # from where you want to extract the PDF eg: 3", value = 1)32 33# run this only when a PDF is uploaded34 35if input_pdf is not None:36    # byte object into a PDF file 37    with open("input.pdf", "wb") as f:38        base64_pdf = base64.b64encode(input_pdf.read()).decode('utf-8')39        f.write(base64.b64decode(base64_pdf))40    f.close()41 42    # read the pdf and parse it using stream43    table = cam.read_pdf("input.pdf", pages = page_number, flavor = 'stream')44 45    st.markdown("### Number of Tables")46 47    # display the output after parsing 48    st.write(table)49 50    # display the table51 52    if len(table) > 0:53 54        # extract the index value of the table55        56        option = st.selectbox(label = "Select the Table to be displayed", options = range(len(table) + 1))57 58        st.markdown('### Output Table')59 60        # display the dataframe61        62        st.dataframe(table[int(option)-1].df)63