Team Ai
Apppublic

ashcodes/pdf-table-extractor-tabula

sourceHugging Faceupdated 4y agoView on Hugging Face
0likes
app.py64 linesDownload Raw Back to root
1import streamlit as st2import numpy as np 3import pandas as pd 4import subprocess 5from subprocess import STDOUT, check_call 6import os 7import base64  8import camelot 9 10# to run this only once and it's cached11@st.cache12def ghostscript():13    """install ghostscript on the linux machine"""14    proc = subprocess.Popen('apt-get install -y ghostscript', shell=True, stdin=None, stdout=open(os.devnull,"wb"), stderr=STDOUT, executable="/bin/bash")15    proc.wait()16 17ghostscript()18 19#heading20html_temp = """21    <div style="background-color:tomato;padding:10px">22    <h2 style="color:white;text-align:center;">PDF Table Extractor WebApp </h2>23    </div>24    """25st.markdown(html_temp,unsafe_allow_html=True)26 27 28# file uploader on streamlit 29#st.sidebar.markdown('Upload PDF files')30input_pdf = st.sidebar.file_uploader(label = "Upload PDF files here", type = 'pdf')31 32# run this only when a PDF is uploaded33if input_pdf is not None:34    # byte object into a PDF file 35    with open("input.pdf", "wb") as f:36        base64_pdf = base64.b64encode(input_pdf.read()).decode('utf-8')37        f.write(base64.b64decode(base64_pdf))38    f.close()39 40#To print uploaded pdf    41def show_pdf(file_path):42    with open(file_path,"rb") as f:43        base64_pdf = base64.b64encode(f.read()).decode('utf-8')44        pdf_display = f'<iframe src="data:application/pdf;base64,{base64_pdf}" width="800" height="800" type="application/pdf"></iframe>'45    st.markdown('## Uploaded PDF')46    st.markdown(pdf_display, unsafe_allow_html=True)47 48#st.sidebar.markdown('Display Uploaded PDF')    49#if st.sidebar.button('Show'):50    #show_pdf("input.pdf")51 52# read the pdf and parse it using stream53if input_pdf is not None:54    table = camelot.read_pdf('input.pdf', flavor='stream',layout_kwargs={'detect_vertical':True},backend='poppler')55    csv_table = table[0].df   56 57st.sidebar.markdown('Extract tables from PDF')58if st.sidebar.button('Extract Table'):59    st.markdown('## Extracted table from PDF')60    st.dataframe(csv_table)61    62if input_pdf is not None:63    st.sidebar.markdown('Download Extracted Table as CSV file')64    st.sidebar.download_button("Download",csv_table.to_csv(),file_name = 'extracted_table.csv', mime = 'text/csv')