mobenta/HTML_Content_Processor
0
1 2import nltk3from unstructured.documents.html import HTMLDocument4import requests5from bs4 import BeautifulSoup6from reportlab.lib.pagesizes import letter7from reportlab.pdfgen import canvas8import gradio as gr9 10# Download and install NLTK data11nltk.download('punkt')12nltk.download('averaged_perceptron_tagger')13 14# Function to process HTML content from a given URL15def process_html_from_url(url):16 response = requests.get(url)17 18 # Check if the request was successful19 if response.status_code == 200:20 # Get the HTML content of the page21 html_content = response.text22 23 # Extract text content from HTML using BeautifulSoup24 soup = BeautifulSoup(html_content, 'html.parser')25 page_content = soup.get_text()26 27 # Save the parsed content to a text file28 text_filename = 'output.txt'29 with open(text_filename, 'w') as f:30 f.write(page_content)31 32 # Save the parsed content to a PDF file33 pdf_filename = 'output.pdf'34 save_text_to_pdf(page_content, pdf_filename)35 36 return text_filename, pdf_filename37 else:38 return None, None39 40def save_text_to_pdf(text, filename):41 c = canvas.Canvas(filename, pagesize=letter)42 width, height = letter43 44 # Split the text into lines45 lines = text.split('\n')46 47 # Define the starting position48 x = 4049 y = height - 4050 line_height = 1251 52 # Add text to the canvas53 for line in lines:54 if y < 40:55 c.showPage()56 y = height - 4057 c.drawString(x, y, line)58 y -= line_height59 60 # Save the PDF file61 c.save()62 63# Function to be used by Gradio interface64def gradio_process(url):65 text_file, pdf_file = process_html_from_url(url)66 if text_file and pdf_file:67 return text_file, pdf_file68 else:69 return "Failed to retrieve HTML content", ""70 71# Create the Gradio interface72iface = gr.Interface(73 fn=gradio_process,74 inputs=gr.Textbox(label="Enter the URL to process"),75 outputs=[76 gr.File(label="Text File"),77 gr.File(label="PDF File")78 ],79 title="HTML Content Processor",80 description="Enter a URL to download and process its HTML content. You can download the resulting text and PDF files."81)82 83# Launch the Gradio app84iface.launch(debug=True)85 86 