Team Ai
Apppublic

mobenta/HTML_Content_Processor

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
app.py86 linesDownload Raw Back to root
1 2import nltk3from unstructured.documents.html import HTMLDocument4import requests5from bs4 import BeautifulSoup6from reportlab.lib.pagesizes import letter7from reportlab.pdfgen import canvas8import gradio as gr9 10# Download and install NLTK data11nltk.download('punkt')12nltk.download('averaged_perceptron_tagger')13 14# Function to process HTML content from a given URL15def process_html_from_url(url):16    response = requests.get(url)17 18    # Check if the request was successful19    if response.status_code == 200:20        # Get the HTML content of the page21        html_content = response.text22 23        # Extract text content from HTML using BeautifulSoup24        soup = BeautifulSoup(html_content, 'html.parser')25        page_content = soup.get_text()26 27        # Save the parsed content to a text file28        text_filename = 'output.txt'29        with open(text_filename, 'w') as f:30            f.write(page_content)31 32        # Save the parsed content to a PDF file33        pdf_filename = 'output.pdf'34        save_text_to_pdf(page_content, pdf_filename)35 36        return text_filename, pdf_filename37    else:38        return None, None39 40def save_text_to_pdf(text, filename):41    c = canvas.Canvas(filename, pagesize=letter)42    width, height = letter43 44    # Split the text into lines45    lines = text.split('\n')46 47    # Define the starting position48    x = 4049    y = height - 4050    line_height = 1251 52    # Add text to the canvas53    for line in lines:54        if y < 40:55            c.showPage()56            y = height - 4057        c.drawString(x, y, line)58        y -= line_height59 60    # Save the PDF file61    c.save()62 63# Function to be used by Gradio interface64def gradio_process(url):65    text_file, pdf_file = process_html_from_url(url)66    if text_file and pdf_file:67        return text_file, pdf_file68    else:69        return "Failed to retrieve HTML content", ""70 71# Create the Gradio interface72iface = gr.Interface(73    fn=gradio_process,74    inputs=gr.Textbox(label="Enter the URL to process"),75    outputs=[76        gr.File(label="Text File"),77        gr.File(label="PDF File")78    ],79    title="HTML Content Processor",80    description="Enter a URL to download and process its HTML content. You can download the resulting text and PDF files."81)82 83# Launch the Gradio app84iface.launch(debug=True)85 86