Team Ai
Apppublic

LukaszBergiel/Final_Assignment_Template_OpenAPI

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
browser.py62 linesDownload Raw Back to tools
1from smolagents import tool2import requests3from markdownify import markdownify as md4from bs4 import BeautifulSoup5import time6 7@tool8def fetch_webpage(url: str, convert_to_markdown: bool = True) -> str:9    """10    Fetches the HTML content of a given URL. 11    if markdown conversion is enabled, it will remove script and style and return the text content as markdown else return raw unfiltered HTML12    Args:13        url (str): The URL to fetch.14        convert_to_markdown (bool): If True, convert the HTML content to Markdown format. else return the raw HTML.15    Returns:16        str: The HTML content of the URL.17    """    18    content = None19    response = requests.get(url, timeout=30)20    if (convert_to_markdown):21        soup = BeautifulSoup(response.text, "html.parser")22        # remove script and style tags23        for script in soup(["script", "style"]):24            script.extract()25 26        # for wikipedia only keep the main content27        if "wikipedia.org" in url:28            main_content = soup.find("main",{"id":"content"})29            if main_content:30                content = md(str(main_content),strip=['script', 'style'], heading_style="ATX").strip()31            else:32                content = md(response.text,strip=['script', 'style'], heading_style="ATX").strip()33    else:34        content = response.text35    36    save_file_with_timestamp(content, "webpage", ".md" if convert_to_markdown else ".html")37           38    return content39 40def save_file_with_timestamp(content: str, file_name: str, extension: str) -> str:41    """42    Save content to a file with a timestamp.43    Args:44        content (str): The content to save.45        file_name (str): The base name of the file.46    Returns:47        str: The path to the saved file.48    """49    try:50        # save content to a file in test folder before returning51        # compute filepath with correct extension based on convert_to_markdown and add a timestamp for unicity52        53        unicity_suffix = str(int(time.time()))54        55        file_path = f"test/{file_name}_{unicity_suffix}.{extension}"56        with open(file_name, "w", encoding="utf-8") as f:57            f.write(content)58    except Exception as e:59        print(f"Error saving content to file: {e}")60    return file_name61 62