Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
web_reader_tool.py359 linesDownload Raw Back to utils
1import hashlib2import json3import mimetypes4import os5import re6import site7import subprocess8import tempfile9import unicodedata10from contextlib import contextmanager11from pathlib import Path12from typing import Optional13from urllib.parse import unquote14 15import chardet16import cloudscraper17from bs4 import BeautifulSoup, CData, Comment, NavigableString18from regex import regex19 20from core.helper import ssrf_proxy21from core.rag.extractor import extract_processor22from core.rag.extractor.extract_processor import ExtractProcessor23 24FULL_TEMPLATE = """25TITLE: {title}26AUTHORS: {authors}27PUBLISH DATE: {publish_date}28TOP_IMAGE_URL: {top_image}29TEXT:30 31{text}32"""33 34 35def page_result(text: str, cursor: int, max_length: int) -> str:36    """Page through `text` and return a substring of `max_length` characters starting from `cursor`."""37    return text[cursor : cursor + max_length]38 39 40def get_url(url: str, user_agent: Optional[str] = None) -> str:41    """Fetch URL and return the contents as a string."""42    headers = {43        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko)"44        " Chrome/91.0.4472.124 Safari/537.36"45    }46    if user_agent:47        headers["User-Agent"] = user_agent48 49    main_content_type = None50    supported_content_types = extract_processor.SUPPORT_URL_CONTENT_TYPES + ["text/html"]51    response = ssrf_proxy.head(url, headers=headers, follow_redirects=True, timeout=(5, 10))52 53    if response.status_code == 200:54        # check content-type55        content_type = response.headers.get("Content-Type")56        if content_type:57            main_content_type = response.headers.get("Content-Type").split(";")[0].strip()58        else:59            content_disposition = response.headers.get("Content-Disposition", "")60            filename_match = re.search(r'filename="([^"]+)"', content_disposition)61            if filename_match:62                filename = unquote(filename_match.group(1))63                extension = re.search(r"\.(\w+)$", filename)64                if extension:65                    main_content_type = mimetypes.guess_type(filename)[0]66 67        if main_content_type not in supported_content_types:68            return "Unsupported content-type [{}] of URL.".format(main_content_type)69 70        if main_content_type in extract_processor.SUPPORT_URL_CONTENT_TYPES:71            return ExtractProcessor.load_from_url(url, return_text=True)72 73        response = ssrf_proxy.get(url, headers=headers, follow_redirects=True, timeout=(120, 300))74    elif response.status_code == 403:75        scraper = cloudscraper.create_scraper()76        scraper.perform_request = ssrf_proxy.make_request77        response = scraper.get(url, headers=headers, follow_redirects=True, timeout=(120, 300))78 79    if response.status_code != 200:80        return "URL returned status code {}.".format(response.status_code)81 82    # Detect encoding using chardet83    detected_encoding = chardet.detect(response.content)84    encoding = detected_encoding["encoding"]85    if encoding:86        try:87            content = response.content.decode(encoding)88        except (UnicodeDecodeError, TypeError):89            content = response.text90    else:91        content = response.text92 93    a = extract_using_readabilipy(content)94 95    if not a["plain_text"] or not a["plain_text"].strip():96        return ""97 98    res = FULL_TEMPLATE.format(99        title=a["title"],100        authors=a["byline"],101        publish_date=a["date"],102        top_image="",103        text=a["plain_text"] or "",104    )105 106    return res107 108 109def extract_using_readabilipy(html):110    with tempfile.NamedTemporaryFile(delete=False, mode="w+") as f_html:111        f_html.write(html)112        f_html.close()113    html_path = f_html.name114 115    # Call Mozilla's Readability.js Readability.parse() function via node, writing output to a temporary file116    article_json_path = html_path + ".json"117    jsdir = os.path.join(find_module_path("readabilipy"), "javascript")118    with chdir(jsdir):119        subprocess.check_call(["node", "ExtractArticle.js", "-i", html_path, "-o", article_json_path])120 121    # Read output of call to Readability.parse() from JSON file and return as Python dictionary122    input_json = json.loads(Path(article_json_path).read_text(encoding="utf-8"))123 124    # Deleting files after processing125    os.unlink(article_json_path)126    os.unlink(html_path)127 128    article_json = {129        "title": None,130        "byline": None,131        "date": None,132        "content": None,133        "plain_content": None,134        "plain_text": None,135    }136    # Populate article fields from readability fields where present137    if input_json:138        if input_json.get("title"):139            article_json["title"] = input_json["title"]140        if input_json.get("byline"):141            article_json["byline"] = input_json["byline"]142        if input_json.get("date"):143            article_json["date"] = input_json["date"]144        if input_json.get("content"):145            article_json["content"] = input_json["content"]146            article_json["plain_content"] = plain_content(article_json["content"], False, False)147            article_json["plain_text"] = extract_text_blocks_as_plain_text(article_json["plain_content"])148        if input_json.get("textContent"):149            article_json["plain_text"] = input_json["textContent"]150            article_json["plain_text"] = re.sub(r"\n\s*\n", "\n", article_json["plain_text"])151 152    return article_json153 154 155def find_module_path(module_name):156    for package_path in site.getsitepackages():157        potential_path = os.path.join(package_path, module_name)158        if os.path.exists(potential_path):159            return potential_path160 161    return None162 163 164@contextmanager165def chdir(path):166    """Change directory in context and return to original on exit"""167    # From https://stackoverflow.com/a/37996581, couldn't find a built-in168    original_path = os.getcwd()169    os.chdir(path)170    try:171        yield172    finally:173        os.chdir(original_path)174 175 176def extract_text_blocks_as_plain_text(paragraph_html):177    # Load article as DOM178    soup = BeautifulSoup(paragraph_html, "html.parser")179    # Select all lists180    list_elements = soup.find_all(["ul", "ol"])181    # Prefix text in all list items with "* " and make lists paragraphs182    for list_element in list_elements:183        plain_items = "".join(184            list(filter(None, [plain_text_leaf_node(li)["text"] for li in list_element.find_all("li")]))185        )186        list_element.string = plain_items187        list_element.name = "p"188    # Select all text blocks189    text_blocks = [s.parent for s in soup.find_all(string=True)]190    text_blocks = [plain_text_leaf_node(block) for block in text_blocks]191    # Drop empty paragraphs192    text_blocks = list(filter(lambda p: p["text"] is not None, text_blocks))193    return text_blocks194 195 196def plain_text_leaf_node(element):197    # Extract all text, stripped of any child HTML elements and normalize it198    plain_text = normalize_text(element.get_text())199    if plain_text != "" and element.name == "li":200        plain_text = "* {}, ".format(plain_text)201    if plain_text == "":202        plain_text = None203    if "data-node-index" in element.attrs:204        plain = {"node_index": element["data-node-index"], "text": plain_text}205    else:206        plain = {"text": plain_text}207    return plain208 209 210def plain_content(readability_content, content_digests, node_indexes):211    # Load article as DOM212    soup = BeautifulSoup(readability_content, "html.parser")213    # Make all elements plain214    elements = plain_elements(soup.contents, content_digests, node_indexes)215    if node_indexes:216        # Add node index attributes to nodes217        elements = [add_node_indexes(element) for element in elements]218    # Replace article contents with plain elements219    soup.contents = elements220    return str(soup)221 222 223def plain_elements(elements, content_digests, node_indexes):224    # Get plain content versions of all elements225    elements = [plain_element(element, content_digests, node_indexes) for element in elements]226    if content_digests:227        # Add content digest attribute to nodes228        elements = [add_content_digest(element) for element in elements]229    return elements230 231 232def plain_element(element, content_digests, node_indexes):233    # For lists, we make each item plain text234    if is_leaf(element):235        # For leaf node elements, extract the text content, discarding any HTML tags236        # 1. Get element contents as text237        plain_text = element.get_text()238        # 2. Normalize the extracted text string to a canonical representation239        plain_text = normalize_text(plain_text)240        # 3. Update element content to be plain text241        element.string = plain_text242    elif is_text(element):243        if is_non_printing(element):244            # The simplified HTML may have come from Readability.js so might245            # have non-printing text (e.g. Comment or CData). In this case, we246            # keep the structure, but ensure that the string is empty.247            element = type(element)("")248        else:249            plain_text = element.string250            plain_text = normalize_text(plain_text)251            element = type(element)(plain_text)252    else:253        # If not a leaf node or leaf type call recursively on child nodes, replacing254        element.contents = plain_elements(element.contents, content_digests, node_indexes)255    return element256 257 258def add_node_indexes(element, node_index="0"):259    # Can't add attributes to string types260    if is_text(element):261        return element262    # Add index to current element263    element["data-node-index"] = node_index264    # Add index to child elements265    for local_idx, child in enumerate([c for c in element.contents if not is_text(c)], start=1):266        # Can't add attributes to leaf string types267        child_index = "{stem}.{local}".format(stem=node_index, local=local_idx)268        add_node_indexes(child, node_index=child_index)269    return element270 271 272def normalize_text(text):273    """Normalize unicode and whitespace."""274    # Normalize unicode first to try and standardize whitespace characters as much as possible before normalizing them275    text = strip_control_characters(text)276    text = normalize_unicode(text)277    text = normalize_whitespace(text)278    return text279 280 281def strip_control_characters(text):282    """Strip out unicode control characters which might break the parsing."""283    # Unicode control characters284    #   [Cc]: Other, Control [includes new lines]285    #   [Cf]: Other, Format286    #   [Cn]: Other, Not Assigned287    #   [Co]: Other, Private Use288    #   [Cs]: Other, Surrogate289    control_chars = {"Cc", "Cf", "Cn", "Co", "Cs"}290    retained_chars = ["\t", "\n", "\r", "\f"]291 292    # Remove non-printing control characters293    return "".join(294        [295            "" if (unicodedata.category(char) in control_chars) and (char not in retained_chars) else char296            for char in text297        ]298    )299 300 301def normalize_unicode(text):302    """Normalize unicode such that things that are visually equivalent map to the same unicode string where possible."""303    normal_form = "NFKC"304    text = unicodedata.normalize(normal_form, text)305    return text306 307 308def normalize_whitespace(text):309    """Replace runs of whitespace characters with a single space as this is what happens when HTML text is displayed."""310    text = regex.sub(r"\s+", " ", text)311    # Remove leading and trailing whitespace312    text = text.strip()313    return text314 315 316def is_leaf(element):317    return element.name in {"p", "li"}318 319 320def is_text(element):321    return isinstance(element, NavigableString)322 323 324def is_non_printing(element):325    return any(isinstance(element, _e) for _e in [Comment, CData])326 327 328def add_content_digest(element):329    if not is_text(element):330        element["data-content-digest"] = content_digest(element)331    return element332 333 334def content_digest(element):335    if is_text(element):336        # Hash337        trimmed_string = element.string.strip()338        if trimmed_string == "":339            digest = ""340        else:341            digest = hashlib.sha256(trimmed_string.encode("utf-8")).hexdigest()342    else:343        contents = element.contents344        num_contents = len(contents)345        if num_contents == 0:346            # No hash when no child elements exist347            digest = ""348        elif num_contents == 1:349            # If single child, use digest of child350            digest = content_digest(contents[0])351        else:352            # Build content digest from the "non-empty" digests of child nodes353            digest = hashlib.sha256()354            child_digests = list(filter(lambda x: x != "", [content_digest(content) for content in contents]))355            for child in child_digests:356                digest.update(child.encode("utf-8"))357            digest = digest.hexdigest()358    return digest359