Underground-Digital/Workflow-Engine
0
1import hashlib2import json3import mimetypes4import os5import re6import site7import subprocess8import tempfile9import unicodedata10from contextlib import contextmanager11from pathlib import Path12from typing import Optional13from urllib.parse import unquote14 15import chardet16import cloudscraper17from bs4 import BeautifulSoup, CData, Comment, NavigableString18from regex import regex19 20from core.helper import ssrf_proxy21from core.rag.extractor import extract_processor22from core.rag.extractor.extract_processor import ExtractProcessor23 24FULL_TEMPLATE = """25TITLE: {title}26AUTHORS: {authors}27PUBLISH DATE: {publish_date}28TOP_IMAGE_URL: {top_image}29TEXT:30 31{text}32"""33 34 35def page_result(text: str, cursor: int, max_length: int) -> str:36 """Page through `text` and return a substring of `max_length` characters starting from `cursor`."""37 return text[cursor : cursor + max_length]38 39 40def get_url(url: str, user_agent: Optional[str] = None) -> str:41 """Fetch URL and return the contents as a string."""42 headers = {43 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko)"44 " Chrome/91.0.4472.124 Safari/537.36"45 }46 if user_agent:47 headers["User-Agent"] = user_agent48 49 main_content_type = None50 supported_content_types = extract_processor.SUPPORT_URL_CONTENT_TYPES + ["text/html"]51 response = ssrf_proxy.head(url, headers=headers, follow_redirects=True, timeout=(5, 10))52 53 if response.status_code == 200:54 # check content-type55 content_type = response.headers.get("Content-Type")56 if content_type:57 main_content_type = response.headers.get("Content-Type").split(";")[0].strip()58 else:59 content_disposition = response.headers.get("Content-Disposition", "")60 filename_match = re.search(r'filename="([^"]+)"', content_disposition)61 if filename_match:62 filename = unquote(filename_match.group(1))63 extension = re.search(r"\.(\w+)$", filename)64 if extension:65 main_content_type = mimetypes.guess_type(filename)[0]66 67 if main_content_type not in supported_content_types:68 return "Unsupported content-type [{}] of URL.".format(main_content_type)69 70 if main_content_type in extract_processor.SUPPORT_URL_CONTENT_TYPES:71 return ExtractProcessor.load_from_url(url, return_text=True)72 73 response = ssrf_proxy.get(url, headers=headers, follow_redirects=True, timeout=(120, 300))74 elif response.status_code == 403:75 scraper = cloudscraper.create_scraper()76 scraper.perform_request = ssrf_proxy.make_request77 response = scraper.get(url, headers=headers, follow_redirects=True, timeout=(120, 300))78 79 if response.status_code != 200:80 return "URL returned status code {}.".format(response.status_code)81 82 # Detect encoding using chardet83 detected_encoding = chardet.detect(response.content)84 encoding = detected_encoding["encoding"]85 if encoding:86 try:87 content = response.content.decode(encoding)88 except (UnicodeDecodeError, TypeError):89 content = response.text90 else:91 content = response.text92 93 a = extract_using_readabilipy(content)94 95 if not a["plain_text"] or not a["plain_text"].strip():96 return ""97 98 res = FULL_TEMPLATE.format(99 title=a["title"],100 authors=a["byline"],101 publish_date=a["date"],102 top_image="",103 text=a["plain_text"] or "",104 )105 106 return res107 108 109def extract_using_readabilipy(html):110 with tempfile.NamedTemporaryFile(delete=False, mode="w+") as f_html:111 f_html.write(html)112 f_html.close()113 html_path = f_html.name114 115 # Call Mozilla's Readability.js Readability.parse() function via node, writing output to a temporary file116 article_json_path = html_path + ".json"117 jsdir = os.path.join(find_module_path("readabilipy"), "javascript")118 with chdir(jsdir):119 subprocess.check_call(["node", "ExtractArticle.js", "-i", html_path, "-o", article_json_path])120 121 # Read output of call to Readability.parse() from JSON file and return as Python dictionary122 input_json = json.loads(Path(article_json_path).read_text(encoding="utf-8"))123 124 # Deleting files after processing125 os.unlink(article_json_path)126 os.unlink(html_path)127 128 article_json = {129 "title": None,130 "byline": None,131 "date": None,132 "content": None,133 "plain_content": None,134 "plain_text": None,135 }136 # Populate article fields from readability fields where present137 if input_json:138 if input_json.get("title"):139 article_json["title"] = input_json["title"]140 if input_json.get("byline"):141 article_json["byline"] = input_json["byline"]142 if input_json.get("date"):143 article_json["date"] = input_json["date"]144 if input_json.get("content"):145 article_json["content"] = input_json["content"]146 article_json["plain_content"] = plain_content(article_json["content"], False, False)147 article_json["plain_text"] = extract_text_blocks_as_plain_text(article_json["plain_content"])148 if input_json.get("textContent"):149 article_json["plain_text"] = input_json["textContent"]150 article_json["plain_text"] = re.sub(r"\n\s*\n", "\n", article_json["plain_text"])151 152 return article_json153 154 155def find_module_path(module_name):156 for package_path in site.getsitepackages():157 potential_path = os.path.join(package_path, module_name)158 if os.path.exists(potential_path):159 return potential_path160 161 return None162 163 164@contextmanager165def chdir(path):166 """Change directory in context and return to original on exit"""167 # From https://stackoverflow.com/a/37996581, couldn't find a built-in168 original_path = os.getcwd()169 os.chdir(path)170 try:171 yield172 finally:173 os.chdir(original_path)174 175 176def extract_text_blocks_as_plain_text(paragraph_html):177 # Load article as DOM178 soup = BeautifulSoup(paragraph_html, "html.parser")179 # Select all lists180 list_elements = soup.find_all(["ul", "ol"])181 # Prefix text in all list items with "* " and make lists paragraphs182 for list_element in list_elements:183 plain_items = "".join(184 list(filter(None, [plain_text_leaf_node(li)["text"] for li in list_element.find_all("li")]))185 )186 list_element.string = plain_items187 list_element.name = "p"188 # Select all text blocks189 text_blocks = [s.parent for s in soup.find_all(string=True)]190 text_blocks = [plain_text_leaf_node(block) for block in text_blocks]191 # Drop empty paragraphs192 text_blocks = list(filter(lambda p: p["text"] is not None, text_blocks))193 return text_blocks194 195 196def plain_text_leaf_node(element):197 # Extract all text, stripped of any child HTML elements and normalize it198 plain_text = normalize_text(element.get_text())199 if plain_text != "" and element.name == "li":200 plain_text = "* {}, ".format(plain_text)201 if plain_text == "":202 plain_text = None203 if "data-node-index" in element.attrs:204 plain = {"node_index": element["data-node-index"], "text": plain_text}205 else:206 plain = {"text": plain_text}207 return plain208 209 210def plain_content(readability_content, content_digests, node_indexes):211 # Load article as DOM212 soup = BeautifulSoup(readability_content, "html.parser")213 # Make all elements plain214 elements = plain_elements(soup.contents, content_digests, node_indexes)215 if node_indexes:216 # Add node index attributes to nodes217 elements = [add_node_indexes(element) for element in elements]218 # Replace article contents with plain elements219 soup.contents = elements220 return str(soup)221 222 223def plain_elements(elements, content_digests, node_indexes):224 # Get plain content versions of all elements225 elements = [plain_element(element, content_digests, node_indexes) for element in elements]226 if content_digests:227 # Add content digest attribute to nodes228 elements = [add_content_digest(element) for element in elements]229 return elements230 231 232def plain_element(element, content_digests, node_indexes):233 # For lists, we make each item plain text234 if is_leaf(element):235 # For leaf node elements, extract the text content, discarding any HTML tags236 # 1. Get element contents as text237 plain_text = element.get_text()238 # 2. Normalize the extracted text string to a canonical representation239 plain_text = normalize_text(plain_text)240 # 3. Update element content to be plain text241 element.string = plain_text242 elif is_text(element):243 if is_non_printing(element):244 # The simplified HTML may have come from Readability.js so might245 # have non-printing text (e.g. Comment or CData). In this case, we246 # keep the structure, but ensure that the string is empty.247 element = type(element)("")248 else:249 plain_text = element.string250 plain_text = normalize_text(plain_text)251 element = type(element)(plain_text)252 else:253 # If not a leaf node or leaf type call recursively on child nodes, replacing254 element.contents = plain_elements(element.contents, content_digests, node_indexes)255 return element256 257 258def add_node_indexes(element, node_index="0"):259 # Can't add attributes to string types260 if is_text(element):261 return element262 # Add index to current element263 element["data-node-index"] = node_index264 # Add index to child elements265 for local_idx, child in enumerate([c for c in element.contents if not is_text(c)], start=1):266 # Can't add attributes to leaf string types267 child_index = "{stem}.{local}".format(stem=node_index, local=local_idx)268 add_node_indexes(child, node_index=child_index)269 return element270 271 272def normalize_text(text):273 """Normalize unicode and whitespace."""274 # Normalize unicode first to try and standardize whitespace characters as much as possible before normalizing them275 text = strip_control_characters(text)276 text = normalize_unicode(text)277 text = normalize_whitespace(text)278 return text279 280 281def strip_control_characters(text):282 """Strip out unicode control characters which might break the parsing."""283 # Unicode control characters284 # [Cc]: Other, Control [includes new lines]285 # [Cf]: Other, Format286 # [Cn]: Other, Not Assigned287 # [Co]: Other, Private Use288 # [Cs]: Other, Surrogate289 control_chars = {"Cc", "Cf", "Cn", "Co", "Cs"}290 retained_chars = ["\t", "\n", "\r", "\f"]291 292 # Remove non-printing control characters293 return "".join(294 [295 "" if (unicodedata.category(char) in control_chars) and (char not in retained_chars) else char296 for char in text297 ]298 )299 300 301def normalize_unicode(text):302 """Normalize unicode such that things that are visually equivalent map to the same unicode string where possible."""303 normal_form = "NFKC"304 text = unicodedata.normalize(normal_form, text)305 return text306 307 308def normalize_whitespace(text):309 """Replace runs of whitespace characters with a single space as this is what happens when HTML text is displayed."""310 text = regex.sub(r"\s+", " ", text)311 # Remove leading and trailing whitespace312 text = text.strip()313 return text314 315 316def is_leaf(element):317 return element.name in {"p", "li"}318 319 320def is_text(element):321 return isinstance(element, NavigableString)322 323 324def is_non_printing(element):325 return any(isinstance(element, _e) for _e in [Comment, CData])326 327 328def add_content_digest(element):329 if not is_text(element):330 element["data-content-digest"] = content_digest(element)331 return element332 333 334def content_digest(element):335 if is_text(element):336 # Hash337 trimmed_string = element.string.strip()338 if trimmed_string == "":339 digest = ""340 else:341 digest = hashlib.sha256(trimmed_string.encode("utf-8")).hexdigest()342 else:343 contents = element.contents344 num_contents = len(contents)345 if num_contents == 0:346 # No hash when no child elements exist347 digest = ""348 elif num_contents == 1:349 # If single child, use digest of child350 digest = content_digest(contents[0])351 else:352 # Build content digest from the "non-empty" digests of child nodes353 digest = hashlib.sha256()354 child_digests = list(filter(lambda x: x != "", [content_digest(content) for content in contents]))355 for child in child_digests:356 digest.update(child.encode("utf-8"))357 digest = digest.hexdigest()358 return digest359 