codekingpro/portable-devtools
114k
1import logging2import re3import xml.etree.cElementTree # OK: user-must-opt-in4from io import BytesIO5from typing import List, Optional, Sequence6from xml.etree.ElementTree import ElementTree # OK: user-must-opt-in7 8from langchain_core.documents import Document9 10from langchain_community.document_loaders.base import BaseLoader11 12logger = logging.getLogger(__name__)13 14_MAXIMUM_TITLE_LENGTH = 6415 16 17class QuipLoader(BaseLoader):18 """Load `Quip` pages.19 20 Port of https://github.com/quip/quip-api/tree/master/samples/baqup21 """22 23 def __init__(24 self,25 api_url: str,26 access_token: str,27 request_timeout: Optional[int] = 60,28 *,29 allow_dangerous_xml_parsing: bool = False,30 ):31 """32 Args:33 api_url: https://platform.quip.com34 access_token: token of access quip API. Please refer:35 https://quip.com/dev/automation/documentation/current#section/Authentication/Get-Access-to-Quip's-APIs36 request_timeout: timeout of request, default 60s.37 allow_dangerous_xml_parsing: Allow dangerous XML parsing, defaults to False38 """39 try:40 from quip_api.quip import QuipClient41 except ImportError:42 raise ImportError(43 "`quip_api` package not found, please run `pip install quip_api`"44 )45 46 self.quip_client = QuipClient(47 access_token=access_token, base_url=api_url, request_timeout=request_timeout48 )49 50 if not allow_dangerous_xml_parsing:51 raise ValueError(52 "The quip client uses the built-in XML parser which may cause"53 "security issues when parsing XML data in some cases. "54 "Please see "55 "https://docs.python.org/3/library/xml.html#xml-vulnerabilities "56 "For more information, set `allow_dangerous_xml_parsing` as True "57 "if you are sure that your distribution of the standard library "58 "is not vulnerable to XML vulnerabilities."59 )60 61 def load(62 self,63 folder_ids: Optional[List[str]] = None,64 thread_ids: Optional[List[str]] = None,65 max_docs: Optional[int] = 1000,66 include_all_folders: bool = False,67 include_comments: bool = False,68 include_images: bool = False,69 ) -> List[Document]:70 """71 Args:72 :param folder_ids: List of specific folder IDs to load, defaults to None73 :param thread_ids: List of specific thread IDs to load, defaults to None74 :param max_docs: Maximum number of docs to retrieve in total, defaults 100075 :param include_all_folders: Include all folders that your access_token76 can access, but doesn't include your private folder77 :param include_comments: Include comments, defaults to False78 :param include_images: Include images, defaults to False79 """80 if not folder_ids and not thread_ids and not include_all_folders:81 raise ValueError(82 "Must specify at least one among `folder_ids`, `thread_ids` "83 "or set `include_all`_folders as True"84 )85 86 thread_ids = thread_ids or []87 88 if folder_ids:89 for folder_id in folder_ids:90 self.get_thread_ids_by_folder_id(folder_id, 0, thread_ids)91 92 if include_all_folders:93 user = self.quip_client.get_authenticated_user()94 if "group_folder_ids" in user:95 self.get_thread_ids_by_folder_id(96 user["group_folder_ids"], 0, thread_ids97 )98 if "shared_folder_ids" in user:99 self.get_thread_ids_by_folder_id(100 user["shared_folder_ids"], 0, thread_ids101 )102 103 thread_ids = list(set(thread_ids[:max_docs]))104 return self.process_threads(thread_ids, include_images, include_comments)105 106 def get_thread_ids_by_folder_id(107 self, folder_id: str, depth: int, thread_ids: List[str]108 ) -> None:109 """Get thread ids by folder id and update in thread_ids"""110 from quip_api.quip import HTTPError, QuipError111 112 try:113 folder = self.quip_client.get_folder(folder_id)114 except QuipError as e:115 if e.code == 403:116 logging.warning(117 f"depth {depth}, Skipped over restricted folder {folder_id}, {e}"118 )119 else:120 logging.warning(121 f"depth {depth}, Skipped over folder {folder_id} "122 f"due to unknown error {e.code}"123 )124 return125 except HTTPError as e:126 logging.warning(127 f"depth {depth}, Skipped over folder {folder_id} "128 f"due to HTTP error {e.code}"129 )130 return131 132 title = folder["folder"].get("title", "Folder %s" % folder_id)133 134 logging.info(f"depth {depth}, Processing folder {title}")135 for child in folder["children"]:136 if "folder_id" in child:137 self.get_thread_ids_by_folder_id(138 child["folder_id"], depth + 1, thread_ids139 )140 elif "thread_id" in child:141 thread_ids.append(child["thread_id"])142 143 def process_threads(144 self, thread_ids: Sequence[str], include_images: bool, include_messages: bool145 ) -> List[Document]:146 """Process a list of thread into a list of documents."""147 docs = []148 for thread_id in thread_ids:149 doc = self.process_thread(thread_id, include_images, include_messages)150 if doc is not None:151 docs.append(doc)152 return docs153 154 def process_thread(155 self, thread_id: str, include_images: bool, include_messages: bool156 ) -> Optional[Document]:157 thread = self.quip_client.get_thread(thread_id)158 thread_id = thread["thread"]["id"]159 title = thread["thread"]["title"]160 link = thread["thread"]["link"]161 update_ts = thread["thread"]["updated_usec"]162 sanitized_title = QuipLoader._sanitize_title(title)163 164 logger.info(165 f"processing thread {thread_id} title {sanitized_title} "166 f"link {link} update_ts {update_ts}"167 )168 169 if "html" in thread:170 # Parse the document171 try:172 tree = self.quip_client.parse_document_html(thread["html"])173 except xml.etree.cElementTree.ParseError as e:174 logger.error(f"Error parsing thread {title} {thread_id}, skipping, {e}")175 return None176 177 metadata = {178 "title": sanitized_title,179 "update_ts": update_ts,180 "id": thread_id,181 "source": link,182 }183 184 # Download each image and replace with the new URL185 text = ""186 if include_images:187 text = self.process_thread_images(tree)188 189 if include_messages:190 text = text + "/n" + self.process_thread_messages(thread_id)191 192 return Document(193 page_content=thread["html"] + text,194 metadata=metadata,195 )196 return None197 198 def process_thread_images(self, tree: ElementTree) -> str:199 text = ""200 201 try:202 from PIL import Image203 from pytesseract import pytesseract204 except ImportError:205 raise ImportError(206 "`Pillow or pytesseract` package not found, "207 "please run "208 "`pip install Pillow` or `pip install pytesseract`"209 )210 211 for img in tree.iter("img"):212 src = img.get("src")213 if not src or not src.startswith("/blob"):214 continue215 _, _, thread_id, blob_id = src.split("/")216 blob_response = self.quip_client.get_blob(thread_id, blob_id)217 try:218 image = Image.open(BytesIO(blob_response.read()))219 text = text + "\n" + pytesseract.image_to_string(image)220 except OSError as e:221 logger.error(f"failed to convert image to text, {e}")222 raise e223 return text224 225 def process_thread_messages(self, thread_id: str) -> str:226 max_created_usec = None227 messages = []228 while True:229 chunk = self.quip_client.get_messages(230 thread_id, max_created_usec=max_created_usec, count=100231 )232 messages.extend(chunk)233 if chunk:234 max_created_usec = chunk[-1]["created_usec"] - 1235 else:236 break237 messages.reverse()238 239 texts = [message["text"] for message in messages]240 241 return "\n".join(texts)242 243 @staticmethod244 def _sanitize_title(title: str) -> str:245 sanitized_title = re.sub(r"\s", " ", title)246 sanitized_title = re.sub(r"(?u)[^- \w.]", "", sanitized_title)247 if len(sanitized_title) > _MAXIMUM_TITLE_LENGTH:248 sanitized_title = sanitized_title[:_MAXIMUM_TITLE_LENGTH]249 return sanitized_title250 