Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
quip.py250 linesDownload Raw Back to document_loaders
1import logging2import re3import xml.etree.cElementTree  # OK: user-must-opt-in4from io import BytesIO5from typing import List, Optional, Sequence6from xml.etree.ElementTree import ElementTree  # OK: user-must-opt-in7 8from langchain_core.documents import Document9 10from langchain_community.document_loaders.base import BaseLoader11 12logger = logging.getLogger(__name__)13 14_MAXIMUM_TITLE_LENGTH = 6415 16 17class QuipLoader(BaseLoader):18    """Load `Quip` pages.19 20    Port of https://github.com/quip/quip-api/tree/master/samples/baqup21    """22 23    def __init__(24        self,25        api_url: str,26        access_token: str,27        request_timeout: Optional[int] = 60,28        *,29        allow_dangerous_xml_parsing: bool = False,30    ):31        """32        Args:33            api_url: https://platform.quip.com34            access_token: token of access quip API. Please refer:35                https://quip.com/dev/automation/documentation/current#section/Authentication/Get-Access-to-Quip's-APIs36            request_timeout: timeout of request, default 60s.37            allow_dangerous_xml_parsing: Allow dangerous XML parsing, defaults to False38        """39        try:40            from quip_api.quip import QuipClient41        except ImportError:42            raise ImportError(43                "`quip_api` package not found, please run `pip install quip_api`"44            )45 46        self.quip_client = QuipClient(47            access_token=access_token, base_url=api_url, request_timeout=request_timeout48        )49 50        if not allow_dangerous_xml_parsing:51            raise ValueError(52                "The quip client uses the built-in XML parser which may cause"53                "security issues when parsing XML data in some cases. "54                "Please see "55                "https://docs.python.org/3/library/xml.html#xml-vulnerabilities "56                "For more information, set `allow_dangerous_xml_parsing` as True "57                "if you are sure that your distribution of the standard library "58                "is not vulnerable to XML vulnerabilities."59            )60 61    def load(62        self,63        folder_ids: Optional[List[str]] = None,64        thread_ids: Optional[List[str]] = None,65        max_docs: Optional[int] = 1000,66        include_all_folders: bool = False,67        include_comments: bool = False,68        include_images: bool = False,69    ) -> List[Document]:70        """71        Args:72            :param folder_ids: List of specific folder IDs to load, defaults to None73            :param thread_ids: List of specific thread IDs to load, defaults to None74            :param max_docs: Maximum number of docs to retrieve in total, defaults 100075            :param include_all_folders: Include all folders that your access_token76                   can access, but doesn't include your private folder77            :param include_comments: Include comments, defaults to False78            :param include_images: Include images, defaults to False79        """80        if not folder_ids and not thread_ids and not include_all_folders:81            raise ValueError(82                "Must specify at least one among `folder_ids`, `thread_ids` "83                "or set `include_all`_folders as True"84            )85 86        thread_ids = thread_ids or []87 88        if folder_ids:89            for folder_id in folder_ids:90                self.get_thread_ids_by_folder_id(folder_id, 0, thread_ids)91 92        if include_all_folders:93            user = self.quip_client.get_authenticated_user()94            if "group_folder_ids" in user:95                self.get_thread_ids_by_folder_id(96                    user["group_folder_ids"], 0, thread_ids97                )98            if "shared_folder_ids" in user:99                self.get_thread_ids_by_folder_id(100                    user["shared_folder_ids"], 0, thread_ids101                )102 103        thread_ids = list(set(thread_ids[:max_docs]))104        return self.process_threads(thread_ids, include_images, include_comments)105 106    def get_thread_ids_by_folder_id(107        self, folder_id: str, depth: int, thread_ids: List[str]108    ) -> None:109        """Get thread ids by folder id and update in thread_ids"""110        from quip_api.quip import HTTPError, QuipError111 112        try:113            folder = self.quip_client.get_folder(folder_id)114        except QuipError as e:115            if e.code == 403:116                logging.warning(117                    f"depth {depth}, Skipped over restricted folder {folder_id}, {e}"118                )119            else:120                logging.warning(121                    f"depth {depth}, Skipped over folder {folder_id} "122                    f"due to unknown error {e.code}"123                )124            return125        except HTTPError as e:126            logging.warning(127                f"depth {depth}, Skipped over folder {folder_id} "128                f"due to HTTP error {e.code}"129            )130            return131 132        title = folder["folder"].get("title", "Folder %s" % folder_id)133 134        logging.info(f"depth {depth}, Processing folder {title}")135        for child in folder["children"]:136            if "folder_id" in child:137                self.get_thread_ids_by_folder_id(138                    child["folder_id"], depth + 1, thread_ids139                )140            elif "thread_id" in child:141                thread_ids.append(child["thread_id"])142 143    def process_threads(144        self, thread_ids: Sequence[str], include_images: bool, include_messages: bool145    ) -> List[Document]:146        """Process a list of thread into a list of documents."""147        docs = []148        for thread_id in thread_ids:149            doc = self.process_thread(thread_id, include_images, include_messages)150            if doc is not None:151                docs.append(doc)152        return docs153 154    def process_thread(155        self, thread_id: str, include_images: bool, include_messages: bool156    ) -> Optional[Document]:157        thread = self.quip_client.get_thread(thread_id)158        thread_id = thread["thread"]["id"]159        title = thread["thread"]["title"]160        link = thread["thread"]["link"]161        update_ts = thread["thread"]["updated_usec"]162        sanitized_title = QuipLoader._sanitize_title(title)163 164        logger.info(165            f"processing thread {thread_id} title {sanitized_title} "166            f"link {link} update_ts {update_ts}"167        )168 169        if "html" in thread:170            # Parse the document171            try:172                tree = self.quip_client.parse_document_html(thread["html"])173            except xml.etree.cElementTree.ParseError as e:174                logger.error(f"Error parsing thread {title} {thread_id}, skipping, {e}")175                return None176 177            metadata = {178                "title": sanitized_title,179                "update_ts": update_ts,180                "id": thread_id,181                "source": link,182            }183 184            # Download each image and replace with the new URL185            text = ""186            if include_images:187                text = self.process_thread_images(tree)188 189            if include_messages:190                text = text + "/n" + self.process_thread_messages(thread_id)191 192            return Document(193                page_content=thread["html"] + text,194                metadata=metadata,195            )196        return None197 198    def process_thread_images(self, tree: ElementTree) -> str:199        text = ""200 201        try:202            from PIL import Image203            from pytesseract import pytesseract204        except ImportError:205            raise ImportError(206                "`Pillow or pytesseract` package not found, "207                "please run "208                "`pip install Pillow` or `pip install pytesseract`"209            )210 211        for img in tree.iter("img"):212            src = img.get("src")213            if not src or not src.startswith("/blob"):214                continue215            _, _, thread_id, blob_id = src.split("/")216            blob_response = self.quip_client.get_blob(thread_id, blob_id)217            try:218                image = Image.open(BytesIO(blob_response.read()))219                text = text + "\n" + pytesseract.image_to_string(image)220            except OSError as e:221                logger.error(f"failed to convert image to text, {e}")222                raise e223        return text224 225    def process_thread_messages(self, thread_id: str) -> str:226        max_created_usec = None227        messages = []228        while True:229            chunk = self.quip_client.get_messages(230                thread_id, max_created_usec=max_created_usec, count=100231            )232            messages.extend(chunk)233            if chunk:234                max_created_usec = chunk[-1]["created_usec"] - 1235            else:236                break237        messages.reverse()238 239        texts = [message["text"] for message in messages]240 241        return "\n".join(texts)242 243    @staticmethod244    def _sanitize_title(title: str) -> str:245        sanitized_title = re.sub(r"\s", " ", title)246        sanitized_title = re.sub(r"(?u)[^- \w.]", "", sanitized_title)247        if len(sanitized_title) > _MAXIMUM_TITLE_LENGTH:248            sanitized_title = sanitized_title[:_MAXIMUM_TITLE_LENGTH]249        return sanitized_title250 
codekingpro/portable-devtools · Team Ai