Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
confluence.py880 linesDownload Raw Back to document_loaders
1import logging2from enum import Enum3from io import BytesIO4from typing import Any, Callable, Dict, Iterator, List, Optional, Union5 6import requests7from langchain_core.documents import Document8from tenacity import (9    before_sleep_log,10    retry,11    stop_after_attempt,12    wait_exponential,13)14 15from langchain_community.document_loaders.base import BaseLoader16 17logger = logging.getLogger(__name__)18 19 20class ContentFormat(str, Enum):21    """Enumerator of the content formats of Confluence page."""22 23    EDITOR = "body.editor"24    EXPORT_VIEW = "body.export_view"25    ANONYMOUS_EXPORT_VIEW = "body.anonymous_export_view"26    STORAGE = "body.storage"27    VIEW = "body.view"28 29    def get_content(self, page: dict) -> str:30        return page["body"][self.name.lower()]["value"]31 32 33class ConfluenceLoader(BaseLoader):34    """Load `Confluence` pages.35 36    Port of https://llamahub.ai/l/confluence37    This currently supports username/api_key, Oauth2 login, personal access token38    or cookies authentication.39 40    Specify a list page_ids and/or space_key to load in the corresponding pages into41    Document objects, if both are specified the union of both sets will be returned.42 43    You can also specify a boolean `include_attachments` to include attachments, this44    is set to False by default, if set to True all attachments will be downloaded and45    ConfluenceLoader will extract the text from the attachments and add it to the46    Document object. Currently supported attachment types are: PDF, PNG, JPEG/JPG,47    SVG, Word and Excel.48 49    Confluence API supports difference format of page content. The storage format is the50    raw XML representation for storage. The view format is the HTML representation for51    viewing with macros are rendered as though it is viewed by users. You can pass52    a enum `content_format` argument to specify the content format, this is53    set to `ContentFormat.STORAGE` by default, the supported values are:54    `ContentFormat.EDITOR`, `ContentFormat.EXPORT_VIEW`,55    `ContentFormat.ANONYMOUS_EXPORT_VIEW`, `ContentFormat.STORAGE`,56    and `ContentFormat.VIEW`.57 58    Hint: space_key and page_id can both be found in the URL of a page in Confluence59    - https://yoursite.atlassian.com/wiki/spaces/<space_key>/pages/<page_id>60 61    Example:62        .. code-block:: python63 64            from langchain_community.document_loaders import ConfluenceLoader65 66            loader = ConfluenceLoader(67                url="https://yoursite.atlassian.com/wiki",68                username="me",69                api_key="12345",70                space_key="SPACE",71                limit=50,72            )73            documents = loader.load()74 75            # Server on perm76            loader = ConfluenceLoader(77                url="https://confluence.yoursite.com/",78                username="me",79                api_key="your_password",80                cloud=False,81                space_key="SPACE",82                limit=50,83            )84            documents = loader.load()85 86    :param url: _description_87    :type url: str88    :param api_key: _description_, defaults to None89    :type api_key: str, optional90    :param username: _description_, defaults to None91    :type username: str, optional92    :param oauth2: _description_, defaults to {}93    :type oauth2: dict, optional94    :param token: _description_, defaults to None95    :type token: str, optional96    :param cloud: _description_, defaults to True97    :type cloud: bool, optional98    :param number_of_retries: How many times to retry, defaults to 399    :type number_of_retries: Optional[int], optional100    :param min_retry_seconds: defaults to 2101    :type min_retry_seconds: Optional[int], optional102    :param max_retry_seconds:  defaults to 10103    :type max_retry_seconds: Optional[int], optional104    :param confluence_kwargs: additional kwargs to initialize confluence with105    :type confluence_kwargs: dict, optional106    :param cookies: _description_, defaults to {}107    :type cookies: dict, optional108    :param space_key: Space key retrieved from a confluence URL, defaults to None109    :type space_key: Optional[str], optional110    :param page_ids: List of specific page IDs to load, defaults to None111    :type page_ids: Optional[List[str]], optional112    :param label: Get all pages with this label, defaults to None113    :type label: Optional[str], optional114    :param cql: CQL Expression, defaults to None115    :type cql: Optional[str], optional116    :param include_restricted_content: defaults to False117    :type include_restricted_content: bool, optional118    :param include_archived_content: Whether to include archived content,119                                     defaults to False120    :type include_archived_content: bool, optional121    :param include_attachments: defaults to False122    :type include_attachments: bool, optional123    :param attachment_filter_func: A function that takes the attachment information124                                   from Confluence and decides whether or not the125                                   attachment is processed.126    :param include_comments: defaults to False127    :type include_comments: bool, optional128    :param content_format: Specify content format, defaults to129                            ContentFormat.STORAGE, the supported values are:130                            `ContentFormat.EDITOR`, `ContentFormat.EXPORT_VIEW`,131                            `ContentFormat.ANONYMOUS_EXPORT_VIEW`,132                            `ContentFormat.STORAGE`, and `ContentFormat.VIEW`.133    :type content_format: ContentFormat134    :param limit: Maximum number of pages to retrieve per request, defaults to 50135    :type limit: int, optional136    :param max_pages: Maximum number of pages to retrieve in total, defaults 1000137    :type max_pages: int, optional138    :param ocr_languages: The languages to use for the Tesseract agent. To use a139                          language, you'll first need to install the appropriate140                          Tesseract language pack.141    :type ocr_languages: str, optional142    :param keep_markdown_format: Whether to keep the markdown format, defaults to143        False144    :type keep_markdown_format: bool145    :param keep_newlines: Whether to keep the newlines format, defaults to146        False147    :type keep_newlines: bool148    :raises ValueError: Errors while validating input149    :raises ImportError: Required dependencies not installed.150    """151 152    def __init__(153        self,154        url: str,155        api_key: Optional[str] = None,156        username: Optional[str] = None,157        session: Optional[requests.Session] = None,158        oauth2: Optional[dict] = None,159        token: Optional[str] = None,160        cloud: Optional[bool] = True,161        number_of_retries: Optional[int] = 3,162        min_retry_seconds: Optional[int] = 2,163        max_retry_seconds: Optional[int] = 10,164        confluence_kwargs: Optional[dict] = None,165        *,166        cookies: Optional[dict] = None,167        space_key: Optional[str] = None,168        page_ids: Optional[List[str]] = None,169        label: Optional[str] = None,170        cql: Optional[str] = None,171        include_restricted_content: bool = False,172        include_archived_content: bool = False,173        include_attachments: bool = False,174        include_comments: bool = False,175        include_labels: bool = False,176        content_format: ContentFormat = ContentFormat.STORAGE,177        limit: Optional[int] = 50,178        max_pages: Optional[int] = 1000,179        ocr_languages: Optional[str] = None,180        keep_markdown_format: bool = False,181        keep_newlines: bool = False,182        attachment_filter_func: Optional[Callable[[dict], bool]] = None,183    ):184        self.space_key = space_key185        self.page_ids = page_ids186        self.label = label187        self.cql = cql188        self.include_restricted_content = include_restricted_content189        self.include_archived_content = include_archived_content190        self.include_attachments = include_attachments191        self.include_comments = include_comments192        self.include_labels = include_labels193        self.content_format = content_format194        self.limit = limit195        self.max_pages = max_pages196        self.ocr_languages = ocr_languages197        self.keep_markdown_format = keep_markdown_format198        self.keep_newlines = keep_newlines199        self.attachment_filter_func = attachment_filter_func200 201        confluence_kwargs = confluence_kwargs or {}202        errors = ConfluenceLoader.validate_init_args(203            url=url,204            api_key=api_key,205            username=username,206            session=session,207            oauth2=oauth2,208            cookies=cookies,209            token=token,210        )211        if errors:212            raise ValueError(f"Error(s) while validating input: {errors}")213        try:214            from atlassian import Confluence215        except ImportError:216            raise ImportError(217                "`atlassian` package not found, please run "218                "`pip install atlassian-python-api`"219            )220 221        self.base_url = url222        self.number_of_retries = number_of_retries223        self.min_retry_seconds = min_retry_seconds224        self.max_retry_seconds = max_retry_seconds225 226        if session:227            self.confluence = Confluence(url=url, session=session, **confluence_kwargs)228        elif oauth2:229            self.confluence = Confluence(230                url=url, oauth2=oauth2, cloud=cloud, **confluence_kwargs231            )232        elif token:233            self.confluence = Confluence(234                url=url, token=token, cloud=cloud, **confluence_kwargs235            )236        elif cookies:237            self.confluence = Confluence(238                url=url, cookies=cookies, cloud=cloud, **confluence_kwargs239            )240        else:241            self.confluence = Confluence(242                url=url,243                username=username,244                password=api_key,245                cloud=cloud,246                **confluence_kwargs,247            )248 249    @staticmethod250    def validate_init_args(251        url: Optional[str] = None,252        api_key: Optional[str] = None,253        username: Optional[str] = None,254        session: Optional[requests.Session] = None,255        oauth2: Optional[dict] = None,256        token: Optional[str] = None,257        cookies: Optional[dict] = None,258    ) -> Union[List, None]:259        """Validates proper combinations of init arguments"""260 261        errors = []262        if url is None:263            errors.append("Must provide `base_url`")264 265        if (api_key and not username) or (username and not api_key):266            errors.append(267                "If one of `api_key` or `username` is provided, "268                "the other must be as well."269            )270 271        non_null_creds = list(272            x is not None273            for x in ((api_key or username), session, oauth2, token, cookies)274        )275        if sum(non_null_creds) > 1:276            all_names = ("(api_key, username)", "session", "oauth2", "token", "cookies")277            provided = tuple(n for x, n in zip(non_null_creds, all_names) if x)278            errors.append(279                f"Cannot provide a value for more than one of: {all_names}. Received "280                f"values for: {provided}"281            )282 283        if (284            oauth2285            and set(oauth2.keys())286            == {287                "token",288                "client_id",289            }290            and set(oauth2["token"].keys())291            != {292                "access_token",293                "token_type",294            }295        ):296            # OAuth2 token authentication297            errors.append(298                "You have either omitted require keys or added extra "299                "keys to the oauth2 dictionary. key values should be "300                "`['client_id', 'token': ['access_token', 'token_type']]`"301            )302 303        if (304            oauth2305            and set(oauth2.keys())306            != {307                "access_token",308                "access_token_secret",309                "consumer_key",310                "key_cert",311            }312            and set(oauth2.keys())313            != {314                "token",315                "client_id",316            }317        ):318            errors.append(319                "You have either omitted required keys or added extra "320                "keys to the oauth2 dictionary. key values should be "321                "`['access_token', 'access_token_secret', 'consumer_key', 'key_cert']` "322                "or `['client_id', 'token': ['access_token', 'token_type']]`"323            )324        return errors or None325 326    def _resolve_param(self, param_name: str, kwargs: Any) -> Any:327        return kwargs[param_name] if param_name in kwargs else getattr(self, param_name)328 329    def _lazy_load(self, **kwargs: Any) -> Iterator[Document]:330        if kwargs:331            logger.warning(332                f"Received runtime arguments {kwargs}. Passing runtime args to `load`"333                f" is deprecated. Please pass arguments during initialization instead."334            )335        space_key = self._resolve_param("space_key", kwargs)336        page_ids = self._resolve_param("page_ids", kwargs)337        label = self._resolve_param("label", kwargs)338        cql = self._resolve_param("cql", kwargs)339        include_restricted_content = self._resolve_param(340            "include_restricted_content", kwargs341        )342        include_archived_content = self._resolve_param(343            "include_archived_content", kwargs344        )345        include_attachments = self._resolve_param("include_attachments", kwargs)346        include_comments = self._resolve_param("include_comments", kwargs)347        include_labels = self._resolve_param("include_labels", kwargs)348        content_format = self._resolve_param("content_format", kwargs)349        limit = self._resolve_param("limit", kwargs)350        max_pages = self._resolve_param("max_pages", kwargs)351        ocr_languages = self._resolve_param("ocr_languages", kwargs)352        keep_markdown_format = self._resolve_param("keep_markdown_format", kwargs)353        keep_newlines = self._resolve_param("keep_newlines", kwargs)354        expand = ",".join(355            [356                content_format.value,357                "version",358                *(["metadata.labels"] if include_labels else []),359            ]360        )361 362        if not space_key and not page_ids and not label and not cql:363            raise ValueError(364                "Must specify at least one among `space_key`, `page_ids`, "365                "`label`, `cql` parameters."366            )367 368        if space_key:369            pages = self.paginate_request(370                self.confluence.get_all_pages_from_space,371                space=space_key,372                limit=limit,373                max_pages=max_pages,374                status="any" if include_archived_content else "current",375                expand=expand,376            )377            yield from self.process_pages(378                pages,379                include_restricted_content,380                include_attachments,381                include_comments,382                include_labels,383                content_format,384                ocr_languages=ocr_languages,385                keep_markdown_format=keep_markdown_format,386                keep_newlines=keep_newlines,387            )388 389        if label:390            pages = self.paginate_request(391                self.confluence.get_all_pages_by_label,392                label=label,393                limit=limit,394                max_pages=max_pages,395            )396            ids_by_label = [page["id"] for page in pages]397            if page_ids:398                page_ids = list(set(page_ids + ids_by_label))399            else:400                page_ids = list(set(ids_by_label))401 402        if cql:403            pages = self.paginate_request(404                self._search_content_by_cql,405                cql=cql,406                limit=limit,407                max_pages=max_pages,408                include_archived_spaces=include_archived_content,409                expand=expand,410            )411            yield from self.process_pages(412                pages,413                include_restricted_content,414                include_attachments,415                include_comments,416                include_labels,417                content_format,418                ocr_languages,419                keep_markdown_format,420                keep_newlines=keep_newlines,421            )422 423        if page_ids:424            for page_id in page_ids:425                get_page = retry(426                    reraise=True,427                    stop=stop_after_attempt(428                        self.number_of_retries  # type: ignore[arg-type]429                    ),430                    wait=wait_exponential(431                        multiplier=1,432                        min=self.min_retry_seconds,  # type: ignore[arg-type]433                        max=self.max_retry_seconds,  # type: ignore[arg-type]434                    ),435                    before_sleep=before_sleep_log(logger, logging.WARNING),436                )(self.confluence.get_page_by_id)437                page = get_page(438                    page_id=page_id,439                    expand=expand,440                )441                if not include_restricted_content and not self.is_public_page(page):442                    continue443                yield self.process_page(444                    page,445                    include_attachments,446                    include_comments,447                    include_labels,448                    content_format,449                    ocr_languages,450                    keep_markdown_format,451                    keep_newlines=keep_newlines,452                )453 454    def load(self, **kwargs: Any) -> List[Document]:455        return list(self._lazy_load(**kwargs))456 457    def lazy_load(self) -> Iterator[Document]:458        yield from self._lazy_load()459 460    def _search_content_by_cql(461        self,462        cql: str,463        include_archived_spaces: Optional[bool] = None,464        next_url: str = "",465        **kwargs: Any,466    ) -> tuple[List[dict], str]:467        if next_url:468            response = self.confluence.get(next_url)469        else:470            url = "rest/api/content/search"471 472            params: Dict[str, Any] = {"cql": cql}473            params.update(kwargs)474            if include_archived_spaces is not None:475                params["includeArchivedSpaces"] = include_archived_spaces476 477            response = self.confluence.get(url, params=params)478 479        return response.get("results", []), response.get("_links", {}).get("next", "")480 481    def paginate_request(self, retrieval_method: Callable, **kwargs: Any) -> List:482        """Paginate the various methods to retrieve groups of pages.483 484        Unfortunately, due to page size, sometimes the Confluence API485        doesn't match the limit value. If `limit` is >100 confluence486        seems to cap the response to 100. Also, due to the Atlassian Python487        package, we don't get the "next" values from the "_links" key because488        they only return the value from the result key. So here, the pagination489        starts from 0 and goes until the max_pages, getting the `limit` number490        of pages with each request. We have to manually check if there491        are more docs based on the length of the returned list of pages, rather than492        just checking for the presence of a `next` key in the response like this page493        would have you do:494        https://developer.atlassian.com/server/confluence/pagination-in-the-rest-api/495 496        :param retrieval_method: Function used to retrieve docs497        :type retrieval_method: callable498        :return: List of documents499        :rtype: List500        """501 502        max_pages = kwargs.pop("max_pages")503        docs: List[dict] = []504        next_url: str = ""505        while len(docs) < max_pages:506            get_pages = retry(507                reraise=True,508                stop=stop_after_attempt(509                    self.number_of_retries  # type: ignore[arg-type]510                ),511                wait=wait_exponential(512                    multiplier=1,513                    min=self.min_retry_seconds,  # type: ignore[arg-type]514                    max=self.max_retry_seconds,  # type: ignore[arg-type]515                ),516                before_sleep=before_sleep_log(logger, logging.WARNING),517            )(retrieval_method)518            if self.cql:  # cursor pagination for CQL519                batch, next_url = get_pages(**kwargs, next_url=next_url)520                if not next_url:521                    docs.extend(batch)522                    break523            else:524                batch = get_pages(**kwargs, start=len(docs))525                if not batch:526                    break527            docs.extend(batch)528        return docs[:max_pages]529 530    def is_public_page(self, page: dict) -> bool:531        """Check if a page is publicly accessible."""532 533        if page["status"] != "current":534            return False535 536        restrictions = self.confluence.get_all_restrictions_for_content(page["id"])537 538        return (539            not restrictions["read"]["restrictions"]["user"]["results"]540            and not restrictions["read"]["restrictions"]["group"]["results"]541        )542 543    def process_pages(544        self,545        pages: List[dict],546        include_restricted_content: bool,547        include_attachments: bool,548        include_comments: bool,549        include_labels: bool,550        content_format: ContentFormat,551        ocr_languages: Optional[str] = None,552        keep_markdown_format: Optional[bool] = False,553        keep_newlines: bool = False,554    ) -> Iterator[Document]:555        """Process a list of pages into a list of documents."""556        for page in pages:557            if not include_restricted_content and not self.is_public_page(page):558                continue559            yield self.process_page(560                page,561                include_attachments,562                include_comments,563                include_labels,564                content_format,565                ocr_languages=ocr_languages,566                keep_markdown_format=keep_markdown_format,567                keep_newlines=keep_newlines,568            )569 570    def process_page(571        self,572        page: dict,573        include_attachments: bool,574        include_comments: bool,575        include_labels: bool,576        content_format: ContentFormat,577        ocr_languages: Optional[str] = None,578        keep_markdown_format: Optional[bool] = False,579        keep_newlines: bool = False,580    ) -> Document:581        if keep_markdown_format:582            try:583                from markdownify import markdownify584            except ImportError:585                raise ImportError(586                    "`markdownify` package not found, please run "587                    "`pip install markdownify`"588                )589        if include_comments or not keep_markdown_format:590            try:591                from bs4 import BeautifulSoup592            except ImportError:593                raise ImportError(594                    "`beautifulsoup4` package not found, please run "595                    "`pip install beautifulsoup4`"596                )597        if include_attachments:598            attachment_texts = self.process_attachment(page["id"], ocr_languages)599        else:600            attachment_texts = []601 602        content = content_format.get_content(page)603        if keep_markdown_format:604            # Use markdownify to keep the page Markdown style605            text = markdownify(content, heading_style="ATX") + "".join(attachment_texts)606 607        else:608            if keep_newlines:609                text = BeautifulSoup(610                    content.replace("</p>", "\n</p>").replace("<br />", "\n"), "lxml"611                ).get_text(" ") + "".join(attachment_texts)612            else:613                text = BeautifulSoup(content, "lxml").get_text(614                    " ", strip=True615                ) + "".join(attachment_texts)616 617        if include_comments:618            comments = self.confluence.get_page_comments(619                page["id"], expand="body.view.value", depth="all"620            )["results"]621            comment_texts = [622                BeautifulSoup(comment["body"]["view"]["value"], "lxml").get_text(623                    " ", strip=True624                )625                for comment in comments626            ]627            text = text + "".join(comment_texts)628 629        if include_labels:630            labels = [631                label["name"]632                for label in page.get("metadata", {})633                .get("labels", {})634                .get("results", [])635            ]636 637        metadata = {638            "title": page["title"],639            "id": page["id"],640            "source": self.base_url.strip("/") + page["_links"]["webui"],641            **({"labels": labels} if include_labels else {}),642        }643 644        if "version" in page and "when" in page["version"]:645            metadata["when"] = page["version"]["when"]646 647        return Document(648            page_content=text,649            metadata=metadata,650        )651 652    def process_attachment(653        self,654        page_id: str,655        ocr_languages: Optional[str] = None,656    ) -> List[str]:657        try:658            from PIL import Image  # noqa: F401659        except ImportError:660            raise ImportError(661                "`Pillow` package not found, please run `pip install Pillow`"662            )663 664        # depending on setup you may also need to set the correct path for665        # poppler and tesseract666        attachments = self.confluence.get_attachments_from_content(page_id)["results"]667        texts = []668        for attachment in attachments:669            if self.attachment_filter_func and not self.attachment_filter_func(670                attachment671            ):672                continue673 674            media_type = attachment["metadata"]["mediaType"]675            absolute_url = self.base_url + attachment["_links"]["download"]676            title = attachment["title"]677            try:678                if media_type == "application/pdf":679                    text = title + self.process_pdf(absolute_url, ocr_languages)680                elif (681                    media_type == "image/png"682                    or media_type == "image/jpg"683                    or media_type == "image/jpeg"684                ):685                    text = title + self.process_image(absolute_url, ocr_languages)686                elif (687                    media_type == "application/vnd.openxmlformats-officedocument"688                    ".wordprocessingml.document"689                ):690                    text = title + self.process_doc(absolute_url)691                elif media_type == "application/vnd.ms-excel":692                    text = title + self.process_xls(absolute_url)693                elif media_type == "image/svg+xml":694                    text = title + self.process_svg(absolute_url, ocr_languages)695                else:696                    continue697                texts.append(text)698            except requests.HTTPError as e:699                if e.response.status_code == 404:700                    print(f"Attachment not found at {absolute_url}")  # noqa: T201701                    continue702                else:703                    raise704 705        return texts706 707    def process_pdf(708        self,709        link: str,710        ocr_languages: Optional[str] = None,711    ) -> str:712        try:713            import pytesseract714            from pdf2image import convert_from_bytes715        except ImportError:716            raise ImportError(717                "`pytesseract` or `pdf2image` package not found, "718                "please run `pip install pytesseract pdf2image`"719            )720 721        response = self.confluence.request(path=link, absolute=True)722        text = ""723 724        if (725            response.status_code != 200726            or response.content == b""727            or response.content is None728        ):729            return text730        try:731            images = convert_from_bytes(response.content)732        except ValueError:733            return text734 735        for i, image in enumerate(images):736            try:737                image_text = pytesseract.image_to_string(image, lang=ocr_languages)738                text += f"Page {i + 1}:\n{image_text}\n\n"739            except pytesseract.TesseractError as ex:740                logger.warning(f"TesseractError: {ex}")741 742        return text743 744    def process_image(745        self,746        link: str,747        ocr_languages: Optional[str] = None,748    ) -> str:749        try:750            import pytesseract751            from PIL import Image752        except ImportError:753            raise ImportError(754                "`pytesseract` or `Pillow` package not found, "755                "please run `pip install pytesseract Pillow`"756            )757 758        response = self.confluence.request(path=link, absolute=True)759        text = ""760 761        if (762            response.status_code != 200763            or response.content == b""764            or response.content is None765        ):766            return text767        try:768            image = Image.open(BytesIO(response.content))769        except OSError:770            return text771 772        return pytesseract.image_to_string(image, lang=ocr_languages)773 774    def process_doc(self, link: str) -> str:775        try:776            import docx2txt777        except ImportError:778            raise ImportError(779                "`docx2txt` package not found, please run `pip install docx2txt`"780            )781 782        response = self.confluence.request(path=link, absolute=True)783        text = ""784 785        if (786            response.status_code != 200787            or response.content == b""788            or response.content is None789        ):790            return text791        file_data = BytesIO(response.content)792 793        return docx2txt.process(file_data)794 795    def process_xls(self, link: str) -> str:796        import io797        import os798 799        try:800            import xlrd801 802        except ImportError:803            raise ImportError("`xlrd` package not found, please run `pip install xlrd`")804 805        try:806            import pandas as pd807 808        except ImportError:809            raise ImportError(810                "`pandas` package not found, please run `pip install pandas`"811            )812 813        response = self.confluence.request(path=link, absolute=True)814        text = ""815 816        if (817            response.status_code != 200818            or response.content == b""819            or response.content is None820        ):821            return text822 823        filename = os.path.basename(link)824        # Getting the whole content of the url after filename,825        # Example: ".csv?version=2&modificationDate=1631800010678&cacheVersion=1&api=v2"826        file_extension = os.path.splitext(filename)[1]827 828        if file_extension.startswith(829            ".csv"830        ):  # if the extension found in the url is ".csv"831            content_string = response.content.decode("utf-8")832            df = pd.read_csv(io.StringIO(content_string))833            text += df.to_string(index=False, header=False) + "\n\n"834        else:835            workbook = xlrd.open_workbook(file_contents=response.content)836            for sheet in workbook.sheets():837                text += f"{sheet.name}:\n"838                for row in range(sheet.nrows):839                    for col in range(sheet.ncols):840                        text += f"{sheet.cell_value(row, col)}\t"841                    text += "\n"842                text += "\n"843 844        return text845 846    def process_svg(847        self,848        link: str,849        ocr_languages: Optional[str] = None,850    ) -> str:851        try:852            import pytesseract853            from PIL import Image854            from reportlab.graphics import renderPM855            from svglib.svglib import svg2rlg856        except ImportError:857            raise ImportError(858                "`pytesseract`, `Pillow`, `reportlab` or `svglib` package not found, "859                "please run `pip install pytesseract Pillow reportlab svglib`"860            )861 862        response = self.confluence.request(path=link, absolute=True)863        text = ""864 865        if (866            response.status_code != 200867            or response.content == b""868            or response.content is None869        ):870            return text871 872        drawing = svg2rlg(BytesIO(response.content))873 874        img_data = BytesIO()875        renderPM.drawToFile(drawing, img_data, fmt="PNG")876        img_data.seek(0)877        image = Image.open(img_data)878 879        return pytesseract.image_to_string(image, lang=ocr_languages)880 
codekingpro/portable-devtools · Team Ai