codekingpro/portable-devtools
114k
1import logging2from enum import Enum3from io import BytesIO4from typing import Any, Callable, Dict, Iterator, List, Optional, Union5 6import requests7from langchain_core.documents import Document8from tenacity import (9 before_sleep_log,10 retry,11 stop_after_attempt,12 wait_exponential,13)14 15from langchain_community.document_loaders.base import BaseLoader16 17logger = logging.getLogger(__name__)18 19 20class ContentFormat(str, Enum):21 """Enumerator of the content formats of Confluence page."""22 23 EDITOR = "body.editor"24 EXPORT_VIEW = "body.export_view"25 ANONYMOUS_EXPORT_VIEW = "body.anonymous_export_view"26 STORAGE = "body.storage"27 VIEW = "body.view"28 29 def get_content(self, page: dict) -> str:30 return page["body"][self.name.lower()]["value"]31 32 33class ConfluenceLoader(BaseLoader):34 """Load `Confluence` pages.35 36 Port of https://llamahub.ai/l/confluence37 This currently supports username/api_key, Oauth2 login, personal access token38 or cookies authentication.39 40 Specify a list page_ids and/or space_key to load in the corresponding pages into41 Document objects, if both are specified the union of both sets will be returned.42 43 You can also specify a boolean `include_attachments` to include attachments, this44 is set to False by default, if set to True all attachments will be downloaded and45 ConfluenceLoader will extract the text from the attachments and add it to the46 Document object. Currently supported attachment types are: PDF, PNG, JPEG/JPG,47 SVG, Word and Excel.48 49 Confluence API supports difference format of page content. The storage format is the50 raw XML representation for storage. The view format is the HTML representation for51 viewing with macros are rendered as though it is viewed by users. You can pass52 a enum `content_format` argument to specify the content format, this is53 set to `ContentFormat.STORAGE` by default, the supported values are:54 `ContentFormat.EDITOR`, `ContentFormat.EXPORT_VIEW`,55 `ContentFormat.ANONYMOUS_EXPORT_VIEW`, `ContentFormat.STORAGE`,56 and `ContentFormat.VIEW`.57 58 Hint: space_key and page_id can both be found in the URL of a page in Confluence59 - https://yoursite.atlassian.com/wiki/spaces/<space_key>/pages/<page_id>60 61 Example:62 .. code-block:: python63 64 from langchain_community.document_loaders import ConfluenceLoader65 66 loader = ConfluenceLoader(67 url="https://yoursite.atlassian.com/wiki",68 username="me",69 api_key="12345",70 space_key="SPACE",71 limit=50,72 )73 documents = loader.load()74 75 # Server on perm76 loader = ConfluenceLoader(77 url="https://confluence.yoursite.com/",78 username="me",79 api_key="your_password",80 cloud=False,81 space_key="SPACE",82 limit=50,83 )84 documents = loader.load()85 86 :param url: _description_87 :type url: str88 :param api_key: _description_, defaults to None89 :type api_key: str, optional90 :param username: _description_, defaults to None91 :type username: str, optional92 :param oauth2: _description_, defaults to {}93 :type oauth2: dict, optional94 :param token: _description_, defaults to None95 :type token: str, optional96 :param cloud: _description_, defaults to True97 :type cloud: bool, optional98 :param number_of_retries: How many times to retry, defaults to 399 :type number_of_retries: Optional[int], optional100 :param min_retry_seconds: defaults to 2101 :type min_retry_seconds: Optional[int], optional102 :param max_retry_seconds: defaults to 10103 :type max_retry_seconds: Optional[int], optional104 :param confluence_kwargs: additional kwargs to initialize confluence with105 :type confluence_kwargs: dict, optional106 :param cookies: _description_, defaults to {}107 :type cookies: dict, optional108 :param space_key: Space key retrieved from a confluence URL, defaults to None109 :type space_key: Optional[str], optional110 :param page_ids: List of specific page IDs to load, defaults to None111 :type page_ids: Optional[List[str]], optional112 :param label: Get all pages with this label, defaults to None113 :type label: Optional[str], optional114 :param cql: CQL Expression, defaults to None115 :type cql: Optional[str], optional116 :param include_restricted_content: defaults to False117 :type include_restricted_content: bool, optional118 :param include_archived_content: Whether to include archived content,119 defaults to False120 :type include_archived_content: bool, optional121 :param include_attachments: defaults to False122 :type include_attachments: bool, optional123 :param attachment_filter_func: A function that takes the attachment information124 from Confluence and decides whether or not the125 attachment is processed.126 :param include_comments: defaults to False127 :type include_comments: bool, optional128 :param content_format: Specify content format, defaults to129 ContentFormat.STORAGE, the supported values are:130 `ContentFormat.EDITOR`, `ContentFormat.EXPORT_VIEW`,131 `ContentFormat.ANONYMOUS_EXPORT_VIEW`,132 `ContentFormat.STORAGE`, and `ContentFormat.VIEW`.133 :type content_format: ContentFormat134 :param limit: Maximum number of pages to retrieve per request, defaults to 50135 :type limit: int, optional136 :param max_pages: Maximum number of pages to retrieve in total, defaults 1000137 :type max_pages: int, optional138 :param ocr_languages: The languages to use for the Tesseract agent. To use a139 language, you'll first need to install the appropriate140 Tesseract language pack.141 :type ocr_languages: str, optional142 :param keep_markdown_format: Whether to keep the markdown format, defaults to143 False144 :type keep_markdown_format: bool145 :param keep_newlines: Whether to keep the newlines format, defaults to146 False147 :type keep_newlines: bool148 :raises ValueError: Errors while validating input149 :raises ImportError: Required dependencies not installed.150 """151 152 def __init__(153 self,154 url: str,155 api_key: Optional[str] = None,156 username: Optional[str] = None,157 session: Optional[requests.Session] = None,158 oauth2: Optional[dict] = None,159 token: Optional[str] = None,160 cloud: Optional[bool] = True,161 number_of_retries: Optional[int] = 3,162 min_retry_seconds: Optional[int] = 2,163 max_retry_seconds: Optional[int] = 10,164 confluence_kwargs: Optional[dict] = None,165 *,166 cookies: Optional[dict] = None,167 space_key: Optional[str] = None,168 page_ids: Optional[List[str]] = None,169 label: Optional[str] = None,170 cql: Optional[str] = None,171 include_restricted_content: bool = False,172 include_archived_content: bool = False,173 include_attachments: bool = False,174 include_comments: bool = False,175 include_labels: bool = False,176 content_format: ContentFormat = ContentFormat.STORAGE,177 limit: Optional[int] = 50,178 max_pages: Optional[int] = 1000,179 ocr_languages: Optional[str] = None,180 keep_markdown_format: bool = False,181 keep_newlines: bool = False,182 attachment_filter_func: Optional[Callable[[dict], bool]] = None,183 ):184 self.space_key = space_key185 self.page_ids = page_ids186 self.label = label187 self.cql = cql188 self.include_restricted_content = include_restricted_content189 self.include_archived_content = include_archived_content190 self.include_attachments = include_attachments191 self.include_comments = include_comments192 self.include_labels = include_labels193 self.content_format = content_format194 self.limit = limit195 self.max_pages = max_pages196 self.ocr_languages = ocr_languages197 self.keep_markdown_format = keep_markdown_format198 self.keep_newlines = keep_newlines199 self.attachment_filter_func = attachment_filter_func200 201 confluence_kwargs = confluence_kwargs or {}202 errors = ConfluenceLoader.validate_init_args(203 url=url,204 api_key=api_key,205 username=username,206 session=session,207 oauth2=oauth2,208 cookies=cookies,209 token=token,210 )211 if errors:212 raise ValueError(f"Error(s) while validating input: {errors}")213 try:214 from atlassian import Confluence215 except ImportError:216 raise ImportError(217 "`atlassian` package not found, please run "218 "`pip install atlassian-python-api`"219 )220 221 self.base_url = url222 self.number_of_retries = number_of_retries223 self.min_retry_seconds = min_retry_seconds224 self.max_retry_seconds = max_retry_seconds225 226 if session:227 self.confluence = Confluence(url=url, session=session, **confluence_kwargs)228 elif oauth2:229 self.confluence = Confluence(230 url=url, oauth2=oauth2, cloud=cloud, **confluence_kwargs231 )232 elif token:233 self.confluence = Confluence(234 url=url, token=token, cloud=cloud, **confluence_kwargs235 )236 elif cookies:237 self.confluence = Confluence(238 url=url, cookies=cookies, cloud=cloud, **confluence_kwargs239 )240 else:241 self.confluence = Confluence(242 url=url,243 username=username,244 password=api_key,245 cloud=cloud,246 **confluence_kwargs,247 )248 249 @staticmethod250 def validate_init_args(251 url: Optional[str] = None,252 api_key: Optional[str] = None,253 username: Optional[str] = None,254 session: Optional[requests.Session] = None,255 oauth2: Optional[dict] = None,256 token: Optional[str] = None,257 cookies: Optional[dict] = None,258 ) -> Union[List, None]:259 """Validates proper combinations of init arguments"""260 261 errors = []262 if url is None:263 errors.append("Must provide `base_url`")264 265 if (api_key and not username) or (username and not api_key):266 errors.append(267 "If one of `api_key` or `username` is provided, "268 "the other must be as well."269 )270 271 non_null_creds = list(272 x is not None273 for x in ((api_key or username), session, oauth2, token, cookies)274 )275 if sum(non_null_creds) > 1:276 all_names = ("(api_key, username)", "session", "oauth2", "token", "cookies")277 provided = tuple(n for x, n in zip(non_null_creds, all_names) if x)278 errors.append(279 f"Cannot provide a value for more than one of: {all_names}. Received "280 f"values for: {provided}"281 )282 283 if (284 oauth2285 and set(oauth2.keys())286 == {287 "token",288 "client_id",289 }290 and set(oauth2["token"].keys())291 != {292 "access_token",293 "token_type",294 }295 ):296 # OAuth2 token authentication297 errors.append(298 "You have either omitted require keys or added extra "299 "keys to the oauth2 dictionary. key values should be "300 "`['client_id', 'token': ['access_token', 'token_type']]`"301 )302 303 if (304 oauth2305 and set(oauth2.keys())306 != {307 "access_token",308 "access_token_secret",309 "consumer_key",310 "key_cert",311 }312 and set(oauth2.keys())313 != {314 "token",315 "client_id",316 }317 ):318 errors.append(319 "You have either omitted required keys or added extra "320 "keys to the oauth2 dictionary. key values should be "321 "`['access_token', 'access_token_secret', 'consumer_key', 'key_cert']` "322 "or `['client_id', 'token': ['access_token', 'token_type']]`"323 )324 return errors or None325 326 def _resolve_param(self, param_name: str, kwargs: Any) -> Any:327 return kwargs[param_name] if param_name in kwargs else getattr(self, param_name)328 329 def _lazy_load(self, **kwargs: Any) -> Iterator[Document]:330 if kwargs:331 logger.warning(332 f"Received runtime arguments {kwargs}. Passing runtime args to `load`"333 f" is deprecated. Please pass arguments during initialization instead."334 )335 space_key = self._resolve_param("space_key", kwargs)336 page_ids = self._resolve_param("page_ids", kwargs)337 label = self._resolve_param("label", kwargs)338 cql = self._resolve_param("cql", kwargs)339 include_restricted_content = self._resolve_param(340 "include_restricted_content", kwargs341 )342 include_archived_content = self._resolve_param(343 "include_archived_content", kwargs344 )345 include_attachments = self._resolve_param("include_attachments", kwargs)346 include_comments = self._resolve_param("include_comments", kwargs)347 include_labels = self._resolve_param("include_labels", kwargs)348 content_format = self._resolve_param("content_format", kwargs)349 limit = self._resolve_param("limit", kwargs)350 max_pages = self._resolve_param("max_pages", kwargs)351 ocr_languages = self._resolve_param("ocr_languages", kwargs)352 keep_markdown_format = self._resolve_param("keep_markdown_format", kwargs)353 keep_newlines = self._resolve_param("keep_newlines", kwargs)354 expand = ",".join(355 [356 content_format.value,357 "version",358 *(["metadata.labels"] if include_labels else []),359 ]360 )361 362 if not space_key and not page_ids and not label and not cql:363 raise ValueError(364 "Must specify at least one among `space_key`, `page_ids`, "365 "`label`, `cql` parameters."366 )367 368 if space_key:369 pages = self.paginate_request(370 self.confluence.get_all_pages_from_space,371 space=space_key,372 limit=limit,373 max_pages=max_pages,374 status="any" if include_archived_content else "current",375 expand=expand,376 )377 yield from self.process_pages(378 pages,379 include_restricted_content,380 include_attachments,381 include_comments,382 include_labels,383 content_format,384 ocr_languages=ocr_languages,385 keep_markdown_format=keep_markdown_format,386 keep_newlines=keep_newlines,387 )388 389 if label:390 pages = self.paginate_request(391 self.confluence.get_all_pages_by_label,392 label=label,393 limit=limit,394 max_pages=max_pages,395 )396 ids_by_label = [page["id"] for page in pages]397 if page_ids:398 page_ids = list(set(page_ids + ids_by_label))399 else:400 page_ids = list(set(ids_by_label))401 402 if cql:403 pages = self.paginate_request(404 self._search_content_by_cql,405 cql=cql,406 limit=limit,407 max_pages=max_pages,408 include_archived_spaces=include_archived_content,409 expand=expand,410 )411 yield from self.process_pages(412 pages,413 include_restricted_content,414 include_attachments,415 include_comments,416 include_labels,417 content_format,418 ocr_languages,419 keep_markdown_format,420 keep_newlines=keep_newlines,421 )422 423 if page_ids:424 for page_id in page_ids:425 get_page = retry(426 reraise=True,427 stop=stop_after_attempt(428 self.number_of_retries # type: ignore[arg-type]429 ),430 wait=wait_exponential(431 multiplier=1,432 min=self.min_retry_seconds, # type: ignore[arg-type]433 max=self.max_retry_seconds, # type: ignore[arg-type]434 ),435 before_sleep=before_sleep_log(logger, logging.WARNING),436 )(self.confluence.get_page_by_id)437 page = get_page(438 page_id=page_id,439 expand=expand,440 )441 if not include_restricted_content and not self.is_public_page(page):442 continue443 yield self.process_page(444 page,445 include_attachments,446 include_comments,447 include_labels,448 content_format,449 ocr_languages,450 keep_markdown_format,451 keep_newlines=keep_newlines,452 )453 454 def load(self, **kwargs: Any) -> List[Document]:455 return list(self._lazy_load(**kwargs))456 457 def lazy_load(self) -> Iterator[Document]:458 yield from self._lazy_load()459 460 def _search_content_by_cql(461 self,462 cql: str,463 include_archived_spaces: Optional[bool] = None,464 next_url: str = "",465 **kwargs: Any,466 ) -> tuple[List[dict], str]:467 if next_url:468 response = self.confluence.get(next_url)469 else:470 url = "rest/api/content/search"471 472 params: Dict[str, Any] = {"cql": cql}473 params.update(kwargs)474 if include_archived_spaces is not None:475 params["includeArchivedSpaces"] = include_archived_spaces476 477 response = self.confluence.get(url, params=params)478 479 return response.get("results", []), response.get("_links", {}).get("next", "")480 481 def paginate_request(self, retrieval_method: Callable, **kwargs: Any) -> List:482 """Paginate the various methods to retrieve groups of pages.483 484 Unfortunately, due to page size, sometimes the Confluence API485 doesn't match the limit value. If `limit` is >100 confluence486 seems to cap the response to 100. Also, due to the Atlassian Python487 package, we don't get the "next" values from the "_links" key because488 they only return the value from the result key. So here, the pagination489 starts from 0 and goes until the max_pages, getting the `limit` number490 of pages with each request. We have to manually check if there491 are more docs based on the length of the returned list of pages, rather than492 just checking for the presence of a `next` key in the response like this page493 would have you do:494 https://developer.atlassian.com/server/confluence/pagination-in-the-rest-api/495 496 :param retrieval_method: Function used to retrieve docs497 :type retrieval_method: callable498 :return: List of documents499 :rtype: List500 """501 502 max_pages = kwargs.pop("max_pages")503 docs: List[dict] = []504 next_url: str = ""505 while len(docs) < max_pages:506 get_pages = retry(507 reraise=True,508 stop=stop_after_attempt(509 self.number_of_retries # type: ignore[arg-type]510 ),511 wait=wait_exponential(512 multiplier=1,513 min=self.min_retry_seconds, # type: ignore[arg-type]514 max=self.max_retry_seconds, # type: ignore[arg-type]515 ),516 before_sleep=before_sleep_log(logger, logging.WARNING),517 )(retrieval_method)518 if self.cql: # cursor pagination for CQL519 batch, next_url = get_pages(**kwargs, next_url=next_url)520 if not next_url:521 docs.extend(batch)522 break523 else:524 batch = get_pages(**kwargs, start=len(docs))525 if not batch:526 break527 docs.extend(batch)528 return docs[:max_pages]529 530 def is_public_page(self, page: dict) -> bool:531 """Check if a page is publicly accessible."""532 533 if page["status"] != "current":534 return False535 536 restrictions = self.confluence.get_all_restrictions_for_content(page["id"])537 538 return (539 not restrictions["read"]["restrictions"]["user"]["results"]540 and not restrictions["read"]["restrictions"]["group"]["results"]541 )542 543 def process_pages(544 self,545 pages: List[dict],546 include_restricted_content: bool,547 include_attachments: bool,548 include_comments: bool,549 include_labels: bool,550 content_format: ContentFormat,551 ocr_languages: Optional[str] = None,552 keep_markdown_format: Optional[bool] = False,553 keep_newlines: bool = False,554 ) -> Iterator[Document]:555 """Process a list of pages into a list of documents."""556 for page in pages:557 if not include_restricted_content and not self.is_public_page(page):558 continue559 yield self.process_page(560 page,561 include_attachments,562 include_comments,563 include_labels,564 content_format,565 ocr_languages=ocr_languages,566 keep_markdown_format=keep_markdown_format,567 keep_newlines=keep_newlines,568 )569 570 def process_page(571 self,572 page: dict,573 include_attachments: bool,574 include_comments: bool,575 include_labels: bool,576 content_format: ContentFormat,577 ocr_languages: Optional[str] = None,578 keep_markdown_format: Optional[bool] = False,579 keep_newlines: bool = False,580 ) -> Document:581 if keep_markdown_format:582 try:583 from markdownify import markdownify584 except ImportError:585 raise ImportError(586 "`markdownify` package not found, please run "587 "`pip install markdownify`"588 )589 if include_comments or not keep_markdown_format:590 try:591 from bs4 import BeautifulSoup592 except ImportError:593 raise ImportError(594 "`beautifulsoup4` package not found, please run "595 "`pip install beautifulsoup4`"596 )597 if include_attachments:598 attachment_texts = self.process_attachment(page["id"], ocr_languages)599 else:600 attachment_texts = []601 602 content = content_format.get_content(page)603 if keep_markdown_format:604 # Use markdownify to keep the page Markdown style605 text = markdownify(content, heading_style="ATX") + "".join(attachment_texts)606 607 else:608 if keep_newlines:609 text = BeautifulSoup(610 content.replace("</p>", "\n</p>").replace("<br />", "\n"), "lxml"611 ).get_text(" ") + "".join(attachment_texts)612 else:613 text = BeautifulSoup(content, "lxml").get_text(614 " ", strip=True615 ) + "".join(attachment_texts)616 617 if include_comments:618 comments = self.confluence.get_page_comments(619 page["id"], expand="body.view.value", depth="all"620 )["results"]621 comment_texts = [622 BeautifulSoup(comment["body"]["view"]["value"], "lxml").get_text(623 " ", strip=True624 )625 for comment in comments626 ]627 text = text + "".join(comment_texts)628 629 if include_labels:630 labels = [631 label["name"]632 for label in page.get("metadata", {})633 .get("labels", {})634 .get("results", [])635 ]636 637 metadata = {638 "title": page["title"],639 "id": page["id"],640 "source": self.base_url.strip("/") + page["_links"]["webui"],641 **({"labels": labels} if include_labels else {}),642 }643 644 if "version" in page and "when" in page["version"]:645 metadata["when"] = page["version"]["when"]646 647 return Document(648 page_content=text,649 metadata=metadata,650 )651 652 def process_attachment(653 self,654 page_id: str,655 ocr_languages: Optional[str] = None,656 ) -> List[str]:657 try:658 from PIL import Image # noqa: F401659 except ImportError:660 raise ImportError(661 "`Pillow` package not found, please run `pip install Pillow`"662 )663 664 # depending on setup you may also need to set the correct path for665 # poppler and tesseract666 attachments = self.confluence.get_attachments_from_content(page_id)["results"]667 texts = []668 for attachment in attachments:669 if self.attachment_filter_func and not self.attachment_filter_func(670 attachment671 ):672 continue673 674 media_type = attachment["metadata"]["mediaType"]675 absolute_url = self.base_url + attachment["_links"]["download"]676 title = attachment["title"]677 try:678 if media_type == "application/pdf":679 text = title + self.process_pdf(absolute_url, ocr_languages)680 elif (681 media_type == "image/png"682 or media_type == "image/jpg"683 or media_type == "image/jpeg"684 ):685 text = title + self.process_image(absolute_url, ocr_languages)686 elif (687 media_type == "application/vnd.openxmlformats-officedocument"688 ".wordprocessingml.document"689 ):690 text = title + self.process_doc(absolute_url)691 elif media_type == "application/vnd.ms-excel":692 text = title + self.process_xls(absolute_url)693 elif media_type == "image/svg+xml":694 text = title + self.process_svg(absolute_url, ocr_languages)695 else:696 continue697 texts.append(text)698 except requests.HTTPError as e:699 if e.response.status_code == 404:700 print(f"Attachment not found at {absolute_url}") # noqa: T201701 continue702 else:703 raise704 705 return texts706 707 def process_pdf(708 self,709 link: str,710 ocr_languages: Optional[str] = None,711 ) -> str:712 try:713 import pytesseract714 from pdf2image import convert_from_bytes715 except ImportError:716 raise ImportError(717 "`pytesseract` or `pdf2image` package not found, "718 "please run `pip install pytesseract pdf2image`"719 )720 721 response = self.confluence.request(path=link, absolute=True)722 text = ""723 724 if (725 response.status_code != 200726 or response.content == b""727 or response.content is None728 ):729 return text730 try:731 images = convert_from_bytes(response.content)732 except ValueError:733 return text734 735 for i, image in enumerate(images):736 try:737 image_text = pytesseract.image_to_string(image, lang=ocr_languages)738 text += f"Page {i + 1}:\n{image_text}\n\n"739 except pytesseract.TesseractError as ex:740 logger.warning(f"TesseractError: {ex}")741 742 return text743 744 def process_image(745 self,746 link: str,747 ocr_languages: Optional[str] = None,748 ) -> str:749 try:750 import pytesseract751 from PIL import Image752 except ImportError:753 raise ImportError(754 "`pytesseract` or `Pillow` package not found, "755 "please run `pip install pytesseract Pillow`"756 )757 758 response = self.confluence.request(path=link, absolute=True)759 text = ""760 761 if (762 response.status_code != 200763 or response.content == b""764 or response.content is None765 ):766 return text767 try:768 image = Image.open(BytesIO(response.content))769 except OSError:770 return text771 772 return pytesseract.image_to_string(image, lang=ocr_languages)773 774 def process_doc(self, link: str) -> str:775 try:776 import docx2txt777 except ImportError:778 raise ImportError(779 "`docx2txt` package not found, please run `pip install docx2txt`"780 )781 782 response = self.confluence.request(path=link, absolute=True)783 text = ""784 785 if (786 response.status_code != 200787 or response.content == b""788 or response.content is None789 ):790 return text791 file_data = BytesIO(response.content)792 793 return docx2txt.process(file_data)794 795 def process_xls(self, link: str) -> str:796 import io797 import os798 799 try:800 import xlrd801 802 except ImportError:803 raise ImportError("`xlrd` package not found, please run `pip install xlrd`")804 805 try:806 import pandas as pd807 808 except ImportError:809 raise ImportError(810 "`pandas` package not found, please run `pip install pandas`"811 )812 813 response = self.confluence.request(path=link, absolute=True)814 text = ""815 816 if (817 response.status_code != 200818 or response.content == b""819 or response.content is None820 ):821 return text822 823 filename = os.path.basename(link)824 # Getting the whole content of the url after filename,825 # Example: ".csv?version=2&modificationDate=1631800010678&cacheVersion=1&api=v2"826 file_extension = os.path.splitext(filename)[1]827 828 if file_extension.startswith(829 ".csv"830 ): # if the extension found in the url is ".csv"831 content_string = response.content.decode("utf-8")832 df = pd.read_csv(io.StringIO(content_string))833 text += df.to_string(index=False, header=False) + "\n\n"834 else:835 workbook = xlrd.open_workbook(file_contents=response.content)836 for sheet in workbook.sheets():837 text += f"{sheet.name}:\n"838 for row in range(sheet.nrows):839 for col in range(sheet.ncols):840 text += f"{sheet.cell_value(row, col)}\t"841 text += "\n"842 text += "\n"843 844 return text845 846 def process_svg(847 self,848 link: str,849 ocr_languages: Optional[str] = None,850 ) -> str:851 try:852 import pytesseract853 from PIL import Image854 from reportlab.graphics import renderPM855 from svglib.svglib import svg2rlg856 except ImportError:857 raise ImportError(858 "`pytesseract`, `Pillow`, `reportlab` or `svglib` package not found, "859 "please run `pip install pytesseract Pillow reportlab svglib`"860 )861 862 response = self.confluence.request(path=link, absolute=True)863 text = ""864 865 if (866 response.status_code != 200867 or response.content == b""868 or response.content is None869 ):870 return text871 872 drawing = svg2rlg(BytesIO(response.content))873 874 img_data = BytesIO()875 renderPM.drawToFile(drawing, img_data, fmt="PNG")876 img_data.seek(0)877 image = Image.open(img_data)878 879 return pytesseract.image_to_string(image, lang=ocr_languages)880 