Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
docugami.py370 linesDownload Raw Back to document_loaders
1import hashlib2import io3import logging4import os5from pathlib import Path6from typing import Any, Dict, List, Mapping, Optional, Sequence, Union7 8import requests9from langchain_core._api.deprecation import deprecated10from langchain_core.documents import Document11from pydantic import BaseModel, model_validator12 13from langchain_community.document_loaders.base import BaseLoader14 15TABLE_NAME = "{http://www.w3.org/1999/xhtml}table"16 17XPATH_KEY = "xpath"18ID_KEY = "id"19DOCUMENT_SOURCE_KEY = "source"20DOCUMENT_NAME_KEY = "name"21STRUCTURE_KEY = "structure"22TAG_KEY = "tag"23PROJECTS_KEY = "projects"24 25DEFAULT_API_ENDPOINT = "https://api.docugami.com/v1preview1"26 27logger = logging.getLogger(__name__)28 29 30@deprecated(31    since="0.0.24",32    removal="1.0",33    alternative_import="docugami_langchain.DocugamiLoader",34)35class DocugamiLoader(BaseLoader, BaseModel):36    """Load from `Docugami`.37 38    To use, you should have the ``dgml-utils`` python package installed.39    """40 41    api: str = DEFAULT_API_ENDPOINT42    """The Docugami API endpoint to use."""43 44    access_token: Optional[str] = os.environ.get("DOCUGAMI_API_KEY")45    """The Docugami API access token to use."""46 47    max_text_length: int = 409648    """Max length of chunk text returned."""49 50    min_text_length: int = 3251    """Threshold under which chunks are appended to next to avoid over-chunking."""52 53    max_metadata_length: int = 51254    """Max length of metadata text returned."""55 56    include_xml_tags: bool = False57    """Set to true for XML tags in chunk output text."""58 59    parent_hierarchy_levels: int = 060    """Set appropriately to get parent chunks using the chunk hierarchy."""61 62    parent_id_key: str = "doc_id"63    """Metadata key for parent doc ID."""64 65    sub_chunk_tables: bool = False66    """Set to True to return sub-chunks within tables."""67 68    whitespace_normalize_text: bool = True69    """Set to False if you want to full whitespace formatting in the original70    XML doc, including indentation."""71 72    docset_id: Optional[str] = None73    """The Docugami API docset ID to use."""74 75    document_ids: Optional[Sequence[str]] = None76    """The Docugami API document IDs to use."""77 78    file_paths: Optional[Sequence[Union[Path, str]]]79    """The local file paths to use."""80 81    include_project_metadata_in_doc_metadata: bool = True82    """Set to True if you want to include the project metadata in the doc metadata."""83 84    @model_validator(mode="before")85    @classmethod86    def validate_local_or_remote(cls, values: Dict[str, Any]) -> Any:87        """Validate that either local file paths are given, or remote API docset ID.88 89        Args:90            values: The values to validate.91 92        Returns:93            The validated values.94        """95        if values.get("file_paths") and values.get("docset_id"):96            raise ValueError("Cannot specify both file_paths and remote API docset_id")97 98        if not values.get("file_paths") and not values.get("docset_id"):99            raise ValueError("Must specify either file_paths or remote API docset_id")100 101        if values.get("docset_id") and not values.get("access_token"):102            raise ValueError("Must specify access token if using remote API docset_id")103 104        return values105 106    def _parse_dgml(107        self,108        content: bytes,109        document_name: Optional[str] = None,110        additional_doc_metadata: Optional[Mapping] = None,111    ) -> List[Document]:112        """Parse a single DGML document into a list of `Document` objects."""113        try:114            from lxml import etree115        except ImportError:116            raise ImportError(117                "Could not import lxml python package. "118                "Please install it with `pip install lxml`."119            )120 121        try:122            from dgml_utils.models import Chunk123            from dgml_utils.segmentation import get_chunks124        except ImportError:125            raise ImportError(126                "Could not import from dgml-utils python package. "127                "Please install it with `pip install dgml-utils`."128            )129 130        def _build_framework_chunk(dg_chunk: Chunk) -> Document:131            # Stable IDs for chunks with the same text.132            _hashed_id = hashlib.md5(dg_chunk.text.encode()).hexdigest()133            metadata = {134                XPATH_KEY: dg_chunk.xpath,135                ID_KEY: _hashed_id,136                DOCUMENT_NAME_KEY: document_name,137                DOCUMENT_SOURCE_KEY: document_name,138                STRUCTURE_KEY: dg_chunk.structure,139                TAG_KEY: dg_chunk.tag,140            }141 142            text = dg_chunk.text143            if additional_doc_metadata:144                if self.include_project_metadata_in_doc_metadata:145                    metadata.update(additional_doc_metadata)146 147            return Document(148                page_content=text[: self.max_text_length],149                metadata=metadata,150            )151 152        # Parse the tree and return chunks153        tree = etree.parse(io.BytesIO(content))154        root = tree.getroot()155 156        dg_chunks = get_chunks(157            root,158            min_text_length=self.min_text_length,159            max_text_length=self.max_text_length,160            whitespace_normalize_text=self.whitespace_normalize_text,161            sub_chunk_tables=self.sub_chunk_tables,162            include_xml_tags=self.include_xml_tags,163            parent_hierarchy_levels=self.parent_hierarchy_levels,164        )165 166        framework_chunks: Dict[str, Document] = {}167        for dg_chunk in dg_chunks:168            framework_chunk = _build_framework_chunk(dg_chunk)169            chunk_id = framework_chunk.metadata.get(ID_KEY)170            if chunk_id:171                framework_chunks[chunk_id] = framework_chunk172                if dg_chunk.parent:173                    framework_parent_chunk = _build_framework_chunk(dg_chunk.parent)174                    parent_id = framework_parent_chunk.metadata.get(ID_KEY)175                    if parent_id and framework_parent_chunk.page_content:176                        framework_chunk.metadata[self.parent_id_key] = parent_id177                        framework_chunks[parent_id] = framework_parent_chunk178 179        return list(framework_chunks.values())180 181    def _document_details_for_docset_id(self, docset_id: str) -> List[Dict]:182        """Gets all document details for the given docset ID"""183        url = f"{self.api}/docsets/{docset_id}/documents"184        all_documents = []185 186        while url:187            response = requests.get(188                url,189                headers={"Authorization": f"Bearer {self.access_token}"},190            )191            if response.ok:192                data = response.json()193                all_documents.extend(data["documents"])194                url = data.get("next", None)195            else:196                raise Exception(197                    f"Failed to download {url} (status: {response.status_code})"198                )199 200        return all_documents201 202    def _project_details_for_docset_id(self, docset_id: str) -> List[Dict]:203        """Gets all project details for the given docset ID"""204        url = f"{self.api}/projects?docset.id={docset_id}"205        all_projects = []206 207        while url:208            response = requests.request(209                "GET",210                url,211                headers={"Authorization": f"Bearer {self.access_token}"},212                data={},213            )214            if response.ok:215                data = response.json()216                all_projects.extend(data["projects"])217                url = data.get("next", None)218            else:219                raise Exception(220                    f"Failed to download {url} (status: {response.status_code})"221                )222 223        return all_projects224 225    def _metadata_for_project(self, project: Dict) -> Dict:226        """Gets project metadata for all files"""227        project_id = project.get(ID_KEY)228 229        url = f"{self.api}/projects/{project_id}/artifacts/latest"230        all_artifacts = []231 232        per_file_metadata: Dict = {}233        while url:234            response = requests.request(235                "GET",236                url,237                headers={"Authorization": f"Bearer {self.access_token}"},238                data={},239            )240            if response.ok:241                data = response.json()242                all_artifacts.extend(data["artifacts"])243                url = data.get("next", None)244            elif response.status_code == 404:245                # Not found is ok, just means no published projects246                return per_file_metadata247            else:248                raise Exception(249                    f"Failed to download {url} (status: {response.status_code})"250                )251 252        for artifact in all_artifacts:253            artifact_name = artifact.get("name")254            artifact_url = artifact.get("url")255            artifact_doc = artifact.get("document")256 257            if artifact_name == "report-values.xml" and artifact_url and artifact_doc:258                doc_id = artifact_doc[ID_KEY]259                metadata: Dict = {}260 261                # The evaluated XML for each document is named after the project262                response = requests.request(263                    "GET",264                    f"{artifact_url}/content",265                    headers={"Authorization": f"Bearer {self.access_token}"},266                    data={},267                )268 269                if response.ok:270                    try:271                        from lxml import etree272                    except ImportError:273                        raise ImportError(274                            "Could not import lxml python package. "275                            "Please install it with `pip install lxml`."276                        )277                    artifact_tree = etree.parse(io.BytesIO(response.content))278                    artifact_root = artifact_tree.getroot()279                    ns = artifact_root.nsmap280                    entries = artifact_root.xpath("//pr:Entry", namespaces=ns)281                    for entry in entries:282                        heading = entry.xpath("./pr:Heading", namespaces=ns)[0].text283                        value = " ".join(284                            entry.xpath("./pr:Value", namespaces=ns)[0].itertext()285                        ).strip()286                        metadata[heading] = value[: self.max_metadata_length]287                    per_file_metadata[doc_id] = metadata288                else:289                    raise Exception(290                        f"Failed to download {artifact_url}/content "291                        + "(status: {response.status_code})"292                    )293 294        return per_file_metadata295 296    def _load_chunks_for_document(297        self,298        document_id: str,299        docset_id: str,300        document_name: Optional[str] = None,301        additional_metadata: Optional[Mapping] = None,302    ) -> List[Document]:303        """Load chunks for a document."""304        url = f"{self.api}/docsets/{docset_id}/documents/{document_id}/dgml"305 306        response = requests.request(307            "GET",308            url,309            headers={"Authorization": f"Bearer {self.access_token}"},310            data={},311        )312 313        if response.ok:314            return self._parse_dgml(315                content=response.content,316                document_name=document_name,317                additional_doc_metadata=additional_metadata,318            )319        else:320            raise Exception(321                f"Failed to download {url} (status: {response.status_code})"322            )323 324    def load(self) -> List[Document]:325        """Load documents."""326        chunks: List[Document] = []327 328        if self.access_token and self.docset_id:329            # Remote mode330            _document_details = self._document_details_for_docset_id(self.docset_id)331            if self.document_ids:332                _document_details = [333                    d for d in _document_details if d[ID_KEY] in self.document_ids334                ]335 336            _project_details = self._project_details_for_docset_id(self.docset_id)337            combined_project_metadata: Dict[str, Dict] = {}338            if _project_details and self.include_project_metadata_in_doc_metadata:339                # If there are any projects for this docset and the caller requested340                # project metadata, load it.341                for project in _project_details:342                    metadata = self._metadata_for_project(project)343                    for file_id in metadata:344                        if file_id not in combined_project_metadata:345                            combined_project_metadata[file_id] = metadata[file_id]346                        else:347                            combined_project_metadata[file_id].update(metadata[file_id])348 349            for doc in _document_details:350                doc_id = doc[ID_KEY]351                doc_name = doc.get(DOCUMENT_NAME_KEY)352                doc_metadata = combined_project_metadata.get(doc_id)353                chunks += self._load_chunks_for_document(354                    document_id=doc_id,355                    docset_id=self.docset_id,356                    document_name=doc_name,357                    additional_metadata=doc_metadata,358                )359        elif self.file_paths:360            # Local mode (for integration testing, or pre-downloaded XML)361            for path in self.file_paths:362                path = Path(path)363                with open(path, "rb") as file:364                    chunks += self._parse_dgml(365                        content=file.read(),366                        document_name=path.name,367                    )368 369        return chunks370 
codekingpro/portable-devtools · Team Ai