codekingpro/portable-devtools
114k
1import hashlib2import io3import logging4import os5from pathlib import Path6from typing import Any, Dict, List, Mapping, Optional, Sequence, Union7 8import requests9from langchain_core._api.deprecation import deprecated10from langchain_core.documents import Document11from pydantic import BaseModel, model_validator12 13from langchain_community.document_loaders.base import BaseLoader14 15TABLE_NAME = "{http://www.w3.org/1999/xhtml}table"16 17XPATH_KEY = "xpath"18ID_KEY = "id"19DOCUMENT_SOURCE_KEY = "source"20DOCUMENT_NAME_KEY = "name"21STRUCTURE_KEY = "structure"22TAG_KEY = "tag"23PROJECTS_KEY = "projects"24 25DEFAULT_API_ENDPOINT = "https://api.docugami.com/v1preview1"26 27logger = logging.getLogger(__name__)28 29 30@deprecated(31 since="0.0.24",32 removal="1.0",33 alternative_import="docugami_langchain.DocugamiLoader",34)35class DocugamiLoader(BaseLoader, BaseModel):36 """Load from `Docugami`.37 38 To use, you should have the ``dgml-utils`` python package installed.39 """40 41 api: str = DEFAULT_API_ENDPOINT42 """The Docugami API endpoint to use."""43 44 access_token: Optional[str] = os.environ.get("DOCUGAMI_API_KEY")45 """The Docugami API access token to use."""46 47 max_text_length: int = 409648 """Max length of chunk text returned."""49 50 min_text_length: int = 3251 """Threshold under which chunks are appended to next to avoid over-chunking."""52 53 max_metadata_length: int = 51254 """Max length of metadata text returned."""55 56 include_xml_tags: bool = False57 """Set to true for XML tags in chunk output text."""58 59 parent_hierarchy_levels: int = 060 """Set appropriately to get parent chunks using the chunk hierarchy."""61 62 parent_id_key: str = "doc_id"63 """Metadata key for parent doc ID."""64 65 sub_chunk_tables: bool = False66 """Set to True to return sub-chunks within tables."""67 68 whitespace_normalize_text: bool = True69 """Set to False if you want to full whitespace formatting in the original70 XML doc, including indentation."""71 72 docset_id: Optional[str] = None73 """The Docugami API docset ID to use."""74 75 document_ids: Optional[Sequence[str]] = None76 """The Docugami API document IDs to use."""77 78 file_paths: Optional[Sequence[Union[Path, str]]]79 """The local file paths to use."""80 81 include_project_metadata_in_doc_metadata: bool = True82 """Set to True if you want to include the project metadata in the doc metadata."""83 84 @model_validator(mode="before")85 @classmethod86 def validate_local_or_remote(cls, values: Dict[str, Any]) -> Any:87 """Validate that either local file paths are given, or remote API docset ID.88 89 Args:90 values: The values to validate.91 92 Returns:93 The validated values.94 """95 if values.get("file_paths") and values.get("docset_id"):96 raise ValueError("Cannot specify both file_paths and remote API docset_id")97 98 if not values.get("file_paths") and not values.get("docset_id"):99 raise ValueError("Must specify either file_paths or remote API docset_id")100 101 if values.get("docset_id") and not values.get("access_token"):102 raise ValueError("Must specify access token if using remote API docset_id")103 104 return values105 106 def _parse_dgml(107 self,108 content: bytes,109 document_name: Optional[str] = None,110 additional_doc_metadata: Optional[Mapping] = None,111 ) -> List[Document]:112 """Parse a single DGML document into a list of `Document` objects."""113 try:114 from lxml import etree115 except ImportError:116 raise ImportError(117 "Could not import lxml python package. "118 "Please install it with `pip install lxml`."119 )120 121 try:122 from dgml_utils.models import Chunk123 from dgml_utils.segmentation import get_chunks124 except ImportError:125 raise ImportError(126 "Could not import from dgml-utils python package. "127 "Please install it with `pip install dgml-utils`."128 )129 130 def _build_framework_chunk(dg_chunk: Chunk) -> Document:131 # Stable IDs for chunks with the same text.132 _hashed_id = hashlib.md5(dg_chunk.text.encode()).hexdigest()133 metadata = {134 XPATH_KEY: dg_chunk.xpath,135 ID_KEY: _hashed_id,136 DOCUMENT_NAME_KEY: document_name,137 DOCUMENT_SOURCE_KEY: document_name,138 STRUCTURE_KEY: dg_chunk.structure,139 TAG_KEY: dg_chunk.tag,140 }141 142 text = dg_chunk.text143 if additional_doc_metadata:144 if self.include_project_metadata_in_doc_metadata:145 metadata.update(additional_doc_metadata)146 147 return Document(148 page_content=text[: self.max_text_length],149 metadata=metadata,150 )151 152 # Parse the tree and return chunks153 tree = etree.parse(io.BytesIO(content))154 root = tree.getroot()155 156 dg_chunks = get_chunks(157 root,158 min_text_length=self.min_text_length,159 max_text_length=self.max_text_length,160 whitespace_normalize_text=self.whitespace_normalize_text,161 sub_chunk_tables=self.sub_chunk_tables,162 include_xml_tags=self.include_xml_tags,163 parent_hierarchy_levels=self.parent_hierarchy_levels,164 )165 166 framework_chunks: Dict[str, Document] = {}167 for dg_chunk in dg_chunks:168 framework_chunk = _build_framework_chunk(dg_chunk)169 chunk_id = framework_chunk.metadata.get(ID_KEY)170 if chunk_id:171 framework_chunks[chunk_id] = framework_chunk172 if dg_chunk.parent:173 framework_parent_chunk = _build_framework_chunk(dg_chunk.parent)174 parent_id = framework_parent_chunk.metadata.get(ID_KEY)175 if parent_id and framework_parent_chunk.page_content:176 framework_chunk.metadata[self.parent_id_key] = parent_id177 framework_chunks[parent_id] = framework_parent_chunk178 179 return list(framework_chunks.values())180 181 def _document_details_for_docset_id(self, docset_id: str) -> List[Dict]:182 """Gets all document details for the given docset ID"""183 url = f"{self.api}/docsets/{docset_id}/documents"184 all_documents = []185 186 while url:187 response = requests.get(188 url,189 headers={"Authorization": f"Bearer {self.access_token}"},190 )191 if response.ok:192 data = response.json()193 all_documents.extend(data["documents"])194 url = data.get("next", None)195 else:196 raise Exception(197 f"Failed to download {url} (status: {response.status_code})"198 )199 200 return all_documents201 202 def _project_details_for_docset_id(self, docset_id: str) -> List[Dict]:203 """Gets all project details for the given docset ID"""204 url = f"{self.api}/projects?docset.id={docset_id}"205 all_projects = []206 207 while url:208 response = requests.request(209 "GET",210 url,211 headers={"Authorization": f"Bearer {self.access_token}"},212 data={},213 )214 if response.ok:215 data = response.json()216 all_projects.extend(data["projects"])217 url = data.get("next", None)218 else:219 raise Exception(220 f"Failed to download {url} (status: {response.status_code})"221 )222 223 return all_projects224 225 def _metadata_for_project(self, project: Dict) -> Dict:226 """Gets project metadata for all files"""227 project_id = project.get(ID_KEY)228 229 url = f"{self.api}/projects/{project_id}/artifacts/latest"230 all_artifacts = []231 232 per_file_metadata: Dict = {}233 while url:234 response = requests.request(235 "GET",236 url,237 headers={"Authorization": f"Bearer {self.access_token}"},238 data={},239 )240 if response.ok:241 data = response.json()242 all_artifacts.extend(data["artifacts"])243 url = data.get("next", None)244 elif response.status_code == 404:245 # Not found is ok, just means no published projects246 return per_file_metadata247 else:248 raise Exception(249 f"Failed to download {url} (status: {response.status_code})"250 )251 252 for artifact in all_artifacts:253 artifact_name = artifact.get("name")254 artifact_url = artifact.get("url")255 artifact_doc = artifact.get("document")256 257 if artifact_name == "report-values.xml" and artifact_url and artifact_doc:258 doc_id = artifact_doc[ID_KEY]259 metadata: Dict = {}260 261 # The evaluated XML for each document is named after the project262 response = requests.request(263 "GET",264 f"{artifact_url}/content",265 headers={"Authorization": f"Bearer {self.access_token}"},266 data={},267 )268 269 if response.ok:270 try:271 from lxml import etree272 except ImportError:273 raise ImportError(274 "Could not import lxml python package. "275 "Please install it with `pip install lxml`."276 )277 artifact_tree = etree.parse(io.BytesIO(response.content))278 artifact_root = artifact_tree.getroot()279 ns = artifact_root.nsmap280 entries = artifact_root.xpath("//pr:Entry", namespaces=ns)281 for entry in entries:282 heading = entry.xpath("./pr:Heading", namespaces=ns)[0].text283 value = " ".join(284 entry.xpath("./pr:Value", namespaces=ns)[0].itertext()285 ).strip()286 metadata[heading] = value[: self.max_metadata_length]287 per_file_metadata[doc_id] = metadata288 else:289 raise Exception(290 f"Failed to download {artifact_url}/content "291 + "(status: {response.status_code})"292 )293 294 return per_file_metadata295 296 def _load_chunks_for_document(297 self,298 document_id: str,299 docset_id: str,300 document_name: Optional[str] = None,301 additional_metadata: Optional[Mapping] = None,302 ) -> List[Document]:303 """Load chunks for a document."""304 url = f"{self.api}/docsets/{docset_id}/documents/{document_id}/dgml"305 306 response = requests.request(307 "GET",308 url,309 headers={"Authorization": f"Bearer {self.access_token}"},310 data={},311 )312 313 if response.ok:314 return self._parse_dgml(315 content=response.content,316 document_name=document_name,317 additional_doc_metadata=additional_metadata,318 )319 else:320 raise Exception(321 f"Failed to download {url} (status: {response.status_code})"322 )323 324 def load(self) -> List[Document]:325 """Load documents."""326 chunks: List[Document] = []327 328 if self.access_token and self.docset_id:329 # Remote mode330 _document_details = self._document_details_for_docset_id(self.docset_id)331 if self.document_ids:332 _document_details = [333 d for d in _document_details if d[ID_KEY] in self.document_ids334 ]335 336 _project_details = self._project_details_for_docset_id(self.docset_id)337 combined_project_metadata: Dict[str, Dict] = {}338 if _project_details and self.include_project_metadata_in_doc_metadata:339 # If there are any projects for this docset and the caller requested340 # project metadata, load it.341 for project in _project_details:342 metadata = self._metadata_for_project(project)343 for file_id in metadata:344 if file_id not in combined_project_metadata:345 combined_project_metadata[file_id] = metadata[file_id]346 else:347 combined_project_metadata[file_id].update(metadata[file_id])348 349 for doc in _document_details:350 doc_id = doc[ID_KEY]351 doc_name = doc.get(DOCUMENT_NAME_KEY)352 doc_metadata = combined_project_metadata.get(doc_id)353 chunks += self._load_chunks_for_document(354 document_id=doc_id,355 docset_id=self.docset_id,356 document_name=doc_name,357 additional_metadata=doc_metadata,358 )359 elif self.file_paths:360 # Local mode (for integration testing, or pre-downloaded XML)361 for path in self.file_paths:362 path = Path(path)363 with open(path, "rb") as file:364 chunks += self._parse_dgml(365 content=file.read(),366 document_name=path.name,367 )368 369 return chunks370 