Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
zep_cloud.py478 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import logging4import warnings5from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Tuple6 7from langchain_core.documents import Document8from langchain_core.embeddings import Embeddings9from langchain_core.vectorstores import VectorStore10 11if TYPE_CHECKING:12    from zep_cloud import CreateDocumentRequest, DocumentCollectionResponse, SearchType13 14logger = logging.getLogger()15 16 17class ZepCloudVectorStore(VectorStore):18    """`Zep` vector store.19 20    It provides methods for adding texts or documents to the store,21    searching for similar documents, and deleting documents.22 23    Search scores are calculated using cosine similarity normalized to [0, 1].24 25    Args:26        collection_name (str): The name of the collection in the Zep store.27        api_key (str): The API key for the Zep API.28    """29 30    def __init__(31        self,32        collection_name: str,33        api_key: str,34    ) -> None:35        super().__init__()36        if not collection_name:37            raise ValueError(38                "collection_name must be specified when using ZepVectorStore."39            )40        try:41            from zep_cloud.client import AsyncZep, Zep42        except ImportError:43            raise ImportError(44                "Could not import zep-python python package. "45                "Please install it with `pip install zep-python`."46            )47        self._client = Zep(api_key=api_key)48        self._client_async = AsyncZep(api_key=api_key)49 50        self.collection_name = collection_name51 52        self._load_collection()53 54    @property55    def embeddings(self) -> Optional[Embeddings]:56        """Unavailable for ZepCloud"""57        return None58 59    def _load_collection(self) -> DocumentCollectionResponse:60        """61        Load the collection from the Zep backend.62        """63        from zep_cloud import NotFoundError64 65        try:66            collection = self._client.document.get_collection(self.collection_name)67        except NotFoundError:68            logger.info(69                f"Collection {self.collection_name} not found. Creating new collection."70            )71            collection = self._create_collection()72 73        return collection74 75    def _create_collection(self) -> DocumentCollectionResponse:76        """77        Create a new collection in the Zep backend.78        """79        self._client.document.add_collection(self.collection_name)80        collection = self._client.document.get_collection(self.collection_name)81        return collection82 83    def _generate_documents_to_add(84        self,85        texts: Iterable[str],86        metadatas: Optional[List[Dict[Any, Any]]] = None,87        document_ids: Optional[List[str]] = None,88    ) -> List[CreateDocumentRequest]:89        from zep_cloud import CreateDocumentRequest as ZepDocument90 91        documents: List[ZepDocument] = []92        for i, d in enumerate(texts):93            documents.append(94                ZepDocument(95                    content=d,96                    metadata=metadatas[i] if metadatas else None,97                    document_id=document_ids[i] if document_ids else None,98                )99            )100        return documents101 102    def add_texts(103        self,104        texts: Iterable[str],105        metadatas: Optional[List[Dict[str, Any]]] = None,106        document_ids: Optional[List[str]] = None,107        **kwargs: Any,108    ) -> List[str]:109        """Run more texts through the embeddings and add to the vectorstore.110 111        Args:112            texts: Iterable of strings to add to the vectorstore.113            metadatas: Optional list of metadatas associated with the texts.114            document_ids: Optional list of document ids associated with the texts.115            kwargs: vectorstore specific parameters116 117        Returns:118            List of ids from adding the texts into the vectorstore.119        """120 121        documents = self._generate_documents_to_add(texts, metadatas, document_ids)122        uuids = self._client.document.add_documents(123            self.collection_name, request=documents124        )125 126        return uuids127 128    async def aadd_texts(129        self,130        texts: Iterable[str],131        metadatas: Optional[List[Dict[str, Any]]] = None,132        document_ids: Optional[List[str]] = None,133        **kwargs: Any,134    ) -> List[str]:135        """Run more texts through the embeddings and add to the vectorstore."""136        documents = self._generate_documents_to_add(texts, metadatas, document_ids)137        uuids = await self._client_async.document.add_documents(138            self.collection_name, request=documents139        )140 141        return uuids142 143    def search(144        self,145        query: str,146        search_type: SearchType,147        metadata: Optional[Dict[str, Any]] = None,148        k: int = 3,149        **kwargs: Any,150    ) -> List[Document]:151        """Return docs most similar to query using specified search type."""152        if search_type == "similarity":153            return self.similarity_search(query, k=k, metadata=metadata, **kwargs)154        elif search_type == "mmr":155            return self.max_marginal_relevance_search(156                query, k=k, metadata=metadata, **kwargs157            )158        else:159            raise ValueError(160                f"search_type of {search_type} not allowed. Expected "161                "search_type to be 'similarity' or 'mmr'."162            )163 164    async def asearch(165        self,166        query: str,167        search_type: str,168        metadata: Optional[Dict[str, Any]] = None,169        k: int = 3,170        **kwargs: Any,171    ) -> List[Document]:172        """Return docs most similar to query using specified search type."""173        if search_type == "similarity":174            return await self.asimilarity_search(175                query, k=k, metadata=metadata, **kwargs176            )177        elif search_type == "mmr":178            return await self.amax_marginal_relevance_search(179                query, k=k, metadata=metadata, **kwargs180            )181        else:182            raise ValueError(183                f"search_type of {search_type} not allowed. Expected "184                "search_type to be 'similarity' or 'mmr'."185            )186 187    def similarity_search(188        self,189        query: str,190        k: int = 4,191        metadata: Optional[Dict[str, Any]] = None,192        **kwargs: Any,193    ) -> List[Document]:194        """Return docs most similar to query."""195 196        results = self._similarity_search_with_relevance_scores(197            query, k=k, metadata=metadata, **kwargs198        )199        return [doc for doc, _ in results]200 201    def similarity_search_with_score(202        self,203        query: str,204        k: int = 4,205        metadata: Optional[Dict[str, Any]] = None,206        **kwargs: Any,207    ) -> List[Tuple[Document, float]]:208        """Run similarity search with distance."""209 210        return self._similarity_search_with_relevance_scores(211            query, k=k, metadata=metadata, **kwargs212        )213 214    def _similarity_search_with_relevance_scores(215        self,216        query: str,217        k: int = 4,218        metadata: Optional[Dict[str, Any]] = None,219        **kwargs: Any,220    ) -> List[Tuple[Document, float]]:221        """222        Default similarity search with relevance scores. Modify if necessary223        in subclass.224        Return docs and relevance scores in the range [0, 1].225 226        0 is dissimilar, 1 is most similar.227 228        Args:229            query: input text230            k: Number of Documents to return. Defaults to 4.231            metadata: Optional, metadata filter232            **kwargs: kwargs to be passed to similarity search. Should include:233                score_threshold: Optional, a floating point value between 0 to 1 and234                    filter the resulting set of retrieved docs235 236        Returns:237            List of Tuples of (doc, similarity_score)238        """239 240        results = self._client.document.search(241            collection_name=self.collection_name,242            text=query,243            limit=k,244            metadata=metadata,245            **kwargs,246        )247 248        return [249            (250                Document(251                    page_content=str(doc.content),252                    metadata=doc.metadata,253                ),254                doc.score or 0.0,255            )256            for doc in results.results or []257        ]258 259    async def asimilarity_search_with_relevance_scores(260        self,261        query: str,262        k: int = 4,263        metadata: Optional[Dict[str, Any]] = None,264        **kwargs: Any,265    ) -> List[Tuple[Document, float]]:266        """Return docs most similar to query."""267 268        results = await self._client_async.document.search(269            collection_name=self.collection_name,270            text=query,271            limit=k,272            metadata=metadata,273            **kwargs,274        )275 276        return [277            (278                Document(279                    page_content=str(doc.content),280                    metadata=doc.metadata,281                ),282                doc.score or 0.0,283            )284            for doc in results.results or []285        ]286 287    async def asimilarity_search(288        self,289        query: str,290        k: int = 4,291        metadata: Optional[Dict[str, Any]] = None,292        **kwargs: Any,293    ) -> List[Document]:294        """Return docs most similar to query."""295 296        results = await self.asimilarity_search_with_relevance_scores(297            query, k, metadata=metadata, **kwargs298        )299 300        return [doc for doc, _ in results]301 302    def similarity_search_by_vector(303        self,304        embedding: List[float],305        k: int = 4,306        metadata: Optional[Dict[str, Any]] = None,307        **kwargs: Any,308    ) -> List[Document]:309        """Unsupported in Zep Cloud"""310        warnings.warn("similarity_search_by_vector is not supported in Zep Cloud")311        return []312 313    async def asimilarity_search_by_vector(314        self,315        embedding: List[float],316        k: int = 4,317        metadata: Optional[Dict[str, Any]] = None,318        **kwargs: Any,319    ) -> List[Document]:320        """Unsupported in Zep Cloud"""321        warnings.warn("asimilarity_search_by_vector is not supported in Zep Cloud")322        return []323 324    def max_marginal_relevance_search(325        self,326        query: str,327        k: int = 4,328        fetch_k: int = 20,329        lambda_mult: float = 0.5,330        metadata: Optional[Dict[str, Any]] = None,331        **kwargs: Any,332    ) -> List[Document]:333        """Return docs selected using the maximal marginal relevance.334 335        Maximal marginal relevance optimizes for similarity to query AND diversity336        among selected documents.337 338        Args:339            query: Text to look up documents similar to.340            k: Number of Documents to return. Defaults to 4.341            fetch_k: Number of Documents to fetch to pass to MMR algorithm.342                     Zep determines this automatically and this parameter is343                        ignored.344            lambda_mult: Number between 0 and 1 that determines the degree345                        of diversity among the results with 0 corresponding346                        to maximum diversity and 1 to minimum diversity.347                        Defaults to 0.5.348            metadata: Optional, metadata to filter the resulting set of retrieved docs349        Returns:350            List of Documents selected by maximal marginal relevance.351        """352 353        results = self._client.document.search(354            collection_name=self.collection_name,355            text=query,356            limit=k,357            metadata=metadata,358            search_type="mmr",359            mmr_lambda=lambda_mult,360            **kwargs,361        )362 363        return [364            Document(page_content=str(d.content), metadata=d.metadata)365            for d in results.results or []366        ]367 368    async def amax_marginal_relevance_search(369        self,370        query: str,371        k: int = 4,372        fetch_k: int = 20,373        lambda_mult: float = 0.5,374        metadata: Optional[Dict[str, Any]] = None,375        **kwargs: Any,376    ) -> List[Document]:377        """Return docs selected using the maximal marginal relevance."""378 379        results = await self._client_async.document.search(380            collection_name=self.collection_name,381            text=query,382            limit=k,383            metadata=metadata,384            search_type="mmr",385            mmr_lambda=lambda_mult,386            **kwargs,387        )388 389        return [390            Document(page_content=str(d.content), metadata=d.metadata)391            for d in results.results or []392        ]393 394    def max_marginal_relevance_search_by_vector(395        self,396        embedding: List[float],397        k: int = 4,398        fetch_k: int = 20,399        lambda_mult: float = 0.5,400        metadata: Optional[Dict[str, Any]] = None,401        **kwargs: Any,402    ) -> List[Document]:403        """Unsupported in Zep Cloud"""404        warnings.warn(405            "max_marginal_relevance_search_by_vector is not supported in Zep Cloud"406        )407        return []408 409    async def amax_marginal_relevance_search_by_vector(410        self,411        embedding: List[float],412        k: int = 4,413        fetch_k: int = 20,414        lambda_mult: float = 0.5,415        metadata: Optional[Dict[str, Any]] = None,416        **kwargs: Any,417    ) -> List[Document]:418        """Unsupported in Zep Cloud"""419        warnings.warn(420            "amax_marginal_relevance_search_by_vector is not supported in Zep Cloud"421        )422        return []423 424    @classmethod425    def from_texts(426        cls,427        texts: List[str],428        embedding: Embeddings,429        metadatas: Optional[List[dict]] = None,430        collection_name: str = "",431        api_key: Optional[str] = None,432        **kwargs: Any,433    ) -> ZepCloudVectorStore:434        """435        Class method that returns a ZepVectorStore instance initialized from texts.436 437        If the collection does not exist, it will be created.438 439        Args:440            texts (List[str]): The list of texts to add to the vectorstore.441            metadatas (Optional[List[Dict[str, Any]]]): Optional list of metadata442               associated with the texts.443            collection_name (str): The name of the collection in the Zep store.444            api_key (str): The API key for the Zep API.445            kwargs: Additional parameters specific to the vectorstore.446 447        Returns:448            ZepVectorStore: An instance of ZepVectorStore.449        """450        if not api_key:451            raise ValueError("api_key must be specified when using ZepVectorStore.")452        vecstore = cls(453            collection_name=collection_name,454            api_key=api_key,455        )456        vecstore.add_texts(texts, metadatas)457        return vecstore458 459    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> None:460        """Delete by Zep vector UUIDs.461 462        Parameters463        ----------464        ids : Optional[List[str]]465            The UUIDs of the vectors to delete.466 467        Raises468        ------469        ValueError470            If no UUIDs are provided.471        """472 473        if ids is None or len(ids) == 0:474            raise ValueError("No uuids provided to delete.")475 476        for u in ids:477            self._client.document.delete_document(self.collection_name, u)478 
codekingpro/portable-devtools · Team Ai