codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import logging4import warnings5from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Tuple6 7from langchain_core.documents import Document8from langchain_core.embeddings import Embeddings9from langchain_core.vectorstores import VectorStore10 11if TYPE_CHECKING:12 from zep_cloud import CreateDocumentRequest, DocumentCollectionResponse, SearchType13 14logger = logging.getLogger()15 16 17class ZepCloudVectorStore(VectorStore):18 """`Zep` vector store.19 20 It provides methods for adding texts or documents to the store,21 searching for similar documents, and deleting documents.22 23 Search scores are calculated using cosine similarity normalized to [0, 1].24 25 Args:26 collection_name (str): The name of the collection in the Zep store.27 api_key (str): The API key for the Zep API.28 """29 30 def __init__(31 self,32 collection_name: str,33 api_key: str,34 ) -> None:35 super().__init__()36 if not collection_name:37 raise ValueError(38 "collection_name must be specified when using ZepVectorStore."39 )40 try:41 from zep_cloud.client import AsyncZep, Zep42 except ImportError:43 raise ImportError(44 "Could not import zep-python python package. "45 "Please install it with `pip install zep-python`."46 )47 self._client = Zep(api_key=api_key)48 self._client_async = AsyncZep(api_key=api_key)49 50 self.collection_name = collection_name51 52 self._load_collection()53 54 @property55 def embeddings(self) -> Optional[Embeddings]:56 """Unavailable for ZepCloud"""57 return None58 59 def _load_collection(self) -> DocumentCollectionResponse:60 """61 Load the collection from the Zep backend.62 """63 from zep_cloud import NotFoundError64 65 try:66 collection = self._client.document.get_collection(self.collection_name)67 except NotFoundError:68 logger.info(69 f"Collection {self.collection_name} not found. Creating new collection."70 )71 collection = self._create_collection()72 73 return collection74 75 def _create_collection(self) -> DocumentCollectionResponse:76 """77 Create a new collection in the Zep backend.78 """79 self._client.document.add_collection(self.collection_name)80 collection = self._client.document.get_collection(self.collection_name)81 return collection82 83 def _generate_documents_to_add(84 self,85 texts: Iterable[str],86 metadatas: Optional[List[Dict[Any, Any]]] = None,87 document_ids: Optional[List[str]] = None,88 ) -> List[CreateDocumentRequest]:89 from zep_cloud import CreateDocumentRequest as ZepDocument90 91 documents: List[ZepDocument] = []92 for i, d in enumerate(texts):93 documents.append(94 ZepDocument(95 content=d,96 metadata=metadatas[i] if metadatas else None,97 document_id=document_ids[i] if document_ids else None,98 )99 )100 return documents101 102 def add_texts(103 self,104 texts: Iterable[str],105 metadatas: Optional[List[Dict[str, Any]]] = None,106 document_ids: Optional[List[str]] = None,107 **kwargs: Any,108 ) -> List[str]:109 """Run more texts through the embeddings and add to the vectorstore.110 111 Args:112 texts: Iterable of strings to add to the vectorstore.113 metadatas: Optional list of metadatas associated with the texts.114 document_ids: Optional list of document ids associated with the texts.115 kwargs: vectorstore specific parameters116 117 Returns:118 List of ids from adding the texts into the vectorstore.119 """120 121 documents = self._generate_documents_to_add(texts, metadatas, document_ids)122 uuids = self._client.document.add_documents(123 self.collection_name, request=documents124 )125 126 return uuids127 128 async def aadd_texts(129 self,130 texts: Iterable[str],131 metadatas: Optional[List[Dict[str, Any]]] = None,132 document_ids: Optional[List[str]] = None,133 **kwargs: Any,134 ) -> List[str]:135 """Run more texts through the embeddings and add to the vectorstore."""136 documents = self._generate_documents_to_add(texts, metadatas, document_ids)137 uuids = await self._client_async.document.add_documents(138 self.collection_name, request=documents139 )140 141 return uuids142 143 def search(144 self,145 query: str,146 search_type: SearchType,147 metadata: Optional[Dict[str, Any]] = None,148 k: int = 3,149 **kwargs: Any,150 ) -> List[Document]:151 """Return docs most similar to query using specified search type."""152 if search_type == "similarity":153 return self.similarity_search(query, k=k, metadata=metadata, **kwargs)154 elif search_type == "mmr":155 return self.max_marginal_relevance_search(156 query, k=k, metadata=metadata, **kwargs157 )158 else:159 raise ValueError(160 f"search_type of {search_type} not allowed. Expected "161 "search_type to be 'similarity' or 'mmr'."162 )163 164 async def asearch(165 self,166 query: str,167 search_type: str,168 metadata: Optional[Dict[str, Any]] = None,169 k: int = 3,170 **kwargs: Any,171 ) -> List[Document]:172 """Return docs most similar to query using specified search type."""173 if search_type == "similarity":174 return await self.asimilarity_search(175 query, k=k, metadata=metadata, **kwargs176 )177 elif search_type == "mmr":178 return await self.amax_marginal_relevance_search(179 query, k=k, metadata=metadata, **kwargs180 )181 else:182 raise ValueError(183 f"search_type of {search_type} not allowed. Expected "184 "search_type to be 'similarity' or 'mmr'."185 )186 187 def similarity_search(188 self,189 query: str,190 k: int = 4,191 metadata: Optional[Dict[str, Any]] = None,192 **kwargs: Any,193 ) -> List[Document]:194 """Return docs most similar to query."""195 196 results = self._similarity_search_with_relevance_scores(197 query, k=k, metadata=metadata, **kwargs198 )199 return [doc for doc, _ in results]200 201 def similarity_search_with_score(202 self,203 query: str,204 k: int = 4,205 metadata: Optional[Dict[str, Any]] = None,206 **kwargs: Any,207 ) -> List[Tuple[Document, float]]:208 """Run similarity search with distance."""209 210 return self._similarity_search_with_relevance_scores(211 query, k=k, metadata=metadata, **kwargs212 )213 214 def _similarity_search_with_relevance_scores(215 self,216 query: str,217 k: int = 4,218 metadata: Optional[Dict[str, Any]] = None,219 **kwargs: Any,220 ) -> List[Tuple[Document, float]]:221 """222 Default similarity search with relevance scores. Modify if necessary223 in subclass.224 Return docs and relevance scores in the range [0, 1].225 226 0 is dissimilar, 1 is most similar.227 228 Args:229 query: input text230 k: Number of Documents to return. Defaults to 4.231 metadata: Optional, metadata filter232 **kwargs: kwargs to be passed to similarity search. Should include:233 score_threshold: Optional, a floating point value between 0 to 1 and234 filter the resulting set of retrieved docs235 236 Returns:237 List of Tuples of (doc, similarity_score)238 """239 240 results = self._client.document.search(241 collection_name=self.collection_name,242 text=query,243 limit=k,244 metadata=metadata,245 **kwargs,246 )247 248 return [249 (250 Document(251 page_content=str(doc.content),252 metadata=doc.metadata,253 ),254 doc.score or 0.0,255 )256 for doc in results.results or []257 ]258 259 async def asimilarity_search_with_relevance_scores(260 self,261 query: str,262 k: int = 4,263 metadata: Optional[Dict[str, Any]] = None,264 **kwargs: Any,265 ) -> List[Tuple[Document, float]]:266 """Return docs most similar to query."""267 268 results = await self._client_async.document.search(269 collection_name=self.collection_name,270 text=query,271 limit=k,272 metadata=metadata,273 **kwargs,274 )275 276 return [277 (278 Document(279 page_content=str(doc.content),280 metadata=doc.metadata,281 ),282 doc.score or 0.0,283 )284 for doc in results.results or []285 ]286 287 async def asimilarity_search(288 self,289 query: str,290 k: int = 4,291 metadata: Optional[Dict[str, Any]] = None,292 **kwargs: Any,293 ) -> List[Document]:294 """Return docs most similar to query."""295 296 results = await self.asimilarity_search_with_relevance_scores(297 query, k, metadata=metadata, **kwargs298 )299 300 return [doc for doc, _ in results]301 302 def similarity_search_by_vector(303 self,304 embedding: List[float],305 k: int = 4,306 metadata: Optional[Dict[str, Any]] = None,307 **kwargs: Any,308 ) -> List[Document]:309 """Unsupported in Zep Cloud"""310 warnings.warn("similarity_search_by_vector is not supported in Zep Cloud")311 return []312 313 async def asimilarity_search_by_vector(314 self,315 embedding: List[float],316 k: int = 4,317 metadata: Optional[Dict[str, Any]] = None,318 **kwargs: Any,319 ) -> List[Document]:320 """Unsupported in Zep Cloud"""321 warnings.warn("asimilarity_search_by_vector is not supported in Zep Cloud")322 return []323 324 def max_marginal_relevance_search(325 self,326 query: str,327 k: int = 4,328 fetch_k: int = 20,329 lambda_mult: float = 0.5,330 metadata: Optional[Dict[str, Any]] = None,331 **kwargs: Any,332 ) -> List[Document]:333 """Return docs selected using the maximal marginal relevance.334 335 Maximal marginal relevance optimizes for similarity to query AND diversity336 among selected documents.337 338 Args:339 query: Text to look up documents similar to.340 k: Number of Documents to return. Defaults to 4.341 fetch_k: Number of Documents to fetch to pass to MMR algorithm.342 Zep determines this automatically and this parameter is343 ignored.344 lambda_mult: Number between 0 and 1 that determines the degree345 of diversity among the results with 0 corresponding346 to maximum diversity and 1 to minimum diversity.347 Defaults to 0.5.348 metadata: Optional, metadata to filter the resulting set of retrieved docs349 Returns:350 List of Documents selected by maximal marginal relevance.351 """352 353 results = self._client.document.search(354 collection_name=self.collection_name,355 text=query,356 limit=k,357 metadata=metadata,358 search_type="mmr",359 mmr_lambda=lambda_mult,360 **kwargs,361 )362 363 return [364 Document(page_content=str(d.content), metadata=d.metadata)365 for d in results.results or []366 ]367 368 async def amax_marginal_relevance_search(369 self,370 query: str,371 k: int = 4,372 fetch_k: int = 20,373 lambda_mult: float = 0.5,374 metadata: Optional[Dict[str, Any]] = None,375 **kwargs: Any,376 ) -> List[Document]:377 """Return docs selected using the maximal marginal relevance."""378 379 results = await self._client_async.document.search(380 collection_name=self.collection_name,381 text=query,382 limit=k,383 metadata=metadata,384 search_type="mmr",385 mmr_lambda=lambda_mult,386 **kwargs,387 )388 389 return [390 Document(page_content=str(d.content), metadata=d.metadata)391 for d in results.results or []392 ]393 394 def max_marginal_relevance_search_by_vector(395 self,396 embedding: List[float],397 k: int = 4,398 fetch_k: int = 20,399 lambda_mult: float = 0.5,400 metadata: Optional[Dict[str, Any]] = None,401 **kwargs: Any,402 ) -> List[Document]:403 """Unsupported in Zep Cloud"""404 warnings.warn(405 "max_marginal_relevance_search_by_vector is not supported in Zep Cloud"406 )407 return []408 409 async def amax_marginal_relevance_search_by_vector(410 self,411 embedding: List[float],412 k: int = 4,413 fetch_k: int = 20,414 lambda_mult: float = 0.5,415 metadata: Optional[Dict[str, Any]] = None,416 **kwargs: Any,417 ) -> List[Document]:418 """Unsupported in Zep Cloud"""419 warnings.warn(420 "amax_marginal_relevance_search_by_vector is not supported in Zep Cloud"421 )422 return []423 424 @classmethod425 def from_texts(426 cls,427 texts: List[str],428 embedding: Embeddings,429 metadatas: Optional[List[dict]] = None,430 collection_name: str = "",431 api_key: Optional[str] = None,432 **kwargs: Any,433 ) -> ZepCloudVectorStore:434 """435 Class method that returns a ZepVectorStore instance initialized from texts.436 437 If the collection does not exist, it will be created.438 439 Args:440 texts (List[str]): The list of texts to add to the vectorstore.441 metadatas (Optional[List[Dict[str, Any]]]): Optional list of metadata442 associated with the texts.443 collection_name (str): The name of the collection in the Zep store.444 api_key (str): The API key for the Zep API.445 kwargs: Additional parameters specific to the vectorstore.446 447 Returns:448 ZepVectorStore: An instance of ZepVectorStore.449 """450 if not api_key:451 raise ValueError("api_key must be specified when using ZepVectorStore.")452 vecstore = cls(453 collection_name=collection_name,454 api_key=api_key,455 )456 vecstore.add_texts(texts, metadatas)457 return vecstore458 459 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> None:460 """Delete by Zep vector UUIDs.461 462 Parameters463 ----------464 ids : Optional[List[str]]465 The UUIDs of the vectors to delete.466 467 Raises468 ------469 ValueError470 If no UUIDs are provided.471 """472 473 if ids is None or len(ids) == 0:474 raise ValueError("No uuids provided to delete.")475 476 for u in ids:477 self._client.document.delete_document(self.collection_name, u)478 