codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import logging4from enum import Enum5from typing import (6 TYPE_CHECKING,7 Any,8 Dict,9 Generator,10 Iterable,11 List,12 Optional,13 Tuple,14 Union,15)16 17import numpy as np18from langchain_core.documents import Document19from langchain_core.vectorstores import VectorStore20 21from langchain_community.vectorstores.utils import maximal_marginal_relevance22 23if TYPE_CHECKING:24 from langchain_core.embeddings import Embeddings25 from pymongo.collection import Collection26 27 28# Before Python 3.11 native StrEnum is not available29class CosmosDBSimilarityType(str, Enum):30 """Cosmos DB Similarity Type as enumerator."""31 32 COS = "COS"33 """CosineSimilarity"""34 IP = "IP"35 """inner - product"""36 L2 = "L2"37 """Euclidean distance"""38 39 40class CosmosDBVectorSearchType(str, Enum):41 """Cosmos DB Vector Search Type as enumerator."""42 43 VECTOR_IVF = "vector-ivf"44 """IVF vector index"""45 VECTOR_HNSW = "vector-hnsw"46 """HNSW vector index"""47 VECTOR_DISKANN = "vector-diskann"48 """DISKANN vector index"""49 50 51logger = logging.getLogger(__name__)52 53DEFAULT_INSERT_BATCH_SIZE = 12854 55 56class AzureCosmosDBVectorSearch(VectorStore):57 """`Azure Cosmos DB for MongoDB vCore` vector store.58 59 To use, you should have both:60 - the ``pymongo`` python package installed61 - a connection string associated with a MongoDB VCore Cluster62 63 Example:64 . code-block:: python65 66 from langchain_community.vectorstores import67 AzureCosmosDBVectorSearch68 from langchain_community.embeddings.openai import OpenAIEmbeddings69 from pymongo import MongoClient70 71 mongo_client = MongoClient("<YOUR-CONNECTION-STRING>")72 collection = mongo_client["<db_name>"]["<collection_name>"]73 embeddings = OpenAIEmbeddings()74 vectorstore = AzureCosmosDBVectorSearch(collection, embeddings)75 """76 77 def __init__(78 self,79 collection: Collection,80 embedding: Embeddings,81 *,82 index_name: str = "vectorSearchIndex",83 text_key: str = "textContent",84 embedding_key: str = "vectorContent",85 application_name: str = "LangChain-CDBMongoVCore-VectorStore-Python",86 ):87 """Constructor for AzureCosmosDBVectorSearch88 89 Args:90 collection: MongoDB collection to add the texts to.91 embedding: Text embedding model to use.92 index_name: Name of the Atlas Search index.93 text_key: MongoDB field that will contain the text94 for each document.95 embedding_key: MongoDB field that will contain the embedding96 for each document.97 """98 self._collection = collection99 self._embedding = embedding100 self._index_name = index_name101 self._text_key = text_key102 self._embedding_key = embedding_key103 self._application_name = application_name104 105 @property106 def embeddings(self) -> Embeddings:107 return self._embedding108 109 def get_index_name(self) -> str:110 """Returns the index name111 112 Returns:113 Returns the index name114 115 """116 return self._index_name117 118 @classmethod119 def from_connection_string(120 cls,121 connection_string: str,122 namespace: str,123 embedding: Embeddings,124 application_name: str = "LangChain-CDBMongoVCore-VectorStore-Python",125 **kwargs: Any,126 ) -> AzureCosmosDBVectorSearch:127 """Creates an Instance of AzureCosmosDBVectorSearch128 from a Connection String129 130 Args:131 connection_string: The MongoDB vCore instance connection string132 namespace: The namespace (database.collection)133 embedding: The embedding utility134 application_name: The user agent for telemetry135 **kwargs: Dynamic keyword arguments136 137 Returns:138 an instance of the vector store139 140 """141 try:142 from pymongo import MongoClient143 except ImportError:144 raise ImportError(145 "Could not import pymongo, please install it with "146 "`pip install pymongo`."147 )148 appname = application_name149 client: MongoClient = MongoClient(connection_string, appname=appname)150 db_name, collection_name = namespace.split(".")151 collection = client[db_name][collection_name]152 return cls(collection, embedding, **kwargs)153 154 def index_exists(self) -> bool:155 """Verifies if the specified index name during instance156 construction exists on the collection157 158 Returns:159 Returns True on success and False if no such index exists160 on the collection161 """162 cursor = self._collection.list_indexes()163 index_name = self._index_name164 165 for res in cursor:166 current_index_name = res.pop("name")167 if current_index_name == index_name:168 return True169 170 return False171 172 def delete_index(self) -> None:173 """Deletes the index specified during instance construction if it exists"""174 if self.index_exists():175 self._collection.drop_index(self._index_name)176 # Raises OperationFailure on an error (e.g. trying to drop177 # an index that does not exist)178 179 def create_index(180 self,181 num_lists: int = 100,182 dimensions: int = 1536,183 similarity: CosmosDBSimilarityType = CosmosDBSimilarityType.COS,184 kind: str = "vector-ivf",185 m: int = 16,186 ef_construction: int = 64,187 max_degree: int = 32,188 l_build: int = 50,189 ) -> dict[str, Any]:190 """Creates an index using the index name specified at191 instance construction192 193 Setting the numLists parameter correctly is important for achieving194 good accuracy and performance.195 Since the vector store uses IVF as the indexing strategy,196 you should create the index only after you197 have loaded a large enough sample documents to ensure that the198 centroids for the respective buckets are199 faily distributed.200 201 We recommend that numLists is set to documentCount/1000 for up202 to 1 million documents203 and to sqrt(documentCount) for more than 1 million documents.204 As the number of items in your database grows, you should205 tune numLists to be larger206 in order to achieve good latency performance for vector search.207 208 If you're experimenting with a new scenario or creating a209 small demo, you can start with numLists210 set to 1 to perform a brute-force search across all vectors.211 This should provide you with the most212 accurate results from the vector search, however be aware that213 the search speed and latency will be slow.214 After your initial setup, you should go ahead and tune215 the numLists parameter using the above guidance.216 217 Args:218 kind: Type of vector index to create.219 Possible options are:220 - vector-ivf221 - vector-hnsw: available as a preview feature only,222 to enable visit https://learn.microsoft.com/en-us/azure/azure-resource-manager/management/preview-features223 - vector-diskann: available as a preview feature only224 num_lists: This integer is the number of clusters that the225 inverted file (IVF) index uses to group the vector data.226 We recommend that numLists is set to documentCount/1000227 for up to 1 million documents and to sqrt(documentCount)228 for more than 1 million documents.229 Using a numLists value of 1 is akin to performing230 brute-force search, which has limited performance231 dimensions: Number of dimensions for vector similarity.232 The maximum number of supported dimensions is 2000233 similarity: Similarity metric to use with the IVF index.234 235 Possible options are:236 - CosmosDBSimilarityType.COS (cosine distance),237 - CosmosDBSimilarityType.L2 (Euclidean distance), and238 - CosmosDBSimilarityType.IP (inner product).239 m: The max number of connections per layer (16 by default, minimum240 value is 2, maximum value is 100). Higher m is suitable for datasets241 with high dimensionality and/or high accuracy requirements.242 ef_construction: the size of the dynamic candidate list for constructing243 the graph (64 by default, minimum value is 4, maximum244 value is 1000). Higher ef_construction will result in245 better index quality and higher accuracy, but it will246 also increase the time required to build the index.247 ef_construction has to be at least 2 * m248 max_degree: Max number of neighbors.249 Default value is 32, range from 20 to 2048.250 Only vector-diskann search supports this for now.251 l_build: l value for index building.252 Default value is 50, range from 10 to 500.253 Only vector-diskann search supports this for now.254 Returns:255 An object describing the created index256 257 """258 # check the kind of vector search to be performed259 # prepare the command accordingly260 create_index_commands = {}261 if kind == CosmosDBVectorSearchType.VECTOR_IVF:262 create_index_commands = self._get_vector_index_ivf(263 kind, num_lists, similarity, dimensions264 )265 elif kind == CosmosDBVectorSearchType.VECTOR_HNSW:266 create_index_commands = self._get_vector_index_hnsw(267 kind, m, ef_construction, similarity, dimensions268 )269 elif kind == CosmosDBVectorSearchType.VECTOR_DISKANN:270 create_index_commands = self._get_vector_index_diskann(271 kind, max_degree, l_build, similarity, dimensions272 )273 274 # retrieve the database object275 current_database = self._collection.database276 277 # invoke the command from the database object278 create_index_responses: dict[str, Any] = current_database.command(279 create_index_commands280 )281 282 return create_index_responses283 284 def _get_vector_index_ivf(285 self, kind: str, num_lists: int, similarity: str, dimensions: int286 ) -> Dict[str, Any]:287 command = {288 "createIndexes": self._collection.name,289 "indexes": [290 {291 "name": self._index_name,292 "key": {self._embedding_key: "cosmosSearch"},293 "cosmosSearchOptions": {294 "kind": kind,295 "numLists": num_lists,296 "similarity": similarity,297 "dimensions": dimensions,298 },299 }300 ],301 }302 return command303 304 def _get_vector_index_hnsw(305 self, kind: str, m: int, ef_construction: int, similarity: str, dimensions: int306 ) -> Dict[str, Any]:307 command = {308 "createIndexes": self._collection.name,309 "indexes": [310 {311 "name": self._index_name,312 "key": {self._embedding_key: "cosmosSearch"},313 "cosmosSearchOptions": {314 "kind": kind,315 "m": m,316 "efConstruction": ef_construction,317 "similarity": similarity,318 "dimensions": dimensions,319 },320 }321 ],322 }323 return command324 325 def _get_vector_index_diskann(326 self, kind: str, max_degree: int, l_build: int, similarity: str, dimensions: int327 ) -> Dict[str, Any]:328 command = {329 "createIndexes": self._collection.name,330 "indexes": [331 {332 "name": self._index_name,333 "key": {self._embedding_key: "cosmosSearch"},334 "cosmosSearchOptions": {335 "kind": kind,336 "maxDegree": max_degree,337 "lBuild": l_build,338 "similarity": similarity,339 "dimensions": dimensions,340 },341 }342 ],343 }344 return command345 346 def create_filter_index(347 self,348 property_to_filter: str,349 index_name: str,350 ) -> dict[str, Any]:351 command = {352 "createIndexes": self._collection.name,353 "indexes": [354 {355 "key": {property_to_filter: 1},356 "name": index_name,357 }358 ],359 }360 # retrieve the database object361 current_database = self._collection.database362 363 # invoke the command from the database object364 create_index_responses: dict[str, Any] = current_database.command(command)365 return create_index_responses366 367 def add_texts(368 self,369 texts: Iterable[str],370 metadatas: Optional[List[Dict[str, Any]]] = None,371 **kwargs: Any,372 ) -> List:373 batch_size = kwargs.get("batch_size", DEFAULT_INSERT_BATCH_SIZE)374 _metadatas: Union[List, Generator] = metadatas or ({} for _ in texts)375 texts_batch = []376 metadatas_batch = []377 result_ids = []378 for i, (text, metadata) in enumerate(zip(texts, _metadatas)):379 texts_batch.append(text)380 metadatas_batch.append(metadata)381 if (i + 1) % batch_size == 0:382 result_ids.extend(self._insert_texts(texts_batch, metadatas_batch))383 texts_batch = []384 metadatas_batch = []385 if texts_batch:386 result_ids.extend(self._insert_texts(texts_batch, metadatas_batch))387 return result_ids388 389 def _insert_texts(self, texts: List[str], metadatas: List[Dict[str, Any]]) -> List:390 """Used to Load Documents into the collection391 392 Args:393 texts: The list of documents strings to load394 metadatas: The list of metadata objects associated with each document395 396 Returns:397 398 """399 # If the text is empty, then exit early400 if not texts:401 return []402 403 # Embed and create the documents404 embeddings = self._embedding.embed_documents(texts)405 to_insert = [406 {self._text_key: t, self._embedding_key: embedding, "metadata": m}407 for t, m, embedding in zip(texts, metadatas, embeddings)408 ]409 # insert the documents in Cosmos DB410 insert_result = self._collection.insert_many(to_insert)411 return insert_result.inserted_ids412 413 @classmethod414 def from_texts(415 cls,416 texts: List[str],417 embedding: Embeddings,418 metadatas: Optional[List[dict]] = None,419 collection: Optional[Collection] = None,420 **kwargs: Any,421 ) -> AzureCosmosDBVectorSearch:422 if collection is None:423 raise ValueError("Must provide 'collection' named parameter.")424 vectorstore = cls(collection, embedding, **kwargs)425 vectorstore.add_texts(texts, metadatas=metadatas)426 return vectorstore427 428 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> Optional[bool]:429 if ids is None:430 raise ValueError("No document ids provided to delete.")431 432 for document_id in ids:433 self.delete_document_by_id(document_id)434 return True435 436 def delete_document_by_id(self, document_id: Optional[str] = None) -> None:437 """Removes a Specific Document by Id438 439 Args:440 document_id: The document identifier441 """442 try:443 from bson.objectid import ObjectId444 except ImportError as e:445 raise ImportError(446 "Unable to import bson, please install with `pip install bson`."447 ) from e448 if document_id is None:449 raise ValueError("No document id provided to delete.")450 451 self._collection.delete_one({"_id": ObjectId(document_id)})452 453 def _similarity_search_with_score(454 self,455 embeddings: List[float],456 k: int = 4,457 kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,458 pre_filter: Optional[Dict] = None,459 ef_search: int = 40,460 score_threshold: float = 0.0,461 l_search: int = 40,462 with_embedding: bool = False,463 ) -> List[Tuple[Document, float]]:464 """Returns a list of documents with their scores465 466 Args:467 embeddings: The query vector468 k: the number of documents to return469 kind: Type of vector index to create.470 Possible options are:471 - vector-ivf472 - vector-hnsw: available as a preview feature only,473 to enable visit https://learn.microsoft.com/en-us/azure/azure-resource-manager/management/preview-features474 - vector-diskann: available as a preview feature only475 ef_search: The size of the dynamic candidate list for search476 (40 by default). A higher value provides better477 recall at the cost of speed.478 score_threshold: (Optional[float], optional): Maximum vector distance479 between selected documents and the query vector. Defaults to None.480 Only vector-ivf search supports this for now.481 l_search: l value for index searching.482 Default value is 40, range from 10 to 10000.483 Only vector-diskann search supports this.484 485 Returns:486 A list of documents closest to the query vector487 """488 pipeline: List[dict[str, Any]] = []489 if kind == CosmosDBVectorSearchType.VECTOR_IVF:490 pipeline = self._get_pipeline_vector_ivf(embeddings, k, pre_filter)491 elif kind == CosmosDBVectorSearchType.VECTOR_HNSW:492 pipeline = self._get_pipeline_vector_hnsw(493 embeddings, k, ef_search, pre_filter494 )495 elif kind == CosmosDBVectorSearchType.VECTOR_DISKANN:496 pipeline = self._get_pipeline_vector_diskann(497 embeddings, k, l_search, pre_filter498 )499 500 cursor = self._collection.aggregate(pipeline)501 502 docs = []503 for res in cursor:504 score = res.pop("similarityScore")505 if score < score_threshold:506 continue507 document_object_field = res.pop("document")508 text = document_object_field.pop(self._text_key)509 metadata = document_object_field.pop("metadata", {})510 metadata["_id"] = document_object_field.pop(511 "_id"512 ) # '_id' is in new position513 if with_embedding:514 metadata[self._embedding_key] = document_object_field.pop(515 self._embedding_key516 )517 518 docs.append((Document(page_content=text, metadata=metadata), score))519 return docs520 521 def _get_pipeline_vector_ivf(522 self, embeddings: List[float], k: int = 4, pre_filter: Optional[Dict] = None523 ) -> List[dict[str, Any]]:524 params = {525 "vector": embeddings,526 "path": self._embedding_key,527 "k": k,528 }529 if pre_filter:530 params["filter"] = pre_filter531 532 pipeline: List[dict[str, Any]] = [533 {534 "$search": {535 "cosmosSearch": params,536 "returnStoredSource": True,537 }538 },539 {540 "$project": {541 "similarityScore": {"$meta": "searchScore"},542 "document": "$$ROOT",543 }544 },545 ]546 return pipeline547 548 def _get_pipeline_vector_hnsw(549 self,550 embeddings: List[float],551 k: int = 4,552 ef_search: int = 40,553 pre_filter: Optional[Dict] = None,554 ) -> List[dict[str, Any]]:555 params = {556 "vector": embeddings,557 "path": self._embedding_key,558 "k": k,559 "efSearch": ef_search,560 }561 if pre_filter:562 params["filter"] = pre_filter563 564 pipeline: List[dict[str, Any]] = [565 {566 "$search": {567 "cosmosSearch": params,568 }569 },570 {571 "$project": {572 "similarityScore": {"$meta": "searchScore"},573 "document": "$$ROOT",574 }575 },576 ]577 return pipeline578 579 def _get_pipeline_vector_diskann(580 self,581 embeddings: List[float],582 k: int = 4,583 l_search: int = 40,584 pre_filter: Optional[Dict] = None,585 ) -> List[dict[str, Any]]:586 params = {587 "vector": embeddings,588 "path": self._embedding_key,589 "k": k,590 "lSearch": l_search,591 }592 if pre_filter:593 params["filter"] = pre_filter594 595 pipeline: List[dict[str, Any]] = [596 {597 "$search": {598 "cosmosSearch": params,599 }600 },601 {602 "$project": {603 "similarityScore": {"$meta": "searchScore"},604 "document": "$$ROOT",605 }606 },607 ]608 return pipeline609 610 def similarity_search_with_score(611 self,612 query: str,613 k: int = 4,614 kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,615 pre_filter: Optional[Dict] = None,616 ef_search: int = 40,617 score_threshold: float = 0.0,618 l_search: int = 40,619 with_embedding: bool = False,620 ) -> List[Tuple[Document, float]]:621 embeddings = self._embedding.embed_query(query)622 docs = self._similarity_search_with_score(623 embeddings=embeddings,624 k=k,625 kind=kind,626 pre_filter=pre_filter,627 ef_search=ef_search,628 score_threshold=score_threshold,629 l_search=l_search,630 with_embedding=with_embedding,631 )632 return docs633 634 def similarity_search(635 self,636 query: str,637 k: int = 4,638 kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,639 pre_filter: Optional[Dict] = None,640 ef_search: int = 40,641 score_threshold: float = 0.0,642 l_search: int = 40,643 with_embedding: bool = False,644 **kwargs: Any,645 ) -> List[Document]:646 docs_and_scores = self.similarity_search_with_score(647 query,648 k=k,649 kind=kind,650 pre_filter=pre_filter,651 ef_search=ef_search,652 score_threshold=score_threshold,653 l_search=l_search,654 with_embedding=with_embedding,655 )656 return [doc for doc, _ in docs_and_scores]657 658 def max_marginal_relevance_search_by_vector(659 self,660 embedding: List[float],661 k: int = 4,662 fetch_k: int = 20,663 lambda_mult: float = 0.5,664 kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,665 pre_filter: Optional[Dict] = None,666 ef_search: int = 40,667 score_threshold: float = 0.0,668 l_search: int = 40,669 with_embedding: bool = False,670 **kwargs: Any,671 ) -> List[Document]:672 # Retrieves the docs with similarity scores673 # sorted by similarity scores in DESC order674 docs = self._similarity_search_with_score(675 embedding,676 k=fetch_k,677 kind=kind,678 pre_filter=pre_filter,679 ef_search=ef_search,680 score_threshold=score_threshold,681 l_search=l_search,682 with_embedding=with_embedding,683 )684 685 # Re-ranks the docs using MMR686 mmr_doc_indexes = maximal_marginal_relevance(687 np.array(embedding),688 [doc.metadata[self._embedding_key] for doc, _ in docs],689 k=k,690 lambda_mult=lambda_mult,691 )692 mmr_docs = [docs[i][0] for i in mmr_doc_indexes]693 return mmr_docs694 695 def max_marginal_relevance_search(696 self,697 query: str,698 k: int = 4,699 fetch_k: int = 20,700 lambda_mult: float = 0.5,701 kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,702 pre_filter: Optional[Dict] = None,703 ef_search: int = 40,704 score_threshold: float = 0.0,705 l_search: int = 40,706 with_embedding: bool = False,707 **kwargs: Any,708 ) -> List[Document]:709 # compute the embeddings vector from the query string710 embeddings = self._embedding.embed_query(query)711 712 docs = self.max_marginal_relevance_search_by_vector(713 embeddings,714 k=k,715 fetch_k=fetch_k,716 lambda_mult=lambda_mult,717 kind=kind,718 pre_filter=pre_filter,719 ef_search=ef_search,720 score_threshold=score_threshold,721 l_search=l_search,722 with_embedding=with_embedding,723 )724 return docs725 726 def get_collection(self) -> Collection:727 return self._collection728 