Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
azure_cosmos_db.py728 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import logging4from enum import Enum5from typing import (6    TYPE_CHECKING,7    Any,8    Dict,9    Generator,10    Iterable,11    List,12    Optional,13    Tuple,14    Union,15)16 17import numpy as np18from langchain_core.documents import Document19from langchain_core.vectorstores import VectorStore20 21from langchain_community.vectorstores.utils import maximal_marginal_relevance22 23if TYPE_CHECKING:24    from langchain_core.embeddings import Embeddings25    from pymongo.collection import Collection26 27 28# Before Python 3.11 native StrEnum is not available29class CosmosDBSimilarityType(str, Enum):30    """Cosmos DB Similarity Type as enumerator."""31 32    COS = "COS"33    """CosineSimilarity"""34    IP = "IP"35    """inner - product"""36    L2 = "L2"37    """Euclidean distance"""38 39 40class CosmosDBVectorSearchType(str, Enum):41    """Cosmos DB Vector Search Type as enumerator."""42 43    VECTOR_IVF = "vector-ivf"44    """IVF vector index"""45    VECTOR_HNSW = "vector-hnsw"46    """HNSW vector index"""47    VECTOR_DISKANN = "vector-diskann"48    """DISKANN vector index"""49 50 51logger = logging.getLogger(__name__)52 53DEFAULT_INSERT_BATCH_SIZE = 12854 55 56class AzureCosmosDBVectorSearch(VectorStore):57    """`Azure Cosmos DB for MongoDB vCore` vector store.58 59    To use, you should have both:60    - the ``pymongo`` python package installed61    - a connection string associated with a MongoDB VCore Cluster62 63    Example:64        . code-block:: python65 66            from langchain_community.vectorstores import67            AzureCosmosDBVectorSearch68            from langchain_community.embeddings.openai import OpenAIEmbeddings69            from pymongo import MongoClient70 71            mongo_client = MongoClient("<YOUR-CONNECTION-STRING>")72            collection = mongo_client["<db_name>"]["<collection_name>"]73            embeddings = OpenAIEmbeddings()74            vectorstore = AzureCosmosDBVectorSearch(collection, embeddings)75    """76 77    def __init__(78        self,79        collection: Collection,80        embedding: Embeddings,81        *,82        index_name: str = "vectorSearchIndex",83        text_key: str = "textContent",84        embedding_key: str = "vectorContent",85        application_name: str = "LangChain-CDBMongoVCore-VectorStore-Python",86    ):87        """Constructor for AzureCosmosDBVectorSearch88 89        Args:90            collection: MongoDB collection to add the texts to.91            embedding: Text embedding model to use.92            index_name: Name of the Atlas Search index.93            text_key: MongoDB field that will contain the text94                for each document.95            embedding_key: MongoDB field that will contain the embedding96                for each document.97        """98        self._collection = collection99        self._embedding = embedding100        self._index_name = index_name101        self._text_key = text_key102        self._embedding_key = embedding_key103        self._application_name = application_name104 105    @property106    def embeddings(self) -> Embeddings:107        return self._embedding108 109    def get_index_name(self) -> str:110        """Returns the index name111 112        Returns:113            Returns the index name114 115        """116        return self._index_name117 118    @classmethod119    def from_connection_string(120        cls,121        connection_string: str,122        namespace: str,123        embedding: Embeddings,124        application_name: str = "LangChain-CDBMongoVCore-VectorStore-Python",125        **kwargs: Any,126    ) -> AzureCosmosDBVectorSearch:127        """Creates an Instance of AzureCosmosDBVectorSearch128        from a Connection String129 130        Args:131            connection_string: The MongoDB vCore instance connection string132            namespace: The namespace (database.collection)133            embedding: The embedding utility134            application_name: The user agent for telemetry135            **kwargs: Dynamic keyword arguments136 137        Returns:138            an instance of the vector store139 140        """141        try:142            from pymongo import MongoClient143        except ImportError:144            raise ImportError(145                "Could not import pymongo, please install it with "146                "`pip install pymongo`."147            )148        appname = application_name149        client: MongoClient = MongoClient(connection_string, appname=appname)150        db_name, collection_name = namespace.split(".")151        collection = client[db_name][collection_name]152        return cls(collection, embedding, **kwargs)153 154    def index_exists(self) -> bool:155        """Verifies if the specified index name during instance156            construction exists on the collection157 158        Returns:159          Returns True on success and False if no such index exists160            on the collection161        """162        cursor = self._collection.list_indexes()163        index_name = self._index_name164 165        for res in cursor:166            current_index_name = res.pop("name")167            if current_index_name == index_name:168                return True169 170        return False171 172    def delete_index(self) -> None:173        """Deletes the index specified during instance construction if it exists"""174        if self.index_exists():175            self._collection.drop_index(self._index_name)176            # Raises OperationFailure on an error (e.g. trying to drop177            # an index that does not exist)178 179    def create_index(180        self,181        num_lists: int = 100,182        dimensions: int = 1536,183        similarity: CosmosDBSimilarityType = CosmosDBSimilarityType.COS,184        kind: str = "vector-ivf",185        m: int = 16,186        ef_construction: int = 64,187        max_degree: int = 32,188        l_build: int = 50,189    ) -> dict[str, Any]:190        """Creates an index using the index name specified at191            instance construction192 193        Setting the numLists parameter correctly is important for achieving194            good accuracy and performance.195            Since the vector store uses IVF as the indexing strategy,196            you should create the index only after you197            have loaded a large enough sample documents to ensure that the198            centroids for the respective buckets are199            faily distributed.200 201        We recommend that numLists is set to documentCount/1000 for up202            to 1 million documents203            and to sqrt(documentCount) for more than 1 million documents.204            As the number of items in your database grows, you should205            tune numLists to be larger206            in order to achieve good latency performance for vector search.207 208            If you're experimenting with a new scenario or creating a209            small demo, you can start with numLists210            set to 1 to perform a brute-force search across all vectors.211            This should provide you with the most212            accurate results from the vector search, however be aware that213            the search speed and latency will be slow.214            After your initial setup, you should go ahead and tune215            the numLists parameter using the above guidance.216 217        Args:218            kind: Type of vector index to create.219                Possible options are:220                    - vector-ivf221                    - vector-hnsw: available as a preview feature only,222                                   to enable visit https://learn.microsoft.com/en-us/azure/azure-resource-manager/management/preview-features223                    - vector-diskann: available as a preview feature only224            num_lists: This integer is the number of clusters that the225                inverted file (IVF) index uses to group the vector data.226                We recommend that numLists is set to documentCount/1000227                for up to 1 million documents and to sqrt(documentCount)228                for more than 1 million documents.229                Using a numLists value of 1 is akin to performing230                brute-force search, which has limited performance231            dimensions: Number of dimensions for vector similarity.232                The maximum number of supported dimensions is 2000233            similarity: Similarity metric to use with the IVF index.234 235                Possible options are:236                    - CosmosDBSimilarityType.COS (cosine distance),237                    - CosmosDBSimilarityType.L2 (Euclidean distance), and238                    - CosmosDBSimilarityType.IP (inner product).239            m: The max number of connections per layer (16 by default, minimum240               value is 2, maximum value is 100). Higher m is suitable for datasets241               with high dimensionality and/or high accuracy requirements.242            ef_construction: the size of the dynamic candidate list for constructing243                            the graph (64 by default, minimum value is 4, maximum244                            value is 1000). Higher ef_construction will result in245                            better index quality and higher accuracy, but it will246                            also increase the time required to build the index.247                            ef_construction has to be at least 2 * m248            max_degree: Max number of neighbors.249                Default value is 32, range from 20 to 2048.250                Only vector-diskann search supports this for now.251            l_build: l value for index building.252                Default value is 50, range from 10 to 500.253                Only vector-diskann search supports this for now.254        Returns:255            An object describing the created index256 257        """258        # check the kind of vector search to be performed259        # prepare the command accordingly260        create_index_commands = {}261        if kind == CosmosDBVectorSearchType.VECTOR_IVF:262            create_index_commands = self._get_vector_index_ivf(263                kind, num_lists, similarity, dimensions264            )265        elif kind == CosmosDBVectorSearchType.VECTOR_HNSW:266            create_index_commands = self._get_vector_index_hnsw(267                kind, m, ef_construction, similarity, dimensions268            )269        elif kind == CosmosDBVectorSearchType.VECTOR_DISKANN:270            create_index_commands = self._get_vector_index_diskann(271                kind, max_degree, l_build, similarity, dimensions272            )273 274        # retrieve the database object275        current_database = self._collection.database276 277        # invoke the command from the database object278        create_index_responses: dict[str, Any] = current_database.command(279            create_index_commands280        )281 282        return create_index_responses283 284    def _get_vector_index_ivf(285        self, kind: str, num_lists: int, similarity: str, dimensions: int286    ) -> Dict[str, Any]:287        command = {288            "createIndexes": self._collection.name,289            "indexes": [290                {291                    "name": self._index_name,292                    "key": {self._embedding_key: "cosmosSearch"},293                    "cosmosSearchOptions": {294                        "kind": kind,295                        "numLists": num_lists,296                        "similarity": similarity,297                        "dimensions": dimensions,298                    },299                }300            ],301        }302        return command303 304    def _get_vector_index_hnsw(305        self, kind: str, m: int, ef_construction: int, similarity: str, dimensions: int306    ) -> Dict[str, Any]:307        command = {308            "createIndexes": self._collection.name,309            "indexes": [310                {311                    "name": self._index_name,312                    "key": {self._embedding_key: "cosmosSearch"},313                    "cosmosSearchOptions": {314                        "kind": kind,315                        "m": m,316                        "efConstruction": ef_construction,317                        "similarity": similarity,318                        "dimensions": dimensions,319                    },320                }321            ],322        }323        return command324 325    def _get_vector_index_diskann(326        self, kind: str, max_degree: int, l_build: int, similarity: str, dimensions: int327    ) -> Dict[str, Any]:328        command = {329            "createIndexes": self._collection.name,330            "indexes": [331                {332                    "name": self._index_name,333                    "key": {self._embedding_key: "cosmosSearch"},334                    "cosmosSearchOptions": {335                        "kind": kind,336                        "maxDegree": max_degree,337                        "lBuild": l_build,338                        "similarity": similarity,339                        "dimensions": dimensions,340                    },341                }342            ],343        }344        return command345 346    def create_filter_index(347        self,348        property_to_filter: str,349        index_name: str,350    ) -> dict[str, Any]:351        command = {352            "createIndexes": self._collection.name,353            "indexes": [354                {355                    "key": {property_to_filter: 1},356                    "name": index_name,357                }358            ],359        }360        # retrieve the database object361        current_database = self._collection.database362 363        # invoke the command from the database object364        create_index_responses: dict[str, Any] = current_database.command(command)365        return create_index_responses366 367    def add_texts(368        self,369        texts: Iterable[str],370        metadatas: Optional[List[Dict[str, Any]]] = None,371        **kwargs: Any,372    ) -> List:373        batch_size = kwargs.get("batch_size", DEFAULT_INSERT_BATCH_SIZE)374        _metadatas: Union[List, Generator] = metadatas or ({} for _ in texts)375        texts_batch = []376        metadatas_batch = []377        result_ids = []378        for i, (text, metadata) in enumerate(zip(texts, _metadatas)):379            texts_batch.append(text)380            metadatas_batch.append(metadata)381            if (i + 1) % batch_size == 0:382                result_ids.extend(self._insert_texts(texts_batch, metadatas_batch))383                texts_batch = []384                metadatas_batch = []385        if texts_batch:386            result_ids.extend(self._insert_texts(texts_batch, metadatas_batch))387        return result_ids388 389    def _insert_texts(self, texts: List[str], metadatas: List[Dict[str, Any]]) -> List:390        """Used to Load Documents into the collection391 392        Args:393            texts: The list of documents strings to load394            metadatas: The list of metadata objects associated with each document395 396        Returns:397 398        """399        # If the text is empty, then exit early400        if not texts:401            return []402 403        # Embed and create the documents404        embeddings = self._embedding.embed_documents(texts)405        to_insert = [406            {self._text_key: t, self._embedding_key: embedding, "metadata": m}407            for t, m, embedding in zip(texts, metadatas, embeddings)408        ]409        # insert the documents in Cosmos DB410        insert_result = self._collection.insert_many(to_insert)411        return insert_result.inserted_ids412 413    @classmethod414    def from_texts(415        cls,416        texts: List[str],417        embedding: Embeddings,418        metadatas: Optional[List[dict]] = None,419        collection: Optional[Collection] = None,420        **kwargs: Any,421    ) -> AzureCosmosDBVectorSearch:422        if collection is None:423            raise ValueError("Must provide 'collection' named parameter.")424        vectorstore = cls(collection, embedding, **kwargs)425        vectorstore.add_texts(texts, metadatas=metadatas)426        return vectorstore427 428    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> Optional[bool]:429        if ids is None:430            raise ValueError("No document ids provided to delete.")431 432        for document_id in ids:433            self.delete_document_by_id(document_id)434        return True435 436    def delete_document_by_id(self, document_id: Optional[str] = None) -> None:437        """Removes a Specific Document by Id438 439        Args:440            document_id: The document identifier441        """442        try:443            from bson.objectid import ObjectId444        except ImportError as e:445            raise ImportError(446                "Unable to import bson, please install with `pip install bson`."447            ) from e448        if document_id is None:449            raise ValueError("No document id provided to delete.")450 451        self._collection.delete_one({"_id": ObjectId(document_id)})452 453    def _similarity_search_with_score(454        self,455        embeddings: List[float],456        k: int = 4,457        kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,458        pre_filter: Optional[Dict] = None,459        ef_search: int = 40,460        score_threshold: float = 0.0,461        l_search: int = 40,462        with_embedding: bool = False,463    ) -> List[Tuple[Document, float]]:464        """Returns a list of documents with their scores465 466        Args:467            embeddings: The query vector468            k: the number of documents to return469            kind: Type of vector index to create.470                Possible options are:471                    - vector-ivf472                    - vector-hnsw: available as a preview feature only,473                                   to enable visit https://learn.microsoft.com/en-us/azure/azure-resource-manager/management/preview-features474                    - vector-diskann: available as a preview feature only475            ef_search: The size of the dynamic candidate list for search476                       (40 by default). A higher value provides better477                       recall at the cost of speed.478            score_threshold: (Optional[float], optional): Maximum vector distance479                between selected documents and the query vector. Defaults to None.480                Only vector-ivf search supports this for now.481            l_search: l value for index searching.482                Default value is 40, range from 10 to 10000.483                Only vector-diskann search supports this.484 485        Returns:486            A list of documents closest to the query vector487        """488        pipeline: List[dict[str, Any]] = []489        if kind == CosmosDBVectorSearchType.VECTOR_IVF:490            pipeline = self._get_pipeline_vector_ivf(embeddings, k, pre_filter)491        elif kind == CosmosDBVectorSearchType.VECTOR_HNSW:492            pipeline = self._get_pipeline_vector_hnsw(493                embeddings, k, ef_search, pre_filter494            )495        elif kind == CosmosDBVectorSearchType.VECTOR_DISKANN:496            pipeline = self._get_pipeline_vector_diskann(497                embeddings, k, l_search, pre_filter498            )499 500        cursor = self._collection.aggregate(pipeline)501 502        docs = []503        for res in cursor:504            score = res.pop("similarityScore")505            if score < score_threshold:506                continue507            document_object_field = res.pop("document")508            text = document_object_field.pop(self._text_key)509            metadata = document_object_field.pop("metadata", {})510            metadata["_id"] = document_object_field.pop(511                "_id"512            )  # '_id' is in new position513            if with_embedding:514                metadata[self._embedding_key] = document_object_field.pop(515                    self._embedding_key516                )517 518            docs.append((Document(page_content=text, metadata=metadata), score))519        return docs520 521    def _get_pipeline_vector_ivf(522        self, embeddings: List[float], k: int = 4, pre_filter: Optional[Dict] = None523    ) -> List[dict[str, Any]]:524        params = {525            "vector": embeddings,526            "path": self._embedding_key,527            "k": k,528        }529        if pre_filter:530            params["filter"] = pre_filter531 532        pipeline: List[dict[str, Any]] = [533            {534                "$search": {535                    "cosmosSearch": params,536                    "returnStoredSource": True,537                }538            },539            {540                "$project": {541                    "similarityScore": {"$meta": "searchScore"},542                    "document": "$$ROOT",543                }544            },545        ]546        return pipeline547 548    def _get_pipeline_vector_hnsw(549        self,550        embeddings: List[float],551        k: int = 4,552        ef_search: int = 40,553        pre_filter: Optional[Dict] = None,554    ) -> List[dict[str, Any]]:555        params = {556            "vector": embeddings,557            "path": self._embedding_key,558            "k": k,559            "efSearch": ef_search,560        }561        if pre_filter:562            params["filter"] = pre_filter563 564        pipeline: List[dict[str, Any]] = [565            {566                "$search": {567                    "cosmosSearch": params,568                }569            },570            {571                "$project": {572                    "similarityScore": {"$meta": "searchScore"},573                    "document": "$$ROOT",574                }575            },576        ]577        return pipeline578 579    def _get_pipeline_vector_diskann(580        self,581        embeddings: List[float],582        k: int = 4,583        l_search: int = 40,584        pre_filter: Optional[Dict] = None,585    ) -> List[dict[str, Any]]:586        params = {587            "vector": embeddings,588            "path": self._embedding_key,589            "k": k,590            "lSearch": l_search,591        }592        if pre_filter:593            params["filter"] = pre_filter594 595        pipeline: List[dict[str, Any]] = [596            {597                "$search": {598                    "cosmosSearch": params,599                }600            },601            {602                "$project": {603                    "similarityScore": {"$meta": "searchScore"},604                    "document": "$$ROOT",605                }606            },607        ]608        return pipeline609 610    def similarity_search_with_score(611        self,612        query: str,613        k: int = 4,614        kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,615        pre_filter: Optional[Dict] = None,616        ef_search: int = 40,617        score_threshold: float = 0.0,618        l_search: int = 40,619        with_embedding: bool = False,620    ) -> List[Tuple[Document, float]]:621        embeddings = self._embedding.embed_query(query)622        docs = self._similarity_search_with_score(623            embeddings=embeddings,624            k=k,625            kind=kind,626            pre_filter=pre_filter,627            ef_search=ef_search,628            score_threshold=score_threshold,629            l_search=l_search,630            with_embedding=with_embedding,631        )632        return docs633 634    def similarity_search(635        self,636        query: str,637        k: int = 4,638        kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,639        pre_filter: Optional[Dict] = None,640        ef_search: int = 40,641        score_threshold: float = 0.0,642        l_search: int = 40,643        with_embedding: bool = False,644        **kwargs: Any,645    ) -> List[Document]:646        docs_and_scores = self.similarity_search_with_score(647            query,648            k=k,649            kind=kind,650            pre_filter=pre_filter,651            ef_search=ef_search,652            score_threshold=score_threshold,653            l_search=l_search,654            with_embedding=with_embedding,655        )656        return [doc for doc, _ in docs_and_scores]657 658    def max_marginal_relevance_search_by_vector(659        self,660        embedding: List[float],661        k: int = 4,662        fetch_k: int = 20,663        lambda_mult: float = 0.5,664        kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,665        pre_filter: Optional[Dict] = None,666        ef_search: int = 40,667        score_threshold: float = 0.0,668        l_search: int = 40,669        with_embedding: bool = False,670        **kwargs: Any,671    ) -> List[Document]:672        # Retrieves the docs with similarity scores673        # sorted by similarity scores in DESC order674        docs = self._similarity_search_with_score(675            embedding,676            k=fetch_k,677            kind=kind,678            pre_filter=pre_filter,679            ef_search=ef_search,680            score_threshold=score_threshold,681            l_search=l_search,682            with_embedding=with_embedding,683        )684 685        # Re-ranks the docs using MMR686        mmr_doc_indexes = maximal_marginal_relevance(687            np.array(embedding),688            [doc.metadata[self._embedding_key] for doc, _ in docs],689            k=k,690            lambda_mult=lambda_mult,691        )692        mmr_docs = [docs[i][0] for i in mmr_doc_indexes]693        return mmr_docs694 695    def max_marginal_relevance_search(696        self,697        query: str,698        k: int = 4,699        fetch_k: int = 20,700        lambda_mult: float = 0.5,701        kind: CosmosDBVectorSearchType = CosmosDBVectorSearchType.VECTOR_IVF,702        pre_filter: Optional[Dict] = None,703        ef_search: int = 40,704        score_threshold: float = 0.0,705        l_search: int = 40,706        with_embedding: bool = False,707        **kwargs: Any,708    ) -> List[Document]:709        # compute the embeddings vector from the query string710        embeddings = self._embedding.embed_query(query)711 712        docs = self.max_marginal_relevance_search_by_vector(713            embeddings,714            k=k,715            fetch_k=fetch_k,716            lambda_mult=lambda_mult,717            kind=kind,718            pre_filter=pre_filter,719            ef_search=ef_search,720            score_threshold=score_threshold,721            l_search=l_search,722            with_embedding=with_embedding,723        )724        return docs725 726    def get_collection(self) -> Collection:727        return self._collection728 
codekingpro/portable-devtools · Team Ai