Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
couchbase.py633 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import uuid4from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Tuple, Type5 6from langchain_core._api.deprecation import deprecated7from langchain_core.documents import Document8from langchain_core.embeddings import Embeddings9from langchain_core.vectorstores import VectorStore10 11if TYPE_CHECKING:12    from couchbase.cluster import Cluster13 14 15@deprecated(16    since="0.2.4",17    removal="1.0",18    alternative_import="langchain_couchbase.CouchbaseSearchVectorStore",19)20class CouchbaseVectorStore(VectorStore):21    """`Couchbase Vector Store` vector store.22 23    To use it, you need24    - a recent installation of the `couchbase` library25    - a Couchbase database with a pre-defined Search index with support for26        vector fields27 28    Example:29        .. code-block:: python30 31            from langchain_community.vectorstores import CouchbaseVectorStore32            from langchain_openai import OpenAIEmbeddings33 34            from couchbase.cluster import Cluster35            from couchbase.auth import PasswordAuthenticator36            from couchbase.options import ClusterOptions37            from datetime import timedelta38 39            auth = PasswordAuthenticator(username, password)40            options = ClusterOptions(auth)41            connect_string = "couchbases://localhost"42            cluster = Cluster(connect_string, options)43 44            # Wait until the cluster is ready for use.45            cluster.wait_until_ready(timedelta(seconds=5))46 47            embeddings = OpenAIEmbeddings()48 49            vectorstore = CouchbaseVectorStore(50                cluster=cluster,51                bucket_name="",52                scope_name="",53                collection_name="",54                embedding=embeddings,55                index_name="vector-index",56            )57 58            vectorstore.add_texts(["hello", "world"])59            results = vectorstore.similarity_search("ola", k=1)60    """61 62    # Default batch size63    DEFAULT_BATCH_SIZE: int = 10064    _metadata_key: str = "metadata"65    _default_text_key: str = "text"66    _default_embedding_key: str = "embedding"67 68    def _check_bucket_exists(self) -> bool:69        """Check if the bucket exists in the linked Couchbase cluster"""70        bucket_manager = self._cluster.buckets()71        try:72            bucket_manager.get_bucket(self._bucket_name)73            return True74        except Exception:75            return False76 77    def _check_scope_and_collection_exists(self) -> bool:78        """Check if the scope and collection exists in the linked Couchbase bucket79        Raises a ValueError if either is not found"""80        scope_collection_map: Dict[str, Any] = {}81 82        # Get a list of all scopes in the bucket83        for scope in self._bucket.collections().get_all_scopes():84            scope_collection_map[scope.name] = []85 86            # Get a list of all the collections in the scope87            for collection in scope.collections:88                scope_collection_map[scope.name].append(collection.name)89 90        # Check if the scope exists91        if self._scope_name not in scope_collection_map.keys():92            raise ValueError(93                f"Scope {self._scope_name} not found in Couchbase "94                f"bucket {self._bucket_name}"95            )96 97        # Check if the collection exists in the scope98        if self._collection_name not in scope_collection_map[self._scope_name]:99            raise ValueError(100                f"Collection {self._collection_name} not found in scope "101                f"{self._scope_name} in Couchbase bucket {self._bucket_name}"102            )103 104        return True105 106    def _check_index_exists(self) -> bool:107        """Check if the Search index exists in the linked Couchbase cluster108        Raises a ValueError if the index does not exist"""109        if self._scoped_index:110            all_indexes = [111                index.name for index in self._scope.search_indexes().get_all_indexes()112            ]113            if self._index_name not in all_indexes:114                raise ValueError(115                    f"Index {self._index_name} does not exist. "116                    " Please create the index before searching."117                )118        else:119            all_indexes = [120                index.name for index in self._cluster.search_indexes().get_all_indexes()121            ]122            if self._index_name not in all_indexes:123                raise ValueError(124                    f"Index {self._index_name} does not exist. "125                    " Please create the index before searching."126                )127 128        return True129 130    def __init__(131        self,132        cluster: Cluster,133        bucket_name: str,134        scope_name: str,135        collection_name: str,136        embedding: Embeddings,137        index_name: str,138        *,139        text_key: Optional[str] = _default_text_key,140        embedding_key: Optional[str] = _default_embedding_key,141        scoped_index: bool = True,142    ) -> None:143        """144        Initialize the Couchbase Vector Store.145 146        Args:147 148            cluster (Cluster): couchbase cluster object with active connection.149            bucket_name (str): name of bucket to store documents in.150            scope_name (str): name of scope in the bucket to store documents in.151            collection_name (str): name of collection in the scope to store documents in152            embedding (Embeddings): embedding function to use.153            index_name (str): name of the Search index to use.154            text_key (optional[str]): key in document to use as text.155                Set to text by default.156            embedding_key (optional[str]): key in document to use for the embeddings.157                Set to embedding by default.158            scoped_index (optional[bool]): specify whether the index is a scoped index.159                Set to True by default.160        """161        try:162            from couchbase.cluster import Cluster163        except ImportError as e:164            raise ImportError(165                "Could not import couchbase python package. "166                "Please install couchbase SDK  with `pip install couchbase`."167            ) from e168 169        if not isinstance(cluster, Cluster):170            raise ValueError(171                f"cluster should be an instance of couchbase.Cluster, "172                f"got {type(cluster)}"173            )174 175        self._cluster = cluster176 177        if not embedding:178            raise ValueError("Embeddings instance must be provided.")179 180        if not bucket_name:181            raise ValueError("bucket_name must be provided.")182 183        if not scope_name:184            raise ValueError("scope_name must be provided.")185 186        if not collection_name:187            raise ValueError("collection_name must be provided.")188 189        if not index_name:190            raise ValueError("index_name must be provided.")191 192        self._bucket_name = bucket_name193        self._scope_name = scope_name194        self._collection_name = collection_name195        self._embedding_function = embedding196        self._text_key = text_key197        self._embedding_key = embedding_key198        self._index_name = index_name199        self._scoped_index = scoped_index200 201        # Check if the bucket exists202        if not self._check_bucket_exists():203            raise ValueError(204                f"Bucket {self._bucket_name} does not exist. "205                " Please create the bucket before searching."206            )207 208        try:209            self._bucket = self._cluster.bucket(self._bucket_name)210            self._scope = self._bucket.scope(self._scope_name)211            self._collection = self._scope.collection(self._collection_name)212        except Exception as e:213            raise ValueError(214                "Error connecting to couchbase. "215                "Please check the connection and credentials."216            ) from e217 218        # Check if the scope and collection exists. Throws ValueError if they don't219        try:220            self._check_scope_and_collection_exists()221        except Exception as e:222            raise e223 224        # Check if the index exists. Throws ValueError if it doesn't225        try:226            self._check_index_exists()227        except Exception as e:228            raise e229 230    def add_texts(231        self,232        texts: Iterable[str],233        metadatas: Optional[List[Dict[str, Any]]] = None,234        ids: Optional[List[str]] = None,235        batch_size: Optional[int] = None,236        **kwargs: Any,237    ) -> List[str]:238        """Run texts through the embeddings and persist in vectorstore.239 240        If the document IDs are passed, the existing documents (if any) will be241        overwritten with the new ones.242 243        Args:244            texts (Iterable[str]): Iterable of strings to add to the vectorstore.245            metadatas (Optional[List[Dict]]): Optional list of metadatas associated246                with the texts.247            ids (Optional[List[str]]): Optional list of ids associated with the texts.248                IDs have to be unique strings across the collection.249                If it is not specified uuids are generated and used as ids.250            batch_size (Optional[int]): Optional batch size for bulk insertions.251                Default is 100.252 253        Returns:254            List[str]:List of ids from adding the texts into the vectorstore.255        """256        from couchbase.exceptions import DocumentExistsException257 258        if not batch_size:259            batch_size = self.DEFAULT_BATCH_SIZE260        doc_ids: List[str] = []261 262        if ids is None:263            ids = [uuid.uuid4().hex for _ in texts]264 265        if metadatas is None:266            metadatas = [{} for _ in texts]267 268        embedded_texts = self._embedding_function.embed_documents(list(texts))269 270        documents_to_insert = [271            {272                id: {273                    self._text_key: text,274                    self._embedding_key: vector,275                    self._metadata_key: metadata,276                }277                for id, text, vector, metadata in zip(278                    ids, texts, embedded_texts, metadatas279                )280            }281        ]282 283        # Insert in batches284        for i in range(0, len(documents_to_insert), batch_size):285            batch = documents_to_insert[i : i + batch_size]286            try:287                result = self._collection.upsert_multi(batch[0])288                if result.all_ok:289                    doc_ids.extend(batch[0].keys())290            except DocumentExistsException as e:291                raise ValueError(f"Document already exists: {e}")292 293        return doc_ids294 295    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> Optional[bool]:296        """Delete documents from the vector store by ids.297 298        Args:299            ids (List[str]): List of IDs of the documents to delete.300            batch_size (Optional[int]): Optional batch size for bulk deletions.301 302        Returns:303            bool: True if all the documents were deleted successfully, False otherwise.304 305        """306        from couchbase.exceptions import DocumentNotFoundException307 308        if ids is None:309            raise ValueError("No document ids provided to delete.")310 311        batch_size = kwargs.get("batch_size", self.DEFAULT_BATCH_SIZE)312        deletion_status = True313 314        # Delete in batches315        for i in range(0, len(ids), batch_size):316            batch = ids[i : i + batch_size]317            try:318                result = self._collection.remove_multi(batch)319            except DocumentNotFoundException as e:320                deletion_status = False321                raise ValueError(f"Document not found: {e}")322 323            deletion_status &= result.all_ok324 325        return deletion_status326 327    @property328    def embeddings(self) -> Embeddings:329        """Return the query embedding object."""330        return self._embedding_function331 332    def _format_metadata(self, row_fields: Dict[str, Any]) -> Dict[str, Any]:333        """Helper method to format the metadata from the Couchbase Search API.334        Args:335            row_fields (Dict[str, Any]): The fields to format.336 337        Returns:338            Dict[str, Any]: The formatted metadata.339        """340        metadata = {}341        for key, value in row_fields.items():342            # Couchbase Search returns the metadata key with a prefix343            # `metadata.` We remove it to get the original metadata key344            if key.startswith(self._metadata_key):345                new_key = key.split(self._metadata_key + ".")[-1]346                metadata[new_key] = value347            else:348                metadata[key] = value349 350        return metadata351 352    def similarity_search_with_score_by_vector(353        self,354        embedding: List[float],355        k: int = 4,356        search_options: Optional[Dict[str, Any]] = {},357        **kwargs: Any,358    ) -> List[Tuple[Document, float]]:359        """Return docs most similar to embedding vector with their scores.360 361        Args:362            embedding (List[float]): Embedding vector to look up documents similar to.363            k (int): Number of Documents to return.364                Defaults to 4.365            search_options (Optional[Dict[str, Any]]): Optional search options that are366                passed to Couchbase search.367                Defaults to empty dictionary.368            fields (Optional[List[str]]): Optional list of fields to include in the369                metadata of results. Note that these need to be stored in the index.370                If nothing is specified, defaults to all the fields stored in the index.371 372        Returns:373            List of (Document, score) that are the most similar to the query vector.374        """375        import couchbase.search as search376        from couchbase.options import SearchOptions377        from couchbase.vector_search import VectorQuery, VectorSearch378 379        fields = kwargs.get("fields", ["*"])380 381        # Document text field needs to be returned from the search382        if fields != ["*"] and self._text_key not in fields:383            fields.append(self._text_key)384 385        search_req = search.SearchRequest.create(386            VectorSearch.from_vector_query(387                VectorQuery(388                    self._embedding_key,389                    embedding,390                    k,391                )392            )393        )394        try:395            if self._scoped_index:396                search_iter = self._scope.search(397                    self._index_name,398                    search_req,399                    SearchOptions(400                        limit=k,401                        fields=fields,402                        raw=search_options,403                    ),404                )405 406            else:407                search_iter = self._cluster.search(408                    index=self._index_name,409                    request=search_req,410                    options=SearchOptions(limit=k, fields=fields, raw=search_options),411                )412 413            docs_with_score = []414 415            # Parse the results416            for row in search_iter.rows():417                text = row.fields.pop(self._text_key, "")418 419                # Format the metadata from Couchbase420                metadata = self._format_metadata(row.fields)421 422                score = row.score423                doc = Document(page_content=text, metadata=metadata)424                docs_with_score.append((doc, score))425 426        except Exception as e:427            raise ValueError(f"Search failed with error: {e}")428 429        return docs_with_score430 431    def similarity_search(432        self,433        query: str,434        k: int = 4,435        search_options: Optional[Dict[str, Any]] = {},436        **kwargs: Any,437    ) -> List[Document]:438        """Return documents most similar to embedding vector with their scores.439 440        Args:441            query (str): Query to look up for similar documents442            k (int): Number of Documents to return.443                Defaults to 4.444            search_options (Optional[Dict[str, Any]]): Optional search options that are445                passed to Couchbase search.446                Defaults to empty dictionary447            fields (Optional[List[str]]): Optional list of fields to include in the448                metadata of results. Note that these need to be stored in the index.449                If nothing is specified, defaults to all the fields stored in the index.450 451        Returns:452            List of Documents most similar to the query.453        """454        query_embedding = self.embeddings.embed_query(query)455        docs_with_scores = self.similarity_search_with_score_by_vector(456            query_embedding, k, search_options, **kwargs457        )458        return [doc for doc, _ in docs_with_scores]459 460    def similarity_search_with_score(461        self,462        query: str,463        k: int = 4,464        search_options: Optional[Dict[str, Any]] = {},465        **kwargs: Any,466    ) -> List[Tuple[Document, float]]:467        """Return documents that are most similar to the query with their scores.468 469        Args:470            query (str): Query to look up for similar documents471            k (int): Number of Documents to return.472                Defaults to 4.473            search_options (Optional[Dict[str, Any]]): Optional search options that are474                passed to Couchbase search.475                Defaults to empty dictionary.476            fields (Optional[List[str]]): Optional list of fields to include in the477                metadata of results. Note that these need to be stored in the index.478                If nothing is specified, defaults to text and metadata fields.479 480        Returns:481            List of (Document, score) that are most similar to the query.482        """483        query_embedding = self.embeddings.embed_query(query)484        docs_with_score = self.similarity_search_with_score_by_vector(485            query_embedding, k, search_options, **kwargs486        )487        return docs_with_score488 489    def similarity_search_by_vector(490        self,491        embedding: List[float],492        k: int = 4,493        search_options: Optional[Dict[str, Any]] = {},494        **kwargs: Any,495    ) -> List[Document]:496        """Return documents that are most similar to the vector embedding.497 498        Args:499            embedding (List[float]): Embedding to look up documents similar to.500            k (int): Number of Documents to return.501                Defaults to 4.502            search_options (Optional[Dict[str, Any]]): Optional search options that are503                passed to Couchbase search.504                Defaults to empty dictionary.505            fields (Optional[List[str]]): Optional list of fields to include in the506                metadata of results. Note that these need to be stored in the index.507                If nothing is specified, defaults to document text and metadata fields.508 509        Returns:510            List of Documents most similar to the query.511        """512        docs_with_score = self.similarity_search_with_score_by_vector(513            embedding, k, search_options, **kwargs514        )515        return [doc for doc, _ in docs_with_score]516 517    @classmethod518    def _from_kwargs(519        cls: Type[CouchbaseVectorStore],520        embedding: Embeddings,521        **kwargs: Any,522    ) -> CouchbaseVectorStore:523        """Initialize the Couchbase vector store from keyword arguments for the524        vector store.525 526        Args:527            embedding: Embedding object to use to embed text.528            **kwargs: Keyword arguments to initialize the vector store with.529                Accepted arguments are:530                    - cluster531                    - bucket_name532                    - scope_name533                    - collection_name534                    - index_name535                    - text_key536                    - embedding_key537                    - scoped_index538 539        """540        cluster = kwargs.get("cluster", None)541        bucket_name = kwargs.get("bucket_name", None)542        scope_name = kwargs.get("scope_name", None)543        collection_name = kwargs.get("collection_name", None)544        index_name = kwargs.get("index_name", None)545        text_key = kwargs.get("text_key", cls._default_text_key)546        embedding_key = kwargs.get("embedding_key", cls._default_embedding_key)547        scoped_index = kwargs.get("scoped_index", True)548 549        if bucket_name is None:550            raise ValueError("bucket_name must be provided")551        if scope_name is None:552            raise ValueError("scope_name must be provided")553        if collection_name is None:554            raise ValueError("collection_name must be provided")555        if index_name is None:556            raise ValueError("index_name must be provided")557 558        return cls(559            embedding=embedding,560            cluster=cluster,561            bucket_name=bucket_name,562            scope_name=scope_name,563            collection_name=collection_name,564            index_name=index_name,565            text_key=text_key,566            embedding_key=embedding_key,567            scoped_index=scoped_index,568        )569 570    @classmethod571    def from_texts(572        cls: Type[CouchbaseVectorStore],573        texts: List[str],574        embedding: Embeddings,575        metadatas: Optional[List[Dict[Any, Any]]] = None,576        **kwargs: Any,577    ) -> CouchbaseVectorStore:578        """Construct a Couchbase vector store from a list of texts.579 580        Example:581            .. code-block:: python582 583            from langchain_community.vectorstores import CouchbaseVectorStore584            from langchain_openai import OpenAIEmbeddings585 586            from couchbase.cluster import Cluster587            from couchbase.auth import PasswordAuthenticator588            from couchbase.options import ClusterOptions589            from datetime import timedelta590 591            auth = PasswordAuthenticator(username, password)592            options = ClusterOptions(auth)593            connect_string = "couchbases://localhost"594            cluster = Cluster(connect_string, options)595 596            # Wait until the cluster is ready for use.597            cluster.wait_until_ready(timedelta(seconds=5))598 599            embeddings = OpenAIEmbeddings()600 601            texts = ["hello", "world"]602 603            vectorstore = CouchbaseVectorStore.from_texts(604                texts,605                embedding=embeddings,606                cluster=cluster,607                bucket_name="",608                scope_name="",609                collection_name="",610                index_name="vector-index",611            )612 613        Args:614            texts (List[str]): list of texts to add to the vector store.615            embedding (Embeddings): embedding function to use.616            metadatas (optional[List[Dict]): list of metadatas to add to documents.617            **kwargs: Keyword arguments used to initialize the vector store with and/or618                passed to `add_texts` method. Check the constructor and/or `add_texts`619                for the list of accepted arguments.620 621        Returns:622            A Couchbase vector store.623 624        """625        vector_store = cls._from_kwargs(embedding, **kwargs)626        batch_size = kwargs.get("batch_size", vector_store.DEFAULT_BATCH_SIZE)627        ids = kwargs.get("ids", None)628        vector_store.add_texts(629            texts, metadatas=metadatas, ids=ids, batch_size=batch_size630        )631 632        return vector_store633 
codekingpro/portable-devtools · Team Ai