codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import uuid4from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Tuple, Type5 6from langchain_core._api.deprecation import deprecated7from langchain_core.documents import Document8from langchain_core.embeddings import Embeddings9from langchain_core.vectorstores import VectorStore10 11if TYPE_CHECKING:12 from couchbase.cluster import Cluster13 14 15@deprecated(16 since="0.2.4",17 removal="1.0",18 alternative_import="langchain_couchbase.CouchbaseSearchVectorStore",19)20class CouchbaseVectorStore(VectorStore):21 """`Couchbase Vector Store` vector store.22 23 To use it, you need24 - a recent installation of the `couchbase` library25 - a Couchbase database with a pre-defined Search index with support for26 vector fields27 28 Example:29 .. code-block:: python30 31 from langchain_community.vectorstores import CouchbaseVectorStore32 from langchain_openai import OpenAIEmbeddings33 34 from couchbase.cluster import Cluster35 from couchbase.auth import PasswordAuthenticator36 from couchbase.options import ClusterOptions37 from datetime import timedelta38 39 auth = PasswordAuthenticator(username, password)40 options = ClusterOptions(auth)41 connect_string = "couchbases://localhost"42 cluster = Cluster(connect_string, options)43 44 # Wait until the cluster is ready for use.45 cluster.wait_until_ready(timedelta(seconds=5))46 47 embeddings = OpenAIEmbeddings()48 49 vectorstore = CouchbaseVectorStore(50 cluster=cluster,51 bucket_name="",52 scope_name="",53 collection_name="",54 embedding=embeddings,55 index_name="vector-index",56 )57 58 vectorstore.add_texts(["hello", "world"])59 results = vectorstore.similarity_search("ola", k=1)60 """61 62 # Default batch size63 DEFAULT_BATCH_SIZE: int = 10064 _metadata_key: str = "metadata"65 _default_text_key: str = "text"66 _default_embedding_key: str = "embedding"67 68 def _check_bucket_exists(self) -> bool:69 """Check if the bucket exists in the linked Couchbase cluster"""70 bucket_manager = self._cluster.buckets()71 try:72 bucket_manager.get_bucket(self._bucket_name)73 return True74 except Exception:75 return False76 77 def _check_scope_and_collection_exists(self) -> bool:78 """Check if the scope and collection exists in the linked Couchbase bucket79 Raises a ValueError if either is not found"""80 scope_collection_map: Dict[str, Any] = {}81 82 # Get a list of all scopes in the bucket83 for scope in self._bucket.collections().get_all_scopes():84 scope_collection_map[scope.name] = []85 86 # Get a list of all the collections in the scope87 for collection in scope.collections:88 scope_collection_map[scope.name].append(collection.name)89 90 # Check if the scope exists91 if self._scope_name not in scope_collection_map.keys():92 raise ValueError(93 f"Scope {self._scope_name} not found in Couchbase "94 f"bucket {self._bucket_name}"95 )96 97 # Check if the collection exists in the scope98 if self._collection_name not in scope_collection_map[self._scope_name]:99 raise ValueError(100 f"Collection {self._collection_name} not found in scope "101 f"{self._scope_name} in Couchbase bucket {self._bucket_name}"102 )103 104 return True105 106 def _check_index_exists(self) -> bool:107 """Check if the Search index exists in the linked Couchbase cluster108 Raises a ValueError if the index does not exist"""109 if self._scoped_index:110 all_indexes = [111 index.name for index in self._scope.search_indexes().get_all_indexes()112 ]113 if self._index_name not in all_indexes:114 raise ValueError(115 f"Index {self._index_name} does not exist. "116 " Please create the index before searching."117 )118 else:119 all_indexes = [120 index.name for index in self._cluster.search_indexes().get_all_indexes()121 ]122 if self._index_name not in all_indexes:123 raise ValueError(124 f"Index {self._index_name} does not exist. "125 " Please create the index before searching."126 )127 128 return True129 130 def __init__(131 self,132 cluster: Cluster,133 bucket_name: str,134 scope_name: str,135 collection_name: str,136 embedding: Embeddings,137 index_name: str,138 *,139 text_key: Optional[str] = _default_text_key,140 embedding_key: Optional[str] = _default_embedding_key,141 scoped_index: bool = True,142 ) -> None:143 """144 Initialize the Couchbase Vector Store.145 146 Args:147 148 cluster (Cluster): couchbase cluster object with active connection.149 bucket_name (str): name of bucket to store documents in.150 scope_name (str): name of scope in the bucket to store documents in.151 collection_name (str): name of collection in the scope to store documents in152 embedding (Embeddings): embedding function to use.153 index_name (str): name of the Search index to use.154 text_key (optional[str]): key in document to use as text.155 Set to text by default.156 embedding_key (optional[str]): key in document to use for the embeddings.157 Set to embedding by default.158 scoped_index (optional[bool]): specify whether the index is a scoped index.159 Set to True by default.160 """161 try:162 from couchbase.cluster import Cluster163 except ImportError as e:164 raise ImportError(165 "Could not import couchbase python package. "166 "Please install couchbase SDK with `pip install couchbase`."167 ) from e168 169 if not isinstance(cluster, Cluster):170 raise ValueError(171 f"cluster should be an instance of couchbase.Cluster, "172 f"got {type(cluster)}"173 )174 175 self._cluster = cluster176 177 if not embedding:178 raise ValueError("Embeddings instance must be provided.")179 180 if not bucket_name:181 raise ValueError("bucket_name must be provided.")182 183 if not scope_name:184 raise ValueError("scope_name must be provided.")185 186 if not collection_name:187 raise ValueError("collection_name must be provided.")188 189 if not index_name:190 raise ValueError("index_name must be provided.")191 192 self._bucket_name = bucket_name193 self._scope_name = scope_name194 self._collection_name = collection_name195 self._embedding_function = embedding196 self._text_key = text_key197 self._embedding_key = embedding_key198 self._index_name = index_name199 self._scoped_index = scoped_index200 201 # Check if the bucket exists202 if not self._check_bucket_exists():203 raise ValueError(204 f"Bucket {self._bucket_name} does not exist. "205 " Please create the bucket before searching."206 )207 208 try:209 self._bucket = self._cluster.bucket(self._bucket_name)210 self._scope = self._bucket.scope(self._scope_name)211 self._collection = self._scope.collection(self._collection_name)212 except Exception as e:213 raise ValueError(214 "Error connecting to couchbase. "215 "Please check the connection and credentials."216 ) from e217 218 # Check if the scope and collection exists. Throws ValueError if they don't219 try:220 self._check_scope_and_collection_exists()221 except Exception as e:222 raise e223 224 # Check if the index exists. Throws ValueError if it doesn't225 try:226 self._check_index_exists()227 except Exception as e:228 raise e229 230 def add_texts(231 self,232 texts: Iterable[str],233 metadatas: Optional[List[Dict[str, Any]]] = None,234 ids: Optional[List[str]] = None,235 batch_size: Optional[int] = None,236 **kwargs: Any,237 ) -> List[str]:238 """Run texts through the embeddings and persist in vectorstore.239 240 If the document IDs are passed, the existing documents (if any) will be241 overwritten with the new ones.242 243 Args:244 texts (Iterable[str]): Iterable of strings to add to the vectorstore.245 metadatas (Optional[List[Dict]]): Optional list of metadatas associated246 with the texts.247 ids (Optional[List[str]]): Optional list of ids associated with the texts.248 IDs have to be unique strings across the collection.249 If it is not specified uuids are generated and used as ids.250 batch_size (Optional[int]): Optional batch size for bulk insertions.251 Default is 100.252 253 Returns:254 List[str]:List of ids from adding the texts into the vectorstore.255 """256 from couchbase.exceptions import DocumentExistsException257 258 if not batch_size:259 batch_size = self.DEFAULT_BATCH_SIZE260 doc_ids: List[str] = []261 262 if ids is None:263 ids = [uuid.uuid4().hex for _ in texts]264 265 if metadatas is None:266 metadatas = [{} for _ in texts]267 268 embedded_texts = self._embedding_function.embed_documents(list(texts))269 270 documents_to_insert = [271 {272 id: {273 self._text_key: text,274 self._embedding_key: vector,275 self._metadata_key: metadata,276 }277 for id, text, vector, metadata in zip(278 ids, texts, embedded_texts, metadatas279 )280 }281 ]282 283 # Insert in batches284 for i in range(0, len(documents_to_insert), batch_size):285 batch = documents_to_insert[i : i + batch_size]286 try:287 result = self._collection.upsert_multi(batch[0])288 if result.all_ok:289 doc_ids.extend(batch[0].keys())290 except DocumentExistsException as e:291 raise ValueError(f"Document already exists: {e}")292 293 return doc_ids294 295 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> Optional[bool]:296 """Delete documents from the vector store by ids.297 298 Args:299 ids (List[str]): List of IDs of the documents to delete.300 batch_size (Optional[int]): Optional batch size for bulk deletions.301 302 Returns:303 bool: True if all the documents were deleted successfully, False otherwise.304 305 """306 from couchbase.exceptions import DocumentNotFoundException307 308 if ids is None:309 raise ValueError("No document ids provided to delete.")310 311 batch_size = kwargs.get("batch_size", self.DEFAULT_BATCH_SIZE)312 deletion_status = True313 314 # Delete in batches315 for i in range(0, len(ids), batch_size):316 batch = ids[i : i + batch_size]317 try:318 result = self._collection.remove_multi(batch)319 except DocumentNotFoundException as e:320 deletion_status = False321 raise ValueError(f"Document not found: {e}")322 323 deletion_status &= result.all_ok324 325 return deletion_status326 327 @property328 def embeddings(self) -> Embeddings:329 """Return the query embedding object."""330 return self._embedding_function331 332 def _format_metadata(self, row_fields: Dict[str, Any]) -> Dict[str, Any]:333 """Helper method to format the metadata from the Couchbase Search API.334 Args:335 row_fields (Dict[str, Any]): The fields to format.336 337 Returns:338 Dict[str, Any]: The formatted metadata.339 """340 metadata = {}341 for key, value in row_fields.items():342 # Couchbase Search returns the metadata key with a prefix343 # `metadata.` We remove it to get the original metadata key344 if key.startswith(self._metadata_key):345 new_key = key.split(self._metadata_key + ".")[-1]346 metadata[new_key] = value347 else:348 metadata[key] = value349 350 return metadata351 352 def similarity_search_with_score_by_vector(353 self,354 embedding: List[float],355 k: int = 4,356 search_options: Optional[Dict[str, Any]] = {},357 **kwargs: Any,358 ) -> List[Tuple[Document, float]]:359 """Return docs most similar to embedding vector with their scores.360 361 Args:362 embedding (List[float]): Embedding vector to look up documents similar to.363 k (int): Number of Documents to return.364 Defaults to 4.365 search_options (Optional[Dict[str, Any]]): Optional search options that are366 passed to Couchbase search.367 Defaults to empty dictionary.368 fields (Optional[List[str]]): Optional list of fields to include in the369 metadata of results. Note that these need to be stored in the index.370 If nothing is specified, defaults to all the fields stored in the index.371 372 Returns:373 List of (Document, score) that are the most similar to the query vector.374 """375 import couchbase.search as search376 from couchbase.options import SearchOptions377 from couchbase.vector_search import VectorQuery, VectorSearch378 379 fields = kwargs.get("fields", ["*"])380 381 # Document text field needs to be returned from the search382 if fields != ["*"] and self._text_key not in fields:383 fields.append(self._text_key)384 385 search_req = search.SearchRequest.create(386 VectorSearch.from_vector_query(387 VectorQuery(388 self._embedding_key,389 embedding,390 k,391 )392 )393 )394 try:395 if self._scoped_index:396 search_iter = self._scope.search(397 self._index_name,398 search_req,399 SearchOptions(400 limit=k,401 fields=fields,402 raw=search_options,403 ),404 )405 406 else:407 search_iter = self._cluster.search(408 index=self._index_name,409 request=search_req,410 options=SearchOptions(limit=k, fields=fields, raw=search_options),411 )412 413 docs_with_score = []414 415 # Parse the results416 for row in search_iter.rows():417 text = row.fields.pop(self._text_key, "")418 419 # Format the metadata from Couchbase420 metadata = self._format_metadata(row.fields)421 422 score = row.score423 doc = Document(page_content=text, metadata=metadata)424 docs_with_score.append((doc, score))425 426 except Exception as e:427 raise ValueError(f"Search failed with error: {e}")428 429 return docs_with_score430 431 def similarity_search(432 self,433 query: str,434 k: int = 4,435 search_options: Optional[Dict[str, Any]] = {},436 **kwargs: Any,437 ) -> List[Document]:438 """Return documents most similar to embedding vector with their scores.439 440 Args:441 query (str): Query to look up for similar documents442 k (int): Number of Documents to return.443 Defaults to 4.444 search_options (Optional[Dict[str, Any]]): Optional search options that are445 passed to Couchbase search.446 Defaults to empty dictionary447 fields (Optional[List[str]]): Optional list of fields to include in the448 metadata of results. Note that these need to be stored in the index.449 If nothing is specified, defaults to all the fields stored in the index.450 451 Returns:452 List of Documents most similar to the query.453 """454 query_embedding = self.embeddings.embed_query(query)455 docs_with_scores = self.similarity_search_with_score_by_vector(456 query_embedding, k, search_options, **kwargs457 )458 return [doc for doc, _ in docs_with_scores]459 460 def similarity_search_with_score(461 self,462 query: str,463 k: int = 4,464 search_options: Optional[Dict[str, Any]] = {},465 **kwargs: Any,466 ) -> List[Tuple[Document, float]]:467 """Return documents that are most similar to the query with their scores.468 469 Args:470 query (str): Query to look up for similar documents471 k (int): Number of Documents to return.472 Defaults to 4.473 search_options (Optional[Dict[str, Any]]): Optional search options that are474 passed to Couchbase search.475 Defaults to empty dictionary.476 fields (Optional[List[str]]): Optional list of fields to include in the477 metadata of results. Note that these need to be stored in the index.478 If nothing is specified, defaults to text and metadata fields.479 480 Returns:481 List of (Document, score) that are most similar to the query.482 """483 query_embedding = self.embeddings.embed_query(query)484 docs_with_score = self.similarity_search_with_score_by_vector(485 query_embedding, k, search_options, **kwargs486 )487 return docs_with_score488 489 def similarity_search_by_vector(490 self,491 embedding: List[float],492 k: int = 4,493 search_options: Optional[Dict[str, Any]] = {},494 **kwargs: Any,495 ) -> List[Document]:496 """Return documents that are most similar to the vector embedding.497 498 Args:499 embedding (List[float]): Embedding to look up documents similar to.500 k (int): Number of Documents to return.501 Defaults to 4.502 search_options (Optional[Dict[str, Any]]): Optional search options that are503 passed to Couchbase search.504 Defaults to empty dictionary.505 fields (Optional[List[str]]): Optional list of fields to include in the506 metadata of results. Note that these need to be stored in the index.507 If nothing is specified, defaults to document text and metadata fields.508 509 Returns:510 List of Documents most similar to the query.511 """512 docs_with_score = self.similarity_search_with_score_by_vector(513 embedding, k, search_options, **kwargs514 )515 return [doc for doc, _ in docs_with_score]516 517 @classmethod518 def _from_kwargs(519 cls: Type[CouchbaseVectorStore],520 embedding: Embeddings,521 **kwargs: Any,522 ) -> CouchbaseVectorStore:523 """Initialize the Couchbase vector store from keyword arguments for the524 vector store.525 526 Args:527 embedding: Embedding object to use to embed text.528 **kwargs: Keyword arguments to initialize the vector store with.529 Accepted arguments are:530 - cluster531 - bucket_name532 - scope_name533 - collection_name534 - index_name535 - text_key536 - embedding_key537 - scoped_index538 539 """540 cluster = kwargs.get("cluster", None)541 bucket_name = kwargs.get("bucket_name", None)542 scope_name = kwargs.get("scope_name", None)543 collection_name = kwargs.get("collection_name", None)544 index_name = kwargs.get("index_name", None)545 text_key = kwargs.get("text_key", cls._default_text_key)546 embedding_key = kwargs.get("embedding_key", cls._default_embedding_key)547 scoped_index = kwargs.get("scoped_index", True)548 549 if bucket_name is None:550 raise ValueError("bucket_name must be provided")551 if scope_name is None:552 raise ValueError("scope_name must be provided")553 if collection_name is None:554 raise ValueError("collection_name must be provided")555 if index_name is None:556 raise ValueError("index_name must be provided")557 558 return cls(559 embedding=embedding,560 cluster=cluster,561 bucket_name=bucket_name,562 scope_name=scope_name,563 collection_name=collection_name,564 index_name=index_name,565 text_key=text_key,566 embedding_key=embedding_key,567 scoped_index=scoped_index,568 )569 570 @classmethod571 def from_texts(572 cls: Type[CouchbaseVectorStore],573 texts: List[str],574 embedding: Embeddings,575 metadatas: Optional[List[Dict[Any, Any]]] = None,576 **kwargs: Any,577 ) -> CouchbaseVectorStore:578 """Construct a Couchbase vector store from a list of texts.579 580 Example:581 .. code-block:: python582 583 from langchain_community.vectorstores import CouchbaseVectorStore584 from langchain_openai import OpenAIEmbeddings585 586 from couchbase.cluster import Cluster587 from couchbase.auth import PasswordAuthenticator588 from couchbase.options import ClusterOptions589 from datetime import timedelta590 591 auth = PasswordAuthenticator(username, password)592 options = ClusterOptions(auth)593 connect_string = "couchbases://localhost"594 cluster = Cluster(connect_string, options)595 596 # Wait until the cluster is ready for use.597 cluster.wait_until_ready(timedelta(seconds=5))598 599 embeddings = OpenAIEmbeddings()600 601 texts = ["hello", "world"]602 603 vectorstore = CouchbaseVectorStore.from_texts(604 texts,605 embedding=embeddings,606 cluster=cluster,607 bucket_name="",608 scope_name="",609 collection_name="",610 index_name="vector-index",611 )612 613 Args:614 texts (List[str]): list of texts to add to the vector store.615 embedding (Embeddings): embedding function to use.616 metadatas (optional[List[Dict]): list of metadatas to add to documents.617 **kwargs: Keyword arguments used to initialize the vector store with and/or618 passed to `add_texts` method. Check the constructor and/or `add_texts`619 for the list of accepted arguments.620 621 Returns:622 A Couchbase vector store.623 624 """625 vector_store = cls._from_kwargs(embedding, **kwargs)626 batch_size = kwargs.get("batch_size", vector_store.DEFAULT_BATCH_SIZE)627 ids = kwargs.get("ids", None)628 vector_store.add_texts(629 texts, metadatas=metadatas, ids=ids, batch_size=batch_size630 )631 632 return vector_store633 