codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import logging4from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, Union5 6import numpy as np7 8try:9 import deeplake10 from deeplake import VectorStore as DeepLakeVectorStore11 from deeplake.core.fast_forwarding import version_compare12 from deeplake.util.exceptions import SampleExtendError13 14 _DEEPLAKE_INSTALLED = True15except ImportError:16 _DEEPLAKE_INSTALLED = False17 18from langchain_core._api import deprecated19from langchain_core.documents import Document20from langchain_core.embeddings import Embeddings21from langchain_core.vectorstores import VectorStore22 23from langchain_community.vectorstores.utils import maximal_marginal_relevance24 25logger = logging.getLogger(__name__)26 27 28@deprecated(29 since="0.3.3",30 removal="1.0",31 message=(32 "This class is deprecated and will be removed in a future version. "33 "You can swap to using the `DeeplakeVectorStore`"34 " implementation in `langchain-deeplake`. "35 "Please do not submit further PRs to this class."36 "See <https://github.com/activeloopai/langchain-deeplake>"37 ),38 alternative_import="langchain_deeplake.DeeplakeVectorStore",39)40class DeepLake(VectorStore):41 """`Activeloop Deep Lake` vector store.42 43 We integrated deeplake's similarity search and filtering for fast prototyping.44 Now, it supports Tensor Query Language (TQL) for production use cases45 over billion rows.46 47 Why Deep Lake?48 49 - Not only stores embeddings, but also the original data with version control.50 - Serverless, doesn't require another service and can be used with major51 cloud providers (S3, GCS, etc.)52 - More than just a multi-modal vector store. You can use the dataset53 to fine-tune your own LLM models.54 55 To use, you should have the ``deeplake`` python package installed.56 57 Example:58 .. code-block:: python59 60 from langchain_community.vectorstores import DeepLake61 from langchain_community.embeddings.openai import OpenAIEmbeddings62 63 embeddings = OpenAIEmbeddings()64 vectorstore = DeepLake("langchain_store", embeddings.embed_query)65 """66 67 _LANGCHAIN_DEFAULT_DEEPLAKE_PATH: str = "./deeplake/"68 _valid_search_kwargs = ["lambda_mult"]69 70 def __init__(71 self,72 dataset_path: str = _LANGCHAIN_DEFAULT_DEEPLAKE_PATH,73 token: Optional[str] = None,74 embedding: Optional[Embeddings] = None,75 embedding_function: Optional[Embeddings] = None,76 read_only: bool = False,77 ingestion_batch_size: int = 1024,78 num_workers: int = 0,79 verbose: bool = True,80 exec_option: Optional[str] = None,81 runtime: Optional[Dict] = None,82 index_params: Optional[Dict[str, Union[int, str]]] = None,83 **kwargs: Any,84 ) -> None:85 """Creates an empty DeepLakeVectorStore or loads an existing one.86 87 The DeepLakeVectorStore is located at the specified ``path``.88 89 Examples:90 >>> # Create a vector store with default tensors91 >>> deeplake_vectorstore = DeepLake(92 ... path = <path_for_storing_Data>,93 ... )94 >>>95 >>> # Create a vector store in the Deep Lake Managed Tensor Database96 >>> data = DeepLake(97 ... path = "hub://org_id/dataset_name",98 ... runtime = {"tensor_db": True},99 ... )100 101 Args:102 dataset_path (str): The full path for storing to the Deep Lake103 Vector Store. It can be:104 - a Deep Lake cloud path of the form ``hub://org_id/dataset_name``.105 Requires registration with Deep Lake.106 - an s3 path of the form ``s3://bucketname/path/to/dataset``.107 Credentials are required in either the environment or passed to108 the creds argument.109 - a local file system path of the form ``./path/to/dataset``110 or ``~/path/to/dataset`` or ``path/to/dataset``.111 - a memory path of the form ``mem://path/to/dataset`` which doesn't112 save the dataset but keeps it in memory instead.113 Should be used only for testing as it does not persist.114 Defaults to _LANGCHAIN_DEFAULT_DEEPLAKE_PATH.115 token (str, optional): Activeloop token, for fetching credentials116 to the dataset at path if it is a Deep Lake dataset.117 Tokens are normally autogenerated. Optional.118 embedding (Embeddings, optional): Function to convert119 either documents or query. Optional.120 embedding_function (Embeddings, optional): Function to convert121 either documents or query. Optional. Deprecated: keeping this122 parameter for backwards compatibility.123 read_only (bool): Open dataset in read-only mode. Default is False.124 ingestion_batch_size (int): During data ingestion, data is divided125 into batches. Batch size is the size of each batch.126 Default is 1024.127 num_workers (int): Number of workers to use during data ingestion.128 Default is 0.129 verbose (bool): Print dataset summary after each operation.130 Default is True.131 exec_option (str, optional): Default method for search execution.132 It could be either ``"auto"``, ``"python"``, ``"compute_engine"``133 or ``"tensor_db"``. Defaults to ``"auto"``.134 If None, it's set to "auto".135 - ``auto``- Selects the best execution method based on the storage136 location of the Vector Store. It is the default option.137 - ``python`` - Pure-python implementation that runs on the client and138 can be used for data stored anywhere. WARNING: using this option139 with big datasets is discouraged because it can lead to140 memory issues.141 - ``compute_engine`` - Performant C++ implementation of the Deep Lake142 Compute Engine that runs on the client and can be used for any data143 stored in or connected to Deep Lake. It cannot be used with144 in-memory or local datasets.145 - ``tensor_db`` - Performant and fully-hosted Managed Tensor Database146 that is responsible for storage and query execution. Only available147 for data stored in the Deep Lake Managed Database. Store datasets148 in this database by specifying runtime = {"tensor_db": True}149 during dataset creation.150 runtime (Dict, optional): Parameters for creating the Vector Store in151 Deep Lake's Managed Tensor Database. Not applicable when loading an152 existing Vector Store. To create a Vector Store in the Managed Tensor153 Database, set `runtime = {"tensor_db": True}`.154 index_params (Optional[Dict[str, Union[int, str]]], optional): Dictionary155 containing information about vector index that will be created. Defaults156 to None, which will utilize ``DEFAULT_VECTORSTORE_INDEX_PARAMS`` from157 ``deeplake.constants``. The specified key-values override the default158 ones.159 - threshold: The threshold for the dataset size above which an index160 will be created for the embedding tensor. When the threshold value161 is set to -1, index creation is turned off. Defaults to -1, which162 turns off the index.163 - distance_metric: This key specifies the method of calculating the164 distance between vectors when creating the vector database (VDB)165 index. It can either be a string that corresponds to a member of166 the DistanceType enumeration, or the string value itself.167 - If no value is provided, it defaults to "L2".168 - "L2" corresponds to DistanceType.L2_NORM.169 - "COS" corresponds to DistanceType.COSINE_SIMILARITY.170 - additional_params: Additional parameters for fine-tuning the index.171 **kwargs: Other optional keyword arguments.172 173 Raises:174 ValueError: If some condition is not met.175 """176 177 self.ingestion_batch_size = ingestion_batch_size178 self.num_workers = num_workers179 self.verbose = verbose180 181 if _DEEPLAKE_INSTALLED is False:182 raise ImportError(183 "Could not import deeplake python package. "184 "Please install it with `pip install deeplake[enterprise]<4.0.0`."185 )186 187 if (188 runtime == {"tensor_db": True}189 and version_compare(deeplake.__version__, "3.6.7") == -1190 ):191 raise ImportError(192 "To use tensor_db option you need to update deeplake to `3.6.7` or "193 "higher. "194 f"Currently installed deeplake version is {deeplake.__version__}. "195 )196 197 self.dataset_path = dataset_path198 199 if embedding_function:200 logger.warning(201 "Using embedding function is deprecated and will be removed "202 "in the future. Please use embedding instead."203 )204 205 self.vectorstore = DeepLakeVectorStore(206 path=self.dataset_path,207 embedding_function=embedding_function or embedding,208 read_only=read_only,209 token=token,210 exec_option=exec_option,211 verbose=verbose,212 runtime=runtime,213 index_params=index_params,214 **kwargs,215 )216 217 self._embedding_function = embedding_function or embedding218 self._id_tensor_name = "ids" if "ids" in self.vectorstore.tensors() else "id"219 220 @property221 def embeddings(self) -> Optional[Embeddings]:222 return self._embedding_function223 224 def add_texts(225 self,226 texts: Iterable[str],227 metadatas: Optional[List[dict]] = None,228 ids: Optional[List[str]] = None,229 **kwargs: Any,230 ) -> List[str]:231 """Run more texts through the embeddings and add to the vectorstore.232 233 Examples:234 >>> ids = deeplake_vectorstore.add_texts(235 ... texts = <list_of_texts>,236 ... metadatas = <list_of_metadata_jsons>,237 ... ids = <list_of_ids>,238 ... )239 240 Args:241 texts (Iterable[str]): Texts to add to the vectorstore.242 metadatas (Optional[List[dict]], optional): Optional list of metadatas.243 ids (Optional[List[str]], optional): Optional list of IDs.244 embedding_function (Optional[Embeddings], optional): Embedding function245 to use to convert the text into embeddings.246 **kwargs (Any): Any additional keyword arguments passed is not supported247 by this method.248 249 Returns:250 List[str]: List of IDs of the added texts.251 """252 self._validate_kwargs(kwargs, "add_texts")253 254 kwargs = {}255 if ids:256 if self._id_tensor_name == "ids": # for backwards compatibility257 kwargs["ids"] = ids258 else:259 kwargs["id"] = ids260 261 if metadatas is None:262 metadatas = [{}] * len(list(texts))263 264 if not isinstance(texts, list):265 texts = list(texts)266 267 if texts is None:268 raise ValueError("`texts` parameter shouldn't be None.")269 elif len(texts) == 0:270 raise ValueError("`texts` parameter shouldn't be empty.")271 272 try:273 return self.vectorstore.add(274 text=texts,275 metadata=metadatas,276 embedding_data=texts,277 embedding_tensor="embedding",278 embedding_function=self._embedding_function.embed_documents, # type: ignore[union-attr]279 return_ids=True,280 **kwargs,281 )282 except SampleExtendError as e:283 if "Failed to append a sample to the tensor 'metadata'" in str(e):284 msg = (285 "**Hint: You might be using invalid type of argument in "286 "document loader (e.g. 'pathlib.PosixPath' instead of 'str')"287 )288 raise ValueError(e.args[0] + "\n\n" + msg)289 else:290 raise e291 292 def _search_tql(293 self,294 tql: Optional[str],295 exec_option: Optional[str] = None,296 **kwargs: Any,297 ) -> List[Document]:298 """Function for performing tql_search.299 300 Args:301 tql (str): TQL Query string for direct evaluation.302 Available only for `compute_engine` and `tensor_db`.303 exec_option (str, optional): Supports 3 ways to search.304 Could be "python", "compute_engine" or "tensor_db". Default is "python".305 - ``python`` - Pure-python implementation for the client.306 WARNING: not recommended for big datasets due to potential memory307 issues.308 - ``compute_engine`` - C++ implementation of Deep Lake Compute309 Engine for the client. Not for in-memory or local datasets.310 - ``tensor_db`` - Hosted Managed Tensor Database for storage311 and query execution. Only for data in Deep Lake Managed Database.312 Use runtime = {"db_engine": True} during dataset creation.313 return_score (bool): Return score with document. Default is False.314 315 Returns:316 Tuple[List[Document], List[Tuple[Document, float]]] - A tuple of two lists.317 The first list contains Documents, and the second list contains318 tuples of Document and float score.319 320 Raises:321 ValueError: If return_score is True but some condition is not met.322 """323 result = self.vectorstore.search(324 query=tql,325 exec_option=exec_option,326 )327 metadatas = result["metadata"]328 texts = result["text"]329 330 docs = [331 Document(332 page_content=text,333 metadata=metadata,334 )335 for text, metadata in zip(texts, metadatas)336 ]337 338 if kwargs:339 unsupported_argument = next(iter(kwargs))340 if kwargs[unsupported_argument] is not False:341 raise ValueError(342 f"specifying {unsupported_argument} is "343 "not supported with tql search."344 )345 346 return docs347 348 def _search(349 self,350 query: Optional[str] = None,351 embedding: Optional[Union[List[float], np.ndarray]] = None,352 embedding_function: Optional[Callable] = None,353 k: int = 4,354 distance_metric: Optional[str] = None,355 use_maximal_marginal_relevance: bool = False,356 fetch_k: Optional[int] = 20,357 filter: Optional[Union[Dict, Callable]] = None,358 return_score: bool = False,359 exec_option: Optional[str] = None,360 deep_memory: bool = False,361 **kwargs: Any,362 ) -> Any[List[Document], List[Tuple[Document, float]]]:363 """364 Return docs similar to query.365 366 Args:367 query (str, optional): Text to look up similar docs.368 embedding (Union[List[float], np.ndarray], optional): Query's embedding.369 embedding_function (Callable, optional): Function to convert `query`370 into embedding.371 k (int): Number of Documents to return.372 distance_metric (Optional[str], optional): `L2` for Euclidean, `L1` for373 Nuclear, `max` for L-infinity distance, `cos` for cosine similarity,374 'dot' for dot product.375 filter (Union[Dict, Callable], optional): Additional filter prior376 to the embedding search.377 - ``Dict`` - Key-value search on tensors of htype json, on an378 AND basis (a sample must satisfy all key-value filters to be True)379 Dict = {"tensor_name_1": {"key": value},380 "tensor_name_2": {"key": value}}381 - ``Function`` - Any function compatible with `deeplake.filter`.382 use_maximal_marginal_relevance (bool): Use maximal marginal relevance.383 fetch_k (int): Number of Documents for MMR algorithm.384 return_score (bool): Return the score.385 exec_option (str, optional): Supports 3 ways to perform searching.386 Could be "python", "compute_engine" or "tensor_db".387 - ``python`` - Pure-python implementation for the client.388 WARNING: not recommended for big datasets.389 - ``compute_engine`` - C++ implementation of Deep Lake Compute390 Engine for the client. Not for in-memory or local datasets.391 - ``tensor_db`` - Hosted Managed Tensor Database for storage392 and query execution. Only for data in Deep Lake Managed Database.393 Use runtime = {"db_engine": True} during dataset creation.394 deep_memory (bool): Whether to use the Deep Memory model for improving395 search results. Defaults to False if deep_memory is not specified in396 the Vector Store initialization. If True, the distance metric is set397 to "deepmemory_distance", which represents the metric with which the398 model was trained. The search is performed using the Deep Memory model.399 If False, the distance metric is set to "COS" or whatever distance400 metric user specifies.401 kwargs: Additional keyword arguments.402 403 Returns:404 List of Documents by the specified distance metric,405 if return_score True, return a tuple of (Document, score)406 407 Raises:408 ValueError: if both `embedding` and `embedding_function` are not specified.409 """410 if kwargs.get("tql_query"):411 logger.warning("`tql_query` is deprecated. Please use `tql` instead.")412 kwargs["tql"] = kwargs.pop("tql_query")413 414 if kwargs.get("tql"):415 return self._search_tql(416 tql=kwargs["tql"],417 exec_option=exec_option,418 return_score=return_score,419 embedding=embedding,420 embedding_function=embedding_function,421 distance_metric=distance_metric,422 use_maximal_marginal_relevance=use_maximal_marginal_relevance,423 filter=filter,424 )425 426 self._validate_kwargs(kwargs, "search")427 428 if embedding_function:429 if isinstance(embedding_function, Embeddings):430 _embedding_function = embedding_function.embed_query431 else:432 _embedding_function = embedding_function433 elif self._embedding_function:434 _embedding_function = self._embedding_function.embed_query435 else:436 _embedding_function = None437 438 if embedding is None:439 if _embedding_function is None:440 raise ValueError(441 "Either `embedding` or `embedding_function` needs to be specified."442 )443 444 embedding = _embedding_function(query) if query else None445 446 if isinstance(embedding, list):447 embedding = np.array(embedding, dtype=np.float32)448 if len(embedding.shape) > 1:449 embedding = embedding[0]450 451 result = self.vectorstore.search(452 embedding=embedding,453 k=fetch_k if use_maximal_marginal_relevance else k,454 distance_metric=distance_metric,455 filter=filter,456 exec_option=exec_option,457 return_tensors=["embedding", "metadata", "text", self._id_tensor_name],458 deep_memory=deep_memory,459 )460 scores = result["score"]461 embeddings = result["embedding"]462 metadatas = result["metadata"]463 texts = result["text"]464 465 if use_maximal_marginal_relevance:466 lambda_mult = kwargs.get("lambda_mult", 0.5)467 indices = maximal_marginal_relevance(468 embedding, # type: ignore[arg-type]469 embeddings,470 k=min(k, len(texts)),471 lambda_mult=lambda_mult,472 )473 474 scores = [scores[i] for i in indices]475 texts = [texts[i] for i in indices]476 metadatas = [metadatas[i] for i in indices]477 478 docs = [479 Document(480 page_content=text,481 metadata=metadata,482 )483 for text, metadata in zip(texts, metadatas)484 ]485 486 if return_score:487 if not isinstance(scores, list):488 scores = [scores]489 490 return [(doc, score) for doc, score in zip(docs, scores)]491 492 return docs493 494 def similarity_search(495 self,496 query: str,497 k: int = 4,498 **kwargs: Any,499 ) -> List[Document]:500 """501 Return docs most similar to query.502 503 Examples:504 >>> # Search using an embedding505 >>> data = vector_store.similarity_search(506 ... query=<your_query>,507 ... k=<num_items>,508 ... exec_option=<preferred_exec_option>,509 ... )510 >>> # Run tql search:511 >>> data = vector_store.similarity_search(512 ... query=None,513 ... tql="SELECT * WHERE id == <id>",514 ... exec_option="compute_engine",515 ... )516 517 Args:518 k (int): Number of Documents to return. Defaults to 4.519 query (str): Text to look up similar documents.520 kwargs: Additional keyword arguments include:521 embedding (Callable): Embedding function to use. Defaults to None.522 distance_metric (str): 'L2' for Euclidean, 'L1' for Nuclear, 'max'523 for L-infinity, 'cos' for cosine, 'dot' for dot product.524 Defaults to 'L2'.525 filter (Union[Dict, Callable], optional): Additional filter526 before embedding search.527 - Dict: Key-value search on tensors of htype json,528 (sample must satisfy all key-value filters)529 Dict = {"tensor_1": {"key": value}, "tensor_2": {"key": value}}530 - Function: Compatible with `deeplake.filter`.531 Defaults to None.532 exec_option (str): Supports 3 ways to perform searching.533 'python', 'compute_engine', or 'tensor_db'. Defaults to 'python'.534 - 'python': Pure-python implementation for the client.535 WARNING: not recommended for big datasets.536 - 'compute_engine': C++ implementation of the Compute Engine for537 the client. Not for in-memory or local datasets.538 - 'tensor_db': Managed Tensor Database for storage and query.539 Only for data in Deep Lake Managed Database.540 Use `runtime = {"db_engine": True}` during dataset creation.541 deep_memory (bool): Whether to use the Deep Memory model for improving542 search results. Defaults to False if deep_memory is not specified543 in the Vector Store initialization. If True, the distance metric544 is set to "deepmemory_distance", which represents the metric with545 which the model was trained. The search is performed using the Deep546 Memory model. If False, the distance metric is set to "COS" or547 whatever distance metric user specifies.548 549 Returns:550 List[Document]: List of Documents most similar to the query vector.551 """552 553 return self._search(554 query=query,555 k=k,556 use_maximal_marginal_relevance=False,557 return_score=False,558 **kwargs,559 )560 561 def similarity_search_by_vector(562 self,563 embedding: Union[List[float], np.ndarray],564 k: int = 4,565 **kwargs: Any,566 ) -> List[Document]:567 """568 Return docs most similar to embedding vector.569 570 Examples:571 >>> # Search using an embedding572 >>> data = vector_store.similarity_search_by_vector(573 ... embedding=<your_embedding>,574 ... k=<num_items_to_return>,575 ... exec_option=<preferred_exec_option>,576 ... )577 578 Args:579 embedding (Union[List[float], np.ndarray]):580 Embedding to find similar docs.581 k (int): Number of Documents to return. Defaults to 4.582 kwargs: Additional keyword arguments including:583 filter (Union[Dict, Callable], optional):584 Additional filter before embedding search.585 - ``Dict`` - Key-value search on tensors of htype json. True586 if all key-value filters are satisfied.587 Dict = {"tensor_name_1": {"key": value},588 "tensor_name_2": {"key": value}}589 - ``Function`` - Any function compatible with590 `deeplake.filter`.591 Defaults to None.592 exec_option (str): Options for search execution include593 "python", "compute_engine", or "tensor_db". Defaults to594 "python".595 - "python" - Pure-python implementation running on the client.596 Can be used for data stored anywhere. WARNING: using this597 option with big datasets is discouraged due to potential598 memory issues.599 - "compute_engine" - Performant C++ implementation of the Deep600 Lake Compute Engine. Runs on the client and can be used for601 any data stored in or connected to Deep Lake. It cannot be602 used with in-memory or local datasets.603 - "tensor_db" - Performant, fully-hosted Managed Tensor Database.604 Responsible for storage and query execution. Only available605 for data stored in the Deep Lake Managed Database.606 To store datasets in this database, specify607 `runtime = {"db_engine": True}` during dataset creation.608 distance_metric (str): `L2` for Euclidean, `L1` for Nuclear,609 `max` for L-infinity distance, `cos` for cosine similarity,610 'dot' for dot product. Defaults to `L2`.611 deep_memory (bool): Whether to use the Deep Memory model for improving612 search results. Defaults to False if deep_memory is not specified613 in the Vector Store initialization. If True, the distance metric614 is set to "deepmemory_distance", which represents the metric with615 which the model was trained. The search is performed using the Deep616 Memory model. If False, the distance metric is set to "COS" or617 whatever distance metric user specifies.618 619 Returns:620 List[Document]: List of Documents most similar to the query vector.621 """622 623 return self._search(624 embedding=embedding,625 k=k,626 use_maximal_marginal_relevance=False,627 return_score=False,628 **kwargs,629 )630 631 def similarity_search_with_score(632 self,633 query: str,634 k: int = 4,635 **kwargs: Any,636 ) -> List[Tuple[Document, float]]:637 """638 Run similarity search with Deep Lake with distance returned.639 640 Examples:641 >>> data = vector_store.similarity_search_with_score(642 ... query=<your_query>,643 ... embedding=<your_embedding_function>644 ... k=<number_of_items_to_return>,645 ... exec_option=<preferred_exec_option>,646 ... )647 648 Args:649 query (str): Query text to search for.650 k (int): Number of results to return. Defaults to 4.651 kwargs: Additional keyword arguments. Some of these arguments are:652 distance_metric: `L2` for Euclidean, `L1` for Nuclear, `max` L-infinity653 distance, `cos` for cosine similarity, 'dot' for dot product.654 Defaults to `L2`.655 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.656 embedding_function (Callable): Embedding function to use. Defaults657 to None.658 exec_option (str): DeepLakeVectorStore supports 3 ways to perform659 searching. It could be either "python", "compute_engine" or660 "tensor_db". Defaults to "python".661 - "python" - Pure-python implementation running on the client.662 Can be used for data stored anywhere. WARNING: using this663 option with big datasets is discouraged due to potential664 memory issues.665 - "compute_engine" - Performant C++ implementation of the Deep666 Lake Compute Engine. Runs on the client and can be used for667 any data stored in or connected to Deep Lake. It cannot be used668 with in-memory or local datasets.669 - "tensor_db" - Performant, fully-hosted Managed Tensor Database.670 Responsible for storage and query execution. Only available for671 data stored in the Deep Lake Managed Database. To store datasets672 in this database, specify `runtime = {"db_engine": True}`673 during dataset creation.674 deep_memory (bool): Whether to use the Deep Memory model for improving675 search results. Defaults to False if deep_memory is not specified676 in the Vector Store initialization. If True, the distance metric677 is set to "deepmemory_distance", which represents the metric with678 which the model was trained. The search is performed using the Deep679 Memory model. If False, the distance metric is set to "COS" or680 whatever distance metric user specifies.681 682 Returns:683 List[Tuple[Document, float]]: List of documents most similar to the query684 text with distance in float."""685 686 return self._search(687 query=query,688 k=k,689 return_score=True,690 **kwargs,691 )692 693 def max_marginal_relevance_search_by_vector(694 self,695 embedding: List[float],696 k: int = 4,697 fetch_k: int = 20,698 lambda_mult: float = 0.5,699 exec_option: Optional[str] = None,700 **kwargs: Any,701 ) -> List[Document]:702 """703 Return docs selected using the maximal marginal relevance. Maximal marginal704 relevance optimizes for similarity to query AND diversity among selected docs.705 706 Examples:707 >>> data = vector_store.max_marginal_relevance_search_by_vector(708 ... embedding=<your_embedding>,709 ... fetch_k=<elements_to_fetch_before_mmr_search>,710 ... k=<number_of_items_to_return>,711 ... exec_option=<preferred_exec_option>,712 ... )713 714 Args:715 embedding: Embedding to look up documents similar to.716 k: Number of Documents to return. Defaults to 4.717 fetch_k: Number of Documents to fetch for MMR algorithm.718 lambda_mult: Number between 0 and 1 determining the degree of diversity.719 0 corresponds to max diversity and 1 to min diversity. Defaults to 0.5.720 exec_option (str): DeepLakeVectorStore supports 3 ways for searching.721 Could be "python", "compute_engine" or "tensor_db". Defaults to722 "python".723 - "python" - Pure-python implementation running on the client.724 Can be used for data stored anywhere. WARNING: using this725 option with big datasets is discouraged due to potential726 memory issues.727 - "compute_engine" - Performant C++ implementation of the Deep728 Lake Compute Engine. Runs on the client and can be used for729 any data stored in or connected to Deep Lake. It cannot be used730 with in-memory or local datasets.731 - "tensor_db" - Performant, fully-hosted Managed Tensor Database.732 Responsible for storage and query execution. Only available for733 data stored in the Deep Lake Managed Database. To store datasets734 in this database, specify `runtime = {"db_engine": True}`735 during dataset creation.736 deep_memory (bool): Whether to use the Deep Memory model for improving737 search results. Defaults to False if deep_memory is not specified738 in the Vector Store initialization. If True, the distance metric739 is set to "deepmemory_distance", which represents the metric with740 which the model was trained. The search is performed using the Deep741 Memory model. If False, the distance metric is set to "COS" or742 whatever distance metric user specifies.743 kwargs: Additional keyword arguments.744 745 Returns:746 List[Documents] - A list of documents.747 """748 749 return self._search(750 embedding=embedding,751 k=k,752 fetch_k=fetch_k,753 use_maximal_marginal_relevance=True,754 lambda_mult=lambda_mult,755 exec_option=exec_option,756 **kwargs,757 )758 759 def max_marginal_relevance_search(760 self,761 query: str,762 k: int = 4,763 fetch_k: int = 20,764 lambda_mult: float = 0.5,765 exec_option: Optional[str] = None,766 **kwargs: Any,767 ) -> List[Document]:768 """Return docs selected using maximal marginal relevance.769 770 Maximal marginal relevance optimizes for similarity to query AND diversity771 among selected documents.772 773 Examples:774 >>> # Search using an embedding775 >>> data = vector_store.max_marginal_relevance_search(776 ... query = <query_to_search>,777 ... embedding_function = <embedding_function_for_query>,778 ... k = <number_of_items_to_return>,779 ... exec_option = <preferred_exec_option>,780 ... )781 782 Args:783 query: Text to look up documents similar to.784 k: Number of Documents to return. Defaults to 4.785 fetch_k: Number of Documents for MMR algorithm.786 lambda_mult: Value between 0 and 1. 0 corresponds787 to maximum diversity and 1 to minimum.788 Defaults to 0.5.789 exec_option (str): Supports 3 ways to perform searching.790 - "python" - Pure-python implementation running on the client.791 Can be used for data stored anywhere. WARNING: using this792 option with big datasets is discouraged due to potential793 memory issues.794 - "compute_engine" - Performant C++ implementation of the Deep795 Lake Compute Engine. Runs on the client and can be used for796 any data stored in or connected to Deep Lake. It cannot be797 used with in-memory or local datasets.798 - "tensor_db" - Performant, fully-hosted Managed Tensor Database.799 Responsible for storage and query execution. Only available800 for data stored in the Deep Lake Managed Database. To store801 datasets in this database, specify802 `runtime = {"db_engine": True}` during dataset creation.803 deep_memory (bool): Whether to use the Deep Memory model for improving804 search results. Defaults to False if deep_memory is not specified805 in the Vector Store initialization. If True, the distance metric806 is set to "deepmemory_distance", which represents the metric with807 which the model was trained. The search is performed using the Deep808 Memory model. If False, the distance metric is set to "COS" or809 whatever distance metric user specifies.810 kwargs: Additional keyword arguments811 812 Returns:813 List of Documents selected by maximal marginal relevance.814 815 Raises:816 ValueError: when MRR search is on but embedding function is817 not specified.818 """819 embedding_function = kwargs.get("embedding") or self._embedding_function820 if embedding_function is None:821 raise ValueError(822 "For MMR search, you must specify an embedding function on"823 " `creation` or during add call."824 )825 return self._search(826 query=query,827 k=k,828 fetch_k=fetch_k,829 use_maximal_marginal_relevance=True,830 lambda_mult=lambda_mult,831 exec_option=exec_option,832 embedding_function=embedding_function, # type: ignore[arg-type]833 **kwargs,834 )835 836 @classmethod837 def from_texts(838 cls,839 texts: List[str],840 embedding: Optional[Embeddings] = None,841 metadatas: Optional[List[dict]] = None,842 ids: Optional[List[str]] = None,843 dataset_path: str = _LANGCHAIN_DEFAULT_DEEPLAKE_PATH,844 **kwargs: Any,845 ) -> DeepLake:846 """Create a Deep Lake dataset from a raw documents.847 848 If a dataset_path is specified, the dataset will be persisted in that location,849 otherwise by default at `./deeplake`850 851 Examples:852 >>> # Search using an embedding853 >>> vector_store = DeepLake.from_texts(854 ... texts = <the_texts_that_you_want_to_embed>,855 ... embedding_function = <embedding_function_for_query>,856 ... k = <number_of_items_to_return>,857 ... exec_option = <preferred_exec_option>,858 ... )859 860 Args:861 dataset_path (str): - The full path to the dataset. Can be:862 - Deep Lake cloud path of the form ``hub://username/dataset_name``.863 To write to Deep Lake cloud datasets,864 ensure that you are logged in to Deep Lake865 (use 'activeloop login' from command line)866 - AWS S3 path of the form ``s3://bucketname/path/to/dataset``.867 Credentials are required in either the environment868 - Google Cloud Storage path of the form869 ``gcs://bucketname/path/to/dataset`` Credentials are required870 in either the environment871 - Local file system path of the form ``./path/to/dataset`` or872 ``~/path/to/dataset`` or ``path/to/dataset``.873 - In-memory path of the form ``mem://path/to/dataset`` which doesn't874 save the dataset, but keeps it in memory instead.875 Should be used only for testing as it does not persist.876 texts (List[Document]): List of documents to add.877 embedding (Optional[Embeddings]): Embedding function. Defaults to None.878 Note, in other places, it is called embedding_function.879 metadatas (Optional[List[dict]]): List of metadatas. Defaults to None.880 ids (Optional[List[str]]): List of document IDs. Defaults to None.881 kwargs: Additional keyword arguments.882 883 Returns:884 DeepLake: Deep Lake dataset.885 """886 deeplake_dataset = cls(dataset_path=dataset_path, embedding=embedding, **kwargs)887 deeplake_dataset.add_texts(888 texts=texts,889 metadatas=metadatas,890 ids=ids,891 )892 return deeplake_dataset893 894 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> bool:895 """Delete the entities in the dataset.896 897 Args:898 ids (Optional[List[str]], optional): The document_ids to delete.899 Defaults to None.900 **kwargs: Other keyword arguments that subclasses might use.901 - filter (Optional[Dict[str, str]], optional): The filter to delete by.902 - delete_all (Optional[bool], optional): Whether to drop the dataset.903 904 Returns:905 bool: Whether the delete operation was successful.906 """907 filter = kwargs.get("filter")908 delete_all = kwargs.get("delete_all")909 910 self.vectorstore.delete(ids=ids, filter=filter, delete_all=delete_all)911 912 return True913 914 @classmethod915 def force_delete_by_path(cls, path: str) -> None:916 """Force delete dataset by path.917 918 Args:919 path (str): path of the dataset to delete.920 921 Raises:922 ValueError: if deeplake is not installed.923 """924 925 try:926 import deeplake927 except ImportError:928 raise ImportError(929 "Could not import deeplake python package. "930 "Please install it with `pip install deeplake`."931 )932 deeplake.delete(path, large_ok=True, force=True)933 934 def delete_dataset(self) -> None:935 """Delete the collection."""936 self.delete(delete_all=True)937 938 def ds(self) -> Any:939 logger.warning(940 "this method is deprecated and will be removed, "941 "better to use `db.vectorstore.dataset` instead."942 )943 return self.vectorstore.dataset944 945 @classmethod946 def _validate_kwargs(cls, kwargs: Any, method_name: str) -> None:947 if kwargs:948 valid_items = cls._get_valid_args(method_name)949 unsupported_items = cls._get_unsupported_items(kwargs, valid_items)950 951 if unsupported_items:952 raise TypeError(953 f"`{unsupported_items}` are not a valid "954 f"argument to {method_name} method"955 )956 957 @classmethod958 def _get_valid_args(cls, method_name: str) -> list[str]:959 if method_name == "search":960 return cls._valid_search_kwargs961 else:962 return []963 964 @staticmethod965 def _get_unsupported_items(kwargs: Any, valid_items: list[str]) -> Optional[str]:966 kwargs = {k: v for k, v in kwargs.items() if k not in valid_items}967 unsupported_items = None968 if kwargs:969 unsupported_items = "`, `".join(set(kwargs.keys()))970 return unsupported_items971 