Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
deeplake.py971 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import logging4from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, Union5 6import numpy as np7 8try:9    import deeplake10    from deeplake import VectorStore as DeepLakeVectorStore11    from deeplake.core.fast_forwarding import version_compare12    from deeplake.util.exceptions import SampleExtendError13 14    _DEEPLAKE_INSTALLED = True15except ImportError:16    _DEEPLAKE_INSTALLED = False17 18from langchain_core._api import deprecated19from langchain_core.documents import Document20from langchain_core.embeddings import Embeddings21from langchain_core.vectorstores import VectorStore22 23from langchain_community.vectorstores.utils import maximal_marginal_relevance24 25logger = logging.getLogger(__name__)26 27 28@deprecated(29    since="0.3.3",30    removal="1.0",31    message=(32        "This class is deprecated and will be removed in a future version. "33        "You can swap to using the `DeeplakeVectorStore`"34        " implementation in `langchain-deeplake`. "35        "Please do not submit further PRs to this class."36        "See <https://github.com/activeloopai/langchain-deeplake>"37    ),38    alternative_import="langchain_deeplake.DeeplakeVectorStore",39)40class DeepLake(VectorStore):41    """`Activeloop Deep Lake` vector store.42 43    We integrated deeplake's similarity search and filtering for fast prototyping.44    Now, it supports Tensor Query Language (TQL) for production use cases45    over billion rows.46 47    Why Deep Lake?48 49    - Not only stores embeddings, but also the original data with version control.50    - Serverless, doesn't require another service and can be used with major51        cloud providers (S3, GCS, etc.)52    - More than just a multi-modal vector store. You can use the dataset53        to fine-tune your own LLM models.54 55    To use, you should have the ``deeplake`` python package installed.56 57    Example:58        .. code-block:: python59 60                from langchain_community.vectorstores import DeepLake61                from langchain_community.embeddings.openai import OpenAIEmbeddings62 63                embeddings = OpenAIEmbeddings()64                vectorstore = DeepLake("langchain_store", embeddings.embed_query)65    """66 67    _LANGCHAIN_DEFAULT_DEEPLAKE_PATH: str = "./deeplake/"68    _valid_search_kwargs = ["lambda_mult"]69 70    def __init__(71        self,72        dataset_path: str = _LANGCHAIN_DEFAULT_DEEPLAKE_PATH,73        token: Optional[str] = None,74        embedding: Optional[Embeddings] = None,75        embedding_function: Optional[Embeddings] = None,76        read_only: bool = False,77        ingestion_batch_size: int = 1024,78        num_workers: int = 0,79        verbose: bool = True,80        exec_option: Optional[str] = None,81        runtime: Optional[Dict] = None,82        index_params: Optional[Dict[str, Union[int, str]]] = None,83        **kwargs: Any,84    ) -> None:85        """Creates an empty DeepLakeVectorStore or loads an existing one.86 87        The DeepLakeVectorStore is located at the specified ``path``.88 89        Examples:90            >>> # Create a vector store with default tensors91            >>> deeplake_vectorstore = DeepLake(92            ...        path = <path_for_storing_Data>,93            ... )94            >>>95            >>> # Create a vector store in the Deep Lake Managed Tensor Database96            >>> data = DeepLake(97            ...        path = "hub://org_id/dataset_name",98            ...        runtime = {"tensor_db": True},99            ... )100 101        Args:102            dataset_path (str): The full path for storing to the Deep Lake103                Vector Store. It can be:104                - a Deep Lake cloud path of the form ``hub://org_id/dataset_name``.105                    Requires registration with Deep Lake.106                - an s3 path of the form ``s3://bucketname/path/to/dataset``.107                    Credentials are required in either the environment or passed to108                    the creds argument.109                - a local file system path of the form ``./path/to/dataset``110                    or ``~/path/to/dataset`` or ``path/to/dataset``.111                - a memory path of the form ``mem://path/to/dataset`` which doesn't112                    save the dataset but keeps it in memory instead.113                    Should be used only for testing as it does not persist.114                    Defaults to _LANGCHAIN_DEFAULT_DEEPLAKE_PATH.115            token (str, optional):  Activeloop token, for fetching credentials116                to the dataset at path if it is a Deep Lake dataset.117                Tokens are normally autogenerated. Optional.118            embedding (Embeddings, optional): Function to convert119                either documents or query. Optional.120            embedding_function (Embeddings, optional): Function to convert121                either documents or query. Optional. Deprecated: keeping this122                parameter for backwards compatibility.123            read_only (bool): Open dataset in read-only mode. Default is False.124            ingestion_batch_size (int): During data ingestion, data is divided125                into batches. Batch size is the size of each batch.126                Default is 1024.127            num_workers (int): Number of workers to use during data ingestion.128                Default is 0.129            verbose (bool): Print dataset summary after each operation.130                Default is True.131            exec_option (str, optional): Default method for search execution.132                It could be either ``"auto"``, ``"python"``, ``"compute_engine"``133                or ``"tensor_db"``. Defaults to ``"auto"``.134                If None, it's set to "auto".135                - ``auto``- Selects the best execution method based on the storage136                    location of the Vector Store. It is the default option.137                - ``python`` - Pure-python implementation that runs on the client and138                    can be used for data stored anywhere. WARNING: using this option139                    with big datasets is discouraged because it can lead to140                    memory issues.141                - ``compute_engine`` - Performant C++ implementation of the Deep Lake142                    Compute Engine that runs on the client and can be used for any data143                    stored in or connected to Deep Lake. It cannot be used with144                    in-memory or local datasets.145                - ``tensor_db`` - Performant and fully-hosted Managed Tensor Database146                    that is responsible for storage and query execution. Only available147                    for data stored in the Deep Lake Managed Database. Store datasets148                    in this database by specifying runtime = {"tensor_db": True}149                    during dataset creation.150            runtime (Dict, optional): Parameters for creating the Vector Store in151                Deep Lake's Managed Tensor Database. Not applicable when loading an152                existing Vector Store. To create a Vector Store in the Managed Tensor153                Database, set `runtime = {"tensor_db": True}`.154            index_params (Optional[Dict[str, Union[int, str]]], optional): Dictionary155                containing information about vector index that will be created. Defaults156                to None, which will utilize ``DEFAULT_VECTORSTORE_INDEX_PARAMS`` from157                ``deeplake.constants``. The specified key-values override the default158                ones.159                - threshold: The threshold for the dataset size above which an index160                    will be created for the embedding tensor. When the threshold value161                    is set to -1, index creation is turned off. Defaults to -1, which162                    turns off the index.163                - distance_metric: This key specifies the method of calculating the164                    distance between vectors when creating the vector database (VDB)165                    index. It can either be a string that corresponds to a member of166                    the DistanceType enumeration, or the string value itself.167                    - If no value is provided, it defaults to "L2".168                    - "L2" corresponds to DistanceType.L2_NORM.169                    - "COS" corresponds to DistanceType.COSINE_SIMILARITY.170                - additional_params: Additional parameters for fine-tuning the index.171            **kwargs: Other optional keyword arguments.172 173        Raises:174            ValueError: If some condition is not met.175        """176 177        self.ingestion_batch_size = ingestion_batch_size178        self.num_workers = num_workers179        self.verbose = verbose180 181        if _DEEPLAKE_INSTALLED is False:182            raise ImportError(183                "Could not import deeplake python package. "184                "Please install it with `pip install deeplake[enterprise]<4.0.0`."185            )186 187        if (188            runtime == {"tensor_db": True}189            and version_compare(deeplake.__version__, "3.6.7") == -1190        ):191            raise ImportError(192                "To use tensor_db option you need to update deeplake to `3.6.7` or "193                "higher. "194                f"Currently installed deeplake version is {deeplake.__version__}. "195            )196 197        self.dataset_path = dataset_path198 199        if embedding_function:200            logger.warning(201                "Using embedding function is deprecated and will be removed "202                "in the future. Please use embedding instead."203            )204 205        self.vectorstore = DeepLakeVectorStore(206            path=self.dataset_path,207            embedding_function=embedding_function or embedding,208            read_only=read_only,209            token=token,210            exec_option=exec_option,211            verbose=verbose,212            runtime=runtime,213            index_params=index_params,214            **kwargs,215        )216 217        self._embedding_function = embedding_function or embedding218        self._id_tensor_name = "ids" if "ids" in self.vectorstore.tensors() else "id"219 220    @property221    def embeddings(self) -> Optional[Embeddings]:222        return self._embedding_function223 224    def add_texts(225        self,226        texts: Iterable[str],227        metadatas: Optional[List[dict]] = None,228        ids: Optional[List[str]] = None,229        **kwargs: Any,230    ) -> List[str]:231        """Run more texts through the embeddings and add to the vectorstore.232 233        Examples:234            >>> ids = deeplake_vectorstore.add_texts(235            ...     texts = <list_of_texts>,236            ...     metadatas = <list_of_metadata_jsons>,237            ...     ids = <list_of_ids>,238            ... )239 240        Args:241            texts (Iterable[str]): Texts to add to the vectorstore.242            metadatas (Optional[List[dict]], optional): Optional list of metadatas.243            ids (Optional[List[str]], optional): Optional list of IDs.244            embedding_function (Optional[Embeddings], optional): Embedding function245                to use to convert the text into embeddings.246            **kwargs (Any): Any additional keyword arguments passed is not supported247                by this method.248 249        Returns:250            List[str]: List of IDs of the added texts.251        """252        self._validate_kwargs(kwargs, "add_texts")253 254        kwargs = {}255        if ids:256            if self._id_tensor_name == "ids":  # for backwards compatibility257                kwargs["ids"] = ids258            else:259                kwargs["id"] = ids260 261        if metadatas is None:262            metadatas = [{}] * len(list(texts))263 264        if not isinstance(texts, list):265            texts = list(texts)266 267        if texts is None:268            raise ValueError("`texts` parameter shouldn't be None.")269        elif len(texts) == 0:270            raise ValueError("`texts` parameter shouldn't be empty.")271 272        try:273            return self.vectorstore.add(274                text=texts,275                metadata=metadatas,276                embedding_data=texts,277                embedding_tensor="embedding",278                embedding_function=self._embedding_function.embed_documents,  # type: ignore[union-attr]279                return_ids=True,280                **kwargs,281            )282        except SampleExtendError as e:283            if "Failed to append a sample to the tensor 'metadata'" in str(e):284                msg = (285                    "**Hint: You might be using invalid type of argument in "286                    "document loader (e.g. 'pathlib.PosixPath' instead of 'str')"287                )288                raise ValueError(e.args[0] + "\n\n" + msg)289            else:290                raise e291 292    def _search_tql(293        self,294        tql: Optional[str],295        exec_option: Optional[str] = None,296        **kwargs: Any,297    ) -> List[Document]:298        """Function for performing tql_search.299 300        Args:301            tql (str): TQL Query string for direct evaluation.302                Available only for `compute_engine` and `tensor_db`.303            exec_option (str, optional): Supports 3 ways to search.304                Could be "python", "compute_engine" or "tensor_db". Default is "python".305                - ``python`` - Pure-python implementation for the client.306                    WARNING: not recommended for big datasets due to potential memory307                    issues.308                - ``compute_engine`` - C++ implementation of Deep Lake Compute309                    Engine for the client. Not for in-memory or local datasets.310                - ``tensor_db`` - Hosted Managed Tensor Database for storage311                    and query execution. Only for data in Deep Lake Managed Database.312                        Use runtime = {"db_engine": True} during dataset creation.313            return_score (bool): Return score with document. Default is False.314 315        Returns:316            Tuple[List[Document], List[Tuple[Document, float]]] - A tuple of two lists.317                The first list contains Documents, and the second list contains318                tuples of Document and float score.319 320        Raises:321            ValueError: If return_score is True but some condition is not met.322        """323        result = self.vectorstore.search(324            query=tql,325            exec_option=exec_option,326        )327        metadatas = result["metadata"]328        texts = result["text"]329 330        docs = [331            Document(332                page_content=text,333                metadata=metadata,334            )335            for text, metadata in zip(texts, metadatas)336        ]337 338        if kwargs:339            unsupported_argument = next(iter(kwargs))340            if kwargs[unsupported_argument] is not False:341                raise ValueError(342                    f"specifying {unsupported_argument} is "343                    "not supported with tql search."344                )345 346        return docs347 348    def _search(349        self,350        query: Optional[str] = None,351        embedding: Optional[Union[List[float], np.ndarray]] = None,352        embedding_function: Optional[Callable] = None,353        k: int = 4,354        distance_metric: Optional[str] = None,355        use_maximal_marginal_relevance: bool = False,356        fetch_k: Optional[int] = 20,357        filter: Optional[Union[Dict, Callable]] = None,358        return_score: bool = False,359        exec_option: Optional[str] = None,360        deep_memory: bool = False,361        **kwargs: Any,362    ) -> Any[List[Document], List[Tuple[Document, float]]]:363        """364        Return docs similar to query.365 366        Args:367            query (str, optional): Text to look up similar docs.368            embedding (Union[List[float], np.ndarray], optional): Query's embedding.369            embedding_function (Callable, optional): Function to convert `query`370                into embedding.371            k (int): Number of Documents to return.372            distance_metric (Optional[str], optional): `L2` for Euclidean, `L1` for373                Nuclear, `max` for L-infinity distance, `cos` for cosine similarity,374                'dot' for dot product.375            filter (Union[Dict, Callable], optional): Additional filter prior376                to the embedding search.377                - ``Dict`` - Key-value search on tensors of htype json, on an378                    AND basis (a sample must satisfy all key-value filters to be True)379                    Dict = {"tensor_name_1": {"key": value},380                            "tensor_name_2": {"key": value}}381                - ``Function`` - Any function compatible with `deeplake.filter`.382            use_maximal_marginal_relevance (bool): Use maximal marginal relevance.383            fetch_k (int): Number of Documents for MMR algorithm.384            return_score (bool): Return the score.385            exec_option (str, optional): Supports 3 ways to perform searching.386                Could be "python", "compute_engine" or "tensor_db".387                - ``python`` - Pure-python implementation for the client.388                    WARNING: not recommended for big datasets.389                - ``compute_engine`` - C++ implementation of Deep Lake Compute390                    Engine for the client. Not for in-memory or local datasets.391                - ``tensor_db`` - Hosted Managed Tensor Database for storage392                    and query execution. Only for data in Deep Lake Managed Database.393                    Use runtime = {"db_engine": True} during dataset creation.394            deep_memory (bool): Whether to use the Deep Memory model for improving395                search results. Defaults to False if deep_memory is not specified in396                the Vector Store initialization. If True, the distance metric is set397                to "deepmemory_distance", which represents the metric with which the398                model was trained. The search is performed using the Deep Memory model.399                If False, the distance metric is set to "COS" or whatever distance400                metric user specifies.401            kwargs: Additional keyword arguments.402 403        Returns:404            List of Documents by the specified distance metric,405            if return_score True, return a tuple of (Document, score)406 407        Raises:408            ValueError: if both `embedding` and `embedding_function` are not specified.409        """410        if kwargs.get("tql_query"):411            logger.warning("`tql_query` is deprecated. Please use `tql` instead.")412            kwargs["tql"] = kwargs.pop("tql_query")413 414        if kwargs.get("tql"):415            return self._search_tql(416                tql=kwargs["tql"],417                exec_option=exec_option,418                return_score=return_score,419                embedding=embedding,420                embedding_function=embedding_function,421                distance_metric=distance_metric,422                use_maximal_marginal_relevance=use_maximal_marginal_relevance,423                filter=filter,424            )425 426        self._validate_kwargs(kwargs, "search")427 428        if embedding_function:429            if isinstance(embedding_function, Embeddings):430                _embedding_function = embedding_function.embed_query431            else:432                _embedding_function = embedding_function433        elif self._embedding_function:434            _embedding_function = self._embedding_function.embed_query435        else:436            _embedding_function = None437 438        if embedding is None:439            if _embedding_function is None:440                raise ValueError(441                    "Either `embedding` or `embedding_function` needs to be specified."442                )443 444            embedding = _embedding_function(query) if query else None445 446        if isinstance(embedding, list):447            embedding = np.array(embedding, dtype=np.float32)448            if len(embedding.shape) > 1:449                embedding = embedding[0]450 451        result = self.vectorstore.search(452            embedding=embedding,453            k=fetch_k if use_maximal_marginal_relevance else k,454            distance_metric=distance_metric,455            filter=filter,456            exec_option=exec_option,457            return_tensors=["embedding", "metadata", "text", self._id_tensor_name],458            deep_memory=deep_memory,459        )460        scores = result["score"]461        embeddings = result["embedding"]462        metadatas = result["metadata"]463        texts = result["text"]464 465        if use_maximal_marginal_relevance:466            lambda_mult = kwargs.get("lambda_mult", 0.5)467            indices = maximal_marginal_relevance(468                embedding,  # type: ignore[arg-type]469                embeddings,470                k=min(k, len(texts)),471                lambda_mult=lambda_mult,472            )473 474            scores = [scores[i] for i in indices]475            texts = [texts[i] for i in indices]476            metadatas = [metadatas[i] for i in indices]477 478        docs = [479            Document(480                page_content=text,481                metadata=metadata,482            )483            for text, metadata in zip(texts, metadatas)484        ]485 486        if return_score:487            if not isinstance(scores, list):488                scores = [scores]489 490            return [(doc, score) for doc, score in zip(docs, scores)]491 492        return docs493 494    def similarity_search(495        self,496        query: str,497        k: int = 4,498        **kwargs: Any,499    ) -> List[Document]:500        """501        Return docs most similar to query.502 503        Examples:504            >>> # Search using an embedding505            >>> data = vector_store.similarity_search(506            ...     query=<your_query>,507            ...     k=<num_items>,508            ...     exec_option=<preferred_exec_option>,509            ... )510            >>> # Run tql search:511            >>> data = vector_store.similarity_search(512            ...     query=None,513            ...     tql="SELECT * WHERE id == <id>",514            ...     exec_option="compute_engine",515            ... )516 517        Args:518            k (int): Number of Documents to return. Defaults to 4.519            query (str): Text to look up similar documents.520            kwargs: Additional keyword arguments include:521                embedding (Callable): Embedding function to use. Defaults to None.522                distance_metric (str): 'L2' for Euclidean, 'L1' for Nuclear, 'max'523                    for L-infinity, 'cos' for cosine, 'dot' for dot product.524                    Defaults to 'L2'.525                filter (Union[Dict, Callable], optional): Additional filter526                    before embedding search.527                    - Dict: Key-value search on tensors of htype json,528                        (sample must satisfy all key-value filters)529                        Dict = {"tensor_1": {"key": value}, "tensor_2": {"key": value}}530                    - Function: Compatible with `deeplake.filter`.531                    Defaults to None.532                exec_option (str): Supports 3 ways to perform searching.533                    'python', 'compute_engine', or 'tensor_db'. Defaults to 'python'.534                    - 'python': Pure-python implementation for the client.535                        WARNING: not recommended for big datasets.536                    - 'compute_engine': C++ implementation of the Compute Engine for537                        the client. Not for in-memory or local datasets.538                    - 'tensor_db': Managed Tensor Database for storage and query.539                        Only for data in Deep Lake Managed Database.540                        Use `runtime = {"db_engine": True}` during dataset creation.541                deep_memory (bool): Whether to use the Deep Memory model for improving542                    search results. Defaults to False if deep_memory is not specified543                    in the Vector Store initialization. If True, the distance metric544                    is set to "deepmemory_distance", which represents the metric with545                    which the model was trained. The search is performed using the Deep546                    Memory model. If False, the distance metric is set to "COS" or547                    whatever distance metric user specifies.548 549        Returns:550            List[Document]: List of Documents most similar to the query vector.551        """552 553        return self._search(554            query=query,555            k=k,556            use_maximal_marginal_relevance=False,557            return_score=False,558            **kwargs,559        )560 561    def similarity_search_by_vector(562        self,563        embedding: Union[List[float], np.ndarray],564        k: int = 4,565        **kwargs: Any,566    ) -> List[Document]:567        """568        Return docs most similar to embedding vector.569 570        Examples:571            >>> # Search using an embedding572            >>> data = vector_store.similarity_search_by_vector(573            ...    embedding=<your_embedding>,574            ...    k=<num_items_to_return>,575            ...    exec_option=<preferred_exec_option>,576            ... )577 578        Args:579            embedding (Union[List[float], np.ndarray]):580                Embedding to find similar docs.581            k (int): Number of Documents to return. Defaults to 4.582            kwargs: Additional keyword arguments including:583                filter (Union[Dict, Callable], optional):584                    Additional filter before embedding search.585                    - ``Dict`` - Key-value search on tensors of htype json. True586                        if all key-value filters are satisfied.587                        Dict = {"tensor_name_1": {"key": value},588                                "tensor_name_2": {"key": value}}589                    - ``Function`` - Any function compatible with590                        `deeplake.filter`.591                    Defaults to None.592                exec_option (str): Options for search execution include593                    "python", "compute_engine", or "tensor_db". Defaults to594                    "python".595                    - "python" - Pure-python implementation running on the client.596                        Can be used for data stored anywhere. WARNING: using this597                        option with big datasets is discouraged due to potential598                        memory issues.599                    - "compute_engine" - Performant C++ implementation of the Deep600                        Lake Compute Engine. Runs on the client and can be used for601                        any data stored in or connected to Deep Lake. It cannot be602                        used with in-memory or local datasets.603                    - "tensor_db" - Performant, fully-hosted Managed Tensor Database.604                        Responsible for storage and query execution. Only available605                        for data stored in the Deep Lake Managed Database.606                        To store datasets in this database, specify607                        `runtime = {"db_engine": True}` during dataset creation.608                distance_metric (str): `L2` for Euclidean, `L1` for Nuclear,609                    `max` for L-infinity distance, `cos` for cosine similarity,610                    'dot' for dot product. Defaults to `L2`.611                deep_memory (bool): Whether to use the Deep Memory model for improving612                    search results. Defaults to False if deep_memory is not specified613                    in the Vector Store initialization. If True, the distance metric614                    is set to "deepmemory_distance", which represents the metric with615                    which the model was trained. The search is performed using the Deep616                    Memory model. If False, the distance metric is set to "COS" or617                    whatever distance metric user specifies.618 619        Returns:620            List[Document]: List of Documents most similar to the query vector.621        """622 623        return self._search(624            embedding=embedding,625            k=k,626            use_maximal_marginal_relevance=False,627            return_score=False,628            **kwargs,629        )630 631    def similarity_search_with_score(632        self,633        query: str,634        k: int = 4,635        **kwargs: Any,636    ) -> List[Tuple[Document, float]]:637        """638        Run similarity search with Deep Lake with distance returned.639 640        Examples:641        >>> data = vector_store.similarity_search_with_score(642        ...     query=<your_query>,643        ...     embedding=<your_embedding_function>644        ...     k=<number_of_items_to_return>,645        ...     exec_option=<preferred_exec_option>,646        ... )647 648        Args:649            query (str): Query text to search for.650            k (int): Number of results to return. Defaults to 4.651            kwargs: Additional keyword arguments. Some of these arguments are:652                distance_metric: `L2` for Euclidean, `L1` for Nuclear, `max` L-infinity653                    distance, `cos` for cosine similarity, 'dot' for dot product.654                    Defaults to `L2`.655                filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.656                    embedding_function (Callable): Embedding function to use. Defaults657                    to None.658                exec_option (str): DeepLakeVectorStore supports 3 ways to perform659                    searching. It could be either "python", "compute_engine" or660                    "tensor_db". Defaults to "python".661                    - "python" - Pure-python implementation running on the client.662                        Can be used for data stored anywhere. WARNING: using this663                        option with big datasets is discouraged due to potential664                        memory issues.665                    - "compute_engine" - Performant C++ implementation of the Deep666                        Lake Compute Engine. Runs on the client and can be used for667                        any data stored in or connected to Deep Lake. It cannot be used668                        with in-memory or local datasets.669                    - "tensor_db" - Performant, fully-hosted Managed Tensor Database.670                        Responsible for storage and query execution. Only available for671                        data stored in the Deep Lake Managed Database. To store datasets672                        in this database, specify `runtime = {"db_engine": True}`673                        during dataset creation.674                deep_memory (bool): Whether to use the Deep Memory model for improving675                    search results. Defaults to False if deep_memory is not specified676                    in the Vector Store initialization. If True, the distance metric677                    is set to "deepmemory_distance", which represents the metric with678                    which the model was trained. The search is performed using the Deep679                    Memory model. If False, the distance metric is set to "COS" or680                    whatever distance metric user specifies.681 682        Returns:683            List[Tuple[Document, float]]: List of documents most similar to the query684                text with distance in float."""685 686        return self._search(687            query=query,688            k=k,689            return_score=True,690            **kwargs,691        )692 693    def max_marginal_relevance_search_by_vector(694        self,695        embedding: List[float],696        k: int = 4,697        fetch_k: int = 20,698        lambda_mult: float = 0.5,699        exec_option: Optional[str] = None,700        **kwargs: Any,701    ) -> List[Document]:702        """703        Return docs selected using the maximal marginal relevance. Maximal marginal704        relevance optimizes for similarity to query AND diversity among selected docs.705 706        Examples:707        >>> data = vector_store.max_marginal_relevance_search_by_vector(708        ...        embedding=<your_embedding>,709        ...        fetch_k=<elements_to_fetch_before_mmr_search>,710        ...        k=<number_of_items_to_return>,711        ...        exec_option=<preferred_exec_option>,712        ... )713 714        Args:715            embedding: Embedding to look up documents similar to.716            k: Number of Documents to return. Defaults to 4.717            fetch_k: Number of Documents to fetch for MMR algorithm.718            lambda_mult: Number between 0 and 1 determining the degree of diversity.719                0 corresponds to max diversity and 1 to min diversity. Defaults to 0.5.720            exec_option (str): DeepLakeVectorStore supports 3 ways for searching.721                Could be "python", "compute_engine" or "tensor_db". Defaults to722                "python".723                - "python" - Pure-python implementation running on the client.724                    Can be used for data stored anywhere. WARNING: using this725                    option with big datasets is discouraged due to potential726                    memory issues.727                - "compute_engine" - Performant C++ implementation of the Deep728                    Lake Compute Engine. Runs on the client and can be used for729                    any data stored in or connected to Deep Lake. It cannot be used730                    with in-memory or local datasets.731                - "tensor_db" - Performant, fully-hosted Managed Tensor Database.732                    Responsible for storage and query execution. Only available for733                    data stored in the Deep Lake Managed Database. To store datasets734                    in this database, specify `runtime = {"db_engine": True}`735                    during dataset creation.736            deep_memory (bool): Whether to use the Deep Memory model for improving737                search results. Defaults to False if deep_memory is not specified738                in the Vector Store initialization. If True, the distance metric739                is set to "deepmemory_distance", which represents the metric with740                which the model was trained. The search is performed using the Deep741                Memory model. If False, the distance metric is set to "COS" or742                whatever distance metric user specifies.743            kwargs: Additional keyword arguments.744 745        Returns:746            List[Documents] - A list of documents.747        """748 749        return self._search(750            embedding=embedding,751            k=k,752            fetch_k=fetch_k,753            use_maximal_marginal_relevance=True,754            lambda_mult=lambda_mult,755            exec_option=exec_option,756            **kwargs,757        )758 759    def max_marginal_relevance_search(760        self,761        query: str,762        k: int = 4,763        fetch_k: int = 20,764        lambda_mult: float = 0.5,765        exec_option: Optional[str] = None,766        **kwargs: Any,767    ) -> List[Document]:768        """Return docs selected using maximal marginal relevance.769 770        Maximal marginal relevance optimizes for similarity to query AND diversity771        among selected documents.772 773        Examples:774        >>> # Search using an embedding775        >>> data = vector_store.max_marginal_relevance_search(776        ...        query = <query_to_search>,777        ...        embedding_function = <embedding_function_for_query>,778        ...        k = <number_of_items_to_return>,779        ...        exec_option = <preferred_exec_option>,780        ... )781 782        Args:783            query: Text to look up documents similar to.784            k: Number of Documents to return. Defaults to 4.785            fetch_k: Number of Documents for MMR algorithm.786            lambda_mult: Value between 0 and 1. 0 corresponds787                        to maximum diversity and 1 to minimum.788                        Defaults to 0.5.789            exec_option (str): Supports 3 ways to perform searching.790                - "python" - Pure-python implementation running on the client.791                        Can be used for data stored anywhere. WARNING: using this792                        option with big datasets is discouraged due to potential793                        memory issues.794                    - "compute_engine" - Performant C++ implementation of the Deep795                        Lake Compute Engine. Runs on the client and can be used for796                        any data stored in or connected to Deep Lake. It cannot be797                        used with in-memory or local datasets.798                    - "tensor_db" - Performant, fully-hosted Managed Tensor Database.799                        Responsible for storage and query execution. Only available800                        for data stored in the Deep Lake Managed Database. To store801                        datasets in this database, specify802                        `runtime = {"db_engine": True}` during dataset creation.803            deep_memory (bool): Whether to use the Deep Memory model for improving804                search results. Defaults to False if deep_memory is not specified805                in the Vector Store initialization. If True, the distance metric806                is set to "deepmemory_distance", which represents the metric with807                which the model was trained. The search is performed using the Deep808                Memory model. If False, the distance metric is set to "COS" or809                whatever distance metric user specifies.810            kwargs: Additional keyword arguments811 812        Returns:813            List of Documents selected by maximal marginal relevance.814 815        Raises:816            ValueError: when MRR search is on but embedding function is817                not specified.818        """819        embedding_function = kwargs.get("embedding") or self._embedding_function820        if embedding_function is None:821            raise ValueError(822                "For MMR search, you must specify an embedding function on"823                " `creation` or during add call."824            )825        return self._search(826            query=query,827            k=k,828            fetch_k=fetch_k,829            use_maximal_marginal_relevance=True,830            lambda_mult=lambda_mult,831            exec_option=exec_option,832            embedding_function=embedding_function,  # type: ignore[arg-type]833            **kwargs,834        )835 836    @classmethod837    def from_texts(838        cls,839        texts: List[str],840        embedding: Optional[Embeddings] = None,841        metadatas: Optional[List[dict]] = None,842        ids: Optional[List[str]] = None,843        dataset_path: str = _LANGCHAIN_DEFAULT_DEEPLAKE_PATH,844        **kwargs: Any,845    ) -> DeepLake:846        """Create a Deep Lake dataset from a raw documents.847 848        If a dataset_path is specified, the dataset will be persisted in that location,849        otherwise by default at `./deeplake`850 851        Examples:852        >>> # Search using an embedding853        >>> vector_store = DeepLake.from_texts(854        ...        texts = <the_texts_that_you_want_to_embed>,855        ...        embedding_function = <embedding_function_for_query>,856        ...        k = <number_of_items_to_return>,857        ...        exec_option = <preferred_exec_option>,858        ... )859 860        Args:861            dataset_path (str): - The full path to the dataset. Can be:862                - Deep Lake cloud path of the form ``hub://username/dataset_name``.863                    To write to Deep Lake cloud datasets,864                    ensure that you are logged in to Deep Lake865                    (use 'activeloop login' from command line)866                - AWS S3 path of the form ``s3://bucketname/path/to/dataset``.867                    Credentials are required in either the environment868                - Google Cloud Storage path of the form869                    ``gcs://bucketname/path/to/dataset`` Credentials are required870                    in either the environment871                - Local file system path of the form ``./path/to/dataset`` or872                    ``~/path/to/dataset`` or ``path/to/dataset``.873                - In-memory path of the form ``mem://path/to/dataset`` which doesn't874                    save the dataset, but keeps it in memory instead.875                    Should be used only for testing as it does not persist.876            texts (List[Document]): List of documents to add.877            embedding (Optional[Embeddings]): Embedding function. Defaults to None.878                Note, in other places, it is called embedding_function.879            metadatas (Optional[List[dict]]): List of metadatas. Defaults to None.880            ids (Optional[List[str]]): List of document IDs. Defaults to None.881            kwargs: Additional keyword arguments.882 883        Returns:884            DeepLake: Deep Lake dataset.885        """886        deeplake_dataset = cls(dataset_path=dataset_path, embedding=embedding, **kwargs)887        deeplake_dataset.add_texts(888            texts=texts,889            metadatas=metadatas,890            ids=ids,891        )892        return deeplake_dataset893 894    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> bool:895        """Delete the entities in the dataset.896 897        Args:898            ids (Optional[List[str]], optional): The document_ids to delete.899                Defaults to None.900            **kwargs: Other keyword arguments that subclasses might use.901                - filter (Optional[Dict[str, str]], optional): The filter to delete by.902                - delete_all (Optional[bool], optional): Whether to drop the dataset.903 904        Returns:905            bool: Whether the delete operation was successful.906        """907        filter = kwargs.get("filter")908        delete_all = kwargs.get("delete_all")909 910        self.vectorstore.delete(ids=ids, filter=filter, delete_all=delete_all)911 912        return True913 914    @classmethod915    def force_delete_by_path(cls, path: str) -> None:916        """Force delete dataset by path.917 918        Args:919            path (str): path of the dataset to delete.920 921        Raises:922            ValueError: if deeplake is not installed.923        """924 925        try:926            import deeplake927        except ImportError:928            raise ImportError(929                "Could not import deeplake python package. "930                "Please install it with `pip install deeplake`."931            )932        deeplake.delete(path, large_ok=True, force=True)933 934    def delete_dataset(self) -> None:935        """Delete the collection."""936        self.delete(delete_all=True)937 938    def ds(self) -> Any:939        logger.warning(940            "this method is deprecated and will be removed, "941            "better to use `db.vectorstore.dataset` instead."942        )943        return self.vectorstore.dataset944 945    @classmethod946    def _validate_kwargs(cls, kwargs: Any, method_name: str) -> None:947        if kwargs:948            valid_items = cls._get_valid_args(method_name)949            unsupported_items = cls._get_unsupported_items(kwargs, valid_items)950 951            if unsupported_items:952                raise TypeError(953                    f"`{unsupported_items}` are not a valid "954                    f"argument to {method_name} method"955                )956 957    @classmethod958    def _get_valid_args(cls, method_name: str) -> list[str]:959        if method_name == "search":960            return cls._valid_search_kwargs961        else:962            return []963 964    @staticmethod965    def _get_unsupported_items(kwargs: Any, valid_items: list[str]) -> Optional[str]:966        kwargs = {k: v for k, v in kwargs.items() if k not in valid_items}967        unsupported_items = None968        if kwargs:969            unsupported_items = "`, `".join(set(kwargs.keys()))970        return unsupported_items971 
codekingpro/portable-devtools · Team Ai