Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
faiss.py1484 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import logging4import operator5import os6import pickle7import uuid8import warnings9from pathlib import Path10from typing import (11    Any,12    Callable,13    Dict,14    Iterable,15    List,16    Optional,17    Sequence,18    Sized,19    Tuple,20    Union,21)22 23import numpy as np24from langchain_core.documents import Document25from langchain_core.embeddings import Embeddings26from langchain_core.runnables.config import run_in_executor27from langchain_core.vectorstores import VectorStore28 29from langchain_community.docstore.base import AddableMixin, Docstore30from langchain_community.docstore.in_memory import InMemoryDocstore31from langchain_community.vectorstores.utils import (32    DistanceStrategy,33    maximal_marginal_relevance,34)35 36logger = logging.getLogger(__name__)37 38 39def dependable_faiss_import(no_avx2: Optional[bool] = None) -> Any:40    """41    Import faiss if available, otherwise raise error.42    If FAISS_NO_AVX2 environment variable is set, it will be considered43    to load FAISS with no AVX2 optimization.44 45    Args:46        no_avx2: Load FAISS strictly with no AVX2 optimization47            so that the vectorstore is portable and compatible with other devices.48    """49    if no_avx2 is None and "FAISS_NO_AVX2" in os.environ:50        no_avx2 = bool(os.getenv("FAISS_NO_AVX2"))51 52    try:53        if no_avx2:54            from faiss import swigfaiss as faiss55        else:56            import faiss57    except ImportError:58        raise ImportError(59            "Could not import faiss python package. "60            "Please install it with `pip install faiss-gpu` (for CUDA supported GPU) "61            "or `pip install faiss-cpu` (depending on Python version)."62        )63    return faiss64 65 66def _len_check_if_sized(x: Any, y: Any, x_name: str, y_name: str) -> None:67    if isinstance(x, Sized) and isinstance(y, Sized) and len(x) != len(y):68        raise ValueError(69            f"{x_name} and {y_name} expected to be equal length but "70            f"len({x_name})={len(x)} and len({y_name})={len(y)}"71        )72    return73 74 75class FAISS(VectorStore):76    """FAISS vector store integration.77 78    See [The FAISS Library](https://arxiv.org/pdf/2401.08281) paper.79 80    Setup:81        Install ``langchain_community`` and ``faiss-cpu`` python packages.82 83        .. code-block:: bash84 85            pip install -qU langchain_community faiss-cpu86 87    Key init args — indexing params:88        embedding_function: Embeddings89            Embedding function to use.90 91    Key init args — client params:92        index: Any93            FAISS index to use.94        docstore: Docstore95            Docstore to use.96        index_to_docstore_id: Dict[int, str]97            Mapping of index to docstore id.98 99    Instantiate:100        .. code-block:: python101 102            import faiss103            from langchain_community.vectorstores import FAISS104            from langchain_community.docstore.in_memory import InMemoryDocstore105            from langchain_openai import OpenAIEmbeddings106 107            index = faiss.IndexFlatL2(len(OpenAIEmbeddings().embed_query("hello world")))108 109            vector_store = FAISS(110                embedding_function=OpenAIEmbeddings(),111                index=index,112                docstore= InMemoryDocstore(),113                index_to_docstore_id={}114            )115 116    Add Documents:117        .. code-block:: python118 119            from langchain_core.documents import Document120 121            document_1 = Document(page_content="foo", metadata={"baz": "bar"})122            document_2 = Document(page_content="thud", metadata={"bar": "baz"})123            document_3 = Document(page_content="i will be deleted :(")124 125            documents = [document_1, document_2, document_3]126            ids = ["1", "2", "3"]127            vector_store.add_documents(documents=documents, ids=ids)128 129    Delete Documents:130        .. code-block:: python131 132            vector_store.delete(ids=["3"])133 134    Search:135        .. code-block:: python136 137            results = vector_store.similarity_search(query="thud",k=1)138            for doc in results:139                print(f"* {doc.page_content} [{doc.metadata}]")140 141        .. code-block:: python142 143            * thud [{'bar': 'baz'}]144 145    Search with filter:146        .. code-block:: python147 148            results = vector_store.similarity_search(query="thud",k=1,filter={"bar": "baz"})149            for doc in results:150                print(f"* {doc.page_content} [{doc.metadata}]")151 152        .. code-block:: python153 154            * thud [{'bar': 'baz'}]155 156    Search with score:157        .. code-block:: python158 159            results = vector_store.similarity_search_with_score(query="qux",k=1)160            for doc, score in results:161                print(f"* [SIM={score:3f}] {doc.page_content} [{doc.metadata}]")162 163        .. code-block:: python164 165            * [SIM=0.335304] foo [{'baz': 'bar'}]166 167    Async:168        .. code-block:: python169 170            # add documents171            # await vector_store.aadd_documents(documents=documents, ids=ids)172 173            # delete documents174            # await vector_store.adelete(ids=["3"])175 176            # search177            # results = vector_store.asimilarity_search(query="thud",k=1)178 179            # search with score180            results = await vector_store.asimilarity_search_with_score(query="qux",k=1)181            for doc,score in results:182                print(f"* [SIM={score:3f}] {doc.page_content} [{doc.metadata}]")183 184        .. code-block:: python185 186            * [SIM=0.335304] foo [{'baz': 'bar'}]187 188    Use as Retriever:189        .. code-block:: python190 191            retriever = vector_store.as_retriever(192                search_type="mmr",193                search_kwargs={"k": 1, "fetch_k": 2, "lambda_mult": 0.5},194            )195            retriever.invoke("thud")196 197        .. code-block:: python198 199            [Document(metadata={'bar': 'baz'}, page_content='thud')]200 201    """  # noqa: E501202 203    def __init__(204        self,205        embedding_function: Union[206            Callable[[str], List[float]],207            Embeddings,208        ],209        index: Any,210        docstore: Docstore,211        index_to_docstore_id: Dict[int, str],212        relevance_score_fn: Optional[Callable[[float], float]] = None,213        normalize_L2: bool = False,214        distance_strategy: DistanceStrategy = DistanceStrategy.EUCLIDEAN_DISTANCE,215    ):216        """Initialize with necessary components."""217        if not isinstance(embedding_function, Embeddings):218            logger.warning(219                "`embedding_function` is expected to be an Embeddings object, support "220                "for passing in a function will soon be removed."221            )222        self.embedding_function = embedding_function223        self.index = index224        self.docstore = docstore225        self.index_to_docstore_id = index_to_docstore_id226        self.distance_strategy = distance_strategy227        self.override_relevance_score_fn = relevance_score_fn228        self._normalize_L2 = normalize_L2229        if (230            self.distance_strategy != DistanceStrategy.EUCLIDEAN_DISTANCE231            and self._normalize_L2232        ):233            warnings.warn(234                "Normalizing L2 is not applicable for "235                f"metric type: {self.distance_strategy}"236            )237 238    @property239    def embeddings(self) -> Optional[Embeddings]:240        return (241            self.embedding_function242            if isinstance(self.embedding_function, Embeddings)243            else None244        )245 246    def _embed_documents(self, texts: List[str]) -> List[List[float]]:247        if isinstance(self.embedding_function, Embeddings):248            return self.embedding_function.embed_documents(texts)249        else:250            return [self.embedding_function(text) for text in texts]251 252    async def _aembed_documents(self, texts: List[str]) -> List[List[float]]:253        if isinstance(self.embedding_function, Embeddings):254            return await self.embedding_function.aembed_documents(texts)255        else:256            # return await asyncio.gather(257            #     [self.embedding_function(text) for text in texts]258            # )259            raise Exception(260                "`embedding_function` is expected to be an Embeddings object, support "261                "for passing in a function will soon be removed."262            )263 264    def _embed_query(self, text: str) -> List[float]:265        if isinstance(self.embedding_function, Embeddings):266            return self.embedding_function.embed_query(text)267        else:268            return self.embedding_function(text)269 270    async def _aembed_query(self, text: str) -> List[float]:271        if isinstance(self.embedding_function, Embeddings):272            return await self.embedding_function.aembed_query(text)273        else:274            # return await self.embedding_function(text)275            raise Exception(276                "`embedding_function` is expected to be an Embeddings object, support "277                "for passing in a function will soon be removed."278            )279 280    def __add(281        self,282        texts: Iterable[str],283        embeddings: Iterable[List[float]],284        metadatas: Optional[Iterable[dict]] = None,285        ids: Optional[List[str]] = None,286    ) -> List[str]:287        faiss = dependable_faiss_import()288        if not isinstance(self.docstore, AddableMixin):289            raise ValueError(290                "If trying to add texts, the underlying docstore should support "291                f"adding items, which {self.docstore} does not"292            )293 294        _len_check_if_sized(texts, metadatas, "texts", "metadatas")295 296        ids = ids or [str(uuid.uuid4()) for _ in texts]297        _len_check_if_sized(texts, ids, "texts", "ids")298 299        _metadatas = metadatas or ({} for _ in texts)300        documents = [301            Document(id=id_, page_content=t, metadata=m)302            for id_, t, m in zip(ids, texts, _metadatas)303        ]304 305        _len_check_if_sized(documents, embeddings, "documents", "embeddings")306 307        if ids and len(ids) != len(set(ids)):308            raise ValueError("Duplicate ids found in the ids list.")309        # Add to the index.310        vector = np.array(embeddings, dtype=np.float32)311        if self._normalize_L2:312            faiss.normalize_L2(vector)313        self.index.add(vector)314 315        # Add information to docstore and index.316        self.docstore.add({id_: doc for id_, doc in zip(ids, documents)})317        starting_len = len(self.index_to_docstore_id)318        index_to_id = {starting_len + j: id_ for j, id_ in enumerate(ids)}319        self.index_to_docstore_id.update(index_to_id)320        return ids321 322    def add_texts(323        self,324        texts: Iterable[str],325        metadatas: Optional[List[dict]] = None,326        ids: Optional[List[str]] = None,327        **kwargs: Any,328    ) -> List[str]:329        """Run more texts through the embeddings and add to the vectorstore.330 331        Args:332            texts: Iterable of strings to add to the vectorstore.333            metadatas: Optional list of metadatas associated with the texts.334            ids: Optional list of unique IDs.335 336        Returns:337            List of ids from adding the texts into the vectorstore.338        """339        texts = list(texts)340        embeddings = self._embed_documents(texts)341        return self.__add(texts, embeddings, metadatas=metadatas, ids=ids)342 343    async def aadd_texts(344        self,345        texts: Iterable[str],346        metadatas: Optional[List[dict]] = None,347        ids: Optional[List[str]] = None,348        **kwargs: Any,349    ) -> List[str]:350        """Run more texts through the embeddings and add to the vectorstore351            asynchronously.352 353        Args:354            texts: Iterable of strings to add to the vectorstore.355            metadatas: Optional list of metadatas associated with the texts.356            ids: Optional list of unique IDs.357 358        Returns:359            List of ids from adding the texts into the vectorstore.360        """361        texts = list(texts)362        embeddings = await self._aembed_documents(texts)363        return self.__add(texts, embeddings, metadatas=metadatas, ids=ids)364 365    def add_embeddings(366        self,367        text_embeddings: Iterable[Tuple[str, List[float]]],368        metadatas: Optional[List[dict]] = None,369        ids: Optional[List[str]] = None,370        **kwargs: Any,371    ) -> List[str]:372        """Add the given texts and embeddings to the vectorstore.373 374        Args:375            text_embeddings: Iterable pairs of string and embedding to376                add to the vectorstore.377            metadatas: Optional list of metadatas associated with the texts.378            ids: Optional list of unique IDs.379 380        Returns:381            List of ids from adding the texts into the vectorstore.382        """383        # Embed and create the documents.384        texts, embeddings = zip(*text_embeddings)385        return self.__add(texts, embeddings, metadatas=metadatas, ids=ids)386 387    def similarity_search_with_score_by_vector(388        self,389        embedding: List[float],390        k: int = 4,391        filter: Optional[Union[Callable, Dict[str, Any]]] = None,392        fetch_k: int = 20,393        **kwargs: Any,394    ) -> List[Tuple[Document, float]]:395        """Return docs most similar to query.396 397        Args:398            embedding: Embedding vector to look up documents similar to.399            k: Number of Documents to return. Defaults to 4.400            filter (Optional[Union[Callable, Dict[str, Any]]]): Filter by metadata.401                Defaults to None. If a callable, it must take as input the402                metadata dict of Document and return a bool.403            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.404                      Defaults to 20.405            **kwargs: kwargs to be passed to similarity search. Can include:406                score_threshold: Optional, a floating point value between 0 to 1 to407                    filter the resulting set of retrieved docs408 409        Returns:410            List of documents most similar to the query text and L2 distance411            in float for each. Lower score represents more similarity.412        """413        faiss = dependable_faiss_import()414        vector = np.array([embedding], dtype=np.float32)415        if self._normalize_L2:416            faiss.normalize_L2(vector)417        scores, indices = self.index.search(vector, k if filter is None else fetch_k)418        docs = []419 420        if filter is not None:421            filter_func = self._create_filter_func(filter)422 423        for j, i in enumerate(indices[0]):424            if i == -1:425                # This happens when not enough docs are returned.426                continue427            _id = self.index_to_docstore_id[i]428            doc = self.docstore.search(_id)429            if not isinstance(doc, Document):430                raise ValueError(f"Could not find document for id {_id}, got {doc}")431            if filter is not None:432                if filter_func(doc.metadata):433                    docs.append((doc, scores[0][j]))434            else:435                docs.append((doc, scores[0][j]))436 437        score_threshold = kwargs.get("score_threshold")438        if score_threshold is not None:439            cmp = (440                operator.ge441                if self.distance_strategy442                in (DistanceStrategy.MAX_INNER_PRODUCT, DistanceStrategy.JACCARD)443                else operator.le444            )445            docs = [446                (doc, similarity)447                for doc, similarity in docs448                if cmp(similarity, score_threshold)449            ]450        return docs[:k]451 452    async def asimilarity_search_with_score_by_vector(453        self,454        embedding: List[float],455        k: int = 4,456        filter: Optional[Union[Callable, Dict[str, Any]]] = None,457        fetch_k: int = 20,458        **kwargs: Any,459    ) -> List[Tuple[Document, float]]:460        """Return docs most similar to query asynchronously.461 462        Args:463            embedding: Embedding vector to look up documents similar to.464            k: Number of Documents to return. Defaults to 4.465            filter (Optional[Dict[str, Any]]): Filter by metadata.466                Defaults to None. If a callable, it must take as input the467                metadata dict of Document and return a bool.468 469            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.470                      Defaults to 20.471            **kwargs: kwargs to be passed to similarity search. Can include:472                score_threshold: Optional, a floating point value between 0 to 1 to473                    filter the resulting set of retrieved docs474 475        Returns:476            List of documents most similar to the query text and L2 distance477            in float for each. Lower score represents more similarity.478        """479 480        # This is a temporary workaround to make the similarity search asynchronous.481        return await run_in_executor(482            None,483            self.similarity_search_with_score_by_vector,484            embedding,485            k=k,486            filter=filter,487            fetch_k=fetch_k,488            **kwargs,489        )490 491    def similarity_search_with_score(492        self,493        query: str,494        k: int = 4,495        filter: Optional[Union[Callable, Dict[str, Any]]] = None,496        fetch_k: int = 20,497        **kwargs: Any,498    ) -> List[Tuple[Document, float]]:499        """Return docs most similar to query.500 501        Args:502            query: Text to look up documents similar to.503            k: Number of Documents to return. Defaults to 4.504            filter (Optional[Dict[str, str]]): Filter by metadata.505                Defaults to None. If a callable, it must take as input the506                metadata dict of Document and return a bool.507 508            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.509                      Defaults to 20.510 511        Returns:512            List of documents most similar to the query text with513            L2 distance in float. Lower score represents more similarity.514        """515        embedding = self._embed_query(query)516        docs = self.similarity_search_with_score_by_vector(517            embedding,518            k,519            filter=filter,520            fetch_k=fetch_k,521            **kwargs,522        )523        return docs524 525    async def asimilarity_search_with_score(526        self,527        query: str,528        k: int = 4,529        filter: Optional[Union[Callable, Dict[str, Any]]] = None,530        fetch_k: int = 20,531        **kwargs: Any,532    ) -> List[Tuple[Document, float]]:533        """Return docs most similar to query asynchronously.534 535        Args:536            query: Text to look up documents similar to.537            k: Number of Documents to return. Defaults to 4.538            filter (Optional[Dict[str, str]]): Filter by metadata.539                Defaults to None. If a callable, it must take as input the540                metadata dict of Document and return a bool.541 542            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.543                      Defaults to 20.544 545        Returns:546            List of documents most similar to the query text with547            L2 distance in float. Lower score represents more similarity.548        """549        embedding = await self._aembed_query(query)550        docs = await self.asimilarity_search_with_score_by_vector(551            embedding,552            k,553            filter=filter,554            fetch_k=fetch_k,555            **kwargs,556        )557        return docs558 559    def similarity_search_by_vector(560        self,561        embedding: List[float],562        k: int = 4,563        filter: Optional[Dict[str, Any]] = None,564        fetch_k: int = 20,565        **kwargs: Any,566    ) -> List[Document]:567        """Return docs most similar to embedding vector.568 569        Args:570            embedding: Embedding to look up documents similar to.571            k: Number of Documents to return. Defaults to 4.572            filter (Optional[Dict[str, str]]): Filter by metadata.573                Defaults to None. If a callable, it must take as input the574                metadata dict of Document and return a bool.575 576            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.577                      Defaults to 20.578 579        Returns:580            List of Documents most similar to the embedding.581        """582        docs_and_scores = self.similarity_search_with_score_by_vector(583            embedding,584            k,585            filter=filter,586            fetch_k=fetch_k,587            **kwargs,588        )589        return [doc for doc, _ in docs_and_scores]590 591    async def asimilarity_search_by_vector(592        self,593        embedding: List[float],594        k: int = 4,595        filter: Optional[Union[Callable, Dict[str, Any]]] = None,596        fetch_k: int = 20,597        **kwargs: Any,598    ) -> List[Document]:599        """Return docs most similar to embedding vector asynchronously.600 601        Args:602            embedding: Embedding to look up documents similar to.603            k: Number of Documents to return. Defaults to 4.604            filter (Optional[Dict[str, str]]): Filter by metadata.605                Defaults to None. If a callable, it must take as input the606                metadata dict of Document and return a bool.607 608            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.609                      Defaults to 20.610 611        Returns:612            List of Documents most similar to the embedding.613        """614        docs_and_scores = await self.asimilarity_search_with_score_by_vector(615            embedding,616            k,617            filter=filter,618            fetch_k=fetch_k,619            **kwargs,620        )621        return [doc for doc, _ in docs_and_scores]622 623    def similarity_search(624        self,625        query: str,626        k: int = 4,627        filter: Optional[Union[Callable, Dict[str, Any]]] = None,628        fetch_k: int = 20,629        **kwargs: Any,630    ) -> List[Document]:631        """Return docs most similar to query.632 633        Args:634            query: Text to look up documents similar to.635            k: Number of Documents to return. Defaults to 4.636            filter: (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.637            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.638                      Defaults to 20.639 640        Returns:641            List of Documents most similar to the query.642        """643        docs_and_scores = self.similarity_search_with_score(644            query, k, filter=filter, fetch_k=fetch_k, **kwargs645        )646        return [doc for doc, _ in docs_and_scores]647 648    async def asimilarity_search(649        self,650        query: str,651        k: int = 4,652        filter: Optional[Union[Callable, Dict[str, Any]]] = None,653        fetch_k: int = 20,654        **kwargs: Any,655    ) -> List[Document]:656        """Return docs most similar to query asynchronously.657 658        Args:659            query: Text to look up documents similar to.660            k: Number of Documents to return. Defaults to 4.661            filter: (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.662            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.663                      Defaults to 20.664 665        Returns:666            List of Documents most similar to the query.667        """668        docs_and_scores = await self.asimilarity_search_with_score(669            query, k, filter=filter, fetch_k=fetch_k, **kwargs670        )671        return [doc for doc, _ in docs_and_scores]672 673    def max_marginal_relevance_search_with_score_by_vector(674        self,675        embedding: List[float],676        *,677        k: int = 4,678        fetch_k: int = 20,679        lambda_mult: float = 0.5,680        filter: Optional[Union[Callable, Dict[str, Any]]] = None,681    ) -> List[Tuple[Document, float]]:682        """Return docs and their similarity scores selected using the maximal marginal683            relevance.684 685        Maximal marginal relevance optimizes for similarity to query AND diversity686        among selected documents.687 688        Args:689            embedding: Embedding to look up documents similar to.690            k: Number of Documents to return. Defaults to 4.691            fetch_k: Number of Documents to fetch before filtering to692                     pass to MMR algorithm.693            lambda_mult: Number between 0 and 1 that determines the degree694                        of diversity among the results with 0 corresponding695                        to maximum diversity and 1 to minimum diversity.696                        Defaults to 0.5.697        Returns:698            List of Documents and similarity scores selected by maximal marginal699                relevance and score for each.700        """701        scores, indices = self.index.search(702            np.array([embedding], dtype=np.float32),703            fetch_k if filter is None else fetch_k * 2,704        )705        if filter is not None:706            filter_func = self._create_filter_func(filter)707            filtered_indices = []708            for i in indices[0]:709                if i == -1:710                    # This happens when not enough docs are returned.711                    continue712                _id = self.index_to_docstore_id[i]713                doc = self.docstore.search(_id)714                if not isinstance(doc, Document):715                    raise ValueError(f"Could not find document for id {_id}, got {doc}")716                if filter_func(doc.metadata):717                    filtered_indices.append(i)718            indices = np.array([filtered_indices])719        # -1 happens when not enough docs are returned.720        embeddings = [self.index.reconstruct(int(i)) for i in indices[0] if i != -1]721        mmr_selected = maximal_marginal_relevance(722            np.array([embedding], dtype=np.float32),723            embeddings,724            k=k,725            lambda_mult=lambda_mult,726        )727 728        docs_and_scores = []729        for i in mmr_selected:730            if indices[0][i] == -1:731                # This happens when not enough docs are returned.732                continue733            _id = self.index_to_docstore_id[indices[0][i]]734            doc = self.docstore.search(_id)735            if not isinstance(doc, Document):736                raise ValueError(f"Could not find document for id {_id}, got {doc}")737            docs_and_scores.append((doc, scores[0][i]))738 739        return docs_and_scores740 741    async def amax_marginal_relevance_search_with_score_by_vector(742        self,743        embedding: List[float],744        *,745        k: int = 4,746        fetch_k: int = 20,747        lambda_mult: float = 0.5,748        filter: Optional[Union[Callable, Dict[str, Any]]] = None,749    ) -> List[Tuple[Document, float]]:750        """Return docs and their similarity scores selected using the maximal marginal751            relevance asynchronously.752 753        Maximal marginal relevance optimizes for similarity to query AND diversity754        among selected documents.755 756        Args:757            embedding: Embedding to look up documents similar to.758            k: Number of Documents to return. Defaults to 4.759            fetch_k: Number of Documents to fetch before filtering to760                     pass to MMR algorithm.761            lambda_mult: Number between 0 and 1 that determines the degree762                        of diversity among the results with 0 corresponding763                        to maximum diversity and 1 to minimum diversity.764                        Defaults to 0.5.765        Returns:766            List of Documents and similarity scores selected by maximal marginal767                relevance and score for each.768        """769        # This is a temporary workaround to make the similarity search asynchronous.770        return await run_in_executor(771            None,772            self.max_marginal_relevance_search_with_score_by_vector,773            embedding,774            k=k,775            fetch_k=fetch_k,776            lambda_mult=lambda_mult,777            filter=filter,778        )779 780    def max_marginal_relevance_search_by_vector(781        self,782        embedding: List[float],783        k: int = 4,784        fetch_k: int = 20,785        lambda_mult: float = 0.5,786        filter: Optional[Union[Callable, Dict[str, Any]]] = None,787        **kwargs: Any,788    ) -> List[Document]:789        """Return docs selected using the maximal marginal relevance.790 791        Maximal marginal relevance optimizes for similarity to query AND diversity792        among selected documents.793 794        Args:795            embedding: Embedding to look up documents similar to.796            k: Number of Documents to return. Defaults to 4.797            fetch_k: Number of Documents to fetch before filtering to798                     pass to MMR algorithm.799            lambda_mult: Number between 0 and 1 that determines the degree800                        of diversity among the results with 0 corresponding801                        to maximum diversity and 1 to minimum diversity.802                        Defaults to 0.5.803        Returns:804            List of Documents selected by maximal marginal relevance.805        """806        docs_and_scores = self.max_marginal_relevance_search_with_score_by_vector(807            embedding, k=k, fetch_k=fetch_k, lambda_mult=lambda_mult, filter=filter808        )809        return [doc for doc, _ in docs_and_scores]810 811    async def amax_marginal_relevance_search_by_vector(812        self,813        embedding: List[float],814        k: int = 4,815        fetch_k: int = 20,816        lambda_mult: float = 0.5,817        filter: Optional[Union[Callable, Dict[str, Any]]] = None,818        **kwargs: Any,819    ) -> List[Document]:820        """Return docs selected using the maximal marginal relevance asynchronously.821 822        Maximal marginal relevance optimizes for similarity to query AND diversity823        among selected documents.824 825        Args:826            embedding: Embedding to look up documents similar to.827            k: Number of Documents to return. Defaults to 4.828            fetch_k: Number of Documents to fetch before filtering to829                     pass to MMR algorithm.830            lambda_mult: Number between 0 and 1 that determines the degree831                        of diversity among the results with 0 corresponding832                        to maximum diversity and 1 to minimum diversity.833                        Defaults to 0.5.834        Returns:835            List of Documents selected by maximal marginal relevance.836        """837        docs_and_scores = (838            await self.amax_marginal_relevance_search_with_score_by_vector(839                embedding, k=k, fetch_k=fetch_k, lambda_mult=lambda_mult, filter=filter840            )841        )842        return [doc for doc, _ in docs_and_scores]843 844    def max_marginal_relevance_search(845        self,846        query: str,847        k: int = 4,848        fetch_k: int = 20,849        lambda_mult: float = 0.5,850        filter: Optional[Union[Callable, Dict[str, Any]]] = None,851        **kwargs: Any,852    ) -> List[Document]:853        """Return docs selected using the maximal marginal relevance.854 855        Maximal marginal relevance optimizes for similarity to query AND diversity856        among selected documents.857 858        Args:859            query: Text to look up documents similar to.860            k: Number of Documents to return. Defaults to 4.861            fetch_k: Number of Documents to fetch before filtering (if needed) to862                     pass to MMR algorithm.863            lambda_mult: Number between 0 and 1 that determines the degree864                        of diversity among the results with 0 corresponding865                        to maximum diversity and 1 to minimum diversity.866                        Defaults to 0.5.867        Returns:868            List of Documents selected by maximal marginal relevance.869        """870        embedding = self._embed_query(query)871        docs = self.max_marginal_relevance_search_by_vector(872            embedding,873            k=k,874            fetch_k=fetch_k,875            lambda_mult=lambda_mult,876            filter=filter,877            **kwargs,878        )879        return docs880 881    async def amax_marginal_relevance_search(882        self,883        query: str,884        k: int = 4,885        fetch_k: int = 20,886        lambda_mult: float = 0.5,887        filter: Optional[Union[Callable, Dict[str, Any]]] = None,888        **kwargs: Any,889    ) -> List[Document]:890        """Return docs selected using the maximal marginal relevance asynchronously.891 892        Maximal marginal relevance optimizes for similarity to query AND diversity893        among selected documents.894 895        Args:896            query: Text to look up documents similar to.897            k: Number of Documents to return. Defaults to 4.898            fetch_k: Number of Documents to fetch before filtering (if needed) to899                     pass to MMR algorithm.900            lambda_mult: Number between 0 and 1 that determines the degree901                        of diversity among the results with 0 corresponding902                        to maximum diversity and 1 to minimum diversity.903                        Defaults to 0.5.904        Returns:905            List of Documents selected by maximal marginal relevance.906        """907        embedding = await self._aembed_query(query)908        docs = await self.amax_marginal_relevance_search_by_vector(909            embedding,910            k=k,911            fetch_k=fetch_k,912            lambda_mult=lambda_mult,913            filter=filter,914            **kwargs,915        )916        return docs917 918    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> Optional[bool]:919        """Delete by ID. These are the IDs in the vectorstore.920 921        Args:922            ids: List of ids to delete.923 924        Returns:925            Optional[bool]: True if deletion is successful,926            False otherwise, None if not implemented.927        """928        if ids is None:929            raise ValueError("No ids provided to delete.")930        missing_ids = set(ids).difference(self.index_to_docstore_id.values())931        if missing_ids:932            raise ValueError(933                f"Some specified ids do not exist in the current store. Ids not found: "934                f"{missing_ids}"935            )936 937        reversed_index = {id_: idx for idx, id_ in self.index_to_docstore_id.items()}938        index_to_delete = {reversed_index[id_] for id_ in ids}939 940        self.index.remove_ids(np.fromiter(index_to_delete, dtype=np.int64))941        self.docstore.delete(ids)942 943        remaining_ids = [944            id_945            for i, id_ in sorted(self.index_to_docstore_id.items())946            if i not in index_to_delete947        ]948        self.index_to_docstore_id = {i: id_ for i, id_ in enumerate(remaining_ids)}949 950        return True951 952    def merge_from(self, target: FAISS) -> None:953        """Merge another FAISS object with the current one.954 955        Add the target FAISS to the current one.956 957        Args:958            target: FAISS object you wish to merge into the current one959 960        Returns:961            None.962        """963        if not isinstance(self.docstore, AddableMixin):964            raise ValueError("Cannot merge with this type of docstore")965        # Numerical index for target docs are incremental on existing ones966        starting_len = len(self.index_to_docstore_id)967 968        # Merge two IndexFlatL2969        self.index.merge_from(target.index)970 971        # Get id and docs from target FAISS object972        full_info = []973        for i, target_id in target.index_to_docstore_id.items():974            doc = target.docstore.search(target_id)975            if not isinstance(doc, Document):976                raise ValueError("Document should be returned")977            full_info.append((starting_len + i, target_id, doc))978 979        # Add information to docstore and index_to_docstore_id.980        self.docstore.add({_id: doc for _, _id, doc in full_info})981        index_to_id = {index: _id for index, _id, _ in full_info}982        self.index_to_docstore_id.update(index_to_id)983 984    @classmethod985    def __from(986        cls,987        texts: Iterable[str],988        embeddings: List[List[float]],989        embedding: Embeddings,990        metadatas: Optional[Iterable[dict]] = None,991        ids: Optional[List[str]] = None,992        normalize_L2: bool = False,993        distance_strategy: DistanceStrategy = DistanceStrategy.EUCLIDEAN_DISTANCE,994        **kwargs: Any,995    ) -> FAISS:996        faiss = dependable_faiss_import()997        if distance_strategy == DistanceStrategy.MAX_INNER_PRODUCT:998            index = faiss.IndexFlatIP(len(embeddings[0]))999        else:1000            # Default to L2, currently other metric types not initialized.1001            index = faiss.IndexFlatL2(len(embeddings[0]))1002        docstore = kwargs.pop("docstore", InMemoryDocstore())1003        index_to_docstore_id = kwargs.pop("index_to_docstore_id", {})1004        vecstore = cls(1005            embedding,1006            index,1007            docstore,1008            index_to_docstore_id,1009            normalize_L2=normalize_L2,1010            distance_strategy=distance_strategy,1011            **kwargs,1012        )1013        vecstore.__add(texts, embeddings, metadatas=metadatas, ids=ids)1014        return vecstore1015 1016    @classmethod1017    def from_texts(1018        cls,1019        texts: List[str],1020        embedding: Embeddings,1021        metadatas: Optional[List[dict]] = None,1022        ids: Optional[List[str]] = None,1023        **kwargs: Any,1024    ) -> FAISS:1025        """Construct FAISS wrapper from raw documents.1026 1027        This is a user friendly interface that:1028            1. Embeds documents.1029            2. Creates an in memory docstore1030            3. Initializes the FAISS database1031 1032        This is intended to be a quick way to get started.1033 1034        Example:1035            .. code-block:: python1036 1037                from langchain_community.vectorstores import FAISS1038                from langchain_community.embeddings import OpenAIEmbeddings1039 1040                embeddings = OpenAIEmbeddings()1041                faiss = FAISS.from_texts(texts, embeddings)1042        """1043        embeddings = embedding.embed_documents(texts)1044        return cls.__from(1045            texts,1046            embeddings,1047            embedding,1048            metadatas=metadatas,1049            ids=ids,1050            **kwargs,1051        )1052 1053    @classmethod1054    async def afrom_texts(1055        cls,1056        texts: list[str],1057        embedding: Embeddings,1058        metadatas: Optional[List[dict]] = None,1059        ids: Optional[List[str]] = None,1060        **kwargs: Any,1061    ) -> FAISS:1062        """Construct FAISS wrapper from raw documents asynchronously.1063 1064        This is a user friendly interface that:1065            1. Embeds documents.1066            2. Creates an in memory docstore1067            3. Initializes the FAISS database1068 1069        This is intended to be a quick way to get started.1070 1071        Example:1072            .. code-block:: python1073 1074                from langchain_community.vectorstores import FAISS1075                from langchain_community.embeddings import OpenAIEmbeddings1076 1077                embeddings = OpenAIEmbeddings()1078                faiss = await FAISS.afrom_texts(texts, embeddings)1079        """1080        embeddings = await embedding.aembed_documents(texts)1081        return cls.__from(1082            texts,1083            embeddings,1084            embedding,1085            metadatas=metadatas,1086            ids=ids,1087            **kwargs,1088        )1089 1090    @classmethod1091    def from_embeddings(1092        cls,1093        text_embeddings: Iterable[Tuple[str, List[float]]],1094        embedding: Embeddings,1095        metadatas: Optional[Iterable[dict]] = None,1096        ids: Optional[List[str]] = None,1097        **kwargs: Any,1098    ) -> FAISS:1099        """Construct FAISS wrapper from raw documents.1100 1101        This is a user friendly interface that:1102            1. Embeds documents.1103            2. Creates an in memory docstore1104            3. Initializes the FAISS database1105 1106        This is intended to be a quick way to get started.1107 1108        Example:1109            .. code-block:: python1110 1111                from langchain_community.vectorstores import FAISS1112                from langchain_community.embeddings import OpenAIEmbeddings1113 1114                embeddings = OpenAIEmbeddings()1115                text_embeddings = embeddings.embed_documents(texts)1116                text_embedding_pairs = zip(texts, text_embeddings)1117                faiss = FAISS.from_embeddings(text_embedding_pairs, embeddings)1118        """1119        texts, embeddings = zip(*text_embeddings)1120        return cls.__from(1121            list(texts),1122            list(embeddings),1123            embedding,1124            metadatas=metadatas,1125            ids=ids,1126            **kwargs,1127        )1128 1129    @classmethod1130    async def afrom_embeddings(1131        cls,1132        text_embeddings: Iterable[Tuple[str, List[float]]],1133        embedding: Embeddings,1134        metadatas: Optional[Iterable[dict]] = None,1135        ids: Optional[List[str]] = None,1136        **kwargs: Any,1137    ) -> FAISS:1138        """Construct FAISS wrapper from raw documents asynchronously."""1139        return cls.from_embeddings(1140            text_embeddings,1141            embedding,1142            metadatas=metadatas,1143            ids=ids,1144            **kwargs,1145        )1146 1147    def save_local(self, folder_path: str, index_name: str = "index") -> None:1148        """Save FAISS index, docstore, and index_to_docstore_id to disk.1149 1150        Args:1151            folder_path: folder path to save index, docstore,1152                and index_to_docstore_id to.1153            index_name: for saving with a specific index file name1154        """1155        path = Path(folder_path)1156        path.mkdir(exist_ok=True, parents=True)1157 1158        # save index separately since it is not picklable1159        faiss = dependable_faiss_import()1160        faiss.write_index(self.index, str(path / f"{index_name}.faiss"))1161 1162        # save docstore and index_to_docstore_id1163        with open(path / f"{index_name}.pkl", "wb") as f:1164            pickle.dump((self.docstore, self.index_to_docstore_id), f)1165 1166    @classmethod1167    def load_local(1168        cls,1169        folder_path: str,1170        embeddings: Embeddings,1171        index_name: str = "index",1172        *,1173        allow_dangerous_deserialization: bool = False,1174        **kwargs: Any,1175    ) -> FAISS:1176        """Load FAISS index, docstore, and index_to_docstore_id from disk.1177 1178        Args:1179            folder_path: folder path to load index, docstore,1180                and index_to_docstore_id from.1181            embeddings: Embeddings to use when generating queries1182            index_name: for saving with a specific index file name1183            allow_dangerous_deserialization: whether to allow deserialization1184                of the data which involves loading a pickle file.1185                Pickle files can be modified by malicious actors to deliver a1186                malicious payload that results in execution of1187                arbitrary code on your machine.1188        """1189        if not allow_dangerous_deserialization:1190            raise ValueError(1191                "The de-serialization relies loading a pickle file. "1192                "Pickle files can be modified to deliver a malicious payload that "1193                "results in execution of arbitrary code on your machine."1194                "You will need to set `allow_dangerous_deserialization` to `True` to "1195                "enable deserialization. If you do this, make sure that you "1196                "trust the source of the data. For example, if you are loading a "1197                "file that you created, and know that no one else has modified the "1198                "file, then this is safe to do. Do not set this to `True` if you are "1199                "loading a file from an untrusted source (e.g., some random site on "1200                "the internet.)."

Showing the first 1,200 of 1484 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai