codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import logging4import operator5import os6import pickle7import uuid8import warnings9from pathlib import Path10from typing import (11 Any,12 Callable,13 Dict,14 Iterable,15 List,16 Optional,17 Sequence,18 Sized,19 Tuple,20 Union,21)22 23import numpy as np24from langchain_core.documents import Document25from langchain_core.embeddings import Embeddings26from langchain_core.runnables.config import run_in_executor27from langchain_core.vectorstores import VectorStore28 29from langchain_community.docstore.base import AddableMixin, Docstore30from langchain_community.docstore.in_memory import InMemoryDocstore31from langchain_community.vectorstores.utils import (32 DistanceStrategy,33 maximal_marginal_relevance,34)35 36logger = logging.getLogger(__name__)37 38 39def dependable_faiss_import(no_avx2: Optional[bool] = None) -> Any:40 """41 Import faiss if available, otherwise raise error.42 If FAISS_NO_AVX2 environment variable is set, it will be considered43 to load FAISS with no AVX2 optimization.44 45 Args:46 no_avx2: Load FAISS strictly with no AVX2 optimization47 so that the vectorstore is portable and compatible with other devices.48 """49 if no_avx2 is None and "FAISS_NO_AVX2" in os.environ:50 no_avx2 = bool(os.getenv("FAISS_NO_AVX2"))51 52 try:53 if no_avx2:54 from faiss import swigfaiss as faiss55 else:56 import faiss57 except ImportError:58 raise ImportError(59 "Could not import faiss python package. "60 "Please install it with `pip install faiss-gpu` (for CUDA supported GPU) "61 "or `pip install faiss-cpu` (depending on Python version)."62 )63 return faiss64 65 66def _len_check_if_sized(x: Any, y: Any, x_name: str, y_name: str) -> None:67 if isinstance(x, Sized) and isinstance(y, Sized) and len(x) != len(y):68 raise ValueError(69 f"{x_name} and {y_name} expected to be equal length but "70 f"len({x_name})={len(x)} and len({y_name})={len(y)}"71 )72 return73 74 75class FAISS(VectorStore):76 """FAISS vector store integration.77 78 See [The FAISS Library](https://arxiv.org/pdf/2401.08281) paper.79 80 Setup:81 Install ``langchain_community`` and ``faiss-cpu`` python packages.82 83 .. code-block:: bash84 85 pip install -qU langchain_community faiss-cpu86 87 Key init args — indexing params:88 embedding_function: Embeddings89 Embedding function to use.90 91 Key init args — client params:92 index: Any93 FAISS index to use.94 docstore: Docstore95 Docstore to use.96 index_to_docstore_id: Dict[int, str]97 Mapping of index to docstore id.98 99 Instantiate:100 .. code-block:: python101 102 import faiss103 from langchain_community.vectorstores import FAISS104 from langchain_community.docstore.in_memory import InMemoryDocstore105 from langchain_openai import OpenAIEmbeddings106 107 index = faiss.IndexFlatL2(len(OpenAIEmbeddings().embed_query("hello world")))108 109 vector_store = FAISS(110 embedding_function=OpenAIEmbeddings(),111 index=index,112 docstore= InMemoryDocstore(),113 index_to_docstore_id={}114 )115 116 Add Documents:117 .. code-block:: python118 119 from langchain_core.documents import Document120 121 document_1 = Document(page_content="foo", metadata={"baz": "bar"})122 document_2 = Document(page_content="thud", metadata={"bar": "baz"})123 document_3 = Document(page_content="i will be deleted :(")124 125 documents = [document_1, document_2, document_3]126 ids = ["1", "2", "3"]127 vector_store.add_documents(documents=documents, ids=ids)128 129 Delete Documents:130 .. code-block:: python131 132 vector_store.delete(ids=["3"])133 134 Search:135 .. code-block:: python136 137 results = vector_store.similarity_search(query="thud",k=1)138 for doc in results:139 print(f"* {doc.page_content} [{doc.metadata}]")140 141 .. code-block:: python142 143 * thud [{'bar': 'baz'}]144 145 Search with filter:146 .. code-block:: python147 148 results = vector_store.similarity_search(query="thud",k=1,filter={"bar": "baz"})149 for doc in results:150 print(f"* {doc.page_content} [{doc.metadata}]")151 152 .. code-block:: python153 154 * thud [{'bar': 'baz'}]155 156 Search with score:157 .. code-block:: python158 159 results = vector_store.similarity_search_with_score(query="qux",k=1)160 for doc, score in results:161 print(f"* [SIM={score:3f}] {doc.page_content} [{doc.metadata}]")162 163 .. code-block:: python164 165 * [SIM=0.335304] foo [{'baz': 'bar'}]166 167 Async:168 .. code-block:: python169 170 # add documents171 # await vector_store.aadd_documents(documents=documents, ids=ids)172 173 # delete documents174 # await vector_store.adelete(ids=["3"])175 176 # search177 # results = vector_store.asimilarity_search(query="thud",k=1)178 179 # search with score180 results = await vector_store.asimilarity_search_with_score(query="qux",k=1)181 for doc,score in results:182 print(f"* [SIM={score:3f}] {doc.page_content} [{doc.metadata}]")183 184 .. code-block:: python185 186 * [SIM=0.335304] foo [{'baz': 'bar'}]187 188 Use as Retriever:189 .. code-block:: python190 191 retriever = vector_store.as_retriever(192 search_type="mmr",193 search_kwargs={"k": 1, "fetch_k": 2, "lambda_mult": 0.5},194 )195 retriever.invoke("thud")196 197 .. code-block:: python198 199 [Document(metadata={'bar': 'baz'}, page_content='thud')]200 201 """ # noqa: E501202 203 def __init__(204 self,205 embedding_function: Union[206 Callable[[str], List[float]],207 Embeddings,208 ],209 index: Any,210 docstore: Docstore,211 index_to_docstore_id: Dict[int, str],212 relevance_score_fn: Optional[Callable[[float], float]] = None,213 normalize_L2: bool = False,214 distance_strategy: DistanceStrategy = DistanceStrategy.EUCLIDEAN_DISTANCE,215 ):216 """Initialize with necessary components."""217 if not isinstance(embedding_function, Embeddings):218 logger.warning(219 "`embedding_function` is expected to be an Embeddings object, support "220 "for passing in a function will soon be removed."221 )222 self.embedding_function = embedding_function223 self.index = index224 self.docstore = docstore225 self.index_to_docstore_id = index_to_docstore_id226 self.distance_strategy = distance_strategy227 self.override_relevance_score_fn = relevance_score_fn228 self._normalize_L2 = normalize_L2229 if (230 self.distance_strategy != DistanceStrategy.EUCLIDEAN_DISTANCE231 and self._normalize_L2232 ):233 warnings.warn(234 "Normalizing L2 is not applicable for "235 f"metric type: {self.distance_strategy}"236 )237 238 @property239 def embeddings(self) -> Optional[Embeddings]:240 return (241 self.embedding_function242 if isinstance(self.embedding_function, Embeddings)243 else None244 )245 246 def _embed_documents(self, texts: List[str]) -> List[List[float]]:247 if isinstance(self.embedding_function, Embeddings):248 return self.embedding_function.embed_documents(texts)249 else:250 return [self.embedding_function(text) for text in texts]251 252 async def _aembed_documents(self, texts: List[str]) -> List[List[float]]:253 if isinstance(self.embedding_function, Embeddings):254 return await self.embedding_function.aembed_documents(texts)255 else:256 # return await asyncio.gather(257 # [self.embedding_function(text) for text in texts]258 # )259 raise Exception(260 "`embedding_function` is expected to be an Embeddings object, support "261 "for passing in a function will soon be removed."262 )263 264 def _embed_query(self, text: str) -> List[float]:265 if isinstance(self.embedding_function, Embeddings):266 return self.embedding_function.embed_query(text)267 else:268 return self.embedding_function(text)269 270 async def _aembed_query(self, text: str) -> List[float]:271 if isinstance(self.embedding_function, Embeddings):272 return await self.embedding_function.aembed_query(text)273 else:274 # return await self.embedding_function(text)275 raise Exception(276 "`embedding_function` is expected to be an Embeddings object, support "277 "for passing in a function will soon be removed."278 )279 280 def __add(281 self,282 texts: Iterable[str],283 embeddings: Iterable[List[float]],284 metadatas: Optional[Iterable[dict]] = None,285 ids: Optional[List[str]] = None,286 ) -> List[str]:287 faiss = dependable_faiss_import()288 if not isinstance(self.docstore, AddableMixin):289 raise ValueError(290 "If trying to add texts, the underlying docstore should support "291 f"adding items, which {self.docstore} does not"292 )293 294 _len_check_if_sized(texts, metadatas, "texts", "metadatas")295 296 ids = ids or [str(uuid.uuid4()) for _ in texts]297 _len_check_if_sized(texts, ids, "texts", "ids")298 299 _metadatas = metadatas or ({} for _ in texts)300 documents = [301 Document(id=id_, page_content=t, metadata=m)302 for id_, t, m in zip(ids, texts, _metadatas)303 ]304 305 _len_check_if_sized(documents, embeddings, "documents", "embeddings")306 307 if ids and len(ids) != len(set(ids)):308 raise ValueError("Duplicate ids found in the ids list.")309 # Add to the index.310 vector = np.array(embeddings, dtype=np.float32)311 if self._normalize_L2:312 faiss.normalize_L2(vector)313 self.index.add(vector)314 315 # Add information to docstore and index.316 self.docstore.add({id_: doc for id_, doc in zip(ids, documents)})317 starting_len = len(self.index_to_docstore_id)318 index_to_id = {starting_len + j: id_ for j, id_ in enumerate(ids)}319 self.index_to_docstore_id.update(index_to_id)320 return ids321 322 def add_texts(323 self,324 texts: Iterable[str],325 metadatas: Optional[List[dict]] = None,326 ids: Optional[List[str]] = None,327 **kwargs: Any,328 ) -> List[str]:329 """Run more texts through the embeddings and add to the vectorstore.330 331 Args:332 texts: Iterable of strings to add to the vectorstore.333 metadatas: Optional list of metadatas associated with the texts.334 ids: Optional list of unique IDs.335 336 Returns:337 List of ids from adding the texts into the vectorstore.338 """339 texts = list(texts)340 embeddings = self._embed_documents(texts)341 return self.__add(texts, embeddings, metadatas=metadatas, ids=ids)342 343 async def aadd_texts(344 self,345 texts: Iterable[str],346 metadatas: Optional[List[dict]] = None,347 ids: Optional[List[str]] = None,348 **kwargs: Any,349 ) -> List[str]:350 """Run more texts through the embeddings and add to the vectorstore351 asynchronously.352 353 Args:354 texts: Iterable of strings to add to the vectorstore.355 metadatas: Optional list of metadatas associated with the texts.356 ids: Optional list of unique IDs.357 358 Returns:359 List of ids from adding the texts into the vectorstore.360 """361 texts = list(texts)362 embeddings = await self._aembed_documents(texts)363 return self.__add(texts, embeddings, metadatas=metadatas, ids=ids)364 365 def add_embeddings(366 self,367 text_embeddings: Iterable[Tuple[str, List[float]]],368 metadatas: Optional[List[dict]] = None,369 ids: Optional[List[str]] = None,370 **kwargs: Any,371 ) -> List[str]:372 """Add the given texts and embeddings to the vectorstore.373 374 Args:375 text_embeddings: Iterable pairs of string and embedding to376 add to the vectorstore.377 metadatas: Optional list of metadatas associated with the texts.378 ids: Optional list of unique IDs.379 380 Returns:381 List of ids from adding the texts into the vectorstore.382 """383 # Embed and create the documents.384 texts, embeddings = zip(*text_embeddings)385 return self.__add(texts, embeddings, metadatas=metadatas, ids=ids)386 387 def similarity_search_with_score_by_vector(388 self,389 embedding: List[float],390 k: int = 4,391 filter: Optional[Union[Callable, Dict[str, Any]]] = None,392 fetch_k: int = 20,393 **kwargs: Any,394 ) -> List[Tuple[Document, float]]:395 """Return docs most similar to query.396 397 Args:398 embedding: Embedding vector to look up documents similar to.399 k: Number of Documents to return. Defaults to 4.400 filter (Optional[Union[Callable, Dict[str, Any]]]): Filter by metadata.401 Defaults to None. If a callable, it must take as input the402 metadata dict of Document and return a bool.403 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.404 Defaults to 20.405 **kwargs: kwargs to be passed to similarity search. Can include:406 score_threshold: Optional, a floating point value between 0 to 1 to407 filter the resulting set of retrieved docs408 409 Returns:410 List of documents most similar to the query text and L2 distance411 in float for each. Lower score represents more similarity.412 """413 faiss = dependable_faiss_import()414 vector = np.array([embedding], dtype=np.float32)415 if self._normalize_L2:416 faiss.normalize_L2(vector)417 scores, indices = self.index.search(vector, k if filter is None else fetch_k)418 docs = []419 420 if filter is not None:421 filter_func = self._create_filter_func(filter)422 423 for j, i in enumerate(indices[0]):424 if i == -1:425 # This happens when not enough docs are returned.426 continue427 _id = self.index_to_docstore_id[i]428 doc = self.docstore.search(_id)429 if not isinstance(doc, Document):430 raise ValueError(f"Could not find document for id {_id}, got {doc}")431 if filter is not None:432 if filter_func(doc.metadata):433 docs.append((doc, scores[0][j]))434 else:435 docs.append((doc, scores[0][j]))436 437 score_threshold = kwargs.get("score_threshold")438 if score_threshold is not None:439 cmp = (440 operator.ge441 if self.distance_strategy442 in (DistanceStrategy.MAX_INNER_PRODUCT, DistanceStrategy.JACCARD)443 else operator.le444 )445 docs = [446 (doc, similarity)447 for doc, similarity in docs448 if cmp(similarity, score_threshold)449 ]450 return docs[:k]451 452 async def asimilarity_search_with_score_by_vector(453 self,454 embedding: List[float],455 k: int = 4,456 filter: Optional[Union[Callable, Dict[str, Any]]] = None,457 fetch_k: int = 20,458 **kwargs: Any,459 ) -> List[Tuple[Document, float]]:460 """Return docs most similar to query asynchronously.461 462 Args:463 embedding: Embedding vector to look up documents similar to.464 k: Number of Documents to return. Defaults to 4.465 filter (Optional[Dict[str, Any]]): Filter by metadata.466 Defaults to None. If a callable, it must take as input the467 metadata dict of Document and return a bool.468 469 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.470 Defaults to 20.471 **kwargs: kwargs to be passed to similarity search. Can include:472 score_threshold: Optional, a floating point value between 0 to 1 to473 filter the resulting set of retrieved docs474 475 Returns:476 List of documents most similar to the query text and L2 distance477 in float for each. Lower score represents more similarity.478 """479 480 # This is a temporary workaround to make the similarity search asynchronous.481 return await run_in_executor(482 None,483 self.similarity_search_with_score_by_vector,484 embedding,485 k=k,486 filter=filter,487 fetch_k=fetch_k,488 **kwargs,489 )490 491 def similarity_search_with_score(492 self,493 query: str,494 k: int = 4,495 filter: Optional[Union[Callable, Dict[str, Any]]] = None,496 fetch_k: int = 20,497 **kwargs: Any,498 ) -> List[Tuple[Document, float]]:499 """Return docs most similar to query.500 501 Args:502 query: Text to look up documents similar to.503 k: Number of Documents to return. Defaults to 4.504 filter (Optional[Dict[str, str]]): Filter by metadata.505 Defaults to None. If a callable, it must take as input the506 metadata dict of Document and return a bool.507 508 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.509 Defaults to 20.510 511 Returns:512 List of documents most similar to the query text with513 L2 distance in float. Lower score represents more similarity.514 """515 embedding = self._embed_query(query)516 docs = self.similarity_search_with_score_by_vector(517 embedding,518 k,519 filter=filter,520 fetch_k=fetch_k,521 **kwargs,522 )523 return docs524 525 async def asimilarity_search_with_score(526 self,527 query: str,528 k: int = 4,529 filter: Optional[Union[Callable, Dict[str, Any]]] = None,530 fetch_k: int = 20,531 **kwargs: Any,532 ) -> List[Tuple[Document, float]]:533 """Return docs most similar to query asynchronously.534 535 Args:536 query: Text to look up documents similar to.537 k: Number of Documents to return. Defaults to 4.538 filter (Optional[Dict[str, str]]): Filter by metadata.539 Defaults to None. If a callable, it must take as input the540 metadata dict of Document and return a bool.541 542 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.543 Defaults to 20.544 545 Returns:546 List of documents most similar to the query text with547 L2 distance in float. Lower score represents more similarity.548 """549 embedding = await self._aembed_query(query)550 docs = await self.asimilarity_search_with_score_by_vector(551 embedding,552 k,553 filter=filter,554 fetch_k=fetch_k,555 **kwargs,556 )557 return docs558 559 def similarity_search_by_vector(560 self,561 embedding: List[float],562 k: int = 4,563 filter: Optional[Dict[str, Any]] = None,564 fetch_k: int = 20,565 **kwargs: Any,566 ) -> List[Document]:567 """Return docs most similar to embedding vector.568 569 Args:570 embedding: Embedding to look up documents similar to.571 k: Number of Documents to return. Defaults to 4.572 filter (Optional[Dict[str, str]]): Filter by metadata.573 Defaults to None. If a callable, it must take as input the574 metadata dict of Document and return a bool.575 576 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.577 Defaults to 20.578 579 Returns:580 List of Documents most similar to the embedding.581 """582 docs_and_scores = self.similarity_search_with_score_by_vector(583 embedding,584 k,585 filter=filter,586 fetch_k=fetch_k,587 **kwargs,588 )589 return [doc for doc, _ in docs_and_scores]590 591 async def asimilarity_search_by_vector(592 self,593 embedding: List[float],594 k: int = 4,595 filter: Optional[Union[Callable, Dict[str, Any]]] = None,596 fetch_k: int = 20,597 **kwargs: Any,598 ) -> List[Document]:599 """Return docs most similar to embedding vector asynchronously.600 601 Args:602 embedding: Embedding to look up documents similar to.603 k: Number of Documents to return. Defaults to 4.604 filter (Optional[Dict[str, str]]): Filter by metadata.605 Defaults to None. If a callable, it must take as input the606 metadata dict of Document and return a bool.607 608 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.609 Defaults to 20.610 611 Returns:612 List of Documents most similar to the embedding.613 """614 docs_and_scores = await self.asimilarity_search_with_score_by_vector(615 embedding,616 k,617 filter=filter,618 fetch_k=fetch_k,619 **kwargs,620 )621 return [doc for doc, _ in docs_and_scores]622 623 def similarity_search(624 self,625 query: str,626 k: int = 4,627 filter: Optional[Union[Callable, Dict[str, Any]]] = None,628 fetch_k: int = 20,629 **kwargs: Any,630 ) -> List[Document]:631 """Return docs most similar to query.632 633 Args:634 query: Text to look up documents similar to.635 k: Number of Documents to return. Defaults to 4.636 filter: (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.637 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.638 Defaults to 20.639 640 Returns:641 List of Documents most similar to the query.642 """643 docs_and_scores = self.similarity_search_with_score(644 query, k, filter=filter, fetch_k=fetch_k, **kwargs645 )646 return [doc for doc, _ in docs_and_scores]647 648 async def asimilarity_search(649 self,650 query: str,651 k: int = 4,652 filter: Optional[Union[Callable, Dict[str, Any]]] = None,653 fetch_k: int = 20,654 **kwargs: Any,655 ) -> List[Document]:656 """Return docs most similar to query asynchronously.657 658 Args:659 query: Text to look up documents similar to.660 k: Number of Documents to return. Defaults to 4.661 filter: (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.662 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.663 Defaults to 20.664 665 Returns:666 List of Documents most similar to the query.667 """668 docs_and_scores = await self.asimilarity_search_with_score(669 query, k, filter=filter, fetch_k=fetch_k, **kwargs670 )671 return [doc for doc, _ in docs_and_scores]672 673 def max_marginal_relevance_search_with_score_by_vector(674 self,675 embedding: List[float],676 *,677 k: int = 4,678 fetch_k: int = 20,679 lambda_mult: float = 0.5,680 filter: Optional[Union[Callable, Dict[str, Any]]] = None,681 ) -> List[Tuple[Document, float]]:682 """Return docs and their similarity scores selected using the maximal marginal683 relevance.684 685 Maximal marginal relevance optimizes for similarity to query AND diversity686 among selected documents.687 688 Args:689 embedding: Embedding to look up documents similar to.690 k: Number of Documents to return. Defaults to 4.691 fetch_k: Number of Documents to fetch before filtering to692 pass to MMR algorithm.693 lambda_mult: Number between 0 and 1 that determines the degree694 of diversity among the results with 0 corresponding695 to maximum diversity and 1 to minimum diversity.696 Defaults to 0.5.697 Returns:698 List of Documents and similarity scores selected by maximal marginal699 relevance and score for each.700 """701 scores, indices = self.index.search(702 np.array([embedding], dtype=np.float32),703 fetch_k if filter is None else fetch_k * 2,704 )705 if filter is not None:706 filter_func = self._create_filter_func(filter)707 filtered_indices = []708 for i in indices[0]:709 if i == -1:710 # This happens when not enough docs are returned.711 continue712 _id = self.index_to_docstore_id[i]713 doc = self.docstore.search(_id)714 if not isinstance(doc, Document):715 raise ValueError(f"Could not find document for id {_id}, got {doc}")716 if filter_func(doc.metadata):717 filtered_indices.append(i)718 indices = np.array([filtered_indices])719 # -1 happens when not enough docs are returned.720 embeddings = [self.index.reconstruct(int(i)) for i in indices[0] if i != -1]721 mmr_selected = maximal_marginal_relevance(722 np.array([embedding], dtype=np.float32),723 embeddings,724 k=k,725 lambda_mult=lambda_mult,726 )727 728 docs_and_scores = []729 for i in mmr_selected:730 if indices[0][i] == -1:731 # This happens when not enough docs are returned.732 continue733 _id = self.index_to_docstore_id[indices[0][i]]734 doc = self.docstore.search(_id)735 if not isinstance(doc, Document):736 raise ValueError(f"Could not find document for id {_id}, got {doc}")737 docs_and_scores.append((doc, scores[0][i]))738 739 return docs_and_scores740 741 async def amax_marginal_relevance_search_with_score_by_vector(742 self,743 embedding: List[float],744 *,745 k: int = 4,746 fetch_k: int = 20,747 lambda_mult: float = 0.5,748 filter: Optional[Union[Callable, Dict[str, Any]]] = None,749 ) -> List[Tuple[Document, float]]:750 """Return docs and their similarity scores selected using the maximal marginal751 relevance asynchronously.752 753 Maximal marginal relevance optimizes for similarity to query AND diversity754 among selected documents.755 756 Args:757 embedding: Embedding to look up documents similar to.758 k: Number of Documents to return. Defaults to 4.759 fetch_k: Number of Documents to fetch before filtering to760 pass to MMR algorithm.761 lambda_mult: Number between 0 and 1 that determines the degree762 of diversity among the results with 0 corresponding763 to maximum diversity and 1 to minimum diversity.764 Defaults to 0.5.765 Returns:766 List of Documents and similarity scores selected by maximal marginal767 relevance and score for each.768 """769 # This is a temporary workaround to make the similarity search asynchronous.770 return await run_in_executor(771 None,772 self.max_marginal_relevance_search_with_score_by_vector,773 embedding,774 k=k,775 fetch_k=fetch_k,776 lambda_mult=lambda_mult,777 filter=filter,778 )779 780 def max_marginal_relevance_search_by_vector(781 self,782 embedding: List[float],783 k: int = 4,784 fetch_k: int = 20,785 lambda_mult: float = 0.5,786 filter: Optional[Union[Callable, Dict[str, Any]]] = None,787 **kwargs: Any,788 ) -> List[Document]:789 """Return docs selected using the maximal marginal relevance.790 791 Maximal marginal relevance optimizes for similarity to query AND diversity792 among selected documents.793 794 Args:795 embedding: Embedding to look up documents similar to.796 k: Number of Documents to return. Defaults to 4.797 fetch_k: Number of Documents to fetch before filtering to798 pass to MMR algorithm.799 lambda_mult: Number between 0 and 1 that determines the degree800 of diversity among the results with 0 corresponding801 to maximum diversity and 1 to minimum diversity.802 Defaults to 0.5.803 Returns:804 List of Documents selected by maximal marginal relevance.805 """806 docs_and_scores = self.max_marginal_relevance_search_with_score_by_vector(807 embedding, k=k, fetch_k=fetch_k, lambda_mult=lambda_mult, filter=filter808 )809 return [doc for doc, _ in docs_and_scores]810 811 async def amax_marginal_relevance_search_by_vector(812 self,813 embedding: List[float],814 k: int = 4,815 fetch_k: int = 20,816 lambda_mult: float = 0.5,817 filter: Optional[Union[Callable, Dict[str, Any]]] = None,818 **kwargs: Any,819 ) -> List[Document]:820 """Return docs selected using the maximal marginal relevance asynchronously.821 822 Maximal marginal relevance optimizes for similarity to query AND diversity823 among selected documents.824 825 Args:826 embedding: Embedding to look up documents similar to.827 k: Number of Documents to return. Defaults to 4.828 fetch_k: Number of Documents to fetch before filtering to829 pass to MMR algorithm.830 lambda_mult: Number between 0 and 1 that determines the degree831 of diversity among the results with 0 corresponding832 to maximum diversity and 1 to minimum diversity.833 Defaults to 0.5.834 Returns:835 List of Documents selected by maximal marginal relevance.836 """837 docs_and_scores = (838 await self.amax_marginal_relevance_search_with_score_by_vector(839 embedding, k=k, fetch_k=fetch_k, lambda_mult=lambda_mult, filter=filter840 )841 )842 return [doc for doc, _ in docs_and_scores]843 844 def max_marginal_relevance_search(845 self,846 query: str,847 k: int = 4,848 fetch_k: int = 20,849 lambda_mult: float = 0.5,850 filter: Optional[Union[Callable, Dict[str, Any]]] = None,851 **kwargs: Any,852 ) -> List[Document]:853 """Return docs selected using the maximal marginal relevance.854 855 Maximal marginal relevance optimizes for similarity to query AND diversity856 among selected documents.857 858 Args:859 query: Text to look up documents similar to.860 k: Number of Documents to return. Defaults to 4.861 fetch_k: Number of Documents to fetch before filtering (if needed) to862 pass to MMR algorithm.863 lambda_mult: Number between 0 and 1 that determines the degree864 of diversity among the results with 0 corresponding865 to maximum diversity and 1 to minimum diversity.866 Defaults to 0.5.867 Returns:868 List of Documents selected by maximal marginal relevance.869 """870 embedding = self._embed_query(query)871 docs = self.max_marginal_relevance_search_by_vector(872 embedding,873 k=k,874 fetch_k=fetch_k,875 lambda_mult=lambda_mult,876 filter=filter,877 **kwargs,878 )879 return docs880 881 async def amax_marginal_relevance_search(882 self,883 query: str,884 k: int = 4,885 fetch_k: int = 20,886 lambda_mult: float = 0.5,887 filter: Optional[Union[Callable, Dict[str, Any]]] = None,888 **kwargs: Any,889 ) -> List[Document]:890 """Return docs selected using the maximal marginal relevance asynchronously.891 892 Maximal marginal relevance optimizes for similarity to query AND diversity893 among selected documents.894 895 Args:896 query: Text to look up documents similar to.897 k: Number of Documents to return. Defaults to 4.898 fetch_k: Number of Documents to fetch before filtering (if needed) to899 pass to MMR algorithm.900 lambda_mult: Number between 0 and 1 that determines the degree901 of diversity among the results with 0 corresponding902 to maximum diversity and 1 to minimum diversity.903 Defaults to 0.5.904 Returns:905 List of Documents selected by maximal marginal relevance.906 """907 embedding = await self._aembed_query(query)908 docs = await self.amax_marginal_relevance_search_by_vector(909 embedding,910 k=k,911 fetch_k=fetch_k,912 lambda_mult=lambda_mult,913 filter=filter,914 **kwargs,915 )916 return docs917 918 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> Optional[bool]:919 """Delete by ID. These are the IDs in the vectorstore.920 921 Args:922 ids: List of ids to delete.923 924 Returns:925 Optional[bool]: True if deletion is successful,926 False otherwise, None if not implemented.927 """928 if ids is None:929 raise ValueError("No ids provided to delete.")930 missing_ids = set(ids).difference(self.index_to_docstore_id.values())931 if missing_ids:932 raise ValueError(933 f"Some specified ids do not exist in the current store. Ids not found: "934 f"{missing_ids}"935 )936 937 reversed_index = {id_: idx for idx, id_ in self.index_to_docstore_id.items()}938 index_to_delete = {reversed_index[id_] for id_ in ids}939 940 self.index.remove_ids(np.fromiter(index_to_delete, dtype=np.int64))941 self.docstore.delete(ids)942 943 remaining_ids = [944 id_945 for i, id_ in sorted(self.index_to_docstore_id.items())946 if i not in index_to_delete947 ]948 self.index_to_docstore_id = {i: id_ for i, id_ in enumerate(remaining_ids)}949 950 return True951 952 def merge_from(self, target: FAISS) -> None:953 """Merge another FAISS object with the current one.954 955 Add the target FAISS to the current one.956 957 Args:958 target: FAISS object you wish to merge into the current one959 960 Returns:961 None.962 """963 if not isinstance(self.docstore, AddableMixin):964 raise ValueError("Cannot merge with this type of docstore")965 # Numerical index for target docs are incremental on existing ones966 starting_len = len(self.index_to_docstore_id)967 968 # Merge two IndexFlatL2969 self.index.merge_from(target.index)970 971 # Get id and docs from target FAISS object972 full_info = []973 for i, target_id in target.index_to_docstore_id.items():974 doc = target.docstore.search(target_id)975 if not isinstance(doc, Document):976 raise ValueError("Document should be returned")977 full_info.append((starting_len + i, target_id, doc))978 979 # Add information to docstore and index_to_docstore_id.980 self.docstore.add({_id: doc for _, _id, doc in full_info})981 index_to_id = {index: _id for index, _id, _ in full_info}982 self.index_to_docstore_id.update(index_to_id)983 984 @classmethod985 def __from(986 cls,987 texts: Iterable[str],988 embeddings: List[List[float]],989 embedding: Embeddings,990 metadatas: Optional[Iterable[dict]] = None,991 ids: Optional[List[str]] = None,992 normalize_L2: bool = False,993 distance_strategy: DistanceStrategy = DistanceStrategy.EUCLIDEAN_DISTANCE,994 **kwargs: Any,995 ) -> FAISS:996 faiss = dependable_faiss_import()997 if distance_strategy == DistanceStrategy.MAX_INNER_PRODUCT:998 index = faiss.IndexFlatIP(len(embeddings[0]))999 else:1000 # Default to L2, currently other metric types not initialized.1001 index = faiss.IndexFlatL2(len(embeddings[0]))1002 docstore = kwargs.pop("docstore", InMemoryDocstore())1003 index_to_docstore_id = kwargs.pop("index_to_docstore_id", {})1004 vecstore = cls(1005 embedding,1006 index,1007 docstore,1008 index_to_docstore_id,1009 normalize_L2=normalize_L2,1010 distance_strategy=distance_strategy,1011 **kwargs,1012 )1013 vecstore.__add(texts, embeddings, metadatas=metadatas, ids=ids)1014 return vecstore1015 1016 @classmethod1017 def from_texts(1018 cls,1019 texts: List[str],1020 embedding: Embeddings,1021 metadatas: Optional[List[dict]] = None,1022 ids: Optional[List[str]] = None,1023 **kwargs: Any,1024 ) -> FAISS:1025 """Construct FAISS wrapper from raw documents.1026 1027 This is a user friendly interface that:1028 1. Embeds documents.1029 2. Creates an in memory docstore1030 3. Initializes the FAISS database1031 1032 This is intended to be a quick way to get started.1033 1034 Example:1035 .. code-block:: python1036 1037 from langchain_community.vectorstores import FAISS1038 from langchain_community.embeddings import OpenAIEmbeddings1039 1040 embeddings = OpenAIEmbeddings()1041 faiss = FAISS.from_texts(texts, embeddings)1042 """1043 embeddings = embedding.embed_documents(texts)1044 return cls.__from(1045 texts,1046 embeddings,1047 embedding,1048 metadatas=metadatas,1049 ids=ids,1050 **kwargs,1051 )1052 1053 @classmethod1054 async def afrom_texts(1055 cls,1056 texts: list[str],1057 embedding: Embeddings,1058 metadatas: Optional[List[dict]] = None,1059 ids: Optional[List[str]] = None,1060 **kwargs: Any,1061 ) -> FAISS:1062 """Construct FAISS wrapper from raw documents asynchronously.1063 1064 This is a user friendly interface that:1065 1. Embeds documents.1066 2. Creates an in memory docstore1067 3. Initializes the FAISS database1068 1069 This is intended to be a quick way to get started.1070 1071 Example:1072 .. code-block:: python1073 1074 from langchain_community.vectorstores import FAISS1075 from langchain_community.embeddings import OpenAIEmbeddings1076 1077 embeddings = OpenAIEmbeddings()1078 faiss = await FAISS.afrom_texts(texts, embeddings)1079 """1080 embeddings = await embedding.aembed_documents(texts)1081 return cls.__from(1082 texts,1083 embeddings,1084 embedding,1085 metadatas=metadatas,1086 ids=ids,1087 **kwargs,1088 )1089 1090 @classmethod1091 def from_embeddings(1092 cls,1093 text_embeddings: Iterable[Tuple[str, List[float]]],1094 embedding: Embeddings,1095 metadatas: Optional[Iterable[dict]] = None,1096 ids: Optional[List[str]] = None,1097 **kwargs: Any,1098 ) -> FAISS:1099 """Construct FAISS wrapper from raw documents.1100 1101 This is a user friendly interface that:1102 1. Embeds documents.1103 2. Creates an in memory docstore1104 3. Initializes the FAISS database1105 1106 This is intended to be a quick way to get started.1107 1108 Example:1109 .. code-block:: python1110 1111 from langchain_community.vectorstores import FAISS1112 from langchain_community.embeddings import OpenAIEmbeddings1113 1114 embeddings = OpenAIEmbeddings()1115 text_embeddings = embeddings.embed_documents(texts)1116 text_embedding_pairs = zip(texts, text_embeddings)1117 faiss = FAISS.from_embeddings(text_embedding_pairs, embeddings)1118 """1119 texts, embeddings = zip(*text_embeddings)1120 return cls.__from(1121 list(texts),1122 list(embeddings),1123 embedding,1124 metadatas=metadatas,1125 ids=ids,1126 **kwargs,1127 )1128 1129 @classmethod1130 async def afrom_embeddings(1131 cls,1132 text_embeddings: Iterable[Tuple[str, List[float]]],1133 embedding: Embeddings,1134 metadatas: Optional[Iterable[dict]] = None,1135 ids: Optional[List[str]] = None,1136 **kwargs: Any,1137 ) -> FAISS:1138 """Construct FAISS wrapper from raw documents asynchronously."""1139 return cls.from_embeddings(1140 text_embeddings,1141 embedding,1142 metadatas=metadatas,1143 ids=ids,1144 **kwargs,1145 )1146 1147 def save_local(self, folder_path: str, index_name: str = "index") -> None:1148 """Save FAISS index, docstore, and index_to_docstore_id to disk.1149 1150 Args:1151 folder_path: folder path to save index, docstore,1152 and index_to_docstore_id to.1153 index_name: for saving with a specific index file name1154 """1155 path = Path(folder_path)1156 path.mkdir(exist_ok=True, parents=True)1157 1158 # save index separately since it is not picklable1159 faiss = dependable_faiss_import()1160 faiss.write_index(self.index, str(path / f"{index_name}.faiss"))1161 1162 # save docstore and index_to_docstore_id1163 with open(path / f"{index_name}.pkl", "wb") as f:1164 pickle.dump((self.docstore, self.index_to_docstore_id), f)1165 1166 @classmethod1167 def load_local(1168 cls,1169 folder_path: str,1170 embeddings: Embeddings,1171 index_name: str = "index",1172 *,1173 allow_dangerous_deserialization: bool = False,1174 **kwargs: Any,1175 ) -> FAISS:1176 """Load FAISS index, docstore, and index_to_docstore_id from disk.1177 1178 Args:1179 folder_path: folder path to load index, docstore,1180 and index_to_docstore_id from.1181 embeddings: Embeddings to use when generating queries1182 index_name: for saving with a specific index file name1183 allow_dangerous_deserialization: whether to allow deserialization1184 of the data which involves loading a pickle file.1185 Pickle files can be modified by malicious actors to deliver a1186 malicious payload that results in execution of1187 arbitrary code on your machine.1188 """1189 if not allow_dangerous_deserialization:1190 raise ValueError(1191 "The de-serialization relies loading a pickle file. "1192 "Pickle files can be modified to deliver a malicious payload that "1193 "results in execution of arbitrary code on your machine."1194 "You will need to set `allow_dangerous_deserialization` to `True` to "1195 "enable deserialization. If you do this, make sure that you "1196 "trust the source of the data. For example, if you are loading a "1197 "file that you created, and know that no one else has modified the "1198 "file, then this is safe to do. Do not set this to `True` if you are "1199 "loading a file from an untrusted source (e.g., some random site on "1200 "the internet.)."