codekingpro/portable-devtools
114k
1from __future__ import annotations2 3from typing import Any, Dict, Iterable, List, Optional, Tuple, Union, cast4 5import numpy as np6from langchain_core.documents import Document7from langchain_core.embeddings import Embeddings8from langchain_core.utils import guard_import9from langchain_core.vectorstores import VectorStore10 11from langchain_community.docstore.base import AddableMixin, Docstore12from langchain_community.docstore.in_memory import InMemoryDocstore13 14 15def dependable_usearch_import() -> Any:16 """17 Import usearch if available, otherwise raise error.18 """19 return guard_import("usearch.index")20 21 22class USearch(VectorStore):23 """`USearch` vector store.24 25 To use, you should have the ``usearch`` python package installed.26 """27 28 def __init__(29 self,30 embedding: Embeddings,31 index: Any,32 docstore: Docstore,33 ids: List[str],34 ):35 """Initialize with necessary components."""36 self.embedding = embedding37 self.index = index38 self.docstore = docstore39 self.ids = ids40 41 def add_texts(42 self,43 texts: Iterable[str],44 metadatas: Optional[List[Dict]] = None,45 ids: Optional[Union[np.ndarray, list[str]]] = None,46 **kwargs: Any,47 ) -> List[str]:48 """Run more texts through the embeddings and add to the vectorstore.49 50 Args:51 texts: Iterable of strings to add to the vectorstore.52 metadatas: Optional list of metadatas associated with the texts.53 ids: Optional list of unique IDs.54 55 Returns:56 List of ids from adding the texts into the vectorstore.57 """58 if not isinstance(self.docstore, AddableMixin):59 raise ValueError(60 "If trying to add texts, the underlying docstore should support "61 f"adding items, which {self.docstore} does not"62 )63 64 embeddings = self.embedding.embed_documents(list(texts))65 documents = []66 for i, text in enumerate(texts):67 metadata = metadatas[i] if metadatas else {}68 documents.append(Document(page_content=text, metadata=metadata))69 70 if ids is None:71 if self.ids:72 last_id = int(self.ids[-1]) + 173 ids = np.array([str(last_id + id) for id, _ in enumerate(texts)])74 else:75 ids = np.array([str(id) for id, _ in enumerate(texts)])76 elif isinstance(ids, list):77 ids = np.array(ids)78 79 self.index.add(np.array(ids), np.array(embeddings))80 self.docstore.add(dict(zip(ids, documents)))81 self.ids.extend(ids)82 return cast(List[str], ids.tolist())83 84 def similarity_search_with_score(85 self,86 query: str,87 k: int = 4,88 ) -> List[Tuple[Document, float]]:89 """Return docs most similar to query.90 91 Args:92 query: Text to look up documents similar to.93 k: Number of Documents to return. Defaults to 4.94 95 Returns:96 List of documents most similar to the query with distance.97 """98 query_embedding = self.embedding.embed_query(query)99 matches = self.index.search(np.array(query_embedding), k)100 101 docs_with_scores: List[Tuple[Document, float]] = []102 for id, score in zip(matches.keys, matches.distances):103 doc = self.docstore.search(str(id))104 if not isinstance(doc, Document):105 raise ValueError(f"Could not find document for id {id}, got {doc}")106 docs_with_scores.append((doc, score))107 108 return docs_with_scores109 110 def similarity_search(111 self,112 query: str,113 k: int = 4,114 **kwargs: Any,115 ) -> List[Document]:116 """Return docs most similar to query.117 118 Args:119 query: Text to look up documents similar to.120 k: Number of Documents to return. Defaults to 4.121 122 Returns:123 List of Documents most similar to the query.124 """125 query_embedding = self.embedding.embed_query(query)126 matches = self.index.search(np.array(query_embedding), k)127 128 docs: List[Document] = []129 for id in matches.keys:130 doc = self.docstore.search(str(id))131 if not isinstance(doc, Document):132 raise ValueError(f"Could not find document for id {id}, got {doc}")133 docs.append(doc)134 135 return docs136 137 @classmethod138 def from_texts(139 cls,140 texts: List[str],141 embedding: Embeddings,142 metadatas: Optional[List[Dict]] = None,143 ids: Optional[Union[np.ndarray, list[str]]] = None,144 metric: str = "cos",145 **kwargs: Any,146 ) -> USearch:147 """Construct USearch wrapper from raw documents.148 This is a user friendly interface that:149 1. Embeds documents.150 2. Creates an in memory docstore151 3. Initializes the USearch database152 This is intended to be a quick way to get started.153 154 Example:155 .. code-block:: python156 157 from langchain_community.vectorstores import USearch158 from langchain_community.embeddings import OpenAIEmbeddings159 160 embeddings = OpenAIEmbeddings()161 usearch = USearch.from_texts(texts, embeddings)162 """163 embeddings = embedding.embed_documents(texts)164 165 documents: List[Document] = []166 if ids is None:167 ids = np.array([str(id) for id, _ in enumerate(texts)])168 elif isinstance(ids, list):169 ids = np.array(ids)170 for i, text in enumerate(texts):171 metadata = metadatas[i] if metadatas else {}172 documents.append(Document(page_content=text, metadata=metadata))173 174 docstore = InMemoryDocstore(dict(zip(ids, documents)))175 usearch = guard_import("usearch.index")176 index = usearch.Index(ndim=len(embeddings[0]), metric=metric)177 index.add(np.array(ids), np.array(embeddings))178 return cls(embedding, index, docstore, cast(List[str], ids.tolist()))179 