Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
usearch.py179 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3from typing import Any, Dict, Iterable, List, Optional, Tuple, Union, cast4 5import numpy as np6from langchain_core.documents import Document7from langchain_core.embeddings import Embeddings8from langchain_core.utils import guard_import9from langchain_core.vectorstores import VectorStore10 11from langchain_community.docstore.base import AddableMixin, Docstore12from langchain_community.docstore.in_memory import InMemoryDocstore13 14 15def dependable_usearch_import() -> Any:16    """17    Import usearch if available, otherwise raise error.18    """19    return guard_import("usearch.index")20 21 22class USearch(VectorStore):23    """`USearch` vector store.24 25    To use, you should have the ``usearch`` python package installed.26    """27 28    def __init__(29        self,30        embedding: Embeddings,31        index: Any,32        docstore: Docstore,33        ids: List[str],34    ):35        """Initialize with necessary components."""36        self.embedding = embedding37        self.index = index38        self.docstore = docstore39        self.ids = ids40 41    def add_texts(42        self,43        texts: Iterable[str],44        metadatas: Optional[List[Dict]] = None,45        ids: Optional[Union[np.ndarray, list[str]]] = None,46        **kwargs: Any,47    ) -> List[str]:48        """Run more texts through the embeddings and add to the vectorstore.49 50        Args:51            texts: Iterable of strings to add to the vectorstore.52            metadatas: Optional list of metadatas associated with the texts.53            ids: Optional list of unique IDs.54 55        Returns:56            List of ids from adding the texts into the vectorstore.57        """58        if not isinstance(self.docstore, AddableMixin):59            raise ValueError(60                "If trying to add texts, the underlying docstore should support "61                f"adding items, which {self.docstore} does not"62            )63 64        embeddings = self.embedding.embed_documents(list(texts))65        documents = []66        for i, text in enumerate(texts):67            metadata = metadatas[i] if metadatas else {}68            documents.append(Document(page_content=text, metadata=metadata))69 70        if ids is None:71            if self.ids:72                last_id = int(self.ids[-1]) + 173                ids = np.array([str(last_id + id) for id, _ in enumerate(texts)])74            else:75                ids = np.array([str(id) for id, _ in enumerate(texts)])76        elif isinstance(ids, list):77            ids = np.array(ids)78 79        self.index.add(np.array(ids), np.array(embeddings))80        self.docstore.add(dict(zip(ids, documents)))81        self.ids.extend(ids)82        return cast(List[str], ids.tolist())83 84    def similarity_search_with_score(85        self,86        query: str,87        k: int = 4,88    ) -> List[Tuple[Document, float]]:89        """Return docs most similar to query.90 91        Args:92            query: Text to look up documents similar to.93            k: Number of Documents to return. Defaults to 4.94 95        Returns:96            List of documents most similar to the query with distance.97        """98        query_embedding = self.embedding.embed_query(query)99        matches = self.index.search(np.array(query_embedding), k)100 101        docs_with_scores: List[Tuple[Document, float]] = []102        for id, score in zip(matches.keys, matches.distances):103            doc = self.docstore.search(str(id))104            if not isinstance(doc, Document):105                raise ValueError(f"Could not find document for id {id}, got {doc}")106            docs_with_scores.append((doc, score))107 108        return docs_with_scores109 110    def similarity_search(111        self,112        query: str,113        k: int = 4,114        **kwargs: Any,115    ) -> List[Document]:116        """Return docs most similar to query.117 118        Args:119            query: Text to look up documents similar to.120            k: Number of Documents to return. Defaults to 4.121 122        Returns:123            List of Documents most similar to the query.124        """125        query_embedding = self.embedding.embed_query(query)126        matches = self.index.search(np.array(query_embedding), k)127 128        docs: List[Document] = []129        for id in matches.keys:130            doc = self.docstore.search(str(id))131            if not isinstance(doc, Document):132                raise ValueError(f"Could not find document for id {id}, got {doc}")133            docs.append(doc)134 135        return docs136 137    @classmethod138    def from_texts(139        cls,140        texts: List[str],141        embedding: Embeddings,142        metadatas: Optional[List[Dict]] = None,143        ids: Optional[Union[np.ndarray, list[str]]] = None,144        metric: str = "cos",145        **kwargs: Any,146    ) -> USearch:147        """Construct USearch wrapper from raw documents.148        This is a user friendly interface that:149            1. Embeds documents.150            2. Creates an in memory docstore151            3. Initializes the USearch database152        This is intended to be a quick way to get started.153 154        Example:155            .. code-block:: python156 157                from langchain_community.vectorstores import USearch158                from langchain_community.embeddings import OpenAIEmbeddings159 160                embeddings = OpenAIEmbeddings()161                usearch = USearch.from_texts(texts, embeddings)162        """163        embeddings = embedding.embed_documents(texts)164 165        documents: List[Document] = []166        if ids is None:167            ids = np.array([str(id) for id, _ in enumerate(texts)])168        elif isinstance(ids, list):169            ids = np.array(ids)170        for i, text in enumerate(texts):171            metadata = metadatas[i] if metadatas else {}172            documents.append(Document(page_content=text, metadata=metadata))173 174        docstore = InMemoryDocstore(dict(zip(ids, documents)))175        usearch = guard_import("usearch.index")176        index = usearch.Index(ndim=len(embeddings[0]), metric=metric)177        index.add(np.array(ids), np.array(embeddings))178        return cls(embedding, index, docstore, cast(List[str], ids.tolist()))179 
codekingpro/portable-devtools · Team Ai