Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
marqo.py477 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import json4import uuid5from typing import (6    TYPE_CHECKING,7    Any,8    Callable,9    Dict,10    Iterable,11    List,12    Optional,13    Tuple,14    Type,15    Union,16)17 18from langchain_core.documents import Document19from langchain_core.embeddings import Embeddings20from langchain_core.vectorstores import VectorStore21 22if TYPE_CHECKING:23    import marqo24 25 26class Marqo(VectorStore):27    """`Marqo` vector store.28 29    Marqo indexes have their own models associated with them to generate your30    embeddings. This means that you can selected from a range of different models31    and also use CLIP models to create multimodal indexes32    with images and text together.33 34    Marqo also supports more advanced queries with multiple weighted terms, see See35    https://docs.marqo.ai/latest/#searching-using-weights-in-queries.36    This class can flexibly take strings or dictionaries for weighted queries37    in its similarity search methods.38 39    To use, you should have the `marqo` python package installed, you can do this with40    `pip install marqo`.41 42    Example:43        .. code-block:: python44 45            import marqo46            from langchain_community.vectorstores import Marqo47            client = marqo.Client(url=os.environ["MARQO_URL"], ...)48            vectorstore = Marqo(client, index_name)49 50    """51 52    def __init__(53        self,54        client: marqo.Client,55        index_name: str,56        add_documents_settings: Optional[Dict[str, Any]] = None,57        searchable_attributes: Optional[List[str]] = None,58        page_content_builder: Optional[Callable[[Dict[str, Any]], str]] = None,59    ):60        """Initialize with Marqo client."""61        try:62            import marqo63        except ImportError:64            raise ImportError(65                "Could not import marqo python package. "66                "Please install it with `pip install marqo`."67            )68        if not isinstance(client, marqo.Client):69            raise ValueError(70                f"client should be an instance of marqo.Client, got {type(client)}"71            )72        self._client = client73        self._index_name = index_name74        self._add_documents_settings = (75            {} if add_documents_settings is None else add_documents_settings76        )77        self._searchable_attributes = searchable_attributes78        self.page_content_builder = page_content_builder79 80        self.tensor_fields = ["text"]81 82        self._document_batch_size = 102483 84    @property85    def embeddings(self) -> Optional[Embeddings]:86        return None87 88    def add_texts(89        self,90        texts: Iterable[str],91        metadatas: Optional[List[dict]] = None,92        **kwargs: Any,93    ) -> List[str]:94        """Upload texts with metadata (properties) to Marqo.95 96        You can either have marqo generate ids for each document or you can provide97        your own by including a "_id" field in the metadata objects.98 99        Args:100            texts (Iterable[str]): am iterator of texts - assumed to preserve an101            order that matches the metadatas.102            metadatas (Optional[List[dict]], optional): a list of metadatas.103 104        Raises:105            ValueError: if metadatas is provided and the number of metadatas differs106            from the number of texts.107 108        Returns:109            List[str]: The list of ids that were added.110        """111 112        settings = self._client.index(self._index_name).get_settings()113        if (114            "index_defaults" in settings115            and settings["index_defaults"]["treat_urls_and_pointers_as_images"]116            or settings.get("treat_urls_and_pointers_as_images")117        ):118            raise ValueError(119                "Marqo.add_texts is disabled for multimodal indexes. To add documents "120                "with a multimodal index use the Python client for Marqo directly."121            )122        documents: List[Dict[str, str]] = []123 124        num_docs = 0125        for i, text in enumerate(texts):126            doc = {127                "text": text,128                "metadata": json.dumps(metadatas[i]) if metadatas else json.dumps({}),129            }130            documents.append(doc)131            num_docs += 1132 133        ids = []134        for i in range(0, num_docs, self._document_batch_size):135            response = self._client.index(self._index_name).add_documents(136                documents[i : i + self._document_batch_size],137                tensor_fields=self.tensor_fields,138                **self._add_documents_settings,139            )140            if response["errors"]:141                err_msg = (142                    f"Error in upload for documents in index range [{i},"143                    f"{i + self._document_batch_size}], "144                    f"check Marqo logs."145                )146                raise RuntimeError(err_msg)147 148            ids += [item["_id"] for item in response["items"]]149 150        return ids151 152    def similarity_search(153        self,154        query: Union[str, Dict[str, float]],155        k: int = 4,156        **kwargs: Any,157    ) -> List[Document]:158        """Search the marqo index for the most similar documents.159 160        Args:161            query (Union[str, Dict[str, float]]): The query for the search, either162            as a string or a weighted query.163            k (int, optional): The number of documents to return. Defaults to 4.164 165        Returns:166            List[Document]: k documents ordered from best to worst match.167        """168        results = self.marqo_similarity_search(query=query, k=k)169 170        documents = self._construct_documents_from_results_without_score(results)171        return documents172 173    def similarity_search_with_score(174        self,175        query: Union[str, Dict[str, float]],176        k: int = 4,177    ) -> List[Tuple[Document, float]]:178        """Return documents from Marqo that are similar to the query as well179        as their scores.180 181        Args:182            query (str): The query to search with, either as a string or a weighted183            query.184            k (int, optional): The number of documents to return. Defaults to 4.185 186        Returns:187            List[Tuple[Document, float]]: The matching documents and their scores,188            ordered by descending score.189        """190        results = self.marqo_similarity_search(query=query, k=k)191 192        scored_documents = self._construct_documents_from_results_with_score(results)193        return scored_documents194 195    def bulk_similarity_search(196        self,197        queries: Iterable[Union[str, Dict[str, float]]],198        k: int = 4,199        **kwargs: Any,200    ) -> List[List[Document]]:201        """Search the marqo index for the most similar documents in bulk with multiple202        queries.203 204        Args:205            queries (Iterable[Union[str, Dict[str, float]]]): An iterable of queries to206            execute in bulk, queries in the list can be strings or dictionaries of207            weighted queries.208            k (int, optional): The number of documents to return for each query.209            Defaults to 4.210 211        Returns:212            List[List[Document]]: A list of results for each query.213        """214        bulk_results = self.marqo_bulk_similarity_search(queries=queries, k=k)215        bulk_documents: List[List[Document]] = []216        for results in bulk_results["result"]:217            documents = self._construct_documents_from_results_without_score(results)218            bulk_documents.append(documents)219 220        return bulk_documents221 222    def bulk_similarity_search_with_score(223        self,224        queries: Iterable[Union[str, Dict[str, float]]],225        k: int = 4,226        **kwargs: Any,227    ) -> List[List[Tuple[Document, float]]]:228        """Return documents from Marqo that are similar to the query as well as229        their scores using a batch of queries.230 231        Args:232            query (Iterable[Union[str, Dict[str, float]]]): An iterable of queries233            to execute in bulk, queries in the list can be strings or dictionaries234            of weighted queries.235            k (int, optional): The number of documents to return. Defaults to 4.236 237        Returns:238            List[Tuple[Document, float]]: A list of lists of the matching239            documents and their scores for each query240        """241        bulk_results = self.marqo_bulk_similarity_search(queries=queries, k=k)242        bulk_documents: List[List[Tuple[Document, float]]] = []243        for results in bulk_results["result"]:244            documents = self._construct_documents_from_results_with_score(results)245            bulk_documents.append(documents)246 247        return bulk_documents248 249    def _construct_documents_from_results_with_score(250        self, results: Dict[str, List[Dict[str, str]]]251    ) -> List[Tuple[Document, Any]]:252        """Helper to convert Marqo results into documents.253 254        Args:255            results (List[dict]): A marqo results object with the 'hits'.256            include_scores (bool, optional): Include scores alongside documents.257            Defaults to False.258 259        Returns:260            Union[List[Document], List[Tuple[Document, float]]]: The documents or261            document score pairs if `include_scores` is true.262        """263        documents: List[Tuple[Document, Any]] = []264        for res in results["hits"]:265            if self.page_content_builder is None:266                text = res["text"]267            else:268                text = self.page_content_builder(res)269 270            metadata = json.loads(res.get("metadata", "{}"))271            documents.append(272                (273                    Document(page_content=text, metadata=metadata),274                    res["_score"],275                )276            )277        return documents278 279    def _construct_documents_from_results_without_score(280        self, results: Dict[str, List[Dict[str, str]]]281    ) -> List[Document]:282        """Helper to convert Marqo results into documents.283 284        Args:285            results (List[dict]): A marqo results object with the 'hits'.286            include_scores (bool, optional): Include scores alongside documents.287            Defaults to False.288 289        Returns:290            Union[List[Document], List[Tuple[Document, float]]]: The documents or291            document score pairs if `include_scores` is true.292        """293        documents: List[Document] = []294        for res in results["hits"]:295            if self.page_content_builder is None:296                text = res["text"]297            else:298                text = self.page_content_builder(res)299 300            metadata = json.loads(res.get("metadata", "{}"))301            documents.append(Document(page_content=text, metadata=metadata))302        return documents303 304    def marqo_similarity_search(305        self,306        query: Union[str, Dict[str, float]],307        k: int = 4,308    ) -> Dict[str, List[Dict[str, str]]]:309        """Return documents from Marqo exposing Marqo's output directly310 311        Args:312            query (str): The query to search with.313            k (int, optional): The number of documents to return. Defaults to 4.314 315        Returns:316            List[Dict[str, Any]]: This hits from marqo.317        """318        results = self._client.index(self._index_name).search(319            q=query, searchable_attributes=self._searchable_attributes, limit=k320        )321        return results322 323    def marqo_bulk_similarity_search(324        self, queries: Iterable[Union[str, Dict[str, float]]], k: int = 4325    ) -> Dict[str, List[Dict[str, List[Dict[str, str]]]]]:326        """Return documents from Marqo using a bulk search, exposes Marqo's327        output directly328 329        Args:330            queries (Iterable[Union[str, Dict[str, float]]]): A list of queries.331            k (int, optional): The number of documents to return for each query.332            Defaults to 4.333 334        Returns:335            Dict[str, Dict[List[Dict[str, Dict[str, Any]]]]]: A bulk search results336            object337        """338        bulk_results = {339            "result": [340                self._client.index(self._index_name).search(341                    q=query, searchable_attributes=self._searchable_attributes, limit=k342                )343                for query in queries344            ]345        }346 347        return bulk_results348 349    @classmethod350    def from_documents(351        cls: Type[Marqo],352        documents: List[Document],353        embedding: Union[Embeddings, None] = None,354        **kwargs: Any,355    ) -> Marqo:356        """Return VectorStore initialized from documents. Note that Marqo does not357        need embeddings, we retain the parameter to adhere to the Liskov substitution358        principle.359 360 361        Args:362            documents (List[Document]): Input documents363            embedding (Any, optional): Embeddings (not required). Defaults to None.364 365        Returns:366            VectorStore: A Marqo vectorstore367        """368        texts = [d.page_content for d in documents]369        metadatas = [d.metadata for d in documents]370        return cls.from_texts(texts, metadatas=metadatas, **kwargs)371 372    @classmethod373    def from_texts(374        cls,375        texts: List[str],376        embedding: Any = None,377        metadatas: Optional[List[dict]] = None,378        index_name: str = "",379        url: str = "http://localhost:8882",380        api_key: str = "",381        add_documents_settings: Optional[Dict[str, Any]] = None,382        searchable_attributes: Optional[List[str]] = None,383        page_content_builder: Optional[Callable[[Dict[str, str]], str]] = None,384        index_settings: Optional[Dict[str, Any]] = None,385        verbose: bool = True,386        **kwargs: Any,387    ) -> Marqo:388        """Return Marqo initialized from texts. Note that Marqo does not need389        embeddings, we retain the parameter to adhere to the Liskov390        substitution principle.391 392        This is a quick way to get started with marqo - simply provide your texts and393        metadatas and this will create an instance of the data store and index the394        provided data.395 396        To know the ids of your documents with this approach you will need to include397        them in under the key "_id" in your metadatas for each text398 399        Example:400        .. code-block:: python401 402                from langchain_community.vectorstores import Marqo403 404                datastore = Marqo(texts=['text'], index_name='my-first-index',405                url='http://localhost:8882')406 407        Args:408            texts (List[str]): A list of texts to index into marqo upon creation.409            embedding (Any, optional): Embeddings (not required). Defaults to None.410            index_name (str, optional): The name of the index to use, if none is411            provided then one will be created with a UUID. Defaults to None.412            url (str, optional): The URL for Marqo. Defaults to "http://localhost:8882".413            api_key (str, optional): The API key for Marqo. Defaults to "".414            metadatas (Optional[List[dict]], optional): A list of metadatas, to415            accompany the texts. Defaults to None.416            this is only used when a new index is being created. Defaults to "cpu". Can417            be "cpu" or "cuda".418            add_documents_settings (Optional[Dict[str, Any]], optional): Settings419            for adding documents, see420            https://docs.marqo.ai/0.0.16/API-Reference/documents/#query-parameters.421            Defaults to {}.422            index_settings (Optional[Dict[str, Any]], optional): Index settings if423            the index doesn't exist, see424            https://docs.marqo.ai/0.0.16/API-Reference/indexes/#index-defaults-object.425            Defaults to {}.426 427        Returns:428            Marqo: An instance of the Marqo vector store429        """430        try:431            import marqo432        except ImportError:433            raise ImportError(434                "Could not import marqo python package. "435                "Please install it with `pip install marqo`."436            )437 438        if not index_name:439            index_name = str(uuid.uuid4())440 441        client = marqo.Client(url=url, api_key=api_key)442 443        try:444            client.create_index(index_name, settings_dict=index_settings or {})445            if verbose:446                print(f"Created {index_name} successfully.")  # noqa: T201447        except Exception:448            if verbose:449                print(f"Index {index_name} exists.")  # noqa: T201450 451        instance: Marqo = cls(452            client,453            index_name,454            searchable_attributes=searchable_attributes,455            add_documents_settings=add_documents_settings or {},456            page_content_builder=page_content_builder,457        )458        instance.add_texts(texts, metadatas)459        return instance460 461    def get_indexes(self) -> List[Dict[str, str]]:462        """Helper to see your available indexes in marqo, useful if the463        from_texts method was used without an index name specified464 465        Returns:466            List[Dict[str, str]]: The list of indexes467        """468        return self._client.get_indexes()["results"]469 470    def get_number_of_documents(self) -> int:471        """Helper to see the number of documents in the index472 473        Returns:474            int: The number of documents475        """476        return self._client.index(self._index_name).get_stats()["numberOfDocuments"]477 
codekingpro/portable-devtools · Team Ai