codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import json4import uuid5from typing import (6 TYPE_CHECKING,7 Any,8 Callable,9 Dict,10 Iterable,11 List,12 Optional,13 Tuple,14 Type,15 Union,16)17 18from langchain_core.documents import Document19from langchain_core.embeddings import Embeddings20from langchain_core.vectorstores import VectorStore21 22if TYPE_CHECKING:23 import marqo24 25 26class Marqo(VectorStore):27 """`Marqo` vector store.28 29 Marqo indexes have their own models associated with them to generate your30 embeddings. This means that you can selected from a range of different models31 and also use CLIP models to create multimodal indexes32 with images and text together.33 34 Marqo also supports more advanced queries with multiple weighted terms, see See35 https://docs.marqo.ai/latest/#searching-using-weights-in-queries.36 This class can flexibly take strings or dictionaries for weighted queries37 in its similarity search methods.38 39 To use, you should have the `marqo` python package installed, you can do this with40 `pip install marqo`.41 42 Example:43 .. code-block:: python44 45 import marqo46 from langchain_community.vectorstores import Marqo47 client = marqo.Client(url=os.environ["MARQO_URL"], ...)48 vectorstore = Marqo(client, index_name)49 50 """51 52 def __init__(53 self,54 client: marqo.Client,55 index_name: str,56 add_documents_settings: Optional[Dict[str, Any]] = None,57 searchable_attributes: Optional[List[str]] = None,58 page_content_builder: Optional[Callable[[Dict[str, Any]], str]] = None,59 ):60 """Initialize with Marqo client."""61 try:62 import marqo63 except ImportError:64 raise ImportError(65 "Could not import marqo python package. "66 "Please install it with `pip install marqo`."67 )68 if not isinstance(client, marqo.Client):69 raise ValueError(70 f"client should be an instance of marqo.Client, got {type(client)}"71 )72 self._client = client73 self._index_name = index_name74 self._add_documents_settings = (75 {} if add_documents_settings is None else add_documents_settings76 )77 self._searchable_attributes = searchable_attributes78 self.page_content_builder = page_content_builder79 80 self.tensor_fields = ["text"]81 82 self._document_batch_size = 102483 84 @property85 def embeddings(self) -> Optional[Embeddings]:86 return None87 88 def add_texts(89 self,90 texts: Iterable[str],91 metadatas: Optional[List[dict]] = None,92 **kwargs: Any,93 ) -> List[str]:94 """Upload texts with metadata (properties) to Marqo.95 96 You can either have marqo generate ids for each document or you can provide97 your own by including a "_id" field in the metadata objects.98 99 Args:100 texts (Iterable[str]): am iterator of texts - assumed to preserve an101 order that matches the metadatas.102 metadatas (Optional[List[dict]], optional): a list of metadatas.103 104 Raises:105 ValueError: if metadatas is provided and the number of metadatas differs106 from the number of texts.107 108 Returns:109 List[str]: The list of ids that were added.110 """111 112 settings = self._client.index(self._index_name).get_settings()113 if (114 "index_defaults" in settings115 and settings["index_defaults"]["treat_urls_and_pointers_as_images"]116 or settings.get("treat_urls_and_pointers_as_images")117 ):118 raise ValueError(119 "Marqo.add_texts is disabled for multimodal indexes. To add documents "120 "with a multimodal index use the Python client for Marqo directly."121 )122 documents: List[Dict[str, str]] = []123 124 num_docs = 0125 for i, text in enumerate(texts):126 doc = {127 "text": text,128 "metadata": json.dumps(metadatas[i]) if metadatas else json.dumps({}),129 }130 documents.append(doc)131 num_docs += 1132 133 ids = []134 for i in range(0, num_docs, self._document_batch_size):135 response = self._client.index(self._index_name).add_documents(136 documents[i : i + self._document_batch_size],137 tensor_fields=self.tensor_fields,138 **self._add_documents_settings,139 )140 if response["errors"]:141 err_msg = (142 f"Error in upload for documents in index range [{i},"143 f"{i + self._document_batch_size}], "144 f"check Marqo logs."145 )146 raise RuntimeError(err_msg)147 148 ids += [item["_id"] for item in response["items"]]149 150 return ids151 152 def similarity_search(153 self,154 query: Union[str, Dict[str, float]],155 k: int = 4,156 **kwargs: Any,157 ) -> List[Document]:158 """Search the marqo index for the most similar documents.159 160 Args:161 query (Union[str, Dict[str, float]]): The query for the search, either162 as a string or a weighted query.163 k (int, optional): The number of documents to return. Defaults to 4.164 165 Returns:166 List[Document]: k documents ordered from best to worst match.167 """168 results = self.marqo_similarity_search(query=query, k=k)169 170 documents = self._construct_documents_from_results_without_score(results)171 return documents172 173 def similarity_search_with_score(174 self,175 query: Union[str, Dict[str, float]],176 k: int = 4,177 ) -> List[Tuple[Document, float]]:178 """Return documents from Marqo that are similar to the query as well179 as their scores.180 181 Args:182 query (str): The query to search with, either as a string or a weighted183 query.184 k (int, optional): The number of documents to return. Defaults to 4.185 186 Returns:187 List[Tuple[Document, float]]: The matching documents and their scores,188 ordered by descending score.189 """190 results = self.marqo_similarity_search(query=query, k=k)191 192 scored_documents = self._construct_documents_from_results_with_score(results)193 return scored_documents194 195 def bulk_similarity_search(196 self,197 queries: Iterable[Union[str, Dict[str, float]]],198 k: int = 4,199 **kwargs: Any,200 ) -> List[List[Document]]:201 """Search the marqo index for the most similar documents in bulk with multiple202 queries.203 204 Args:205 queries (Iterable[Union[str, Dict[str, float]]]): An iterable of queries to206 execute in bulk, queries in the list can be strings or dictionaries of207 weighted queries.208 k (int, optional): The number of documents to return for each query.209 Defaults to 4.210 211 Returns:212 List[List[Document]]: A list of results for each query.213 """214 bulk_results = self.marqo_bulk_similarity_search(queries=queries, k=k)215 bulk_documents: List[List[Document]] = []216 for results in bulk_results["result"]:217 documents = self._construct_documents_from_results_without_score(results)218 bulk_documents.append(documents)219 220 return bulk_documents221 222 def bulk_similarity_search_with_score(223 self,224 queries: Iterable[Union[str, Dict[str, float]]],225 k: int = 4,226 **kwargs: Any,227 ) -> List[List[Tuple[Document, float]]]:228 """Return documents from Marqo that are similar to the query as well as229 their scores using a batch of queries.230 231 Args:232 query (Iterable[Union[str, Dict[str, float]]]): An iterable of queries233 to execute in bulk, queries in the list can be strings or dictionaries234 of weighted queries.235 k (int, optional): The number of documents to return. Defaults to 4.236 237 Returns:238 List[Tuple[Document, float]]: A list of lists of the matching239 documents and their scores for each query240 """241 bulk_results = self.marqo_bulk_similarity_search(queries=queries, k=k)242 bulk_documents: List[List[Tuple[Document, float]]] = []243 for results in bulk_results["result"]:244 documents = self._construct_documents_from_results_with_score(results)245 bulk_documents.append(documents)246 247 return bulk_documents248 249 def _construct_documents_from_results_with_score(250 self, results: Dict[str, List[Dict[str, str]]]251 ) -> List[Tuple[Document, Any]]:252 """Helper to convert Marqo results into documents.253 254 Args:255 results (List[dict]): A marqo results object with the 'hits'.256 include_scores (bool, optional): Include scores alongside documents.257 Defaults to False.258 259 Returns:260 Union[List[Document], List[Tuple[Document, float]]]: The documents or261 document score pairs if `include_scores` is true.262 """263 documents: List[Tuple[Document, Any]] = []264 for res in results["hits"]:265 if self.page_content_builder is None:266 text = res["text"]267 else:268 text = self.page_content_builder(res)269 270 metadata = json.loads(res.get("metadata", "{}"))271 documents.append(272 (273 Document(page_content=text, metadata=metadata),274 res["_score"],275 )276 )277 return documents278 279 def _construct_documents_from_results_without_score(280 self, results: Dict[str, List[Dict[str, str]]]281 ) -> List[Document]:282 """Helper to convert Marqo results into documents.283 284 Args:285 results (List[dict]): A marqo results object with the 'hits'.286 include_scores (bool, optional): Include scores alongside documents.287 Defaults to False.288 289 Returns:290 Union[List[Document], List[Tuple[Document, float]]]: The documents or291 document score pairs if `include_scores` is true.292 """293 documents: List[Document] = []294 for res in results["hits"]:295 if self.page_content_builder is None:296 text = res["text"]297 else:298 text = self.page_content_builder(res)299 300 metadata = json.loads(res.get("metadata", "{}"))301 documents.append(Document(page_content=text, metadata=metadata))302 return documents303 304 def marqo_similarity_search(305 self,306 query: Union[str, Dict[str, float]],307 k: int = 4,308 ) -> Dict[str, List[Dict[str, str]]]:309 """Return documents from Marqo exposing Marqo's output directly310 311 Args:312 query (str): The query to search with.313 k (int, optional): The number of documents to return. Defaults to 4.314 315 Returns:316 List[Dict[str, Any]]: This hits from marqo.317 """318 results = self._client.index(self._index_name).search(319 q=query, searchable_attributes=self._searchable_attributes, limit=k320 )321 return results322 323 def marqo_bulk_similarity_search(324 self, queries: Iterable[Union[str, Dict[str, float]]], k: int = 4325 ) -> Dict[str, List[Dict[str, List[Dict[str, str]]]]]:326 """Return documents from Marqo using a bulk search, exposes Marqo's327 output directly328 329 Args:330 queries (Iterable[Union[str, Dict[str, float]]]): A list of queries.331 k (int, optional): The number of documents to return for each query.332 Defaults to 4.333 334 Returns:335 Dict[str, Dict[List[Dict[str, Dict[str, Any]]]]]: A bulk search results336 object337 """338 bulk_results = {339 "result": [340 self._client.index(self._index_name).search(341 q=query, searchable_attributes=self._searchable_attributes, limit=k342 )343 for query in queries344 ]345 }346 347 return bulk_results348 349 @classmethod350 def from_documents(351 cls: Type[Marqo],352 documents: List[Document],353 embedding: Union[Embeddings, None] = None,354 **kwargs: Any,355 ) -> Marqo:356 """Return VectorStore initialized from documents. Note that Marqo does not357 need embeddings, we retain the parameter to adhere to the Liskov substitution358 principle.359 360 361 Args:362 documents (List[Document]): Input documents363 embedding (Any, optional): Embeddings (not required). Defaults to None.364 365 Returns:366 VectorStore: A Marqo vectorstore367 """368 texts = [d.page_content for d in documents]369 metadatas = [d.metadata for d in documents]370 return cls.from_texts(texts, metadatas=metadatas, **kwargs)371 372 @classmethod373 def from_texts(374 cls,375 texts: List[str],376 embedding: Any = None,377 metadatas: Optional[List[dict]] = None,378 index_name: str = "",379 url: str = "http://localhost:8882",380 api_key: str = "",381 add_documents_settings: Optional[Dict[str, Any]] = None,382 searchable_attributes: Optional[List[str]] = None,383 page_content_builder: Optional[Callable[[Dict[str, str]], str]] = None,384 index_settings: Optional[Dict[str, Any]] = None,385 verbose: bool = True,386 **kwargs: Any,387 ) -> Marqo:388 """Return Marqo initialized from texts. Note that Marqo does not need389 embeddings, we retain the parameter to adhere to the Liskov390 substitution principle.391 392 This is a quick way to get started with marqo - simply provide your texts and393 metadatas and this will create an instance of the data store and index the394 provided data.395 396 To know the ids of your documents with this approach you will need to include397 them in under the key "_id" in your metadatas for each text398 399 Example:400 .. code-block:: python401 402 from langchain_community.vectorstores import Marqo403 404 datastore = Marqo(texts=['text'], index_name='my-first-index',405 url='http://localhost:8882')406 407 Args:408 texts (List[str]): A list of texts to index into marqo upon creation.409 embedding (Any, optional): Embeddings (not required). Defaults to None.410 index_name (str, optional): The name of the index to use, if none is411 provided then one will be created with a UUID. Defaults to None.412 url (str, optional): The URL for Marqo. Defaults to "http://localhost:8882".413 api_key (str, optional): The API key for Marqo. Defaults to "".414 metadatas (Optional[List[dict]], optional): A list of metadatas, to415 accompany the texts. Defaults to None.416 this is only used when a new index is being created. Defaults to "cpu". Can417 be "cpu" or "cuda".418 add_documents_settings (Optional[Dict[str, Any]], optional): Settings419 for adding documents, see420 https://docs.marqo.ai/0.0.16/API-Reference/documents/#query-parameters.421 Defaults to {}.422 index_settings (Optional[Dict[str, Any]], optional): Index settings if423 the index doesn't exist, see424 https://docs.marqo.ai/0.0.16/API-Reference/indexes/#index-defaults-object.425 Defaults to {}.426 427 Returns:428 Marqo: An instance of the Marqo vector store429 """430 try:431 import marqo432 except ImportError:433 raise ImportError(434 "Could not import marqo python package. "435 "Please install it with `pip install marqo`."436 )437 438 if not index_name:439 index_name = str(uuid.uuid4())440 441 client = marqo.Client(url=url, api_key=api_key)442 443 try:444 client.create_index(index_name, settings_dict=index_settings or {})445 if verbose:446 print(f"Created {index_name} successfully.") # noqa: T201447 except Exception:448 if verbose:449 print(f"Index {index_name} exists.") # noqa: T201450 451 instance: Marqo = cls(452 client,453 index_name,454 searchable_attributes=searchable_attributes,455 add_documents_settings=add_documents_settings or {},456 page_content_builder=page_content_builder,457 )458 instance.add_texts(texts, metadatas)459 return instance460 461 def get_indexes(self) -> List[Dict[str, str]]:462 """Helper to see your available indexes in marqo, useful if the463 from_texts method was used without an index name specified464 465 Returns:466 List[Dict[str, str]]: The list of indexes467 """468 return self._client.get_indexes()["results"]469 470 def get_number_of_documents(self) -> int:471 """Helper to see the number of documents in the index472 473 Returns:474 int: The number of documents475 """476 return self._client.index(self._index_name).get_stats()["numberOfDocuments"]477 