codekingpro/portable-devtools
114k
1"""Wrapper around TileDB vector database."""2 3from __future__ import annotations4 5import pickle6import random7import sys8from typing import Any, Dict, Iterable, List, Mapping, Optional, Tuple9 10import numpy as np11from langchain_core.documents import Document12from langchain_core.embeddings import Embeddings13from langchain_core.utils import guard_import14from langchain_core.vectorstores import VectorStore15 16from langchain_community.vectorstores.utils import maximal_marginal_relevance17 18INDEX_METRICS = frozenset(["euclidean"])19DEFAULT_METRIC = "euclidean"20DOCUMENTS_ARRAY_NAME = "documents"21VECTOR_INDEX_NAME = "vectors"22MAX_UINT64 = np.iinfo(np.dtype("uint64")).max23MAX_FLOAT_32 = np.finfo(np.dtype("float32")).max24MAX_FLOAT = sys.float_info.max25 26 27def dependable_tiledb_import() -> Any:28 """Import tiledb-vector-search if available, otherwise raise error."""29 return (30 guard_import("tiledb.vector_search"),31 guard_import("tiledb"),32 )33 34 35def get_vector_index_uri_from_group(group: Any) -> str:36 """Get the URI of the vector index."""37 return group[VECTOR_INDEX_NAME].uri38 39 40def get_documents_array_uri_from_group(group: Any) -> str:41 """Get the URI of the documents array from group.42 43 Args:44 group: TileDB group object.45 46 Returns:47 URI of the documents array.48 """49 return group[DOCUMENTS_ARRAY_NAME].uri50 51 52def get_vector_index_uri(uri: str) -> str:53 """Get the URI of the vector index."""54 return f"{uri}/{VECTOR_INDEX_NAME}"55 56 57def get_documents_array_uri(uri: str) -> str:58 """Get the URI of the documents array."""59 return f"{uri}/{DOCUMENTS_ARRAY_NAME}"60 61 62class TileDB(VectorStore):63 """TileDB vector store.64 65 To use, you should have the ``tiledb-vector-search`` python package installed.66 67 Example:68 .. code-block:: python69 70 from langchain_community import TileDB71 embeddings = OpenAIEmbeddings()72 db = TileDB(embeddings, index_uri, metric)73 74 """75 76 def __init__(77 self,78 embedding: Embeddings,79 index_uri: str,80 metric: str,81 *,82 vector_index_uri: str = "",83 docs_array_uri: str = "",84 config: Optional[Mapping[str, Any]] = None,85 timestamp: Any = None,86 allow_dangerous_deserialization: bool = False,87 **kwargs: Any,88 ):89 """Initialize with necessary components.90 91 Args:92 allow_dangerous_deserialization: whether to allow deserialization93 of the data which involves loading data using pickle.94 data can be modified by malicious actors to deliver a95 malicious payload that results in execution of96 arbitrary code on your machine.97 """98 if not allow_dangerous_deserialization:99 raise ValueError(100 "TileDB relies on pickle for serialization and deserialization. "101 "This can be dangerous if the data is intercepted and/or modified "102 "by malicious actors prior to being de-serialized. "103 "If you are sure that the data is safe from modification, you can "104 " set allow_dangerous_deserialization=True to proceed. "105 "Loading of compromised data using pickle can result in execution of "106 "arbitrary code on your machine."107 )108 self.embedding = embedding109 self.embedding_function = embedding.embed_query110 self.index_uri = index_uri111 self.metric = metric112 self.config = config113 114 tiledb_vs, tiledb = (115 guard_import("tiledb.vector_search"),116 guard_import("tiledb"),117 )118 with tiledb.scope_ctx(ctx_or_config=config):119 index_group = tiledb.Group(self.index_uri, "r")120 self.vector_index_uri = (121 vector_index_uri122 if vector_index_uri != ""123 else get_vector_index_uri_from_group(index_group)124 )125 self.docs_array_uri = (126 docs_array_uri127 if docs_array_uri != ""128 else get_documents_array_uri_from_group(index_group)129 )130 index_group.close()131 group = tiledb.Group(self.vector_index_uri, "r")132 self.index_type = group.meta.get("index_type")133 group.close()134 self.timestamp = timestamp135 if self.index_type == "FLAT":136 self.vector_index = tiledb_vs.flat_index.FlatIndex(137 uri=self.vector_index_uri,138 config=self.config,139 timestamp=self.timestamp,140 **kwargs,141 )142 elif self.index_type == "IVF_FLAT":143 self.vector_index = tiledb_vs.ivf_flat_index.IVFFlatIndex(144 uri=self.vector_index_uri,145 config=self.config,146 timestamp=self.timestamp,147 **kwargs,148 )149 150 @property151 def embeddings(self) -> Optional[Embeddings]:152 return self.embedding153 154 def process_index_results(155 self,156 ids: List[int],157 scores: List[float],158 *,159 k: int = 4,160 filter: Optional[Dict[str, Any]] = None,161 score_threshold: float = MAX_FLOAT,162 ) -> List[Tuple[Document, float]]:163 """Turns TileDB results into a list of documents and scores.164 165 Args:166 ids: List of indices of the documents in the index.167 scores: List of distances of the documents in the index.168 k: Number of Documents to return. Defaults to 4.169 filter (Optional[Dict[str, Any]]): Filter by metadata. Defaults to None.170 score_threshold: Optional, a floating point value to filter the171 resulting set of retrieved docs172 Returns:173 List of Documents and scores.174 """175 tiledb = guard_import("tiledb")176 docs = []177 docs_array = tiledb.open(178 self.docs_array_uri, "r", timestamp=self.timestamp, config=self.config179 )180 for idx, score in zip(ids, scores):181 if idx == 0 and score == 0:182 continue183 if idx == MAX_UINT64 and score == MAX_FLOAT_32:184 continue185 doc = docs_array[idx]186 if doc is None or len(doc["text"]) == 0:187 raise ValueError(f"Could not find document for id {idx}, got {doc}")188 pickled_metadata = doc.get("metadata")189 result_doc = Document(page_content=str(doc["text"][0]))190 if pickled_metadata is not None:191 metadata = pickle.loads( # ignore[pickle]: explicit-opt-in192 np.array(pickled_metadata.tolist()).astype(np.uint8).tobytes()193 )194 result_doc.metadata = metadata195 if filter is not None:196 filter = {197 key: [value] if not isinstance(value, list) else value198 for key, value in filter.items()199 }200 if all(201 result_doc.metadata.get(key) in value202 for key, value in filter.items()203 ):204 docs.append((result_doc, score))205 else:206 docs.append((result_doc, score))207 docs_array.close()208 docs = [(doc, score) for doc, score in docs if score <= score_threshold]209 return docs[:k]210 211 def similarity_search_with_score_by_vector(212 self,213 embedding: List[float],214 *,215 k: int = 4,216 filter: Optional[Dict[str, Any]] = None,217 fetch_k: int = 20,218 **kwargs: Any,219 ) -> List[Tuple[Document, float]]:220 """Return docs most similar to query.221 222 Args:223 embedding: Embedding vector to look up documents similar to.224 k: Number of Documents to return. Defaults to 4.225 filter (Optional[Dict[str, Any]]): Filter by metadata. Defaults to None.226 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.227 Defaults to 20.228 **kwargs: kwargs to be passed to similarity search. Can include:229 nprobe: Optional, number of partitions to check if using IVF_FLAT index230 score_threshold: Optional, a floating point value to filter the231 resulting set of retrieved docs232 233 Returns:234 List of documents most similar to the query text and distance235 in float for each. Lower score represents more similarity.236 """237 if "score_threshold" in kwargs:238 score_threshold = kwargs.pop("score_threshold")239 else:240 score_threshold = MAX_FLOAT241 d, i = self.vector_index.query(242 np.array([np.array(embedding).astype(np.float32)]).astype(np.float32),243 k=k if filter is None else fetch_k,244 **kwargs,245 )246 return self.process_index_results(247 ids=i[0], scores=d[0], filter=filter, k=k, score_threshold=score_threshold248 )249 250 def similarity_search_with_score(251 self,252 query: str,253 *,254 k: int = 4,255 filter: Optional[Dict[str, Any]] = None,256 fetch_k: int = 20,257 **kwargs: Any,258 ) -> List[Tuple[Document, float]]:259 """Return docs most similar to query.260 261 Args:262 query: Text to look up documents similar to.263 k: Number of Documents to return. Defaults to 4.264 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.265 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.266 Defaults to 20.267 268 Returns:269 List of documents most similar to the query text with270 Distance as float. Lower score represents more similarity.271 """272 embedding = self.embedding_function(query)273 docs = self.similarity_search_with_score_by_vector(274 embedding,275 k=k,276 filter=filter,277 fetch_k=fetch_k,278 **kwargs,279 )280 return docs281 282 def similarity_search_by_vector(283 self,284 embedding: List[float],285 k: int = 4,286 filter: Optional[Dict[str, Any]] = None,287 fetch_k: int = 20,288 **kwargs: Any,289 ) -> List[Document]:290 """Return docs most similar to embedding vector.291 292 Args:293 embedding: Embedding to look up documents similar to.294 k: Number of Documents to return. Defaults to 4.295 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.296 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.297 Defaults to 20.298 299 Returns:300 List of Documents most similar to the embedding.301 """302 docs_and_scores = self.similarity_search_with_score_by_vector(303 embedding,304 k=k,305 filter=filter,306 fetch_k=fetch_k,307 **kwargs,308 )309 return [doc for doc, _ in docs_and_scores]310 311 def similarity_search(312 self,313 query: str,314 k: int = 4,315 filter: Optional[Dict[str, Any]] = None,316 fetch_k: int = 20,317 **kwargs: Any,318 ) -> List[Document]:319 """Return docs most similar to query.320 321 Args:322 query: Text to look up documents similar to.323 k: Number of Documents to return. Defaults to 4.324 filter: (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.325 fetch_k: (Optional[int]) Number of Documents to fetch before filtering.326 Defaults to 20.327 328 Returns:329 List of Documents most similar to the query.330 """331 docs_and_scores = self.similarity_search_with_score(332 query, k=k, filter=filter, fetch_k=fetch_k, **kwargs333 )334 return [doc for doc, _ in docs_and_scores]335 336 def max_marginal_relevance_search_with_score_by_vector(337 self,338 embedding: List[float],339 *,340 k: int = 4,341 fetch_k: int = 20,342 lambda_mult: float = 0.5,343 filter: Optional[Dict[str, Any]] = None,344 **kwargs: Any,345 ) -> List[Tuple[Document, float]]:346 """Return docs and their similarity scores selected using the maximal marginal347 relevance.348 349 Maximal marginal relevance optimizes for similarity to query AND diversity350 among selected documents.351 352 Args:353 embedding: Embedding to look up documents similar to.354 k: Number of Documents to return. Defaults to 4.355 fetch_k: Number of Documents to fetch before filtering to356 pass to MMR algorithm.357 lambda_mult: Number between 0 and 1 that determines the degree358 of diversity among the results with 0 corresponding359 to maximum diversity and 1 to minimum diversity.360 Defaults to 0.5.361 Returns:362 List of Documents and similarity scores selected by maximal marginal363 relevance and score for each.364 """365 if "score_threshold" in kwargs:366 score_threshold = kwargs.pop("score_threshold")367 else:368 score_threshold = MAX_FLOAT369 scores, indices = self.vector_index.query(370 np.array([np.array(embedding).astype(np.float32)]).astype(np.float32),371 k=fetch_k if filter is None else fetch_k * 2,372 **kwargs,373 )374 results = self.process_index_results(375 ids=indices[0],376 scores=scores[0],377 filter=filter,378 k=fetch_k if filter is None else fetch_k * 2,379 score_threshold=score_threshold,380 )381 embeddings = [382 self.embedding.embed_documents([doc.page_content])[0] for doc, _ in results383 ]384 mmr_selected = maximal_marginal_relevance(385 np.array([embedding], dtype=np.float32),386 embeddings,387 k=k,388 lambda_mult=lambda_mult,389 )390 docs_and_scores = []391 for i in mmr_selected:392 docs_and_scores.append(results[i])393 return docs_and_scores394 395 def max_marginal_relevance_search_by_vector(396 self,397 embedding: List[float],398 k: int = 4,399 fetch_k: int = 20,400 lambda_mult: float = 0.5,401 filter: Optional[Dict[str, Any]] = None,402 **kwargs: Any,403 ) -> List[Document]:404 """Return docs selected using the maximal marginal relevance.405 406 Maximal marginal relevance optimizes for similarity to query AND diversity407 among selected documents.408 409 Args:410 embedding: Embedding to look up documents similar to.411 k: Number of Documents to return. Defaults to 4.412 fetch_k: Number of Documents to fetch before filtering to413 pass to MMR algorithm.414 lambda_mult: Number between 0 and 1 that determines the degree415 of diversity among the results with 0 corresponding416 to maximum diversity and 1 to minimum diversity.417 Defaults to 0.5.418 Returns:419 List of Documents selected by maximal marginal relevance.420 """421 docs_and_scores = self.max_marginal_relevance_search_with_score_by_vector(422 embedding,423 k=k,424 fetch_k=fetch_k,425 lambda_mult=lambda_mult,426 filter=filter,427 **kwargs,428 )429 return [doc for doc, _ in docs_and_scores]430 431 def max_marginal_relevance_search(432 self,433 query: str,434 k: int = 4,435 fetch_k: int = 20,436 lambda_mult: float = 0.5,437 filter: Optional[Dict[str, Any]] = None,438 **kwargs: Any,439 ) -> List[Document]:440 """Return docs selected using the maximal marginal relevance.441 442 Maximal marginal relevance optimizes for similarity to query AND diversity443 among selected documents.444 445 Args:446 query: Text to look up documents similar to.447 k: Number of Documents to return. Defaults to 4.448 fetch_k: Number of Documents to fetch before filtering (if needed) to449 pass to MMR algorithm.450 lambda_mult: Number between 0 and 1 that determines the degree451 of diversity among the results with 0 corresponding452 to maximum diversity and 1 to minimum diversity.453 Defaults to 0.5.454 Returns:455 List of Documents selected by maximal marginal relevance.456 """457 embedding = self.embedding_function(query)458 docs = self.max_marginal_relevance_search_by_vector(459 embedding,460 k=k,461 fetch_k=fetch_k,462 lambda_mult=lambda_mult,463 filter=filter,464 **kwargs,465 )466 return docs467 468 @classmethod469 def create(470 cls,471 index_uri: str,472 index_type: str,473 dimensions: int,474 vector_type: np.dtype,475 *,476 metadatas: bool = True,477 config: Optional[Mapping[str, Any]] = None,478 ) -> None:479 tiledb_vs, tiledb = (480 guard_import("tiledb.vector_search"),481 guard_import("tiledb"),482 )483 with tiledb.scope_ctx(ctx_or_config=config):484 try:485 tiledb.group_create(index_uri)486 except tiledb.TileDBError as err:487 raise err488 group = tiledb.Group(index_uri, "w")489 vector_index_uri = get_vector_index_uri(group.uri)490 docs_uri = get_documents_array_uri(group.uri)491 if index_type == "FLAT":492 tiledb_vs.flat_index.create(493 uri=vector_index_uri,494 dimensions=dimensions,495 vector_type=vector_type,496 config=config,497 )498 elif index_type == "IVF_FLAT":499 tiledb_vs.ivf_flat_index.create(500 uri=vector_index_uri,501 dimensions=dimensions,502 vector_type=vector_type,503 config=config,504 )505 group.add(vector_index_uri, name=VECTOR_INDEX_NAME)506 507 # Create TileDB array to store Documents508 # TODO add a Document store API to tiledb-vector-search to allow storing509 # different types of objects and metadata in a more generic way.510 dim = tiledb.Dim(511 name="id",512 domain=(0, MAX_UINT64 - 1),513 dtype=np.dtype(np.uint64),514 )515 dom = tiledb.Domain(dim)516 517 text_attr = tiledb.Attr(name="text", dtype=np.dtype("U1"), var=True)518 attrs = [text_attr]519 if metadatas:520 metadata_attr = tiledb.Attr(name="metadata", dtype=np.uint8, var=True)521 attrs.append(metadata_attr)522 schema = tiledb.ArraySchema(523 domain=dom,524 sparse=True,525 allows_duplicates=False,526 attrs=attrs,527 )528 tiledb.Array.create(docs_uri, schema)529 group.add(docs_uri, name=DOCUMENTS_ARRAY_NAME)530 group.close()531 532 @classmethod533 def __from(534 cls,535 texts: List[str],536 embeddings: List[List[float]],537 embedding: Embeddings,538 index_uri: str,539 *,540 metadatas: Optional[List[dict]] = None,541 ids: Optional[List[str]] = None,542 metric: str = DEFAULT_METRIC,543 index_type: str = "FLAT",544 config: Optional[Mapping[str, Any]] = None,545 index_timestamp: int = 0,546 **kwargs: Any,547 ) -> TileDB:548 if metric not in INDEX_METRICS:549 raise ValueError(550 (551 f"Unsupported distance metric: {metric}. "552 f"Expected one of {list(INDEX_METRICS)}"553 )554 )555 tiledb_vs, tiledb = (556 guard_import("tiledb.vector_search"),557 guard_import("tiledb"),558 )559 input_vectors = np.array(embeddings).astype(np.float32)560 cls.create(561 index_uri=index_uri,562 index_type=index_type,563 dimensions=input_vectors.shape[1],564 vector_type=input_vectors.dtype,565 metadatas=metadatas is not None,566 config=config,567 )568 with tiledb.scope_ctx(ctx_or_config=config):569 if not embeddings:570 raise ValueError("embeddings must be provided to build a TileDB index")571 572 vector_index_uri = get_vector_index_uri(index_uri)573 docs_uri = get_documents_array_uri(index_uri)574 if ids is None:575 ids = [str(random.randint(0, MAX_UINT64 - 1)) for _ in texts]576 external_ids = np.array(ids).astype(np.uint64)577 578 tiledb_vs.ingestion.ingest(579 index_type=index_type,580 index_uri=vector_index_uri,581 input_vectors=input_vectors,582 external_ids=external_ids,583 index_timestamp=index_timestamp if index_timestamp != 0 else None,584 config=config,585 **kwargs,586 )587 with tiledb.open(docs_uri, "w") as A:588 if external_ids is None:589 external_ids = np.zeros(len(texts), dtype=np.uint64)590 for i in range(len(texts)):591 external_ids[i] = i592 data = {}593 data["text"] = np.array(texts)594 if metadatas is not None:595 metadata_attr = np.empty([len(metadatas)], dtype=object)596 i = 0597 for metadata in metadatas:598 metadata_attr[i] = np.frombuffer(599 pickle.dumps(metadata), dtype=np.uint8600 )601 i += 1602 data["metadata"] = metadata_attr603 604 A[external_ids] = data605 return cls(606 embedding=embedding,607 index_uri=index_uri,608 metric=metric,609 config=config,610 **kwargs,611 )612 613 def delete(614 self, ids: Optional[List[str]] = None, timestamp: int = 0, **kwargs: Any615 ) -> Optional[bool]:616 """Delete by vector ID or other criteria.617 618 Args:619 ids: List of ids to delete.620 timestamp: Optional timestamp to delete with.621 **kwargs: Other keyword arguments that subclasses might use.622 623 Returns:624 Optional[bool]: True if deletion is successful,625 False otherwise, None if not implemented.626 """627 628 external_ids = np.array(ids).astype(np.uint64)629 self.vector_index.delete_batch(630 external_ids=external_ids, timestamp=timestamp if timestamp != 0 else None631 )632 return True633 634 def add_texts(635 self,636 texts: Iterable[str],637 metadatas: Optional[List[dict]] = None,638 ids: Optional[List[str]] = None,639 timestamp: int = 0,640 **kwargs: Any,641 ) -> List[str]:642 """Run more texts through the embeddings and add to the vectorstore.643 644 Args:645 texts: Iterable of strings to add to the vectorstore.646 metadatas: Optional list of metadatas associated with the texts.647 ids: Optional ids of each text object.648 timestamp: Optional timestamp to write new texts with.649 kwargs: vectorstore specific parameters650 651 Returns:652 List of ids from adding the texts into the vectorstore.653 """654 tiledb = guard_import("tiledb")655 embeddings = self.embedding.embed_documents(list(texts))656 if ids is None:657 ids = [str(random.randint(0, MAX_UINT64 - 1)) for _ in texts]658 659 external_ids = np.array(ids).astype(np.uint64)660 vectors = np.empty((len(embeddings)), dtype="O")661 for i in range(len(embeddings)):662 vectors[i] = np.array(embeddings[i], dtype=np.float32)663 self.vector_index.update_batch(664 vectors=vectors,665 external_ids=external_ids,666 timestamp=timestamp if timestamp != 0 else None,667 )668 669 docs = {}670 docs["text"] = np.array(texts)671 if metadatas is not None:672 metadata_attr = np.empty([len(metadatas)], dtype=object)673 i = 0674 for metadata in metadatas:675 metadata_attr[i] = np.frombuffer(pickle.dumps(metadata), dtype=np.uint8)676 i += 1677 docs["metadata"] = metadata_attr678 679 docs_array = tiledb.open(680 self.docs_array_uri,681 "w",682 timestamp=timestamp if timestamp != 0 else None,683 config=self.config,684 )685 docs_array[external_ids] = docs686 docs_array.close()687 return ids688 689 @classmethod690 def from_texts(691 cls,692 texts: List[str],693 embedding: Embeddings,694 metadatas: Optional[List[dict]] = None,695 ids: Optional[List[str]] = None,696 metric: str = DEFAULT_METRIC,697 index_uri: str = "/tmp/tiledb_array",698 index_type: str = "FLAT",699 config: Optional[Mapping[str, Any]] = None,700 index_timestamp: int = 0,701 **kwargs: Any,702 ) -> TileDB:703 """Construct a TileDB index from raw documents.704 705 Args:706 texts: List of documents to index.707 embedding: Embedding function to use.708 metadatas: List of metadata dictionaries to associate with documents.709 ids: Optional ids of each text object.710 metric: Metric to use for indexing. Defaults to "euclidean".711 index_uri: The URI to write the TileDB arrays712 index_type: Optional, Vector index type ("FLAT", IVF_FLAT")713 config: Optional, TileDB config714 index_timestamp: Optional, timestamp to write new texts with.715 716 Example:717 .. code-block:: python718 719 from langchain_community import TileDB720 from langchain_community.embeddings import OpenAIEmbeddings721 embeddings = OpenAIEmbeddings()722 index = TileDB.from_texts(texts, embeddings)723 """724 embeddings = []725 embeddings = embedding.embed_documents(texts)726 return cls.__from(727 texts=texts,728 embeddings=embeddings,729 embedding=embedding,730 metadatas=metadatas,731 ids=ids,732 metric=metric,733 index_uri=index_uri,734 index_type=index_type,735 config=config,736 index_timestamp=index_timestamp,737 **kwargs,738 )739 740 @classmethod741 def from_embeddings(742 cls,743 text_embeddings: List[Tuple[str, List[float]]],744 embedding: Embeddings,745 index_uri: str,746 *,747 metadatas: Optional[List[dict]] = None,748 ids: Optional[List[str]] = None,749 metric: str = DEFAULT_METRIC,750 index_type: str = "FLAT",751 config: Optional[Mapping[str, Any]] = None,752 index_timestamp: int = 0,753 **kwargs: Any,754 ) -> TileDB:755 """Construct TileDB index from embeddings.756 757 Args:758 text_embeddings: List of tuples of (text, embedding)759 embedding: Embedding function to use.760 index_uri: The URI to write the TileDB arrays761 metadatas: List of metadata dictionaries to associate with documents.762 metric: Optional, Metric to use for indexing. Defaults to "euclidean".763 index_type: Optional, Vector index type ("FLAT", IVF_FLAT")764 config: Optional, TileDB config765 index_timestamp: Optional, timestamp to write new texts with.766 767 Example:768 .. code-block:: python769 770 from langchain_community import TileDB771 from langchain_community.embeddings import OpenAIEmbeddings772 embeddings = OpenAIEmbeddings()773 text_embeddings = embeddings.embed_documents(texts)774 text_embedding_pairs = list(zip(texts, text_embeddings))775 db = TileDB.from_embeddings(text_embedding_pairs, embeddings)776 """777 texts = [t[0] for t in text_embeddings]778 embeddings = [t[1] for t in text_embeddings]779 780 return cls.__from(781 texts=texts,782 embeddings=embeddings,783 embedding=embedding,784 metadatas=metadatas,785 ids=ids,786 metric=metric,787 index_uri=index_uri,788 index_type=index_type,789 config=config,790 index_timestamp=index_timestamp,791 **kwargs,792 )793 794 @classmethod795 def load(796 cls,797 index_uri: str,798 embedding: Embeddings,799 *,800 metric: str = DEFAULT_METRIC,801 config: Optional[Mapping[str, Any]] = None,802 timestamp: Any = None,803 **kwargs: Any,804 ) -> TileDB:805 """Load a TileDB index from a URI.806 807 Args:808 index_uri: The URI of the TileDB vector index.809 embedding: Embeddings to use when generating queries.810 metric: Optional, Metric to use for indexing. Defaults to "euclidean".811 config: Optional, TileDB config812 timestamp: Optional, timestamp to use for opening the arrays.813 """814 return cls(815 embedding=embedding,816 index_uri=index_uri,817 metric=metric,818 config=config,819 timestamp=timestamp,820 **kwargs,821 )822 823 def consolidate_updates(self, **kwargs: Any) -> None:824 self.vector_index = self.vector_index.consolidate_updates(**kwargs)825 