Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
tiledb.py825 linesDownload Raw Back to vectorstores
1"""Wrapper around TileDB vector database."""2 3from __future__ import annotations4 5import pickle6import random7import sys8from typing import Any, Dict, Iterable, List, Mapping, Optional, Tuple9 10import numpy as np11from langchain_core.documents import Document12from langchain_core.embeddings import Embeddings13from langchain_core.utils import guard_import14from langchain_core.vectorstores import VectorStore15 16from langchain_community.vectorstores.utils import maximal_marginal_relevance17 18INDEX_METRICS = frozenset(["euclidean"])19DEFAULT_METRIC = "euclidean"20DOCUMENTS_ARRAY_NAME = "documents"21VECTOR_INDEX_NAME = "vectors"22MAX_UINT64 = np.iinfo(np.dtype("uint64")).max23MAX_FLOAT_32 = np.finfo(np.dtype("float32")).max24MAX_FLOAT = sys.float_info.max25 26 27def dependable_tiledb_import() -> Any:28    """Import tiledb-vector-search if available, otherwise raise error."""29    return (30        guard_import("tiledb.vector_search"),31        guard_import("tiledb"),32    )33 34 35def get_vector_index_uri_from_group(group: Any) -> str:36    """Get the URI of the vector index."""37    return group[VECTOR_INDEX_NAME].uri38 39 40def get_documents_array_uri_from_group(group: Any) -> str:41    """Get the URI of the documents array from group.42 43    Args:44        group: TileDB group object.45 46    Returns:47        URI of the documents array.48    """49    return group[DOCUMENTS_ARRAY_NAME].uri50 51 52def get_vector_index_uri(uri: str) -> str:53    """Get the URI of the vector index."""54    return f"{uri}/{VECTOR_INDEX_NAME}"55 56 57def get_documents_array_uri(uri: str) -> str:58    """Get the URI of the documents array."""59    return f"{uri}/{DOCUMENTS_ARRAY_NAME}"60 61 62class TileDB(VectorStore):63    """TileDB vector store.64 65    To use, you should have the ``tiledb-vector-search`` python package installed.66 67    Example:68        .. code-block:: python69 70            from langchain_community import TileDB71            embeddings = OpenAIEmbeddings()72            db = TileDB(embeddings, index_uri, metric)73 74    """75 76    def __init__(77        self,78        embedding: Embeddings,79        index_uri: str,80        metric: str,81        *,82        vector_index_uri: str = "",83        docs_array_uri: str = "",84        config: Optional[Mapping[str, Any]] = None,85        timestamp: Any = None,86        allow_dangerous_deserialization: bool = False,87        **kwargs: Any,88    ):89        """Initialize with necessary components.90 91        Args:92            allow_dangerous_deserialization: whether to allow deserialization93                of the data which involves loading data using pickle.94                data can be modified by malicious actors to deliver a95                malicious payload that results in execution of96                arbitrary code on your machine.97        """98        if not allow_dangerous_deserialization:99            raise ValueError(100                "TileDB relies on pickle for serialization and deserialization. "101                "This can be dangerous if the data is intercepted and/or modified "102                "by malicious actors prior to being de-serialized. "103                "If you are sure that the data is safe from modification, you can "104                " set allow_dangerous_deserialization=True to proceed. "105                "Loading of compromised data using pickle can result in execution of "106                "arbitrary code on your machine."107            )108        self.embedding = embedding109        self.embedding_function = embedding.embed_query110        self.index_uri = index_uri111        self.metric = metric112        self.config = config113 114        tiledb_vs, tiledb = (115            guard_import("tiledb.vector_search"),116            guard_import("tiledb"),117        )118        with tiledb.scope_ctx(ctx_or_config=config):119            index_group = tiledb.Group(self.index_uri, "r")120            self.vector_index_uri = (121                vector_index_uri122                if vector_index_uri != ""123                else get_vector_index_uri_from_group(index_group)124            )125            self.docs_array_uri = (126                docs_array_uri127                if docs_array_uri != ""128                else get_documents_array_uri_from_group(index_group)129            )130            index_group.close()131            group = tiledb.Group(self.vector_index_uri, "r")132            self.index_type = group.meta.get("index_type")133            group.close()134            self.timestamp = timestamp135            if self.index_type == "FLAT":136                self.vector_index = tiledb_vs.flat_index.FlatIndex(137                    uri=self.vector_index_uri,138                    config=self.config,139                    timestamp=self.timestamp,140                    **kwargs,141                )142            elif self.index_type == "IVF_FLAT":143                self.vector_index = tiledb_vs.ivf_flat_index.IVFFlatIndex(144                    uri=self.vector_index_uri,145                    config=self.config,146                    timestamp=self.timestamp,147                    **kwargs,148                )149 150    @property151    def embeddings(self) -> Optional[Embeddings]:152        return self.embedding153 154    def process_index_results(155        self,156        ids: List[int],157        scores: List[float],158        *,159        k: int = 4,160        filter: Optional[Dict[str, Any]] = None,161        score_threshold: float = MAX_FLOAT,162    ) -> List[Tuple[Document, float]]:163        """Turns TileDB results into a list of documents and scores.164 165        Args:166            ids: List of indices of the documents in the index.167            scores: List of distances of the documents in the index.168            k: Number of Documents to return. Defaults to 4.169            filter (Optional[Dict[str, Any]]): Filter by metadata. Defaults to None.170            score_threshold: Optional, a floating point value to filter the171                resulting set of retrieved docs172        Returns:173            List of Documents and scores.174        """175        tiledb = guard_import("tiledb")176        docs = []177        docs_array = tiledb.open(178            self.docs_array_uri, "r", timestamp=self.timestamp, config=self.config179        )180        for idx, score in zip(ids, scores):181            if idx == 0 and score == 0:182                continue183            if idx == MAX_UINT64 and score == MAX_FLOAT_32:184                continue185            doc = docs_array[idx]186            if doc is None or len(doc["text"]) == 0:187                raise ValueError(f"Could not find document for id {idx}, got {doc}")188            pickled_metadata = doc.get("metadata")189            result_doc = Document(page_content=str(doc["text"][0]))190            if pickled_metadata is not None:191                metadata = pickle.loads(  # ignore[pickle]: explicit-opt-in192                    np.array(pickled_metadata.tolist()).astype(np.uint8).tobytes()193                )194                result_doc.metadata = metadata195            if filter is not None:196                filter = {197                    key: [value] if not isinstance(value, list) else value198                    for key, value in filter.items()199                }200                if all(201                    result_doc.metadata.get(key) in value202                    for key, value in filter.items()203                ):204                    docs.append((result_doc, score))205            else:206                docs.append((result_doc, score))207        docs_array.close()208        docs = [(doc, score) for doc, score in docs if score <= score_threshold]209        return docs[:k]210 211    def similarity_search_with_score_by_vector(212        self,213        embedding: List[float],214        *,215        k: int = 4,216        filter: Optional[Dict[str, Any]] = None,217        fetch_k: int = 20,218        **kwargs: Any,219    ) -> List[Tuple[Document, float]]:220        """Return docs most similar to query.221 222        Args:223            embedding: Embedding vector to look up documents similar to.224            k: Number of Documents to return. Defaults to 4.225            filter (Optional[Dict[str, Any]]): Filter by metadata. Defaults to None.226            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.227                      Defaults to 20.228            **kwargs: kwargs to be passed to similarity search. Can include:229                nprobe: Optional, number of partitions to check if using IVF_FLAT index230                score_threshold: Optional, a floating point value to filter the231                    resulting set of retrieved docs232 233        Returns:234            List of documents most similar to the query text and distance235            in float for each. Lower score represents more similarity.236        """237        if "score_threshold" in kwargs:238            score_threshold = kwargs.pop("score_threshold")239        else:240            score_threshold = MAX_FLOAT241        d, i = self.vector_index.query(242            np.array([np.array(embedding).astype(np.float32)]).astype(np.float32),243            k=k if filter is None else fetch_k,244            **kwargs,245        )246        return self.process_index_results(247            ids=i[0], scores=d[0], filter=filter, k=k, score_threshold=score_threshold248        )249 250    def similarity_search_with_score(251        self,252        query: str,253        *,254        k: int = 4,255        filter: Optional[Dict[str, Any]] = None,256        fetch_k: int = 20,257        **kwargs: Any,258    ) -> List[Tuple[Document, float]]:259        """Return docs most similar to query.260 261        Args:262            query: Text to look up documents similar to.263            k: Number of Documents to return. Defaults to 4.264            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.265            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.266                      Defaults to 20.267 268        Returns:269            List of documents most similar to the query text with270            Distance as float. Lower score represents more similarity.271        """272        embedding = self.embedding_function(query)273        docs = self.similarity_search_with_score_by_vector(274            embedding,275            k=k,276            filter=filter,277            fetch_k=fetch_k,278            **kwargs,279        )280        return docs281 282    def similarity_search_by_vector(283        self,284        embedding: List[float],285        k: int = 4,286        filter: Optional[Dict[str, Any]] = None,287        fetch_k: int = 20,288        **kwargs: Any,289    ) -> List[Document]:290        """Return docs most similar to embedding vector.291 292        Args:293            embedding: Embedding to look up documents similar to.294            k: Number of Documents to return. Defaults to 4.295            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.296            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.297                      Defaults to 20.298 299        Returns:300            List of Documents most similar to the embedding.301        """302        docs_and_scores = self.similarity_search_with_score_by_vector(303            embedding,304            k=k,305            filter=filter,306            fetch_k=fetch_k,307            **kwargs,308        )309        return [doc for doc, _ in docs_and_scores]310 311    def similarity_search(312        self,313        query: str,314        k: int = 4,315        filter: Optional[Dict[str, Any]] = None,316        fetch_k: int = 20,317        **kwargs: Any,318    ) -> List[Document]:319        """Return docs most similar to query.320 321        Args:322            query: Text to look up documents similar to.323            k: Number of Documents to return. Defaults to 4.324            filter: (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.325            fetch_k: (Optional[int]) Number of Documents to fetch before filtering.326                      Defaults to 20.327 328        Returns:329            List of Documents most similar to the query.330        """331        docs_and_scores = self.similarity_search_with_score(332            query, k=k, filter=filter, fetch_k=fetch_k, **kwargs333        )334        return [doc for doc, _ in docs_and_scores]335 336    def max_marginal_relevance_search_with_score_by_vector(337        self,338        embedding: List[float],339        *,340        k: int = 4,341        fetch_k: int = 20,342        lambda_mult: float = 0.5,343        filter: Optional[Dict[str, Any]] = None,344        **kwargs: Any,345    ) -> List[Tuple[Document, float]]:346        """Return docs and their similarity scores selected using the maximal marginal347            relevance.348 349        Maximal marginal relevance optimizes for similarity to query AND diversity350        among selected documents.351 352        Args:353            embedding: Embedding to look up documents similar to.354            k: Number of Documents to return. Defaults to 4.355            fetch_k: Number of Documents to fetch before filtering to356                     pass to MMR algorithm.357            lambda_mult: Number between 0 and 1 that determines the degree358                        of diversity among the results with 0 corresponding359                        to maximum diversity and 1 to minimum diversity.360                        Defaults to 0.5.361        Returns:362            List of Documents and similarity scores selected by maximal marginal363                relevance and score for each.364        """365        if "score_threshold" in kwargs:366            score_threshold = kwargs.pop("score_threshold")367        else:368            score_threshold = MAX_FLOAT369        scores, indices = self.vector_index.query(370            np.array([np.array(embedding).astype(np.float32)]).astype(np.float32),371            k=fetch_k if filter is None else fetch_k * 2,372            **kwargs,373        )374        results = self.process_index_results(375            ids=indices[0],376            scores=scores[0],377            filter=filter,378            k=fetch_k if filter is None else fetch_k * 2,379            score_threshold=score_threshold,380        )381        embeddings = [382            self.embedding.embed_documents([doc.page_content])[0] for doc, _ in results383        ]384        mmr_selected = maximal_marginal_relevance(385            np.array([embedding], dtype=np.float32),386            embeddings,387            k=k,388            lambda_mult=lambda_mult,389        )390        docs_and_scores = []391        for i in mmr_selected:392            docs_and_scores.append(results[i])393        return docs_and_scores394 395    def max_marginal_relevance_search_by_vector(396        self,397        embedding: List[float],398        k: int = 4,399        fetch_k: int = 20,400        lambda_mult: float = 0.5,401        filter: Optional[Dict[str, Any]] = None,402        **kwargs: Any,403    ) -> List[Document]:404        """Return docs selected using the maximal marginal relevance.405 406        Maximal marginal relevance optimizes for similarity to query AND diversity407        among selected documents.408 409        Args:410            embedding: Embedding to look up documents similar to.411            k: Number of Documents to return. Defaults to 4.412            fetch_k: Number of Documents to fetch before filtering to413                     pass to MMR algorithm.414            lambda_mult: Number between 0 and 1 that determines the degree415                        of diversity among the results with 0 corresponding416                        to maximum diversity and 1 to minimum diversity.417                        Defaults to 0.5.418        Returns:419            List of Documents selected by maximal marginal relevance.420        """421        docs_and_scores = self.max_marginal_relevance_search_with_score_by_vector(422            embedding,423            k=k,424            fetch_k=fetch_k,425            lambda_mult=lambda_mult,426            filter=filter,427            **kwargs,428        )429        return [doc for doc, _ in docs_and_scores]430 431    def max_marginal_relevance_search(432        self,433        query: str,434        k: int = 4,435        fetch_k: int = 20,436        lambda_mult: float = 0.5,437        filter: Optional[Dict[str, Any]] = None,438        **kwargs: Any,439    ) -> List[Document]:440        """Return docs selected using the maximal marginal relevance.441 442        Maximal marginal relevance optimizes for similarity to query AND diversity443        among selected documents.444 445        Args:446            query: Text to look up documents similar to.447            k: Number of Documents to return. Defaults to 4.448            fetch_k: Number of Documents to fetch before filtering (if needed) to449                     pass to MMR algorithm.450            lambda_mult: Number between 0 and 1 that determines the degree451                        of diversity among the results with 0 corresponding452                        to maximum diversity and 1 to minimum diversity.453                        Defaults to 0.5.454        Returns:455            List of Documents selected by maximal marginal relevance.456        """457        embedding = self.embedding_function(query)458        docs = self.max_marginal_relevance_search_by_vector(459            embedding,460            k=k,461            fetch_k=fetch_k,462            lambda_mult=lambda_mult,463            filter=filter,464            **kwargs,465        )466        return docs467 468    @classmethod469    def create(470        cls,471        index_uri: str,472        index_type: str,473        dimensions: int,474        vector_type: np.dtype,475        *,476        metadatas: bool = True,477        config: Optional[Mapping[str, Any]] = None,478    ) -> None:479        tiledb_vs, tiledb = (480            guard_import("tiledb.vector_search"),481            guard_import("tiledb"),482        )483        with tiledb.scope_ctx(ctx_or_config=config):484            try:485                tiledb.group_create(index_uri)486            except tiledb.TileDBError as err:487                raise err488            group = tiledb.Group(index_uri, "w")489            vector_index_uri = get_vector_index_uri(group.uri)490            docs_uri = get_documents_array_uri(group.uri)491            if index_type == "FLAT":492                tiledb_vs.flat_index.create(493                    uri=vector_index_uri,494                    dimensions=dimensions,495                    vector_type=vector_type,496                    config=config,497                )498            elif index_type == "IVF_FLAT":499                tiledb_vs.ivf_flat_index.create(500                    uri=vector_index_uri,501                    dimensions=dimensions,502                    vector_type=vector_type,503                    config=config,504                )505            group.add(vector_index_uri, name=VECTOR_INDEX_NAME)506 507            # Create TileDB array to store Documents508            # TODO add a Document store API to tiledb-vector-search to allow storing509            #  different types of objects and metadata in a more generic way.510            dim = tiledb.Dim(511                name="id",512                domain=(0, MAX_UINT64 - 1),513                dtype=np.dtype(np.uint64),514            )515            dom = tiledb.Domain(dim)516 517            text_attr = tiledb.Attr(name="text", dtype=np.dtype("U1"), var=True)518            attrs = [text_attr]519            if metadatas:520                metadata_attr = tiledb.Attr(name="metadata", dtype=np.uint8, var=True)521                attrs.append(metadata_attr)522            schema = tiledb.ArraySchema(523                domain=dom,524                sparse=True,525                allows_duplicates=False,526                attrs=attrs,527            )528            tiledb.Array.create(docs_uri, schema)529            group.add(docs_uri, name=DOCUMENTS_ARRAY_NAME)530            group.close()531 532    @classmethod533    def __from(534        cls,535        texts: List[str],536        embeddings: List[List[float]],537        embedding: Embeddings,538        index_uri: str,539        *,540        metadatas: Optional[List[dict]] = None,541        ids: Optional[List[str]] = None,542        metric: str = DEFAULT_METRIC,543        index_type: str = "FLAT",544        config: Optional[Mapping[str, Any]] = None,545        index_timestamp: int = 0,546        **kwargs: Any,547    ) -> TileDB:548        if metric not in INDEX_METRICS:549            raise ValueError(550                (551                    f"Unsupported distance metric: {metric}. "552                    f"Expected one of {list(INDEX_METRICS)}"553                )554            )555        tiledb_vs, tiledb = (556            guard_import("tiledb.vector_search"),557            guard_import("tiledb"),558        )559        input_vectors = np.array(embeddings).astype(np.float32)560        cls.create(561            index_uri=index_uri,562            index_type=index_type,563            dimensions=input_vectors.shape[1],564            vector_type=input_vectors.dtype,565            metadatas=metadatas is not None,566            config=config,567        )568        with tiledb.scope_ctx(ctx_or_config=config):569            if not embeddings:570                raise ValueError("embeddings must be provided to build a TileDB index")571 572            vector_index_uri = get_vector_index_uri(index_uri)573            docs_uri = get_documents_array_uri(index_uri)574            if ids is None:575                ids = [str(random.randint(0, MAX_UINT64 - 1)) for _ in texts]576            external_ids = np.array(ids).astype(np.uint64)577 578            tiledb_vs.ingestion.ingest(579                index_type=index_type,580                index_uri=vector_index_uri,581                input_vectors=input_vectors,582                external_ids=external_ids,583                index_timestamp=index_timestamp if index_timestamp != 0 else None,584                config=config,585                **kwargs,586            )587            with tiledb.open(docs_uri, "w") as A:588                if external_ids is None:589                    external_ids = np.zeros(len(texts), dtype=np.uint64)590                    for i in range(len(texts)):591                        external_ids[i] = i592                data = {}593                data["text"] = np.array(texts)594                if metadatas is not None:595                    metadata_attr = np.empty([len(metadatas)], dtype=object)596                    i = 0597                    for metadata in metadatas:598                        metadata_attr[i] = np.frombuffer(599                            pickle.dumps(metadata), dtype=np.uint8600                        )601                        i += 1602                    data["metadata"] = metadata_attr603 604                A[external_ids] = data605        return cls(606            embedding=embedding,607            index_uri=index_uri,608            metric=metric,609            config=config,610            **kwargs,611        )612 613    def delete(614        self, ids: Optional[List[str]] = None, timestamp: int = 0, **kwargs: Any615    ) -> Optional[bool]:616        """Delete by vector ID or other criteria.617 618        Args:619            ids: List of ids to delete.620            timestamp: Optional timestamp to delete with.621            **kwargs: Other keyword arguments that subclasses might use.622 623        Returns:624            Optional[bool]: True if deletion is successful,625            False otherwise, None if not implemented.626        """627 628        external_ids = np.array(ids).astype(np.uint64)629        self.vector_index.delete_batch(630            external_ids=external_ids, timestamp=timestamp if timestamp != 0 else None631        )632        return True633 634    def add_texts(635        self,636        texts: Iterable[str],637        metadatas: Optional[List[dict]] = None,638        ids: Optional[List[str]] = None,639        timestamp: int = 0,640        **kwargs: Any,641    ) -> List[str]:642        """Run more texts through the embeddings and add to the vectorstore.643 644        Args:645            texts: Iterable of strings to add to the vectorstore.646            metadatas: Optional list of metadatas associated with the texts.647            ids: Optional ids of each text object.648            timestamp: Optional timestamp to write new texts with.649            kwargs: vectorstore specific parameters650 651        Returns:652            List of ids from adding the texts into the vectorstore.653        """654        tiledb = guard_import("tiledb")655        embeddings = self.embedding.embed_documents(list(texts))656        if ids is None:657            ids = [str(random.randint(0, MAX_UINT64 - 1)) for _ in texts]658 659        external_ids = np.array(ids).astype(np.uint64)660        vectors = np.empty((len(embeddings)), dtype="O")661        for i in range(len(embeddings)):662            vectors[i] = np.array(embeddings[i], dtype=np.float32)663        self.vector_index.update_batch(664            vectors=vectors,665            external_ids=external_ids,666            timestamp=timestamp if timestamp != 0 else None,667        )668 669        docs = {}670        docs["text"] = np.array(texts)671        if metadatas is not None:672            metadata_attr = np.empty([len(metadatas)], dtype=object)673            i = 0674            for metadata in metadatas:675                metadata_attr[i] = np.frombuffer(pickle.dumps(metadata), dtype=np.uint8)676                i += 1677            docs["metadata"] = metadata_attr678 679        docs_array = tiledb.open(680            self.docs_array_uri,681            "w",682            timestamp=timestamp if timestamp != 0 else None,683            config=self.config,684        )685        docs_array[external_ids] = docs686        docs_array.close()687        return ids688 689    @classmethod690    def from_texts(691        cls,692        texts: List[str],693        embedding: Embeddings,694        metadatas: Optional[List[dict]] = None,695        ids: Optional[List[str]] = None,696        metric: str = DEFAULT_METRIC,697        index_uri: str = "/tmp/tiledb_array",698        index_type: str = "FLAT",699        config: Optional[Mapping[str, Any]] = None,700        index_timestamp: int = 0,701        **kwargs: Any,702    ) -> TileDB:703        """Construct a TileDB index from raw documents.704 705        Args:706            texts: List of documents to index.707            embedding: Embedding function to use.708            metadatas: List of metadata dictionaries to associate with documents.709            ids: Optional ids of each text object.710            metric: Metric to use for indexing. Defaults to "euclidean".711            index_uri: The URI to write the TileDB arrays712            index_type: Optional,  Vector index type ("FLAT", IVF_FLAT")713            config: Optional, TileDB config714            index_timestamp: Optional, timestamp to write new texts with.715 716        Example:717            .. code-block:: python718 719                from langchain_community import TileDB720                from langchain_community.embeddings import OpenAIEmbeddings721                embeddings = OpenAIEmbeddings()722                index = TileDB.from_texts(texts, embeddings)723        """724        embeddings = []725        embeddings = embedding.embed_documents(texts)726        return cls.__from(727            texts=texts,728            embeddings=embeddings,729            embedding=embedding,730            metadatas=metadatas,731            ids=ids,732            metric=metric,733            index_uri=index_uri,734            index_type=index_type,735            config=config,736            index_timestamp=index_timestamp,737            **kwargs,738        )739 740    @classmethod741    def from_embeddings(742        cls,743        text_embeddings: List[Tuple[str, List[float]]],744        embedding: Embeddings,745        index_uri: str,746        *,747        metadatas: Optional[List[dict]] = None,748        ids: Optional[List[str]] = None,749        metric: str = DEFAULT_METRIC,750        index_type: str = "FLAT",751        config: Optional[Mapping[str, Any]] = None,752        index_timestamp: int = 0,753        **kwargs: Any,754    ) -> TileDB:755        """Construct TileDB index from embeddings.756 757        Args:758            text_embeddings: List of tuples of (text, embedding)759            embedding: Embedding function to use.760            index_uri: The URI to write the TileDB arrays761            metadatas: List of metadata dictionaries to associate with documents.762            metric: Optional, Metric to use for indexing. Defaults to "euclidean".763            index_type: Optional, Vector index type ("FLAT", IVF_FLAT")764            config: Optional, TileDB config765            index_timestamp: Optional, timestamp to write new texts with.766 767        Example:768            .. code-block:: python769 770                from langchain_community import TileDB771                from langchain_community.embeddings import OpenAIEmbeddings772                embeddings = OpenAIEmbeddings()773                text_embeddings = embeddings.embed_documents(texts)774                text_embedding_pairs = list(zip(texts, text_embeddings))775                db = TileDB.from_embeddings(text_embedding_pairs, embeddings)776        """777        texts = [t[0] for t in text_embeddings]778        embeddings = [t[1] for t in text_embeddings]779 780        return cls.__from(781            texts=texts,782            embeddings=embeddings,783            embedding=embedding,784            metadatas=metadatas,785            ids=ids,786            metric=metric,787            index_uri=index_uri,788            index_type=index_type,789            config=config,790            index_timestamp=index_timestamp,791            **kwargs,792        )793 794    @classmethod795    def load(796        cls,797        index_uri: str,798        embedding: Embeddings,799        *,800        metric: str = DEFAULT_METRIC,801        config: Optional[Mapping[str, Any]] = None,802        timestamp: Any = None,803        **kwargs: Any,804    ) -> TileDB:805        """Load a TileDB index from a URI.806 807        Args:808            index_uri: The URI of the TileDB vector index.809            embedding: Embeddings to use when generating queries.810            metric: Optional, Metric to use for indexing. Defaults to "euclidean".811            config: Optional, TileDB config812            timestamp: Optional, timestamp to use for opening the arrays.813        """814        return cls(815            embedding=embedding,816            index_uri=index_uri,817            metric=metric,818            config=config,819            timestamp=timestamp,820            **kwargs,821        )822 823    def consolidate_updates(self, **kwargs: Any) -> None:824        self.vector_index = self.vector_index.consolidate_updates(**kwargs)825 
codekingpro/portable-devtools · Team Ai