Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
kinetica.py955 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import asyncio4import enum5import json6import logging7import struct8import uuid9from collections import OrderedDict10from enum import Enum11from functools import partial12from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, Type13 14import numpy as np15from langchain_core.documents import Document16from langchain_core.embeddings import Embeddings17from langchain_core.vectorstores import VectorStore18from pydantic_settings import BaseSettings, SettingsConfigDict19 20from langchain_community.vectorstores.utils import maximal_marginal_relevance21 22 23class DistanceStrategy(str, enum.Enum):24    """Enumerator of the Distance strategies."""25 26    EUCLIDEAN = "l2"27    COSINE = "cosine"28    MAX_INNER_PRODUCT = "inner"29 30 31def _results_to_docs(docs_and_scores: Any) -> List[Document]:32    """Return docs from docs and scores."""33    return [doc for doc, _ in docs_and_scores]34 35 36class Dimension(int, Enum):37    """Some default dimensions for known embeddings."""38 39    OPENAI = 153640 41 42DEFAULT_DISTANCE_STRATEGY = DistanceStrategy.EUCLIDEAN43 44_LANGCHAIN_DEFAULT_SCHEMA_NAME = "langchain"  ## Default Kinetica schema name45_LANGCHAIN_DEFAULT_COLLECTION_NAME = (46    "langchain_kinetica_embeddings"  ## Default Kinetica table name47)48 49 50class KineticaSettings(BaseSettings):51    """`Kinetica` client configuration.52 53    Attribute:54        host (str) : An URL to connect to MyScale backend.55                             Defaults to 'localhost'.56        port (int) : URL port to connect with HTTP. Defaults to 8443.57        username (str) : Username to login. Defaults to None.58        password (str) : Password to login. Defaults to None.59        database (str) : Database name to find the table. Defaults to 'default'.60        table (str) : Table name to operate on.61                      Defaults to 'vector_table'.62        metric (str) : Metric to compute distance,63                       supported are ('angular', 'euclidean', 'manhattan', 'hamming',64                       'dot'). Defaults to 'angular'.65                       https://github.com/spotify/annoy/blob/main/src/annoymodule.cc#L149-L16966 67    """68 69    host: str = "http://127.0.0.1"70    port: int = 919171 72    username: Optional[str] = None73    password: Optional[str] = None74 75    database: str = _LANGCHAIN_DEFAULT_SCHEMA_NAME76    table: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME77    metric: str = DEFAULT_DISTANCE_STRATEGY.value78 79    def __getitem__(self, item: str) -> Any:80        return getattr(self, item)81 82    model_config = SettingsConfigDict(83        env_file=".env",84        env_file_encoding="utf-8",85        env_prefix="kinetica_",86        extra="ignore",87    )88 89 90class Kinetica(VectorStore):91    """`Kinetica` vector store.92 93    To use, you should have the ``gpudb`` python package installed.94 95    Args:96        config: Kinetica connection settings class.97        embedding_function: Any embedding function implementing98            `langchain.embeddings.base.Embeddings` interface.99        collection_name: The name of the collection to use. (default: langchain)100            NOTE: This is not the name of the table, but the name of the collection.101            The tables will be created when initializing the store (if not exists)102            So, make sure the user has the right permissions to create tables.103        distance_strategy: The distance strategy to use. (default: COSINE)104        pre_delete_collection: If True, will delete the collection if it exists.105            (default: False). Useful for testing.106        engine_args: SQLAlchemy's create engine arguments.107 108    Example:109        .. code-block:: python110 111            from langchain_community.vectorstores import Kinetica, KineticaSettings112            from langchain_community.embeddings.openai import OpenAIEmbeddings113 114            kinetica_settings = KineticaSettings(115                host="http://127.0.0.1", username="", password=""116                )117            COLLECTION_NAME = "kinetica_store"118            embeddings = OpenAIEmbeddings()119            vectorstore = Kinetica.from_documents(120                documents=docs,121                embedding=embeddings,122                collection_name=COLLECTION_NAME,123                config=kinetica_settings,124            )125    """126 127    def __init__(128        self,129        config: KineticaSettings,130        embedding_function: Embeddings,131        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,132        schema_name: str = _LANGCHAIN_DEFAULT_SCHEMA_NAME,133        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,134        pre_delete_collection: bool = False,135        logger: Optional[logging.Logger] = None,136        relevance_score_fn: Optional[Callable[[float], float]] = None,137    ) -> None:138        """Constructor for the Kinetica class139 140        Args:141            config (KineticaSettings): a `KineticaSettings` instance142            embedding_function (Embeddings): embedding function to use143            collection_name (str, optional): the Kinetica table name.144                            Defaults to _LANGCHAIN_DEFAULT_COLLECTION_NAME.145            schema_name (str, optional): the Kinetica table name.146                            Defaults to _LANGCHAIN_DEFAULT_SCHEMA_NAME.147            distance_strategy (DistanceStrategy, optional): _description_.148                            Defaults to DEFAULT_DISTANCE_STRATEGY.149            pre_delete_collection (bool, optional): _description_. Defaults to False.150            logger (Optional[logging.Logger], optional): _description_.151                            Defaults to None.152        """153 154        self._config = config155        self.embedding_function = embedding_function156        self.collection_name = collection_name157        self.schema_name = schema_name158        self._distance_strategy = distance_strategy159        self.pre_delete_collection = pre_delete_collection160        self.logger = logger or logging.getLogger(__name__)161        self.override_relevance_score_fn = relevance_score_fn162        self._db = self.__get_db(self._config)163 164    def __post_init__(self, dimensions: int) -> None:165        """166        Initialize the store.167        """168        try:169            from gpudb import GPUdbTable170        except ImportError:171            raise ImportError(172                "Could not import Kinetica python API. "173                "Please install it with `pip install gpudb>=7.2.2.0`."174            )175 176        self.dimensions = dimensions177        dimension_field = f"vector({dimensions})"178 179        if self.pre_delete_collection:180            self.delete_schema()181 182        self.table_name = self.collection_name183        if self.schema_name is not None and len(self.schema_name) > 0:184            self.table_name = f"{self.schema_name}.{self.collection_name}"185 186        self.table_schema = [187            ["text", "string"],188            ["embedding", "bytes", dimension_field],189            ["metadata", "string", "json"],190            ["id", "string", "uuid"],191        ]192 193        self.create_schema()194        self.EmbeddingStore: GPUdbTable = self.create_tables_if_not_exists()195 196    def __get_db(self, config: KineticaSettings) -> Any:197        try:198            from gpudb import GPUdb199        except ImportError:200            raise ImportError(201                "Could not import Kinetica python API. "202                "Please install it with `pip install gpudb>=7.2.2.0`."203            )204 205        options = GPUdb.Options()206        options.username = config.username207        options.password = config.password208        options.skip_ssl_cert_verification = True209        return GPUdb(host=config.host, options=options)210 211    @property212    def embeddings(self) -> Embeddings:213        return self.embedding_function214 215    @classmethod216    def __from(217        cls,218        config: KineticaSettings,219        texts: List[str],220        embeddings: List[List[float]],221        embedding: Embeddings,222        dimensions: int,223        metadatas: Optional[List[dict]] = None,224        ids: Optional[List[str]] = None,225        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,226        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,227        pre_delete_collection: bool = False,228        logger: Optional[logging.Logger] = None,229        *,230        schema_name: str = _LANGCHAIN_DEFAULT_SCHEMA_NAME,231        **kwargs: Any,232    ) -> Kinetica:233        """Class method to assist in constructing the `Kinetica` store instance234            using different combinations of parameters235 236        Args:237            config (KineticaSettings): a `KineticaSettings` instance238            texts (List[str]): The list of texts to generate embeddings for and store239            embeddings (List[List[float]]): List of embeddings240            embedding (Embeddings): the Embedding function241            dimensions (int): The number of dimensions the embeddings have242            metadatas (Optional[List[dict]], optional): List of JSON data associated243                        with each text. Defaults to None.244            ids (Optional[List[str]], optional): List of unique IDs (UUID by default)245                        associated with each text. Defaults to None.246            collection_name (str, optional): Kinetica table name.247                        Defaults to _LANGCHAIN_DEFAULT_COLLECTION_NAME.248            schema_name (str, optional): Kinetica schema name.249                        Defaults to _LANGCHAIN_DEFAULT_SCHEMA_NAME.250            distance_strategy (DistanceStrategy, optional): Not used for now.251                        Defaults to DEFAULT_DISTANCE_STRATEGY.252            pre_delete_collection (bool, optional): Whether to delete the Kinetica253                        schema or not. Defaults to False.254            logger (Optional[logging.Logger], optional): Logger to use for logging at255                        different levels. Defaults to None.256 257        Returns:258            Kinetica: An instance of Kinetica class259        """260        if ids is None:261            ids = [str(uuid.uuid4()) for _ in texts]262 263        if not metadatas:264            metadatas = [{} for _ in texts]265 266        store = cls(267            config=config,268            collection_name=collection_name,269            schema_name=schema_name,270            embedding_function=embedding,271            distance_strategy=distance_strategy,272            pre_delete_collection=pre_delete_collection,273            logger=logger,274            **kwargs,275        )276 277        store.__post_init__(dimensions)278 279        store.add_embeddings(280            texts=texts, embeddings=embeddings, metadatas=metadatas, ids=ids, **kwargs281        )282 283        return store284 285    def create_tables_if_not_exists(self) -> Any:286        """Create the table to store the texts and embeddings"""287 288        try:289            from gpudb import GPUdbTable290        except ImportError:291            raise ImportError(292                "Could not import Kinetica python API. "293                "Please install it with `pip install gpudb>=7.2.2.0`."294            )295        return GPUdbTable(296            _type=self.table_schema,297            name=self.table_name,298            db=self._db,299            options={"is_replicated": "true"},300        )301 302    def drop_tables(self) -> None:303        """Delete the table"""304        self._db.clear_table(305            f"{self.table_name}", options={"no_error_if_not_exists": "true"}306        )307 308    def create_schema(self) -> None:309        """Create a new Kinetica schema"""310        self._db.create_schema(self.schema_name)311 312    def delete_schema(self) -> None:313        """Delete a Kinetica schema with cascade set to `true`314        This method will delete a schema with all tables in it.315        """316        self.logger.debug("Trying to delete collection")317        self._db.drop_schema(318            self.schema_name, {"no_error_if_not_exists": "true", "cascade": "true"}319        )320 321    def add_embeddings(322        self,323        texts: Iterable[str],324        embeddings: List[List[float]],325        metadatas: Optional[List[dict]] = None,326        ids: Optional[List[str]] = None,327        **kwargs: Any,328    ) -> List[str]:329        """Add embeddings to the vectorstore.330 331        Args:332            texts: Iterable of strings to add to the vectorstore.333            embeddings: List of list of embedding vectors.334            metadatas: List of metadatas associated with the texts.335            ids: List of ids for the text embedding pairs336            kwargs: vectorstore specific parameters337        """338        if ids is None:339            ids = [str(uuid.uuid4()) for _ in texts]340 341        if not metadatas:342            metadatas = [{} for _ in texts]343 344        records = []345        for text, embedding, metadata, id in zip(texts, embeddings, metadatas, ids):346            buf = struct.pack("%sf" % self.dimensions, *embedding)347            records.append([text, buf, json.dumps(metadata), id])348 349        self.EmbeddingStore.insert_records(records)350 351        return ids352 353    def add_texts(354        self,355        texts: Iterable[str],356        metadatas: Optional[List[dict]] = None,357        ids: Optional[List[str]] = None,358        **kwargs: Any,359    ) -> List[str]:360        """Run more texts through the embeddings and add to the vectorstore.361 362        Args:363            texts: Iterable of strings to add to the vectorstore.364            metadatas: Optional list of metadatas (JSON data) associated with the texts.365            ids: List of IDs (UUID) for the texts supplied; will be generated if None366            kwargs: vectorstore specific parameters367 368        Returns:369            List of ids from adding the texts into the vectorstore.370        """371        embeddings = self.embedding_function.embed_documents(list(texts))372        self.dimensions = len(embeddings[0])373        if not hasattr(self, "EmbeddingStore"):374            self.__post_init__(self.dimensions)375        return self.add_embeddings(376            texts=texts, embeddings=embeddings, metadatas=metadatas, ids=ids, **kwargs377        )378 379    def similarity_search(380        self,381        query: str,382        k: int = 4,383        filter: Optional[dict] = None,384        **kwargs: Any,385    ) -> List[Document]:386        """Run similarity search with Kinetica with distance.387 388        Args:389            query (str): Query text to search for.390            k (int): Number of results to return. Defaults to 4.391            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.392 393        Returns:394            List of Documents most similar to the query.395        """396        embedding = self.embedding_function.embed_query(text=query)397        return self.similarity_search_by_vector(398            embedding=embedding,399            k=k,400            filter=filter,401        )402 403    def similarity_search_with_score(404        self,405        query: str,406        k: int = 4,407        filter: Optional[dict] = None,408    ) -> List[Tuple[Document, float]]:409        """Return docs most similar to query.410 411        Args:412            query: Text to look up documents similar to.413            k: Number of Documents to return. Defaults to 4.414            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.415 416        Returns:417            List of Documents most similar to the query and score for each418        """419        embedding = self.embedding_function.embed_query(query)420        docs = self.similarity_search_with_score_by_vector(421            embedding=embedding, k=k, filter=filter422        )423        return docs424 425    def similarity_search_with_score_by_vector(426        self,427        embedding: List[float],428        k: int = 4,429        filter: Optional[dict] = None,430    ) -> List[Tuple[Document, float]]:431        # from gpudb import GPUdbException432 433        results = []434        resp: Dict = self.__query_collection(embedding, k, filter)435        if resp and resp["status_info"]["status"] == "OK":436            total_records = resp["total_number_of_records"]437            if total_records > 0:438                records: OrderedDict = resp["records"]439                results = list(zip(*list(records.values())))440 441                return self._results_to_docs_and_scores(results)442            else:443                self.logger.warning(444                    f"No records found; status: {resp['status_info']['status']}"445                )446        return results447 448    def similarity_search_by_vector(449        self,450        embedding: List[float],451        k: int = 4,452        filter: Optional[dict] = None,453        **kwargs: Any,454    ) -> List[Document]:455        """Return docs most similar to embedding vector.456 457        Args:458            embedding: Embedding to look up documents similar to.459            k: Number of Documents to return. Defaults to 4.460            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.461 462        Returns:463            List of Documents most similar to the query vector.464        """465        docs_and_scores = self.similarity_search_with_score_by_vector(466            embedding=embedding, k=k, filter=filter467        )468        return [doc for doc, _ in docs_and_scores]469 470    def _results_to_docs_and_scores(self, results: Any) -> List[Tuple[Document, float]]:471        """Return docs and scores from results."""472        docs = (473            [474                (475                    Document(476                        page_content=result[0],477                        metadata=json.loads(result[1]),478                    ),479                    result[2] if self.embedding_function is not None else None,480                )481                for result in results482            ]483            if len(results) > 0484            else []485        )486        return docs487 488    def _select_relevance_score_fn(self) -> Callable[[float], float]:489        """490        The 'correct' relevance function491        may differ depending on a few things, including:492        - the distance / similarity metric used by the VectorStore493        - the scale of your embeddings (OpenAI's are unit normed. Many others are not!)494        - embedding dimensionality495        - etc.496        """497        if self.override_relevance_score_fn is not None:498            return self.override_relevance_score_fn499 500        # Default strategy is to rely on distance strategy provided501        # in vectorstore constructor502        if self._distance_strategy == DistanceStrategy.COSINE:503            return self._cosine_relevance_score_fn504        elif self._distance_strategy == DistanceStrategy.EUCLIDEAN:505            return self._euclidean_relevance_score_fn506        elif self._distance_strategy == DistanceStrategy.MAX_INNER_PRODUCT:507            return self._max_inner_product_relevance_score_fn508        else:509            raise ValueError(510                "No supported normalization function"511                f" for distance_strategy of {self._distance_strategy}."512                "Consider providing relevance_score_fn to Kinetica constructor."513            )514 515    @property516    def distance_strategy(self) -> str:517        if self._distance_strategy == DistanceStrategy.EUCLIDEAN:518            return "l2_distance"519        elif self._distance_strategy == DistanceStrategy.COSINE:520            return "cosine_distance"521        elif self._distance_strategy == DistanceStrategy.MAX_INNER_PRODUCT:522            return "dot_product"523        else:524            raise ValueError(525                f"Got unexpected value for distance: {self._distance_strategy}. "526                f"Should be one of {', '.join([ds.value for ds in DistanceStrategy])}."527            )528 529    def __query_collection(530        self,531        embedding: List[float],532        k: int = 4,533        filter: Optional[Dict[str, str]] = None,534    ) -> Dict:535        """Query the collection."""536        # if filter is not None:537        #     filter_clauses = []538        #     for key, value in filter.items():539        #         IN = "in"540        #         if isinstance(value, dict) and IN in map(str.lower, value):541        #             value_case_insensitive = {542        #                 k.lower(): v for k, v in value.items()543        #             }544        #             filter_by_metadata = self.EmbeddingStore.cmetadata[545        #                 key546        #             ].astext.in_(value_case_insensitive[IN])547        #             filter_clauses.append(filter_by_metadata)548        #         else:549        #             filter_by_metadata = self.EmbeddingStore.cmetadata[550        #                 key551        #             ].astext == str(value)552        #             filter_clauses.append(filter_by_metadata)553 554        json_filter = json.dumps(filter) if filter is not None else None555        where_clause = (556            f" where '{json_filter}' = JSON(metadata) "557            if json_filter is not None558            else ""559        )560 561        embedding_str = "[" + ",".join([str(x) for x in embedding]) + "]"562 563        dist_strategy = self.distance_strategy564 565        query_string = f"""566                SELECT text, metadata, {dist_strategy}(embedding, '{embedding_str}') 567                as distance, embedding568                FROM "{self.schema_name}"."{self.collection_name}"569                {where_clause}570                ORDER BY distance asc NULLS LAST571                LIMIT {k}572        """573 574        self.logger.debug(query_string)575        resp = self._db.execute_sql_and_decode(query_string)576        self.logger.debug(resp)577        return resp578 579    def max_marginal_relevance_search_with_score_by_vector(580        self,581        embedding: List[float],582        k: int = 4,583        fetch_k: int = 20,584        lambda_mult: float = 0.5,585        filter: Optional[Dict[str, str]] = None,586        **kwargs: Any,587    ) -> List[Tuple[Document, float]]:588        """Return docs selected using the maximal marginal relevance with score589            to embedding vector.590 591        Maximal marginal relevance optimizes for similarity to query AND diversity592            among selected documents.593 594        Args:595            embedding: Embedding to look up documents similar to.596            k (int): Number of Documents to return. Defaults to 4.597            fetch_k (int): Number of Documents to fetch to pass to MMR algorithm.598                Defaults to 20.599            lambda_mult (float): Number between 0 and 1 that determines the degree600                of diversity among the results with 0 corresponding601                to maximum diversity and 1 to minimum diversity.602                Defaults to 0.5.603            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.604 605        Returns:606            List[Tuple[Document, float]]: List of Documents selected by maximal marginal607                relevance to the query and score for each.608        """609        resp = self.__query_collection(embedding=embedding, k=fetch_k, filter=filter)610        records: OrderedDict = resp["records"]611        results = list(zip(*list(records.values())))612 613        embedding_list = [614            struct.unpack("%sf" % self.dimensions, embedding)615            for embedding in records["embedding"]616        ]617 618        mmr_selected = maximal_marginal_relevance(619            np.array(embedding, dtype=np.float32),620            embedding_list,621            k=k,622            lambda_mult=lambda_mult,623        )624 625        candidates = self._results_to_docs_and_scores(results)626 627        return [r for i, r in enumerate(candidates) if i in mmr_selected]628 629    def max_marginal_relevance_search(630        self,631        query: str,632        k: int = 4,633        fetch_k: int = 20,634        lambda_mult: float = 0.5,635        filter: Optional[Dict[str, str]] = None,636        **kwargs: Any,637    ) -> List[Document]:638        """Return docs selected using the maximal marginal relevance.639 640        Maximal marginal relevance optimizes for similarity to query AND diversity641            among selected documents.642 643        Args:644            query (str): Text to look up documents similar to.645            k (int): Number of Documents to return. Defaults to 4.646            fetch_k (int): Number of Documents to fetch to pass to MMR algorithm.647                Defaults to 20.648            lambda_mult (float): Number between 0 and 1 that determines the degree649                of diversity among the results with 0 corresponding650                to maximum diversity and 1 to minimum diversity.651                Defaults to 0.5.652            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.653 654        Returns:655            List[Document]: List of Documents selected by maximal marginal relevance.656        """657        embedding = self.embedding_function.embed_query(query)658        return self.max_marginal_relevance_search_by_vector(659            embedding,660            k=k,661            fetch_k=fetch_k,662            lambda_mult=lambda_mult,663            filter=filter,664            **kwargs,665        )666 667    def max_marginal_relevance_search_with_score(668        self,669        query: str,670        k: int = 4,671        fetch_k: int = 20,672        lambda_mult: float = 0.5,673        filter: Optional[dict] = None,674        **kwargs: Any,675    ) -> List[Tuple[Document, float]]:676        """Return docs selected using the maximal marginal relevance with score.677 678        Maximal marginal relevance optimizes for similarity to query AND diversity679            among selected documents.680 681        Args:682            query (str): Text to look up documents similar to.683            k (int): Number of Documents to return. Defaults to 4.684            fetch_k (int): Number of Documents to fetch to pass to MMR algorithm.685                Defaults to 20.686            lambda_mult (float): Number between 0 and 1 that determines the degree687                of diversity among the results with 0 corresponding688                to maximum diversity and 1 to minimum diversity.689                Defaults to 0.5.690            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.691 692        Returns:693            List[Tuple[Document, float]]: List of Documents selected by maximal marginal694                relevance to the query and score for each.695        """696        embedding = self.embedding_function.embed_query(query)697        docs = self.max_marginal_relevance_search_with_score_by_vector(698            embedding=embedding,699            k=k,700            fetch_k=fetch_k,701            lambda_mult=lambda_mult,702            filter=filter,703            **kwargs,704        )705        return docs706 707    def max_marginal_relevance_search_by_vector(708        self,709        embedding: List[float],710        k: int = 4,711        fetch_k: int = 20,712        lambda_mult: float = 0.5,713        filter: Optional[Dict[str, str]] = None,714        **kwargs: Any,715    ) -> List[Document]:716        """Return docs selected using the maximal marginal relevance717            to embedding vector.718 719        Maximal marginal relevance optimizes for similarity to query AND diversity720            among selected documents.721 722        Args:723            embedding (str): Text to look up documents similar to.724            k (int): Number of Documents to return. Defaults to 4.725            fetch_k (int): Number of Documents to fetch to pass to MMR algorithm.726                Defaults to 20.727            lambda_mult (float): Number between 0 and 1 that determines the degree728                of diversity among the results with 0 corresponding729                to maximum diversity and 1 to minimum diversity.730                Defaults to 0.5.731            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.732 733        Returns:734            List[Document]: List of Documents selected by maximal marginal relevance.735        """736        docs_and_scores = self.max_marginal_relevance_search_with_score_by_vector(737            embedding,738            k=k,739            fetch_k=fetch_k,740            lambda_mult=lambda_mult,741            filter=filter,742            **kwargs,743        )744 745        return _results_to_docs(docs_and_scores)746 747    async def amax_marginal_relevance_search_by_vector(748        self,749        embedding: List[float],750        k: int = 4,751        fetch_k: int = 20,752        lambda_mult: float = 0.5,753        filter: Optional[Dict[str, str]] = None,754        **kwargs: Any,755    ) -> List[Document]:756        """Return docs selected using the maximal marginal relevance."""757 758        # This is a temporary workaround to make the similarity search759        # asynchronous. The proper solution is to make the similarity search760        # asynchronous in the vector store implementations.761        func = partial(762            self.max_marginal_relevance_search_by_vector,763            embedding,764            k=k,765            fetch_k=fetch_k,766            lambda_mult=lambda_mult,767            filter=filter,768            **kwargs,769        )770        return await asyncio.get_event_loop().run_in_executor(None, func)771 772    @classmethod773    def from_texts(774        cls: Type[Kinetica],775        texts: List[str],776        embedding: Embeddings,777        metadatas: Optional[List[dict]] = None,778        config: KineticaSettings = KineticaSettings(),779        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,780        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,781        ids: Optional[List[str]] = None,782        pre_delete_collection: bool = False,783        *,784        schema_name: str = _LANGCHAIN_DEFAULT_SCHEMA_NAME,785        **kwargs: Any,786    ) -> Kinetica:787        """Adds the texts passed in to the vector store and returns it788 789        Args:790            cls (Type[Kinetica]): Kinetica class791            texts (List[str]): A list of texts for which the embeddings are generated792            embedding (Embeddings): List of embeddings793            metadatas (Optional[List[dict]], optional): List of dicts, JSON794                        describing the texts/documents. Defaults to None.795            config (KineticaSettings): a `KineticaSettings` instance796            collection_name (str, optional): Kinetica schema name.797                        Defaults to _LANGCHAIN_DEFAULT_COLLECTION_NAME.798            schema_name (str, optional): Kinetica schema name.799                        Defaults to _LANGCHAIN_DEFAULT_SCHEMA_NAME.800            distance_strategy (DistanceStrategy, optional): Distance strategy801                        e.g., l2, cosine etc.. Defaults to DEFAULT_DISTANCE_STRATEGY.802            ids (Optional[List[str]], optional): A list of UUIDs for each803                        text/document. Defaults to None.804            pre_delete_collection (bool, optional): Indicates whether the Kinetica805                        schema is to be deleted or not. Defaults to False.806 807        Returns:808            Kinetica: a `Kinetica` instance809        """810 811        if len(texts) == 0:812            raise ValueError("texts is empty")813 814        try:815            first_embedding = embedding.embed_documents(texts[0:1])816        except NotImplementedError:817            first_embedding = [embedding.embed_query(texts[0])]818 819        dimensions = len(first_embedding[0])820        embeddings = embedding.embed_documents(list(texts))821 822        kinetica_store = cls.__from(823            texts=texts,824            embeddings=embeddings,825            embedding=embedding,826            dimensions=dimensions,827            config=config,828            metadatas=metadatas,829            ids=ids,830            collection_name=collection_name,831            schema_name=schema_name,832            distance_strategy=distance_strategy,833            pre_delete_collection=pre_delete_collection,834            **kwargs,835        )836 837        return kinetica_store838 839    @classmethod840    def from_embeddings(841        cls: Type[Kinetica],842        text_embeddings: List[Tuple[str, List[float]]],843        embedding: Embeddings,844        metadatas: Optional[List[dict]] = None,845        config: KineticaSettings = KineticaSettings(),846        dimensions: int = Dimension.OPENAI,847        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,848        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,849        ids: Optional[List[str]] = None,850        pre_delete_collection: bool = False,851        *,852        schema_name: str = _LANGCHAIN_DEFAULT_SCHEMA_NAME,853        **kwargs: Any,854    ) -> Kinetica:855        """Adds the embeddings passed in to the vector store and returns it856 857        Args:858            cls (Type[Kinetica]): Kinetica class859            text_embeddings (List[Tuple[str, List[float]]]): A list of texts860                            and the embeddings861            embedding (Embeddings): List of embeddings862            metadatas (Optional[List[dict]], optional): List of dicts, JSON describing863                        the texts/documents. Defaults to None.864            config (KineticaSettings): a `KineticaSettings` instance865            dimensions (int, optional): Dimension for the vector data, if not passed a866                        default will be used. Defaults to Dimension.OPENAI.867            collection_name (str, optional): Kinetica schema name.868                        Defaults to _LANGCHAIN_DEFAULT_COLLECTION_NAME.869            schema_name (str, optional): Kinetica schema name.870                        Defaults to _LANGCHAIN_DEFAULT_SCHEMA_NAME.871            distance_strategy (DistanceStrategy, optional): Distance strategy872                        e.g., l2, cosine etc.. Defaults to DEFAULT_DISTANCE_STRATEGY.873            ids (Optional[List[str]], optional): A list of UUIDs for each text/document.874                        Defaults to None.875            pre_delete_collection (bool, optional): Indicates whether the876                        Kinetica schema is to be deleted or not. Defaults to False.877 878        Returns:879            Kinetica: a `Kinetica` instance880        """881 882        texts = [t[0] for t in text_embeddings]883        embeddings = [t[1] for t in text_embeddings]884        dimensions = len(embeddings[0])885 886        return cls.__from(887            texts=texts,888            embeddings=embeddings,889            embedding=embedding,890            dimensions=dimensions,891            config=config,892            metadatas=metadatas,893            ids=ids,894            collection_name=collection_name,895            schema_name=schema_name,896            distance_strategy=distance_strategy,897            pre_delete_collection=pre_delete_collection,898            **kwargs,899        )900 901    @classmethod902    def from_documents(903        cls: Type[Kinetica],904        documents: List[Document],905        embedding: Embeddings,906        config: KineticaSettings = KineticaSettings(),907        metadatas: Optional[List[dict]] = None,908        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,909        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,910        ids: Optional[List[str]] = None,911        pre_delete_collection: bool = False,912        *,913        schema_name: str = _LANGCHAIN_DEFAULT_SCHEMA_NAME,914        **kwargs: Any,915    ) -> Kinetica:916        """Adds the list of `Document` passed in to the vector store and returns it917 918        Args:919            cls (Type[Kinetica]): Kinetica class920            texts (List[str]): A list of texts for which the embeddings are generated921            embedding (Embeddings): List of embeddings922            config (KineticaSettings): a `KineticaSettings` instance923            metadatas (Optional[List[dict]], optional): List of dicts, JSON describing924                        the texts/documents. Defaults to None.925            collection_name (str, optional): Kinetica schema name.926                        Defaults to _LANGCHAIN_DEFAULT_COLLECTION_NAME.927            schema_name (str, optional): Kinetica schema name.928                        Defaults to _LANGCHAIN_DEFAULT_SCHEMA_NAME.929            distance_strategy (DistanceStrategy, optional): Distance strategy930                        e.g., l2, cosine etc.. Defaults to DEFAULT_DISTANCE_STRATEGY.931            ids (Optional[List[str]], optional): A list of UUIDs for each text/document.932                        Defaults to None.933            pre_delete_collection (bool, optional): Indicates whether the Kinetica934                        schema is to be deleted or not. Defaults to False.935 936        Returns:937            Kinetica: a `Kinetica` instance938        """939 940        texts = [d.page_content for d in documents]941        metadatas = [d.metadata for d in documents]942 943        return cls.from_texts(944            texts=texts,945            embedding=embedding,946            metadatas=metadatas,947            config=config,948            collection_name=collection_name,949            schema_name=schema_name,950            distance_strategy=distance_strategy,951            ids=ids,952            pre_delete_collection=pre_delete_collection,953            **kwargs,954        )955 
codekingpro/portable-devtools · Team Ai