Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
singlestoredb.py1070 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import json4import re5from enum import Enum6from typing import (7    Any,8    Callable,9    Iterable,10    List,11    Optional,12    Tuple,13    Type,14)15 16from langchain_core._api import deprecated17from langchain_core.documents import Document18from langchain_core.embeddings import Embeddings19from langchain_core.vectorstores import VectorStore, VectorStoreRetriever20from sqlalchemy.pool import QueuePool21 22from langchain_community.vectorstores.utils import DistanceStrategy23 24DEFAULT_DISTANCE_STRATEGY = DistanceStrategy.DOT_PRODUCT25 26ORDERING_DIRECTIVE: dict = {27    DistanceStrategy.EUCLIDEAN_DISTANCE: "",28    DistanceStrategy.DOT_PRODUCT: "DESC",29}30 31 32@deprecated(33    since="0.3.22",34    message=(35        "This class is pending deprecation and may be removed in a future version. "36        "You can swap to using the `SingleStoreVectorStore` "37        "implementation in `langchain_singlestore`. "38        "See <https://github.com/singlestore-labs/langchain-singlestore> for details "39        "about the new implementation."40    ),41    alternative="from langchain_singlestore import SingleStoreVectorStore",42    pending=True,43)44class SingleStoreDB(VectorStore):45    """`SingleStore DB` vector store.46 47    The prerequisite for using this class is the installation of the ``singlestoredb``48    Python package.49 50    The SingleStoreDB vectorstore can be created by providing an embedding function and51    the relevant parameters for the database connection, connection pool, and52    optionally, the names of the table and the fields to use.53    """54 55    class SearchStrategy(str, Enum):56        """Enumerator of the Search strategies for searching in the vectorstore."""57 58        VECTOR_ONLY = "VECTOR_ONLY"59        TEXT_ONLY = "TEXT_ONLY"60        FILTER_BY_TEXT = "FILTER_BY_TEXT"61        FILTER_BY_VECTOR = "FILTER_BY_VECTOR"62        WEIGHTED_SUM = "WEIGHTED_SUM"63 64    def _get_connection(self: SingleStoreDB) -> Any:65        try:66            import singlestoredb as s267        except ImportError:68            raise ImportError(69                "Could not import singlestoredb python package. "70                "Please install it with `pip install singlestoredb`."71            )72        return s2.connect(**self.connection_kwargs)73 74    def __init__(75        self,76        embedding: Embeddings,77        *,78        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,79        table_name: str = "embeddings",80        content_field: str = "content",81        metadata_field: str = "metadata",82        vector_field: str = "vector",83        id_field: str = "id",84        use_vector_index: bool = False,85        vector_index_name: str = "",86        vector_index_options: Optional[dict] = None,87        vector_size: int = 1536,88        use_full_text_search: bool = False,89        pool_size: int = 5,90        max_overflow: int = 10,91        timeout: float = 30,92        **kwargs: Any,93    ):94        """Initialize with necessary components.95 96        Args:97            embedding (Embeddings): A text embedding model.98 99            distance_strategy (DistanceStrategy, optional):100                Determines the strategy employed for calculating101                the distance between vectors in the embedding space.102                Defaults to DOT_PRODUCT.103                Available options are:104                - DOT_PRODUCT: Computes the scalar product of two vectors.105                    This is the default behavior106                - EUCLIDEAN_DISTANCE: Computes the Euclidean distance between107                    two vectors. This metric considers the geometric distance in108                    the vector space, and might be more suitable for embeddings109                    that rely on spatial relationships. This metric is not110                    compatible with the WEIGHTED_SUM search strategy.111 112            table_name (str, optional): Specifies the name of the table in use.113                Defaults to "embeddings".114            content_field (str, optional): Specifies the field to store the content.115                Defaults to "content".116            metadata_field (str, optional): Specifies the field to store metadata.117                Defaults to "metadata".118            vector_field (str, optional): Specifies the field to store the vector.119                Defaults to "vector".120            id_field (str, optional): Specifies the field to store the id.121                Defaults to "id".122 123            use_vector_index (bool, optional): Toggles the use of a vector index.124                Works only with SingleStoreDB 8.5 or later. Defaults to False.125                If set to True, vector_size parameter is required to be set to126                a proper value.127 128            vector_index_name (str, optional): Specifies the name of the vector index.129                Defaults to empty. Will be ignored if use_vector_index is set to False.130 131            vector_index_options (dict, optional): Specifies the options for132                the vector index. Defaults to {}.133                Will be ignored if use_vector_index is set to False. The options are:134                index_type (str, optional): Specifies the type of the index.135                    Defaults to IVF_PQFS.136                For more options, please refer to the SingleStoreDB documentation:137                https://docs.singlestore.com/cloud/reference/sql-reference/vector-functions/vector-indexing/138 139            vector_size (int, optional): Specifies the size of the vector.140                Defaults to 1536. Required if use_vector_index is set to True.141                Should be set to the same value as the size of the vectors142                stored in the vector_field.143 144            use_full_text_search (bool, optional): Toggles the use a full-text index145                on the document content. Defaults to False. If set to True, the table146                will be created with a full-text index on the content field,147                and the simularity_search method will all using TEXT_ONLY,148                FILTER_BY_TEXT, FILTER_BY_VECTOR, and WIGHTED_SUM search strategies.149                If set to False, the simularity_search method will only allow150                VECTOR_ONLY search strategy.151 152            Following arguments pertain to the connection pool:153 154            pool_size (int, optional): Determines the number of active connections in155                the pool. Defaults to 5.156            max_overflow (int, optional): Determines the maximum number of connections157                allowed beyond the pool_size. Defaults to 10.158            timeout (float, optional): Specifies the maximum wait time in seconds for159                establishing a connection. Defaults to 30.160 161            Following arguments pertain to the database connection:162 163            host (str, optional): Specifies the hostname, IP address, or URL for the164                database connection. The default scheme is "mysql".165            user (str, optional): Database username.166            password (str, optional): Database password.167            port (int, optional): Database port. Defaults to 3306 for non-HTTP168                connections, 80 for HTTP connections, and 443 for HTTPS connections.169            database (str, optional): Database name.170 171            Additional optional arguments provide further customization over the172            database connection:173 174            pure_python (bool, optional): Toggles the connector mode. If True,175                operates in pure Python mode.176            local_infile (bool, optional): Allows local file uploads.177            charset (str, optional): Specifies the character set for string values.178            ssl_key (str, optional): Specifies the path of the file containing the SSL179                key.180            ssl_cert (str, optional): Specifies the path of the file containing the SSL181                certificate.182            ssl_ca (str, optional): Specifies the path of the file containing the SSL183                certificate authority.184            ssl_cipher (str, optional): Sets the SSL cipher list.185            ssl_disabled (bool, optional): Disables SSL usage.186            ssl_verify_cert (bool, optional): Verifies the server's certificate.187                Automatically enabled if ``ssl_ca`` is specified.188            ssl_verify_identity (bool, optional): Verifies the server's identity.189            conv (dict[int, Callable], optional): A dictionary of data conversion190                functions.191            credential_type (str, optional): Specifies the type of authentication to192                use: auth.PASSWORD, auth.JWT, or auth.BROWSER_SSO.193            autocommit (bool, optional): Enables autocommits.194            results_type (str, optional): Determines the structure of the query results:195                tuples, namedtuples, dicts.196            results_format (str, optional): Deprecated. This option has been renamed to197                results_type.198 199        Examples:200            Basic Usage:201 202            .. code-block:: python203 204                from langchain_openai import OpenAIEmbeddings205                from langchain_community.vectorstores import SingleStoreDB206 207                vectorstore = SingleStoreDB(208                    OpenAIEmbeddings(),209                    host="https://user:password@127.0.0.1:3306/database"210                )211 212            Advanced Usage:213 214            .. code-block:: python215 216                from langchain_openai import OpenAIEmbeddings217                from langchain_community.vectorstores import SingleStoreDB218 219                vectorstore = SingleStoreDB(220                    OpenAIEmbeddings(),221                    distance_strategy=DistanceStrategy.EUCLIDEAN_DISTANCE,222                    host="127.0.0.1",223                    port=3306,224                    user="user",225                    password="password",226                    database="db",227                    table_name="my_custom_table",228                    pool_size=10,229                    timeout=60,230                )231 232            Using environment variables:233 234            .. code-block:: python235 236                from langchain_openai import OpenAIEmbeddings237                from langchain_community.vectorstores import SingleStoreDB238 239                os.environ['SINGLESTOREDB_URL'] = 'me:p455w0rd@s2-host.com/my_db'240                vectorstore = SingleStoreDB(OpenAIEmbeddings())241 242            Using vector index:243 244            .. code-block:: python245 246                from langchain_openai import OpenAIEmbeddings247                from langchain_community.vectorstores import SingleStoreDB248 249                os.environ['SINGLESTOREDB_URL'] = 'me:p455w0rd@s2-host.com/my_db'250                vectorstore = SingleStoreDB(251                    OpenAIEmbeddings(),252                    use_vector_index=True,253                )254 255            Using full-text index:256 257            .. code-block:: python258                from langchain_openai import OpenAIEmbeddings259                from langchain_community.vectorstores import SingleStoreDB260 261                os.environ['SINGLESTOREDB_URL'] = 'me:p455w0rd@s2-host.com/my_db'262                vectorstore = SingleStoreDB(263                    OpenAIEmbeddings(),264                    use_full_text_search=True,265                )266        """267 268        self.embedding = embedding269        self.distance_strategy = distance_strategy270        self.table_name = self._sanitize_input(table_name)271        self.content_field = self._sanitize_input(content_field)272        self.metadata_field = self._sanitize_input(metadata_field)273        self.vector_field = self._sanitize_input(vector_field)274        self.id_field = self._sanitize_input(id_field)275 276        self.use_vector_index = bool(use_vector_index)277        self.vector_index_name = self._sanitize_input(vector_index_name)278        self.vector_index_options = dict(vector_index_options or {})279        self.vector_index_options["metric_type"] = self.distance_strategy280        self.vector_size = int(vector_size)281 282        self.use_full_text_search = bool(use_full_text_search)283 284        # Pass the rest of the kwargs to the connection.285        self.connection_kwargs = kwargs286 287        # Add program name and version to connection attributes.288        if "conn_attrs" not in self.connection_kwargs:289            self.connection_kwargs["conn_attrs"] = dict()290 291        self.connection_kwargs["conn_attrs"]["_connector_name"] = "langchain python sdk"292        self.connection_kwargs["conn_attrs"]["_connector_version"] = "2.1.0"293 294        # Create connection pool.295        self.connection_pool = QueuePool(296            self._get_connection,297            max_overflow=max_overflow,298            pool_size=pool_size,299            timeout=timeout,300        )301        self._create_table()302 303    @property304    def embeddings(self) -> Embeddings:305        return self.embedding306 307    def _sanitize_input(self, input_str: str) -> str:308        # Remove characters that are not alphanumeric or underscores309        return re.sub(r"[^a-zA-Z0-9_]", "", input_str)310 311    def _select_relevance_score_fn(self) -> Callable[[float], float]:312        return self._max_inner_product_relevance_score_fn313 314    def _create_table(self: SingleStoreDB) -> None:315        """Create table if it doesn't exist."""316        conn = self.connection_pool.connect()317        try:318            cur = conn.cursor()319            try:320                full_text_index = ""321                if self.use_full_text_search:322                    full_text_index = ", FULLTEXT({})".format(self.content_field)323                if self.use_vector_index:324                    index_options = ""325                    if self.vector_index_options and len(self.vector_index_options) > 0:326                        index_options = "INDEX_OPTIONS '{}'".format(327                            json.dumps(self.vector_index_options)328                        )329                    cur.execute(330                        """CREATE TABLE IF NOT EXISTS {}331                        ({} BIGINT AUTO_INCREMENT PRIMARY KEY, {} LONGTEXT CHARACTER332                        SET utf8mb4 COLLATE utf8mb4_general_ci, {} VECTOR({}, F32)333                        NOT NULL, {} JSON, VECTOR INDEX {} ({}) {}{});""".format(334                            self.table_name,335                            self.id_field,336                            self.content_field,337                            self.vector_field,338                            self.vector_size,339                            self.metadata_field,340                            self.vector_index_name,341                            self.vector_field,342                            index_options,343                            full_text_index,344                        ),345                    )346                else:347                    cur.execute(348                        """CREATE TABLE IF NOT EXISTS {}349                        ({} BIGINT AUTO_INCREMENT PRIMARY KEY, {} LONGTEXT CHARACTER350                        SET utf8mb4 COLLATE utf8mb4_general_ci, {} BLOB, {} JSON{});351                        """.format(352                            self.table_name,353                            self.id_field,354                            self.content_field,355                            self.vector_field,356                            self.metadata_field,357                            full_text_index,358                        ),359                    )360            finally:361                cur.close()362        finally:363            conn.close()364 365    def add_images(366        self,367        uris: List[str],368        metadatas: Optional[List[dict]] = None,369        embeddings: Optional[List[List[float]]] = None,370        return_ids: bool = False,371        **kwargs: Any,372    ) -> List[str]:373        """Run images through the embeddings and add to the vectorstore.374 375        Args:376            uris List[str]: File path to images.377                Each URI will be added to the vectorstore as document content.378            metadatas (Optional[List[dict]], optional): Optional list of metadatas.379                Defaults to None.380            embeddings (Optional[List[List[float]]], optional): Optional pre-generated381                embeddings. Defaults to None.382 383        Returns:384            List[str]: list of document ids added to the vectorstore385                if return_ids is True. Otherwise, an empty list.386        """387        # Set embeddings388        if (389            embeddings is None390            and self.embedding is not None391            and hasattr(self.embedding, "embed_image")392        ):393            embeddings = self.embedding.embed_image(uris=uris)394        return self.add_texts(395            uris, metadatas, embeddings, return_ids=return_ids, **kwargs396        )397 398    def add_texts(399        self,400        texts: Iterable[str],401        metadatas: Optional[List[dict]] = None,402        embeddings: Optional[List[List[float]]] = None,403        return_ids: bool = False,404        **kwargs: Any,405    ) -> List[str]:406        """Add more texts to the vectorstore.407 408        Args:409            texts (Iterable[str]): Iterable of strings/text to add to the vectorstore.410            metadatas (Optional[List[dict]], optional): Optional list of metadatas.411                Defaults to None.412            embeddings (Optional[List[List[float]]], optional): Optional pre-generated413                embeddings. Defaults to None.414 415        Returns:416            List[str]: list of document ids added to the vectorstore417                if return_ids is True. Otherwise, an empty list.418        """419        ids: List[str] = []420        conn = self.connection_pool.connect()421        try:422            cur = conn.cursor()423            try:424                # Write data to singlestore db425                for i, text in enumerate(texts):426                    # Use provided values by default or fallback427                    metadata = metadatas[i] if metadatas else {}428                    embedding = (429                        embeddings[i]430                        if embeddings431                        else self.embedding.embed_documents([text])[0]432                    )433                    cur.execute(434                        """INSERT INTO {}({}, {}, {})435                        VALUES (%s, JSON_ARRAY_PACK(%s), %s)""".format(436                            self.table_name,437                            self.content_field,438                            self.vector_field,439                            self.metadata_field,440                        ),441                        (442                            text,443                            "[{}]".format(",".join(map(str, embedding))),444                            json.dumps(metadata),445                        ),446                    )447                    if return_ids:448                        cur.execute("SELECT LAST_INSERT_ID();")449                        row = cur.fetchone()450                        if row:451                            ids.append(str(row[0]))452                if self.use_vector_index or self.use_full_text_search:453                    cur.execute("OPTIMIZE TABLE {} FLUSH;".format(self.table_name))454            finally:455                cur.close()456        finally:457            conn.close()458        return ids459 460    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> bool | None:461        """Delete documents from the vectorstore.462 463        Args:464            ids (List[str], optional): List of document ids to delete.465                If None, all documents will be deleted. Defaults to None.466 467        Returns:468            bool: True if deletion was successful, False otherwise.469        """470        if ids is None:471            return True472 473        conn = self.connection_pool.connect()474        try:475            cur = conn.cursor()476            try:477                cur.execute(478                    "DELETE FROM {} WHERE {} IN ({})".format(479                        self.table_name, self.id_field, ",".join(ids)480                    )481                )482                if self.use_vector_index or self.use_full_text_search:483                    cur.execute("OPTIMIZE TABLE {} FLUSH;".format(self.table_name))484            finally:485                cur.close()486        finally:487            conn.close()488        return True489 490    def similarity_search(491        self,492        query: str,493        k: int = 4,494        filter: Optional[dict] = None,495        search_strategy: SearchStrategy = SearchStrategy.VECTOR_ONLY,496        filter_threshold: float = 0,497        text_weight: float = 0.5,498        vector_weight: float = 0.5,499        vector_select_count_multiplier: int = 10,500        **kwargs: Any,501    ) -> List[Document]:502        """Returns the most similar indexed documents to the query text.503 504        Uses cosine similarity.505 506        Args:507            query (str): The query text for which to find similar documents.508            k (int): The number of documents to return. Default is 4.509            filter (dict): A dictionary of metadata fields and values to filter by.510                Default is None.511            search_strategy (SearchStrategy): The search strategy to use.512                Default is SearchStrategy.VECTOR_ONLY.513                Available options are:514                - SearchStrategy.VECTOR_ONLY: Searches only by vector similarity.515                - SearchStrategy.TEXT_ONLY: Searches only by text similarity. This516                    option is only available if use_full_text_search is True.517                - SearchStrategy.FILTER_BY_TEXT: Filters by text similarity and518                    searches by vector similarity. This option is only available if519                    use_full_text_search is True.520                - SearchStrategy.FILTER_BY_VECTOR: Filters by vector similarity and521                    searches by text similarity. This option is only available if522                    use_full_text_search is True.523                - SearchStrategy.WEIGHTED_SUM: Searches by a weighted sum of text and524                    vector similarity. This option is only available if525                    use_full_text_search is True and distance_strategy is DOT_PRODUCT.526            filter_threshold (float): The threshold for filtering by text or vector527                similarity. Default is 0. This option has effect only if search_strategy528                is SearchStrategy.FILTER_BY_TEXT or SearchStrategy.FILTER_BY_VECTOR.529            text_weight (float): The weight of text similarity in the weighted sum530                search strategy. Default is 0.5. This option has effect only if531                search_strategy is SearchStrategy.WEIGHTED_SUM.532            vector_weight (float): The weight of vector similarity in the weighted sum533                search strategy. Default is 0.5. This option has effect only if534                search_strategy is SearchStrategy.WEIGHTED_SUM.535            vector_select_count_multiplier (int): The multiplier for the number of536                vectors to select when using the vector index. Default is 10.537                This parameter has effect only if use_vector_index is True and538                search_strategy is SearchStrategy.WEIGHTED_SUM or539                SearchStrategy.FILTER_BY_TEXT.540                The number of vectors selected will541                be k * vector_select_count_multiplier.542                This is needed due to the limitations of the vector index.543 544 545        Returns:546            List[Document]: A list of documents that are most similar to the query text.547 548        Examples:549 550            Basic Usage:551            .. code-block:: python552 553                from langchain_community.vectorstores import SingleStoreDB554                from langchain_openai import OpenAIEmbeddings555 556                s2 = SingleStoreDB.from_documents(557                    docs,558                    OpenAIEmbeddings(),559                    host="username:password@localhost:3306/database"560                )561                results = s2.similarity_search("query text", 1,562                                    {"metadata_field": "metadata_value"})563 564            Different Search Strategies:565            .. code-block:: python566 567                from langchain_community.vectorstores import SingleStoreDB568                from langchain_openai import OpenAIEmbeddings569 570                s2 = SingleStoreDB.from_documents(571                    docs,572                    OpenAIEmbeddings(),573                    host="username:password@localhost:3306/database",574                    use_full_text_search=True,575                    use_vector_index=True,576                )577                results = s2.similarity_search("query text", 1,578                        search_strategy=SingleStoreDB.SearchStrategy.FILTER_BY_TEXT,579                        filter_threshold=0.5)580 581            Weighted Sum Search Strategy:582            .. code-block:: python583 584                from langchain_community.vectorstores import SingleStoreDB585                from langchain_openai import OpenAIEmbeddings586 587                s2 = SingleStoreDB.from_documents(588                    docs,589                    OpenAIEmbeddings(),590                    host="username:password@localhost:3306/database",591                    use_full_text_search=True,592                    use_vector_index=True,593                )594                results = s2.similarity_search("query text", 1,595                    search_strategy=SingleStoreDB.SearchStrategy.WEIGHTED_SUM,596                    text_weight=0.3,597                    vector_weight=0.7)598        """599        docs_and_scores = self.similarity_search_with_score(600            query=query,601            k=k,602            filter=filter,603            search_strategy=search_strategy,604            filter_threshold=filter_threshold,605            text_weight=text_weight,606            vector_weight=vector_weight,607            vector_select_count_multiplier=vector_select_count_multiplier,608            **kwargs,609        )610        return [doc for doc, _ in docs_and_scores]611 612    def similarity_search_with_score(613        self,614        query: str,615        k: int = 4,616        filter: Optional[dict] = None,617        search_strategy: SearchStrategy = SearchStrategy.VECTOR_ONLY,618        filter_threshold: float = 1,619        text_weight: float = 0.5,620        vector_weight: float = 0.5,621        vector_select_count_multiplier: int = 10,622        **kwargs: Any,623    ) -> List[Tuple[Document, float]]:624        """Return docs most similar to query. Uses cosine similarity.625 626        Args:627            query: Text to look up documents similar to.628            k: Number of Documents to return. Defaults to 4.629            filter: A dictionary of metadata fields and values to filter by.630                    Defaults to None.631            search_strategy (SearchStrategy): The search strategy to use.632                Default is SearchStrategy.VECTOR_ONLY.633                Available options are:634                - SearchStrategy.VECTOR_ONLY: Searches only by vector similarity.635                - SearchStrategy.TEXT_ONLY: Searches only by text similarity. This636                    option is only available if use_full_text_search is True.637                - SearchStrategy.FILTER_BY_TEXT: Filters by text similarity and638                    searches by vector similarity. This option is only available if639                    use_full_text_search is True.640                - SearchStrategy.FILTER_BY_VECTOR: Filters by vector similarity and641                    searches by text similarity. This option is only available if642                    use_full_text_search is True.643                - SearchStrategy.WEIGHTED_SUM: Searches by a weighted sum of text and644                    vector similarity. This option is only available if645                    use_full_text_search is True and distance_strategy is DOT_PRODUCT.646            filter_threshold (float): The threshold for filtering by text or vector647                similarity. Default is 0. This option has effect only if search_strategy648                is SearchStrategy.FILTER_BY_TEXT or SearchStrategy.FILTER_BY_VECTOR.649            text_weight (float): The weight of text similarity in the weighted sum650                search strategy. Default is 0.5. This option has effect only if651                search_strategy is SearchStrategy.WEIGHTED_SUM.652            vector_weight (float): The weight of vector similarity in the weighted sum653                search strategy. Default is 0.5. This option has effect only if654                search_strategy is SearchStrategy.WEIGHTED_SUM.655            vector_select_count_multiplier (int): The multiplier for the number of656                vectors to select when using the vector index. Default is 10.657                This parameter has effect only if use_vector_index is True and658                search_strategy is SearchStrategy.WEIGHTED_SUM or659                SearchStrategy.FILTER_BY_TEXT.660                The number of vectors selected will661                be k * vector_select_count_multiplier.662                This is needed due to the limitations of the vector index.663        Returns:664            List of Documents most similar to the query and score for each665            document.666 667        Raises:668            ValueError: If the search strategy is not supported with the669                distance strategy.670 671        Examples:672            Basic Usage:673            .. code-block:: python674 675                from langchain_community.vectorstores import SingleStoreDB676                from langchain_openai import OpenAIEmbeddings677 678                s2 = SingleStoreDB.from_documents(679                    docs,680                    OpenAIEmbeddings(),681                    host="username:password@localhost:3306/database"682                )683                results = s2.similarity_search_with_score("query text", 1,684                                    {"metadata_field": "metadata_value"})685 686            Different Search Strategies:687 688            .. code-block:: python689 690                from langchain_community.vectorstores import SingleStoreDB691                from langchain_openai import OpenAIEmbeddings692 693                s2 = SingleStoreDB.from_documents(694                    docs,695                    OpenAIEmbeddings(),696                    host="username:password@localhost:3306/database",697                    use_full_text_search=True,698                    use_vector_index=True,699                )700                results = s2.similarity_search_with_score("query text", 1,701                        search_strategy=SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR,702                        filter_threshold=0.5)703 704            Weighted Sum Search Strategy:705            .. code-block:: python706 707                from langchain_community.vectorstores import SingleStoreDB708                from langchain_openai import OpenAIEmbeddings709 710                s2 = SingleStoreDB.from_documents(711                    docs,712                    OpenAIEmbeddings(),713                    host="username:password@localhost:3306/database",714                    use_full_text_search=True,715                    use_vector_index=True,716                )717                results = s2.similarity_search_with_score("query text", 1,718                    search_strategy=SingleStoreDB.SearchStrategy.WEIGHTED_SUM,719                    text_weight=0.3,720                    vector_weight=0.7)721        """722 723        if (724            search_strategy != SingleStoreDB.SearchStrategy.VECTOR_ONLY725            and not self.use_full_text_search726        ):727            raise ValueError(728                """Search strategy {} is not supported729                when use_full_text_search is False""".format(search_strategy)730            )731 732        if (733            search_strategy == SingleStoreDB.SearchStrategy.WEIGHTED_SUM734            and self.distance_strategy != DistanceStrategy.DOT_PRODUCT735        ):736            raise ValueError(737                "Search strategy {} is not supported with distance strategy {}".format(738                    search_strategy, self.distance_strategy739                )740            )741 742        # Creates embedding vector from user query743        embedding = []744        if search_strategy != SingleStoreDB.SearchStrategy.TEXT_ONLY:745            embedding = self.embedding.embed_query(query)746 747        self.embedding.embed_query(query)748        conn = self.connection_pool.connect()749        result = []750        where_clause: str = ""751        where_clause_values: List[Any] = []752        if filter or search_strategy in [753            SingleStoreDB.SearchStrategy.FILTER_BY_TEXT,754            SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR,755        ]:756            where_clause = "WHERE "757            arguments = []758 759            if search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_TEXT:760                arguments.append(761                    "MATCH ({}) AGAINST (%s) > %s".format(self.content_field)762                )763                where_clause_values.append(query)764                where_clause_values.append(float(filter_threshold))765 766            if search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR:767                condition = "{}({}, JSON_ARRAY_PACK(%s)) ".format(768                    self.distance_strategy.name769                    if isinstance(self.distance_strategy, DistanceStrategy)770                    else self.distance_strategy,771                    self.vector_field,772                )773                if self.distance_strategy == DistanceStrategy.EUCLIDEAN_DISTANCE:774                    condition += "< %s"775                else:776                    condition += "> %s"777                arguments.append(condition)778                where_clause_values.append("[{}]".format(",".join(map(str, embedding))))779                where_clause_values.append(float(filter_threshold))780 781            def build_where_clause(782                where_clause_values: List[Any],783                sub_filter: dict,784                prefix_args: Optional[List[str]] = None,785            ) -> None:786                prefix_args = prefix_args or []787                for key in sub_filter.keys():788                    if isinstance(sub_filter[key], dict):789                        build_where_clause(790                            where_clause_values, sub_filter[key], prefix_args + [key]791                        )792                    else:793                        arguments.append(794                            "JSON_EXTRACT_JSON({}, {}) = %s".format(795                                self.metadata_field,796                                ", ".join(["%s"] * (len(prefix_args) + 1)),797                            )798                        )799                        where_clause_values += prefix_args + [key]800                        where_clause_values.append(json.dumps(sub_filter[key]))801 802            if filter:803                build_where_clause(where_clause_values, filter)804            where_clause += " AND ".join(arguments)805 806        try:807            cur = conn.cursor()808            try:809                if (810                    search_strategy == SingleStoreDB.SearchStrategy.VECTOR_ONLY811                    or search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_TEXT812                ):813                    search_options = ""814                    if (815                        self.use_vector_index816                        and search_strategy817                        == SingleStoreDB.SearchStrategy.FILTER_BY_TEXT818                    ):819                        search_options = "SEARCH_OPTIONS '{\"k\":%d}'" % (820                            k * vector_select_count_multiplier821                        )822                    cur.execute(823                        """SELECT {}, {}, {}({}, JSON_ARRAY_PACK(%s)) as __score824                        FROM {} {} ORDER BY __score {}{} LIMIT %s""".format(825                            self.content_field,826                            self.metadata_field,827                            self.distance_strategy.name828                            if isinstance(self.distance_strategy, DistanceStrategy)829                            else self.distance_strategy,830                            self.vector_field,831                            self.table_name,832                            where_clause,833                            search_options,834                            ORDERING_DIRECTIVE[self.distance_strategy],835                        ),836                        ("[{}]".format(",".join(map(str, embedding))),)837                        + tuple(where_clause_values)838                        + (k,),839                    )840                elif (841                    search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR842                    or search_strategy == SingleStoreDB.SearchStrategy.TEXT_ONLY843                ):844                    cur.execute(845                        """SELECT {}, {}, MATCH ({}) AGAINST (%s) as __score846                        FROM {} {} ORDER BY __score DESC LIMIT %s""".format(847                            self.content_field,848                            self.metadata_field,849                            self.content_field,850                            self.table_name,851                            where_clause,852                        ),853                        (query,) + tuple(where_clause_values) + (k,),854                    )855                elif search_strategy == SingleStoreDB.SearchStrategy.WEIGHTED_SUM:856                    cur.execute(857                        """SELECT {}, {}, __score1 * %s + __score2 * %s as __score858                        FROM (859                            SELECT {}, {}, {}, MATCH ({}) AGAINST (%s) as __score1 860                        FROM {} {}) r1 FULL OUTER JOIN (861                            SELECT {}, {}({}, JSON_ARRAY_PACK(%s)) as __score2862                            FROM {} {} ORDER BY __score2 {} LIMIT %s863                        ) r2 ON r1.{} = r2.{} ORDER BY __score {} LIMIT %s""".format(864                            self.content_field,865                            self.metadata_field,866                            self.id_field,867                            self.content_field,868                            self.metadata_field,869                            self.content_field,870                            self.table_name,871                            where_clause,872                            self.id_field,873                            self.distance_strategy.name874                            if isinstance(self.distance_strategy, DistanceStrategy)875                            else self.distance_strategy,876                            self.vector_field,877                            self.table_name,878                            where_clause,879                            ORDERING_DIRECTIVE[self.distance_strategy],880                            self.id_field,881                            self.id_field,882                            ORDERING_DIRECTIVE[self.distance_strategy],883                        ),884                        (text_weight, vector_weight, query)885                        + tuple(where_clause_values)886                        + ("[{}]".format(",".join(map(str, embedding))),)887                        + tuple(where_clause_values)888                        + (k * vector_select_count_multiplier, k),889                    )890                else:891                    raise ValueError(892                        "Invalid search strategy: {}".format(search_strategy)893                    )894 895                for row in cur.fetchall():896                    doc = Document(page_content=row[0], metadata=row[1])897                    result.append((doc, float(row[2])))898            finally:899                cur.close()900        finally:901            conn.close()902        return result903 904    @classmethod905    def from_texts(906        cls: Type[SingleStoreDB],907        texts: List[str],908        embedding: Embeddings,909        metadatas: Optional[List[dict]] = None,910        distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,911        table_name: str = "embeddings",912        content_field: str = "content",913        metadata_field: str = "metadata",914        vector_field: str = "vector",915        id_field: str = "id",916        use_vector_index: bool = False,917        vector_index_name: str = "",918        vector_index_options: Optional[dict] = None,919        vector_size: int = 1536,920        use_full_text_search: bool = False,921        pool_size: int = 5,922        max_overflow: int = 10,923        timeout: float = 30,924        **kwargs: Any,925    ) -> SingleStoreDB:926        """Create a SingleStoreDB vectorstore from raw documents.927        This is a user-friendly interface that:928            1. Embeds documents.929            2. Creates a new table for the embeddings in SingleStoreDB.930            3. Adds the documents to the newly created table.931        This is intended to be a quick way to get started.932        Args:933            texts (List[str]): List of texts to add to the vectorstore.934            embedding (Embeddings): A text embedding model.935            metadatas (Optional[List[dict]], optional): Optional list of metadatas.936                Defaults to None.937            distance_strategy (DistanceStrategy, optional):938                Determines the strategy employed for calculating939                the distance between vectors in the embedding space.940                Defaults to DOT_PRODUCT.941                Available options are:942                - DOT_PRODUCT: Computes the scalar product of two vectors.943                    This is the default behavior944                - EUCLIDEAN_DISTANCE: Computes the Euclidean distance between945                    two vectors. This metric considers the geometric distance in946                    the vector space, and might be more suitable for embeddings947                    that rely on spatial relationships. This metric is not948                    compatible with the WEIGHTED_SUM search strategy.949            table_name (str, optional): Specifies the name of the table in use.950                Defaults to "embeddings".951            content_field (str, optional): Specifies the field to store the content.952                Defaults to "content".953            metadata_field (str, optional): Specifies the field to store metadata.954                Defaults to "metadata".955            vector_field (str, optional): Specifies the field to store the vector.956                Defaults to "vector".957            id_field (str, optional): Specifies the field to store the id.958                Defaults to "id".959            use_vector_index (bool, optional): Toggles the use of a vector index.960                Works only with SingleStoreDB 8.5 or later. Defaults to False.961                If set to True, vector_size parameter is required to be set to962                a proper value.963            vector_index_name (str, optional): Specifies the name of the vector index.964                Defaults to empty. Will be ignored if use_vector_index is set to False.965            vector_index_options (dict, optional): Specifies the options for966                the vector index. Defaults to {}.967                Will be ignored if use_vector_index is set to False. The options are:968                index_type (str, optional): Specifies the type of the index.969                    Defaults to IVF_PQFS.970                For more options, please refer to the SingleStoreDB documentation:971                https://docs.singlestore.com/cloud/reference/sql-reference/vector-functions/vector-indexing/972            vector_size (int, optional): Specifies the size of the vector.973                Defaults to 1536. Required if use_vector_index is set to True.974                Should be set to the same value as the size of the vectors975                stored in the vector_field.976            use_full_text_search (bool, optional): Toggles the use a full-text index977                on the document content. Defaults to False. If set to True, the table978                will be created with a full-text index on the content field,979                and the simularity_search method will all using TEXT_ONLY,980                FILTER_BY_TEXT, FILTER_BY_VECTOR, and WIGHTED_SUM search strategies.981                If set to False, the simularity_search method will only allow982                VECTOR_ONLY search strategy.983 984            pool_size (int, optional): Determines the number of active connections in985                the pool. Defaults to 5.986            max_overflow (int, optional): Determines the maximum number of connections987                allowed beyond the pool_size. Defaults to 10.988            timeout (float, optional): Specifies the maximum wait time in seconds for989                establishing a connection. Defaults to 30.990 991            Additional optional arguments provide further customization over the992            database connection:993 994            pure_python (bool, optional): Toggles the connector mode. If True,995                operates in pure Python mode.996            local_infile (bool, optional): Allows local file uploads.997            charset (str, optional): Specifies the character set for string values.998            ssl_key (str, optional): Specifies the path of the file containing the SSL999                key.1000            ssl_cert (str, optional): Specifies the path of the file containing the SSL1001                certificate.1002            ssl_ca (str, optional): Specifies the path of the file containing the SSL1003                certificate authority.1004            ssl_cipher (str, optional): Sets the SSL cipher list.1005            ssl_disabled (bool, optional): Disables SSL usage.1006            ssl_verify_cert (bool, optional): Verifies the server's certificate.1007                Automatically enabled if ``ssl_ca`` is specified.1008            ssl_verify_identity (bool, optional): Verifies the server's identity.1009            conv (dict[int, Callable], optional): A dictionary of data conversion1010                functions.1011            credential_type (str, optional): Specifies the type of authentication to1012                use: auth.PASSWORD, auth.JWT, or auth.BROWSER_SSO.1013            autocommit (bool, optional): Enables autocommits.1014            results_type (str, optional): Determines the structure of the query results:1015                tuples, namedtuples, dicts.1016            results_format (str, optional): Deprecated. This option has been renamed to1017                results_type.1018 1019        Example:1020            .. code-block:: python1021 1022                from langchain_community.vectorstores import SingleStoreDB1023                from langchain_openai import OpenAIEmbeddings1024 1025                s2 = SingleStoreDB.from_texts(1026                    texts,1027                    OpenAIEmbeddings(),1028                    host="username:password@localhost:3306/database"1029                )1030        """1031 1032        instance = cls(1033            embedding,1034            distance_strategy=distance_strategy,1035            table_name=table_name,1036            content_field=content_field,1037            metadata_field=metadata_field,1038            vector_field=vector_field,1039            id_field=id_field,1040            pool_size=pool_size,1041            max_overflow=max_overflow,1042            timeout=timeout,1043            use_vector_index=use_vector_index,1044            vector_index_name=vector_index_name,1045            vector_index_options=vector_index_options,1046            vector_size=vector_size,1047            use_full_text_search=use_full_text_search,1048            **kwargs,1049        )1050        instance.add_texts(texts, metadatas, embedding.embed_documents(texts), **kwargs)1051        return instance1052 1053    def drop(self) -> None:1054        """Drop the table and delete all data from the vectorstore.1055        Vector store will be unusable after this operation.1056        """1057        conn = self.connection_pool.connect()1058        try:1059            cur = conn.cursor()1060            try:1061                cur.execute("DROP TABLE IF EXISTS {}".format(self.table_name))1062            finally:1063                cur.close()1064        finally:1065            conn.close()1066 1067 1068# SingleStoreDBRetriever is not needed, but we keep it for backwards compatibility1069SingleStoreDBRetriever = VectorStoreRetriever1070 
codekingpro/portable-devtools · Team Ai