codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import json4import re5from enum import Enum6from typing import (7 Any,8 Callable,9 Iterable,10 List,11 Optional,12 Tuple,13 Type,14)15 16from langchain_core._api import deprecated17from langchain_core.documents import Document18from langchain_core.embeddings import Embeddings19from langchain_core.vectorstores import VectorStore, VectorStoreRetriever20from sqlalchemy.pool import QueuePool21 22from langchain_community.vectorstores.utils import DistanceStrategy23 24DEFAULT_DISTANCE_STRATEGY = DistanceStrategy.DOT_PRODUCT25 26ORDERING_DIRECTIVE: dict = {27 DistanceStrategy.EUCLIDEAN_DISTANCE: "",28 DistanceStrategy.DOT_PRODUCT: "DESC",29}30 31 32@deprecated(33 since="0.3.22",34 message=(35 "This class is pending deprecation and may be removed in a future version. "36 "You can swap to using the `SingleStoreVectorStore` "37 "implementation in `langchain_singlestore`. "38 "See <https://github.com/singlestore-labs/langchain-singlestore> for details "39 "about the new implementation."40 ),41 alternative="from langchain_singlestore import SingleStoreVectorStore",42 pending=True,43)44class SingleStoreDB(VectorStore):45 """`SingleStore DB` vector store.46 47 The prerequisite for using this class is the installation of the ``singlestoredb``48 Python package.49 50 The SingleStoreDB vectorstore can be created by providing an embedding function and51 the relevant parameters for the database connection, connection pool, and52 optionally, the names of the table and the fields to use.53 """54 55 class SearchStrategy(str, Enum):56 """Enumerator of the Search strategies for searching in the vectorstore."""57 58 VECTOR_ONLY = "VECTOR_ONLY"59 TEXT_ONLY = "TEXT_ONLY"60 FILTER_BY_TEXT = "FILTER_BY_TEXT"61 FILTER_BY_VECTOR = "FILTER_BY_VECTOR"62 WEIGHTED_SUM = "WEIGHTED_SUM"63 64 def _get_connection(self: SingleStoreDB) -> Any:65 try:66 import singlestoredb as s267 except ImportError:68 raise ImportError(69 "Could not import singlestoredb python package. "70 "Please install it with `pip install singlestoredb`."71 )72 return s2.connect(**self.connection_kwargs)73 74 def __init__(75 self,76 embedding: Embeddings,77 *,78 distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,79 table_name: str = "embeddings",80 content_field: str = "content",81 metadata_field: str = "metadata",82 vector_field: str = "vector",83 id_field: str = "id",84 use_vector_index: bool = False,85 vector_index_name: str = "",86 vector_index_options: Optional[dict] = None,87 vector_size: int = 1536,88 use_full_text_search: bool = False,89 pool_size: int = 5,90 max_overflow: int = 10,91 timeout: float = 30,92 **kwargs: Any,93 ):94 """Initialize with necessary components.95 96 Args:97 embedding (Embeddings): A text embedding model.98 99 distance_strategy (DistanceStrategy, optional):100 Determines the strategy employed for calculating101 the distance between vectors in the embedding space.102 Defaults to DOT_PRODUCT.103 Available options are:104 - DOT_PRODUCT: Computes the scalar product of two vectors.105 This is the default behavior106 - EUCLIDEAN_DISTANCE: Computes the Euclidean distance between107 two vectors. This metric considers the geometric distance in108 the vector space, and might be more suitable for embeddings109 that rely on spatial relationships. This metric is not110 compatible with the WEIGHTED_SUM search strategy.111 112 table_name (str, optional): Specifies the name of the table in use.113 Defaults to "embeddings".114 content_field (str, optional): Specifies the field to store the content.115 Defaults to "content".116 metadata_field (str, optional): Specifies the field to store metadata.117 Defaults to "metadata".118 vector_field (str, optional): Specifies the field to store the vector.119 Defaults to "vector".120 id_field (str, optional): Specifies the field to store the id.121 Defaults to "id".122 123 use_vector_index (bool, optional): Toggles the use of a vector index.124 Works only with SingleStoreDB 8.5 or later. Defaults to False.125 If set to True, vector_size parameter is required to be set to126 a proper value.127 128 vector_index_name (str, optional): Specifies the name of the vector index.129 Defaults to empty. Will be ignored if use_vector_index is set to False.130 131 vector_index_options (dict, optional): Specifies the options for132 the vector index. Defaults to {}.133 Will be ignored if use_vector_index is set to False. The options are:134 index_type (str, optional): Specifies the type of the index.135 Defaults to IVF_PQFS.136 For more options, please refer to the SingleStoreDB documentation:137 https://docs.singlestore.com/cloud/reference/sql-reference/vector-functions/vector-indexing/138 139 vector_size (int, optional): Specifies the size of the vector.140 Defaults to 1536. Required if use_vector_index is set to True.141 Should be set to the same value as the size of the vectors142 stored in the vector_field.143 144 use_full_text_search (bool, optional): Toggles the use a full-text index145 on the document content. Defaults to False. If set to True, the table146 will be created with a full-text index on the content field,147 and the simularity_search method will all using TEXT_ONLY,148 FILTER_BY_TEXT, FILTER_BY_VECTOR, and WIGHTED_SUM search strategies.149 If set to False, the simularity_search method will only allow150 VECTOR_ONLY search strategy.151 152 Following arguments pertain to the connection pool:153 154 pool_size (int, optional): Determines the number of active connections in155 the pool. Defaults to 5.156 max_overflow (int, optional): Determines the maximum number of connections157 allowed beyond the pool_size. Defaults to 10.158 timeout (float, optional): Specifies the maximum wait time in seconds for159 establishing a connection. Defaults to 30.160 161 Following arguments pertain to the database connection:162 163 host (str, optional): Specifies the hostname, IP address, or URL for the164 database connection. The default scheme is "mysql".165 user (str, optional): Database username.166 password (str, optional): Database password.167 port (int, optional): Database port. Defaults to 3306 for non-HTTP168 connections, 80 for HTTP connections, and 443 for HTTPS connections.169 database (str, optional): Database name.170 171 Additional optional arguments provide further customization over the172 database connection:173 174 pure_python (bool, optional): Toggles the connector mode. If True,175 operates in pure Python mode.176 local_infile (bool, optional): Allows local file uploads.177 charset (str, optional): Specifies the character set for string values.178 ssl_key (str, optional): Specifies the path of the file containing the SSL179 key.180 ssl_cert (str, optional): Specifies the path of the file containing the SSL181 certificate.182 ssl_ca (str, optional): Specifies the path of the file containing the SSL183 certificate authority.184 ssl_cipher (str, optional): Sets the SSL cipher list.185 ssl_disabled (bool, optional): Disables SSL usage.186 ssl_verify_cert (bool, optional): Verifies the server's certificate.187 Automatically enabled if ``ssl_ca`` is specified.188 ssl_verify_identity (bool, optional): Verifies the server's identity.189 conv (dict[int, Callable], optional): A dictionary of data conversion190 functions.191 credential_type (str, optional): Specifies the type of authentication to192 use: auth.PASSWORD, auth.JWT, or auth.BROWSER_SSO.193 autocommit (bool, optional): Enables autocommits.194 results_type (str, optional): Determines the structure of the query results:195 tuples, namedtuples, dicts.196 results_format (str, optional): Deprecated. This option has been renamed to197 results_type.198 199 Examples:200 Basic Usage:201 202 .. code-block:: python203 204 from langchain_openai import OpenAIEmbeddings205 from langchain_community.vectorstores import SingleStoreDB206 207 vectorstore = SingleStoreDB(208 OpenAIEmbeddings(),209 host="https://user:password@127.0.0.1:3306/database"210 )211 212 Advanced Usage:213 214 .. code-block:: python215 216 from langchain_openai import OpenAIEmbeddings217 from langchain_community.vectorstores import SingleStoreDB218 219 vectorstore = SingleStoreDB(220 OpenAIEmbeddings(),221 distance_strategy=DistanceStrategy.EUCLIDEAN_DISTANCE,222 host="127.0.0.1",223 port=3306,224 user="user",225 password="password",226 database="db",227 table_name="my_custom_table",228 pool_size=10,229 timeout=60,230 )231 232 Using environment variables:233 234 .. code-block:: python235 236 from langchain_openai import OpenAIEmbeddings237 from langchain_community.vectorstores import SingleStoreDB238 239 os.environ['SINGLESTOREDB_URL'] = 'me:p455w0rd@s2-host.com/my_db'240 vectorstore = SingleStoreDB(OpenAIEmbeddings())241 242 Using vector index:243 244 .. code-block:: python245 246 from langchain_openai import OpenAIEmbeddings247 from langchain_community.vectorstores import SingleStoreDB248 249 os.environ['SINGLESTOREDB_URL'] = 'me:p455w0rd@s2-host.com/my_db'250 vectorstore = SingleStoreDB(251 OpenAIEmbeddings(),252 use_vector_index=True,253 )254 255 Using full-text index:256 257 .. code-block:: python258 from langchain_openai import OpenAIEmbeddings259 from langchain_community.vectorstores import SingleStoreDB260 261 os.environ['SINGLESTOREDB_URL'] = 'me:p455w0rd@s2-host.com/my_db'262 vectorstore = SingleStoreDB(263 OpenAIEmbeddings(),264 use_full_text_search=True,265 )266 """267 268 self.embedding = embedding269 self.distance_strategy = distance_strategy270 self.table_name = self._sanitize_input(table_name)271 self.content_field = self._sanitize_input(content_field)272 self.metadata_field = self._sanitize_input(metadata_field)273 self.vector_field = self._sanitize_input(vector_field)274 self.id_field = self._sanitize_input(id_field)275 276 self.use_vector_index = bool(use_vector_index)277 self.vector_index_name = self._sanitize_input(vector_index_name)278 self.vector_index_options = dict(vector_index_options or {})279 self.vector_index_options["metric_type"] = self.distance_strategy280 self.vector_size = int(vector_size)281 282 self.use_full_text_search = bool(use_full_text_search)283 284 # Pass the rest of the kwargs to the connection.285 self.connection_kwargs = kwargs286 287 # Add program name and version to connection attributes.288 if "conn_attrs" not in self.connection_kwargs:289 self.connection_kwargs["conn_attrs"] = dict()290 291 self.connection_kwargs["conn_attrs"]["_connector_name"] = "langchain python sdk"292 self.connection_kwargs["conn_attrs"]["_connector_version"] = "2.1.0"293 294 # Create connection pool.295 self.connection_pool = QueuePool(296 self._get_connection,297 max_overflow=max_overflow,298 pool_size=pool_size,299 timeout=timeout,300 )301 self._create_table()302 303 @property304 def embeddings(self) -> Embeddings:305 return self.embedding306 307 def _sanitize_input(self, input_str: str) -> str:308 # Remove characters that are not alphanumeric or underscores309 return re.sub(r"[^a-zA-Z0-9_]", "", input_str)310 311 def _select_relevance_score_fn(self) -> Callable[[float], float]:312 return self._max_inner_product_relevance_score_fn313 314 def _create_table(self: SingleStoreDB) -> None:315 """Create table if it doesn't exist."""316 conn = self.connection_pool.connect()317 try:318 cur = conn.cursor()319 try:320 full_text_index = ""321 if self.use_full_text_search:322 full_text_index = ", FULLTEXT({})".format(self.content_field)323 if self.use_vector_index:324 index_options = ""325 if self.vector_index_options and len(self.vector_index_options) > 0:326 index_options = "INDEX_OPTIONS '{}'".format(327 json.dumps(self.vector_index_options)328 )329 cur.execute(330 """CREATE TABLE IF NOT EXISTS {}331 ({} BIGINT AUTO_INCREMENT PRIMARY KEY, {} LONGTEXT CHARACTER332 SET utf8mb4 COLLATE utf8mb4_general_ci, {} VECTOR({}, F32)333 NOT NULL, {} JSON, VECTOR INDEX {} ({}) {}{});""".format(334 self.table_name,335 self.id_field,336 self.content_field,337 self.vector_field,338 self.vector_size,339 self.metadata_field,340 self.vector_index_name,341 self.vector_field,342 index_options,343 full_text_index,344 ),345 )346 else:347 cur.execute(348 """CREATE TABLE IF NOT EXISTS {}349 ({} BIGINT AUTO_INCREMENT PRIMARY KEY, {} LONGTEXT CHARACTER350 SET utf8mb4 COLLATE utf8mb4_general_ci, {} BLOB, {} JSON{});351 """.format(352 self.table_name,353 self.id_field,354 self.content_field,355 self.vector_field,356 self.metadata_field,357 full_text_index,358 ),359 )360 finally:361 cur.close()362 finally:363 conn.close()364 365 def add_images(366 self,367 uris: List[str],368 metadatas: Optional[List[dict]] = None,369 embeddings: Optional[List[List[float]]] = None,370 return_ids: bool = False,371 **kwargs: Any,372 ) -> List[str]:373 """Run images through the embeddings and add to the vectorstore.374 375 Args:376 uris List[str]: File path to images.377 Each URI will be added to the vectorstore as document content.378 metadatas (Optional[List[dict]], optional): Optional list of metadatas.379 Defaults to None.380 embeddings (Optional[List[List[float]]], optional): Optional pre-generated381 embeddings. Defaults to None.382 383 Returns:384 List[str]: list of document ids added to the vectorstore385 if return_ids is True. Otherwise, an empty list.386 """387 # Set embeddings388 if (389 embeddings is None390 and self.embedding is not None391 and hasattr(self.embedding, "embed_image")392 ):393 embeddings = self.embedding.embed_image(uris=uris)394 return self.add_texts(395 uris, metadatas, embeddings, return_ids=return_ids, **kwargs396 )397 398 def add_texts(399 self,400 texts: Iterable[str],401 metadatas: Optional[List[dict]] = None,402 embeddings: Optional[List[List[float]]] = None,403 return_ids: bool = False,404 **kwargs: Any,405 ) -> List[str]:406 """Add more texts to the vectorstore.407 408 Args:409 texts (Iterable[str]): Iterable of strings/text to add to the vectorstore.410 metadatas (Optional[List[dict]], optional): Optional list of metadatas.411 Defaults to None.412 embeddings (Optional[List[List[float]]], optional): Optional pre-generated413 embeddings. Defaults to None.414 415 Returns:416 List[str]: list of document ids added to the vectorstore417 if return_ids is True. Otherwise, an empty list.418 """419 ids: List[str] = []420 conn = self.connection_pool.connect()421 try:422 cur = conn.cursor()423 try:424 # Write data to singlestore db425 for i, text in enumerate(texts):426 # Use provided values by default or fallback427 metadata = metadatas[i] if metadatas else {}428 embedding = (429 embeddings[i]430 if embeddings431 else self.embedding.embed_documents([text])[0]432 )433 cur.execute(434 """INSERT INTO {}({}, {}, {})435 VALUES (%s, JSON_ARRAY_PACK(%s), %s)""".format(436 self.table_name,437 self.content_field,438 self.vector_field,439 self.metadata_field,440 ),441 (442 text,443 "[{}]".format(",".join(map(str, embedding))),444 json.dumps(metadata),445 ),446 )447 if return_ids:448 cur.execute("SELECT LAST_INSERT_ID();")449 row = cur.fetchone()450 if row:451 ids.append(str(row[0]))452 if self.use_vector_index or self.use_full_text_search:453 cur.execute("OPTIMIZE TABLE {} FLUSH;".format(self.table_name))454 finally:455 cur.close()456 finally:457 conn.close()458 return ids459 460 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> bool | None:461 """Delete documents from the vectorstore.462 463 Args:464 ids (List[str], optional): List of document ids to delete.465 If None, all documents will be deleted. Defaults to None.466 467 Returns:468 bool: True if deletion was successful, False otherwise.469 """470 if ids is None:471 return True472 473 conn = self.connection_pool.connect()474 try:475 cur = conn.cursor()476 try:477 cur.execute(478 "DELETE FROM {} WHERE {} IN ({})".format(479 self.table_name, self.id_field, ",".join(ids)480 )481 )482 if self.use_vector_index or self.use_full_text_search:483 cur.execute("OPTIMIZE TABLE {} FLUSH;".format(self.table_name))484 finally:485 cur.close()486 finally:487 conn.close()488 return True489 490 def similarity_search(491 self,492 query: str,493 k: int = 4,494 filter: Optional[dict] = None,495 search_strategy: SearchStrategy = SearchStrategy.VECTOR_ONLY,496 filter_threshold: float = 0,497 text_weight: float = 0.5,498 vector_weight: float = 0.5,499 vector_select_count_multiplier: int = 10,500 **kwargs: Any,501 ) -> List[Document]:502 """Returns the most similar indexed documents to the query text.503 504 Uses cosine similarity.505 506 Args:507 query (str): The query text for which to find similar documents.508 k (int): The number of documents to return. Default is 4.509 filter (dict): A dictionary of metadata fields and values to filter by.510 Default is None.511 search_strategy (SearchStrategy): The search strategy to use.512 Default is SearchStrategy.VECTOR_ONLY.513 Available options are:514 - SearchStrategy.VECTOR_ONLY: Searches only by vector similarity.515 - SearchStrategy.TEXT_ONLY: Searches only by text similarity. This516 option is only available if use_full_text_search is True.517 - SearchStrategy.FILTER_BY_TEXT: Filters by text similarity and518 searches by vector similarity. This option is only available if519 use_full_text_search is True.520 - SearchStrategy.FILTER_BY_VECTOR: Filters by vector similarity and521 searches by text similarity. This option is only available if522 use_full_text_search is True.523 - SearchStrategy.WEIGHTED_SUM: Searches by a weighted sum of text and524 vector similarity. This option is only available if525 use_full_text_search is True and distance_strategy is DOT_PRODUCT.526 filter_threshold (float): The threshold for filtering by text or vector527 similarity. Default is 0. This option has effect only if search_strategy528 is SearchStrategy.FILTER_BY_TEXT or SearchStrategy.FILTER_BY_VECTOR.529 text_weight (float): The weight of text similarity in the weighted sum530 search strategy. Default is 0.5. This option has effect only if531 search_strategy is SearchStrategy.WEIGHTED_SUM.532 vector_weight (float): The weight of vector similarity in the weighted sum533 search strategy. Default is 0.5. This option has effect only if534 search_strategy is SearchStrategy.WEIGHTED_SUM.535 vector_select_count_multiplier (int): The multiplier for the number of536 vectors to select when using the vector index. Default is 10.537 This parameter has effect only if use_vector_index is True and538 search_strategy is SearchStrategy.WEIGHTED_SUM or539 SearchStrategy.FILTER_BY_TEXT.540 The number of vectors selected will541 be k * vector_select_count_multiplier.542 This is needed due to the limitations of the vector index.543 544 545 Returns:546 List[Document]: A list of documents that are most similar to the query text.547 548 Examples:549 550 Basic Usage:551 .. code-block:: python552 553 from langchain_community.vectorstores import SingleStoreDB554 from langchain_openai import OpenAIEmbeddings555 556 s2 = SingleStoreDB.from_documents(557 docs,558 OpenAIEmbeddings(),559 host="username:password@localhost:3306/database"560 )561 results = s2.similarity_search("query text", 1,562 {"metadata_field": "metadata_value"})563 564 Different Search Strategies:565 .. code-block:: python566 567 from langchain_community.vectorstores import SingleStoreDB568 from langchain_openai import OpenAIEmbeddings569 570 s2 = SingleStoreDB.from_documents(571 docs,572 OpenAIEmbeddings(),573 host="username:password@localhost:3306/database",574 use_full_text_search=True,575 use_vector_index=True,576 )577 results = s2.similarity_search("query text", 1,578 search_strategy=SingleStoreDB.SearchStrategy.FILTER_BY_TEXT,579 filter_threshold=0.5)580 581 Weighted Sum Search Strategy:582 .. code-block:: python583 584 from langchain_community.vectorstores import SingleStoreDB585 from langchain_openai import OpenAIEmbeddings586 587 s2 = SingleStoreDB.from_documents(588 docs,589 OpenAIEmbeddings(),590 host="username:password@localhost:3306/database",591 use_full_text_search=True,592 use_vector_index=True,593 )594 results = s2.similarity_search("query text", 1,595 search_strategy=SingleStoreDB.SearchStrategy.WEIGHTED_SUM,596 text_weight=0.3,597 vector_weight=0.7)598 """599 docs_and_scores = self.similarity_search_with_score(600 query=query,601 k=k,602 filter=filter,603 search_strategy=search_strategy,604 filter_threshold=filter_threshold,605 text_weight=text_weight,606 vector_weight=vector_weight,607 vector_select_count_multiplier=vector_select_count_multiplier,608 **kwargs,609 )610 return [doc for doc, _ in docs_and_scores]611 612 def similarity_search_with_score(613 self,614 query: str,615 k: int = 4,616 filter: Optional[dict] = None,617 search_strategy: SearchStrategy = SearchStrategy.VECTOR_ONLY,618 filter_threshold: float = 1,619 text_weight: float = 0.5,620 vector_weight: float = 0.5,621 vector_select_count_multiplier: int = 10,622 **kwargs: Any,623 ) -> List[Tuple[Document, float]]:624 """Return docs most similar to query. Uses cosine similarity.625 626 Args:627 query: Text to look up documents similar to.628 k: Number of Documents to return. Defaults to 4.629 filter: A dictionary of metadata fields and values to filter by.630 Defaults to None.631 search_strategy (SearchStrategy): The search strategy to use.632 Default is SearchStrategy.VECTOR_ONLY.633 Available options are:634 - SearchStrategy.VECTOR_ONLY: Searches only by vector similarity.635 - SearchStrategy.TEXT_ONLY: Searches only by text similarity. This636 option is only available if use_full_text_search is True.637 - SearchStrategy.FILTER_BY_TEXT: Filters by text similarity and638 searches by vector similarity. This option is only available if639 use_full_text_search is True.640 - SearchStrategy.FILTER_BY_VECTOR: Filters by vector similarity and641 searches by text similarity. This option is only available if642 use_full_text_search is True.643 - SearchStrategy.WEIGHTED_SUM: Searches by a weighted sum of text and644 vector similarity. This option is only available if645 use_full_text_search is True and distance_strategy is DOT_PRODUCT.646 filter_threshold (float): The threshold for filtering by text or vector647 similarity. Default is 0. This option has effect only if search_strategy648 is SearchStrategy.FILTER_BY_TEXT or SearchStrategy.FILTER_BY_VECTOR.649 text_weight (float): The weight of text similarity in the weighted sum650 search strategy. Default is 0.5. This option has effect only if651 search_strategy is SearchStrategy.WEIGHTED_SUM.652 vector_weight (float): The weight of vector similarity in the weighted sum653 search strategy. Default is 0.5. This option has effect only if654 search_strategy is SearchStrategy.WEIGHTED_SUM.655 vector_select_count_multiplier (int): The multiplier for the number of656 vectors to select when using the vector index. Default is 10.657 This parameter has effect only if use_vector_index is True and658 search_strategy is SearchStrategy.WEIGHTED_SUM or659 SearchStrategy.FILTER_BY_TEXT.660 The number of vectors selected will661 be k * vector_select_count_multiplier.662 This is needed due to the limitations of the vector index.663 Returns:664 List of Documents most similar to the query and score for each665 document.666 667 Raises:668 ValueError: If the search strategy is not supported with the669 distance strategy.670 671 Examples:672 Basic Usage:673 .. code-block:: python674 675 from langchain_community.vectorstores import SingleStoreDB676 from langchain_openai import OpenAIEmbeddings677 678 s2 = SingleStoreDB.from_documents(679 docs,680 OpenAIEmbeddings(),681 host="username:password@localhost:3306/database"682 )683 results = s2.similarity_search_with_score("query text", 1,684 {"metadata_field": "metadata_value"})685 686 Different Search Strategies:687 688 .. code-block:: python689 690 from langchain_community.vectorstores import SingleStoreDB691 from langchain_openai import OpenAIEmbeddings692 693 s2 = SingleStoreDB.from_documents(694 docs,695 OpenAIEmbeddings(),696 host="username:password@localhost:3306/database",697 use_full_text_search=True,698 use_vector_index=True,699 )700 results = s2.similarity_search_with_score("query text", 1,701 search_strategy=SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR,702 filter_threshold=0.5)703 704 Weighted Sum Search Strategy:705 .. code-block:: python706 707 from langchain_community.vectorstores import SingleStoreDB708 from langchain_openai import OpenAIEmbeddings709 710 s2 = SingleStoreDB.from_documents(711 docs,712 OpenAIEmbeddings(),713 host="username:password@localhost:3306/database",714 use_full_text_search=True,715 use_vector_index=True,716 )717 results = s2.similarity_search_with_score("query text", 1,718 search_strategy=SingleStoreDB.SearchStrategy.WEIGHTED_SUM,719 text_weight=0.3,720 vector_weight=0.7)721 """722 723 if (724 search_strategy != SingleStoreDB.SearchStrategy.VECTOR_ONLY725 and not self.use_full_text_search726 ):727 raise ValueError(728 """Search strategy {} is not supported729 when use_full_text_search is False""".format(search_strategy)730 )731 732 if (733 search_strategy == SingleStoreDB.SearchStrategy.WEIGHTED_SUM734 and self.distance_strategy != DistanceStrategy.DOT_PRODUCT735 ):736 raise ValueError(737 "Search strategy {} is not supported with distance strategy {}".format(738 search_strategy, self.distance_strategy739 )740 )741 742 # Creates embedding vector from user query743 embedding = []744 if search_strategy != SingleStoreDB.SearchStrategy.TEXT_ONLY:745 embedding = self.embedding.embed_query(query)746 747 self.embedding.embed_query(query)748 conn = self.connection_pool.connect()749 result = []750 where_clause: str = ""751 where_clause_values: List[Any] = []752 if filter or search_strategy in [753 SingleStoreDB.SearchStrategy.FILTER_BY_TEXT,754 SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR,755 ]:756 where_clause = "WHERE "757 arguments = []758 759 if search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_TEXT:760 arguments.append(761 "MATCH ({}) AGAINST (%s) > %s".format(self.content_field)762 )763 where_clause_values.append(query)764 where_clause_values.append(float(filter_threshold))765 766 if search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR:767 condition = "{}({}, JSON_ARRAY_PACK(%s)) ".format(768 self.distance_strategy.name769 if isinstance(self.distance_strategy, DistanceStrategy)770 else self.distance_strategy,771 self.vector_field,772 )773 if self.distance_strategy == DistanceStrategy.EUCLIDEAN_DISTANCE:774 condition += "< %s"775 else:776 condition += "> %s"777 arguments.append(condition)778 where_clause_values.append("[{}]".format(",".join(map(str, embedding))))779 where_clause_values.append(float(filter_threshold))780 781 def build_where_clause(782 where_clause_values: List[Any],783 sub_filter: dict,784 prefix_args: Optional[List[str]] = None,785 ) -> None:786 prefix_args = prefix_args or []787 for key in sub_filter.keys():788 if isinstance(sub_filter[key], dict):789 build_where_clause(790 where_clause_values, sub_filter[key], prefix_args + [key]791 )792 else:793 arguments.append(794 "JSON_EXTRACT_JSON({}, {}) = %s".format(795 self.metadata_field,796 ", ".join(["%s"] * (len(prefix_args) + 1)),797 )798 )799 where_clause_values += prefix_args + [key]800 where_clause_values.append(json.dumps(sub_filter[key]))801 802 if filter:803 build_where_clause(where_clause_values, filter)804 where_clause += " AND ".join(arguments)805 806 try:807 cur = conn.cursor()808 try:809 if (810 search_strategy == SingleStoreDB.SearchStrategy.VECTOR_ONLY811 or search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_TEXT812 ):813 search_options = ""814 if (815 self.use_vector_index816 and search_strategy817 == SingleStoreDB.SearchStrategy.FILTER_BY_TEXT818 ):819 search_options = "SEARCH_OPTIONS '{\"k\":%d}'" % (820 k * vector_select_count_multiplier821 )822 cur.execute(823 """SELECT {}, {}, {}({}, JSON_ARRAY_PACK(%s)) as __score824 FROM {} {} ORDER BY __score {}{} LIMIT %s""".format(825 self.content_field,826 self.metadata_field,827 self.distance_strategy.name828 if isinstance(self.distance_strategy, DistanceStrategy)829 else self.distance_strategy,830 self.vector_field,831 self.table_name,832 where_clause,833 search_options,834 ORDERING_DIRECTIVE[self.distance_strategy],835 ),836 ("[{}]".format(",".join(map(str, embedding))),)837 + tuple(where_clause_values)838 + (k,),839 )840 elif (841 search_strategy == SingleStoreDB.SearchStrategy.FILTER_BY_VECTOR842 or search_strategy == SingleStoreDB.SearchStrategy.TEXT_ONLY843 ):844 cur.execute(845 """SELECT {}, {}, MATCH ({}) AGAINST (%s) as __score846 FROM {} {} ORDER BY __score DESC LIMIT %s""".format(847 self.content_field,848 self.metadata_field,849 self.content_field,850 self.table_name,851 where_clause,852 ),853 (query,) + tuple(where_clause_values) + (k,),854 )855 elif search_strategy == SingleStoreDB.SearchStrategy.WEIGHTED_SUM:856 cur.execute(857 """SELECT {}, {}, __score1 * %s + __score2 * %s as __score858 FROM (859 SELECT {}, {}, {}, MATCH ({}) AGAINST (%s) as __score1 860 FROM {} {}) r1 FULL OUTER JOIN (861 SELECT {}, {}({}, JSON_ARRAY_PACK(%s)) as __score2862 FROM {} {} ORDER BY __score2 {} LIMIT %s863 ) r2 ON r1.{} = r2.{} ORDER BY __score {} LIMIT %s""".format(864 self.content_field,865 self.metadata_field,866 self.id_field,867 self.content_field,868 self.metadata_field,869 self.content_field,870 self.table_name,871 where_clause,872 self.id_field,873 self.distance_strategy.name874 if isinstance(self.distance_strategy, DistanceStrategy)875 else self.distance_strategy,876 self.vector_field,877 self.table_name,878 where_clause,879 ORDERING_DIRECTIVE[self.distance_strategy],880 self.id_field,881 self.id_field,882 ORDERING_DIRECTIVE[self.distance_strategy],883 ),884 (text_weight, vector_weight, query)885 + tuple(where_clause_values)886 + ("[{}]".format(",".join(map(str, embedding))),)887 + tuple(where_clause_values)888 + (k * vector_select_count_multiplier, k),889 )890 else:891 raise ValueError(892 "Invalid search strategy: {}".format(search_strategy)893 )894 895 for row in cur.fetchall():896 doc = Document(page_content=row[0], metadata=row[1])897 result.append((doc, float(row[2])))898 finally:899 cur.close()900 finally:901 conn.close()902 return result903 904 @classmethod905 def from_texts(906 cls: Type[SingleStoreDB],907 texts: List[str],908 embedding: Embeddings,909 metadatas: Optional[List[dict]] = None,910 distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,911 table_name: str = "embeddings",912 content_field: str = "content",913 metadata_field: str = "metadata",914 vector_field: str = "vector",915 id_field: str = "id",916 use_vector_index: bool = False,917 vector_index_name: str = "",918 vector_index_options: Optional[dict] = None,919 vector_size: int = 1536,920 use_full_text_search: bool = False,921 pool_size: int = 5,922 max_overflow: int = 10,923 timeout: float = 30,924 **kwargs: Any,925 ) -> SingleStoreDB:926 """Create a SingleStoreDB vectorstore from raw documents.927 This is a user-friendly interface that:928 1. Embeds documents.929 2. Creates a new table for the embeddings in SingleStoreDB.930 3. Adds the documents to the newly created table.931 This is intended to be a quick way to get started.932 Args:933 texts (List[str]): List of texts to add to the vectorstore.934 embedding (Embeddings): A text embedding model.935 metadatas (Optional[List[dict]], optional): Optional list of metadatas.936 Defaults to None.937 distance_strategy (DistanceStrategy, optional):938 Determines the strategy employed for calculating939 the distance between vectors in the embedding space.940 Defaults to DOT_PRODUCT.941 Available options are:942 - DOT_PRODUCT: Computes the scalar product of two vectors.943 This is the default behavior944 - EUCLIDEAN_DISTANCE: Computes the Euclidean distance between945 two vectors. This metric considers the geometric distance in946 the vector space, and might be more suitable for embeddings947 that rely on spatial relationships. This metric is not948 compatible with the WEIGHTED_SUM search strategy.949 table_name (str, optional): Specifies the name of the table in use.950 Defaults to "embeddings".951 content_field (str, optional): Specifies the field to store the content.952 Defaults to "content".953 metadata_field (str, optional): Specifies the field to store metadata.954 Defaults to "metadata".955 vector_field (str, optional): Specifies the field to store the vector.956 Defaults to "vector".957 id_field (str, optional): Specifies the field to store the id.958 Defaults to "id".959 use_vector_index (bool, optional): Toggles the use of a vector index.960 Works only with SingleStoreDB 8.5 or later. Defaults to False.961 If set to True, vector_size parameter is required to be set to962 a proper value.963 vector_index_name (str, optional): Specifies the name of the vector index.964 Defaults to empty. Will be ignored if use_vector_index is set to False.965 vector_index_options (dict, optional): Specifies the options for966 the vector index. Defaults to {}.967 Will be ignored if use_vector_index is set to False. The options are:968 index_type (str, optional): Specifies the type of the index.969 Defaults to IVF_PQFS.970 For more options, please refer to the SingleStoreDB documentation:971 https://docs.singlestore.com/cloud/reference/sql-reference/vector-functions/vector-indexing/972 vector_size (int, optional): Specifies the size of the vector.973 Defaults to 1536. Required if use_vector_index is set to True.974 Should be set to the same value as the size of the vectors975 stored in the vector_field.976 use_full_text_search (bool, optional): Toggles the use a full-text index977 on the document content. Defaults to False. If set to True, the table978 will be created with a full-text index on the content field,979 and the simularity_search method will all using TEXT_ONLY,980 FILTER_BY_TEXT, FILTER_BY_VECTOR, and WIGHTED_SUM search strategies.981 If set to False, the simularity_search method will only allow982 VECTOR_ONLY search strategy.983 984 pool_size (int, optional): Determines the number of active connections in985 the pool. Defaults to 5.986 max_overflow (int, optional): Determines the maximum number of connections987 allowed beyond the pool_size. Defaults to 10.988 timeout (float, optional): Specifies the maximum wait time in seconds for989 establishing a connection. Defaults to 30.990 991 Additional optional arguments provide further customization over the992 database connection:993 994 pure_python (bool, optional): Toggles the connector mode. If True,995 operates in pure Python mode.996 local_infile (bool, optional): Allows local file uploads.997 charset (str, optional): Specifies the character set for string values.998 ssl_key (str, optional): Specifies the path of the file containing the SSL999 key.1000 ssl_cert (str, optional): Specifies the path of the file containing the SSL1001 certificate.1002 ssl_ca (str, optional): Specifies the path of the file containing the SSL1003 certificate authority.1004 ssl_cipher (str, optional): Sets the SSL cipher list.1005 ssl_disabled (bool, optional): Disables SSL usage.1006 ssl_verify_cert (bool, optional): Verifies the server's certificate.1007 Automatically enabled if ``ssl_ca`` is specified.1008 ssl_verify_identity (bool, optional): Verifies the server's identity.1009 conv (dict[int, Callable], optional): A dictionary of data conversion1010 functions.1011 credential_type (str, optional): Specifies the type of authentication to1012 use: auth.PASSWORD, auth.JWT, or auth.BROWSER_SSO.1013 autocommit (bool, optional): Enables autocommits.1014 results_type (str, optional): Determines the structure of the query results:1015 tuples, namedtuples, dicts.1016 results_format (str, optional): Deprecated. This option has been renamed to1017 results_type.1018 1019 Example:1020 .. code-block:: python1021 1022 from langchain_community.vectorstores import SingleStoreDB1023 from langchain_openai import OpenAIEmbeddings1024 1025 s2 = SingleStoreDB.from_texts(1026 texts,1027 OpenAIEmbeddings(),1028 host="username:password@localhost:3306/database"1029 )1030 """1031 1032 instance = cls(1033 embedding,1034 distance_strategy=distance_strategy,1035 table_name=table_name,1036 content_field=content_field,1037 metadata_field=metadata_field,1038 vector_field=vector_field,1039 id_field=id_field,1040 pool_size=pool_size,1041 max_overflow=max_overflow,1042 timeout=timeout,1043 use_vector_index=use_vector_index,1044 vector_index_name=vector_index_name,1045 vector_index_options=vector_index_options,1046 vector_size=vector_size,1047 use_full_text_search=use_full_text_search,1048 **kwargs,1049 )1050 instance.add_texts(texts, metadatas, embedding.embed_documents(texts), **kwargs)1051 return instance1052 1053 def drop(self) -> None:1054 """Drop the table and delete all data from the vectorstore.1055 Vector store will be unusable after this operation.1056 """1057 conn = self.connection_pool.connect()1058 try:1059 cur = conn.cursor()1060 try:1061 cur.execute("DROP TABLE IF EXISTS {}".format(self.table_name))1062 finally:1063 cur.close()1064 finally:1065 conn.close()1066 1067 1068# SingleStoreDBRetriever is not needed, but we keep it for backwards compatibility1069SingleStoreDBRetriever = VectorStoreRetriever1070 