codekingpro/portable-devtools
114k
1# mypy: disable-error-code=func-returns-value2from __future__ import annotations3 4import json5import logging6import uuid7import warnings8from typing import Any, Iterable, List, Optional, Type9 10from langchain_core.documents import Document11from langchain_core.embeddings import Embeddings12from langchain_core.vectorstores import VST, VectorStore13 14logger = logging.getLogger(__name__)15 16DEFAULT_VECTOR_KEY = "embedding"17DEFAULT_ID_KEY = "id"18DEFAULT_TEXT_KEY = "text"19DEFAULT_TABLE_NAME = "embeddings"20SIMILARITY_ALIAS = "similarity_score"21DUCKDB_FETCHALL_PAGE_CONTENT_INDEX = 122DUCKDB_FETCHALL_METADATA_INDEX = 323DUCKDB_FETCHALL_SIMILARITY_SCORE_INDEX = 424 25 26class DuckDB(VectorStore):27 """`DuckDB` vector store.28 29 This class provides a vector store interface for adding texts and performing30 similarity searches using DuckDB.31 32 For more information about DuckDB, see: https://duckdb.org/33 34 This integration requires the `duckdb` Python package.35 You can install it with `pip install duckdb`.36 37 *Security Notice*: The default DuckDB configuration is not secure.38 39 By **default**, DuckDB can interact with files across the entire file system,40 which includes abilities to read, write, and list files and directories.41 It can also access some python variables present in the global namespace.42 43 When using this DuckDB vectorstore, we suggest that you initialize the44 DuckDB connection with a secure configuration.45 46 For example, you can set `enable_external_access` to `false` in the connection47 configuration to disable external access to the DuckDB connection.48 49 You can view the DuckDB configuration options here:50 51 https://duckdb.org/docs/configuration/overview.html52 53 Please review other relevant security considerations in the DuckDB54 documentation. (e.g., "autoinstall_known_extensions": "false",55 "autoload_known_extensions": "false")56 57 See https://python.langchain.com/docs/security for more information.58 59 Args:60 connection: Optional DuckDB connection61 embedding: The embedding function or model to use for generating embeddings.62 vector_key: The column name for storing vectors. Defaults to `embedding`.63 id_key: The column name for storing unique identifiers. Defaults to `id`.64 text_key: The column name for storing text. Defaults to `text`.65 table_name: The name of the table to use for storing embeddings. Defaults to66 `embeddings`.67 68 Example:69 .. code-block:: python70 71 import duckdb72 conn = duckdb.connect(database=':memory:',73 config={74 # Sample configuration to restrict some DuckDB capabilities75 # List is not exhaustive. Please review DuckDB documentation.76 "enable_external_access": "false",77 "autoinstall_known_extensions": "false",78 "autoload_known_extensions": "false"79 }80 )81 embedding_function = ... # Define or import your embedding function here82 vector_store = DuckDB(conn, embedding_function)83 vector_store.add_texts(['text1', 'text2'])84 result = vector_store.similarity_search('text1')85 """86 87 def __init__(88 self,89 *,90 connection: Optional[Any] = None,91 embedding: Embeddings,92 vector_key: str = DEFAULT_VECTOR_KEY,93 id_key: str = DEFAULT_ID_KEY,94 text_key: str = DEFAULT_TEXT_KEY,95 table_name: str = DEFAULT_TABLE_NAME,96 ):97 """Initialize with DuckDB connection and setup for vector storage."""98 try:99 import duckdb100 except ImportError:101 raise ImportError(102 "Could not import duckdb package. "103 "Please install it with `pip install duckdb`."104 )105 106 self.duckdb = duckdb107 self._embedding = embedding108 self._vector_key = vector_key109 self._id_key = id_key110 self._text_key = text_key111 self._table_name = table_name112 113 if self._embedding is None:114 raise ValueError("An embedding function or model must be provided.")115 116 if connection is None:117 warnings.warn(118 "No DuckDB connection provided. A new connection will be created."119 "This connection is running in memory and no data will be persisted."120 "To persist data, specify `connection=duckdb.connect(...)` when using "121 "the API. Please review the documentation of the vectorstore for "122 "security recommendations on configuring the connection."123 )124 125 self._connection = connection or self.duckdb.connect(126 database=":memory:", config={"enable_external_access": "false"}127 )128 self._ensure_table()129 self._table = self._connection.table(self._table_name)130 131 @property132 def embeddings(self) -> Optional[Embeddings]:133 """Returns the embedding object used by the vector store."""134 return self._embedding135 136 def add_texts(137 self,138 texts: Iterable[str],139 metadatas: Optional[List[dict]] = None,140 **kwargs: Any,141 ) -> List[str]:142 """Turn texts into embedding and add it to the database using Pandas DataFrame143 144 Args:145 texts: Iterable of strings to add to the vectorstore.146 metadatas: Optional list of metadatas associated with the texts.147 kwargs: Additional parameters including optional 'ids' to associate148 with the texts.149 150 Returns:151 List of ids of the added texts.152 """153 have_pandas = False154 try:155 import pandas as pd156 157 have_pandas = True158 except ImportError:159 logger.info(160 "Unable to import pandas. "161 "Install it with `pip install -U pandas` "162 "to improve performance of add_texts()."163 )164 165 # Extract ids from kwargs or generate new ones if not provided166 ids = kwargs.pop("ids", [str(uuid.uuid4()) for _ in texts])167 168 # Embed texts and create documents169 ids = ids or [str(uuid.uuid4()) for _ in texts]170 embeddings = self._embedding.embed_documents(list(texts))171 data = []172 for idx, text in enumerate(texts):173 embedding = embeddings[idx]174 # Serialize metadata if present, else default to None175 metadata = (176 json.dumps(metadatas[idx])177 if metadatas and idx < len(metadatas)178 else None179 )180 if have_pandas:181 data.append(182 {183 self._id_key: ids[idx],184 self._text_key: text,185 self._vector_key: embedding,186 "metadata": metadata,187 }188 )189 else:190 self._connection.execute(191 f"INSERT INTO {self._table_name} VALUES (?,?,?,?)",192 [ids[idx], text, embedding, metadata],193 )194 195 if have_pandas:196 # noinspection PyUnusedLocal197 df = pd.DataFrame.from_dict(data) # noqa: F841198 self._connection.register("df", df)199 self._connection.execute(200 f"INSERT INTO {self._table_name} SELECT * FROM df",201 )202 return ids203 204 def similarity_search_pd(205 self, query: str, k: int = 4, **kwargs: Any206 ) -> List[Document]:207 """Performs a similarity search for a given query string.208 Requires pandas to be installed.209 This was the previously executed method for similarity search.210 211 Args:212 query: The query string to search for.213 k: The number of similar texts to return.214 215 Returns:216 A list of Documents most similar to the query.217 """218 try:219 import pandas as pandas220 except ImportError:221 warnings.warn("You may need to `pip install pandas` to use this method.")222 223 embedding = self._embedding.embed_query(query)224 list_cosine_similarity = self.duckdb.FunctionExpression(225 "list_cosine_similarity",226 self.duckdb.ColumnExpression(self._vector_key),227 self.duckdb.ConstantExpression(embedding),228 )229 docs = (230 self._table.select(231 *[232 self.duckdb.StarExpression(exclude=[]),233 list_cosine_similarity.alias(SIMILARITY_ALIAS),234 ]235 )236 .order(f"{SIMILARITY_ALIAS} desc")237 .limit(k)238 .fetchdf()239 )240 return [241 Document(242 page_content=docs[self._text_key][idx],243 metadata={244 **json.loads(docs["metadata"][idx]),245 # using underscore prefix to avoid conflicts with user metadata keys246 f"_{SIMILARITY_ALIAS}": docs[SIMILARITY_ALIAS][idx],247 }248 if docs["metadata"][idx]249 else {},250 )251 for idx in range(len(docs))252 ]253 254 def similarity_search(255 self, query: str, k: int = 4, **kwargs: Any256 ) -> List[Document]:257 """Performs a similarity search for a given query string.258 Does not require pandas to be installed.259 260 Args:261 query: The query string to search for.262 k: The number of similar texts to return.263 264 Returns:265 A list of Documents most similar to the query.266 """267 268 embedding = self._embedding.embed_query(query)269 list_cosine_similarity = self.duckdb.FunctionExpression(270 "list_cosine_similarity",271 self.duckdb.ColumnExpression(self._vector_key),272 self.duckdb.ConstantExpression(embedding),273 )274 docs = (275 self._table.select(276 *[277 self.duckdb.StarExpression(exclude=[]),278 list_cosine_similarity.alias(SIMILARITY_ALIAS),279 ]280 )281 .order(f"{SIMILARITY_ALIAS} desc")282 .limit(k)283 .fetchall()284 )285 return [286 Document(287 page_content=docs[idx][DUCKDB_FETCHALL_PAGE_CONTENT_INDEX],288 metadata={289 **json.loads(docs[idx][DUCKDB_FETCHALL_METADATA_INDEX]),290 # using underscore prefix to avoid conflicts with user metadata keys291 f"_{SIMILARITY_ALIAS}": docs[idx][292 DUCKDB_FETCHALL_SIMILARITY_SCORE_INDEX293 ],294 }295 if docs[idx][DUCKDB_FETCHALL_METADATA_INDEX]296 else {},297 )298 for idx in range(len(docs))299 ]300 301 @classmethod302 def from_texts(303 cls: Type[VST],304 texts: List[str],305 embedding: Embeddings,306 metadatas: Optional[List[dict]] = None,307 **kwargs: Any,308 ) -> DuckDB:309 """Creates an instance of DuckDB and populates it with texts and310 their embeddings.311 312 Args:313 texts: List of strings to add to the vector store.314 embedding: The embedding function or model to use for generating embeddings.315 metadatas: Optional list of metadata dictionaries associated with the texts.316 kwargs: Additional keyword arguments including:317 - connection: DuckDB connection. If not provided, a new connection will318 be created.319 - vector_key: The column name for storing vectors. Default "vector".320 - id_key: The column name for storing unique identifiers. Default "id".321 - text_key: The column name for storing text. Defaults to "text".322 - table_name: The name of the table to use for storing embeddings.323 Defaults to "embeddings".324 325 Returns:326 An instance of DuckDB with the provided texts and their embeddings added.327 """328 329 # Extract kwargs for DuckDB instance creation330 connection = kwargs.get("connection", None)331 vector_key = kwargs.get("vector_key", DEFAULT_VECTOR_KEY)332 id_key = kwargs.get("id_key", DEFAULT_ID_KEY)333 text_key = kwargs.get("text_key", DEFAULT_TEXT_KEY)334 table_name = kwargs.get("table_name", DEFAULT_TABLE_NAME)335 336 # Create an instance of DuckDB337 instance = DuckDB(338 connection=connection,339 embedding=embedding,340 vector_key=vector_key,341 id_key=id_key,342 text_key=text_key,343 table_name=table_name,344 )345 # Add texts and their embeddings to the DuckDB vector store346 instance.add_texts(texts, metadatas=metadatas, **kwargs)347 348 return instance349 350 def _ensure_table(self) -> None:351 """Ensures the table for storing embeddings exists."""352 create_table_sql = f"""353 CREATE TABLE IF NOT EXISTS {self._table_name} (354 {self._id_key} VARCHAR PRIMARY KEY,355 {self._text_key} VARCHAR,356 {self._vector_key} FLOAT[],357 metadata VARCHAR358 )359 """360 self._connection.execute(create_table_sql)361 