Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
duckdb.py361 linesDownload Raw Back to vectorstores
1# mypy: disable-error-code=func-returns-value2from __future__ import annotations3 4import json5import logging6import uuid7import warnings8from typing import Any, Iterable, List, Optional, Type9 10from langchain_core.documents import Document11from langchain_core.embeddings import Embeddings12from langchain_core.vectorstores import VST, VectorStore13 14logger = logging.getLogger(__name__)15 16DEFAULT_VECTOR_KEY = "embedding"17DEFAULT_ID_KEY = "id"18DEFAULT_TEXT_KEY = "text"19DEFAULT_TABLE_NAME = "embeddings"20SIMILARITY_ALIAS = "similarity_score"21DUCKDB_FETCHALL_PAGE_CONTENT_INDEX = 122DUCKDB_FETCHALL_METADATA_INDEX = 323DUCKDB_FETCHALL_SIMILARITY_SCORE_INDEX = 424 25 26class DuckDB(VectorStore):27    """`DuckDB` vector store.28 29    This class provides a vector store interface for adding texts and performing30    similarity searches using DuckDB.31 32    For more information about DuckDB, see: https://duckdb.org/33 34    This integration requires the `duckdb` Python package.35    You can install it with `pip install duckdb`.36 37    *Security Notice*: The default DuckDB configuration is not secure.38 39        By **default**, DuckDB can interact with files across the entire file system,40        which includes abilities to read, write, and list files and directories.41        It can also access some python variables present in the global namespace.42 43        When using this DuckDB vectorstore, we suggest that you initialize the44        DuckDB connection with a secure configuration.45 46        For example, you can set `enable_external_access` to `false` in the connection47        configuration to disable external access to the DuckDB connection.48 49        You can view the DuckDB configuration options here:50 51        https://duckdb.org/docs/configuration/overview.html52 53        Please review other relevant security considerations in the DuckDB54        documentation. (e.g., "autoinstall_known_extensions": "false",55        "autoload_known_extensions": "false")56 57        See https://python.langchain.com/docs/security for more information.58 59    Args:60        connection: Optional DuckDB connection61        embedding: The embedding function or model to use for generating embeddings.62        vector_key: The column name for storing vectors. Defaults to `embedding`.63        id_key: The column name for storing unique identifiers. Defaults to `id`.64        text_key: The column name for storing text. Defaults to `text`.65        table_name: The name of the table to use for storing embeddings. Defaults to66          `embeddings`.67 68    Example:69        .. code-block:: python70 71            import duckdb72            conn = duckdb.connect(database=':memory:',73                config={74                    # Sample configuration to restrict some DuckDB capabilities75                    # List is not exhaustive. Please review DuckDB documentation.76                        "enable_external_access": "false",77                        "autoinstall_known_extensions": "false",78                        "autoload_known_extensions": "false"79                    }80            )81            embedding_function = ... # Define or import your embedding function here82            vector_store = DuckDB(conn, embedding_function)83            vector_store.add_texts(['text1', 'text2'])84            result = vector_store.similarity_search('text1')85    """86 87    def __init__(88        self,89        *,90        connection: Optional[Any] = None,91        embedding: Embeddings,92        vector_key: str = DEFAULT_VECTOR_KEY,93        id_key: str = DEFAULT_ID_KEY,94        text_key: str = DEFAULT_TEXT_KEY,95        table_name: str = DEFAULT_TABLE_NAME,96    ):97        """Initialize with DuckDB connection and setup for vector storage."""98        try:99            import duckdb100        except ImportError:101            raise ImportError(102                "Could not import duckdb package. "103                "Please install it with `pip install duckdb`."104            )105 106        self.duckdb = duckdb107        self._embedding = embedding108        self._vector_key = vector_key109        self._id_key = id_key110        self._text_key = text_key111        self._table_name = table_name112 113        if self._embedding is None:114            raise ValueError("An embedding function or model must be provided.")115 116        if connection is None:117            warnings.warn(118                "No DuckDB connection provided. A new connection will be created."119                "This connection is running in memory and no data will be persisted."120                "To persist data, specify `connection=duckdb.connect(...)` when using "121                "the API. Please review the documentation of the vectorstore for "122                "security recommendations on configuring the connection."123            )124 125        self._connection = connection or self.duckdb.connect(126            database=":memory:", config={"enable_external_access": "false"}127        )128        self._ensure_table()129        self._table = self._connection.table(self._table_name)130 131    @property132    def embeddings(self) -> Optional[Embeddings]:133        """Returns the embedding object used by the vector store."""134        return self._embedding135 136    def add_texts(137        self,138        texts: Iterable[str],139        metadatas: Optional[List[dict]] = None,140        **kwargs: Any,141    ) -> List[str]:142        """Turn texts into embedding and add it to the database using Pandas DataFrame143 144        Args:145            texts: Iterable of strings to add to the vectorstore.146            metadatas: Optional list of metadatas associated with the texts.147            kwargs: Additional parameters including optional 'ids' to associate148              with the texts.149 150        Returns:151            List of ids of the added texts.152        """153        have_pandas = False154        try:155            import pandas as pd156 157            have_pandas = True158        except ImportError:159            logger.info(160                "Unable to import pandas. "161                "Install it with `pip install -U pandas` "162                "to improve performance of add_texts()."163            )164 165        # Extract ids from kwargs or generate new ones if not provided166        ids = kwargs.pop("ids", [str(uuid.uuid4()) for _ in texts])167 168        # Embed texts and create documents169        ids = ids or [str(uuid.uuid4()) for _ in texts]170        embeddings = self._embedding.embed_documents(list(texts))171        data = []172        for idx, text in enumerate(texts):173            embedding = embeddings[idx]174            # Serialize metadata if present, else default to None175            metadata = (176                json.dumps(metadatas[idx])177                if metadatas and idx < len(metadatas)178                else None179            )180            if have_pandas:181                data.append(182                    {183                        self._id_key: ids[idx],184                        self._text_key: text,185                        self._vector_key: embedding,186                        "metadata": metadata,187                    }188                )189            else:190                self._connection.execute(191                    f"INSERT INTO {self._table_name} VALUES (?,?,?,?)",192                    [ids[idx], text, embedding, metadata],193                )194 195        if have_pandas:196            # noinspection PyUnusedLocal197            df = pd.DataFrame.from_dict(data)  # noqa: F841198            self._connection.register("df", df)199            self._connection.execute(200                f"INSERT INTO {self._table_name} SELECT * FROM df",201            )202        return ids203 204    def similarity_search_pd(205        self, query: str, k: int = 4, **kwargs: Any206    ) -> List[Document]:207        """Performs a similarity search for a given query string.208        Requires pandas to be installed.209        This was the previously executed method for similarity search.210 211        Args:212            query: The query string to search for.213            k: The number of similar texts to return.214 215        Returns:216            A list of Documents most similar to the query.217        """218        try:219            import pandas as pandas220        except ImportError:221            warnings.warn("You may need to `pip install pandas` to use this method.")222 223        embedding = self._embedding.embed_query(query)224        list_cosine_similarity = self.duckdb.FunctionExpression(225            "list_cosine_similarity",226            self.duckdb.ColumnExpression(self._vector_key),227            self.duckdb.ConstantExpression(embedding),228        )229        docs = (230            self._table.select(231                *[232                    self.duckdb.StarExpression(exclude=[]),233                    list_cosine_similarity.alias(SIMILARITY_ALIAS),234                ]235            )236            .order(f"{SIMILARITY_ALIAS} desc")237            .limit(k)238            .fetchdf()239        )240        return [241            Document(242                page_content=docs[self._text_key][idx],243                metadata={244                    **json.loads(docs["metadata"][idx]),245                    # using underscore prefix to avoid conflicts with user metadata keys246                    f"_{SIMILARITY_ALIAS}": docs[SIMILARITY_ALIAS][idx],247                }248                if docs["metadata"][idx]249                else {},250            )251            for idx in range(len(docs))252        ]253 254    def similarity_search(255        self, query: str, k: int = 4, **kwargs: Any256    ) -> List[Document]:257        """Performs a similarity search for a given query string.258        Does not require pandas to be installed.259 260        Args:261            query: The query string to search for.262            k: The number of similar texts to return.263 264        Returns:265            A list of Documents most similar to the query.266        """267 268        embedding = self._embedding.embed_query(query)269        list_cosine_similarity = self.duckdb.FunctionExpression(270            "list_cosine_similarity",271            self.duckdb.ColumnExpression(self._vector_key),272            self.duckdb.ConstantExpression(embedding),273        )274        docs = (275            self._table.select(276                *[277                    self.duckdb.StarExpression(exclude=[]),278                    list_cosine_similarity.alias(SIMILARITY_ALIAS),279                ]280            )281            .order(f"{SIMILARITY_ALIAS} desc")282            .limit(k)283            .fetchall()284        )285        return [286            Document(287                page_content=docs[idx][DUCKDB_FETCHALL_PAGE_CONTENT_INDEX],288                metadata={289                    **json.loads(docs[idx][DUCKDB_FETCHALL_METADATA_INDEX]),290                    # using underscore prefix to avoid conflicts with user metadata keys291                    f"_{SIMILARITY_ALIAS}": docs[idx][292                        DUCKDB_FETCHALL_SIMILARITY_SCORE_INDEX293                    ],294                }295                if docs[idx][DUCKDB_FETCHALL_METADATA_INDEX]296                else {},297            )298            for idx in range(len(docs))299        ]300 301    @classmethod302    def from_texts(303        cls: Type[VST],304        texts: List[str],305        embedding: Embeddings,306        metadatas: Optional[List[dict]] = None,307        **kwargs: Any,308    ) -> DuckDB:309        """Creates an instance of DuckDB and populates it with texts and310          their embeddings.311 312        Args:313            texts: List of strings to add to the vector store.314            embedding: The embedding function or model to use for generating embeddings.315            metadatas: Optional list of metadata dictionaries associated with the texts.316            kwargs: Additional keyword arguments including:317                - connection: DuckDB connection. If not provided, a new connection will318                  be created.319                - vector_key: The column name for storing vectors. Default "vector".320                - id_key: The column name for storing unique identifiers. Default "id".321                - text_key: The column name for storing text. Defaults to "text".322                - table_name: The name of the table to use for storing embeddings.323                    Defaults to "embeddings".324 325        Returns:326            An instance of DuckDB with the provided texts and their embeddings added.327        """328 329        # Extract kwargs for DuckDB instance creation330        connection = kwargs.get("connection", None)331        vector_key = kwargs.get("vector_key", DEFAULT_VECTOR_KEY)332        id_key = kwargs.get("id_key", DEFAULT_ID_KEY)333        text_key = kwargs.get("text_key", DEFAULT_TEXT_KEY)334        table_name = kwargs.get("table_name", DEFAULT_TABLE_NAME)335 336        # Create an instance of DuckDB337        instance = DuckDB(338            connection=connection,339            embedding=embedding,340            vector_key=vector_key,341            id_key=id_key,342            text_key=text_key,343            table_name=table_name,344        )345        # Add texts and their embeddings to the DuckDB vector store346        instance.add_texts(texts, metadatas=metadatas, **kwargs)347 348        return instance349 350    def _ensure_table(self) -> None:351        """Ensures the table for storing embeddings exists."""352        create_table_sql = f"""353        CREATE TABLE IF NOT EXISTS {self._table_name} (354            {self._id_key} VARCHAR PRIMARY KEY,355            {self._text_key} VARCHAR,356            {self._vector_key} FLOAT[],357            metadata VARCHAR358        )359        """360        self._connection.execute(create_table_sql)361 
codekingpro/portable-devtools · Team Ai