Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
dingo.py383 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import logging4import uuid5from typing import Any, Iterable, List, Optional, Tuple6 7import numpy as np8from langchain_core.documents import Document9from langchain_core.embeddings import Embeddings10from langchain_core.vectorstores import VectorStore11 12from langchain_community.vectorstores.utils import maximal_marginal_relevance13 14logger = logging.getLogger(__name__)15 16 17class Dingo(VectorStore):18    """`Dingo` vector store.19 20    To use, you should have the ``dingodb`` python package installed.21 22    Example:23        .. code-block:: python24 25            from langchain_community.vectorstores import Dingo26            from langchain_community.embeddings.openai import OpenAIEmbeddings27 28            embeddings = OpenAIEmbeddings()29            dingo = Dingo(embeddings, "text")30    """31 32    def __init__(33        self,34        embedding: Embeddings,35        text_key: str,36        *,37        client: Any = None,38        index_name: Optional[str] = None,39        dimension: int = 1024,40        host: Optional[List[str]] = None,41        user: str = "root",42        password: str = "123123",43        self_id: bool = False,44    ):45        """Initialize with Dingo client."""46        try:47            import dingodb48        except ImportError:49            raise ImportError(50                "Could not import dingo python package. "51                "Please install it with `pip install dingodb."52            )53 54        host = host if host is not None else ["172.20.31.10:13000"]55 56        # collection57        if client is not None:58            dingo_client = client59        else:60            try:61                # connect to dingo db62                dingo_client = dingodb.DingoDB(user, password, host)63            except ValueError as e:64                raise ValueError(f"Dingo failed to connect: {e}")65 66        self._text_key = text_key67        self._client = dingo_client68 69        if (70            index_name is not None71            and index_name not in dingo_client.get_index()72            and index_name.upper() not in dingo_client.get_index()73        ):74            if self_id is True:75                dingo_client.create_index(76                    index_name, dimension=dimension, auto_id=False77                )78            else:79                dingo_client.create_index(index_name, dimension=dimension)80 81        self._index_name = index_name82        self._embedding = embedding83 84    @property85    def embeddings(self) -> Optional[Embeddings]:86        return self._embedding87 88    def add_texts(89        self,90        texts: Iterable[str],91        metadatas: Optional[List[dict]] = None,92        ids: Optional[List[str]] = None,93        text_key: str = "text",94        batch_size: int = 500,95        **kwargs: Any,96    ) -> List[str]:97        """Run more texts through the embeddings and add to the vectorstore.98 99        Args:100            texts: Iterable of strings to add to the vectorstore.101            metadatas: Optional list of metadatas associated with the texts.102            ids: Optional list of ids to associate with the texts.103 104        Returns:105            List of ids from adding the texts into the vectorstore.106 107        """108 109        # Embed and create the documents110        ids = ids or [str(uuid.uuid4().int)[:13] for _ in texts]111        metadatas_list = []112        texts = list(texts)113        embeds = self._embedding.embed_documents(texts)114        for i, text in enumerate(texts):115            metadata = metadatas[i] if metadatas else {}116            metadata[self._text_key] = text117            metadatas_list.append(metadata)118        # upsert to Dingo119        for i in range(0, len(list(texts)), batch_size):120            j = i + batch_size121            add_res = self._client.vector_add(122                self._index_name, metadatas_list[i:j], embeds[i:j], ids[i:j]123            )124            if not add_res:125                raise Exception("vector add fail")126 127        return ids128 129    def similarity_search(130        self,131        query: str,132        k: int = 4,133        search_params: Optional[dict] = None,134        timeout: Optional[int] = None,135        **kwargs: Any,136    ) -> List[Document]:137        """Return Dingo documents most similar to query, along with scores.138 139        Args:140            query: Text to look up documents similar to.141            k: Number of Documents to return. Defaults to 4.142            search_params: Dictionary of argument(s) to filter on metadata143 144        Returns:145            List of Documents most similar to the query and score for each146        """147        docs_and_scores = self.similarity_search_with_score(148            query, k=k, search_params=search_params, **kwargs149        )150        return [doc for doc, _ in docs_and_scores]151 152    def similarity_search_with_score(153        self,154        query: str,155        k: int = 4,156        search_params: Optional[dict] = None,157        timeout: Optional[int] = None,158        **kwargs: Any,159    ) -> List[Tuple[Document, float]]:160        """Return Dingo documents most similar to query, along with scores.161 162        Args:163            query: Text to look up documents similar to.164            k: Number of Documents to return. Defaults to 4.165            search_params: Dictionary of argument(s) to filter on metadata166 167        Returns:168            List of Documents most similar to the query and score for each169        """170        docs = []171        query_obj = self._embedding.embed_query(query)172        results = self._client.vector_search(173            self._index_name, xq=query_obj, top_k=k, search_params=search_params174        )175 176        if not results:177            return []178 179        for res in results[0]["vectorWithDistances"]:180            score = res["distance"]181            if (182                "score_threshold" in kwargs183                and kwargs.get("score_threshold") is not None184            ):185                if score > kwargs.get("score_threshold"):186                    continue187            metadatas = res["scalarData"]188            id = res["id"]189            text = metadatas[self._text_key]["fields"][0]["data"]190            metadata = {"id": id, "text": text, "score": score}191            for meta_key in metadatas.keys():192                metadata[meta_key] = metadatas[meta_key]["fields"][0]["data"]193            docs.append((Document(page_content=text, metadata=metadata), score))194 195        return docs196 197    def max_marginal_relevance_search_by_vector(198        self,199        embedding: List[float],200        k: int = 4,201        fetch_k: int = 20,202        lambda_mult: float = 0.5,203        search_params: Optional[dict] = None,204        **kwargs: Any,205    ) -> List[Document]:206        """Return docs selected using the maximal marginal relevance.207 208        Maximal marginal relevance optimizes for similarity to query AND diversity209        among selected documents.210 211        Args:212            embedding: Embedding to look up documents similar to.213            k: Number of Documents to return. Defaults to 4.214            fetch_k: Number of Documents to fetch to pass to MMR algorithm.215            lambda_mult: Number between 0 and 1 that determines the degree216                        of diversity among the results with 0 corresponding217                        to maximum diversity and 1 to minimum diversity.218                        Defaults to 0.5.219        Returns:220            List of Documents selected by maximal marginal relevance.221        """222        results = self._client.vector_search(223            self._index_name, [embedding], search_params=search_params, top_k=k224        )225 226        mmr_selected = maximal_marginal_relevance(227            np.array([embedding], dtype=np.float32),228            [229                item["vector"]["floatValues"]230                for item in results[0]["vectorWithDistances"]231            ],232            k=k,233            lambda_mult=lambda_mult,234        )235        selected = []236        for i in mmr_selected:237            meta_data = {}238            for k, v in results[0]["vectorWithDistances"][i]["scalarData"].items():239                meta_data.update({str(k): v["fields"][0]["data"]})240            selected.append(meta_data)241        return [242            Document(page_content=metadata.pop(self._text_key), metadata=metadata)243            for metadata in selected244        ]245 246    def max_marginal_relevance_search(247        self,248        query: str,249        k: int = 4,250        fetch_k: int = 20,251        lambda_mult: float = 0.5,252        search_params: Optional[dict] = None,253        **kwargs: Any,254    ) -> List[Document]:255        """Return docs selected using the maximal marginal relevance.256 257        Maximal marginal relevance optimizes for similarity to query AND diversity258        among selected documents.259 260        Args:261            query: Text to look up documents similar to.262            k: Number of Documents to return. Defaults to 4.263            fetch_k: Number of Documents to fetch to pass to MMR algorithm.264            lambda_mult: Number between 0 and 1 that determines the degree265                        of diversity among the results with 0 corresponding266                        to maximum diversity and 1 to minimum diversity.267                        Defaults to 0.5.268        Returns:269            List of Documents selected by maximal marginal relevance.270        """271        embedding = self._embedding.embed_query(query)272        return self.max_marginal_relevance_search_by_vector(273            embedding, k, fetch_k, lambda_mult, search_params274        )275 276    @classmethod277    def from_texts(278        cls,279        texts: List[str],280        embedding: Embeddings,281        metadatas: Optional[List[dict]] = None,282        ids: Optional[List[str]] = None,283        text_key: str = "text",284        index_name: Optional[str] = None,285        dimension: int = 1024,286        client: Any = None,287        host: List[str] = ["172.20.31.10:13000"],288        user: str = "root",289        password: str = "123123",290        batch_size: int = 500,291        **kwargs: Any,292    ) -> Dingo:293        """Construct Dingo wrapper from raw documents.294 295                This is a user friendly interface that:296                    1. Embeds documents.297                    2. Adds the documents to a provided Dingo index298 299                This is intended to be a quick way to get started.300 301                Example:302                    .. code-block:: python303 304                        from langchain_community.vectorstores import Dingo305                        from langchain_community.embeddings import OpenAIEmbeddings306                        import dingodb307        sss308                        embeddings = OpenAIEmbeddings()309                        dingo = Dingo.from_texts(310                            texts,311                            embeddings,312                            index_name="langchain-demo"313                        )314        """315        try:316            import dingodb317        except ImportError:318            raise ImportError(319                "Could not import dingo python package. "320                "Please install it with `pip install dingodb`."321            )322 323        if client is not None:324            dingo_client = client325        else:326            try:327                # connect to dingo db328                dingo_client = dingodb.DingoDB(user, password, host)329            except ValueError as e:330                raise ValueError(f"Dingo failed to connect: {e}")331        if kwargs is not None and kwargs.get("self_id") is True:332            if (333                index_name is not None334                and index_name not in dingo_client.get_index()335                and index_name.upper() not in dingo_client.get_index()336            ):337                dingo_client.create_index(338                    index_name, dimension=dimension, auto_id=False339                )340        else:341            if (342                index_name is not None343                and index_name not in dingo_client.get_index()344                and index_name.upper() not in dingo_client.get_index()345            ):346                dingo_client.create_index(index_name, dimension=dimension)347 348        # Embed and create the documents349 350        ids = ids or [str(uuid.uuid4().int)[:13] for _ in texts]351        metadatas_list = []352        texts = list(texts)353        embeds = embedding.embed_documents(texts)354        for i, text in enumerate(texts):355            metadata = metadatas[i] if metadatas else {}356            metadata[text_key] = text357            metadatas_list.append(metadata)358 359        # upsert to Dingo360        for i in range(0, len(list(texts)), batch_size):361            j = i + batch_size362            add_res = dingo_client.vector_add(363                index_name, metadatas_list[i:j], embeds[i:j], ids[i:j]364            )365            if not add_res:366                raise Exception("vector add fail")367        return cls(embedding, text_key, client=dingo_client, index_name=index_name)368 369    def delete(370        self,371        ids: Optional[List[str]] = None,372        **kwargs: Any,373    ) -> Any:374        """Delete by vector IDs or filter.375        Args:376            ids: List of ids to delete.377        """378 379        if ids is None:380            raise ValueError("No ids provided to delete.")381 382        return self._client.vector_delete(self._index_name, ids=ids)383 
codekingpro/portable-devtools · Team Ai