Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
chroma.py911 linesDownload Raw Back to vectorstores
1from __future__ import annotations2 3import base644import logging5import uuid6from typing import (7    TYPE_CHECKING,8    Any,9    Callable,10    Dict,11    Iterable,12    List,13    Optional,14    Tuple,15    Type,16)17 18import numpy as np19from langchain_core._api import deprecated20from langchain_core.documents import Document21from langchain_core.embeddings import Embeddings22from langchain_core.utils import xor_args23from langchain_core.vectorstores import VectorStore24 25from langchain_community.vectorstores.utils import maximal_marginal_relevance26 27if TYPE_CHECKING:28    import chromadb29    import chromadb.config30    from chromadb.api.types import ID, OneOrMany, Where, WhereDocument31 32logger = logging.getLogger()33DEFAULT_K = 4  # Number of Documents to return.34 35 36def _results_to_docs(results: Any) -> List[Document]:37    return [doc for doc, _ in _results_to_docs_and_scores(results)]38 39 40def _results_to_docs_and_scores(results: Any) -> List[Tuple[Document, float]]:41    return [42        # TODO: Chroma can do batch querying,43        # we shouldn't hard code to the 1st result44        (Document(page_content=result[0], metadata=result[1] or {}), result[2])45        for result in zip(46            results["documents"][0],47            results["metadatas"][0],48            results["distances"][0],49        )50    ]51 52 53@deprecated(since="0.2.9", removal="1.0", alternative_import="langchain_chroma.Chroma")54class Chroma(VectorStore):55    """`ChromaDB` vector store.56 57    To use, you should have the ``chromadb`` python package installed.58 59    Example:60        .. code-block:: python61 62                from langchain_community.vectorstores import Chroma63                from langchain_community.embeddings.openai import OpenAIEmbeddings64 65                embeddings = OpenAIEmbeddings()66                vectorstore = Chroma("langchain_store", embeddings)67    """68 69    _LANGCHAIN_DEFAULT_COLLECTION_NAME: str = "langchain"70 71    def __init__(72        self,73        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,74        embedding_function: Optional[Embeddings] = None,75        persist_directory: Optional[str] = None,76        client_settings: Optional[chromadb.config.Settings] = None,77        collection_metadata: Optional[Dict] = None,78        client: Optional[chromadb.Client] = None,79        relevance_score_fn: Optional[Callable[[float], float]] = None,80    ) -> None:81        """Initialize with a Chroma client."""82        try:83            import chromadb84            import chromadb.config85        except ImportError:86            raise ImportError(87                "Could not import chromadb python package. "88                "Please install it with `pip install chromadb`."89            )90 91        if client is not None:92            self._client_settings = client_settings93            self._client = client94            self._persist_directory = persist_directory95        else:96            if client_settings:97                # If client_settings is provided with persist_directory specified,98                # then it is "in-memory and persisting to disk" mode.99                client_settings.persist_directory = (100                    persist_directory or client_settings.persist_directory101                )102                if client_settings.persist_directory is not None:103                    # Maintain backwards compatibility with chromadb < 0.4.0104                    major, minor, _ = chromadb.__version__.split(".")105                    if int(major) == 0 and int(minor) < 4:106                        client_settings.chroma_db_impl = "duckdb+parquet"107 108                _client_settings = client_settings109            elif persist_directory:110                # Maintain backwards compatibility with chromadb < 0.4.0111                major, minor, _ = chromadb.__version__.split(".")112                if int(major) == 0 and int(minor) < 4:113                    _client_settings = chromadb.config.Settings(114                        chroma_db_impl="duckdb+parquet",115                    )116                else:117                    _client_settings = chromadb.config.Settings(is_persistent=True)118                _client_settings.persist_directory = persist_directory119            else:120                _client_settings = chromadb.config.Settings()121            self._client_settings = _client_settings122            self._client = chromadb.Client(_client_settings)123            self._persist_directory = (124                _client_settings.persist_directory or persist_directory125            )126 127        self._embedding_function = embedding_function128        self._collection = self._client.get_or_create_collection(129            name=collection_name,130            embedding_function=None,131            metadata=collection_metadata,132        )133        self.override_relevance_score_fn = relevance_score_fn134 135    @property136    def embeddings(self) -> Optional[Embeddings]:137        return self._embedding_function138 139    @xor_args(("query_texts", "query_embeddings"))140    def __query_collection(141        self,142        query_texts: Optional[List[str]] = None,143        query_embeddings: Optional[List[List[float]]] = None,144        n_results: int = 4,145        where: Optional[Dict[str, str]] = None,146        where_document: Optional[Dict[str, str]] = None,147        **kwargs: Any,148    ) -> List[Document]:149        """Query the chroma collection."""150        try:151            import chromadb  # noqa: F401152        except ImportError:153            raise ImportError(154                "Could not import chromadb python package. "155                "Please install it with `pip install chromadb`."156            )157        return self._collection.query(158            query_texts=query_texts,159            query_embeddings=query_embeddings,160            n_results=n_results,161            where=where,162            where_document=where_document,163            **kwargs,164        )165 166    def encode_image(self, uri: str) -> str:167        """Get base64 string from image URI."""168        with open(uri, "rb") as image_file:169            return base64.b64encode(image_file.read()).decode("utf-8")170 171    def add_images(172        self,173        uris: List[str],174        metadatas: Optional[List[dict]] = None,175        ids: Optional[List[str]] = None,176        **kwargs: Any,177    ) -> List[str]:178        """Run more images through the embeddings and add to the vectorstore.179 180        Args:181            uris List[str]: File path to the image.182            metadatas (Optional[List[dict]], optional): Optional list of metadatas.183            ids (Optional[List[str]], optional): Optional list of IDs.184 185        Returns:186            List[str]: List of IDs of the added images.187        """188        # Map from uris to b64 encoded strings189        b64_texts = [self.encode_image(uri=uri) for uri in uris]190        # Populate IDs191        if ids is None:192            ids = [str(uuid.uuid4()) for _ in uris]193        embeddings = None194        # Set embeddings195        if self._embedding_function is not None and hasattr(196            self._embedding_function, "embed_image"197        ):198            embeddings = self._embedding_function.embed_image(uris=uris)199        if metadatas:200            # fill metadatas with empty dicts if somebody201            # did not specify metadata for all images202            length_diff = len(uris) - len(metadatas)203            if length_diff:204                metadatas = metadatas + [{}] * length_diff205            empty_ids = []206            non_empty_ids = []207            for idx, m in enumerate(metadatas):208                if m:209                    non_empty_ids.append(idx)210                else:211                    empty_ids.append(idx)212            if non_empty_ids:213                metadatas = [metadatas[idx] for idx in non_empty_ids]214                images_with_metadatas = [b64_texts[idx] for idx in non_empty_ids]215                embeddings_with_metadatas = (216                    [embeddings[idx] for idx in non_empty_ids] if embeddings else None217                )218                ids_with_metadata = [ids[idx] for idx in non_empty_ids]219                try:220                    self._collection.upsert(221                        metadatas=metadatas,222                        embeddings=embeddings_with_metadatas,223                        documents=images_with_metadatas,224                        ids=ids_with_metadata,225                    )226                except ValueError as e:227                    if "Expected metadata value to be" in str(e):228                        msg = (229                            "Try filtering complex metadata using "230                            "langchain_community.vectorstores.utils.filter_complex_metadata."231                        )232                        raise ValueError(e.args[0] + "\n\n" + msg)233                    else:234                        raise e235            if empty_ids:236                images_without_metadatas = [b64_texts[j] for j in empty_ids]237                embeddings_without_metadatas = (238                    [embeddings[j] for j in empty_ids] if embeddings else None239                )240                ids_without_metadatas = [ids[j] for j in empty_ids]241                self._collection.upsert(242                    embeddings=embeddings_without_metadatas,243                    documents=images_without_metadatas,244                    ids=ids_without_metadatas,245                )246        else:247            self._collection.upsert(248                embeddings=embeddings,249                documents=b64_texts,250                ids=ids,251            )252        return ids253 254    def add_texts(255        self,256        texts: Iterable[str],257        metadatas: Optional[List[dict]] = None,258        ids: Optional[List[str]] = None,259        **kwargs: Any,260    ) -> List[str]:261        """Run more texts through the embeddings and add to the vectorstore.262 263        Args:264            texts (Iterable[str]): Texts to add to the vectorstore.265            metadatas (Optional[List[dict]], optional): Optional list of metadatas.266            ids (Optional[List[str]], optional): Optional list of IDs.267 268        Returns:269            List[str]: List of IDs of the added texts.270        """271        # TODO: Handle the case where the user doesn't provide ids on the Collection272        if ids is None:273            ids = [str(uuid.uuid4()) for _ in texts]274        embeddings = None275        texts = list(texts)276        if self._embedding_function is not None:277            embeddings = self._embedding_function.embed_documents(texts)278        if metadatas:279            # fill metadatas with empty dicts if somebody280            # did not specify metadata for all texts281            length_diff = len(texts) - len(metadatas)282            if length_diff:283                metadatas = metadatas + [{}] * length_diff284            empty_ids = []285            non_empty_ids = []286            for idx, m in enumerate(metadatas):287                if m:288                    non_empty_ids.append(idx)289                else:290                    empty_ids.append(idx)291            if non_empty_ids:292                metadatas = [metadatas[idx] for idx in non_empty_ids]293                texts_with_metadatas = [texts[idx] for idx in non_empty_ids]294                embeddings_with_metadatas = (295                    [embeddings[idx] for idx in non_empty_ids] if embeddings else None296                )297                ids_with_metadata = [ids[idx] for idx in non_empty_ids]298                try:299                    self._collection.upsert(300                        metadatas=metadatas,301                        embeddings=embeddings_with_metadatas,302                        documents=texts_with_metadatas,303                        ids=ids_with_metadata,304                    )305                except ValueError as e:306                    if "Expected metadata value to be" in str(e):307                        msg = (308                            "Try filtering complex metadata from the document using "309                            "langchain_community.vectorstores.utils.filter_complex_metadata."310                        )311                        raise ValueError(e.args[0] + "\n\n" + msg)312                    else:313                        raise e314            if empty_ids:315                texts_without_metadatas = [texts[j] for j in empty_ids]316                embeddings_without_metadatas = (317                    [embeddings[j] for j in empty_ids] if embeddings else None318                )319                ids_without_metadatas = [ids[j] for j in empty_ids]320                self._collection.upsert(321                    embeddings=embeddings_without_metadatas,322                    documents=texts_without_metadatas,323                    ids=ids_without_metadatas,324                )325        else:326            self._collection.upsert(327                embeddings=embeddings,328                documents=texts,329                ids=ids,330            )331        return ids332 333    def similarity_search(334        self,335        query: str,336        k: int = DEFAULT_K,337        filter: Optional[Dict[str, str]] = None,338        **kwargs: Any,339    ) -> List[Document]:340        """Run similarity search with Chroma.341 342        Args:343            query (str): Query text to search for.344            k (int): Number of results to return. Defaults to 4.345            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.346 347        Returns:348            List[Document]: List of documents most similar to the query text.349        """350        docs_and_scores = self.similarity_search_with_score(351            query, k, filter=filter, **kwargs352        )353        return [doc for doc, _ in docs_and_scores]354 355    def similarity_search_by_vector(356        self,357        embedding: List[float],358        k: int = DEFAULT_K,359        filter: Optional[Dict[str, str]] = None,360        where_document: Optional[Dict[str, str]] = None,361        **kwargs: Any,362    ) -> List[Document]:363        """Return docs most similar to embedding vector.364        Args:365            embedding (List[float]): Embedding to look up documents similar to.366            k (int): Number of Documents to return. Defaults to 4.367            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.368        Returns:369            List of Documents most similar to the query vector.370        """371        results = self.__query_collection(372            query_embeddings=embedding,373            n_results=k,374            where=filter,375            where_document=where_document,376            **kwargs,377        )378        return _results_to_docs(results)379 380    def similarity_search_by_vector_with_relevance_scores(381        self,382        embedding: List[float],383        k: int = DEFAULT_K,384        filter: Optional[Dict[str, str]] = None,385        where_document: Optional[Dict[str, str]] = None,386        **kwargs: Any,387    ) -> List[Tuple[Document, float]]:388        """389        Return docs most similar to embedding vector and similarity score.390 391        Args:392            embedding (List[float]): Embedding to look up documents similar to.393            k (int): Number of Documents to return. Defaults to 4.394            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.395 396        Returns:397            List[Tuple[Document, float]]: List of documents most similar to398            the query text and cosine distance in float for each.399            Lower score represents more similarity.400        """401        results = self.__query_collection(402            query_embeddings=embedding,403            n_results=k,404            where=filter,405            where_document=where_document,406            **kwargs,407        )408        return _results_to_docs_and_scores(results)409 410    def similarity_search_with_score(411        self,412        query: str,413        k: int = DEFAULT_K,414        filter: Optional[Dict[str, str]] = None,415        where_document: Optional[Dict[str, str]] = None,416        **kwargs: Any,417    ) -> List[Tuple[Document, float]]:418        """Run similarity search with Chroma with distance.419 420        Args:421            query (str): Query text to search for.422            k (int): Number of results to return. Defaults to 4.423            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.424 425        Returns:426            List[Tuple[Document, float]]: List of documents most similar to427            the query text and cosine distance in float for each.428            Lower score represents more similarity.429        """430        if self._embedding_function is None:431            results = self.__query_collection(432                query_texts=[query],433                n_results=k,434                where=filter,435                where_document=where_document,436                **kwargs,437            )438        else:439            query_embedding = self._embedding_function.embed_query(query)440            results = self.__query_collection(441                query_embeddings=[query_embedding],442                n_results=k,443                where=filter,444                where_document=where_document,445                **kwargs,446            )447 448        return _results_to_docs_and_scores(results)449 450    def _select_relevance_score_fn(self) -> Callable[[float], float]:451        """452        The 'correct' relevance function453        may differ depending on a few things, including:454        - the distance / similarity metric used by the VectorStore455        - the scale of your embeddings (OpenAI's are unit normed. Many others are not!)456        - embedding dimensionality457        - etc.458        """459        if self.override_relevance_score_fn:460            return self.override_relevance_score_fn461 462        distance = "l2"463        distance_key = "hnsw:space"464        metadata = self._collection.metadata465 466        if metadata and distance_key in metadata:467            distance = metadata[distance_key]468 469        if distance == "cosine":470            return self._cosine_relevance_score_fn471        elif distance == "l2":472            return self._euclidean_relevance_score_fn473        elif distance == "ip":474            return self._max_inner_product_relevance_score_fn475        else:476            raise ValueError(477                "No supported normalization function"478                f" for distance metric of type: {distance}."479                "Consider providing relevance_score_fn to Chroma constructor."480            )481 482    def similarity_search_by_image(483        self,484        uri: str,485        k: int = DEFAULT_K,486        filter: Optional[Dict[str, str]] = None,487        **kwargs: Any,488    ) -> List[Document]:489        """Search for similar images based on the given image URI.490 491        Args:492            uri (str): URI of the image to search for.493            k (int, optional): Number of results to return. Defaults to DEFAULT_K.494            filter (Optional[Dict[str, str]], optional): Filter by metadata.495            **kwargs (Any): Additional arguments to pass to function.496 497        Returns:498            List of Images most similar to the provided image.499            Each element in list is a Langchain Document Object.500            The page content is b64 encoded image, metadata is default or501            as defined by user.502 503        Raises:504            ValueError: If the embedding function does not support image embeddings.505        """506        if self._embedding_function is None or not hasattr(507            self._embedding_function, "embed_image"508        ):509            raise ValueError("The embedding function must support image embedding.")510 511        # Obtain image embedding512        # Assuming embed_image returns a single embedding513        image_embedding = self._embedding_function.embed_image(uris=[uri])514 515        # Perform similarity search based on the obtained embedding516        results = self.similarity_search_by_vector(517            embedding=image_embedding,518            k=k,519            filter=filter,520            **kwargs,521        )522 523        return results524 525    def similarity_search_by_image_with_relevance_score(526        self,527        uri: str,528        k: int = DEFAULT_K,529        filter: Optional[Dict[str, str]] = None,530        **kwargs: Any,531    ) -> List[Tuple[Document, float]]:532        """Search for similar images based on the given image URI.533 534        Args:535            uri (str): URI of the image to search for.536            k (int, optional): Number of results to return.537            Defaults to DEFAULT_K.538            filter (Optional[Dict[str, str]], optional): Filter by metadata.539            **kwargs (Any): Additional arguments to pass to function.540 541        Returns:542            List[Tuple[Document, float]]: List of tuples containing documents similar543            to the query image and their similarity scores.544            0th element in each tuple is a Langchain Document Object.545            The page content is b64 encoded img, metadata is default or defined by user.546 547        Raises:548            ValueError: If the embedding function does not support image embeddings.549        """550        if self._embedding_function is None or not hasattr(551            self._embedding_function, "embed_image"552        ):553            raise ValueError("The embedding function must support image embedding.")554 555        # Obtain image embedding556        # Assuming embed_image returns a single embedding557        image_embedding = self._embedding_function.embed_image(uris=[uri])558 559        # Perform similarity search based on the obtained embedding560        results = self.similarity_search_by_vector_with_relevance_scores(561            embedding=image_embedding,562            k=k,563            filter=filter,564            **kwargs,565        )566 567        return results568 569    def max_marginal_relevance_search_by_vector(570        self,571        embedding: List[float],572        k: int = DEFAULT_K,573        fetch_k: int = 20,574        lambda_mult: float = 0.5,575        filter: Optional[Dict[str, str]] = None,576        where_document: Optional[Dict[str, str]] = None,577        **kwargs: Any,578    ) -> List[Document]:579        """Return docs selected using the maximal marginal relevance.580        Maximal marginal relevance optimizes for similarity to query AND diversity581        among selected documents.582 583        Args:584            embedding: Embedding to look up documents similar to.585            k: Number of Documents to return. Defaults to 4.586            fetch_k: Number of Documents to fetch to pass to MMR algorithm.587            lambda_mult: Number between 0 and 1 that determines the degree588                        of diversity among the results with 0 corresponding589                        to maximum diversity and 1 to minimum diversity.590                        Defaults to 0.5.591            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.592 593        Returns:594            List of Documents selected by maximal marginal relevance.595        """596 597        results = self.__query_collection(598            query_embeddings=embedding,599            n_results=fetch_k,600            where=filter,601            where_document=where_document,602            include=["metadatas", "documents", "distances", "embeddings"],603            **kwargs,604        )605        mmr_selected = maximal_marginal_relevance(606            np.array(embedding, dtype=np.float32),607            results["embeddings"][0],608            k=k,609            lambda_mult=lambda_mult,610        )611 612        candidates = _results_to_docs(results)613 614        selected_results = [r for i, r in enumerate(candidates) if i in mmr_selected]615        return selected_results616 617    def max_marginal_relevance_search(618        self,619        query: str,620        k: int = DEFAULT_K,621        fetch_k: int = 20,622        lambda_mult: float = 0.5,623        filter: Optional[Dict[str, str]] = None,624        where_document: Optional[Dict[str, str]] = None,625        **kwargs: Any,626    ) -> List[Document]:627        """Return docs selected using the maximal marginal relevance.628        Maximal marginal relevance optimizes for similarity to query AND diversity629        among selected documents.630 631        Args:632            query: Text to look up documents similar to.633            k: Number of Documents to return. Defaults to 4.634            fetch_k: Number of Documents to fetch to pass to MMR algorithm.635            lambda_mult: Number between 0 and 1 that determines the degree636                        of diversity among the results with 0 corresponding637                        to maximum diversity and 1 to minimum diversity.638                        Defaults to 0.5.639            filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.640 641        Returns:642            List of Documents selected by maximal marginal relevance.643        """644        if self._embedding_function is None:645            raise ValueError(646                "For MMR search, you must specify an embedding function oncreation."647            )648 649        embedding = self._embedding_function.embed_query(query)650        docs = self.max_marginal_relevance_search_by_vector(651            embedding,652            k,653            fetch_k,654            lambda_mult=lambda_mult,655            filter=filter,656            where_document=where_document,657        )658        return docs659 660    def delete_collection(self) -> None:661        """Delete the collection."""662        self._client.delete_collection(self._collection.name)663 664    def get(665        self,666        ids: Optional[OneOrMany[ID]] = None,667        where: Optional[Where] = None,668        limit: Optional[int] = None,669        offset: Optional[int] = None,670        where_document: Optional[WhereDocument] = None,671        include: Optional[List[str]] = None,672    ) -> Dict[str, Any]:673        """Gets the collection.674 675        Args:676            ids: The ids of the embeddings to get. Optional.677            where: A Where type dict used to filter results by.678                   E.g. `{"color" : "red", "price": 4.20}`. Optional.679            limit: The number of documents to return. Optional.680            offset: The offset to start returning results from.681                    Useful for paging results with limit. Optional.682            where_document: A WhereDocument type dict used to filter by the documents.683                            E.g. `{$contains: "hello"}`. Optional.684            include: A list of what to include in the results.685                     Can contain `"embeddings"`, `"metadatas"`, `"documents"`.686                     Ids are always included.687                     Defaults to `["metadatas", "documents"]`. Optional.688        """689        kwargs = {690            "ids": ids,691            "where": where,692            "limit": limit,693            "offset": offset,694            "where_document": where_document,695        }696 697        if include is not None:698            kwargs["include"] = include699 700        return self._collection.get(**kwargs)701 702    @deprecated(703        since="0.1.17",704        message=(705            "Since Chroma 0.4.x the manual persistence method is no longer "706            "supported as docs are automatically persisted."707        ),708        removal="1.0",709    )710    def persist(self) -> None:711        """Persist the collection.712 713        This can be used to explicitly persist the data to disk.714        It will also be called automatically when the object is destroyed.715 716        Since Chroma 0.4.x the manual persistence method is no longer717        supported as docs are automatically persisted.718        """719        if self._persist_directory is None:720            raise ValueError(721                "You must specify a persist_directory on"722                "creation to persist the collection."723            )724        import chromadb725 726        # Maintain backwards compatibility with chromadb < 0.4.0727        major, minor, _ = chromadb.__version__.split(".")728        if int(major) == 0 and int(minor) < 4:729            self._client.persist()730 731    def update_document(self, document_id: str, document: Document) -> None:732        """Update a document in the collection.733 734        Args:735            document_id (str): ID of the document to update.736            document (Document): Document to update.737        """738        return self.update_documents([document_id], [document])739 740    def update_documents(self, ids: List[str], documents: List[Document]) -> None:741        """Update a document in the collection.742 743        Args:744            ids (List[str]): List of ids of the document to update.745            documents (List[Document]): List of documents to update.746        """747        text = [document.page_content for document in documents]748        metadata = [document.metadata for document in documents]749        if self._embedding_function is None:750            raise ValueError(751                "For update, you must specify an embedding function on creation."752            )753        embeddings = self._embedding_function.embed_documents(text)754 755        if hasattr(756            self._collection._client,757            "get_max_batch_size",  # for Chroma 0.5.1 and above758        ) or hasattr(759            self._collection._client, "max_batch_size"760        ):  # for Chroma 0.4.10 and above761            from chromadb.utils.batch_utils import create_batches762 763            for batch in create_batches(764                api=self._collection._client,765                ids=ids,766                metadatas=metadata,767                documents=text,768                embeddings=embeddings,769            ):770                self._collection.update(771                    ids=batch[0],772                    embeddings=batch[1],773                    documents=batch[3],774                    metadatas=batch[2],775                )776        else:777            self._collection.update(778                ids=ids,779                embeddings=embeddings,780                documents=text,781                metadatas=metadata,782            )783 784    @classmethod785    def from_texts(786        cls: Type[Chroma],787        texts: List[str],788        embedding: Optional[Embeddings] = None,789        metadatas: Optional[List[dict]] = None,790        ids: Optional[List[str]] = None,791        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,792        persist_directory: Optional[str] = None,793        client_settings: Optional[chromadb.config.Settings] = None,794        client: Optional[chromadb.Client] = None,795        collection_metadata: Optional[Dict] = None,796        **kwargs: Any,797    ) -> Chroma:798        """Create a Chroma vectorstore from a raw documents.799 800        If a persist_directory is specified, the collection will be persisted there.801        Otherwise, the data will be ephemeral in-memory.802 803        Args:804            texts (List[str]): List of texts to add to the collection.805            collection_name (str): Name of the collection to create.806            persist_directory (Optional[str]): Directory to persist the collection.807            embedding (Optional[Embeddings]): Embedding function. Defaults to None.808            metadatas (Optional[List[dict]]): List of metadatas. Defaults to None.809            ids (Optional[List[str]]): List of document IDs. Defaults to None.810            client_settings (Optional[chromadb.config.Settings]): Chroma client settings811            collection_metadata (Optional[Dict]): Collection configurations.812                                                  Defaults to None.813 814        Returns:815            Chroma: Chroma vectorstore.816        """817        chroma_collection = cls(818            collection_name=collection_name,819            embedding_function=embedding,820            persist_directory=persist_directory,821            client_settings=client_settings,822            client=client,823            collection_metadata=collection_metadata,824            **kwargs,825        )826        if ids is None:827            ids = [str(uuid.uuid4()) for _ in texts]828        if hasattr(829            chroma_collection._client,830            "get_max_batch_size",  # for Chroma 0.5.1 and above831        ) or hasattr(832            chroma_collection._client,833            "max_batch_size",834        ):  # for Chroma 0.4.10 and above835            from chromadb.utils.batch_utils import create_batches836 837            for batch in create_batches(838                api=chroma_collection._client,839                ids=ids,840                metadatas=metadatas,841                documents=texts,842            ):843                chroma_collection.add_texts(844                    texts=batch[3] if batch[3] else [],845                    metadatas=batch[2] if batch[2] else None,846                    ids=batch[0],847                )848        else:849            chroma_collection.add_texts(texts=texts, metadatas=metadatas, ids=ids)850        return chroma_collection851 852    @classmethod853    def from_documents(854        cls: Type[Chroma],855        documents: List[Document],856        embedding: Optional[Embeddings] = None,857        ids: Optional[List[str]] = None,858        collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,859        persist_directory: Optional[str] = None,860        client_settings: Optional[chromadb.config.Settings] = None,861        client: Optional[862            chromadb.Client863        ] = None,  # Add this line  # type: ignore[valid-type]864        collection_metadata: Optional[Dict] = None,865        **kwargs: Any,866    ) -> Chroma:867        """Create a Chroma vectorstore from a list of documents.868 869        If a persist_directory is specified, the collection will be persisted there.870        Otherwise, the data will be ephemeral in-memory.871 872        Args:873            collection_name (str): Name of the collection to create.874            persist_directory (Optional[str]): Directory to persist the collection.875            ids (Optional[List[str]]): List of document IDs. Defaults to None.876            documents (List[Document]): List of documents to add to the vectorstore.877            embedding (Optional[Embeddings]): Embedding function. Defaults to None.878            client_settings (Optional[chromadb.config.Settings]): Chroma client settings879            collection_metadata (Optional[Dict]): Collection configurations.880                                                  Defaults to None.881 882        Returns:883            Chroma: Chroma vectorstore.884        """885        texts = [doc.page_content for doc in documents]886        metadatas = [doc.metadata for doc in documents]887        return cls.from_texts(888            texts=texts,889            embedding=embedding,890            metadatas=metadatas,891            ids=ids,892            collection_name=collection_name,893            persist_directory=persist_directory,894            client_settings=client_settings,895            client=client,896            collection_metadata=collection_metadata,897            **kwargs,898        )899 900    def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> None:901        """Delete by vector IDs.902 903        Args:904            ids: List of ids to delete.905        """906        self._collection.delete(ids=ids, **kwargs)907 908    def __len__(self) -> int:909        """Count the number of documents in the collection."""910        return self._collection.count()911 
codekingpro/portable-devtools · Team Ai