codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import base644import logging5import uuid6from typing import (7 TYPE_CHECKING,8 Any,9 Callable,10 Dict,11 Iterable,12 List,13 Optional,14 Tuple,15 Type,16)17 18import numpy as np19from langchain_core._api import deprecated20from langchain_core.documents import Document21from langchain_core.embeddings import Embeddings22from langchain_core.utils import xor_args23from langchain_core.vectorstores import VectorStore24 25from langchain_community.vectorstores.utils import maximal_marginal_relevance26 27if TYPE_CHECKING:28 import chromadb29 import chromadb.config30 from chromadb.api.types import ID, OneOrMany, Where, WhereDocument31 32logger = logging.getLogger()33DEFAULT_K = 4 # Number of Documents to return.34 35 36def _results_to_docs(results: Any) -> List[Document]:37 return [doc for doc, _ in _results_to_docs_and_scores(results)]38 39 40def _results_to_docs_and_scores(results: Any) -> List[Tuple[Document, float]]:41 return [42 # TODO: Chroma can do batch querying,43 # we shouldn't hard code to the 1st result44 (Document(page_content=result[0], metadata=result[1] or {}), result[2])45 for result in zip(46 results["documents"][0],47 results["metadatas"][0],48 results["distances"][0],49 )50 ]51 52 53@deprecated(since="0.2.9", removal="1.0", alternative_import="langchain_chroma.Chroma")54class Chroma(VectorStore):55 """`ChromaDB` vector store.56 57 To use, you should have the ``chromadb`` python package installed.58 59 Example:60 .. code-block:: python61 62 from langchain_community.vectorstores import Chroma63 from langchain_community.embeddings.openai import OpenAIEmbeddings64 65 embeddings = OpenAIEmbeddings()66 vectorstore = Chroma("langchain_store", embeddings)67 """68 69 _LANGCHAIN_DEFAULT_COLLECTION_NAME: str = "langchain"70 71 def __init__(72 self,73 collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,74 embedding_function: Optional[Embeddings] = None,75 persist_directory: Optional[str] = None,76 client_settings: Optional[chromadb.config.Settings] = None,77 collection_metadata: Optional[Dict] = None,78 client: Optional[chromadb.Client] = None,79 relevance_score_fn: Optional[Callable[[float], float]] = None,80 ) -> None:81 """Initialize with a Chroma client."""82 try:83 import chromadb84 import chromadb.config85 except ImportError:86 raise ImportError(87 "Could not import chromadb python package. "88 "Please install it with `pip install chromadb`."89 )90 91 if client is not None:92 self._client_settings = client_settings93 self._client = client94 self._persist_directory = persist_directory95 else:96 if client_settings:97 # If client_settings is provided with persist_directory specified,98 # then it is "in-memory and persisting to disk" mode.99 client_settings.persist_directory = (100 persist_directory or client_settings.persist_directory101 )102 if client_settings.persist_directory is not None:103 # Maintain backwards compatibility with chromadb < 0.4.0104 major, minor, _ = chromadb.__version__.split(".")105 if int(major) == 0 and int(minor) < 4:106 client_settings.chroma_db_impl = "duckdb+parquet"107 108 _client_settings = client_settings109 elif persist_directory:110 # Maintain backwards compatibility with chromadb < 0.4.0111 major, minor, _ = chromadb.__version__.split(".")112 if int(major) == 0 and int(minor) < 4:113 _client_settings = chromadb.config.Settings(114 chroma_db_impl="duckdb+parquet",115 )116 else:117 _client_settings = chromadb.config.Settings(is_persistent=True)118 _client_settings.persist_directory = persist_directory119 else:120 _client_settings = chromadb.config.Settings()121 self._client_settings = _client_settings122 self._client = chromadb.Client(_client_settings)123 self._persist_directory = (124 _client_settings.persist_directory or persist_directory125 )126 127 self._embedding_function = embedding_function128 self._collection = self._client.get_or_create_collection(129 name=collection_name,130 embedding_function=None,131 metadata=collection_metadata,132 )133 self.override_relevance_score_fn = relevance_score_fn134 135 @property136 def embeddings(self) -> Optional[Embeddings]:137 return self._embedding_function138 139 @xor_args(("query_texts", "query_embeddings"))140 def __query_collection(141 self,142 query_texts: Optional[List[str]] = None,143 query_embeddings: Optional[List[List[float]]] = None,144 n_results: int = 4,145 where: Optional[Dict[str, str]] = None,146 where_document: Optional[Dict[str, str]] = None,147 **kwargs: Any,148 ) -> List[Document]:149 """Query the chroma collection."""150 try:151 import chromadb # noqa: F401152 except ImportError:153 raise ImportError(154 "Could not import chromadb python package. "155 "Please install it with `pip install chromadb`."156 )157 return self._collection.query(158 query_texts=query_texts,159 query_embeddings=query_embeddings,160 n_results=n_results,161 where=where,162 where_document=where_document,163 **kwargs,164 )165 166 def encode_image(self, uri: str) -> str:167 """Get base64 string from image URI."""168 with open(uri, "rb") as image_file:169 return base64.b64encode(image_file.read()).decode("utf-8")170 171 def add_images(172 self,173 uris: List[str],174 metadatas: Optional[List[dict]] = None,175 ids: Optional[List[str]] = None,176 **kwargs: Any,177 ) -> List[str]:178 """Run more images through the embeddings and add to the vectorstore.179 180 Args:181 uris List[str]: File path to the image.182 metadatas (Optional[List[dict]], optional): Optional list of metadatas.183 ids (Optional[List[str]], optional): Optional list of IDs.184 185 Returns:186 List[str]: List of IDs of the added images.187 """188 # Map from uris to b64 encoded strings189 b64_texts = [self.encode_image(uri=uri) for uri in uris]190 # Populate IDs191 if ids is None:192 ids = [str(uuid.uuid4()) for _ in uris]193 embeddings = None194 # Set embeddings195 if self._embedding_function is not None and hasattr(196 self._embedding_function, "embed_image"197 ):198 embeddings = self._embedding_function.embed_image(uris=uris)199 if metadatas:200 # fill metadatas with empty dicts if somebody201 # did not specify metadata for all images202 length_diff = len(uris) - len(metadatas)203 if length_diff:204 metadatas = metadatas + [{}] * length_diff205 empty_ids = []206 non_empty_ids = []207 for idx, m in enumerate(metadatas):208 if m:209 non_empty_ids.append(idx)210 else:211 empty_ids.append(idx)212 if non_empty_ids:213 metadatas = [metadatas[idx] for idx in non_empty_ids]214 images_with_metadatas = [b64_texts[idx] for idx in non_empty_ids]215 embeddings_with_metadatas = (216 [embeddings[idx] for idx in non_empty_ids] if embeddings else None217 )218 ids_with_metadata = [ids[idx] for idx in non_empty_ids]219 try:220 self._collection.upsert(221 metadatas=metadatas,222 embeddings=embeddings_with_metadatas,223 documents=images_with_metadatas,224 ids=ids_with_metadata,225 )226 except ValueError as e:227 if "Expected metadata value to be" in str(e):228 msg = (229 "Try filtering complex metadata using "230 "langchain_community.vectorstores.utils.filter_complex_metadata."231 )232 raise ValueError(e.args[0] + "\n\n" + msg)233 else:234 raise e235 if empty_ids:236 images_without_metadatas = [b64_texts[j] for j in empty_ids]237 embeddings_without_metadatas = (238 [embeddings[j] for j in empty_ids] if embeddings else None239 )240 ids_without_metadatas = [ids[j] for j in empty_ids]241 self._collection.upsert(242 embeddings=embeddings_without_metadatas,243 documents=images_without_metadatas,244 ids=ids_without_metadatas,245 )246 else:247 self._collection.upsert(248 embeddings=embeddings,249 documents=b64_texts,250 ids=ids,251 )252 return ids253 254 def add_texts(255 self,256 texts: Iterable[str],257 metadatas: Optional[List[dict]] = None,258 ids: Optional[List[str]] = None,259 **kwargs: Any,260 ) -> List[str]:261 """Run more texts through the embeddings and add to the vectorstore.262 263 Args:264 texts (Iterable[str]): Texts to add to the vectorstore.265 metadatas (Optional[List[dict]], optional): Optional list of metadatas.266 ids (Optional[List[str]], optional): Optional list of IDs.267 268 Returns:269 List[str]: List of IDs of the added texts.270 """271 # TODO: Handle the case where the user doesn't provide ids on the Collection272 if ids is None:273 ids = [str(uuid.uuid4()) for _ in texts]274 embeddings = None275 texts = list(texts)276 if self._embedding_function is not None:277 embeddings = self._embedding_function.embed_documents(texts)278 if metadatas:279 # fill metadatas with empty dicts if somebody280 # did not specify metadata for all texts281 length_diff = len(texts) - len(metadatas)282 if length_diff:283 metadatas = metadatas + [{}] * length_diff284 empty_ids = []285 non_empty_ids = []286 for idx, m in enumerate(metadatas):287 if m:288 non_empty_ids.append(idx)289 else:290 empty_ids.append(idx)291 if non_empty_ids:292 metadatas = [metadatas[idx] for idx in non_empty_ids]293 texts_with_metadatas = [texts[idx] for idx in non_empty_ids]294 embeddings_with_metadatas = (295 [embeddings[idx] for idx in non_empty_ids] if embeddings else None296 )297 ids_with_metadata = [ids[idx] for idx in non_empty_ids]298 try:299 self._collection.upsert(300 metadatas=metadatas,301 embeddings=embeddings_with_metadatas,302 documents=texts_with_metadatas,303 ids=ids_with_metadata,304 )305 except ValueError as e:306 if "Expected metadata value to be" in str(e):307 msg = (308 "Try filtering complex metadata from the document using "309 "langchain_community.vectorstores.utils.filter_complex_metadata."310 )311 raise ValueError(e.args[0] + "\n\n" + msg)312 else:313 raise e314 if empty_ids:315 texts_without_metadatas = [texts[j] for j in empty_ids]316 embeddings_without_metadatas = (317 [embeddings[j] for j in empty_ids] if embeddings else None318 )319 ids_without_metadatas = [ids[j] for j in empty_ids]320 self._collection.upsert(321 embeddings=embeddings_without_metadatas,322 documents=texts_without_metadatas,323 ids=ids_without_metadatas,324 )325 else:326 self._collection.upsert(327 embeddings=embeddings,328 documents=texts,329 ids=ids,330 )331 return ids332 333 def similarity_search(334 self,335 query: str,336 k: int = DEFAULT_K,337 filter: Optional[Dict[str, str]] = None,338 **kwargs: Any,339 ) -> List[Document]:340 """Run similarity search with Chroma.341 342 Args:343 query (str): Query text to search for.344 k (int): Number of results to return. Defaults to 4.345 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.346 347 Returns:348 List[Document]: List of documents most similar to the query text.349 """350 docs_and_scores = self.similarity_search_with_score(351 query, k, filter=filter, **kwargs352 )353 return [doc for doc, _ in docs_and_scores]354 355 def similarity_search_by_vector(356 self,357 embedding: List[float],358 k: int = DEFAULT_K,359 filter: Optional[Dict[str, str]] = None,360 where_document: Optional[Dict[str, str]] = None,361 **kwargs: Any,362 ) -> List[Document]:363 """Return docs most similar to embedding vector.364 Args:365 embedding (List[float]): Embedding to look up documents similar to.366 k (int): Number of Documents to return. Defaults to 4.367 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.368 Returns:369 List of Documents most similar to the query vector.370 """371 results = self.__query_collection(372 query_embeddings=embedding,373 n_results=k,374 where=filter,375 where_document=where_document,376 **kwargs,377 )378 return _results_to_docs(results)379 380 def similarity_search_by_vector_with_relevance_scores(381 self,382 embedding: List[float],383 k: int = DEFAULT_K,384 filter: Optional[Dict[str, str]] = None,385 where_document: Optional[Dict[str, str]] = None,386 **kwargs: Any,387 ) -> List[Tuple[Document, float]]:388 """389 Return docs most similar to embedding vector and similarity score.390 391 Args:392 embedding (List[float]): Embedding to look up documents similar to.393 k (int): Number of Documents to return. Defaults to 4.394 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.395 396 Returns:397 List[Tuple[Document, float]]: List of documents most similar to398 the query text and cosine distance in float for each.399 Lower score represents more similarity.400 """401 results = self.__query_collection(402 query_embeddings=embedding,403 n_results=k,404 where=filter,405 where_document=where_document,406 **kwargs,407 )408 return _results_to_docs_and_scores(results)409 410 def similarity_search_with_score(411 self,412 query: str,413 k: int = DEFAULT_K,414 filter: Optional[Dict[str, str]] = None,415 where_document: Optional[Dict[str, str]] = None,416 **kwargs: Any,417 ) -> List[Tuple[Document, float]]:418 """Run similarity search with Chroma with distance.419 420 Args:421 query (str): Query text to search for.422 k (int): Number of results to return. Defaults to 4.423 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.424 425 Returns:426 List[Tuple[Document, float]]: List of documents most similar to427 the query text and cosine distance in float for each.428 Lower score represents more similarity.429 """430 if self._embedding_function is None:431 results = self.__query_collection(432 query_texts=[query],433 n_results=k,434 where=filter,435 where_document=where_document,436 **kwargs,437 )438 else:439 query_embedding = self._embedding_function.embed_query(query)440 results = self.__query_collection(441 query_embeddings=[query_embedding],442 n_results=k,443 where=filter,444 where_document=where_document,445 **kwargs,446 )447 448 return _results_to_docs_and_scores(results)449 450 def _select_relevance_score_fn(self) -> Callable[[float], float]:451 """452 The 'correct' relevance function453 may differ depending on a few things, including:454 - the distance / similarity metric used by the VectorStore455 - the scale of your embeddings (OpenAI's are unit normed. Many others are not!)456 - embedding dimensionality457 - etc.458 """459 if self.override_relevance_score_fn:460 return self.override_relevance_score_fn461 462 distance = "l2"463 distance_key = "hnsw:space"464 metadata = self._collection.metadata465 466 if metadata and distance_key in metadata:467 distance = metadata[distance_key]468 469 if distance == "cosine":470 return self._cosine_relevance_score_fn471 elif distance == "l2":472 return self._euclidean_relevance_score_fn473 elif distance == "ip":474 return self._max_inner_product_relevance_score_fn475 else:476 raise ValueError(477 "No supported normalization function"478 f" for distance metric of type: {distance}."479 "Consider providing relevance_score_fn to Chroma constructor."480 )481 482 def similarity_search_by_image(483 self,484 uri: str,485 k: int = DEFAULT_K,486 filter: Optional[Dict[str, str]] = None,487 **kwargs: Any,488 ) -> List[Document]:489 """Search for similar images based on the given image URI.490 491 Args:492 uri (str): URI of the image to search for.493 k (int, optional): Number of results to return. Defaults to DEFAULT_K.494 filter (Optional[Dict[str, str]], optional): Filter by metadata.495 **kwargs (Any): Additional arguments to pass to function.496 497 Returns:498 List of Images most similar to the provided image.499 Each element in list is a Langchain Document Object.500 The page content is b64 encoded image, metadata is default or501 as defined by user.502 503 Raises:504 ValueError: If the embedding function does not support image embeddings.505 """506 if self._embedding_function is None or not hasattr(507 self._embedding_function, "embed_image"508 ):509 raise ValueError("The embedding function must support image embedding.")510 511 # Obtain image embedding512 # Assuming embed_image returns a single embedding513 image_embedding = self._embedding_function.embed_image(uris=[uri])514 515 # Perform similarity search based on the obtained embedding516 results = self.similarity_search_by_vector(517 embedding=image_embedding,518 k=k,519 filter=filter,520 **kwargs,521 )522 523 return results524 525 def similarity_search_by_image_with_relevance_score(526 self,527 uri: str,528 k: int = DEFAULT_K,529 filter: Optional[Dict[str, str]] = None,530 **kwargs: Any,531 ) -> List[Tuple[Document, float]]:532 """Search for similar images based on the given image URI.533 534 Args:535 uri (str): URI of the image to search for.536 k (int, optional): Number of results to return.537 Defaults to DEFAULT_K.538 filter (Optional[Dict[str, str]], optional): Filter by metadata.539 **kwargs (Any): Additional arguments to pass to function.540 541 Returns:542 List[Tuple[Document, float]]: List of tuples containing documents similar543 to the query image and their similarity scores.544 0th element in each tuple is a Langchain Document Object.545 The page content is b64 encoded img, metadata is default or defined by user.546 547 Raises:548 ValueError: If the embedding function does not support image embeddings.549 """550 if self._embedding_function is None or not hasattr(551 self._embedding_function, "embed_image"552 ):553 raise ValueError("The embedding function must support image embedding.")554 555 # Obtain image embedding556 # Assuming embed_image returns a single embedding557 image_embedding = self._embedding_function.embed_image(uris=[uri])558 559 # Perform similarity search based on the obtained embedding560 results = self.similarity_search_by_vector_with_relevance_scores(561 embedding=image_embedding,562 k=k,563 filter=filter,564 **kwargs,565 )566 567 return results568 569 def max_marginal_relevance_search_by_vector(570 self,571 embedding: List[float],572 k: int = DEFAULT_K,573 fetch_k: int = 20,574 lambda_mult: float = 0.5,575 filter: Optional[Dict[str, str]] = None,576 where_document: Optional[Dict[str, str]] = None,577 **kwargs: Any,578 ) -> List[Document]:579 """Return docs selected using the maximal marginal relevance.580 Maximal marginal relevance optimizes for similarity to query AND diversity581 among selected documents.582 583 Args:584 embedding: Embedding to look up documents similar to.585 k: Number of Documents to return. Defaults to 4.586 fetch_k: Number of Documents to fetch to pass to MMR algorithm.587 lambda_mult: Number between 0 and 1 that determines the degree588 of diversity among the results with 0 corresponding589 to maximum diversity and 1 to minimum diversity.590 Defaults to 0.5.591 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.592 593 Returns:594 List of Documents selected by maximal marginal relevance.595 """596 597 results = self.__query_collection(598 query_embeddings=embedding,599 n_results=fetch_k,600 where=filter,601 where_document=where_document,602 include=["metadatas", "documents", "distances", "embeddings"],603 **kwargs,604 )605 mmr_selected = maximal_marginal_relevance(606 np.array(embedding, dtype=np.float32),607 results["embeddings"][0],608 k=k,609 lambda_mult=lambda_mult,610 )611 612 candidates = _results_to_docs(results)613 614 selected_results = [r for i, r in enumerate(candidates) if i in mmr_selected]615 return selected_results616 617 def max_marginal_relevance_search(618 self,619 query: str,620 k: int = DEFAULT_K,621 fetch_k: int = 20,622 lambda_mult: float = 0.5,623 filter: Optional[Dict[str, str]] = None,624 where_document: Optional[Dict[str, str]] = None,625 **kwargs: Any,626 ) -> List[Document]:627 """Return docs selected using the maximal marginal relevance.628 Maximal marginal relevance optimizes for similarity to query AND diversity629 among selected documents.630 631 Args:632 query: Text to look up documents similar to.633 k: Number of Documents to return. Defaults to 4.634 fetch_k: Number of Documents to fetch to pass to MMR algorithm.635 lambda_mult: Number between 0 and 1 that determines the degree636 of diversity among the results with 0 corresponding637 to maximum diversity and 1 to minimum diversity.638 Defaults to 0.5.639 filter (Optional[Dict[str, str]]): Filter by metadata. Defaults to None.640 641 Returns:642 List of Documents selected by maximal marginal relevance.643 """644 if self._embedding_function is None:645 raise ValueError(646 "For MMR search, you must specify an embedding function oncreation."647 )648 649 embedding = self._embedding_function.embed_query(query)650 docs = self.max_marginal_relevance_search_by_vector(651 embedding,652 k,653 fetch_k,654 lambda_mult=lambda_mult,655 filter=filter,656 where_document=where_document,657 )658 return docs659 660 def delete_collection(self) -> None:661 """Delete the collection."""662 self._client.delete_collection(self._collection.name)663 664 def get(665 self,666 ids: Optional[OneOrMany[ID]] = None,667 where: Optional[Where] = None,668 limit: Optional[int] = None,669 offset: Optional[int] = None,670 where_document: Optional[WhereDocument] = None,671 include: Optional[List[str]] = None,672 ) -> Dict[str, Any]:673 """Gets the collection.674 675 Args:676 ids: The ids of the embeddings to get. Optional.677 where: A Where type dict used to filter results by.678 E.g. `{"color" : "red", "price": 4.20}`. Optional.679 limit: The number of documents to return. Optional.680 offset: The offset to start returning results from.681 Useful for paging results with limit. Optional.682 where_document: A WhereDocument type dict used to filter by the documents.683 E.g. `{$contains: "hello"}`. Optional.684 include: A list of what to include in the results.685 Can contain `"embeddings"`, `"metadatas"`, `"documents"`.686 Ids are always included.687 Defaults to `["metadatas", "documents"]`. Optional.688 """689 kwargs = {690 "ids": ids,691 "where": where,692 "limit": limit,693 "offset": offset,694 "where_document": where_document,695 }696 697 if include is not None:698 kwargs["include"] = include699 700 return self._collection.get(**kwargs)701 702 @deprecated(703 since="0.1.17",704 message=(705 "Since Chroma 0.4.x the manual persistence method is no longer "706 "supported as docs are automatically persisted."707 ),708 removal="1.0",709 )710 def persist(self) -> None:711 """Persist the collection.712 713 This can be used to explicitly persist the data to disk.714 It will also be called automatically when the object is destroyed.715 716 Since Chroma 0.4.x the manual persistence method is no longer717 supported as docs are automatically persisted.718 """719 if self._persist_directory is None:720 raise ValueError(721 "You must specify a persist_directory on"722 "creation to persist the collection."723 )724 import chromadb725 726 # Maintain backwards compatibility with chromadb < 0.4.0727 major, minor, _ = chromadb.__version__.split(".")728 if int(major) == 0 and int(minor) < 4:729 self._client.persist()730 731 def update_document(self, document_id: str, document: Document) -> None:732 """Update a document in the collection.733 734 Args:735 document_id (str): ID of the document to update.736 document (Document): Document to update.737 """738 return self.update_documents([document_id], [document])739 740 def update_documents(self, ids: List[str], documents: List[Document]) -> None:741 """Update a document in the collection.742 743 Args:744 ids (List[str]): List of ids of the document to update.745 documents (List[Document]): List of documents to update.746 """747 text = [document.page_content for document in documents]748 metadata = [document.metadata for document in documents]749 if self._embedding_function is None:750 raise ValueError(751 "For update, you must specify an embedding function on creation."752 )753 embeddings = self._embedding_function.embed_documents(text)754 755 if hasattr(756 self._collection._client,757 "get_max_batch_size", # for Chroma 0.5.1 and above758 ) or hasattr(759 self._collection._client, "max_batch_size"760 ): # for Chroma 0.4.10 and above761 from chromadb.utils.batch_utils import create_batches762 763 for batch in create_batches(764 api=self._collection._client,765 ids=ids,766 metadatas=metadata,767 documents=text,768 embeddings=embeddings,769 ):770 self._collection.update(771 ids=batch[0],772 embeddings=batch[1],773 documents=batch[3],774 metadatas=batch[2],775 )776 else:777 self._collection.update(778 ids=ids,779 embeddings=embeddings,780 documents=text,781 metadatas=metadata,782 )783 784 @classmethod785 def from_texts(786 cls: Type[Chroma],787 texts: List[str],788 embedding: Optional[Embeddings] = None,789 metadatas: Optional[List[dict]] = None,790 ids: Optional[List[str]] = None,791 collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,792 persist_directory: Optional[str] = None,793 client_settings: Optional[chromadb.config.Settings] = None,794 client: Optional[chromadb.Client] = None,795 collection_metadata: Optional[Dict] = None,796 **kwargs: Any,797 ) -> Chroma:798 """Create a Chroma vectorstore from a raw documents.799 800 If a persist_directory is specified, the collection will be persisted there.801 Otherwise, the data will be ephemeral in-memory.802 803 Args:804 texts (List[str]): List of texts to add to the collection.805 collection_name (str): Name of the collection to create.806 persist_directory (Optional[str]): Directory to persist the collection.807 embedding (Optional[Embeddings]): Embedding function. Defaults to None.808 metadatas (Optional[List[dict]]): List of metadatas. Defaults to None.809 ids (Optional[List[str]]): List of document IDs. Defaults to None.810 client_settings (Optional[chromadb.config.Settings]): Chroma client settings811 collection_metadata (Optional[Dict]): Collection configurations.812 Defaults to None.813 814 Returns:815 Chroma: Chroma vectorstore.816 """817 chroma_collection = cls(818 collection_name=collection_name,819 embedding_function=embedding,820 persist_directory=persist_directory,821 client_settings=client_settings,822 client=client,823 collection_metadata=collection_metadata,824 **kwargs,825 )826 if ids is None:827 ids = [str(uuid.uuid4()) for _ in texts]828 if hasattr(829 chroma_collection._client,830 "get_max_batch_size", # for Chroma 0.5.1 and above831 ) or hasattr(832 chroma_collection._client,833 "max_batch_size",834 ): # for Chroma 0.4.10 and above835 from chromadb.utils.batch_utils import create_batches836 837 for batch in create_batches(838 api=chroma_collection._client,839 ids=ids,840 metadatas=metadatas,841 documents=texts,842 ):843 chroma_collection.add_texts(844 texts=batch[3] if batch[3] else [],845 metadatas=batch[2] if batch[2] else None,846 ids=batch[0],847 )848 else:849 chroma_collection.add_texts(texts=texts, metadatas=metadatas, ids=ids)850 return chroma_collection851 852 @classmethod853 def from_documents(854 cls: Type[Chroma],855 documents: List[Document],856 embedding: Optional[Embeddings] = None,857 ids: Optional[List[str]] = None,858 collection_name: str = _LANGCHAIN_DEFAULT_COLLECTION_NAME,859 persist_directory: Optional[str] = None,860 client_settings: Optional[chromadb.config.Settings] = None,861 client: Optional[862 chromadb.Client863 ] = None, # Add this line # type: ignore[valid-type]864 collection_metadata: Optional[Dict] = None,865 **kwargs: Any,866 ) -> Chroma:867 """Create a Chroma vectorstore from a list of documents.868 869 If a persist_directory is specified, the collection will be persisted there.870 Otherwise, the data will be ephemeral in-memory.871 872 Args:873 collection_name (str): Name of the collection to create.874 persist_directory (Optional[str]): Directory to persist the collection.875 ids (Optional[List[str]]): List of document IDs. Defaults to None.876 documents (List[Document]): List of documents to add to the vectorstore.877 embedding (Optional[Embeddings]): Embedding function. Defaults to None.878 client_settings (Optional[chromadb.config.Settings]): Chroma client settings879 collection_metadata (Optional[Dict]): Collection configurations.880 Defaults to None.881 882 Returns:883 Chroma: Chroma vectorstore.884 """885 texts = [doc.page_content for doc in documents]886 metadatas = [doc.metadata for doc in documents]887 return cls.from_texts(888 texts=texts,889 embedding=embedding,890 metadatas=metadatas,891 ids=ids,892 collection_name=collection_name,893 persist_directory=persist_directory,894 client_settings=client_settings,895 client=client,896 collection_metadata=collection_metadata,897 **kwargs,898 )899 900 def delete(self, ids: Optional[List[str]] = None, **kwargs: Any) -> None:901 """Delete by vector IDs.902 903 Args:904 ids: List of ids to delete.905 """906 self._collection.delete(ids=ids, **kwargs)907 908 def __len__(self) -> int:909 """Count the number of documents in the collection."""910 return self._collection.count()911 