codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import enum4import os5import random6import string7from hashlib import md58from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, Type9 10import numpy as np11from langchain_core.documents import Document12from langchain_core.embeddings import Embeddings13from langchain_core.vectorstores import VectorStore14 15from langchain_community.graphs import FalkorDBGraph16from langchain_community.vectorstores.utils import (17 DistanceStrategy,18 maximal_marginal_relevance,19)20 21 22def generate_random_string(length: int) -> str:23 # Define the characters to use: uppercase, lowercase, digits, and24 # punctuation25 characters = string.ascii_letters26 # Randomly choose 'length' characters from the pool of possible characters27 random_string = "".join(random.choice(characters) for _ in range(length))28 return random_string29 30 31DEFAULT_DISTANCE_STRATEGY = DistanceStrategy.COSINE32DISTANCE_MAPPING = {33 DistanceStrategy.EUCLIDEAN_DISTANCE: "euclidean",34 DistanceStrategy.COSINE: "cosine",35}36 37 38class SearchType(str, enum.Enum):39 """40 Enumerator for different search strategies in FalkorDB VectorStore.41 42 - `SearchType.VECTOR`: This option searches using only43 the vector indexes in the vectorstore, relying on the44 similarity between vector embeddings to return45 relevant results.46 47 - `SearchType.HYBRID`: This option performs a combined search,48 querying both the full-text indexes and the vector indexes.49 It integrates traditional text search with vector-based50 search for more comprehensive results.51 52 """53 54 VECTOR = "vector"55 HYBRID = "hybrid"56 57 58DEFAULT_SEARCH_TYPE = SearchType.VECTOR59 60 61class IndexType(str, enum.Enum):62 """Enumerator of the index types."""63 64 NODE = "NODE"65 RELATIONSHIP = "RELATIONSHIP"66 67 68DEFAULT_INDEX_TYPE = IndexType.NODE69 70 71def dict_to_yaml_str(input_dict: Dict, indent: int = 0) -> str:72 """73 Convert a dictionary to a YAML-like string without using external libraries.74 75 Parameters:76 - input_dict (dict): The dictionary to convert.77 - indent (int): The current indentation level.78 79 Returns:80 - str: The YAML-like string representation of the input dictionary.81 """82 yaml_str = ""83 for key, value in input_dict.items():84 padding = " " * indent85 if isinstance(value, dict):86 yaml_str += f"{padding}{key}:\n{dict_to_yaml_str(value, indent + 1)}"87 elif isinstance(value, list):88 yaml_str += f"{padding}{key}:\n"89 for item in value:90 yaml_str += f"{padding}- {item}\n"91 else:92 yaml_str += f"{padding}{key}: {value}\n"93 return yaml_str94 95 96def construct_metadata_filter(97 filter: Optional[Dict[str, Any]] = None,98) -> Tuple[str, Dict[str, Any]]:99 """100 Construct a metadata filter by directly injecting101 the filter values into the query.102 103 Args:104 filter (Optional[Dict[str, Any]]): Dictionary105 representing the filter condition.106 107 Returns:108 Tuple[str, Dict[str, Any]]: Filter snippet109 and an empty dictionary (since110 we don't need parameters).111 """112 if not filter:113 return "", {}114 115 filter_snippet = ""116 117 for i, (key, value) in enumerate(filter.items(), start=1):118 if filter_snippet:119 filter_snippet += " AND "120 121 # If the value is a string, wrap it in quotes. Otherwise, directly122 # inject the value.123 if isinstance(value, str):124 filter_snippet += f"n.{key} = '{value}'"125 else:126 filter_snippet += f"n.{key} = {value}"127 128 return filter_snippet, {}129 130 131def _get_search_index_query(132 search_type: SearchType, index_type: IndexType = DEFAULT_INDEX_TYPE133) -> str:134 if index_type == IndexType.NODE:135 if search_type == SearchType.VECTOR:136 return (137 "CALL db.idx.vector.queryNodes($entity_label, "138 "$entity_property, $k, vecf32($embedding)) "139 "YIELD node, score "140 "WITH node, (2 - score) / 2 AS score "141 )142 elif search_type == SearchType.HYBRID:143 return (144 "CALL { "145 "CALL db.idx.vector.queryNodes($entity_label, "146 "$entity_property, $k, vecf32($embedding)) "147 "YIELD node, score "148 "WITH collect({node: node, score: score})"149 " AS nodes, max(score) AS max_score "150 "UNWIND nodes AS n "151 "RETURN n.node AS node, (n.score / max_score) AS score "152 "UNION "153 "CALL db.idx.fulltext.queryNodes($entity_label, $query) "154 "YIELD node, score "155 "WITH collect({node: node, score: score})"156 " AS nodes, max(score) AS max_score "157 "UNWIND nodes AS n "158 "RETURN n.node AS node, (n.score / max_score) AS score "159 "} "160 "WITH node, max(score) AS score "161 "ORDER BY score DESC LIMIT $k "162 )163 elif index_type == IndexType.RELATIONSHIP:164 return (165 "CALL db.idx.vector.queryRelationships"166 "($entity_label, $entity_property, $k, vecf32($embedding)) "167 "YIELD relationship, score "168 )169 170 171def process_index_data(data: List[List[Any]]) -> List[Dict[str, Any]]:172 """173 Processes a nested list of entity data174 to extract information about labels,175 entity types, properties, index types,176 and index details (if applicable).177 178 Args:179 data (List[List[Any]]): A nested list containing180 details about entitys, their properties, index181 types, and configuration information.182 183 Returns:184 List[Dict[str, Any]]: A list of dictionaries where each dictionary185 contains:186 - entity_label (str): The label or name of the187 entity or relationship (e.g., 'Person', 'Song').188 - entity_property (str): The property of the entity189 or relationship on which an index190 was created (e.g., 'first_name').191 - index_type (str or List[str]): The type(s)192 of index applied to the property (e.g.,193 'FULLTEXT', 'VECTOR').194 - index_status (str): The status of the index195 (e.g., 'OPERATIONAL', 'PENDING').196 - index_dimension (Optional[int]): The dimension197 of the vector index, if applicable.198 - index_similarityFunction (Optional[str]): The199 similarity function used by the vector200 index, if applicable.201 - entity_type (str): The type of entity. That is202 either entity or relationship203 204 Notes:205 - The entity label is extracted from the first206 element of each entity list.207 - The entity property and associated index types208 are extracted from the second element.209 - If the index type includes 'VECTOR', additional210 details such as dimension and similarity function211 are extracted from the entity configuration.212 - The function handles cases where entitys have213 multiple index types (e.g., both 'FULLTEXT' and 'VECTOR').214 """215 216 result = []217 218 for entity in data:219 # Extract basic information220 221 entity_label = entity[0]222 223 index_type_dict = entity[2]224 225 index_status = entity[7]226 227 entity_type = entity[6]228 229 # Process each property and its index type(s)230 for prop, index_types in index_type_dict.items():231 entity_info = {232 "entity_label": entity_label,233 "entity_property": prop,234 "entity_type": entity_type,235 "index_type": index_types[0],236 "index_status": index_status,237 "index_dimension": None,238 "index_similarityFunction": None,239 }240 241 # Check for VECTOR type and extract additional details242 if "VECTOR" in index_types:243 if isinstance(entity[3], str):244 entity_info["index_dimension"] = None245 entity_info["index_similarityFunction"] = None246 else:247 vector_info = entity[3].get(prop, {})248 entity_info["index_dimension"] = vector_info.get("dimension")249 entity_info["index_similarityFunction"] = vector_info.get(250 "similarityFunction"251 )252 253 result.append(entity_info)254 255 return result256 257 258class FalkorDBVector(VectorStore):259 """`FalkorDB` vector index.260 261 To use, you should have the ``falkordb`` python package installed262 263 Args:264 host: FalkorDB host265 port: FalkorDB port266 username: Optionally provide your username267 details if you are connecting to a268 FalkorDB Cloud database instance269 password: Optionally provide your password270 details if you are connecting to a271 FalkorDB Cloud database instance272 embedding: Any embedding function implementing273 `langchain.embeddings.base.Embeddings` interface.274 distance_strategy The distance strategy to use.275 (default: "EUCLIDEAN")276 pre_delete_collection: If True, will delete277 existing data if it exists.(default:278 False). Useful for testing.279 search_type: Similiarity search type to use.280 Could be either SearchType.VECTOR or281 SearchType.HYBRID (default:282 SearchType.VECTOR)283 database: Optionally provide the name of the284 database to use else FalkorDBVector will285 generate a random database for you.286 node_label: Provide the label of the node you287 want the embeddings of your data to be288 stored in. (default: "Chunk")289 relation_type: Provide the relationship type290 of the relationship you want the291 embeddings of your data to be stored in.292 (default: "")293 embedding_node_property: Provide the name of294 the property in which you want your295 embeddings to be stored. (default: "embedding")296 text_node_property: Provide the name of297 the property in which you want your texts298 to be stored. (default: "text")299 embedding_dimension: Provide the dimension300 of your embeddings or it will be301 calculated for you.302 retrieval_query: Optionally a provide a303 retrieval_query else the default304 retrieval query will be used.305 index_type: Provide the index type for the306 VectorStore else the default index307 type will be used.308 graph: Optionally provide the graph you309 would like to use310 relevance_score_fn: Optionally provide a311 function that computes a relevance score312 based on the similarity score returned by313 the search.314 ssl: Specify whether the connection to the315 database should be secured using SSL/TLS316 encryption (default: False)317 318 Example:319 .. code-block:: python320 321 from langchain_community.vectorstores.falkordb_vector import FalkorDBVector322 from langchain_community.embeddings.openai import OpenAIEmbeddings323 from langchain_text_splitters import CharacterTextSplitter324 325 326 host="localhost"327 port=6379328 raw_documents = TextLoader('../../../state_of_the_union.txt').load()329 text_splitter = CharacterTextSplitter(chunk_size=1000, chunk_overlap=0)330 documents = text_splitter.split_documents(raw_documents)331 332 embeddings=OpenAIEmbeddings()333 vectorstore = FalkorDBVector.from_documents(334 embedding=embeddings,335 documents=documents,336 host=host,337 port=port,338 )339 """340 341 def __init__(342 self,343 embedding: Embeddings,344 *,345 search_type: SearchType = SearchType.VECTOR,346 username: Optional[str] = None,347 password: Optional[str] = None,348 host: str = "localhost",349 port: int = 6379,350 distance_strategy: DistanceStrategy = DEFAULT_DISTANCE_STRATEGY,351 database: Optional[str] = generate_random_string(4),352 node_label: str = "Chunk",353 relation_type: str = "",354 embedding_node_property: str = "embedding",355 text_node_property: str = "text",356 embedding_dimension: Optional[int] = None,357 retrieval_query: Optional[str] = "",358 index_type: IndexType = DEFAULT_INDEX_TYPE,359 graph: Optional[FalkorDBGraph] = None,360 relevance_score_fn: Optional[Callable[[float], float]] = None,361 ssl: bool = False,362 pre_delete_collection: bool = False,363 metadata: List[Any] = [],364 ) -> None:365 try:366 import falkordb367 except ImportError:368 raise ImportError(369 "Could not import falkordb python package."370 "Please install it with `pip install falkordb`"371 )372 373 try:374 import redis.exceptions375 except ImportError:376 raise ImportError(377 "Could not import redis.exceptions."378 "Please install it with `pip install redis`"379 )380 381 # Allow only cosine and euclidean distance strategies382 if distance_strategy not in [383 DistanceStrategy.EUCLIDEAN_DISTANCE,384 DistanceStrategy.COSINE,385 ]:386 raise ValueError(387 "`distance_strategy` must be either 'EULIDEAN_DISTANCE` or `COSINE`"388 )389 390 # Graph object takes precedent over env or input params391 if graph:392 self._database = graph._graph393 self._driver = graph._driver394 else:395 # Handle credentials via environment variables or input params396 self._host = host397 self._port = port398 self._username = username or os.environ.get("FALKORDB_USERNAME")399 self._password = password or os.environ.get("FALKORDB_PASSWORD")400 self._ssl = ssl401 402 # Initialize the FalkorDB connection403 try:404 self._driver = falkordb.FalkorDB(405 host=self._host,406 port=self._port,407 username=self._username,408 password=self._password,409 ssl=self._ssl,410 )411 except redis.exceptions.ConnectionError:412 raise ValueError(413 "Could not connect to FalkorDB database."414 "Please ensure that the host and port is correct"415 )416 except redis.exceptions.AuthenticationError:417 raise ValueError(418 "Could not connect to FalkorDB database. "419 "Please ensure that the username and password are correct"420 )421 422 # Verify that required values are not null423 if not embedding_node_property:424 raise ValueError(425 "The `embedding_node_property` must not be None or empty string"426 )427 if not node_label:428 raise ValueError("The `node_label` must not be None or empty string")429 430 self._database = self._driver.select_graph(database)431 self.database_name = database432 self.embedding = embedding433 self.node_label = node_label434 self.relation_type = relation_type435 self.embedding_node_property = embedding_node_property436 self.text_node_property = text_node_property437 self._distance_strategy = distance_strategy438 self.override_relevance_score_fn = relevance_score_fn439 self.pre_delete_collection = pre_delete_collection440 self.retrieval_query = retrieval_query441 self.search_type = search_type442 self._index_type = index_type443 self.metadata = metadata444 445 # Calculate embedding_dimensions if not given446 if not embedding_dimension:447 self.embedding_dimension = len(self.embedding.embed_query("foo"))448 449 # Delete existing data if flagged450 if pre_delete_collection:451 self._database.query(f"""MATCH (n:`{self.node_label}`) DELETE n""")452 453 @property454 def embeddings(self) -> Embeddings:455 """Returns the `Embeddings` model being used by the Vectorstore"""456 return self.embedding457 458 def _query(459 self,460 query: str,461 *,462 params: Optional[dict] = None,463 retry_on_timeout: bool = True,464 ) -> List[List]:465 """466 This method sends a Cypher query to the connected FalkorDB database467 and returns the results as a list of lists.468 469 Args:470 query (str): The Cypher query to execute.471 params (dict, optional): Dictionary of query parameters. Defaults to {}.472 473 Returns:474 List[List]: List of Lists containing the query results475 """476 params = params or {}477 try:478 data = self._database.query(query, params)479 return data.result_set480 except Exception as e:481 if "Invalid input" in str(e):482 raise ValueError(f"Cypher Statement is not valid\n{e}")483 if retry_on_timeout:484 return self._query(query, params=params, retry_on_timeout=False)485 else:486 raise e487 488 def retrieve_existing_node_index(489 self, node_label: Optional[str] = ""490 ) -> Tuple[Optional[int], Optional[str], Optional[str], Optional[str]]:491 """492 Check if the vector index exists in the FalkorDB database493 and returns its embedding dimension, entity_type,494 entity_label, entity_property495 496 This method;497 1. queries the FalkorDB database for existing indexes498 2. attempts to retrieve the dimension of499 the vector index with the specified node label500 & index type501 3. If the index exists, its dimension is returned.502 4. Else if the index doesn't exist, `None` is returned.503 504 Returns:505 int or None: The embedding dimension of the506 existing index if found,507 str or None: The entity type found.508 str or None: The label of the entity that the509 vector index was created with510 str or None: The property of the entity for511 which the vector index was created on512 513 514 """515 if node_label:516 pass517 elif self.node_label:518 node_label = self.node_label519 else:520 raise ValueError("`node_label` property must be set to use this function")521 522 embedding_dimension = None523 entity_type = None524 entity_label = None525 entity_property = None526 index_information = self._database.query("CALL db.indexes()")527 528 if index_information:529 processed_index_information = process_index_data(530 index_information.result_set531 )532 for dict in processed_index_information:533 if (534 dict.get("entity_label", False) == node_label535 and dict.get("entity_type", False) == "NODE"536 ):537 if dict["index_type"] == "VECTOR":538 embedding_dimension = int(dict["index_dimension"])539 entity_type = str(dict["entity_type"])540 entity_label = str(dict["entity_label"])541 entity_property = str(dict["entity_property"])542 break543 if embedding_dimension and entity_type and entity_label and entity_property:544 self._index_type = IndexType(entity_type)545 return embedding_dimension, entity_type, entity_label, entity_property546 else:547 return None, None, None, None548 else:549 return None, None, None, None550 551 def retrieve_existing_relationship_index(552 self, relation_type: Optional[str] = ""553 ) -> Tuple[Optional[int], Optional[str], Optional[str], Optional[str]]:554 """555 Check if the vector index exists in the FalkorDB database556 and returns its embedding dimension, entity_type, entity_label, entity_property557 558 This method;559 1. queries the FalkorDB database for existing indexes560 2. attempts to retrieve the dimension of the vector561 index with the specified label & index type562 3. If the index exists, its dimension is returned.563 4. Else if the index doesn't exist, `None` is returned.564 565 Returns:566 int or None: The embedding dimension of the existing index if found,567 str or None: The entity type found.568 str or None: The label of the entity that569 the vector index was created with570 str or None: The property of the entity for571 which the vector index was created on572 573 574 """575 if relation_type:576 pass577 elif self.relation_type:578 relation_type = self.relation_type579 else:580 raise ValueError(581 "Couldn't find any specified `relation_type`."582 " Check if you spelled it correctly"583 )584 585 embedding_dimension = None586 entity_type = None587 entity_label = None588 entity_property = None589 index_information = self._database.query("CALL db.indexes()")590 591 if index_information:592 processed_index_information = process_index_data(593 index_information.result_set594 )595 for dict in processed_index_information:596 if (597 dict.get("entity_label", False) == relation_type598 and dict.get("entity_type", False) == "RELATIONSHIP"599 ):600 if dict["index_type"] == "VECTOR":601 embedding_dimension = int(dict["index_dimension"])602 entity_type = str(dict["entity_type"])603 entity_label = str(dict["entity_label"])604 entity_property = str(dict["entity_property"])605 break606 if embedding_dimension and entity_type and entity_label and entity_property:607 self._index_type = IndexType(entity_type)608 return embedding_dimension, entity_type, entity_label, entity_property609 else:610 return None, None, None, None611 else:612 return None, None, None, None613 614 def retrieve_existing_fts_index(self) -> Optional[str]:615 """616 Check if the fulltext index exists in the FalkorDB database617 618 This method queries the FalkorDB database for existing fts indexes619 with the specified name.620 621 Returns:622 str: fulltext index entity label623 """624 625 entity_label = None626 index_information = self._database.query("CALL db.indexes()")627 if index_information:628 processed_index_information = process_index_data(629 index_information.result_set630 )631 for dict in processed_index_information:632 if dict.get("entity_label", False) == self.node_label:633 if dict["index_type"] == "FULLTEXT":634 entity_label = str(dict["entity_label"])635 break636 637 if entity_label:638 return entity_label639 else:640 return None641 else:642 return None643 644 def create_new_node_index(645 self,646 node_label: Optional[str] = "",647 embedding_node_property: Optional[str] = "",648 embedding_dimension: Optional[int] = None,649 ) -> None:650 """651 This method creates a new vector index652 on a node in FalkorDB.653 """654 if node_label:655 pass656 elif self.node_label:657 node_label = self.node_label658 else:659 raise ValueError("`node_label` property must be set to use this function")660 661 if embedding_node_property:662 pass663 elif self.embedding_node_property:664 embedding_node_property = self.embedding_node_property665 else:666 raise ValueError(667 "`embedding_node_property` property must be set to use this function"668 )669 670 if embedding_dimension:671 pass672 elif self.embedding_dimension:673 embedding_dimension = self.embedding_dimension674 else:675 raise ValueError(676 "`embedding_dimension` property must be set to use this function"677 )678 try:679 self._database.create_node_vector_index(680 node_label,681 embedding_node_property,682 dim=embedding_dimension,683 similarity_function=DISTANCE_MAPPING[self._distance_strategy],684 )685 except Exception as e:686 if "already indexed" in str(e):687 raise ValueError(688 f"A vector index on (:{node_label}"689 "{"690 f"{embedding_node_property}"691 "}) has already been created"692 )693 else:694 raise ValueError(f"Error occurred: {e}")695 696 def create_new_index_on_relationship(697 self,698 relation_type: str = "",699 embedding_node_property: str = "",700 embedding_dimension: int = 0,701 ) -> None:702 """703 This method creates an new vector index704 on a relationship/edge in FalkorDB.705 """706 if relation_type:707 pass708 elif self.relation_type:709 relation_type = self.relation_type710 else:711 raise ValueError("`relation_type` must be set to use this function")712 if embedding_node_property:713 pass714 elif self.embedding_node_property:715 embedding_node_property = self.embedding_node_property716 else:717 raise ValueError(718 "`embedding_node_property` must be set to use this function"719 )720 if embedding_dimension and embedding_dimension != 0:721 pass722 elif self.embedding_dimension:723 embedding_dimension = self.embedding_dimension724 else:725 raise ValueError("`embedding_dimension` must be set to use this function")726 727 try:728 self._database.create_edge_vector_index(729 relation_type,730 embedding_node_property,731 dim=embedding_dimension,732 similarity_function=DISTANCE_MAPPING[DEFAULT_DISTANCE_STRATEGY],733 )734 except Exception as e:735 if "already indexed" in str(e):736 raise ValueError(737 f"A vector index on [:{relation_type}"738 "{"739 f"{embedding_node_property}"740 "}] has already been created"741 )742 else:743 raise ValueError(f"Error occurred: {e}")744 745 def create_new_keyword_index(self, text_node_properties: List[str] = []) -> None:746 """747 This method constructs a Cypher query and executes it748 to create a new full text index in FalkorDB749 Args:750 text_node_properties (List[str]): List of node properties751 to be indexed.If not provided, defaults to752 self.text_node_property.753 """754 # Use the provided properties or default to self.text_node_property755 node_props = text_node_properties or [self.text_node_property]756 757 # Dynamically pass node label and properties to create the full-text758 # index759 self._database.create_node_fulltext_index(self.node_label, *node_props)760 761 def add_embeddings(762 self,763 texts: Iterable[str],764 embeddings: List[List[float]],765 metadatas: Optional[List[dict]] = None,766 ids: Optional[List[str]] = None,767 **kwargs: Any,768 ) -> List[str]:769 """Add embeddings to the vectorstore.770 771 Args:772 texts: Iterable of strings to add to the vectorstore.773 embeddings: List of list of embedding vectors.774 metadatas: List of metadatas associated with the texts.775 kwargs: vectorstore specific parameters776 """777 if ids is None:778 ids = [md5(text.encode("utf-8")).hexdigest() for text in texts]779 780 if not metadatas:781 metadatas = [{} for _ in texts]782 783 self.metadata = []784 785 # Check if all dictionaries are empty786 if all(not metadata for metadata in metadatas):787 pass788 else:789 # Initialize a set to keep track of unique non-empty keys790 unique_non_empty_keys: set[str] = set()791 792 # Iterate over each metadata dictionary793 for metadata in metadatas:794 # Add keys with non-empty values to the set795 unique_non_empty_keys.update(796 key for key, value in metadata.items() if value797 )798 799 # Print unique non-empty keys800 if unique_non_empty_keys:801 self.metadata = list(unique_non_empty_keys)802 803 parameters = {804 "data": [805 {"text": text, "metadata": metadata, "embedding": embedding, "id": id}806 for text, metadata, embedding, id in zip(807 texts, metadatas, embeddings, ids808 )809 ]810 }811 812 self._database.query(813 "UNWIND $data AS row "814 f"MERGE (c:`{self.node_label}` {{id: row.id}}) "815 f"SET c.`{self.embedding_node_property}`"816 f" = vecf32(row.embedding), c.`{self.text_node_property}`"817 " = row.text, c += row.metadata",818 params=parameters,819 )820 821 return ids822 823 def add_texts(824 self,825 texts: Iterable[str],826 metadatas: Optional[List[dict]] = None,827 ids: Optional[List[str]] = None,828 **kwargs: Any,829 ) -> List[str]:830 """Run more texts through the embeddings and add to the vectorstore.831 832 Args:833 texts: Iterable of strings to add to the vectorstore.834 metadatas: Optional list of metadatas associated with the texts.835 kwargs: vectorstore specific parameters836 Returns:837 List of ids from adding the texts into the vectorstore.838 """839 embeddings = self.embedding.embed_documents(list(texts))840 return self.add_embeddings(841 texts=texts, embeddings=embeddings, metadatas=metadatas, ids=ids, **kwargs842 )843 844 def add_documents(845 self,846 documents: List[Document],847 ids: Optional[List[str]] = None,848 **kwargs: Any,849 ) -> List[str]:850 """851 This function takes List[Document] element(s) and populates852 the existing store with a default node or default node(s) that853 represent the element(s) and returns the id(s) of the newly created node(s).854 855 Args:856 documents: the List[Document] element(s).857 ids: Optional List of custom IDs to assign to the documents.858 859 Returns:860 A list containing the id(s) of the newly created node in the store.861 """862 # Ensure the length of the ids matches the length of the documents if863 # provided864 if ids and len(ids) != len(documents):865 raise ValueError("The number of ids must match the number of documents.")866 867 result_ids = []868 869 # Add the documents to the store with custom or generated IDs870 self.from_documents(871 embedding=self.embedding,872 documents=documents,873 )874 875 for i, doc in enumerate(documents):876 page_content = doc.page_content877 if ids:878 # If custom IDs are provided, use them directly879 assigned_id = ids[i]880 self._query(881 """882 MATCH (n)883 WHERE n.text = $page_content884 SET n.id = $assigned_id885 """,886 params={"page_content": page_content, "assigned_id": assigned_id},887 )888 result_ids.append(assigned_id)889 890 else:891 # Use the existing logic to query the ID if no custom IDs were892 # provided893 result = self._query(894 """895 MATCH (n)896 WHERE n.text = $page_content897 RETURN n.id898 """,899 params={"page_content": page_content},900 )901 try:902 result_ids.append(result[0][0])903 904 except Exception:905 raise ValueError(906 "Your document wasn't added to the store"907 " successfully. Check your spellings."908 )909 910 return result_ids911 912 @classmethod913 def from_texts(914 cls: type[FalkorDBVector],915 texts: List[str],916 embedding: Embeddings,917 metadatas: Optional[List[Dict]] = None, # Optional918 distance_strategy: Optional[DistanceStrategy] = None, # Optional919 ids: Optional[List[str]] = None,920 **kwargs: Any,921 ) -> FalkorDBVector:922 """923 Return FalkorDBVector initialized from texts and embeddings.924 """925 embeddings = embedding.embed_documents(list(texts))926 927 # Set default values if None928 if metadatas is None:929 metadatas = [{} for _ in texts]930 if distance_strategy is None:931 distance_strategy = DEFAULT_DISTANCE_STRATEGY932 933 return cls.__from(934 texts,935 embeddings,936 embedding,937 metadatas=metadatas,938 ids=ids,939 distance_strategy=distance_strategy,940 **kwargs,941 )942 943 @classmethod944 def __from(945 cls,946 texts: List[str],947 embeddings: List[List[float]],948 embedding: Embeddings,949 metadatas: Optional[List[dict]] = None,950 ids: Optional[List[str]] = None,951 search_type: SearchType = SearchType.VECTOR,952 **kwargs: Any,953 ) -> FalkorDBVector:954 if ids is None:955 ids = [md5(text.encode("utf-8")).hexdigest() for text in texts]956 957 if not metadatas:958 metadatas = [{} for _ in texts]959 960 store = cls(961 embedding=embedding,962 search_type=search_type,963 **kwargs,964 )965 966 # Check if the vector index already exists967 embedding_dimension, index_type, entity_label, entity_property = (968 store.retrieve_existing_node_index()969 )970 971 # Raise error if relationship index type972 if index_type == "RELATIONSHIP":973 raise ValueError(974 "Data ingestion is not supported with relationship vector index"975 )976 977 # If the vector index doesn't exist yet978 if not index_type:979 store.create_new_node_index()980 embedding_dimension, index_type, entity_label, entity_property = (981 store.retrieve_existing_node_index()982 )983 984 # If the index already exists, check if embedding dimensions match985 elif (986 embedding_dimension and not store.embedding_dimension == embedding_dimension987 ):988 raise ValueError(989 f"A Vector index for {entity_label} on {entity_property} exists"990 "The provided embedding function and vector index "991 "dimensions do not match.\n"992 f"Embedding function dimension: {store.embedding_dimension}\n"993 f"Vector index dimension: {embedding_dimension}"994 )995 996 if search_type == SearchType.HYBRID:997 fts_node_label = store.retrieve_existing_fts_index()998 # If the FTS index doesn't exist yet999 if not fts_node_label:1000 store.create_new_keyword_index()1001 else: # Validate that FTS and Vector Index use the same information1002 if not fts_node_label == store.node_label:1003 raise ValueError(1004 "Vector and keyword index don't index the same node label"1005 )1006 1007 store.add_embeddings(1008 texts=texts, embeddings=embeddings, metadatas=metadatas, ids=ids, **kwargs1009 )1010 1011 return store1012 1013 @classmethod1014 def from_existing_index(1015 cls: Type[FalkorDBVector],1016 embedding: Embeddings,1017 node_label: str,1018 search_type: SearchType = DEFAULT_SEARCH_TYPE,1019 **kwargs: Any,1020 ) -> FalkorDBVector:1021 """1022 Get instance of an existing FalkorDB vector index. This method will1023 return the instance of the store without inserting any new1024 embeddings.1025 """1026 1027 store = cls(1028 embedding=embedding,1029 node_label=node_label,1030 search_type=search_type,1031 **kwargs,1032 )1033 1034 embedding_dimension, index_type, entity_label, entity_property = (1035 store.retrieve_existing_node_index()1036 )1037 1038 # Raise error if relationship index type1039 if index_type == "RELATIONSHIP":1040 raise ValueError(1041 "Relationship vector index is not supported with "1042 "`from_existing_index` method. Please use the "1043 "`from_existing_relationship_index` method."1044 )1045 1046 if not index_type:1047 raise ValueError(1048 f"The specified vector index node label `{node_label}` does not exist. "1049 "Make sure to check if you spelled the node label correctly"1050 )1051 1052 # Check if embedding function and vector index dimensions match1053 if embedding_dimension and not store.embedding_dimension == embedding_dimension:1054 raise ValueError(1055 "The provided embedding function and vector index "1056 "dimensions do not match.\n"1057 f"Embedding function dimension: {store.embedding_dimension}\n"1058 f"Vector index dimension: {embedding_dimension}"1059 )1060 1061 if search_type == SearchType.HYBRID:1062 fts_node_label = store.retrieve_existing_fts_index()1063 # If the FTS index doesn't exist yet1064 if not fts_node_label:1065 raise ValueError(1066 "The specified keyword index name does not exist. "1067 "Make sure to check if you spelled it correctly"1068 )1069 else: # Validate that FTS and Vector index use the same information1070 if not fts_node_label == store.node_label:1071 raise ValueError(1072 "Vector and keyword index don't index the same node label"1073 )1074 1075 return store1076 1077 @classmethod1078 def from_existing_relationship_index(1079 cls: Type[FalkorDBVector],1080 embedding: Embeddings,1081 relation_type: str,1082 search_type: SearchType = DEFAULT_SEARCH_TYPE,1083 **kwargs: Any,1084 ) -> FalkorDBVector:1085 """1086 Get instance of an existing FalkorDB relationship vector index.1087 This method will return the instance of the store without1088 inserting any new embeddings.1089 """1090 if search_type == SearchType.HYBRID:1091 raise ValueError(1092 "Hybrid search is not supported in combination "1093 "with relationship vector index"1094 )1095 1096 store = cls(1097 embedding=embedding,1098 relation_type=relation_type,1099 **kwargs,1100 )1101 1102 embedding_dimension, index_type, entity_label, entity_property = (1103 store.retrieve_existing_relationship_index()1104 )1105 1106 if not index_type:1107 raise ValueError(1108 "The specified vector index on the relationship"1109 f" {relation_type} does not exist. "1110 "Make sure to check if you spelled it correctly"1111 )1112 # Raise error if not relationship index type1113 if index_type == "NODE":1114 raise ValueError(1115 "Node vector index is not supported with "1116 "`from_existing_relationship_index` method. Please use the "1117 "`from_existing_index` method."1118 )1119 1120 # Check if embedding function and vector index dimensions match1121 if embedding_dimension and not store.embedding_dimension == embedding_dimension:1122 raise ValueError(1123 "The provided embedding function and vector index "1124 "dimensions do not match.\n"1125 f"Embedding function dimension: {store.embedding_dimension}\n"1126 f"Vector index dimension: {embedding_dimension}"1127 )1128 1129 return store1130 1131 @classmethod1132 def from_existing_graph(1133 cls: Type[FalkorDBVector],1134 embedding: Embeddings,1135 database: str,1136 node_label: str,1137 embedding_node_property: str,1138 text_node_properties: List[str],1139 *,1140 search_type: SearchType = DEFAULT_SEARCH_TYPE,1141 retrieval_query: str = "",1142 **kwargs: Any,1143 ) -> FalkorDBVector:1144 """1145 Initialize and return a FalkorDBVector instance1146 from an existing graph using the database name1147 1148 This method initializes a FalkorDBVector instance1149 using the provided parameters and the existing graph.1150 It validates the existence of the indices and creates1151 new ones if they don't exist.1152 1153 Args:1154 embedding: The `Embeddings` model you would like to use1155 database: The name of the existing graph/database you1156 would like to initialize1157 node_label: The label of the node you want to initialize.1158 embedding_node_property: The name of the property you1159 want your embeddings to be stored in.1160 1161 Returns:1162 FalkorDBVector: An instance of FalkorDBVector initialized1163 with the provided parameters and existing graph.1164 1165 Example:1166 >>> falkordb_vector = FalkorDBVector.from_existing_graph(1167 ... embedding=my_embedding,1168 ... node_label="Document",1169 ... embedding_node_property="embedding",1170 ... text_node_properties=["title", "content"]1171 ... )1172 1173 """1174 # Validate that database and text_node_properties is not empty1175 if not database:1176 raise ValueError("Parameter `database` must be given")1177 if not text_node_properties:1178 raise ValueError(1179 "Parameter `text_node_properties` must not be an empty list"1180 )1181 1182 # Prefer retrieval query from params, otherwise construct it1183 if not retrieval_query:1184 retrieval_query = (1185 f"RETURN reduce(str='', k IN {text_node_properties} |"1186 " str + '\\n' + k + ': ' + coalesce(node[k], '')) AS text, "1187 "node {.*, `"1188 + embedding_node_property1189 + "`: Null, id: Null, "1190 + ", ".join([f"`{prop}`: Null" for prop in text_node_properties])1191 + "} AS metadata, score"1192 )1193 1194 store = cls(1195 database=database,1196 embedding=embedding,1197 search_type=search_type,1198 retrieval_query=retrieval_query,1199 node_label=node_label,1200 embedding_node_property=embedding_node_property,