codekingpro/portable-devtools
114k
1"""2Pathway Vector Store client.3 4 5The Pathway Vector Server is a pipeline written in the Pathway framweork which indexes6all files in a given folder, embeds them, and builds a vector index. The pipeline reacts7to changes in source files, automatically updating appropriate index entries.8 9The PathwayVectorClient implements the LangChain VectorStore interface and queries the10PathwayVectorServer to retrieve up-to-date documents.11 12You can use the client with managed instances of Pathway Vector Store, or run your own13instance as described at https://pathway.com/developers/user-guide/llm-xpack/vectorstore_pipeline/14 15"""16 17import json18import logging19from typing import Any, Callable, Iterable, List, Optional, Tuple20 21import requests22from langchain_core.documents import Document23from langchain_core.embeddings import Embeddings24from langchain_core.vectorstores import VectorStore25 26 27# Copied from https://github.com/pathwaycom/pathway/blob/main/python/pathway/xpacks/llm/vector_store.py28# to remove dependency on Pathway library.29class _VectorStoreClient:30 def __init__(31 self,32 host: Optional[str] = None,33 port: Optional[int] = None,34 url: Optional[str] = None,35 ):36 """37 A client you can use to query :py:class:`VectorStoreServer`.38 39 Please provide aither the `url`, or `host` and `port`.40 41 Args:42 - host: host on which `:py:class:`VectorStoreServer` listens43 - port: port on which `:py:class:`VectorStoreServer` listens44 - url: url at which `:py:class:`VectorStoreServer` listens45 """46 err = "Either (`host` and `port`) or `url` must be provided, but not both."47 if url is not None:48 if host or port:49 raise ValueError(err)50 self.url = url51 else:52 if host is None:53 raise ValueError(err)54 port = port or 8055 self.url = f"http://{host}:{port}"56 57 def query(58 self, query: str, k: int = 3, metadata_filter: Optional[str] = None59 ) -> List[dict]:60 """61 Perform a query to the vector store and fetch results.62 63 Args:64 - query:65 - k: number of documents to be returned66 - metadata_filter: optional string representing the metadata filtering query67 in the JMESPath format. The search will happen only for documents68 satisfying this filtering.69 """70 71 data = {"query": query, "k": k}72 if metadata_filter is not None:73 data["metadata_filter"] = metadata_filter74 url = self.url + "/v1/retrieve"75 response = requests.post(76 url,77 data=json.dumps(data),78 headers={"Content-Type": "application/json"},79 timeout=3,80 )81 responses = response.json()82 return sorted(responses, key=lambda x: x["dist"])83 84 # Make an alias85 __call__ = query86 87 def get_vectorstore_statistics(self) -> dict:88 """Fetch basic statistics about the vector store."""89 90 url = self.url + "/v1/statistics"91 response = requests.post(92 url,93 json={},94 headers={"Content-Type": "application/json"},95 )96 responses = response.json()97 return responses98 99 def get_input_files(100 self,101 metadata_filter: Optional[str] = None,102 filepath_globpattern: Optional[str] = None,103 ) -> list:104 """105 Fetch information on documents in the vector store.106 107 Args:108 metadata_filter: optional string representing the metadata filtering query109 in the JMESPath format. The search will happen only for documents110 satisfying this filtering.111 filepath_globpattern: optional glob pattern specifying which documents112 will be searched for this query.113 """114 url = self.url + "/v1/inputs"115 response = requests.post(116 url,117 json={118 "metadata_filter": metadata_filter,119 "filepath_globpattern": filepath_globpattern,120 },121 headers={"Content-Type": "application/json"},122 )123 responses = response.json()124 return responses125 126 127class PathwayVectorClient(VectorStore):128 """129 VectorStore connecting to Pathway Vector Store.130 """131 132 def __init__(133 self,134 host: Optional[str] = None,135 port: Optional[int] = None,136 url: Optional[str] = None,137 ) -> None:138 """139 A client you can use to query Pathway Vector Store.140 141 Please provide aither the `url`, or `host` and `port`.142 143 Args:144 - host: host on which Pathway Vector Store listens145 - port: port on which Pathway Vector Store listens146 - url: url at which Pathway Vector Store listens147 """148 self.client = _VectorStoreClient(host, port, url)149 150 def add_texts(151 self,152 texts: Iterable[str],153 metadatas: Optional[List[dict]] = None,154 **kwargs: Any,155 ) -> List[str]:156 """Pathway is not suitable for this method."""157 raise NotImplementedError(158 "Pathway vector store does not support adding or removing texts"159 " from client."160 )161 162 @classmethod163 def from_texts(164 cls,165 texts: List[str],166 embedding: Embeddings,167 metadatas: Optional[List[dict]] = None,168 **kwargs: Any,169 ) -> "PathwayVectorClient":170 raise NotImplementedError(171 "Pathway vector store does not support initializing from_texts."172 )173 174 def similarity_search(175 self, query: str, k: int = 4, **kwargs: Any176 ) -> List[Document]:177 metadata_filter = kwargs.pop("metadata_filter", None)178 if kwargs:179 logging.warning(180 "Unknown kwargs passed to PathwayVectorClient.similarity_search: %s",181 kwargs,182 )183 rets = self.client(query=query, k=k, metadata_filter=metadata_filter)184 185 return [186 Document(page_content=ret["text"], metadata=ret["metadata"]) for ret in rets187 ]188 189 def similarity_search_with_score(190 self,191 query: str,192 k: int = 4,193 metadata_filter: Optional[str] = None,194 ) -> List[Tuple[Document, float]]:195 """Run similarity search with Pathway with distance.196 197 Args:198 - query (str): Query text to search for.199 - k (int): Number of results to return. Defaults to 4.200 - metadata_filter (Optional[str]): Filter by metadata.201 Filtering query should be in JMESPath format. Defaults to None.202 203 Returns:204 List[Tuple[Document, float]]: List of documents most similar to205 the query text and cosine distance in float for each.206 Lower score represents more similarity.207 """208 rets = self.client(query=query, k=k, metadata_filter=metadata_filter)209 210 return [211 (Document(page_content=ret["text"], metadata=ret["metadata"]), ret["dist"])212 for ret in rets213 ]214 215 def _select_relevance_score_fn(self) -> Callable[[float], float]:216 return self._cosine_relevance_score_fn217 218 def get_vectorstore_statistics(self) -> dict:219 """Fetch basic statistics about the Vector Store."""220 return self.client.get_vectorstore_statistics()221 222 def get_input_files(223 self,224 metadata_filter: Optional[str] = None,225 filepath_globpattern: Optional[str] = None,226 ) -> list:227 """List files indexed by the Vector Store."""228 return self.client.get_input_files(metadata_filter, filepath_globpattern)229 