codekingpro/portable-devtools
114k
1from typing import Any, List2 3from langchain_core.embeddings import Embeddings4from pydantic import BaseModel, ConfigDict5 6DEFAULT_MODEL_URL = "https://tfhub.dev/google/universal-sentence-encoder-multilingual/3"7 8 9class TensorflowHubEmbeddings(BaseModel, Embeddings):10 """TensorflowHub embedding models.11 12 To use, you should have the ``tensorflow_text`` python package installed.13 14 Example:15 .. code-block:: python16 17 from langchain_community.embeddings import TensorflowHubEmbeddings18 url = "https://tfhub.dev/google/universal-sentence-encoder-multilingual/3"19 tf = TensorflowHubEmbeddings(model_url=url)20 """21 22 embed: Any = None #: :meta private:23 model_url: str = DEFAULT_MODEL_URL24 """Model name to use."""25 26 def __init__(self, **kwargs: Any):27 """Initialize the tensorflow_hub and tensorflow_text."""28 super().__init__(**kwargs)29 try:30 import tensorflow_hub31 except ImportError:32 raise ImportError(33 "Could not import tensorflow-hub python package. "34 "Please install it with `pip install tensorflow-hub``."35 )36 try:37 import tensorflow_text # noqa38 except ImportError:39 raise ImportError(40 "Could not import tensorflow_text python package. "41 "Please install it with `pip install tensorflow_text``."42 )43 44 self.embed = tensorflow_hub.load(self.model_url)45 46 model_config = ConfigDict(47 extra="forbid",48 protected_namespaces=(),49 )50 51 def embed_documents(self, texts: List[str]) -> List[List[float]]:52 """Compute doc embeddings using a TensorflowHub embedding model.53 54 Args:55 texts: The list of texts to embed.56 57 Returns:58 List of embeddings, one for each text.59 """60 texts = list(map(lambda x: x.replace("\n", " "), texts))61 embeddings = self.embed(texts).numpy()62 return embeddings.tolist()63 64 def embed_query(self, text: str) -> List[float]:65 """Compute query embeddings using a TensorflowHub embedding model.66 67 Args:68 text: The text to embed.69 70 Returns:71 Embeddings for the text.72 """73 text = text.replace("\n", " ")74 embedding = self.embed([text]).numpy()[0]75 return embedding.tolist()76 