codekingpro/portable-devtools
114k
1from typing import Any, Callable, List2 3from langchain_core.embeddings import Embeddings4from pydantic import ConfigDict5 6from langchain_community.llms.self_hosted import SelfHostedPipeline7 8 9def _embed_documents(pipeline: Any, *args: Any, **kwargs: Any) -> List[List[float]]:10 """Inference function to send to the remote hardware.11 12 Accepts a sentence_transformer model_id and13 returns a list of embeddings for each document in the batch.14 """15 return pipeline(*args, **kwargs)16 17 18class SelfHostedEmbeddings(SelfHostedPipeline, Embeddings):19 """Custom embedding models on self-hosted remote hardware.20 21 Supported hardware includes auto-launched instances on AWS, GCP, Azure,22 and Lambda, as well as servers specified23 by IP address and SSH credentials (such as on-prem, or another24 cloud like Paperspace, Coreweave, etc.).25 26 To use, you should have the ``runhouse`` python package installed.27 28 Example using a model load function:29 .. code-block:: python30 31 from langchain_community.embeddings import SelfHostedEmbeddings32 from transformers import AutoModelForCausalLM, AutoTokenizer, pipeline33 import runhouse as rh34 35 gpu = rh.cluster(name="rh-a10x", instance_type="A100:1")36 def get_pipeline():37 model_id = "facebook/bart-large"38 tokenizer = AutoTokenizer.from_pretrained(model_id)39 model = AutoModelForCausalLM.from_pretrained(model_id)40 return pipeline("feature-extraction", model=model, tokenizer=tokenizer)41 embeddings = SelfHostedEmbeddings(42 model_load_fn=get_pipeline,43 hardware=gpu44 model_reqs=["./", "torch", "transformers"],45 )46 Example passing in a pipeline path:47 .. code-block:: python48 49 from langchain_community.embeddings import SelfHostedHFEmbeddings50 import runhouse as rh51 from transformers import pipeline52 53 gpu = rh.cluster(name="rh-a10x", instance_type="A100:1")54 pipeline = pipeline(model="bert-base-uncased", task="feature-extraction")55 rh.blob(pickle.dumps(pipeline),56 path="models/pipeline.pkl").save().to(gpu, path="models")57 embeddings = SelfHostedHFEmbeddings.from_pipeline(58 pipeline="models/pipeline.pkl",59 hardware=gpu,60 model_reqs=["./", "torch", "transformers"],61 )62 """63 64 inference_fn: Callable = _embed_documents65 """Inference function to extract the embeddings on the remote hardware."""66 inference_kwargs: Any = None67 """Any kwargs to pass to the model's inference function."""68 69 model_config = ConfigDict(70 extra="forbid",71 )72 73 def embed_documents(self, texts: List[str]) -> List[List[float]]:74 """Compute doc embeddings using a HuggingFace transformer model.75 76 Args:77 texts: The list of texts to embed.s78 79 Returns:80 List of embeddings, one for each text.81 """82 texts = list(map(lambda x: x.replace("\n", " "), texts))83 embeddings = self.client(self.pipeline_ref, texts)84 if not isinstance(embeddings, list):85 return embeddings.tolist()86 return embeddings87 88 def embed_query(self, text: str) -> List[float]:89 """Compute query embeddings using a HuggingFace transformer model.90 91 Args:92 text: The text to embed.93 94 Returns:95 Embeddings for the text.96 """97 text = text.replace("\n", " ")98 embeddings = self.client(self.pipeline_ref, text)99 if not isinstance(embeddings, list):100 return embeddings.tolist()101 return embeddings102 