codekingpro/portable-devtools
114k
1import logging2from typing import List, Optional3 4import requests5from langchain_core.embeddings import Embeddings6from pydantic import BaseModel7 8logger = logging.getLogger(__name__)9 10 11class LlamafileEmbeddings(BaseModel, Embeddings):12 """Llamafile lets you distribute and run large language models with a13 single file.14 15 To get started, see: https://github.com/Mozilla-Ocho/llamafile16 17 To use this class, you will need to first:18 19 1. Download a llamafile.20 2. Make the downloaded file executable: `chmod +x path/to/model.llamafile`21 3. Start the llamafile in server mode with embeddings enabled:22 23 `./path/to/model.llamafile --server --nobrowser --embedding`24 25 Example:26 .. code-block:: python27 28 from langchain_community.embeddings import LlamafileEmbeddings29 embedder = LlamafileEmbeddings()30 doc_embeddings = embedder.embed_documents(31 [32 "Alpha is the first letter of the Greek alphabet",33 "Beta is the second letter of the Greek alphabet",34 ]35 )36 query_embedding = embedder.embed_query(37 "What is the second letter of the Greek alphabet"38 )39 40 """41 42 base_url: str = "http://localhost:8080"43 """Base url where the llamafile server is listening."""44 45 request_timeout: Optional[int] = None46 """Timeout for server requests"""47 48 def _embed(self, text: str) -> List[float]:49 try:50 response = requests.post(51 url=f"{self.base_url}/embedding",52 headers={53 "Content-Type": "application/json",54 },55 json={56 "content": text,57 },58 timeout=self.request_timeout,59 )60 except requests.exceptions.ConnectionError:61 raise requests.exceptions.ConnectionError(62 f"Could not connect to Llamafile server. Please make sure "63 f"that a server is running at {self.base_url}."64 )65 66 # Raise exception if we got a bad (non-200) response status code67 response.raise_for_status()68 69 contents = response.json()70 if "embedding" not in contents:71 raise KeyError(72 "Unexpected output from /embedding endpoint, output dict "73 "missing 'embedding' key."74 )75 76 embedding = contents["embedding"]77 78 # Sanity check the embedding vector:79 # Prior to llamafile v0.6.2, if the server was not started with the80 # `--embedding` option, the embedding endpoint would always return a81 # 0-vector. See issue:82 # https://github.com/Mozilla-Ocho/llamafile/issues/24383 # So here we raise an exception if the vector sums to exactly 0.84 if sum(embedding) == 0.0:85 raise ValueError(86 "Embedding sums to 0, did you start the llamafile server with "87 "the `--embedding` option enabled?"88 )89 90 return embedding91 92 def embed_documents(self, texts: List[str]) -> List[List[float]]:93 """Embed documents using a llamafile server running at `self.base_url`.94 llamafile server should be started in a separate process before invoking95 this method.96 97 Args:98 texts: The list of texts to embed.99 100 Returns:101 List of embeddings, one for each text.102 """103 doc_embeddings = []104 for text in texts:105 doc_embeddings.append(self._embed(text))106 return doc_embeddings107 108 def embed_query(self, text: str) -> List[float]:109 """Embed a query using a llamafile server running at `self.base_url`.110 llamafile server should be started in a separate process before invoking111 this method.112 113 Args:114 text: The text to embed.115 116 Returns:117 Embeddings for the text.118 """119 return self._embed(text)120 