Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
llamafile.py120 linesDownload Raw Back to embeddings
1import logging2from typing import List, Optional3 4import requests5from langchain_core.embeddings import Embeddings6from pydantic import BaseModel7 8logger = logging.getLogger(__name__)9 10 11class LlamafileEmbeddings(BaseModel, Embeddings):12    """Llamafile lets you distribute and run large language models with a13    single file.14 15    To get started, see: https://github.com/Mozilla-Ocho/llamafile16 17    To use this class, you will need to first:18 19    1. Download a llamafile.20    2. Make the downloaded file executable: `chmod +x path/to/model.llamafile`21    3. Start the llamafile in server mode with embeddings enabled:22 23        `./path/to/model.llamafile --server --nobrowser --embedding`24 25    Example:26        .. code-block:: python27 28            from langchain_community.embeddings import LlamafileEmbeddings29            embedder = LlamafileEmbeddings()30            doc_embeddings = embedder.embed_documents(31                [32                    "Alpha is the first letter of the Greek alphabet",33                    "Beta is the second letter of the Greek alphabet",34                ]35            )36            query_embedding = embedder.embed_query(37                "What is the second letter of the Greek alphabet"38            )39 40    """41 42    base_url: str = "http://localhost:8080"43    """Base url where the llamafile server is listening."""44 45    request_timeout: Optional[int] = None46    """Timeout for server requests"""47 48    def _embed(self, text: str) -> List[float]:49        try:50            response = requests.post(51                url=f"{self.base_url}/embedding",52                headers={53                    "Content-Type": "application/json",54                },55                json={56                    "content": text,57                },58                timeout=self.request_timeout,59            )60        except requests.exceptions.ConnectionError:61            raise requests.exceptions.ConnectionError(62                f"Could not connect to Llamafile server. Please make sure "63                f"that a server is running at {self.base_url}."64            )65 66        # Raise exception if we got a bad (non-200) response status code67        response.raise_for_status()68 69        contents = response.json()70        if "embedding" not in contents:71            raise KeyError(72                "Unexpected output from /embedding endpoint, output dict "73                "missing 'embedding' key."74            )75 76        embedding = contents["embedding"]77 78        # Sanity check the embedding vector:79        # Prior to llamafile v0.6.2, if the server was not started with the80        # `--embedding` option, the embedding endpoint would always return a81        # 0-vector. See issue:82        # https://github.com/Mozilla-Ocho/llamafile/issues/24383        # So here we raise an exception if the vector sums to exactly 0.84        if sum(embedding) == 0.0:85            raise ValueError(86                "Embedding sums to 0, did you start the llamafile server with "87                "the `--embedding` option enabled?"88            )89 90        return embedding91 92    def embed_documents(self, texts: List[str]) -> List[List[float]]:93        """Embed documents using a llamafile server running at `self.base_url`.94        llamafile server should be started in a separate process before invoking95        this method.96 97        Args:98            texts: The list of texts to embed.99 100        Returns:101            List of embeddings, one for each text.102        """103        doc_embeddings = []104        for text in texts:105            doc_embeddings.append(self._embed(text))106        return doc_embeddings107 108    def embed_query(self, text: str) -> List[float]:109        """Embed a query using a llamafile server running at `self.base_url`.110        llamafile server should be started in a separate process before invoking111        this method.112 113        Args:114            text: The text to embed.115 116        Returns:117            Embeddings for the text.118        """119        return self._embed(text)120 
codekingpro/portable-devtools · Team Ai