Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
llamacpp.py354 linesDownload Raw Back to llms
1from __future__ import annotations2 3import logging4from pathlib import Path5from typing import Any, Dict, Iterator, List, Optional, Union6 7from langchain_core.callbacks import CallbackManagerForLLMRun8from langchain_core.language_models.llms import LLM9from langchain_core.outputs import GenerationChunk10from langchain_core.utils import get_pydantic_field_names, pre_init11from langchain_core.utils.utils import _build_model_kwargs12from pydantic import Field, model_validator13 14logger = logging.getLogger(__name__)15 16 17class LlamaCpp(LLM):18    """llama.cpp model.19 20    To use, you should have the llama-cpp-python library installed, and provide the21    path to the Llama model as a named parameter to the constructor.22    Check out: https://github.com/abetlen/llama-cpp-python23 24    Example:25        .. code-block:: python26 27            from langchain_community.llms import LlamaCpp28            llm = LlamaCpp(model_path="/path/to/llama/model")29    """30 31    client: Any = None  #: :meta private:32    model_path: str33    """The path to the Llama model file."""34 35    lora_base: Optional[str] = None36    """The path to the Llama LoRA base model."""37 38    lora_path: Optional[str] = None39    """The path to the Llama LoRA. If None, no LoRa is loaded."""40 41    n_ctx: int = Field(512, alias="n_ctx")42    """Token context window."""43 44    n_parts: int = Field(-1, alias="n_parts")45    """Number of parts to split the model into.46    If -1, the number of parts is automatically determined."""47 48    seed: int = Field(-1, alias="seed")49    """Seed. If -1, a random seed is used."""50 51    f16_kv: bool = Field(True, alias="f16_kv")52    """Use half-precision for key/value cache."""53 54    logits_all: bool = Field(False, alias="logits_all")55    """Return logits for all tokens, not just the last token."""56 57    vocab_only: bool = Field(False, alias="vocab_only")58    """Only load the vocabulary, no weights."""59 60    use_mlock: bool = Field(False, alias="use_mlock")61    """Force system to keep model in RAM."""62 63    n_threads: Optional[int] = Field(None, alias="n_threads")64    """Number of threads to use.65    If None, the number of threads is automatically determined."""66 67    n_batch: Optional[int] = Field(8, alias="n_batch")68    """Number of tokens to process in parallel.69    Should be a number between 1 and n_ctx."""70 71    n_gpu_layers: Optional[int] = Field(None, alias="n_gpu_layers")72    """Number of layers to be loaded into gpu memory. Default None."""73 74    suffix: Optional[str] = Field(None)75    """A suffix to append to the generated text. If None, no suffix is appended."""76 77    max_tokens: Optional[int] = 25678    """The maximum number of tokens to generate."""79 80    temperature: Optional[float] = 0.881    """The temperature to use for sampling."""82 83    top_p: Optional[float] = 0.9584    """The top-p value to use for sampling."""85 86    logprobs: Optional[int] = Field(None)87    """The number of logprobs to return. If None, no logprobs are returned."""88 89    echo: Optional[bool] = False90    """Whether to echo the prompt."""91 92    stop: Optional[List[str]] = []93    """A list of strings to stop generation when encountered."""94 95    repeat_penalty: Optional[float] = 1.196    """The penalty to apply to repeated tokens."""97 98    top_k: Optional[int] = 4099    """The top-k value to use for sampling."""100 101    last_n_tokens_size: Optional[int] = 64102    """The number of tokens to look back when applying the repeat_penalty."""103 104    use_mmap: Optional[bool] = True105    """Whether to keep the model loaded in RAM"""106 107    rope_freq_scale: float = 1.0108    """Scale factor for rope sampling."""109 110    rope_freq_base: float = 10000.0111    """Base frequency for rope sampling."""112 113    model_kwargs: Dict[str, Any] = Field(default_factory=dict)114    """Any additional parameters to pass to llama_cpp.Llama."""115 116    streaming: bool = True117    """Whether to stream the results, token by token."""118 119    grammar_path: Optional[Union[str, Path]] = None120    """121    grammar_path: Path to the .gbnf file that defines formal grammars122    for constraining model outputs. For instance, the grammar can be used123    to force the model to generate valid JSON or to speak exclusively in emojis. At most124    one of grammar_path and grammar should be passed in.125    """126    grammar: Optional[Union[str, Any]] = None127    """128    grammar: formal grammar for constraining model outputs. For instance, the grammar 129    can be used to force the model to generate valid JSON or to speak exclusively in 130    emojis. At most one of grammar_path and grammar should be passed in.131    """132 133    verbose: bool = True134    """Print verbose output to stderr."""135 136    @pre_init137    def validate_environment(cls, values: Dict) -> Dict:138        """Validate that llama-cpp-python library is installed."""139        try:140            from llama_cpp import Llama, LlamaGrammar141        except ImportError:142            raise ImportError(143                "Could not import llama-cpp-python library. "144                "Please install the llama-cpp-python library to "145                "use this embedding model: pip install llama-cpp-python"146            )147 148        model_path = values["model_path"]149        model_param_names = [150            "rope_freq_scale",151            "rope_freq_base",152            "lora_path",153            "lora_base",154            "n_ctx",155            "n_parts",156            "seed",157            "f16_kv",158            "logits_all",159            "vocab_only",160            "use_mlock",161            "n_threads",162            "n_batch",163            "use_mmap",164            "last_n_tokens_size",165            "verbose",166        ]167        model_params = {k: values[k] for k in model_param_names}168        # For backwards compatibility, only include if non-null.169        if values["n_gpu_layers"] is not None:170            model_params["n_gpu_layers"] = values["n_gpu_layers"]171 172        model_params.update(values["model_kwargs"])173 174        try:175            values["client"] = Llama(model_path, **model_params)176        except Exception as e:177            raise ValueError(178                f"Could not load Llama model from path: {model_path}. "179                f"Received error {e}"180            )181 182        if values["grammar"] and values["grammar_path"]:183            grammar = values["grammar"]184            grammar_path = values["grammar_path"]185            raise ValueError(186                "Can only pass in one of grammar and grammar_path. Received "187                f"{grammar=} and {grammar_path=}."188            )189        elif isinstance(values["grammar"], str):190            values["grammar"] = LlamaGrammar.from_string(values["grammar"])191        elif values["grammar_path"]:192            values["grammar"] = LlamaGrammar.from_file(values["grammar_path"])193        else:194            pass195        return values196 197    @model_validator(mode="before")198    @classmethod199    def build_model_kwargs(cls, values: Dict[str, Any]) -> Any:200        """Build extra kwargs from additional params that were passed in."""201        all_required_field_names = get_pydantic_field_names(cls)202        values = _build_model_kwargs(values, all_required_field_names)203        return values204 205    @property206    def _default_params(self) -> Dict[str, Any]:207        """Get the default parameters for calling llama_cpp."""208        params = {209            "suffix": self.suffix,210            "max_tokens": self.max_tokens,211            "temperature": self.temperature,212            "top_p": self.top_p,213            "logprobs": self.logprobs,214            "echo": self.echo,215            "stop_sequences": self.stop,  # key here is convention among LLM classes216            "repeat_penalty": self.repeat_penalty,217            "top_k": self.top_k,218        }219        if self.grammar:220            params["grammar"] = self.grammar221        return params222 223    @property224    def _identifying_params(self) -> Dict[str, Any]:225        """Get the identifying parameters."""226        return {**{"model_path": self.model_path}, **self._default_params}227 228    @property229    def _llm_type(self) -> str:230        """Return type of llm."""231        return "llamacpp"232 233    def _get_parameters(self, stop: Optional[List[str]] = None) -> Dict[str, Any]:234        """235        Performs sanity check, preparing parameters in format needed by llama_cpp.236 237        Args:238            stop (Optional[List[str]]): List of stop sequences for llama_cpp.239 240        Returns:241            Dictionary containing the combined parameters.242        """243 244        # Raise error if stop sequences are in both input and default params245        if self.stop and stop is not None:246            raise ValueError("`stop` found in both the input and default params.")247 248        params = self._default_params249 250        # llama_cpp expects the "stop" key not this, so we remove it:251        params.pop("stop_sequences")252 253        # then sets it as configured, or default to an empty list:254        params["stop"] = self.stop or stop or []255 256        return params257 258    def _call(259        self,260        prompt: str,261        stop: Optional[List[str]] = None,262        run_manager: Optional[CallbackManagerForLLMRun] = None,263        **kwargs: Any,264    ) -> str:265        """Call the Llama model and return the output.266 267        Args:268            prompt: The prompt to use for generation.269            stop: A list of strings to stop generation when encountered.270 271        Returns:272            The generated text.273 274        Example:275            .. code-block:: python276 277                from langchain_community.llms import LlamaCpp278                llm = LlamaCpp(model_path="/path/to/local/llama/model.bin")279                llm.invoke("This is a prompt.")280        """281        if self.streaming:282            # If streaming is enabled, we use the stream283            # method that yields as they are generated284            # and return the combined strings from the first choices's text:285            combined_text_output = ""286            for chunk in self._stream(287                prompt=prompt,288                stop=stop,289                run_manager=run_manager,290                **kwargs,291            ):292                combined_text_output += chunk.text293            return combined_text_output294        else:295            params = self._get_parameters(stop)296            params = {**params, **kwargs}297            result = self.client(prompt=prompt, **params)298            return result["choices"][0]["text"]299 300    def _stream(301        self,302        prompt: str,303        stop: Optional[List[str]] = None,304        run_manager: Optional[CallbackManagerForLLMRun] = None,305        **kwargs: Any,306    ) -> Iterator[GenerationChunk]:307        """Yields results objects as they are generated in real time.308 309        It also calls the callback manager's on_llm_new_token event with310        similar parameters to the OpenAI LLM class method of the same name.311 312        Args:313            prompt: The prompts to pass into the model.314            stop: Optional list of stop words to use when generating.315 316        Returns:317            A generator representing the stream of tokens being generated.318 319        Yields:320            A dictionary like objects containing a string token and metadata.321            See llama-cpp-python docs and below for more.322 323        Example:324            .. code-block:: python325 326                from langchain_community.llms import LlamaCpp327                llm = LlamaCpp(328                    model_path="/path/to/local/model.bin",329                    temperature = 0.5330                )331                for chunk in llm.stream("Ask 'Hi, how are you?' like a pirate:'",332                        stop=["'","\n"]):333                    result = chunk["choices"][0]334                    print(result["text"], end='', flush=True)  # noqa: T201335 336        """337        params = {**self._get_parameters(stop), **kwargs}338        result = self.client(prompt=prompt, stream=True, **params)339        for part in result:340            logprobs = part["choices"][0].get("logprobs", None)341            chunk = GenerationChunk(342                text=part["choices"][0]["text"],343                generation_info={"logprobs": logprobs},344            )345            if run_manager:346                run_manager.on_llm_new_token(347                    token=chunk.text, verbose=self.verbose, log_probs=logprobs348                )349            yield chunk350 351    def get_num_tokens(self, text: str) -> int:352        tokenized_text = self.client.tokenize(text.encode("utf-8"))353        return len(tokenized_text)354 
codekingpro/portable-devtools · Team Ai