codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import logging4from pathlib import Path5from typing import Any, Dict, Iterator, List, Optional, Union6 7from langchain_core.callbacks import CallbackManagerForLLMRun8from langchain_core.language_models.llms import LLM9from langchain_core.outputs import GenerationChunk10from langchain_core.utils import get_pydantic_field_names, pre_init11from langchain_core.utils.utils import _build_model_kwargs12from pydantic import Field, model_validator13 14logger = logging.getLogger(__name__)15 16 17class LlamaCpp(LLM):18 """llama.cpp model.19 20 To use, you should have the llama-cpp-python library installed, and provide the21 path to the Llama model as a named parameter to the constructor.22 Check out: https://github.com/abetlen/llama-cpp-python23 24 Example:25 .. code-block:: python26 27 from langchain_community.llms import LlamaCpp28 llm = LlamaCpp(model_path="/path/to/llama/model")29 """30 31 client: Any = None #: :meta private:32 model_path: str33 """The path to the Llama model file."""34 35 lora_base: Optional[str] = None36 """The path to the Llama LoRA base model."""37 38 lora_path: Optional[str] = None39 """The path to the Llama LoRA. If None, no LoRa is loaded."""40 41 n_ctx: int = Field(512, alias="n_ctx")42 """Token context window."""43 44 n_parts: int = Field(-1, alias="n_parts")45 """Number of parts to split the model into.46 If -1, the number of parts is automatically determined."""47 48 seed: int = Field(-1, alias="seed")49 """Seed. If -1, a random seed is used."""50 51 f16_kv: bool = Field(True, alias="f16_kv")52 """Use half-precision for key/value cache."""53 54 logits_all: bool = Field(False, alias="logits_all")55 """Return logits for all tokens, not just the last token."""56 57 vocab_only: bool = Field(False, alias="vocab_only")58 """Only load the vocabulary, no weights."""59 60 use_mlock: bool = Field(False, alias="use_mlock")61 """Force system to keep model in RAM."""62 63 n_threads: Optional[int] = Field(None, alias="n_threads")64 """Number of threads to use.65 If None, the number of threads is automatically determined."""66 67 n_batch: Optional[int] = Field(8, alias="n_batch")68 """Number of tokens to process in parallel.69 Should be a number between 1 and n_ctx."""70 71 n_gpu_layers: Optional[int] = Field(None, alias="n_gpu_layers")72 """Number of layers to be loaded into gpu memory. Default None."""73 74 suffix: Optional[str] = Field(None)75 """A suffix to append to the generated text. If None, no suffix is appended."""76 77 max_tokens: Optional[int] = 25678 """The maximum number of tokens to generate."""79 80 temperature: Optional[float] = 0.881 """The temperature to use for sampling."""82 83 top_p: Optional[float] = 0.9584 """The top-p value to use for sampling."""85 86 logprobs: Optional[int] = Field(None)87 """The number of logprobs to return. If None, no logprobs are returned."""88 89 echo: Optional[bool] = False90 """Whether to echo the prompt."""91 92 stop: Optional[List[str]] = []93 """A list of strings to stop generation when encountered."""94 95 repeat_penalty: Optional[float] = 1.196 """The penalty to apply to repeated tokens."""97 98 top_k: Optional[int] = 4099 """The top-k value to use for sampling."""100 101 last_n_tokens_size: Optional[int] = 64102 """The number of tokens to look back when applying the repeat_penalty."""103 104 use_mmap: Optional[bool] = True105 """Whether to keep the model loaded in RAM"""106 107 rope_freq_scale: float = 1.0108 """Scale factor for rope sampling."""109 110 rope_freq_base: float = 10000.0111 """Base frequency for rope sampling."""112 113 model_kwargs: Dict[str, Any] = Field(default_factory=dict)114 """Any additional parameters to pass to llama_cpp.Llama."""115 116 streaming: bool = True117 """Whether to stream the results, token by token."""118 119 grammar_path: Optional[Union[str, Path]] = None120 """121 grammar_path: Path to the .gbnf file that defines formal grammars122 for constraining model outputs. For instance, the grammar can be used123 to force the model to generate valid JSON or to speak exclusively in emojis. At most124 one of grammar_path and grammar should be passed in.125 """126 grammar: Optional[Union[str, Any]] = None127 """128 grammar: formal grammar for constraining model outputs. For instance, the grammar 129 can be used to force the model to generate valid JSON or to speak exclusively in 130 emojis. At most one of grammar_path and grammar should be passed in.131 """132 133 verbose: bool = True134 """Print verbose output to stderr."""135 136 @pre_init137 def validate_environment(cls, values: Dict) -> Dict:138 """Validate that llama-cpp-python library is installed."""139 try:140 from llama_cpp import Llama, LlamaGrammar141 except ImportError:142 raise ImportError(143 "Could not import llama-cpp-python library. "144 "Please install the llama-cpp-python library to "145 "use this embedding model: pip install llama-cpp-python"146 )147 148 model_path = values["model_path"]149 model_param_names = [150 "rope_freq_scale",151 "rope_freq_base",152 "lora_path",153 "lora_base",154 "n_ctx",155 "n_parts",156 "seed",157 "f16_kv",158 "logits_all",159 "vocab_only",160 "use_mlock",161 "n_threads",162 "n_batch",163 "use_mmap",164 "last_n_tokens_size",165 "verbose",166 ]167 model_params = {k: values[k] for k in model_param_names}168 # For backwards compatibility, only include if non-null.169 if values["n_gpu_layers"] is not None:170 model_params["n_gpu_layers"] = values["n_gpu_layers"]171 172 model_params.update(values["model_kwargs"])173 174 try:175 values["client"] = Llama(model_path, **model_params)176 except Exception as e:177 raise ValueError(178 f"Could not load Llama model from path: {model_path}. "179 f"Received error {e}"180 )181 182 if values["grammar"] and values["grammar_path"]:183 grammar = values["grammar"]184 grammar_path = values["grammar_path"]185 raise ValueError(186 "Can only pass in one of grammar and grammar_path. Received "187 f"{grammar=} and {grammar_path=}."188 )189 elif isinstance(values["grammar"], str):190 values["grammar"] = LlamaGrammar.from_string(values["grammar"])191 elif values["grammar_path"]:192 values["grammar"] = LlamaGrammar.from_file(values["grammar_path"])193 else:194 pass195 return values196 197 @model_validator(mode="before")198 @classmethod199 def build_model_kwargs(cls, values: Dict[str, Any]) -> Any:200 """Build extra kwargs from additional params that were passed in."""201 all_required_field_names = get_pydantic_field_names(cls)202 values = _build_model_kwargs(values, all_required_field_names)203 return values204 205 @property206 def _default_params(self) -> Dict[str, Any]:207 """Get the default parameters for calling llama_cpp."""208 params = {209 "suffix": self.suffix,210 "max_tokens": self.max_tokens,211 "temperature": self.temperature,212 "top_p": self.top_p,213 "logprobs": self.logprobs,214 "echo": self.echo,215 "stop_sequences": self.stop, # key here is convention among LLM classes216 "repeat_penalty": self.repeat_penalty,217 "top_k": self.top_k,218 }219 if self.grammar:220 params["grammar"] = self.grammar221 return params222 223 @property224 def _identifying_params(self) -> Dict[str, Any]:225 """Get the identifying parameters."""226 return {**{"model_path": self.model_path}, **self._default_params}227 228 @property229 def _llm_type(self) -> str:230 """Return type of llm."""231 return "llamacpp"232 233 def _get_parameters(self, stop: Optional[List[str]] = None) -> Dict[str, Any]:234 """235 Performs sanity check, preparing parameters in format needed by llama_cpp.236 237 Args:238 stop (Optional[List[str]]): List of stop sequences for llama_cpp.239 240 Returns:241 Dictionary containing the combined parameters.242 """243 244 # Raise error if stop sequences are in both input and default params245 if self.stop and stop is not None:246 raise ValueError("`stop` found in both the input and default params.")247 248 params = self._default_params249 250 # llama_cpp expects the "stop" key not this, so we remove it:251 params.pop("stop_sequences")252 253 # then sets it as configured, or default to an empty list:254 params["stop"] = self.stop or stop or []255 256 return params257 258 def _call(259 self,260 prompt: str,261 stop: Optional[List[str]] = None,262 run_manager: Optional[CallbackManagerForLLMRun] = None,263 **kwargs: Any,264 ) -> str:265 """Call the Llama model and return the output.266 267 Args:268 prompt: The prompt to use for generation.269 stop: A list of strings to stop generation when encountered.270 271 Returns:272 The generated text.273 274 Example:275 .. code-block:: python276 277 from langchain_community.llms import LlamaCpp278 llm = LlamaCpp(model_path="/path/to/local/llama/model.bin")279 llm.invoke("This is a prompt.")280 """281 if self.streaming:282 # If streaming is enabled, we use the stream283 # method that yields as they are generated284 # and return the combined strings from the first choices's text:285 combined_text_output = ""286 for chunk in self._stream(287 prompt=prompt,288 stop=stop,289 run_manager=run_manager,290 **kwargs,291 ):292 combined_text_output += chunk.text293 return combined_text_output294 else:295 params = self._get_parameters(stop)296 params = {**params, **kwargs}297 result = self.client(prompt=prompt, **params)298 return result["choices"][0]["text"]299 300 def _stream(301 self,302 prompt: str,303 stop: Optional[List[str]] = None,304 run_manager: Optional[CallbackManagerForLLMRun] = None,305 **kwargs: Any,306 ) -> Iterator[GenerationChunk]:307 """Yields results objects as they are generated in real time.308 309 It also calls the callback manager's on_llm_new_token event with310 similar parameters to the OpenAI LLM class method of the same name.311 312 Args:313 prompt: The prompts to pass into the model.314 stop: Optional list of stop words to use when generating.315 316 Returns:317 A generator representing the stream of tokens being generated.318 319 Yields:320 A dictionary like objects containing a string token and metadata.321 See llama-cpp-python docs and below for more.322 323 Example:324 .. code-block:: python325 326 from langchain_community.llms import LlamaCpp327 llm = LlamaCpp(328 model_path="/path/to/local/model.bin",329 temperature = 0.5330 )331 for chunk in llm.stream("Ask 'Hi, how are you?' like a pirate:'",332 stop=["'","\n"]):333 result = chunk["choices"][0]334 print(result["text"], end='', flush=True) # noqa: T201335 336 """337 params = {**self._get_parameters(stop), **kwargs}338 result = self.client(prompt=prompt, stream=True, **params)339 for part in result:340 logprobs = part["choices"][0].get("logprobs", None)341 chunk = GenerationChunk(342 text=part["choices"][0]["text"],343 generation_info={"logprobs": logprobs},344 )345 if run_manager:346 run_manager.on_llm_new_token(347 token=chunk.text, verbose=self.verbose, log_probs=logprobs348 )349 yield chunk350 351 def get_num_tokens(self, text: str) -> int:352 tokenized_text = self.client.tokenize(text.encode("utf-8"))353 return len(tokenized_text)354 