codekingpro/portable-devtools
114k
1# flake8: noqa2from langchain_core.utils import pre_init3from typing import Any, AsyncIterator, Dict, Iterator, List, Optional, Union4from langchain_core.utils import pre_init5from pydantic import root_validator6from langchain_core.utils import pre_init7from langchain_core.callbacks import (8 AsyncCallbackManagerForLLMRun,9 CallbackManagerForLLMRun,10)11from langchain_core.utils import pre_init12from langchain_core.language_models.llms import LLM13from langchain_core.utils import pre_init14from langchain_community.llms.utils import enforce_stop_tokens15from langchain_core.utils import pre_init16from langchain_core.outputs import GenerationChunk17 18 19class DeepSparse(LLM):20 """Neural Magic DeepSparse LLM interface.21 To use, you should have the ``deepsparse`` or ``deepsparse-nightly``22 python package installed. See https://github.com/neuralmagic/deepsparse23 This interface let's you deploy optimized LLMs straight from the24 [SparseZoo](https://sparsezoo.neuralmagic.com/?useCase=text_generation)25 Example:26 .. code-block:: python27 from langchain_community.llms import DeepSparse28 llm = DeepSparse(model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none")29 """ # noqa: E50130 31 pipeline: Any #: :meta private:32 33 model: str34 """The path to a model file or directory or the name of a SparseZoo model stub."""35 36 model_configuration: Optional[Dict[str, Any]] = None37 """Keyword arguments passed to the pipeline construction.38 Common parameters are sequence_length, prompt_sequence_length"""39 40 generation_config: Union[None, str, Dict] = None41 """GenerationConfig dictionary consisting of parameters used to control42 sequences generated for each prompt. Common parameters are:43 max_length, max_new_tokens, num_return_sequences, output_scores,44 top_p, top_k, repetition_penalty."""45 46 streaming: bool = False47 """Whether to stream the results, token by token."""48 49 @property50 def _identifying_params(self) -> Dict[str, Any]:51 """Get the identifying parameters."""52 return {53 "model": self.model,54 "model_config": self.model_configuration,55 "generation_config": self.generation_config,56 "streaming": self.streaming,57 }58 59 @property60 def _llm_type(self) -> str:61 """Return type of llm."""62 return "deepsparse"63 64 @pre_init65 def validate_environment(cls, values: Dict) -> Dict:66 """Validate that ``deepsparse`` package is installed."""67 try:68 from deepsparse import Pipeline69 except ImportError:70 raise ImportError(71 "Could not import `deepsparse` package. "72 "Please install it with `pip install deepsparse[llm]`"73 )74 75 model_config = values["model_configuration"] or {}76 77 values["pipeline"] = Pipeline.create(78 task="text_generation",79 model_path=values["model"],80 **model_config,81 )82 return values83 84 def _call(85 self,86 prompt: str,87 stop: Optional[List[str]] = None,88 run_manager: Optional[CallbackManagerForLLMRun] = None,89 **kwargs: Any,90 ) -> str:91 """Generate text from a prompt.92 Args:93 prompt: The prompt to generate text from.94 stop: A list of strings to stop generation when encountered.95 Returns:96 The generated text.97 Example:98 .. code-block:: python99 from langchain_community.llms import DeepSparse100 llm = DeepSparse(model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none")101 llm.invoke("Tell me a joke.")102 """103 if self.streaming:104 combined_output = ""105 for chunk in self._stream(106 prompt=prompt, stop=stop, run_manager=run_manager, **kwargs107 ):108 combined_output += chunk.text109 text = combined_output110 else:111 text = (112 self.pipeline(sequences=prompt, **self.generation_config)113 .generations[0]114 .text115 )116 117 if stop is not None:118 text = enforce_stop_tokens(text, stop)119 120 return text121 122 async def _acall(123 self,124 prompt: str,125 stop: Optional[List[str]] = None,126 run_manager: Optional[AsyncCallbackManagerForLLMRun] = None,127 **kwargs: Any,128 ) -> str:129 """Generate text from a prompt.130 Args:131 prompt: The prompt to generate text from.132 stop: A list of strings to stop generation when encountered.133 Returns:134 The generated text.135 Example:136 .. code-block:: python137 from langchain_community.llms import DeepSparse138 llm = DeepSparse(model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none")139 llm.invoke("Tell me a joke.")140 """141 if self.streaming:142 combined_output = ""143 async for chunk in self._astream(144 prompt=prompt, stop=stop, run_manager=run_manager, **kwargs145 ):146 combined_output += chunk.text147 text = combined_output148 else:149 text = (150 self.pipeline(sequences=prompt, **self.generation_config)151 .generations[0]152 .text153 )154 155 if stop is not None:156 text = enforce_stop_tokens(text, stop)157 158 return text159 160 def _stream(161 self,162 prompt: str,163 stop: Optional[List[str]] = None,164 run_manager: Optional[CallbackManagerForLLMRun] = None,165 **kwargs: Any,166 ) -> Iterator[GenerationChunk]:167 """Yields results objects as they are generated in real time.168 It also calls the callback manager's on_llm_new_token event with169 similar parameters to the OpenAI LLM class method of the same name.170 Args:171 prompt: The prompt to pass into the model.172 stop: Optional list of stop words to use when generating.173 Returns:174 A generator representing the stream of tokens being generated.175 Yields:176 A dictionary like object containing a string token.177 Example:178 .. code-block:: python179 from langchain_community.llms import DeepSparse180 llm = DeepSparse(181 model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none",182 streaming=True183 )184 for chunk in llm.stream("Tell me a joke",185 stop=["'","\n"]):186 print(chunk, end='', flush=True) # noqa: T201187 """188 inference = self.pipeline(189 sequences=prompt, streaming=True, **self.generation_config190 )191 for token in inference:192 chunk = GenerationChunk(text=token.generations[0].text)193 194 if run_manager:195 run_manager.on_llm_new_token(token=chunk.text)196 yield chunk197 198 async def _astream(199 self,200 prompt: str,201 stop: Optional[List[str]] = None,202 run_manager: Optional[AsyncCallbackManagerForLLMRun] = None,203 **kwargs: Any,204 ) -> AsyncIterator[GenerationChunk]:205 """Yields results objects as they are generated in real time.206 It also calls the callback manager's on_llm_new_token event with207 similar parameters to the OpenAI LLM class method of the same name.208 Args:209 prompt: The prompt to pass into the model.210 stop: Optional list of stop words to use when generating.211 Returns:212 A generator representing the stream of tokens being generated.213 Yields:214 A dictionary like object containing a string token.215 Example:216 .. code-block:: python217 from langchain_community.llms import DeepSparse218 llm = DeepSparse(219 model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none",220 streaming=True221 )222 for chunk in llm.stream("Tell me a joke",223 stop=["'","\n"]):224 print(chunk, end='', flush=True) # noqa: T201225 """226 inference = self.pipeline(227 sequences=prompt, streaming=True, **self.generation_config228 )229 for token in inference:230 chunk = GenerationChunk(text=token.generations[0].text)231 232 if run_manager:233 await run_manager.on_llm_new_token(token=chunk.text)234 yield chunk235 