Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
deepsparse.py235 linesDownload Raw Back to llms
1# flake8: noqa2from langchain_core.utils import pre_init3from typing import Any, AsyncIterator, Dict, Iterator, List, Optional, Union4from langchain_core.utils import pre_init5from pydantic import root_validator6from langchain_core.utils import pre_init7from langchain_core.callbacks import (8    AsyncCallbackManagerForLLMRun,9    CallbackManagerForLLMRun,10)11from langchain_core.utils import pre_init12from langchain_core.language_models.llms import LLM13from langchain_core.utils import pre_init14from langchain_community.llms.utils import enforce_stop_tokens15from langchain_core.utils import pre_init16from langchain_core.outputs import GenerationChunk17 18 19class DeepSparse(LLM):20    """Neural Magic DeepSparse LLM interface.21    To use, you should have the ``deepsparse`` or ``deepsparse-nightly``22    python package installed. See https://github.com/neuralmagic/deepsparse23    This interface let's you deploy optimized LLMs straight from the24    [SparseZoo](https://sparsezoo.neuralmagic.com/?useCase=text_generation)25    Example:26        .. code-block:: python27            from langchain_community.llms import DeepSparse28            llm = DeepSparse(model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none")29    """  # noqa: E50130 31    pipeline: Any  #: :meta private:32 33    model: str34    """The path to a model file or directory or the name of a SparseZoo model stub."""35 36    model_configuration: Optional[Dict[str, Any]] = None37    """Keyword arguments passed to the pipeline construction.38    Common parameters are sequence_length, prompt_sequence_length"""39 40    generation_config: Union[None, str, Dict] = None41    """GenerationConfig dictionary consisting of parameters used to control42    sequences generated for each prompt. Common parameters are:43    max_length, max_new_tokens, num_return_sequences, output_scores,44    top_p, top_k, repetition_penalty."""45 46    streaming: bool = False47    """Whether to stream the results, token by token."""48 49    @property50    def _identifying_params(self) -> Dict[str, Any]:51        """Get the identifying parameters."""52        return {53            "model": self.model,54            "model_config": self.model_configuration,55            "generation_config": self.generation_config,56            "streaming": self.streaming,57        }58 59    @property60    def _llm_type(self) -> str:61        """Return type of llm."""62        return "deepsparse"63 64    @pre_init65    def validate_environment(cls, values: Dict) -> Dict:66        """Validate that ``deepsparse`` package is installed."""67        try:68            from deepsparse import Pipeline69        except ImportError:70            raise ImportError(71                "Could not import `deepsparse` package. "72                "Please install it with `pip install deepsparse[llm]`"73            )74 75        model_config = values["model_configuration"] or {}76 77        values["pipeline"] = Pipeline.create(78            task="text_generation",79            model_path=values["model"],80            **model_config,81        )82        return values83 84    def _call(85        self,86        prompt: str,87        stop: Optional[List[str]] = None,88        run_manager: Optional[CallbackManagerForLLMRun] = None,89        **kwargs: Any,90    ) -> str:91        """Generate text from a prompt.92        Args:93            prompt: The prompt to generate text from.94            stop: A list of strings to stop generation when encountered.95        Returns:96            The generated text.97        Example:98            .. code-block:: python99                from langchain_community.llms import DeepSparse100                llm = DeepSparse(model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none")101                llm.invoke("Tell me a joke.")102        """103        if self.streaming:104            combined_output = ""105            for chunk in self._stream(106                prompt=prompt, stop=stop, run_manager=run_manager, **kwargs107            ):108                combined_output += chunk.text109            text = combined_output110        else:111            text = (112                self.pipeline(sequences=prompt, **self.generation_config)113                .generations[0]114                .text115            )116 117        if stop is not None:118            text = enforce_stop_tokens(text, stop)119 120        return text121 122    async def _acall(123        self,124        prompt: str,125        stop: Optional[List[str]] = None,126        run_manager: Optional[AsyncCallbackManagerForLLMRun] = None,127        **kwargs: Any,128    ) -> str:129        """Generate text from a prompt.130        Args:131            prompt: The prompt to generate text from.132            stop: A list of strings to stop generation when encountered.133        Returns:134            The generated text.135        Example:136            .. code-block:: python137                from langchain_community.llms import DeepSparse138                llm = DeepSparse(model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none")139                llm.invoke("Tell me a joke.")140        """141        if self.streaming:142            combined_output = ""143            async for chunk in self._astream(144                prompt=prompt, stop=stop, run_manager=run_manager, **kwargs145            ):146                combined_output += chunk.text147            text = combined_output148        else:149            text = (150                self.pipeline(sequences=prompt, **self.generation_config)151                .generations[0]152                .text153            )154 155        if stop is not None:156            text = enforce_stop_tokens(text, stop)157 158        return text159 160    def _stream(161        self,162        prompt: str,163        stop: Optional[List[str]] = None,164        run_manager: Optional[CallbackManagerForLLMRun] = None,165        **kwargs: Any,166    ) -> Iterator[GenerationChunk]:167        """Yields results objects as they are generated in real time.168        It also calls the callback manager's on_llm_new_token event with169        similar parameters to the OpenAI LLM class method of the same name.170        Args:171            prompt: The prompt to pass into the model.172            stop: Optional list of stop words to use when generating.173        Returns:174            A generator representing the stream of tokens being generated.175        Yields:176            A dictionary like object containing a string token.177        Example:178            .. code-block:: python179                from langchain_community.llms import DeepSparse180                llm = DeepSparse(181                    model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none",182                    streaming=True183                )184                for chunk in llm.stream("Tell me a joke",185                        stop=["'","\n"]):186                    print(chunk, end='', flush=True)  # noqa: T201187        """188        inference = self.pipeline(189            sequences=prompt, streaming=True, **self.generation_config190        )191        for token in inference:192            chunk = GenerationChunk(text=token.generations[0].text)193 194            if run_manager:195                run_manager.on_llm_new_token(token=chunk.text)196            yield chunk197 198    async def _astream(199        self,200        prompt: str,201        stop: Optional[List[str]] = None,202        run_manager: Optional[AsyncCallbackManagerForLLMRun] = None,203        **kwargs: Any,204    ) -> AsyncIterator[GenerationChunk]:205        """Yields results objects as they are generated in real time.206        It also calls the callback manager's on_llm_new_token event with207        similar parameters to the OpenAI LLM class method of the same name.208        Args:209            prompt: The prompt to pass into the model.210            stop: Optional list of stop words to use when generating.211        Returns:212            A generator representing the stream of tokens being generated.213        Yields:214            A dictionary like object containing a string token.215        Example:216            .. code-block:: python217                from langchain_community.llms import DeepSparse218                llm = DeepSparse(219                    model="zoo:nlg/text_generation/codegen_mono-350m/pytorch/huggingface/bigpython_bigquery_thepile/base_quant-none",220                    streaming=True221                )222                for chunk in llm.stream("Tell me a joke",223                        stop=["'","\n"]):224                    print(chunk, end='', flush=True)  # noqa: T201225        """226        inference = self.pipeline(227            sequences=prompt, streaming=True, **self.generation_config228        )229        for token in inference:230            chunk = GenerationChunk(text=token.generations[0].text)231 232            if run_manager:233                await run_manager.on_llm_new_token(token=chunk.text)234            yield chunk235 
codekingpro/portable-devtools · Team Ai