Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
vertexai.py394 linesDownload Raw Back to chat_models
1"""Wrapper around Google VertexAI chat-based models."""2 3from __future__ import annotations4 5import base646import logging7import re8from dataclasses import dataclass, field9from typing import TYPE_CHECKING, Any, Dict, Iterator, List, Optional, Union, cast10from urllib.parse import urlparse11 12import requests13from langchain_core._api.deprecation import deprecated14from langchain_core.callbacks import (15    AsyncCallbackManagerForLLMRun,16    CallbackManagerForLLMRun,17)18from langchain_core.language_models.chat_models import (19    BaseChatModel,20    generate_from_stream,21)22from langchain_core.messages import (23    AIMessage,24    AIMessageChunk,25    BaseMessage,26    HumanMessage,27    SystemMessage,28)29from langchain_core.outputs import ChatGeneration, ChatGenerationChunk, ChatResult30from langchain_core.utils import pre_init31 32from langchain_community.llms.vertexai import (33    _VertexAICommon,34    is_codey_model,35    is_gemini_model,36)37from langchain_community.utilities.vertexai import (38    load_image_from_gcs,39    raise_vertex_import_error,40)41 42if TYPE_CHECKING:43    from vertexai.language_models import (44        ChatMessage,45        ChatSession,46        CodeChatSession,47        InputOutputTextPair,48    )49    from vertexai.preview.generative_models import Content50 51logger = logging.getLogger(__name__)52 53 54@dataclass55class _ChatHistory:56    """Represents a context and a history of messages."""57 58    history: List["ChatMessage"] = field(default_factory=list)59    context: Optional[str] = None60 61 62def _parse_chat_history(history: List[BaseMessage]) -> _ChatHistory:63    """Parse a sequence of messages into history.64 65    Args:66        history: The list of messages to re-create the history of the chat.67    Returns:68        A parsed chat history.69    Raises:70        ValueError: If a sequence of message has a SystemMessage not at the71        first place.72    """73    from vertexai.language_models import ChatMessage74 75    vertex_messages, context = [], None76    for i, message in enumerate(history):77        content = cast(str, message.content)78        if i == 0 and isinstance(message, SystemMessage):79            context = content80        elif isinstance(message, AIMessage):81            vertex_message = ChatMessage(content=message.content, author="bot")82            vertex_messages.append(vertex_message)83        elif isinstance(message, HumanMessage):84            vertex_message = ChatMessage(content=message.content, author="user")85            vertex_messages.append(vertex_message)86        else:87            raise ValueError(88                f"Unexpected message with type {type(message)} at the position {i}."89            )90    chat_history = _ChatHistory(context=context, history=vertex_messages)91    return chat_history92 93 94def _is_url(s: str) -> bool:95    try:96        result = urlparse(s)97        return all([result.scheme, result.netloc])98    except Exception as e:99        logger.debug(f"Unable to parse URL: {e}")100        return False101 102 103def _parse_chat_history_gemini(104    history: List[BaseMessage], project: Optional[str]105) -> List["Content"]:106    from vertexai.preview.generative_models import Content, Image, Part107 108    def _convert_to_prompt(part: Union[str, Dict]) -> Part:109        if isinstance(part, str):110            return Part.from_text(part)111 112        if not isinstance(part, Dict):113            raise ValueError(114                f"Message's content is expected to be a dict, got {type(part)}!"115            )116        if part["type"] == "text":117            return Part.from_text(part["text"])118        elif part["type"] == "image_url":119            path = part["image_url"]["url"]120            if path.startswith("gs://"):121                image = load_image_from_gcs(path=path, project=project)122            elif path.startswith("data:image/"):123                # extract base64 component from image uri124                encoded: Any = re.search(r"data:image/\w{2,4};base64,(.*)", path)125                if encoded:126                    encoded = encoded.group(1)127                else:128                    raise ValueError(129                        "Invalid image uri. It should be in the format "130                        "data:image/<image_type>;base64,<base64_encoded_image>."131                    )132                image = Image.from_bytes(base64.b64decode(encoded))133            elif _is_url(path):134                response = requests.get(path)135                response.raise_for_status()136                image = Image.from_bytes(response.content)137            else:138                image = Image.load_from_file(path)139        else:140            raise ValueError("Only text and image_url types are supported!")141        return Part.from_image(image)142 143    vertex_messages = []144    for i, message in enumerate(history):145        if i == 0 and isinstance(message, SystemMessage):146            raise ValueError("SystemMessages are not yet supported!")147        elif isinstance(message, AIMessage):148            role = "model"149        elif isinstance(message, HumanMessage):150            role = "user"151        else:152            raise ValueError(153                f"Unexpected message with type {type(message)} at the position {i}."154            )155 156        raw_content = message.content157        if isinstance(raw_content, str):158            raw_content = [raw_content]159        parts = [_convert_to_prompt(part) for part in raw_content]160        vertex_message = Content(role=role, parts=parts)161        vertex_messages.append(vertex_message)162    return vertex_messages163 164 165def _parse_examples(examples: List[BaseMessage]) -> List["InputOutputTextPair"]:166    from vertexai.language_models import InputOutputTextPair167 168    if len(examples) % 2 != 0:169        raise ValueError(170            f"Expect examples to have an even amount of messages, got {len(examples)}."171        )172    example_pairs = []173    input_text = None174    for i, example in enumerate(examples):175        if i % 2 == 0:176            if not isinstance(example, HumanMessage):177                raise ValueError(178                    f"Expected the first message in a part to be from human, got "179                    f"{type(example)} for the {i}th message."180                )181            input_text = example.content182        if i % 2 == 1:183            if not isinstance(example, AIMessage):184                raise ValueError(185                    f"Expected the second message in a part to be from AI, got "186                    f"{type(example)} for the {i}th message."187                )188            pair = InputOutputTextPair(189                input_text=input_text, output_text=example.content190            )191            example_pairs.append(pair)192    return example_pairs193 194 195def _get_question(messages: List[BaseMessage]) -> HumanMessage:196    """Get the human message at the end of a list of input messages to a chat model."""197    if not messages:198        raise ValueError("You should provide at least one message to start the chat!")199    question = messages[-1]200    if not isinstance(question, HumanMessage):201        raise ValueError(202            f"Last message in the list should be from human, got {question.type}."203        )204    return question205 206 207@deprecated(208    since="0.0.12",209    removal="1.0",210    alternative_import="langchain_google_vertexai.ChatVertexAI",211)212class ChatVertexAI(_VertexAICommon, BaseChatModel):213    """`Vertex AI` Chat large language models API."""214 215    model_name: str = "chat-bison"216    "Underlying model name."217    examples: Optional[List[BaseMessage]] = None218 219    @classmethod220    def is_lc_serializable(self) -> bool:221        return True222 223    @classmethod224    def get_lc_namespace(cls) -> List[str]:225        """Get the namespace of the langchain object."""226        return ["langchain", "chat_models", "vertexai"]227 228    @pre_init229    def validate_environment(cls, values: Dict) -> Dict:230        """Validate that the python package exists in environment."""231        is_gemini = is_gemini_model(values["model_name"])232        cls._try_init_vertexai(values)233        try:234            from vertexai.language_models import ChatModel, CodeChatModel235 236            if is_gemini:237                from vertexai.preview.generative_models import (238                    GenerativeModel,239                )240        except ImportError:241            raise_vertex_import_error()242        if is_gemini:243            values["client"] = GenerativeModel(model_name=values["model_name"])244        else:245            if is_codey_model(values["model_name"]):246                model_cls = CodeChatModel247            else:248                model_cls = ChatModel249            values["client"] = model_cls.from_pretrained(values["model_name"])250        return values251 252    def _generate(253        self,254        messages: List[BaseMessage],255        stop: Optional[List[str]] = None,256        run_manager: Optional[CallbackManagerForLLMRun] = None,257        stream: Optional[bool] = None,258        **kwargs: Any,259    ) -> ChatResult:260        """Generate next turn in the conversation.261 262        Args:263            messages: The history of the conversation as a list of messages. Code chat264                does not support context.265            stop: The list of stop words (optional).266            run_manager: The CallbackManager for LLM run, it's not used at the moment.267            stream: Whether to use the streaming endpoint.268 269        Returns:270            The ChatResult that contains outputs generated by the model.271 272        Raises:273            ValueError: if the last message in the list is not from human.274        """275        should_stream = stream if stream is not None else self.streaming276        if should_stream:277            stream_iter = self._stream(278                messages, stop=stop, run_manager=run_manager, **kwargs279            )280            return generate_from_stream(stream_iter)281 282        question = _get_question(messages)283        params = self._prepare_params(stop=stop, stream=False, **kwargs)284        msg_params = {}285        if "candidate_count" in params:286            msg_params["candidate_count"] = params.pop("candidate_count")287 288        if self._is_gemini_model:289            history_gemini = _parse_chat_history_gemini(messages, project=self.project)290            message = history_gemini.pop()291            chat = self.client.start_chat(history=history_gemini)292            response = chat.send_message(message, generation_config=params)293        else:294            history = _parse_chat_history(messages[:-1])295            examples = kwargs.get("examples") or self.examples296            if examples:297                params["examples"] = _parse_examples(examples)298            chat = self._start_chat(history, **params)299            response = chat.send_message(question.content, **msg_params)300        generations = [301            ChatGeneration(message=AIMessage(content=r.text))302            for r in response.candidates303        ]304        return ChatResult(generations=generations)305 306    async def _agenerate(307        self,308        messages: List[BaseMessage],309        stop: Optional[List[str]] = None,310        run_manager: Optional[AsyncCallbackManagerForLLMRun] = None,311        **kwargs: Any,312    ) -> ChatResult:313        """Asynchronously generate next turn in the conversation.314 315        Args:316            messages: The history of the conversation as a list of messages. Code chat317                does not support context.318            stop: The list of stop words (optional).319            run_manager: The CallbackManager for LLM run, it's not used at the moment.320 321        Returns:322            The ChatResult that contains outputs generated by the model.323 324        Raises:325            ValueError: if the last message in the list is not from human.326        """327        if "stream" in kwargs:328            kwargs.pop("stream")329            logger.warning("ChatVertexAI does not currently support async streaming.")330 331        params = self._prepare_params(stop=stop, **kwargs)332        msg_params = {}333        if "candidate_count" in params:334            msg_params["candidate_count"] = params.pop("candidate_count")335 336        if self._is_gemini_model:337            history_gemini = _parse_chat_history_gemini(messages, project=self.project)338            message = history_gemini.pop()339            chat = self.client.start_chat(history=history_gemini)340            response = await chat.send_message_async(message, generation_config=params)341        else:342            question = _get_question(messages)343            history = _parse_chat_history(messages[:-1])344            examples = kwargs.get("examples", None)345            if examples:346                params["examples"] = _parse_examples(examples)347            chat = self._start_chat(history, **params)348            response = await chat.send_message_async(question.content, **msg_params)349 350        generations = [351            ChatGeneration(message=AIMessage(content=r.text))352            for r in response.candidates353        ]354        return ChatResult(generations=generations)355 356    def _stream(357        self,358        messages: List[BaseMessage],359        stop: Optional[List[str]] = None,360        run_manager: Optional[CallbackManagerForLLMRun] = None,361        **kwargs: Any,362    ) -> Iterator[ChatGenerationChunk]:363        params = self._prepare_params(stop=stop, stream=True, **kwargs)364        if self._is_gemini_model:365            history_gemini = _parse_chat_history_gemini(messages, project=self.project)366            message = history_gemini.pop()367            chat = self.client.start_chat(history=history_gemini)368            responses = chat.send_message(369                message, stream=True, generation_config=params370            )371        else:372            question = _get_question(messages)373            history = _parse_chat_history(messages[:-1])374            examples = kwargs.get("examples", None)375            if examples:376                params["examples"] = _parse_examples(examples)377            chat = self._start_chat(history, **params)378            responses = chat.send_message_streaming(question.content, **params)379        for response in responses:380            chunk = ChatGenerationChunk(message=AIMessageChunk(content=response.text))381            if run_manager:382                run_manager.on_llm_new_token(response.text, chunk=chunk)383            yield chunk384 385    def _start_chat(386        self, history: _ChatHistory, **kwargs: Any387    ) -> Union[ChatSession, CodeChatSession]:388        if not self.is_codey_model:389            return self.client.start_chat(390                context=history.context, message_history=history.history, **kwargs391            )392        else:393            return self.client.start_chat(message_history=history.history, **kwargs)394 
codekingpro/portable-devtools · Team Ai