BarinkDev/LargeLanguageModels
0
1{2 "cells": [3 {4 "cell_type": "markdown",5 "id": "6eee0106-2fee-4263-855f-2c362ce159d1",6 "metadata": {},7 "source": [8 "# New model from google called Gemma "9 ]10 },11 {12 "cell_type": "markdown",13 "id": "37731f28-7b4b-4229-9270-c0a547c13030",14 "metadata": {},15 "source": [16 "Source: [https://huggingface.co/google/gemma-7b](https://huggingface.co/google/gemma-7b)"17 ]18 },19 {20 "cell_type": "code",21 "execution_count": null,22 "id": "4a9c70eb-6f88-4c82-ba14-eb358a62a198",23 "metadata": {},24 "outputs": [],25 "source": [26 "# pip install bitsandbytes accelerate\n",27 "%pip install bitsandbytes accelerate \n"28 ]29 },30 {31 "cell_type": "code",32 "execution_count": null,33 "id": "3f0c0b9f-c457-4d05-b349-d76b0b65dc10",34 "metadata": {},35 "outputs": [36 {37 "data": {38 "application/vnd.jupyter.widget-view+json": {39 "model_id": "d1559a58ffff43609eaa5074b0178cd9",40 "version_major": 2,41 "version_minor": 042 },43 "text/plain": [44 "Loading checkpoint shards: 0%| | 0/4 [00:00<?, ?it/s]"45 ]46 },47 "metadata": {},48 "output_type": "display_data"49 },50 {51 "name": "stderr",52 "output_type": "stream",53 "text": [54 "WARNING:root:Some parameters are on the meta device device because they were offloaded to the disk and cpu.\n",55 "2024-02-21 23:34:48.779292: E external/local_xla/xla/stream_executor/cuda/cuda_dnn.cc:9261] Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered\n",56 "2024-02-21 23:34:48.781146: E external/local_xla/xla/stream_executor/cuda/cuda_fft.cc:607] Unable to register cuFFT factory: Attempting to register factory for plugin cuFFT when one has already been registered\n",57 "2024-02-21 23:34:49.052470: E external/local_xla/xla/stream_executor/cuda/cuda_blas.cc:1515] Unable to register cuBLAS factory: Attempting to register factory for plugin cuBLAS when one has already been registered\n",58 "2024-02-21 23:34:58.007900: W tensorflow/compiler/tf2tensorrt/utils/py_utils.cc:38] TF-TRT Warning: Could not find TensorRT\n",59 "\n",60 "KeyboardInterrupt\n",61 "\n"62 ]63 }64 ],65 "source": [66 "import os \n",67 "from transformers import AutoTokenizer, AutoModelForCausalLM, AutoConfig\n",68 "os.environ['TRANSFORMERS_CACHE'] = '/mnt/f/wkspc-linux/.cache'\n",69 "os.environ['HUGGINGFACE_HUB_CACHE'] = '/mnt/f/wkspc-linux/.cache'\n",70 "os.environ['HF_HOME'] = '/mnt/f/wkspc-linux/.cache'\n",71 "\n",72 "tokenizer = AutoTokenizer.from_pretrained(\"google/gemma-7b\", cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",73 "model = AutoModelForCausalLM.from_pretrained(\"google/gemma-7b\",device_map=\"auto\", cache_dir=\"/mnt/f/wkspc-linux/.cache\" )\n",74 "\n",75 "\n",76 "input_text = \"Write me a poem about Machine Learning.\"\n",77 "input_ids = tokenizer(input_text, return_tensors=\"pt\").to(\"cuda\")\n",78 "\n",79 "outputs = model.generate(**input_ids, max_new_tokens=200)\n",80 "print(tokenizer.decode(outputs[0]))\n"81 ]82 },83 {84 "cell_type": "code",85 "execution_count": 1,86 "id": "c5eeae00-6f0e-4686-a3b8-3ee3538abb87",87 "metadata": {},88 "outputs": [89 {90 "data": {91 "application/vnd.jupyter.widget-view+json": {92 "model_id": "8dfc17e70a904d88b7fa87d23ec44cfc",93 "version_major": 2,94 "version_minor": 095 },96 "text/plain": [97 "tokenizer.model: 0%| | 0.00/4.24M [00:00<?, ?B/s]"98 ]99 },100 "metadata": {},101 "output_type": "display_data"102 },103 {104 "name": "stderr",105 "output_type": "stream",106 "text": [107 "`low_cpu_mem_usage` was None, now set to True since model is quantized.\n"108 ]109 },110 {111 "data": {112 "application/vnd.jupyter.widget-view+json": {113 "model_id": "af8ce3d6f3444aaf8bb0d8968ccd9fec",114 "version_major": 2,115 "version_minor": 0116 },117 "text/plain": [118 "Downloading shards: 0%| | 0/4 [00:00<?, ?it/s]"119 ]120 },121 "metadata": {},122 "output_type": "display_data"123 },124 {125 "data": {126 "application/vnd.jupyter.widget-view+json": {127 "model_id": "ab7543ed101e45aeb18df74325a073e2",128 "version_major": 2,129 "version_minor": 0130 },131 "text/plain": [132 "Loading checkpoint shards: 0%| | 0/4 [00:00<?, ?it/s]"133 ]134 },135 "metadata": {},136 "output_type": "display_data"137 },138 {139 "name": "stderr",140 "output_type": "stream",141 "text": [142 "2024-02-23 15:55:39.248214: E external/local_xla/xla/stream_executor/cuda/cuda_dnn.cc:9261] Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered\n",143 "2024-02-23 15:55:39.251426: E external/local_xla/xla/stream_executor/cuda/cuda_fft.cc:607] Unable to register cuFFT factory: Attempting to register factory for plugin cuFFT when one has already been registered\n",144 "2024-02-23 15:55:39.992353: E external/local_xla/xla/stream_executor/cuda/cuda_blas.cc:1515] Unable to register cuBLAS factory: Attempting to register factory for plugin cuBLAS when one has already been registered\n",145 "2024-02-23 15:56:03.288614: W tensorflow/compiler/tf2tensorrt/utils/py_utils.cc:38] TF-TRT Warning: Could not find TensorRT\n"146 ]147 },148 {149 "name": "stdout",150 "output_type": "stream",151 "text": [152 "<bos>Write me a poem about Machine Learning.\n",153 "\n",154 "I’m not a poet, but I’m a Machine Learning Engineer.\n",155 "\n",156 "I’m not a poet, but I’m a\n"157 ]158 }159 ],160 "source": [161 "\n",162 "from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig\n",163 "\n",164 "quantization_config = BitsAndBytesConfig(load_in_8bit=True)\n",165 "\n",166 "tokenizer = AutoTokenizer.from_pretrained(\"google/gemma-7b\", cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",167 "model = AutoModelForCausalLM.from_pretrained(\"google/gemma-7b\", quantization_config=quantization_config, cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",168 "\n",169 "input_text = \"Write me a poem about Machine Learning.\"\n",170 "input_ids = tokenizer(input_text, return_tensors=\"pt\").to(\"cuda\")\n",171 "\n",172 "outputs = model.generate(**input_ids, max_new_tokens=30)\n",173 "print(tokenizer.decode(outputs[0]))\n"174 ]175 },176 {177 "cell_type": "code",178 "execution_count": 2,179 "id": "1b0c0f09-a1d4-4b9f-baa1-914ead919b06",180 "metadata": {},181 "outputs": [],182 "source": [183 "def prompt(input_text):\n",184 " input_ids = tokenizer(input_text, return_tensors=\"pt\").to(\"cuda\")\n",185 " \n",186 " outputs = model.generate(**input_ids, max_new_tokens=30)\n",187 " return tokenizer.decode(outputs[0])"188 ]189 },190 {191 "cell_type": "code",192 "execution_count": null,193 "id": "3d2e9452-72e3-4467-bc75-cdcaa1cd5693",194 "metadata": {},195 "outputs": [],196 "source": [197 "print(prompt(\"What is a good framework for python webscraping?\"))"198 ]199 },200 {201 "cell_type": "code",202 "execution_count": null,203 "id": "42093c90-7c93-4f3d-a109-9b298688ddea",204 "metadata": {},205 "outputs": [],206 "source": []207 }208 ],209 "metadata": {210 "kernelspec": {211 "display_name": "Python 3 (ipykernel)",212 "language": "python",213 "name": "python3"214 },215 "language_info": {216 "codemirror_mode": {217 "name": "ipython",218 "version": 3219 },220 "file_extension": ".py",221 "mimetype": "text/x-python",222 "name": "python",223 "nbconvert_exporter": "python",224 "pygments_lexer": "ipython3",225 "version": "3.11.2"226 }227 },228 "nbformat": 4,229 "nbformat_minor": 5230}231 