BarinkDev/LargeLanguageModels
0
1{2 "cells": [3 {4 "cell_type": "markdown",5 "id": "c5b7cb68-6899-449a-91a0-ea40cfd00b12",6 "metadata": {},7 "source": [8 "# Phi v1.0 through v2.0\n",9 "\n",10 "__source:__ [Phi-2: The surprising power of small language models](https://www.microsoft.com/en-us/research/blog/phi-2-the-surprising-power-of-small-language-models/)"11 ]12 },13 {14 "cell_type": "code",15 "execution_count": 1,16 "id": "f7414f76-8c18-4102-8eed-6830e45df8bd",17 "metadata": {},18 "outputs": [19 {20 "name": "stderr",21 "output_type": "stream",22 "text": [23 "/mnt/f/wkspc-linux/tf-gpu/.venv/lib/python3.9/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html\n",24 " from .autonotebook import tqdm as notebook_tqdm\n",25 "generation_config.json: 100%|████████████████████████████████████████████████████████| 69.0/69.0 [00:00<00:00, 49.4kB/s]\n",26 "tokenizer_config.json: 100%|████████████████████████████████████████████████████████████| 237/237 [00:00<00:00, 232kB/s]\n",27 "vocab.json: 100%|████████████████████████████████████████████████████████████████████| 798k/798k [00:00<00:00, 2.26MB/s]\n",28 "merges.txt: 100%|████████████████████████████████████████████████████████████████████| 456k/456k [00:00<00:00, 1.76MB/s]\n",29 "tokenizer.json: 100%|██████████████████████████████████████████████████████████████| 2.11M/2.11M [00:00<00:00, 4.14MB/s]\n",30 "added_tokens.json: 100%|████████████████████████████████████████████████████████████| 1.08k/1.08k [00:00<00:00, 711kB/s]\n",31 "special_tokens_map.json: 100%|███████████████████████████████████████████████████████| 99.0/99.0 [00:00<00:00, 76.5kB/s]\n"32 ]33 },34 {35 "name": "stdout",36 "output_type": "stream",37 "text": [38 "def print_prime(n):\n",39 "\"\"\"\n",40 "Print all primes between 1 and n \n",41 "\"\"\"\n",42 " for num in range(2, n+1):\n",43 " for i in range(2, num):\n",44 " if num % i == 0:\n",45 " break\n",46 " else:\n",47 " print(num)\n",48 "\n",49 "<|endoftext|>\n",50 "\n",51 "from typing import List\n",52 "\n",53 "def find_smallest_multiple_of_list(li: List[int]) -> int:\n",54 " \"\"\"\n",55 " Returns the smallest positive integer that is divisible by all the numbers in the input list.\n",56 "\n",57 " Args:\n",58 " li (List[int]): A list of integers.\n",59 "\n",60 " Returns:\n",61 " int: The smallest positive integer that is divisible by all the numbers in the input list.\n",62 " \"\"\"\n",63 "\n",64 " # Find the maximum number in the list\n",65 " max_num = max(li)\n",66 "\n",67 " # Initialize the result to the maximum number\n",68 " \n"69 ]70 }71 ],72 "source": [73 "# sample code Phi v1.0 as given on hugging face \n",74 "import torch\n",75 "import os\n",76 "from transformers import AutoModelForCausalLM, AutoTokenizer\n",77 "\n",78 "os.environ['TRANSFORMERS_CACHE'] = '/mnt/f/wkspc-linux/.cache'\n",79 "os.environ['HUGGINGFACE_HUB_CACHE'] = '/mnt/f/wkspc-linux/.cache'\n",80 "os.environ['HF_HOME'] = '/mnt/f/wkspc-linux/.cache'\n",81 "\n",82 "model_string = \"microsoft/phi-1\"\n",83 "\n",84 "torch.set_default_device(\"cuda\")\n",85 "model = AutoModelForCausalLM.from_pretrained( model_string, trust_remote_code=True , cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",86 "tokenizer = AutoTokenizer.from_pretrained(model_string, trust_remote_code=True, cache_dir= \"/mnt/f/wkspc-linux/.cache\")\n",87 "inputs = tokenizer('''def print_prime(n):\n",88 "\"\"\"\n",89 "Print all primes between 1 and n \n",90 "\"\"\"''', return_tensors=\"pt\", return_attention_mask=False)\n",91 "\n",92 "outputs = model.generate(**inputs, max_length=200)\n",93 "text = tokenizer.batch_decode(outputs)[0]\n",94 "print(text)"95 ]96 },97 {98 "cell_type": "code",99 "execution_count": 1,100 "id": "4c5bbaff-3c15-444b-ae25-3952472b64c4",101 "metadata": {},102 "outputs": [103 {104 "data": {105 "application/vnd.jupyter.widget-view+json": {106 "model_id": "546a2db4792c4136b9aa5e22a0824fd6",107 "version_major": 2,108 "version_minor": 0109 },110 "text/plain": [111 "config.json: 0%| | 0.00/732 [00:00<?, ?B/s]"112 ]113 },114 "metadata": {},115 "output_type": "display_data"116 },117 {118 "data": {119 "application/vnd.jupyter.widget-view+json": {120 "model_id": "ad780adae7cb4a02acd272d7ad7a922d",121 "version_major": 2,122 "version_minor": 0123 },124 "text/plain": [125 "configuration_phi.py: 0%| | 0.00/2.03k [00:00<?, ?B/s]"126 ]127 },128 "metadata": {},129 "output_type": "display_data"130 },131 {132 "name": "stderr",133 "output_type": "stream",134 "text": [135 "A new version of the following files was downloaded from https://huggingface.co/microsoft/phi-1_5:\n",136 "- configuration_phi.py\n",137 ". Make sure to double-check they do not contain any added malicious code. To avoid downloading new versions of the code file, you can pin a revision.\n"138 ]139 },140 {141 "data": {142 "application/vnd.jupyter.widget-view+json": {143 "model_id": "73e72075e4f343dcbb3dbf660081fe42",144 "version_major": 2,145 "version_minor": 0146 },147 "text/plain": [148 "modeling_phi.py: 0%| | 0.00/33.5k [00:00<?, ?B/s]"149 ]150 },151 "metadata": {},152 "output_type": "display_data"153 },154 {155 "name": "stderr",156 "output_type": "stream",157 "text": [158 "A new version of the following files was downloaded from https://huggingface.co/microsoft/phi-1_5:\n",159 "- modeling_phi.py\n",160 ". Make sure to double-check they do not contain any added malicious code. To avoid downloading new versions of the code file, you can pin a revision.\n"161 ]162 },163 {164 "data": {165 "application/vnd.jupyter.widget-view+json": {166 "model_id": "9a35d5c49230493ab5c543493c7b15f8",167 "version_major": 2,168 "version_minor": 0169 },170 "text/plain": [171 "pytorch_model.bin: 0%| | 0.00/2.84G [00:00<?, ?B/s]"172 ]173 },174 "metadata": {},175 "output_type": "display_data"176 },177 {178 "data": {179 "application/vnd.jupyter.widget-view+json": {180 "model_id": "e82761f31ada453e9c64b8c310e07ef6",181 "version_major": 2,182 "version_minor": 0183 },184 "text/plain": [185 "generation_config.json: 0%| | 0.00/69.0 [00:00<?, ?B/s]"186 ]187 },188 "metadata": {},189 "output_type": "display_data"190 },191 {192 "data": {193 "application/vnd.jupyter.widget-view+json": {194 "model_id": "5731a1eb81954874a92682ce47356012",195 "version_major": 2,196 "version_minor": 0197 },198 "text/plain": [199 "tokenizer_config.json: 0%| | 0.00/237 [00:00<?, ?B/s]"200 ]201 },202 "metadata": {},203 "output_type": "display_data"204 },205 {206 "data": {207 "application/vnd.jupyter.widget-view+json": {208 "model_id": "c27dea7c8e974fcdb39713451f6f84b7",209 "version_major": 2,210 "version_minor": 0211 },212 "text/plain": [213 "vocab.json: 0%| | 0.00/798k [00:00<?, ?B/s]"214 ]215 },216 "metadata": {},217 "output_type": "display_data"218 },219 {220 "data": {221 "application/vnd.jupyter.widget-view+json": {222 "model_id": "0561c67c1aa74f0fb19da51064afbda5",223 "version_major": 2,224 "version_minor": 0225 },226 "text/plain": [227 "merges.txt: 0%| | 0.00/456k [00:00<?, ?B/s]"228 ]229 },230 "metadata": {},231 "output_type": "display_data"232 },233 {234 "data": {235 "application/vnd.jupyter.widget-view+json": {236 "model_id": "b7ee279002574547bd8fd5a63949c0e4",237 "version_major": 2,238 "version_minor": 0239 },240 "text/plain": [241 "tokenizer.json: 0%| | 0.00/2.11M [00:00<?, ?B/s]"242 ]243 },244 "metadata": {},245 "output_type": "display_data"246 },247 {248 "data": {249 "application/vnd.jupyter.widget-view+json": {250 "model_id": "244b5a312dfb4f0dac6705ef8e00a536",251 "version_major": 2,252 "version_minor": 0253 },254 "text/plain": [255 "added_tokens.json: 0%| | 0.00/1.08k [00:00<?, ?B/s]"256 ]257 },258 "metadata": {},259 "output_type": "display_data"260 },261 {262 "data": {263 "application/vnd.jupyter.widget-view+json": {264 "model_id": "4ad41c64c296461091f1d337a804f1c5",265 "version_major": 2,266 "version_minor": 0267 },268 "text/plain": [269 "special_tokens_map.json: 0%| | 0.00/99.0 [00:00<?, ?B/s]"270 ]271 },272 "metadata": {},273 "output_type": "display_data"274 },275 {276 "name": "stdout",277 "output_type": "stream",278 "text": [279 "def print_prime(n):\n",280 " \"\"\"\n",281 " Print all primes between 1 and n\n",282 " \"\"\"\n",283 " primes = []\n",284 " for num in range(2, n+1):\n",285 " is_prime = True\n",286 " for i in range(2, int(math.sqrt(num))+1):\n",287 " if num % i == 0:\n",288 " is_prime = False\n",289 " break\n",290 " if is_prime:\n",291 " primes.append(num)\n",292 " print(primes)\n",293 " \n",294 "print_prime(20)\n",295 "```\n",296 "\n",297 "Output:\n",298 "```\n",299 "[2, 3, 5, 7, 11, 13, 17, 19]\n",300 "```\n",301 "\n",302 "Exercise 5:\n",303 "Write a Python function that takes a list of numbers and returns the sum of all even numbers in the list.\n",304 "\n",305 "```python\n",306 "def sum_even(numbers):\n",307 " \"\"\"\n",308 " \n"309 ]310 }311 ],312 "source": [313 "# sample code Phi v1.5 \n",314 "\n",315 "import torch\n",316 "from transformers import AutoModelForCausalLM, AutoTokenizer\n",317 "\n",318 "torch.set_default_device(\"cuda\")\n",319 "\n",320 "model = AutoModelForCausalLM.from_pretrained(\"microsoft/phi-1_5\", torch_dtype=\"auto\", trust_remote_code=True, cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",321 "tokenizer = AutoTokenizer.from_pretrained(\"microsoft/phi-1_5\", trust_remote_code=True, cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",322 "\n",323 "inputs = tokenizer('''def print_prime(n):\n",324 " \"\"\"\n",325 " Print all primes between 1 and n\n",326 " \"\"\"''', return_tensors=\"pt\", return_attention_mask=False)\n",327 "\n",328 "outputs = model.generate(**inputs, max_length=200)\n",329 "text = tokenizer.batch_decode(outputs)[0]\n",330 "print(text)\n"331 ]332 },333 {334 "cell_type": "markdown",335 "id": "ceaf3bdc-35f9-4be8-9cab-a170ae43371e",336 "metadata": {},337 "source": [338 "Sample code Phi2\n",339 "Microsoft wasn't very open about it being on hugging face \n",340 "__SOURCE:__ https://huggingface.co/microsoft/phi-2"341 ]342 },343 {344 "cell_type": "code",345 "execution_count": null,346 "id": "b6765e63-a9ea-4589-a3f1-8207df5daa12",347 "metadata": {},348 "outputs": [349 {350 "data": {351 "application/vnd.jupyter.widget-view+json": {352 "model_id": "949ce3dd53e547f79ddf718e9b27a352",353 "version_major": 2,354 "version_minor": 0355 },356 "text/plain": [357 "config.json: 0%| | 0.00/755 [00:00<?, ?B/s]"358 ]359 },360 "metadata": {},361 "output_type": "display_data"362 },363 {364 "data": {365 "application/vnd.jupyter.widget-view+json": {366 "model_id": "e0700ee929524052bbecd9b766d1fafb",367 "version_major": 2,368 "version_minor": 0369 },370 "text/plain": [371 "configuration_phi.py: 0%| | 0.00/2.03k [00:00<?, ?B/s]"372 ]373 },374 "metadata": {},375 "output_type": "display_data"376 },377 {378 "name": "stderr",379 "output_type": "stream",380 "text": [381 "A new version of the following files was downloaded from https://huggingface.co/microsoft/phi-2:\n",382 "- configuration_phi.py\n",383 ". Make sure to double-check they do not contain any added malicious code. To avoid downloading new versions of the code file, you can pin a revision.\n"384 ]385 },386 {387 "data": {388 "application/vnd.jupyter.widget-view+json": {389 "model_id": "51cffc23be5d4b8c84d2fdbb4135f572",390 "version_major": 2,391 "version_minor": 0392 },393 "text/plain": [394 "modeling_phi.py: 0%| | 0.00/33.4k [00:00<?, ?B/s]"395 ]396 },397 "metadata": {},398 "output_type": "display_data"399 },400 {401 "name": "stderr",402 "output_type": "stream",403 "text": [404 "A new version of the following files was downloaded from https://huggingface.co/microsoft/phi-2:\n",405 "- modeling_phi.py\n",406 ". Make sure to double-check they do not contain any added malicious code. To avoid downloading new versions of the code file, you can pin a revision.\n"407 ]408 },409 {410 "data": {411 "application/vnd.jupyter.widget-view+json": {412 "model_id": "7a1f54ba9a074c1882def000fd0f241a",413 "version_major": 2,414 "version_minor": 0415 },416 "text/plain": [417 "model.safetensors.index.json: 0%| | 0.00/24.3k [00:00<?, ?B/s]"418 ]419 },420 "metadata": {},421 "output_type": "display_data"422 },423 {424 "data": {425 "application/vnd.jupyter.widget-view+json": {426 "model_id": "4c7a8db4aa8c45e790f9c57c66db644d",427 "version_major": 2,428 "version_minor": 0429 },430 "text/plain": [431 "Downloading shards: 0%| | 0/2 [00:00<?, ?it/s]"432 ]433 },434 "metadata": {},435 "output_type": "display_data"436 },437 {438 "data": {439 "application/vnd.jupyter.widget-view+json": {440 "model_id": "48431a40e93b4b768e33dfd0a03aec16",441 "version_major": 2,442 "version_minor": 0443 },444 "text/plain": [445 "model-00001-of-00002.safetensors: 0%| | 0.00/4.98G [00:00<?, ?B/s]"446 ]447 },448 "metadata": {},449 "output_type": "display_data"450 },451 {452 "data": {453 "application/vnd.jupyter.widget-view+json": {454 "model_id": "906a2bdffe78429aa56ba798bf66bc4b",455 "version_major": 2,456 "version_minor": 0457 },458 "text/plain": [459 "model-00002-of-00002.safetensors: 0%| | 0.00/577M [00:00<?, ?B/s]"460 ]461 },462 "metadata": {},463 "output_type": "display_data"464 },465 {466 "data": {467 "application/vnd.jupyter.widget-view+json": {468 "model_id": "97c906f0d10143fab97b8c15547b81fc",469 "version_major": 2,470 "version_minor": 0471 },472 "text/plain": [473 "Loading checkpoint shards: 0%| | 0/2 [00:00<?, ?it/s]"474 ]475 },476 "metadata": {},477 "output_type": "display_data"478 },479 {480 "data": {481 "application/vnd.jupyter.widget-view+json": {482 "model_id": "7a9a1bb6680c4861b3e4612f8bc6f3e2",483 "version_major": 2,484 "version_minor": 0485 },486 "text/plain": [487 "generation_config.json: 0%| | 0.00/69.0 [00:00<?, ?B/s]"488 ]489 },490 "metadata": {},491 "output_type": "display_data"492 },493 {494 "data": {495 "application/vnd.jupyter.widget-view+json": {496 "model_id": "122eb9cdcfd544e6b681c98165aaf74c",497 "version_major": 2,498 "version_minor": 0499 },500 "text/plain": [501 "tokenizer_config.json: 0%| | 0.00/7.34k [00:00<?, ?B/s]"502 ]503 },504 "metadata": {},505 "output_type": "display_data"506 },507 {508 "data": {509 "application/vnd.jupyter.widget-view+json": {510 "model_id": "74b6e1ccb39f465797225af1b8894e6b",511 "version_major": 2,512 "version_minor": 0513 },514 "text/plain": [515 "vocab.json: 0%| | 0.00/798k [00:00<?, ?B/s]"516 ]517 },518 "metadata": {},519 "output_type": "display_data"520 },521 {522 "data": {523 "application/vnd.jupyter.widget-view+json": {524 "model_id": "3ae4e772854c4d11b6f9a195f28e2e51",525 "version_major": 2,526 "version_minor": 0527 },528 "text/plain": [529 "merges.txt: 0%| | 0.00/456k [00:00<?, ?B/s]"530 ]531 },532 "metadata": {},533 "output_type": "display_data"534 },535 {536 "data": {537 "application/vnd.jupyter.widget-view+json": {538 "model_id": "8aee1ac95ae9435c96fa407be15d8246",539 "version_major": 2,540 "version_minor": 0541 },542 "text/plain": [543 "tokenizer.json: 0%| | 0.00/2.11M [00:00<?, ?B/s]"544 ]545 },546 "metadata": {},547 "output_type": "display_data"548 },549 {550 "data": {551 "application/vnd.jupyter.widget-view+json": {552 "model_id": "4dff5b60b0f84c3c8527d7dc3e063907",553 "version_major": 2,554 "version_minor": 0555 },556 "text/plain": [557 "added_tokens.json: 0%| | 0.00/1.08k [00:00<?, ?B/s]"558 ]559 },560 "metadata": {},561 "output_type": "display_data"562 },563 {564 "data": {565 "application/vnd.jupyter.widget-view+json": {566 "model_id": "7ae9dc2d005047aa81dc09232f7560a1",567 "version_major": 2,568 "version_minor": 0569 },570 "text/plain": [571 "special_tokens_map.json: 0%| | 0.00/99.0 [00:00<?, ?B/s]"572 ]573 },574 "metadata": {},575 "output_type": "display_data"576 },577 {578 "name": "stderr",579 "output_type": "stream",580 "text": [581 "Special tokens have been added in the vocabulary, make sure the associated word embeddings are fine-tuned or trained.\n"582 ]583 }584 ],585 "source": [586 "import torch\n",587 "from transformers import AutoModelForCausalLM, AutoTokenizer\n",588 "\n",589 "torch.set_default_device(\"cuda\")\n",590 "\n",591 "model = AutoModelForCausalLM.from_pretrained(\"microsoft/phi-2\", torch_dtype=\"auto\", trust_remote_code=True, cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",592 "tokenizer = AutoTokenizer.from_pretrained(\"microsoft/phi-2\", trust_remote_code=True, cache_dir=\"/mnt/f/wkspc-linux/.cache\")\n",593 "\n",594 "inputs = tokenizer('''def print_prime(n):\n",595 " \"\"\"\n",596 " Print all primes between 1 and n\n",597 " \"\"\"''', return_tensors=\"pt\", return_attention_mask=False)\n",598 "\n",599 "outputs = model.generate(**inputs, max_length=200)\n",600 "text = tokenizer.batch_decode(outputs)[0]\n",601 "print(text)\n"602 ]603 },604 {605 "cell_type": "code",606 "execution_count": null,607 "id": "c59e0b78-784a-4a67-9863-21e92c24c2f5",608 "metadata": {},609 "outputs": [],610 "source": []611 }612 ],613 "metadata": {614 "kernelspec": {615 "display_name": "Python 3 (ipykernel)",616 "language": "python",617 "name": "python3"618 },619 "language_info": {620 "codemirror_mode": {621 "name": "ipython",622 "version": 3623 },624 "file_extension": ".py",625 "mimetype": "text/x-python",626 "name": "python",627 "nbconvert_exporter": "python",628 "pygments_lexer": "ipython3",629 "version": "3.9.2"630 }631 },632 "nbformat": 4,633 "nbformat_minor": 5634}635 