Team Ai
Datasetpublic

ysn-rfd/text-dataset-tiny-code-script-py-format

USED of tahamajs/medicine_ds_persian for .parquet file USED of Alijafarixcs2/persian-it-llama2-2k for .parquet file USED of Abirate/english_quotes for .jsonl file NEW FILES (05/12/2025) NEW FILES (12/26/2025) NEW FILES (02/15/2026)

sourceHugging Faceapache-2.0updated 4mo agoView on Hugging Face
3likes1.7kdownloads
Untitled.ipynb262 linesDownload Raw Back to opencv_test
1{2 "cells": [3  {4   "cell_type": "code",5   "execution_count": 1,6   "id": "9991fa5b-c41f-4117-a81f-75727fa4a74d",7   "metadata": {},8   "outputs": [9    {10     "data": {11      "application/vnd.jupyter.widget-view+json": {12       "model_id": "a77abb8840a9407ebf60f186551f64d4",13       "version_major": 2,14       "version_minor": 015      },16      "text/plain": [17       "pytorch_model.bin:   0%|          | 0.00/485M [00:00<?, ?B/s]"18      ]19     },20     "metadata": {},21     "output_type": "display_data"22    },23    {24     "name": "stdout",25     "output_type": "stream",26     "text": [27      "* Running on local URL:  http://127.0.0.1:7860\n",28      "* To create a public link, set `share=True` in `launch()`.\n"29     ]30    },31    {32     "data": {33      "text/html": [34       "<div><iframe src=\"http://127.0.0.1:7860/\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"35      ],36      "text/plain": [37       "<IPython.core.display.HTML object>"38      ]39     },40     "metadata": {},41     "output_type": "display_data"42    },43    {44     "data": {45      "text/plain": []46     },47     "execution_count": 1,48     "metadata": {},49     "output_type": "execute_result"50    }51   ],52   "source": [53    "import gradio as gr\n",54    "from transformers import AutoTokenizer, AutoModelForCausalLM\n",55    "import torch\n",56    "\n",57    "# بارگذاری مدل و توکنایزر\n",58    "tokenizer = AutoTokenizer.from_pretrained(\"HooshvareLab/gpt2-fa\")\n",59    "model = AutoModelForCausalLM.from_pretrained(\"HooshvareLab/gpt2-fa\")\n",60    "\n",61    "# تابع تولید متن\n",62    "def generate_text(prompt):\n",63    "    input_ids = tokenizer.encode(prompt, return_tensors=\"pt\")\n",64    "    output = model.generate(\n",65    "        input_ids,\n",66    "        max_new_tokens=120,\n",67    "        do_sample=True,\n",68    "        temperature=0.7,\n",69    "        top_k=40,\n",70    "        top_p=0.9,\n",71    "        repetition_penalty=1.3,\n",72    "        pad_token_id=tokenizer.eos_token_id\n",73    "    )\n",74    "    return tokenizer.decode(output[0], skip_special_tokens=True)\n",75    "\n",76    "# رابط گرافیکی\n",77    "gr.Interface(\n",78    "    fn=generate_text,\n",79    "    inputs=gr.Textbox(label=\"متن ورودی (پرامپت)\", placeholder=\"مثلاً: در یک روز بهاری،\"),\n",80    "    outputs=gr.Textbox(label=\"متن تولید شده\"),\n",81    "    title=\"تولید متن فارسی با GPT2\",\n",82    "    description=\"مدل: bolbolzaban/gpt2-persian - تولید خودکار متن طبیعی به زبان فارسی\"\n",83    ").launch()\n"84   ]85  },86  {87   "cell_type": "code",88   "execution_count": 1,89   "id": "c4d50353-c26f-45a9-9962-da748dc618f4",90   "metadata": {},91   "outputs": [92    {93     "name": "stdout",94     "output_type": "stream",95     "text": [96      "* Running on local URL:  http://127.0.0.1:7860\n",97      "* To create a public link, set `share=True` in `launch()`.\n"98     ]99    },100    {101     "data": {102      "text/html": [103       "<div><iframe src=\"http://127.0.0.1:7860/\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"104      ],105      "text/plain": [106       "<IPython.core.display.HTML object>"107      ]108     },109     "metadata": {},110     "output_type": "display_data"111    },112    {113     "data": {114      "text/plain": []115     },116     "execution_count": 1,117     "metadata": {},118     "output_type": "execute_result"119    }120   ],121   "source": [122    "import gradio as gr\n",123    "from transformers import AutoTokenizer, AutoModelForCausalLM\n",124    "import torch\n",125    "\n",126    "# بارگذاری مدل و توکنایزر\n",127    "tokenizer = AutoTokenizer.from_pretrained(\"HooshvareLab/gpt2-fa\")\n",128    "model = AutoModelForCausalLM.from_pretrained(\"HooshvareLab/gpt2-fa\")\n",129    "\n",130    "# تابع تولید متن با تنظیمات دقیق\n",131    "def generate_text(prompt):\n",132    "    input_ids = tokenizer.encode(prompt, return_tensors=\"pt\")\n",133    "    output = model.generate(\n",134    "        input_ids,\n",135    "        max_new_tokens=150,\n",136    "        do_sample=True,\n",137    "        temperature=0.7,\n",138    "        top_k=50,\n",139    "        top_p=0.92,\n",140    "        repetition_penalty=1.3,\n",141    "        no_repeat_ngram_size=3,\n",142    "        pad_token_id=tokenizer.eos_token_id,\n",143    "        eos_token_id=tokenizer.eos_token_id\n",144    "    )\n",145    "    return tokenizer.decode(output[0], skip_special_tokens=True)\n",146    "\n",147    "# ساخت رابط گرافیکی با Gradio\n",148    "gr.Interface(\n",149    "    fn=generate_text,\n",150    "    inputs=gr.Textbox(label=\"📝 متن ورودی (پرامپت)\", placeholder=\"مثلاً: در یک روز بهاری،\", lines=3),\n",151    "    outputs=gr.Textbox(label=\"📄 متن تولید شده\"),\n",152    "    title=\"💬 تولید متن فارسی دقیق با GPT2\",\n",153    "    description=\"این ابزار از مدل bolbolzaban/gpt2-persian استفاده می‌کند و متن روان و طبیعی به زبان فارسی تولید می‌کند.\",\n",154    "    theme=\"soft\"\n",155    ").launch()\n"156   ]157  },158  {159   "cell_type": "code",160   "execution_count": 1,161   "id": "eba7eda5-f61f-434c-b500-58a8b33c6f0e",162   "metadata": {},163   "outputs": [164    {165     "name": "stdout",166     "output_type": "stream",167     "text": [168      "* Running on local URL:  http://127.0.0.1:7860\n",169      "* To create a public link, set `share=True` in `launch()`.\n"170     ]171    },172    {173     "data": {174      "text/html": [175       "<div><iframe src=\"http://127.0.0.1:7860/\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"176      ],177      "text/plain": [178       "<IPython.core.display.HTML object>"179      ]180     },181     "metadata": {},182     "output_type": "display_data"183    },184    {185     "data": {186      "text/plain": []187     },188     "execution_count": 1,189     "metadata": {},190     "output_type": "execute_result"191    }192   ],193   "source": [194    "import gradio as gr\n",195    "from transformers import AutoTokenizer, AutoModelForCausalLM\n",196    "import torch\n",197    "\n",198    "# بارگذاری مدل و توکنایزر\n",199    "tokenizer = AutoTokenizer.from_pretrained(\"HooshvareLab/gpt2-fa\")\n",200    "model = AutoModelForCausalLM.from_pretrained(\"HooshvareLab/gpt2-fa\")\n",201    "\n",202    "# تابع تولید متن با دستور\n",203    "def generate_with_instruction(instruction):\n",204    "    prompt = f\"دستور: {instruction}\\nپاسخ:\"\n",205    "    input_ids = tokenizer.encode(prompt, return_tensors=\"pt\")\n",206    "    output = model.generate(\n",207    "        input_ids,\n",208    "        max_new_tokens=150,\n",209    "        do_sample=True,\n",210    "        temperature=0.7,\n",211    "        top_k=50,\n",212    "        top_p=0.92,\n",213    "        repetition_penalty=1.3,\n",214    "        no_repeat_ngram_size=3,\n",215    "        pad_token_id=tokenizer.eos_token_id,\n",216    "        eos_token_id=tokenizer.eos_token_id\n",217    "    )\n",218    "    return tokenizer.decode(output[0], skip_special_tokens=True)\n",219    "\n",220    "# رابط گرافیکی\n",221    "gr.Interface(\n",222    "    fn=generate_with_instruction,\n",223    "    inputs=gr.Textbox(label=\"📝 دستور وارد کنید\", placeholder=\"مثلاً: درباره‌ی تاثیر ورزش بر ذهن بنویس\", lines=3),\n",224    "    outputs=gr.Textbox(label=\"📄 پاسخ مدل\"),\n",225    "    title=\"💬 دستور به مدل GPT2 فارسی\",\n",226    "    description=\"با وارد کردن دستور در قالب فارسی، مدل شروع به تولید متن می‌کند. مثل: نوشتن، خلاصه‌سازی یا ترجمه.\",\n",227    "    theme=\"soft\"\n",228    ").launch()\n"229   ]230  },231  {232   "cell_type": "code",233   "execution_count": null,234   "id": "b83582e3-9740-46a5-8a1b-6329acb9ef16",235   "metadata": {},236   "outputs": [],237   "source": []238  }239 ],240 "metadata": {241  "kernelspec": {242   "display_name": "Python [conda env:base] *",243   "language": "python",244   "name": "conda-base-py"245  },246  "language_info": {247   "codemirror_mode": {248    "name": "ipython",249    "version": 3250   },251   "file_extension": ".py",252   "mimetype": "text/x-python",253   "name": "python",254   "nbconvert_exporter": "python",255   "pygments_lexer": "ipython3",256   "version": "3.12.7"257  }258 },259 "nbformat": 4,260 "nbformat_minor": 5261}262