FadouaFGM/Stackoverflow_Questions_Categorisation
3
1{2 "cells": [3 {4 "cell_type": "markdown",5 "id": "6dc82e22-12de-4a26-a71c-0cda7c0788d1",6 "metadata": {},7 "source": [8 "# <center> <font color='blue'> **Projet 5: Catégorisez automatiquement des questions**\n",9 "# <center> <font color='goldenrode'> **Notebook: API**\n"10 ]11 },12 {13 "cell_type": "markdown",14 "id": "70a1feae-070b-4742-b953-701c090d1578",15 "metadata": {},16 "source": [17 "**On commence tout d'abord par la définition des fonctions globales utilisées pour le nettoyage du texte, ensuite la vectorisation et le modèle ML choisi lors de l'étude menée dans le notebook précédent.**"18 ]19 },20 {21 "cell_type": "code",22 "execution_count": 1,23 "id": "069f9d87-a932-4eb1-9c76-2cc7c39e1470",24 "metadata": {},25 "outputs": [],26 "source": [27 "from sklearn.pipeline import Pipeline\n",28 "from sklearn.feature_extraction.text import TfidfVectorizer\n",29 "from sklearn.multiclass import OneVsRestClassifier\n",30 "from sklearn.linear_model import LogisticRegression\n",31 "from sklearn.linear_model import SGDClassifier\n",32 "from sklearn.preprocessing import MultiLabelBinarizer\n",33 "import pandas as pd\n",34 "import numpy as np\n",35 "import matplotlib.pyplot as plt\n",36 "import seaborn as sns\n",37 "import time\n",38 "import warnings\n",39 "import re\n",40 "import nltk\n",41 "import spacy\n",42 "import re\n",43 "\n",44 "from nltk.tokenize import WordPunctTokenizer\n",45 "from nltk.corpus import stopwords\n"46 ]47 },48 {49 "cell_type": "code",50 "execution_count": 2,51 "id": "cdf58c5c-aa77-4c5b-aa72-3f6f9c1c2fcb",52 "metadata": {},53 "outputs": [],54 "source": [55 "#!pip install spacy"56 ]57 },58 {59 "cell_type": "code",60 "execution_count": 3,61 "id": "70d93c3b-1fcc-4c9c-a34d-f0b70d011149",62 "metadata": {},63 "outputs": [],64 "source": [65 "nlp = spacy.load(\"en_core_web_sm\")\n",66 "nlp.Defaults.stop_words.add(\"`,\")\n",67 "nlp.Defaults.stop_words.add(\"``\")"68 ]69 },70 {71 "cell_type": "markdown",72 "id": "094930b6-f328-4386-9044-f799e7fefb0c",73 "metadata": {},74 "source": [75 "### **Définition des fonctions et modèle**"76 ]77 },78 {79 "cell_type": "code",80 "execution_count": 4,81 "id": "31a7ec44-9c68-4de7-a2b5-65f1b7b00a38",82 "metadata": {},83 "outputs": [],84 "source": [85 "# Define functions\n",86 "\n",87 "#lemmatize text without stop or punctuation words\n",88 "def lemmatize(text):\n",89 " doc = nlp(text)\n",90 " tokens = [token.lemma_ for token in doc if not (token.is_stop or token.is_digit or token.is_punct)]\n",91 " return ' '.join(tokens)\n",92 "\n",93 "def tokenization(text):\n",94 " tokens = WordPunctTokenizer().tokenize(text)\n",95 " return tokens"96 ]97 },98 {99 "cell_type": "code",100 "execution_count": 5,101 "id": "2897a639-4b4b-41a4-b6d2-5ffe20124b22",102 "metadata": {},103 "outputs": [],104 "source": [105 "# function to preprocess text\n",106 "def clean(text):\n",107 " \n",108 " #Lower case\n",109 " text = text.lower()\n",110 " # removing paragraph numbers\n",111 " text = re.sub('[0-9]+.\\t','',str(text))\n",112 " # change the pattern C# to csharp\n",113 " pattern = r'c#' \n",114 " text = re.sub(pattern, 'csharp', text)\n",115 " #removing web and html links\n",116 " text = re.sub(r'http\\S+', '', text)\n",117 " # removing special characters\n",118 " text = re.sub(\"</p>\",'',str(text))\n",119 " text = re.sub(\"<p>\",'',str(text))\n",120 " text = re.sub(\"</pre>\",'',str(text))\n",121 " text = re.sub(\"<pre>\",'',str(text))\n",122 " text = re.sub(\"&\",'',str(text))\n",123 " text = re.sub(\";\",'',str(text))\n",124 " text = re.sub(\"gt\",' ',str(text))\n",125 " text = re.sub(\"pre\",'',str(text))\n",126 " # removing any reference to outside text\n",127 " text = re.sub(\"[\\(\\[].*?[\\)\\]]\", \"\", str(text))\n",128 " # removing numbers\n",129 " text = re.sub('[0-9]','',str(text))\n",130 " # removing new line characters\n",131 " text = re.sub('\\n ','',str(text))\n",132 " text = re.sub('\\n',' ',str(text))\n",133 " # removing apostrophes\n",134 " text = re.sub(\"'s\",'',str(text))\n",135 " # removing hyphens\n",136 " text = re.sub(\"-\",' ',str(text))\n",137 " text = re.sub(\"—\",'',str(text))\n",138 " # removing > or < or = signs\n",139 " text = re.sub(\"<\",' ',str(text))\n",140 " text = re.sub(\">\",'',str(text))\n",141 " text = re.sub(\"=\",'',str(text))\n",142 " # removing quotation marks\n",143 " text = re.sub('\\\"','',str(text))\n",144 " # removing quotation marks\n",145 " text = re.sub('/','',str(text))\n",146 " # Use regex to delete all what's inside < >\n",147 " CLEANR = re.compile('<.*?>') \n",148 " text = re.sub(CLEANR, '', text)\n",149 " \n",150 " return text"151 ]152 },153 {154 "cell_type": "code",155 "execution_count": 6,156 "id": "8e75252b-50c7-4475-8e8c-9d354c523b89",157 "metadata": {},158 "outputs": [],159 "source": [160 "def remove_code(text):\n",161 " \n",162 " #first position of the code in code\n",163 " codepointer=text.find('<code>')\n",164 " result=''\n",165 " \n",166 " while codepointer!=-1:\n",167 " #last position of /code\n",168 " codeender=text.find(u'</code>',codepointer)\n",169 " #the code between pointer and ender\n",170 " result=result+text[codepointer:codeender+7]\n",171 " codepointer=text.find('<code>',codeender)\n",172 " \n",173 " listOfWords2remove = ([i for i in result.split()])\n",174 " \n",175 " for i in listOfWords2remove:\n",176 " text = text.replace(i, '') \n",177 " \n",178 " return text\n"179 ]180 },181 {182 "cell_type": "code",183 "execution_count": 7,184 "id": "6d53ceb9-3b23-477f-8495-bb050805a412",185 "metadata": {},186 "outputs": [],187 "source": [188 "def text_processing(dfoftext):\n",189 " \n",190 " cleaneddftext = dfoftext.apply(lambda txt : remove_code(txt))\n",191 " cleaneddftext = cleaneddftext.apply(lambda txt : clean(txt))\n",192 " cleaneddftext = cleaneddftext.apply(lambda txt : lemmatize(txt))\n",193 " \n",194 " return cleaneddftext\n"195 ]196 },197 {198 "cell_type": "code",199 "execution_count": 8,200 "id": "140c5699-f51f-4aa2-b50c-068217f75ca9",201 "metadata": {},202 "outputs": [],203 "source": [204 "def multilabeloutputcoding(target):\n",205 " \n",206 " multilablBin = MultiLabelBinarizer()\n",207 " y = pd.DataFrame(multilablBin.fit_transform(target.str.split(\" \")),columns=multilablBin.classes_)\n",208 " tags_list = list(y.columns)\n",209 " \n",210 " return y, tags_list"211 ]212 },213 {214 "cell_type": "code",215 "execution_count": 9,216 "id": "3d88a4d0-4d9b-48f7-8247-69091c4faf59",217 "metadata": {},218 "outputs": [],219 "source": [220 "pipeline = Pipeline([\n",221 " ('vectorizer', TfidfVectorizer(tokenizer = lambda x: x.split(), sublinear_tf=False)),\n",222 " ('classifier', OneVsRestClassifier(SGDClassifier(alpha=0.0001, loss='modified_huber', penalty='none')))\n",223 "])"224 ]225 },226 {227 "cell_type": "code",228 "execution_count": 10,229 "id": "15897789-9f15-4fa9-a005-2f48428f7f96",230 "metadata": {},231 "outputs": [],232 "source": [233 "# Define function to predict with the new list of thresholds with attributing a threshold per label\n",234 "def predict_with_thresholds(y_prob, thresholds):\n",235 " y_pred = np.zeros_like(y_prob)\n",236 " for i in range(y_prob.shape[1]):\n",237 " y_pred[:, i] = (y_prob[:, i] >= thresholds[i]).astype(int)\n",238 " return y_pred"239 ]240 },241 {242 "cell_type": "code",243 "execution_count": 34,244 "id": "61a8cd32-4279-4ba3-b626-1347f25eae06",245 "metadata": {},246 "outputs": [],247 "source": [248 "import joblib\n",249 "\n",250 "def makeprediction(text):\n",251 " # load the pre-trained TfidfVectorizer from disk\n",252 " tfidf = joblib.load('tfidf_vectorizer.joblib')\n",253 " \n",254 " # load the pre-trained Linear_SGD classifier from disk\n",255 " ovr = joblib.load('linear_sgd_classifier.joblib')\n",256 " \n",257 " # Processing the text\n",258 " cleanedtext = text_processing(text)\n",259 " print(cleanedtext)\n",260 " print(type(cleanedtext))\n",261 " \n",262 " # applying the model and reconstruction predicted targets\n",263 " texttfidf = tfidf.transform(cleanedtext)\n",264 " \n",265 " # make prediction with pretrained classifier\n",266 " ypred = ovr.predict_proba(texttfidf)\n",267 " print(ypred)\n",268 " \n",269 " # recontructing tags from predicted y\n",270 " thresholds = joblib.load('thresholds.joblib')\n",271 " labels = joblib.load('labels.joblib')\n",272 " \n",273 " y_pred_thr = predict_with_thresholds(ypred,thresholds)\n",274 " print(y_pred_thr)\n",275 " \n",276 " tags_pred = [[labels[i] for i in range(len(yp)) if yp[i] == 1] for yp in y_pred_thr]\n",277 " #tags_pred = tags_pred.apply(lambda x: x if x else ['no predicted labels'])\n",278 " \n",279 " return tags_pred\n",280 " "281 ]282 },283 {284 "cell_type": "markdown",285 "id": "78207713-1c2b-4e8f-a790-84059621c9e2",286 "metadata": {},287 "source": [288 "### **Prédiction d'un exemple de Texte**"289 ]290 },291 {292 "cell_type": "code",293 "execution_count": 35,294 "id": "7f4ded62-466a-430f-be6a-c8fe8b45fbd2",295 "metadata": {},296 "outputs": [],297 "source": [298 "#!pip install gradio"299 ]300 },301 {302 "cell_type": "code",303 "execution_count": 36,304 "id": "924beecb-242b-4b8b-a0aa-e1d55b124153",305 "metadata": {},306 "outputs": [],307 "source": [308 "text1 = '<p>Is there a way to record the screen, either desktop or window, using .NET technologies.</p>\\n\\n<p>My goal is something free. I like the idea of small, low cpu usage, and simple, but would consider other options if they created a better final product.</p>\\n\\n<p>In a nutshell, I know how to take a screenshot in C#, but how would I record the screen, or area of the screen, as a video?</p>\\n\\n<p>Thanks a lot for your ideas and time!</p>\\n'\n",309 "text2 = \"<p>I've seen <code>default</code> used next to function declarations in a class. What does it do?</p>\\n\\n<pre><code>class C {\\n C(const C&) = default;\\n C(C&&) = default;\\n C& operator=(const C&) & = default;\\n C& operator=(C&&) & = default;\\n virtual ~C() { }\\n};\\n</code></pre>\\n\""310 ]311 },312 {313 "cell_type": "code",314 "execution_count": 37,315 "id": "c6b5e46a-0509-4bed-9e70-076f64db06d9",316 "metadata": {},317 "outputs": [318 {319 "data": {320 "text/plain": [321 "'<p>Is there a way to record the screen, either desktop or window, using .NET technologies.</p>\\n\\n<p>My goal is something free. I like the idea of small, low cpu usage, and simple, but would consider other options if they created a better final product.</p>\\n\\n<p>In a nutshell, I know how to take a screenshot in C#, but how would I record the screen, or area of the screen, as a video?</p>\\n\\n<p>Thanks a lot for your ideas and time!</p>\\n'"322 ]323 },324 "execution_count": 37,325 "metadata": {},326 "output_type": "execute_result"327 }328 ],329 "source": [330 "data = [[text1],[text2]]\n",331 "data = pd.DataFrame(data, columns = ['Text'])\n",332 "data.iloc[0]['Text']"333 ]334 },335 {336 "cell_type": "code",337 "execution_count": 38,338 "id": "e0b43ebc-4428-4b4d-a220-55c69e92d68f",339 "metadata": {},340 "outputs": [341 {342 "name": "stdout",343 "output_type": "stream",344 "text": [345 "0 way record screen desktop window .net technolo...\n",346 "1 see codedefault function declaration class ...\n",347 "Name: Text, dtype: object\n",348 "<class 'pandas.core.series.Series'>\n",349 "[[0.48948313 0. 0. 0. 0. 0.\n",350 " 0.67368617 0. 0. 0. 0. 0.\n",351 " 0. 0. 0. 0. 0. 0.\n",352 " 0. 0. 0. 0. 0. 0.\n",353 " 0. 0. 0. 0. 0. 0.\n",354 " 0. 0. 0. 0. 0. 0.\n",355 " 0. 0. 0. 0. 0. 0.\n",356 " 0. 0. 0. 0. 0. 0. ]\n",357 " [0. 0. 0. 0. 0. 0.\n",358 " 0.3016006 0.49029685 0. 0. 0. 0.\n",359 " 0. 0. 0. 0. 0. 0.\n",360 " 0. 0. 0. 0. 0. 0.05103097\n",361 " 0. 0. 0. 0. 0. 0.\n",362 " 0. 0. 0. 0. 0. 0.\n",363 " 0. 0. 0. 0. 0. 0.\n",364 " 0. 0. 0. 0. 0. 0. ]]\n",365 "[[1. 0. 0. 0. 0. 0. 1. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0.\n",366 " 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0.]\n",367 " [0. 0. 0. 0. 0. 0. 1. 1. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0.\n",368 " 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0.]]\n"369 ]370 },371 {372 "data": {373 "text/plain": [374 "[['.net', 'c'], ['c', 'c++']]"375 ]376 },377 "execution_count": 38,378 "metadata": {},379 "output_type": "execute_result"380 }381 ],382 "source": [383 "tags = makeprediction(data['Text'])\n",384 "tags"385 ]386 },387 {388 "cell_type": "markdown",389 "id": "2285d443-6a28-4822-8bb9-b2f96213e8ff",390 "metadata": {},391 "source": [392 "### **Implémentation de l'API: GradioAPI**"393 ]394 },395 {396 "cell_type": "code",397 "execution_count": 39,398 "id": "53f33875-626c-463f-aa8e-0297a4796eaf",399 "metadata": {},400 "outputs": [401 {402 "name": "stdout",403 "output_type": "stream",404 "text": [405 "Running on local URL: http://127.0.0.1:7874\n",406 "\n",407 "To create a public link, set `share=True` in `launch()`.\n"408 ]409 },410 {411 "data": {412 "text/html": [413 "<div><iframe src=\"http://127.0.0.1:7874/\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"414 ],415 "text/plain": [416 "<IPython.core.display.HTML object>"417 ]418 },419 "metadata": {},420 "output_type": "display_data"421 },422 {423 "data": {424 "text/plain": []425 },426 "execution_count": 39,427 "metadata": {},428 "output_type": "execute_result"429 },430 {431 "name": "stdout",432 "output_type": "stream",433 "text": [434 "0 see codedefault code function declaration cl...\n",435 "Name: Text, dtype: object\n",436 "<class 'pandas.core.series.Series'>\n",437 "[[0. 0. 0. 0. 0. 0.\n",438 " 0.16757153 0.4143708 0. 0. 0. 0.\n",439 " 0. 0. 0. 0. 0. 0.\n",440 " 0. 0. 0. 0. 0. 0.\n",441 " 0. 0. 0. 0. 0. 0.\n",442 " 0. 0. 0. 0. 0. 0.\n",443 " 0. 0. 0. 0. 0. 0.\n",444 " 0. 0. 0. 0. 0. 0. ]]\n",445 "[[0. 0. 0. 0. 0. 0. 1. 1. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0.\n",446 " 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0. 0.]]\n"447 ]448 }449 ],450 "source": [451 "import gradio as gra\n",452 "def predict(text: List[str]):\n",453 " data = [[text]]\n",454 " data = pd.DataFrame(data, columns = ['Text'])\n",455 " tags = makeprediction(data['Text'])\n",456 " return {\"tags!😎\": tags} \n",457 "#define gradio interface and other parameters\n",458 "app = gra.Interface(fn = predict, inputs=\"text\", outputs=\"text\")\n",459 "app.launch()"460 ]461 },462 {463 "cell_type": "code",464 "execution_count": 24,465 "id": "a9d78433-3339-4059-adf8-a9dc161dde03",466 "metadata": {},467 "outputs": [],468 "source": [469 "from fastapi import FastAPI\n",470 "from typing import List\n",471 "\n",472 "app = FastAPI()\n",473 "\n",474 "@app.post('/predict')\n",475 "def predict(text: List[str]):\n",476 " data = [[text]]\n",477 " data = pd.DataFrame(data, columns = ['Text'])\n",478 " tags = makeprediction(data['Text'])\n",479 " return {\"tags\": tags}"480 ]481 },482 {483 "cell_type": "code",484 "execution_count": 20,485 "id": "2415d200-d6e9-4f16-84f0-6f3bb3bd1673",486 "metadata": {},487 "outputs": [],488 "source": [489 "#uvicorn P5_API_1602:predict --reload"490 ]491 },492 {493 "cell_type": "code",494 "execution_count": 21,495 "id": "cada3d8d-9b6c-4579-9dca-bcf7f744ce81",496 "metadata": {},497 "outputs": [],498 "source": [499 "# Execute following commands in the endpoint \n",500 "#result = makeprediction(text, target)\n",501 "#return {\"result\": result}"502 ]503 },504 {505 "cell_type": "markdown",506 "id": "a55ce440-95af-4328-91b7-66071651a99c",507 "metadata": {},508 "source": [509 "Test the endpoint using a tool like Postman or cURL by sending a POST request to http://localhost:8000/predict with the text and target parameters in the request body."510 ]511 },512 {513 "cell_type": "code",514 "execution_count": 23,515 "id": "9529eddc-87df-4270-85e7-7b477e659f44",516 "metadata": {},517 "outputs": [518 {519 "ename": "AttributeError",520 "evalue": "'function' object has no attribute 'run'",521 "output_type": "error",522 "traceback": [523 "\u001b[1;31m---------------------------------------------------------------------------\u001b[0m",524 "\u001b[1;31mAttributeError\u001b[0m Traceback (most recent call last)",525 "\u001b[1;32m~\\AppData\\Local\\Temp/ipykernel_9764/1028344225.py\u001b[0m in \u001b[0;36m<module>\u001b[1;34m\u001b[0m\n\u001b[1;32m----> 1\u001b[1;33m \u001b[0mpredict\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mrun\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mhost\u001b[0m\u001b[1;33m=\u001b[0m\u001b[1;34m\"0.0.0.0\"\u001b[0m\u001b[1;33m,\u001b[0m \u001b[0mport\u001b[0m\u001b[1;33m=\u001b[0m\u001b[1;36m8080\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m",526 "\u001b[1;31mAttributeError\u001b[0m: 'function' object has no attribute 'run'"527 ]528 }529 ],530 "source": [531 "predict.run(host=\"0.0.0.0\", port=8080)"532 ]533 },534 {535 "cell_type": "code",536 "execution_count": 38,537 "id": "241d6319-3cbe-4563-a505-1a9b4f429447",538 "metadata": {},539 "outputs": [],540 "source": [541 "#print('The model is ready')"542 ]543 },544 {545 "cell_type": "markdown",546 "id": "3dd5733a-c1fd-4a7b-90f8-5841c5bd62bb",547 "metadata": {},548 "source": [549 "### "550 ]551 },552 {553 "cell_type": "code",554 "execution_count": null,555 "id": "77df5896-f7f0-4dcc-9e47-7be5c7a131e3",556 "metadata": {},557 "outputs": [],558 "source": []559 }560 ],561 "metadata": {562 "kernelspec": {563 "display_name": "Python 3 (ipykernel)",564 "language": "python",565 "name": "python3"566 },567 "language_info": {568 "codemirror_mode": {569 "name": "ipython",570 "version": 3571 },572 "file_extension": ".py",573 "mimetype": "text/x-python",574 "name": "python",575 "nbconvert_exporter": "python",576 "pygments_lexer": "ipython3",577 "version": "3.9.7"578 }579 },580 "nbformat": 4,581 "nbformat_minor": 5582}583 