Team Ai
Apppublic

RemotelyBest/RemotelyBest_Development_of_AI_Applications

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
Final revision code.ipynb251 linesDownload Raw Back to Other Codes
1{2 "cells": [3  {4   "cell_type": "code",5   "execution_count": 14,6   "metadata": {},7   "outputs": [8    {9     "name": "stdout",10     "output_type": "stream",11     "text": [12      "* Running on local URL:  http://127.0.0.1:7870\n",13      "* Running on public URL: https://a94e18f722148a0463.gradio.live\n",14      "\n",15      "This share link expires in 72 hours. For free permanent hosting and GPU upgrades, run `gradio deploy` from the terminal in the working directory to deploy to Hugging Face Spaces (https://huggingface.co/spaces)\n"16     ]17    },18    {19     "data": {20      "text/html": [21       "<div><iframe src=\"https://a94e18f722148a0463.gradio.live\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"22      ],23      "text/plain": [24       "<IPython.core.display.HTML object>"25      ]26     },27     "metadata": {},28     "output_type": "display_data"29    },30    {31     "data": {32      "text/plain": []33     },34     "execution_count": 14,35     "metadata": {},36     "output_type": "execute_result"37    }38   ],39   "source": [40    "from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, AutoModelForSequenceClassification, TextClassificationPipeline\n",41    "import torch\n",42    "import gradio as gr\n",43    "from openpyxl import load_workbook\n",44    "from numpy import mean\n",45    "import pandas as pd\n",46    "import matplotlib.pyplot as plt\n",47    "\n",48    "theme = gr.themes.Soft(\n",49    "    primary_hue=\"amber\",\n",50    "    secondary_hue=\"amber\",\n",51    "    neutral_hue=\"stone\",\n",52    ")\n",53    "\n",54    "# Load tokenizers and models\n",55    "tokenizer = AutoTokenizer.from_pretrained(\"suriya7/bart-finetuned-text-summarization\")\n",56    "model = AutoModelForSeq2SeqLM.from_pretrained(\"suriya7/bart-finetuned-text-summarization\")\n",57    "\n",58    "tokenizer_keywords = AutoTokenizer.from_pretrained(\"transformer3/H2-keywordextractor\")\n",59    "model_keywords = AutoModelForSeq2SeqLM.from_pretrained(\"transformer3/H2-keywordextractor\")\n",60    "\n",61    "device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n",62    "new_model = AutoModelForSequenceClassification.from_pretrained('roberta-rating')\n",63    "new_tokenizer = AutoTokenizer.from_pretrained('roberta-rating')\n",64    "\n",65    "classifier = TextClassificationPipeline(model=new_model, tokenizer=new_tokenizer, device=device)\n",66    "\n",67    "label_mapping = {1: '1/5', 2: '2/5', 3: '3/5', 4: '4/5', 5: '5/5'}\n",68    "\n",69    "# Function to display and filter the Excel workbook\n",70    "def filter_xl(file, keywords):\n",71    "    # Load the workbook and convert it to a DataFrame\n",72    "    workbook = load_workbook(filename=file)\n",73    "    sheet = workbook.active\n",74    "    data = sheet.values\n",75    "    columns = next(data)[0:]\n",76    "    df = pd.DataFrame(data, columns=columns)\n",77    "    \n",78    "    if keywords:\n",79    "        keyword_list = keywords.split(',')\n",80    "        for keyword in keyword_list:\n",81    "            df = df[df.apply(lambda row: row.astype(str).str.contains(keyword.strip(), case=False).any(), axis=1)]\n",82    "    \n",83    "    return df\n",84    "\n",85    "# Function to calculate overall rating from filtered data\n",86    "def calculate_rating(filtered_df):\n",87    "    reviews = filtered_df.to_numpy().flatten()\n",88    "    ratings = []\n",89    "    for review in reviews:\n",90    "        if pd.notna(review):\n",91    "            rating = int(classifier(review)[0]['label'].split('_')[1])\n",92    "            ratings.append(rating)\n",93    "    \n",94    "    return round(mean(ratings), 2), ratings\n",95    "\n",96    "# Function to calculate results including summary, keywords, and sentiment\n",97    "def calculate_results(file, keywords):\n",98    "    filtered_df = filter_xl(file, keywords)\n",99    "    overall_rating, ratings = calculate_rating(filtered_df)\n",100    "    \n",101    "    # Summarize and extract keywords from the filtered reviews\n",102    "    text = \" \".join(filtered_df.to_numpy().flatten())\n",103    "    inputs = tokenizer([text], max_length=1024, truncation=True, return_tensors=\"pt\")\n",104    "    summary_ids = model.generate(inputs[\"input_ids\"], num_beams=2, min_length=10, max_length=50)\n",105    "    summary = tokenizer.batch_decode(summary_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]\n",106    "    summary = summary.replace(\"I\", \"They\").replace(\"my\", \"their\").replace(\"me\", \"them\")\n",107    "\n",108    "    inputs_keywords = tokenizer_keywords([text], max_length=1024, truncation=True, return_tensors=\"pt\")\n",109    "    summary_ids_keywords = model_keywords.generate(inputs_keywords[\"input_ids\"], num_beams=2, min_length=0, max_length=100)\n",110    "    keywords = tokenizer_keywords.batch_decode(summary_ids_keywords, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]\n",111    "\n",112    "    # Determine overall sentiment\n",113    "    sentiments = []\n",114    "    for review in filtered_df.to_numpy().flatten():\n",115    "        if pd.notna(review):\n",116    "            sentiment = classifier(review)[0]['label']\n",117    "            sentiment_label = \"Positive\" if sentiment == \"LABEL_4\" or sentiment == \"LABEL_5\" else \"Negative\" if sentiment == \"LABEL_1\" or sentiment == \"LABEL_2\" else \"Neutral\"\n",118    "            sentiments.append(sentiment_label)\n",119    "    \n",120    "    overall_sentiment = \"Positive\" if sentiments.count(\"Positive\") > sentiments.count(\"Negative\") else \"Negative\" if sentiments.count(\"Negative\") > sentiments.count(\"Positive\") else \"Neutral\"\n",121    "\n",122    "    return overall_rating, summary, keywords, overall_sentiment, ratings, sentiments\n",123    "\n",124    "# Function to analyze a single review\n",125    "def analyze_review(review):\n",126    "    if not review.strip():\n",127    "        return \"Error: No text provided\", \"Error: No text provided\", \"Error: No text provided\", \"Error: No text provided\"\n",128    "    \n",129    "    # Calculate rating\n",130    "    rating = int(classifier(review)[0]['label'].split('_')[1])\n",131    "    \n",132    "    # Summarize review\n",133    "    inputs = tokenizer([review], max_length=1024, truncation=True, return_tensors=\"pt\")\n",134    "    summary_ids = model.generate(inputs[\"input_ids\"], num_beams=2, min_length=10, max_length=50)\n",135    "    summary = tokenizer.batch_decode(summary_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]\n",136    "    summary = summary.replace(\"I\", \"he/she\").replace(\"my\", \"his/her\").replace(\"me\", \"him/her\")\n",137    "\n",138    "    # Extract keywords\n",139    "    inputs_keywords = tokenizer_keywords([review], max_length=1024, truncation=True, return_tensors=\"pt\")\n",140    "    summary_ids_keywords = model_keywords.generate(inputs_keywords[\"input_ids\"], num_beams=2, min_length=0, max_length=100)\n",141    "    keywords = tokenizer_keywords.batch_decode(summary_ids_keywords, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]\n",142    "\n",143    "    # Determine sentiment\n",144    "    sentiment = classifier(review)[0]['label']\n",145    "    sentiment_label = \"Positive\" if sentiment == \"LABEL_4\" or sentiment == \"LABEL_5\" else \"Negative\" if sentiment == \"LABEL_1\" or sentiment == \"LABEL_2\" else \"Neutral\"\n",146    "\n",147    "    return rating, summary, keywords, sentiment_label\n",148    "\n",149    "# Function to count rows in the filtered DataFrame\n",150    "def count_rows(filtered_df):\n",151    "    return len(filtered_df)\n",152    "\n",153    "# Function to plot ratings\n",154    "def plot_ratings(ratings):\n",155    "    plt.figure(figsize=(10, 5))\n",156    "    plt.hist(ratings, bins=range(1, 7), edgecolor='black', align='left')\n",157    "    plt.xlabel('Rating')\n",158    "    plt.ylabel('Frequency')\n",159    "    plt.title('Distribution of Ratings')\n",160    "    plt.xticks(range(1, 6))\n",161    "    plt.grid(True)\n",162    "    plt.savefig('ratings_distribution.png')\n",163    "    return 'ratings_distribution.png'\n",164    "\n",165    "# Function to plot sentiments\n",166    "def plot_sentiments(sentiments):\n",167    "    sentiment_counts = pd.Series(sentiments).value_counts()\n",168    "    plt.figure(figsize=(10, 5))\n",169    "    sentiment_counts.plot(kind='bar', color=['green', 'red', 'blue'])\n",170    "    plt.xlabel('Sentiment')\n",171    "    plt.ylabel('Frequency')\n",172    "    plt.title('Distribution of Sentiments')\n",173    "    plt.grid(True)\n",174    "    plt.savefig('sentiments_distribution.png')\n",175    "    return 'sentiments_distribution.png'\n",176    "\n",177    "# Gradio interface\n",178    "with gr.Blocks(theme=theme) as demo:\n",179    "    gr.Markdown(\"<h1 style='text-align: center;'>Feedback and Auditing Survey AI Analyzer</h1><br>\")\n",180    "    with gr.Tabs():\n",181    "        with gr.TabItem(\"Upload and Filter\"):\n",182    "            with gr.Row():\n",183    "                with gr.Column(scale=1):\n",184    "                    excel_file = gr.File(label=\"Upload Excel File\")\n",185    "                    #excel_file = gr.File(label=\"Upload Excel File\", file_types=[\".xlsx\", \".xlsm\", \".xltx\", \".xltm\"])\n",186    "                    keywords_input = gr.Textbox(label=\"Filter by Keywords (comma-separated)\")\n",187    "                    display_button = gr.Button(\"Display and Filter Excel Data\")\n",188    "                    clear_button_upload = gr.Button(\"Clear\")\n",189    "                    row_count = gr.Textbox(label=\"Number of Rows\", interactive=False)\n",190    "                with gr.Column(scale=3):\n",191    "                    filtered_data = gr.Dataframe(label=\"Filtered Excel Contents\")\n",192    "        \n",193    "        with gr.TabItem(\"Calculate Results\"):\n",194    "            with gr.Row():\n",195    "                with gr.Column():\n",196    "                    overall_rating = gr.Textbox(label=\"Overall Rating\")\n",197    "                    summary = gr.Textbox(label=\"Summary\")\n",198    "                    keywords_output = gr.Textbox(label=\"Keywords\")\n",199    "                    overall_sentiment = gr.Textbox(label=\"Overall Sentiment\")\n",200    "                    calculate_button = gr.Button(\"Calculate Results\")\n",201    "                with gr.Column():\n",202    "                    ratings_graph = gr.Image(label=\"Ratings Distribution\")\n",203    "                    sentiments_graph = gr.Image(label=\"Sentiments Distribution\")\n",204    "                    calculate_graph_button = gr.Button(\"Calculate Graph Results\")\n",205    "        \n",206    "        with gr.TabItem(\"Testing Area / Write a Review\"):\n",207    "            with gr.Row():\n",208    "                with gr.Column(scale=2):\n",209    "                    review_input = gr.Textbox(label=\"Write your review here\")\n",210    "                    analyze_button = gr.Button(\"Analyze Review\")\n",211    "                    clear_button_review = gr.Button(\"Clear\")\n",212    "                with gr.Column(scale=2):\n",213    "                    review_rating = gr.Textbox(label=\"Rating\")\n",214    "                    review_summary = gr.Textbox(label=\"Summary\")\n",215    "                    review_keywords = gr.Textbox(label=\"Keywords\")\n",216    "                    review_sentiment = gr.Textbox(label=\"Sentiment\")\n",217    "\n",218    "    display_button.click(lambda file, keywords: (filter_xl(file, keywords), count_rows(filter_xl(file, keywords))), inputs=[excel_file, keywords_input], outputs=[filtered_data, row_count])\n",219    "    calculate_graph_button.click(lambda file, keywords: (*calculate_results(file, keywords)[:4], plot_ratings(calculate_results(file, keywords)[4]), plot_sentiments(calculate_results(file, keywords)[5])), inputs=[excel_file, keywords_input], outputs=[overall_rating, summary, keywords_output, overall_sentiment, ratings_graph, sentiments_graph])\n",220    "    calculate_button.click(lambda file, keywords: (*calculate_results(file, keywords)[:4], plot_ratings(calculate_results(file, keywords)[4])), inputs=[excel_file, keywords_input], outputs=[overall_rating, summary, keywords_output, overall_sentiment])\n",221    "    analyze_button.click(analyze_review, inputs=review_input, outputs=[review_rating, review_summary, review_keywords, review_sentiment])\n",222    "    clear_button_upload.click(lambda: (\"\"), outputs=[keywords_input])\n",223    "    clear_button_review.click(lambda: (\"\", \"\", \"\", \"\", \"\"), outputs=[review_input, review_rating, review_summary, review_keywords, review_sentiment])\n",224    "\n",225    "demo.launch(share=True)"226   ]227  }228 ],229 "metadata": {230  "kernelspec": {231   "display_name": "SolutionsInPR",232   "language": "python",233   "name": "python3"234  },235  "language_info": {236   "codemirror_mode": {237    "name": "ipython",238    "version": 3239   },240   "file_extension": ".py",241   "mimetype": "text/x-python",242   "name": "python",243   "nbconvert_exporter": "python",244   "pygments_lexer": "ipython3",245   "version": "3.12.4"246  }247 },248 "nbformat": 4,249 "nbformat_minor": 2250}251