thamesh24/Agentic_Bot_GUI
0
1import streamlit as st2import os3import PyPDF24import logging5import nltk6nltk.data.path.append("/home/user/nltk_data")7import torch8import numpy as np9import random10from nltk.tokenize import sent_tokenize11from transformers import pipeline12from sentence_transformers import SentenceTransformer, util13from sklearn.feature_extraction.text import TfidfVectorizer14 15nltk.download('punkt')16nltk.download('punkt_tab')17print(nltk.data.find('tokenizers/punkt'))18 19logging.basicConfig(filename="support_bot_log.txt", level=logging.INFO,20 format="%(asctime)s - %(levelname)s - %(message)s", force=True)21 22logging.info("Logging initialized successfully.")23 24class SupportBotAgent:25 def __init__(self, document_path):26 logging.info("Initializing SupportBotAgent...")27 self.document_path = document_path28 self.document_text = self.load_document()29 30 logging.info("Loading sentence embedding model...")31 self.embedding_model = SentenceTransformer('all-MiniLM-L6-v2')32 33 logging.info("Loading question-answering model google/flan-t5-base")34 self.nlp_pipeline = pipeline("text2text-generation", model="google/flan-t5-base", tokenizer="google/flan-t5-base")35 36 logging.info("Processing document into structured text chunks...")37 self.paragraphs = self.process_document(self.document_text)38 logging.info(f"Document processed into {len(self.paragraphs)} chunks.")39 40 logging.info("Computing TF-IDF vectors for keyword-based retrieval...")41 self.tfidf_vectorizer = TfidfVectorizer(stop_words='english')42 self.tfidf_matrix = self.tfidf_vectorizer.fit_transform(self.paragraphs)43 44 logging.info("Computing sentence embeddings for semantic search...")45 self.embeddings = self.embedding_model.encode(self.paragraphs, convert_to_tensor=True)46 47 logging.info("SupportBotAgent successfully initialized.")48 49 def load_document(self):50 logging.info(f"Attempting to load document: {self.document_path}")51 file_extension = os.path.splitext(self.document_path)[1].lower()52 try:53 if file_extension == ".pdf":54 with open(self.document_path, "rb") as file:55 reader = PyPDF2.PdfReader(file)56 text = "\n".join([page.extract_text() for page in reader.pages if page.extract_text()])57 elif file_extension == ".txt":58 with open(self.document_path, "r", encoding="utf-8") as file:59 text = file.read()60 else:61 raise ValueError("Unsupported file format. Use PDF or TXT.")62 63 logging.info("Document successfully loaded and processed.")64 return text65 except Exception as e:66 logging.error(f"Error loading document: {e}")67 raise e68 69 def process_document(self, text):70 logging.info("Starting document chunking process...")71 sentences = sent_tokenize(text)72 chunks = []73 current_chunk = []74 max_chunk_length = 40075 sentence_embeddings = self.embedding_model.encode(sentences, convert_to_tensor=True)76 77 for i, sentence in enumerate(sentences):78 if len(" ".join(current_chunk)) + len(sentence) < max_chunk_length:79 if current_chunk:80 sim_score = util.pytorch_cos_sim(sentence_embeddings[i - 1], sentence_embeddings[i])[0].item()81 if sim_score < 0.3:82 chunks.append(" ".join(current_chunk))83 current_chunk = [sentence]84 else:85 current_chunk.append(sentence)86 else:87 current_chunk.append(sentence)88 else:89 chunks.append(" ".join(current_chunk))90 current_chunk = [sentence]91 92 if current_chunk:93 chunks.append(" ".join(current_chunk))94 95 logging.info(f"Document successfully chunked into {len(chunks)} sections.")96 return chunks97 98 def retrieve_relevant_section(self, query, top_n=3):99 logging.info(f"Retrieving relevant section for query: {query}")100 query_tfidf = self.tfidf_vectorizer.transform([query])101 tfidf_scores = np.dot(self.tfidf_matrix, query_tfidf.T).toarray().flatten()102 top_indices = np.argsort(tfidf_scores)[-top_n:][::-1]103 104 query_embedding = self.embedding_model.encode(query, convert_to_tensor=True)105 best_matches = []106 107 for idx in top_indices:108 similarity_score = util.pytorch_cos_sim(query_embedding, self.embeddings[idx])[0].item()109 best_matches.append((self.paragraphs[idx], similarity_score))110 111 best_matches = sorted(best_matches, key=lambda x: x[1], reverse=True)112 113 if best_matches and best_matches[0][1] >= 0.5:114 logging.info(f"Top relevant section found with similarity score: {best_matches[0][1]:.2f}")115 return best_matches[0][0]116 else:117 logging.warning("No relevant section found, returning fallback response.")118 return "I'm sorry, I couldn't find relevant details. Please contact support@example.com."119 120 def answer_query(self, query):121 logging.info(f"Processing query: {query}")122 relevant_section = self.retrieve_relevant_section(query)123 if "I'm sorry" in relevant_section:124 return f"Query: {query}\n\n{relevant_section}"125 126 prompt = f"Based on the following context, answer the question:\n\nContext: {relevant_section}\n\nQuestion: {query}"127 answer_data = self.nlp_pipeline(prompt, max_length=100, truncation=True)128 extracted_answer = answer_data[0]["generated_text"]129 return f"Query: {query}\nBot: {extracted_answer}\n\n"130 131# Streamlit UI Setup132st.title("Chatbot with Document Upload")133 134def save_uploaded_file(uploaded_file):135 file_path = os.path.join("temp", uploaded_file.name)136 os.makedirs("temp", exist_ok=True)137 with open(file_path, "wb") as f:138 f.write(uploaded_file.getbuffer())139 return file_path140 141uploaded_file = st.file_uploader("Upload a PDF or TXT file", type=["pdf", "txt"])142 143if uploaded_file:144 file_path = save_uploaded_file(uploaded_file)145 st.success(f"File uploaded: {uploaded_file.name}")146 bot = SupportBotAgent(document_path=file_path)147 st.session_state.bot = bot148 149if "bot" in st.session_state:150 query = st.text_input("Ask a question:")151 if st.button("Get Answer"):152 if not query:153 st.warning("Please enter a query.")154 else:155 response = st.session_state.bot.answer_query(query)156 st.write("**Bot Response:**", response)157 