Team Ai
Apppublic

thamesh24/Agentic_Bot_GUI

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
app.py157 linesDownload Raw Back to root
1import streamlit as st2import os3import PyPDF24import logging5import nltk6nltk.data.path.append("/home/user/nltk_data")7import torch8import numpy as np9import random10from nltk.tokenize import sent_tokenize11from transformers import pipeline12from sentence_transformers import SentenceTransformer, util13from sklearn.feature_extraction.text import TfidfVectorizer14 15nltk.download('punkt')16nltk.download('punkt_tab')17print(nltk.data.find('tokenizers/punkt'))18 19logging.basicConfig(filename="support_bot_log.txt", level=logging.INFO,20                    format="%(asctime)s - %(levelname)s - %(message)s", force=True)21 22logging.info("Logging initialized successfully.")23 24class SupportBotAgent:25    def __init__(self, document_path):26        logging.info("Initializing SupportBotAgent...")27        self.document_path = document_path28        self.document_text = self.load_document()29 30        logging.info("Loading sentence embedding model...")31        self.embedding_model = SentenceTransformer('all-MiniLM-L6-v2')32 33        logging.info("Loading question-answering model google/flan-t5-base")34        self.nlp_pipeline = pipeline("text2text-generation", model="google/flan-t5-base", tokenizer="google/flan-t5-base")35 36        logging.info("Processing document into structured text chunks...")37        self.paragraphs = self.process_document(self.document_text)38        logging.info(f"Document processed into {len(self.paragraphs)} chunks.")39 40        logging.info("Computing TF-IDF vectors for keyword-based retrieval...")41        self.tfidf_vectorizer = TfidfVectorizer(stop_words='english')42        self.tfidf_matrix = self.tfidf_vectorizer.fit_transform(self.paragraphs)43 44        logging.info("Computing sentence embeddings for semantic search...")45        self.embeddings = self.embedding_model.encode(self.paragraphs, convert_to_tensor=True)46 47        logging.info("SupportBotAgent successfully initialized.")48 49    def load_document(self):50        logging.info(f"Attempting to load document: {self.document_path}")51        file_extension = os.path.splitext(self.document_path)[1].lower()52        try:53            if file_extension == ".pdf":54                with open(self.document_path, "rb") as file:55                    reader = PyPDF2.PdfReader(file)56                    text = "\n".join([page.extract_text() for page in reader.pages if page.extract_text()])57            elif file_extension == ".txt":58                with open(self.document_path, "r", encoding="utf-8") as file:59                    text = file.read()60            else:61                raise ValueError("Unsupported file format. Use PDF or TXT.")62 63            logging.info("Document successfully loaded and processed.")64            return text65        except Exception as e:66            logging.error(f"Error loading document: {e}")67            raise e68 69    def process_document(self, text):70        logging.info("Starting document chunking process...")71        sentences = sent_tokenize(text)72        chunks = []73        current_chunk = []74        max_chunk_length = 40075        sentence_embeddings = self.embedding_model.encode(sentences, convert_to_tensor=True)76 77        for i, sentence in enumerate(sentences):78            if len(" ".join(current_chunk)) + len(sentence) < max_chunk_length:79                if current_chunk:80                    sim_score = util.pytorch_cos_sim(sentence_embeddings[i - 1], sentence_embeddings[i])[0].item()81                    if sim_score < 0.3:82                        chunks.append(" ".join(current_chunk))83                        current_chunk = [sentence]84                    else:85                        current_chunk.append(sentence)86                else:87                    current_chunk.append(sentence)88            else:89                chunks.append(" ".join(current_chunk))90                current_chunk = [sentence]91 92        if current_chunk:93            chunks.append(" ".join(current_chunk))94 95        logging.info(f"Document successfully chunked into {len(chunks)} sections.")96        return chunks97 98    def retrieve_relevant_section(self, query, top_n=3):99        logging.info(f"Retrieving relevant section for query: {query}")100        query_tfidf = self.tfidf_vectorizer.transform([query])101        tfidf_scores = np.dot(self.tfidf_matrix, query_tfidf.T).toarray().flatten()102        top_indices = np.argsort(tfidf_scores)[-top_n:][::-1]103 104        query_embedding = self.embedding_model.encode(query, convert_to_tensor=True)105        best_matches = []106 107        for idx in top_indices:108            similarity_score = util.pytorch_cos_sim(query_embedding, self.embeddings[idx])[0].item()109            best_matches.append((self.paragraphs[idx], similarity_score))110 111        best_matches = sorted(best_matches, key=lambda x: x[1], reverse=True)112 113        if best_matches and best_matches[0][1] >= 0.5:114            logging.info(f"Top relevant section found with similarity score: {best_matches[0][1]:.2f}")115            return best_matches[0][0]116        else:117            logging.warning("No relevant section found, returning fallback response.")118            return "I'm sorry, I couldn't find relevant details. Please contact support@example.com."119 120    def answer_query(self, query):121        logging.info(f"Processing query: {query}")122        relevant_section = self.retrieve_relevant_section(query)123        if "I'm sorry" in relevant_section:124            return f"Query: {query}\n\n{relevant_section}"125 126        prompt = f"Based on the following context, answer the question:\n\nContext: {relevant_section}\n\nQuestion: {query}"127        answer_data = self.nlp_pipeline(prompt, max_length=100, truncation=True)128        extracted_answer = answer_data[0]["generated_text"]129        return f"Query: {query}\nBot: {extracted_answer}\n\n"130 131# Streamlit UI Setup132st.title("Chatbot with Document Upload")133 134def save_uploaded_file(uploaded_file):135    file_path = os.path.join("temp", uploaded_file.name)136    os.makedirs("temp", exist_ok=True)137    with open(file_path, "wb") as f:138        f.write(uploaded_file.getbuffer())139    return file_path140 141uploaded_file = st.file_uploader("Upload a PDF or TXT file", type=["pdf", "txt"])142 143if uploaded_file:144    file_path = save_uploaded_file(uploaded_file)145    st.success(f"File uploaded: {uploaded_file.name}")146    bot = SupportBotAgent(document_path=file_path)147    st.session_state.bot = bot148 149if "bot" in st.session_state:150    query = st.text_input("Ask a question:")151    if st.button("Get Answer"):152        if not query:153            st.warning("Please enter a query.")154        else:155            response = st.session_state.bot.answer_query(query)156            st.write("**Bot Response:**", response)157