Team Ai
Apppublic

HuzaifaTech/Multi_documents

sourceHugging Facemitupdated 5mo agoView on Hugging Face
0likes
app.py249 linesDownload Raw Back to root
1import os2import uuid3import chromadb4import gradio as gr5from pypdf import PdfReader6import docx7from sentence_transformers import SentenceTransformer8from groq import Groq9 10# =========================11# πŸ”‘ GROQ API (HF SECRET)12# =========================13# Set your secret as "GROQ_API_KEY" in HF Space Settings β†’ Variables and secrets14groq_client = Groq(api_key=os.getenv("Multi_doc"))15 16# =========================17# πŸ“„ LOAD DOCUMENTS18# =========================19def load_pdf(path):20    reader = PdfReader(path)21    return "\n".join([p.extract_text() or "" for p in reader.pages])22 23def load_docx(path):24    doc = docx.Document(path)25    return "\n".join([p.text for p in doc.paragraphs])26 27def load_txt(path):28    with open(path, "r", encoding="utf-8") as f:29        return f.read()30 31def load_document(path):32    ext = path.split(".")[-1].lower()33    if ext == "pdf":34        return load_pdf(path)35    if ext == "docx":36        return load_docx(path)37    if ext == "txt":38        return load_txt(path)39    raise ValueError(f"Unsupported file type: .{ext}")40 41# =========================42# βœ‚οΈ CHUNKING43# =========================44def chunk_text(text, size=400, overlap=80):45    words = text.split()46    chunks = []47    i = 048    cid = 049 50    while i < len(words):51        chunks.append({52            "id": cid,53            "text": " ".join(words[i:i + size])54        })55        i += size - overlap56        cid += 157 58    return chunks59 60# =========================61# 🧠 EMBEDDINGS (LOCAL)62# =========================63embed_model = SentenceTransformer("all-MiniLM-L6-v2")64 65def embed(texts):66    return embed_model.encode(texts, show_progress_bar=False).tolist()67 68# =========================69# πŸ—„οΈ CHROMA DB70# HF Spaces has a read-only root β€” use /tmp for writable storage71# =========================72chroma_client = chromadb.PersistentClient(path="/tmp/chroma_db")73collection = chroma_client.get_or_create_collection("rag")74 75# =========================76# πŸ“ PROCESS FILES77# =========================78def process_files(files):79    if not files:80        return "⚠️ No files uploaded."81 82    all_chunks = []83    errors = []84 85    for f in files:86        # Gradio on HF passes file path as a string or NamedString87        file_path = f if isinstance(f, str) else f.name88        if not file_path:89            continue90        try:91            text = load_document(file_path)92            if not text.strip():93                errors.append(f"⚠️ {os.path.basename(file_path)} appears empty.")94                continue95            chunks = chunk_text(text)96            for c in chunks:97                all_chunks.append({98                    "source": os.path.basename(file_path),99                    "text": c["text"]100                })101        except Exception as e:102            errors.append(f"❌ Error reading {os.path.basename(file_path)}: {e}")103 104    if not all_chunks:105        return "\n".join(errors) if errors else "⚠️ No content could be extracted."106 107    texts = [c["text"] for c in all_chunks]108    embeddings = embed(texts)109 110    collection.add(111        ids=[str(uuid.uuid4()) for _ in all_chunks],112        embeddings=embeddings,113        documents=texts,114        metadatas=[{"source": c["source"]} for c in all_chunks]115    )116 117    result = f"βœ… Indexed {len(files)} file(s) β€” {len(all_chunks)} chunks stored."118    if errors:119        result += "\n" + "\n".join(errors)120    return result121 122# =========================123# πŸ” RETRIEVAL124# =========================125def retrieve(query, k=3):126    # Guard: collection might be empty127    count = collection.count()128    if count == 0:129        return []130 131    k = min(k, count)  # Can't retrieve more than what's stored132    q_emb = embed([query])[0]133 134    results = collection.query(135        query_embeddings=[q_emb],136        n_results=k137    )138 139    docs = []140    for i in range(len(results["documents"][0])):141        docs.append({142            "text": results["documents"][0][i],143            "source": results["metadatas"][0][i]["source"]144        })145 146    return docs147 148# =========================149# πŸ€– GROQ GENERATION150# =========================151def generate(query):152    docs = retrieve(query)153 154    if not docs:155        return "⚠️ No documents indexed yet. Please upload and process files first."156 157    context = "\n\n".join(158        [f"[{d['source']}]\n{d['text']}" for d in docs]159    )160 161    prompt = f"""You are a strict RAG assistant.162Answer ONLY from the context below.163If the answer is not found in the context, say: "Not found in documents."164 165CONTEXT:166{context}167 168QUESTION:169{query}170 171ANSWER:"""172 173    try:174        response = groq_client.chat.completions.create(175            model="llama-3.1-8b-instant",176            messages=[{"role": "user", "content": prompt}],177            temperature=0.2,178            max_tokens=1024,179        )180        answer = response.choices[0].message.content181    except Exception as e:182        return f"❌ Groq API error: {e}"183 184    sources = "\n\n".join(185        [f"πŸ“„ **{d['source']}**\n{d['text'][:200]}…" for d in docs]186    )187 188    return f"{answer}\n\n---\nπŸ“š **Sources:**\n{sources}"189 190# =========================191# πŸ’¬ CHAT FUNCTION192# Gradio 5 uses {"role": ..., "content": ...} dicts, not tuples193# =========================194def chat(message, history):195    if not message.strip():196        return "", history197    reply = generate(message)198    history.append({"role": "user", "content": message})199    history.append({"role": "assistant", "content": reply})200    return "", history201 202# =========================203# 🎨 GRADIO UI204# =========================205with gr.Blocks(title="Groq RAG Assistant") as app:206 207    gr.Markdown(208        """# 🧠 Groq RAG Assistant209        Upload your documents, then ask questions about them.210        Powered by **Groq LLaMA3** + **ChromaDB** + **sentence-transformers**.211        """212    )213 214    with gr.Row():215 216        with gr.Column(scale=1):217            gr.Markdown("### πŸ“‚ Upload Documents")218            files = gr.File(219                file_count="multiple",220                file_types=[".pdf", ".docx", ".txt"],221                label="Upload PDF / DOCX / TXT"222            )223            process_btn = gr.Button("πŸš€ Process Files", variant="primary")224            status = gr.Textbox(label="Status", interactive=False)225 226            process_btn.click(fn=process_files, inputs=files, outputs=status)227 228        with gr.Column(scale=2):229            gr.Markdown("### πŸ’¬ Ask Your Documents")230            # Gradio 5: type="messages" uses the new dict format231            chatbot = gr.Chatbot(height=480, type="messages")232            msg = gr.Textbox(233                placeholder="Ask a question about your documents…",234                label="Your question",235                lines=2236            )237            with gr.Row():238                submit_btn = gr.Button("Send", variant="primary")239                clear_btn = gr.Button("Clear Chat")240 241            submit_btn.click(fn=chat, inputs=[msg, chatbot], outputs=[msg, chatbot])242            msg.submit(fn=chat, inputs=[msg, chatbot], outputs=[msg, chatbot])243            clear_btn.click(fn=lambda: ([], ""), outputs=[chatbot, msg])244 245# =========================246# πŸš€ LAUNCH247# =========================248if __name__ == "__main__":249    app.launch()