Team Ai
Apppublic

shivam701171/Invoice_processing_tool

sourceHugging Facemitupdated 1y agoView on Hugging Face
0likes
app.py3259 linesDownload Raw Back to root
1#!/usr/bin/env python32"""3AI Invoice Processing System - Complete Single File for Hugging Face Spaces4A comprehensive system with AI-powered extraction, semantic search, and analytics.5 6Author: AI Assistant7Date: 20248Version: HuggingFace Single File v1.09"""10 11# ===============================================================================12# IMPORTS AND COMPATIBILITY CHECKS13# ===============================================================================14 15import os16import json17import re18import tempfile19import shutil20import pickle21import numpy as np22from datetime import datetime23from typing import Dict, List, Optional, Tuple24from dataclasses import dataclass25from pathlib import Path26import time27import logging28import uuid29 30# Check if running on Hugging Face Spaces31IS_HF_SPACE = os.getenv("SPACE_ID") is not None32 33# Get Hugging Face token from environment or Streamlit secrets34HF_TOKEN = None35try:36    # Try Streamlit secrets first (for HF Spaces)37    HF_TOKEN = st.secrets.get("HF_TOKEN", None)38except:39    # Fall back to environment variable40    HF_TOKEN = os.getenv("HF_TOKEN", None)41 42# Streamlit and core libraries43import streamlit as st44import sqlite345import pandas as pd46import plotly.express as px47import plotly.graph_objects as go48import requests49 50# Vector storage and embeddings (with fallbacks)51try:52    import faiss53    FAISS_AVAILABLE = True54except ImportError:55    FAISS_AVAILABLE = False56    st.warning("⚠️ FAISS not available. Vector search will be disabled.")57 58try:59    from sentence_transformers import SentenceTransformer60    SENTENCE_TRANSFORMERS_AVAILABLE = True61except ImportError:62    SENTENCE_TRANSFORMERS_AVAILABLE = False63    st.warning("⚠️ Sentence Transformers not available. Using fallback methods.")64 65try:66    import torch67    TORCH_AVAILABLE = True68except ImportError:69    TORCH_AVAILABLE = False70 71# Document processing (simplified for HF)72try:73    import pdfplumber74    PDF_PROCESSING_AVAILABLE = True75    PDF_PROCESSOR = "pdfplumber"76except ImportError:77    try:78        import PyPDF279        PDF_PROCESSING_AVAILABLE = True80        PDF_PROCESSOR = "PyPDF2"81    except ImportError:82        PDF_PROCESSING_AVAILABLE = False83        PDF_PROCESSOR = None84 85# ===============================================================================86# STREAMLIT CONFIGURATION87# ===============================================================================88 89st.set_page_config(90    page_title="AI Invoice Processing System",91    page_icon="📄",92    layout="wide",93    initial_sidebar_state="expanded",94    menu_items={95        'Get Help': 'https://huggingface.co/spaces',96        'Report a bug': 'https://huggingface.co/spaces',97        'About': """98        # AI Invoice Processing System99        Built for Hugging Face Spaces with AI-powered extraction and semantic search.100        """101    }102)103 104# ===============================================================================105# CONFIGURATION106# ===============================================================================107 108HF_CONFIG = {109    "max_file_size_mb": 10,110    "max_concurrent_files": 3,111    "timeout_seconds": 30,112    "use_cpu_only": True,113    "embedding_model": "all-MiniLM-L6-v2",114    "cache_dir": "./cache",115    "data_dir": "./data",116    "enable_ollama": False,117}118 119# Create necessary directories120os.makedirs(HF_CONFIG["cache_dir"], exist_ok=True)121os.makedirs(HF_CONFIG["data_dir"], exist_ok=True)122 123# ===============================================================================124# DATA STRUCTURES125# ===============================================================================126 127@dataclass128class InvoiceData:129    """Data structure for extracted invoice information"""130    supplier_name: str = ""131    buyer_name: str = ""132    invoice_number: str = ""133    date: str = ""134    amount: float = 0.0135    quantity: int = 0136    product_description: str = ""137    file_path: str = ""138    extraction_confidence: float = 0.0139    processing_method: str = "regex"140 141@dataclass142class VectorSearchResult:143    """Data structure for vector search results"""144    invoice_id: str145    invoice_number: str146    supplier_name: str147    similarity_score: float148    content_preview: str149    metadata: Dict150 151# ===============================================================================152# DOCUMENT PROCESSING CLASSES153# ===============================================================================154 155class DocumentProcessor:156    """Simplified document processor for Hugging Face Spaces"""157    158    def __init__(self):159        self.setup_processors()160    161    def setup_processors(self):162        """Setup available document processors"""163        self.processors = {}164        165        # PDF processing166        if PDF_PROCESSING_AVAILABLE:167            if PDF_PROCESSOR == "pdfplumber":168                self.processors['pdf'] = self.extract_with_pdfplumber169                st.success("✅ PDF processing available (pdfplumber)")170            elif PDF_PROCESSOR == "PyPDF2":171                self.processors['pdf'] = self.extract_with_pypdf2172                st.success("✅ PDF processing available (PyPDF2)")173        else:174            st.warning("⚠️ No PDF processor available")175        176        # Text files177        self.processors['txt'] = self.extract_text_file178    179    def extract_with_pdfplumber(self, file_path: str) -> str:180        """Extract text using pdfplumber"""181        try:182            import pdfplumber183            text = ""184            with pdfplumber.open(file_path) as pdf:185                for page in pdf.pages:186                    page_text = page.extract_text()187                    if page_text:188                        text += page_text + "\n"189            return text190        except Exception as e:191            st.error(f"PDF extraction failed: {e}")192            return ""193    194    def extract_with_pypdf2(self, file_path: str) -> str:195        """Extract text using PyPDF2"""196        try:197            import PyPDF2198            text = ""199            with open(file_path, 'rb') as file:200                pdf_reader = PyPDF2.PdfReader(file)201                for page in pdf_reader.pages:202                    text += page.extract_text() + "\n"203            return text204        except Exception as e:205            st.error(f"PDF extraction failed: {e}")206            return ""207    208    def extract_text_file(self, file_path: str) -> str:209        """Extract text from text files"""210        try:211            with open(file_path, 'r', encoding='utf-8') as f:212                return f.read()213        except Exception as e:214            st.error(f"Text file extraction failed: {e}")215            return ""216    217    def extract_text_from_document(self, file_path: str) -> str:218        """Extract text from document based on file type"""219        file_ext = Path(file_path).suffix.lower()220        221        if file_ext == '.pdf':222            processor = self.processors.get('pdf')223        elif file_ext == '.txt':224            processor = self.processors.get('txt')225        else:226            st.warning(f"Unsupported file type: {file_ext}")227            return ""228        229        if processor:230            return processor(file_path)231        else:232            st.error(f"No processor available for {file_ext}")233            return ""234 235# ===============================================================================236# AI EXTRACTION CLASS237# ===============================================================================238 239class AIExtractor:240    """AI extraction for Hugging Face Spaces with Mistral 7B support"""241    242    def __init__(self):243        self.use_mistral = self.setup_mistral()244        self.use_transformers = self.setup_transformers() if not self.use_mistral else False245    246    def setup_mistral(self):247        """Try to setup Mistral 7B model with proper authentication"""248        try:249            # Check if we have HF token250            if not HF_TOKEN:251                st.warning("⚠️ Hugging Face token not found. Add HF_TOKEN to secrets for Mistral access.")252                return False253            254            # Check if we're in a high-resource environment255            import psutil256            memory_gb = psutil.virtual_memory().total / (1024**3)257            258            if memory_gb < 8:259                st.warning("⚠️ Insufficient memory for Mistral 7B. Using lighter models.")260                return False261            262            from transformers import AutoModelForCausalLM, AutoTokenizer, pipeline263            from huggingface_hub import login264            265            # Login with HF token266            login(token=HF_TOKEN)267            268            with st.spinner("🔄 Loading Mistral 7B model (this may take a few minutes)..."):269                # Use the instruction-tuned model270                model_name = "mistralai/Mistral-7B-Instruct-v0.1"271                272                # Load with reduced precision for memory efficiency273                self.mistral_tokenizer = AutoTokenizer.from_pretrained(274                    model_name,275                    cache_dir=HF_CONFIG["cache_dir"],276                    token=HF_TOKEN277                )278                279                self.mistral_model = AutoModelForCausalLM.from_pretrained(280                    model_name,281                    torch_dtype=torch.float16 if TORCH_AVAILABLE else None,282                    device_map="auto" if TORCH_AVAILABLE else None,283                    load_in_8bit=True,  # Use 8-bit quantization284                    cache_dir=HF_CONFIG["cache_dir"],285                    token=HF_TOKEN286                )287                288                # Create pipeline289                self.mistral_pipeline = pipeline(290                    "text-generation",291                    model=self.mistral_model,292                    tokenizer=self.mistral_tokenizer,293                    torch_dtype=torch.float16 if TORCH_AVAILABLE else None,294                    device_map="auto" if TORCH_AVAILABLE else None295                )296            297            st.success("✅ Mistral 7B model loaded successfully!")298            return True299            300        except ImportError as e:301            st.warning(f"⚠️ Missing dependencies for Mistral 7B: {e}")302            return False303        except Exception as e:304            st.warning(f"⚠️ Mistral 7B not available: {e}")305            st.info("💡 To use Mistral 7B: Add your Hugging Face token to secrets as 'HF_TOKEN'")306            return False307    308    def setup_transformers(self):309        """Fallback to lighter NER model"""310        try:311            from transformers import pipeline312            313            with st.spinner("Loading fallback AI model..."):314                self.ner_pipeline = pipeline(315                    "ner", 316                    model="dbmdz/bert-large-cased-finetuned-conll03-english",317                    aggregation_strategy="simple"318                )319            320            st.success("✅ Fallback AI extraction model loaded")321            return True322            323        except Exception as e:324            st.warning(f"⚠️ AI extraction not available: {e}")325            return False326    327    def extract_with_mistral(self, text: str) -> InvoiceData:328        """Extract invoice data using Mistral 7B"""329        try:330            # Create a detailed prompt for Mistral331            prompt = f"""<s>[INST] You are an expert at extracting structured information from invoices. 332 333Extract the following information from this invoice text and respond ONLY with valid JSON:334 335{{336    "invoice_number": "invoice or bill number",337    "supplier_name": "company providing goods/services",338    "buyer_name": "company receiving goods/services",339    "date": "date in YYYY-MM-DD format",340    "amount": "total amount as number only",341    "quantity": "total quantity as integer",342    "product_description": "brief description of items/services"343}}344 345Invoice text:346{text[:2000]}347 348Respond with JSON only: [/INST]"""349 350            # Generate response351            response = self.mistral_pipeline(352                prompt,353                max_new_tokens=300,354                temperature=0.1,355                do_sample=True,356                pad_token_id=self.mistral_tokenizer.eos_token_id357            )358            359            # Extract the generated text360            generated_text = response[0]['generated_text']361            362            # Find JSON in the response363            json_start = generated_text.find('{')364            json_end = generated_text.rfind('}') + 1365            366            if json_start != -1 and json_end > json_start:367                json_str = generated_text[json_start:json_end]368                369                # Parse JSON370                import json371                data = json.loads(json_str)372                373                # Create InvoiceData object374                invoice_data = InvoiceData()375                invoice_data.supplier_name = str(data.get('supplier_name', '')).strip()376                invoice_data.buyer_name = str(data.get('buyer_name', '')).strip()377                invoice_data.invoice_number = str(data.get('invoice_number', '')).strip()378                invoice_data.date = self.parse_date(str(data.get('date', '')))379                380                # Parse amount381                try:382                    amount_val = data.get('amount', 0)383                    if isinstance(amount_val, str):384                        amount_clean = re.sub(r'[^\d.]', '', amount_val)385                        invoice_data.amount = float(amount_clean) if amount_clean else 0.0386                    else:387                        invoice_data.amount = float(amount_val)388                except:389                    invoice_data.amount = 0.0390                391                # Parse quantity392                try:393                    qty_val = data.get('quantity', 0)394                    invoice_data.quantity = int(float(str(qty_val).replace(',', '')))395                except:396                    invoice_data.quantity = 0397                398                invoice_data.product_description = str(data.get('product_description', '')).strip()399                invoice_data.extraction_confidence = 0.95  # High confidence for Mistral400                invoice_data.processing_method = "mistral_7b"401                402                return invoice_data403            else:404                st.warning("⚠️ Mistral response didn't contain valid JSON, falling back to regex")405                return self.extract_with_regex(text)406                407        except Exception as e:408            st.error(f"Mistral extraction failed: {e}")409            return self.extract_with_regex(text)410    411    def extract_with_ai(self, text: str) -> InvoiceData:412        """Extract invoice data using available AI method"""413        if self.use_mistral:414            st.info("🤖 Using Mistral 7B for extraction...")415            return self.extract_with_mistral(text)416        elif self.use_transformers:417            st.info("🤖 Using NER model for extraction...")418            return self.extract_with_ner(text)419        else:420            st.info("🔧 Using regex extraction...")421            return self.extract_with_regex(text)422    423    def extract_with_ner(self, text: str) -> InvoiceData:424        """Extract using NER model (fallback method)"""425        try:426            # Use NER to extract entities427            entities = self.ner_pipeline(text[:512])  # Limit text length428            429            invoice_data = InvoiceData()430            invoice_data.processing_method = "ai_ner"431            432            # Extract specific entities433            for entity in entities:434                entity_text = entity['word'].replace('##', '')435                436                if entity['entity_group'] == 'ORG':437                    if not invoice_data.supplier_name:438                        invoice_data.supplier_name = entity_text439                    elif not invoice_data.buyer_name:440                        invoice_data.buyer_name = entity_text441                442                elif entity['entity_group'] == 'MISC':443                    if not invoice_data.invoice_number and any(c.isdigit() for c in entity_text):444                        invoice_data.invoice_number = entity_text445            446            # Fall back to regex for missing fields447            regex_data = self.extract_with_regex(text)448            449            # Combine results450            if not invoice_data.invoice_number:451                invoice_data.invoice_number = regex_data.invoice_number452            if not invoice_data.amount:453                invoice_data.amount = regex_data.amount454            if not invoice_data.date:455                invoice_data.date = regex_data.date456            if not invoice_data.quantity:457                invoice_data.quantity = regex_data.quantity458            459            invoice_data.extraction_confidence = 0.8460            461            return invoice_data462            463        except Exception as e:464            st.error(f"NER extraction failed: {e}")465            return self.extract_with_regex(text)466    467    def extract_with_regex(self, text: str) -> InvoiceData:468        """Enhanced regex extraction with better amount detection"""469        invoice_data = InvoiceData()470        invoice_data.processing_method = "regex"471        472        # Enhanced regex patterns with more comprehensive matching473        patterns = {474            'invoice_number': [475                r'invoice\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',476                r'bill\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',477                r'inv\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',478                r'ref\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',479                r'#\s*([A-Z0-9\-_/]{3,})',480                r'(?:^|\s)([A-Z]{2,}\d{3,}|\d{3,}[A-Z]{2,})',  # Common patterns like ABC123 or 123ABC481            ],482            'amount': [483                # Currency symbols with amounts484                r'total\s*(?:amount)?\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',485                r'amount\s*(?:due|paid|total)?\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',486                r'grand\s*total\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',487                r'net\s*(?:amount|total)\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',488                r'sub\s*total\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',489                490                # Currency symbols at the beginning491                r'[\$₹£€]\s*([0-9,]+\.?\d*)',492                493                # Amounts at end of lines (common in invoices)494                r'([0-9,]+\.?\d*)\s*[\$₹£€]?\s*495    496    def parse_date(self, date_str: str) -> str:497        """Parse date to YYYY-MM-DD format"""498        if not date_str:499            return ""500        501        formats = ['%Y-%m-%d', '%m/%d/%Y', '%d/%m/%Y', '%m-%d-%Y', '%d-%m-%Y', '%Y/%m/%d']502        503        for fmt in formats:504            try:505                parsed_date = datetime.strptime(date_str, fmt)506                return parsed_date.strftime('%Y-%m-%d')507            except ValueError:508                continue509        510        return date_str511 512# ===============================================================================513# VECTOR STORE CLASS514# ===============================================================================515 516class VectorStore:517    """Simplified vector store for Hugging Face Spaces"""518    519    def __init__(self, embedding_model: str = "all-MiniLM-L6-v2"):520        self.embedding_model_name = embedding_model521        self.vector_store_path = os.path.join(HF_CONFIG["data_dir"], "vectors.pkl")522        self.metadata_path = os.path.join(HF_CONFIG["data_dir"], "metadata.pkl")523        self.embedding_model = None524        self.vectors = []525        self.document_metadata = []526        self.embedding_dimension = None527        528        self.setup_embedding_model()529        self.load_vector_store()530    531    def setup_embedding_model(self):532        """Initialize the sentence transformer model"""533        if not SENTENCE_TRANSFORMERS_AVAILABLE:534            st.warning("⚠️ Sentence Transformers not available. Vector search disabled.")535            return536        537        try:538            with st.spinner(f"Loading embedding model: {self.embedding_model_name}..."):539                self.embedding_model = SentenceTransformer(540                    self.embedding_model_name,541                    cache_folder=HF_CONFIG["cache_dir"]542                )543                544                # Get embedding dimension545                test_embedding = self.embedding_model.encode(["test"])546                self.embedding_dimension = test_embedding.shape[0]547                548                st.success(f"✅ Embedding model loaded: {self.embedding_model_name}")549                550        except Exception as e:551            st.error(f"❌ Failed to load embedding model: {e}")552            self.embedding_model = None553    554    def load_vector_store(self):555        """Load existing vector store"""556        try:557            if os.path.exists(self.vector_store_path) and os.path.exists(self.metadata_path):558                with open(self.vector_store_path, 'rb') as f:559                    self.vectors = pickle.load(f)560                561                with open(self.metadata_path, 'rb') as f:562                    self.document_metadata = pickle.load(f)563                564                st.success(f"✅ Vector store loaded: {len(self.document_metadata)} documents")565            else:566                self.vectors = []567                self.document_metadata = []568                st.info("📄 New vector store initialized")569                570        except Exception as e:571            st.error(f"❌ Error loading vector store: {e}")572            self.vectors = []573            self.document_metadata = []574    575    def save_vector_store(self):576        """Save vector store to disk"""577        try:578            with open(self.vector_store_path, 'wb') as f:579                pickle.dump(self.vectors, f)580            581            with open(self.metadata_path, 'wb') as f:582                pickle.dump(self.document_metadata, f)583            584            return True585        except Exception as e:586            st.error(f"Error saving vector store: {e}")587            return False588    589    def create_document_text(self, invoice_data: dict, raw_text: str = "") -> str:590        """Create searchable text from invoice data"""591        text_parts = []592        593        for field, value in invoice_data.items():594            if value and field != 'id':595                text_parts.append(f"{field}: {value}")596        597        if raw_text:598            text_parts.append(f"content: {raw_text[:300]}")599        600        return " | ".join(text_parts)601    602    def add_document(self, invoice_data: dict, raw_text: str = "") -> bool:603        """Add a document to the vector store"""604        if not self.embedding_model:605            return False606        607        try:608            document_text = self.create_document_text(invoice_data, raw_text)609            610            # Generate embedding611            embedding = self.embedding_model.encode(document_text, normalize_embeddings=True)612            613            # Create metadata614            metadata = {615                'invoice_id': invoice_data.get('id', ''),616                'invoice_number': invoice_data.get('invoice_number', ''),617                'supplier_name': invoice_data.get('supplier_name', ''),618                'buyer_name': invoice_data.get('buyer_name', ''),619                'amount': invoice_data.get('amount', 0),620                'date': invoice_data.get('date', ''),621                'file_name': invoice_data.get('file_info', {}).get('file_name', ''),622                'document_text': document_text[:200],623                'timestamp': datetime.now().isoformat()624            }625            626            # Add to store627            self.vectors.append(embedding)628            self.document_metadata.append(metadata)629            630            return True631            632        except Exception as e:633            st.error(f"Error adding document to vector store: {e}")634            return False635    636    def semantic_search(self, query: str, top_k: int = 5) -> List[VectorSearchResult]:637        """Perform semantic search using cosine similarity"""638        if not self.embedding_model or not self.vectors:639            return []640        641        try:642            # Generate query embedding643            query_embedding = self.embedding_model.encode(query, normalize_embeddings=True)644            645            # Calculate similarities646            similarities = []647            for i, doc_embedding in enumerate(self.vectors):648                similarity = np.dot(query_embedding, doc_embedding)649                similarities.append((similarity, i))650            651            # Sort by similarity652            similarities.sort(reverse=True)653            654            # Return top results655            results = []656            for similarity, idx in similarities[:top_k]:657                if similarity > 0.1:  # Relevance threshold658                    metadata = self.document_metadata[idx]659                    result = VectorSearchResult(660                        invoice_id=metadata.get('invoice_id', ''),661                        invoice_number=metadata.get('invoice_number', ''),662                        supplier_name=metadata.get('supplier_name', ''),663                        similarity_score=float(similarity),664                        content_preview=metadata.get('document_text', ''),665                        metadata=metadata666                    )667                    results.append(result)668            669            return results670            671        except Exception as e:672            st.error(f"Error in semantic search: {e}")673            return []674 675# ===============================================================================676# MAIN PROCESSOR CLASS677# ===============================================================================678 679class InvoiceProcessor:680    """Main invoice processor for Hugging Face Spaces"""681    682    def __init__(self):683        self.setup_storage()684        self.document_processor = DocumentProcessor()685        self.ai_extractor = AIExtractor()686        self.vector_store = VectorStore() if SENTENCE_TRANSFORMERS_AVAILABLE else None687        688        # Initialize stats689        self.processing_stats = {690            'total_processed': 0,691            'successful': 0,692            'failed': 0,693            'start_time': datetime.now()694        }695    696    def setup_storage(self):697        """Setup storage paths"""698        self.data_dir = HF_CONFIG["data_dir"]699        self.json_path = os.path.join(self.data_dir, "invoices.json")700        701        # Initialize JSON storage702        if not os.path.exists(self.json_path):703            initial_data = {704                "metadata": {705                    "created_at": datetime.now().isoformat(),706                    "version": "hf_v1.0",707                    "total_invoices": 0708                },709                "invoices": [],710                "summary": {711                    "total_amount": 0.0,712                    "unique_suppliers": [],713                    "processing_stats": {"successful": 0, "failed": 0}714                }715            }716            self.save_json_data(initial_data)717    718    def load_json_data(self) -> dict:719        """Load invoice data from JSON"""720        try:721            with open(self.json_path, 'r', encoding='utf-8') as f:722                return json.load(f)723        except (FileNotFoundError, json.JSONDecodeError):724            self.setup_storage()725            return self.load_json_data()726    727    def save_json_data(self, data: dict):728        """Save invoice data to JSON"""729        try:730            with open(self.json_path, 'w', encoding='utf-8') as f:731                json.dump(data, f, indent=2, ensure_ascii=False)732        except Exception as e:733            st.error(f"Error saving data: {e}")734    735    def process_uploaded_file(self, uploaded_file) -> InvoiceData:736        """Process a single uploaded file with enhanced debugging"""737        self.processing_stats['total_processed'] += 1738        739        try:740            # Debug file info741            file_size = len(uploaded_file.getvalue())742            file_extension = uploaded_file.name.split('.')[-1].lower() if '.' in uploaded_file.name else 'unknown'743            744            st.info(f"📄 Processing: {uploaded_file.name} ({file_size/1024:.1f} KB, .{file_extension})")745            746            # Check file size747            if file_size > HF_CONFIG["max_file_size_mb"] * 1024 * 1024:748                error_msg = f"File too large: {file_size / 1024 / 1024:.2f}MB > {HF_CONFIG['max_file_size_mb']}MB"749                st.error(error_msg)750                self.processing_stats['failed'] += 1751                return InvoiceData()752            753            # Check file type754            if file_extension not in ['pdf', 'txt']:755                error_msg = f"Unsupported file type: .{file_extension} (supported: PDF, TXT)"756                st.warning(error_msg)757                self.processing_stats['failed'] += 1758                return InvoiceData()759            760            # Save temporarily761            with tempfile.NamedTemporaryFile(delete=False, suffix=f".{file_extension}") as tmp_file:762                file_content = uploaded_file.getvalue()763                tmp_file.write(file_content)764                tmp_file_path = tmp_file.name765                766                st.info(f"💾 Saved temporarily to: {tmp_file_path}")767            768            try:769                # Extract text770                st.info("🔍 Extracting text from document...")771                text = self.document_processor.extract_text_from_document(tmp_file_path)772                773                if not text or not text.strip():774                    st.warning(f"❌ No text extracted from {uploaded_file.name}")775                    self.processing_stats['failed'] += 1776                    return InvoiceData()777                778                text_length = len(text)779                st.info(f"📝 Extracted {text_length} characters of text")780                781                # Show text preview782                if text_length > 0:783                    with st.expander("📄 Text Preview (First 500 characters)", expanded=False):784                        st.text(text[:500] + "..." if len(text) > 500 else text)785                786                # Extract invoice data787                st.info("🤖 Extracting invoice data using AI/Regex...")788                invoice_data = self.ai_extractor.extract_with_ai(text)789                invoice_data.file_path = uploaded_file.name790                791                # Show extraction results792                st.info(f"📊 Extraction completed with {invoice_data.extraction_confidence:.1%} confidence")793                794                # Save to storage795                st.info("💾 Saving extracted data...")796                self.save_invoice_data(invoice_data, text, file_size)797                798                self.processing_stats['successful'] += 1799                st.success(f"✅ Successfully processed {uploaded_file.name}")800                801                return invoice_data802                803            finally:804                # Cleanup805                try:806                    os.unlink(tmp_file_path)807                    st.info("🧹 Cleaned up temporary file")808                except:809                    pass810                811        except Exception as e:812            error_msg = f"Error processing {uploaded_file.name}: {str(e)}"813            st.error(error_msg)814            self.processing_stats['failed'] += 1815            816            # Show detailed error for debugging817            with st.expander("🔍 Error Details", expanded=False):818                st.code(str(e))819                import traceback820                st.code(traceback.format_exc())821            822            return InvoiceData()823    824    def save_invoice_data(self, invoice_data: InvoiceData, raw_text: str, file_size: int):825        """Save invoice data to JSON and vector store"""826        try:827            # Load existing data828            data = self.load_json_data()829            830            # Create invoice record831            invoice_record = {832                "id": len(data["invoices"]) + 1,833                "invoice_number": invoice_data.invoice_number,834                "supplier_name": invoice_data.supplier_name,835                "buyer_name": invoice_data.buyer_name,836                "date": invoice_data.date,837                "amount": invoice_data.amount,838                "quantity": invoice_data.quantity,839                "product_description": invoice_data.product_description,840                "file_info": {841                    "file_name": invoice_data.file_path,842                    "file_size": file_size843                },844                "extraction_info": {845                    "confidence": invoice_data.extraction_confidence,846                    "method": invoice_data.processing_method,847                    "raw_text_preview": raw_text[:300]848                },849                "timestamps": {850                    "created_at": datetime.now().isoformat()851                }852            }853            854            # Add to invoices855            data["invoices"].append(invoice_record)856            857            # Update summary858            self.update_summary(data)859            860            # Save JSON861            self.save_json_data(data)862            863            # Add to vector store864            if self.vector_store:865                self.vector_store.add_document(invoice_record, raw_text)866                self.vector_store.save_vector_store()867            868        except Exception as e:869            st.error(f"Error saving invoice data: {e}")870    871    def update_summary(self, data: dict):872        """Update summary statistics"""873        invoices = data["invoices"]874        875        total_amount = sum(inv.get("amount", 0) for inv in invoices)876        unique_suppliers = list(set(inv.get("supplier_name", "") for inv in invoices if inv.get("supplier_name")))877        878        data["summary"] = {879            "total_amount": total_amount,880            "unique_suppliers": unique_suppliers,881            "processing_stats": {882                "successful": self.processing_stats['successful'],883                "failed": self.processing_stats['failed'],884                "total_processed": self.processing_stats['total_processed']885            }886        }887        888        data["metadata"]["last_updated"] = datetime.now().isoformat()889        data["metadata"]["total_invoices"] = len(invoices)890 891# ===============================================================================892# CHATBOT CLASS893# ===============================================================================894 895class ChatBot:896    """Chatbot for invoice queries"""897    898    def __init__(self, processor: InvoiceProcessor):899        self.processor = processor900    901    def query_database(self, query: str) -> str:902        """Process user query and return response"""903        try:904            data = self.processor.load_json_data()905            invoices = data.get("invoices", [])906            907            if not invoices:908                return "No invoice data found. Please upload some invoices first."909            910            query_lower = query.lower()911            912            # Handle different query types913            if any(phrase in query_lower for phrase in ["summary", "overview", "total"]):914                return self.generate_summary(data)915            916            elif "count" in query_lower or "how many" in query_lower:917                return self.handle_count_query(data)918            919            elif any(phrase in query_lower for phrase in ["amount", "value", "money", "cost"]):920                return self.handle_amount_query(data)921            922            elif any(phrase in query_lower for phrase in ["supplier", "vendor", "company"]):923                return self.handle_supplier_query(data, query)924            925            elif self.processor.vector_store:926                return self.handle_semantic_search(query)927            928            else:929                return self.handle_general_query(data, query)930                931        except Exception as e:932            return f"Error processing query: {e}"933    934    def generate_summary(self, data: dict) -> str:935        """Generate comprehensive summary"""936        invoices = data.get("invoices", [])937        summary = data.get("summary", {})938        939        if not invoices:940            return "No invoices found in the system."941        942        total_amount = summary.get("total_amount", 0)943        avg_amount = total_amount / len(invoices) if invoices else 0944        unique_suppliers = len(summary.get("unique_suppliers", []))945        946        response = f"""947**📊 Invoice System Summary**948 949• **Total Invoices**: {len(invoices):,}950• **Total Value**: ₹{total_amount:,.2f}951• **Average Invoice**: ₹{avg_amount:,.2f}952• **Unique Suppliers**: {unique_suppliers}953 954**📈 Processing Stats**955• **Successful**: {summary.get('processing_stats', {}).get('successful', 0)}956• **Failed**: {summary.get('processing_stats', {}).get('failed', 0)}957 958**🔍 Recent Invoices**959"""960        961        # Show recent invoices962        recent = sorted(invoices, key=lambda x: x.get('timestamps', {}).get('created_at', ''), reverse=True)[:5]963        for i, inv in enumerate(recent, 1):964            response += f"\n{i}. **{inv.get('invoice_number', 'N/A')}** - {inv.get('supplier_name', 'Unknown')} (₹{inv.get('amount', 0):,.2f})"965        966        return response967    968    def handle_count_query(self, data: dict) -> str:969        """Handle count-related queries"""970        invoices = data.get("invoices", [])971        total = len(invoices)972        unique_numbers = len(set(inv.get('invoice_number', '') for inv in invoices if inv.get('invoice_number')))973        974        return f"""975**📊 Invoice Count Summary**976 977• **Total Records**: {total}978• **Unique Invoice Numbers**: {unique_numbers}979• **Duplicates**: {total - unique_numbers if total > unique_numbers else 0}980 981**📅 Processing Timeline**982• **First Invoice**: {invoices[0].get('timestamps', {}).get('created_at', 'N/A')[:10] if invoices else 'N/A'}983• **Latest Invoice**: {invoices[-1].get('timestamps', {}).get('created_at', 'N/A')[:10] if invoices else 'N/A'}984"""985    986    def handle_amount_query(self, data: dict) -> str:987        """Handle amount-related queries"""988        invoices = data.get("invoices", [])989        amounts = [inv.get('amount', 0) for inv in invoices if inv.get('amount', 0) > 0]990        991        if not amounts:992            return "No amount information found in invoices."993        994        total_amount = sum(amounts)995        avg_amount = total_amount / len(amounts)996        max_amount = max(amounts)997        min_amount = min(amounts)998        999        # Find high-value invoices1000        high_value_threshold = sorted(amounts, reverse=True)[min(4, len(amounts)-1)] if len(amounts) > 5 else max_amount1001        high_value_invoices = [inv for inv in invoices if inv.get('amount', 0) >= high_value_threshold]1002        1003        response = f"""1004**💰 Financial Analysis**1005 1006• **Total Amount**: ₹{total_amount:,.2f}1007• **Average Amount**: ₹{avg_amount:,.2f}1008• **Highest Invoice**: ₹{max_amount:,.2f}1009• **Lowest Invoice**: ₹{min_amount:,.2f}1010 1011**🎯 High-Value Invoices (₹{high_value_threshold:,.2f}+)**1012"""1013        1014        for i, inv in enumerate(high_value_invoices[:5], 1):1015            response += f"\n{i}. **{inv.get('invoice_number', 'N/A')}** - {inv.get('supplier_name', 'Unknown')} (₹{inv.get('amount', 0):,.2f})"1016        1017        return response1018    1019    def handle_supplier_query(self, data: dict, query: str) -> str:1020        """Handle supplier-related queries"""1021        invoices = data.get("invoices", [])1022        1023        # Count invoices by supplier1024        supplier_counts = {}1025        supplier_amounts = {}1026        1027        for inv in invoices:1028            supplier = inv.get('supplier_name', '').strip()1029            if supplier:1030                supplier_counts[supplier] = supplier_counts.get(supplier, 0) + 11031                supplier_amounts[supplier] = supplier_amounts.get(supplier, 0) + inv.get('amount', 0)1032        1033        if not supplier_counts:1034            return "No supplier information found in invoices."1035        1036        # Sort suppliers by amount1037        top_suppliers = sorted(supplier_amounts.items(), key=lambda x: x[1], reverse=True)[:10]1038        1039        response = f"""1040**🏢 Supplier Analysis**1041 1042• **Total Unique Suppliers**: {len(supplier_counts)}1043• **Most Active**: {max(supplier_counts, key=supplier_counts.get)} ({supplier_counts[max(supplier_counts, key=supplier_counts.get)]} invoices)1044 1045**💰 Top Suppliers by Amount**1046"""1047        1048        for i, (supplier, amount) in enumerate(top_suppliers, 1):1049            count = supplier_counts[supplier]1050            avg = amount / count if count > 0 else 01051            response += f"\n{i}. **{supplier}** - ₹{amount:,.2f} ({count} invoices, avg: ₹{avg:,.2f})"1052        1053        return response1054    1055    def handle_semantic_search(self, query: str) -> str:1056        """Handle semantic search queries"""1057        try:1058            results = self.processor.vector_store.semantic_search(query, top_k=5)1059            1060            if not results:1061                return f"No relevant results found for '{query}'. Try different keywords."1062            1063            response = f"🔍 **Semantic Search Results for '{query}'**\n\n"1064            1065            for i, result in enumerate(results, 1):1066                response += f"{i}. **{result.invoice_number}** - {result.supplier_name}\n"1067                response += f"   • Similarity: {result.similarity_score:.3f}\n"1068                response += f"   • Amount: ₹{result.metadata.get('amount', 0):,.2f}\n"1069                response += f"   • Preview: {result.content_preview[:100]}...\n\n"1070            1071            return response1072            1073        except Exception as e:1074            return f"Semantic search error: {e}"1075    1076    def handle_general_query(self, data: dict, query: str) -> str:1077        """Handle general queries with keyword search"""1078        invoices = data.get("invoices", [])1079        query_words = query.lower().split()1080        1081        # Simple keyword matching1082        matching_invoices = []1083        for inv in invoices:1084            text_to_search = (1085                inv.get('supplier_name', '') + ' ' +1086                inv.get('buyer_name', '') + ' ' +1087                inv.get('product_description', '') + ' ' +1088                inv.get('extraction_info', {}).get('raw_text_preview', '')1089            ).lower()1090            1091            if any(word in text_to_search for word in query_words):1092                matching_invoices.append(inv)1093        1094        if not matching_invoices:1095            return f"No invoices found matching '{query}'. Try different keywords or check the summary."1096        1097        response = f"🔍 **Found {len(matching_invoices)} invoices matching '{query}'**\n\n"1098        1099        for i, inv in enumerate(matching_invoices[:5], 1):1100            response += f"{i}. **{inv.get('invoice_number', 'N/A')}** - {inv.get('supplier_name', 'Unknown')}\n"1101            response += f"   • Amount: ₹{inv.get('amount', 0):,.2f}\n"1102            response += f"   • Date: {inv.get('date', 'N/A')}\n\n"1103        1104        if len(matching_invoices) > 5:1105            response += f"... and {len(matching_invoices) - 5} more results."1106        1107        return response1108 1109# ===============================================================================1110# STREAMLIT APPLICATION1111# ===============================================================================1112 1113def create_app():1114    """Main Streamlit application"""1115    1116    # Generate unique session ID for this run1117    if 'session_id' not in st.session_state:1118        st.session_state.session_id = str(uuid.uuid4())[:8]1119    1120    session_id = st.session_state.session_id1121    1122    # Custom CSS1123    st.markdown("""1124    <style>1125    .main-header {1126        font-size: 2.5rem;1127        font-weight: bold;1128        text-align: center;1129        color: #FF6B35;1130        margin-bottom: 1rem;1131    }1132    .feature-box {1133        background: linear-gradient(135deg, #667eea 0%, #764ba2 100%);1134        padding: 1rem;1135        border-radius: 10px;1136        color: white;1137        margin: 0.5rem 0;1138        text-align: center;1139    }1140    .status-ok { color: #28a745; font-weight: bold; }1141    .status-warning { color: #ffc107; font-weight: bold; }1142    .status-error { color: #dc3545; font-weight: bold; }1143    </style>1144    """, unsafe_allow_html=True)1145    1146    # Header1147    st.markdown('<h1 class="main-header">📄 AI Invoice Processing System</h1>', unsafe_allow_html=True)1148    st.markdown("""1149    <div style="text-align: center; margin-bottom: 2rem;">1150        <p style="font-size: 1.1rem; color: #666;">1151            AI-Powered Document Processing • Semantic Search • Smart Analytics • Hugging Face Spaces1152        </p>1153    </div>1154    """, unsafe_allow_html=True)1155    1156    # Initialize processor1157    if 'processor' not in st.session_state:1158        with st.spinner("🔧 Initializing AI Invoice Processor..."):1159            try:1160                st.session_state.processor = InvoiceProcessor()1161                st.session_state.chatbot = ChatBot(st.session_state.processor)1162                st.session_state.chat_history = []1163                st.success("✅ System initialized successfully!")1164            except Exception as e:1165                st.error(f"❌ Initialization failed: {e}")1166                st.stop()1167    1168    # Sidebar1169    with st.sidebar:1170        st.header("🎛️ System Status")1171        1172        processor = st.session_state.processor1173        1174        # Component status1175        if processor.document_processor.processors:1176            st.markdown('<span class="status-ok">✅ Document Processing</span>', unsafe_allow_html=True)1177        else:1178            st.markdown('<span class="status-error">❌ Document Processing</span>', unsafe_allow_html=True)1179        1180        if processor.ai_extractor.use_transformers:1181            st.markdown('<span class="status-ok">✅ AI Extraction</span>', unsafe_allow_html=True)1182        else:1183            st.markdown('<span class="status-warning">⚠️ Regex Extraction</span>', unsafe_allow_html=True)1184        1185        if processor.vector_store and processor.vector_store.embedding_model:1186            st.markdown('<span class="status-ok">✅ Semantic Search</span>', unsafe_allow_html=True)1187        else:1188            st.markdown('<span class="status-warning">⚠️ Keyword Search Only</span>', unsafe_allow_html=True)1189        1190        # Quick stats1191        st.header("📊 Quick Stats")1192        try:1193            data = processor.load_json_data()1194            total_invoices = len(data.get("invoices", []))1195            total_amount = data.get("summary", {}).get("total_amount", 0)1196            1197            st.metric("Total Invoices", total_invoices)1198            st.metric("Total Value", f"₹{total_amount:,.2f}")1199            st.metric("Success Rate", f"{processor.processing_stats['successful']}/{processor.processing_stats['total_processed']}")1200            

Showing the first 1,200 of 3259 lines. Download the file for the rest.