shivam701171/Invoice_processing_tool
0
1#!/usr/bin/env python32"""3AI Invoice Processing System - Complete Single File for Hugging Face Spaces4A comprehensive system with AI-powered extraction, semantic search, and analytics.5 6Author: AI Assistant7Date: 20248Version: HuggingFace Single File v1.09"""10 11# ===============================================================================12# IMPORTS AND COMPATIBILITY CHECKS13# ===============================================================================14 15import os16import json17import re18import tempfile19import shutil20import pickle21import numpy as np22from datetime import datetime23from typing import Dict, List, Optional, Tuple24from dataclasses import dataclass25from pathlib import Path26import time27import logging28import uuid29 30# Check if running on Hugging Face Spaces31IS_HF_SPACE = os.getenv("SPACE_ID") is not None32 33# Get Hugging Face token from environment or Streamlit secrets34HF_TOKEN = None35try:36 # Try Streamlit secrets first (for HF Spaces)37 HF_TOKEN = st.secrets.get("HF_TOKEN", None)38except:39 # Fall back to environment variable40 HF_TOKEN = os.getenv("HF_TOKEN", None)41 42# Streamlit and core libraries43import streamlit as st44import sqlite345import pandas as pd46import plotly.express as px47import plotly.graph_objects as go48import requests49 50# Vector storage and embeddings (with fallbacks)51try:52 import faiss53 FAISS_AVAILABLE = True54except ImportError:55 FAISS_AVAILABLE = False56 st.warning("⚠️ FAISS not available. Vector search will be disabled.")57 58try:59 from sentence_transformers import SentenceTransformer60 SENTENCE_TRANSFORMERS_AVAILABLE = True61except ImportError:62 SENTENCE_TRANSFORMERS_AVAILABLE = False63 st.warning("⚠️ Sentence Transformers not available. Using fallback methods.")64 65try:66 import torch67 TORCH_AVAILABLE = True68except ImportError:69 TORCH_AVAILABLE = False70 71# Document processing (simplified for HF)72try:73 import pdfplumber74 PDF_PROCESSING_AVAILABLE = True75 PDF_PROCESSOR = "pdfplumber"76except ImportError:77 try:78 import PyPDF279 PDF_PROCESSING_AVAILABLE = True80 PDF_PROCESSOR = "PyPDF2"81 except ImportError:82 PDF_PROCESSING_AVAILABLE = False83 PDF_PROCESSOR = None84 85# ===============================================================================86# STREAMLIT CONFIGURATION87# ===============================================================================88 89st.set_page_config(90 page_title="AI Invoice Processing System",91 page_icon="📄",92 layout="wide",93 initial_sidebar_state="expanded",94 menu_items={95 'Get Help': 'https://huggingface.co/spaces',96 'Report a bug': 'https://huggingface.co/spaces',97 'About': """98 # AI Invoice Processing System99 Built for Hugging Face Spaces with AI-powered extraction and semantic search.100 """101 }102)103 104# ===============================================================================105# CONFIGURATION106# ===============================================================================107 108HF_CONFIG = {109 "max_file_size_mb": 10,110 "max_concurrent_files": 3,111 "timeout_seconds": 30,112 "use_cpu_only": True,113 "embedding_model": "all-MiniLM-L6-v2",114 "cache_dir": "./cache",115 "data_dir": "./data",116 "enable_ollama": False,117}118 119# Create necessary directories120os.makedirs(HF_CONFIG["cache_dir"], exist_ok=True)121os.makedirs(HF_CONFIG["data_dir"], exist_ok=True)122 123# ===============================================================================124# DATA STRUCTURES125# ===============================================================================126 127@dataclass128class InvoiceData:129 """Data structure for extracted invoice information"""130 supplier_name: str = ""131 buyer_name: str = ""132 invoice_number: str = ""133 date: str = ""134 amount: float = 0.0135 quantity: int = 0136 product_description: str = ""137 file_path: str = ""138 extraction_confidence: float = 0.0139 processing_method: str = "regex"140 141@dataclass142class VectorSearchResult:143 """Data structure for vector search results"""144 invoice_id: str145 invoice_number: str146 supplier_name: str147 similarity_score: float148 content_preview: str149 metadata: Dict150 151# ===============================================================================152# DOCUMENT PROCESSING CLASSES153# ===============================================================================154 155class DocumentProcessor:156 """Simplified document processor for Hugging Face Spaces"""157 158 def __init__(self):159 self.setup_processors()160 161 def setup_processors(self):162 """Setup available document processors"""163 self.processors = {}164 165 # PDF processing166 if PDF_PROCESSING_AVAILABLE:167 if PDF_PROCESSOR == "pdfplumber":168 self.processors['pdf'] = self.extract_with_pdfplumber169 st.success("✅ PDF processing available (pdfplumber)")170 elif PDF_PROCESSOR == "PyPDF2":171 self.processors['pdf'] = self.extract_with_pypdf2172 st.success("✅ PDF processing available (PyPDF2)")173 else:174 st.warning("⚠️ No PDF processor available")175 176 # Text files177 self.processors['txt'] = self.extract_text_file178 179 def extract_with_pdfplumber(self, file_path: str) -> str:180 """Extract text using pdfplumber"""181 try:182 import pdfplumber183 text = ""184 with pdfplumber.open(file_path) as pdf:185 for page in pdf.pages:186 page_text = page.extract_text()187 if page_text:188 text += page_text + "\n"189 return text190 except Exception as e:191 st.error(f"PDF extraction failed: {e}")192 return ""193 194 def extract_with_pypdf2(self, file_path: str) -> str:195 """Extract text using PyPDF2"""196 try:197 import PyPDF2198 text = ""199 with open(file_path, 'rb') as file:200 pdf_reader = PyPDF2.PdfReader(file)201 for page in pdf_reader.pages:202 text += page.extract_text() + "\n"203 return text204 except Exception as e:205 st.error(f"PDF extraction failed: {e}")206 return ""207 208 def extract_text_file(self, file_path: str) -> str:209 """Extract text from text files"""210 try:211 with open(file_path, 'r', encoding='utf-8') as f:212 return f.read()213 except Exception as e:214 st.error(f"Text file extraction failed: {e}")215 return ""216 217 def extract_text_from_document(self, file_path: str) -> str:218 """Extract text from document based on file type"""219 file_ext = Path(file_path).suffix.lower()220 221 if file_ext == '.pdf':222 processor = self.processors.get('pdf')223 elif file_ext == '.txt':224 processor = self.processors.get('txt')225 else:226 st.warning(f"Unsupported file type: {file_ext}")227 return ""228 229 if processor:230 return processor(file_path)231 else:232 st.error(f"No processor available for {file_ext}")233 return ""234 235# ===============================================================================236# AI EXTRACTION CLASS237# ===============================================================================238 239class AIExtractor:240 """AI extraction for Hugging Face Spaces with Mistral 7B support"""241 242 def __init__(self):243 self.use_mistral = self.setup_mistral()244 self.use_transformers = self.setup_transformers() if not self.use_mistral else False245 246 def setup_mistral(self):247 """Try to setup Mistral 7B model with proper authentication"""248 try:249 # Check if we have HF token250 if not HF_TOKEN:251 st.warning("⚠️ Hugging Face token not found. Add HF_TOKEN to secrets for Mistral access.")252 return False253 254 # Check if we're in a high-resource environment255 import psutil256 memory_gb = psutil.virtual_memory().total / (1024**3)257 258 if memory_gb < 8:259 st.warning("⚠️ Insufficient memory for Mistral 7B. Using lighter models.")260 return False261 262 from transformers import AutoModelForCausalLM, AutoTokenizer, pipeline263 from huggingface_hub import login264 265 # Login with HF token266 login(token=HF_TOKEN)267 268 with st.spinner("🔄 Loading Mistral 7B model (this may take a few minutes)..."):269 # Use the instruction-tuned model270 model_name = "mistralai/Mistral-7B-Instruct-v0.1"271 272 # Load with reduced precision for memory efficiency273 self.mistral_tokenizer = AutoTokenizer.from_pretrained(274 model_name,275 cache_dir=HF_CONFIG["cache_dir"],276 token=HF_TOKEN277 )278 279 self.mistral_model = AutoModelForCausalLM.from_pretrained(280 model_name,281 torch_dtype=torch.float16 if TORCH_AVAILABLE else None,282 device_map="auto" if TORCH_AVAILABLE else None,283 load_in_8bit=True, # Use 8-bit quantization284 cache_dir=HF_CONFIG["cache_dir"],285 token=HF_TOKEN286 )287 288 # Create pipeline289 self.mistral_pipeline = pipeline(290 "text-generation",291 model=self.mistral_model,292 tokenizer=self.mistral_tokenizer,293 torch_dtype=torch.float16 if TORCH_AVAILABLE else None,294 device_map="auto" if TORCH_AVAILABLE else None295 )296 297 st.success("✅ Mistral 7B model loaded successfully!")298 return True299 300 except ImportError as e:301 st.warning(f"⚠️ Missing dependencies for Mistral 7B: {e}")302 return False303 except Exception as e:304 st.warning(f"⚠️ Mistral 7B not available: {e}")305 st.info("💡 To use Mistral 7B: Add your Hugging Face token to secrets as 'HF_TOKEN'")306 return False307 308 def setup_transformers(self):309 """Fallback to lighter NER model"""310 try:311 from transformers import pipeline312 313 with st.spinner("Loading fallback AI model..."):314 self.ner_pipeline = pipeline(315 "ner", 316 model="dbmdz/bert-large-cased-finetuned-conll03-english",317 aggregation_strategy="simple"318 )319 320 st.success("✅ Fallback AI extraction model loaded")321 return True322 323 except Exception as e:324 st.warning(f"⚠️ AI extraction not available: {e}")325 return False326 327 def extract_with_mistral(self, text: str) -> InvoiceData:328 """Extract invoice data using Mistral 7B"""329 try:330 # Create a detailed prompt for Mistral331 prompt = f"""<s>[INST] You are an expert at extracting structured information from invoices. 332 333Extract the following information from this invoice text and respond ONLY with valid JSON:334 335{{336 "invoice_number": "invoice or bill number",337 "supplier_name": "company providing goods/services",338 "buyer_name": "company receiving goods/services",339 "date": "date in YYYY-MM-DD format",340 "amount": "total amount as number only",341 "quantity": "total quantity as integer",342 "product_description": "brief description of items/services"343}}344 345Invoice text:346{text[:2000]}347 348Respond with JSON only: [/INST]"""349 350 # Generate response351 response = self.mistral_pipeline(352 prompt,353 max_new_tokens=300,354 temperature=0.1,355 do_sample=True,356 pad_token_id=self.mistral_tokenizer.eos_token_id357 )358 359 # Extract the generated text360 generated_text = response[0]['generated_text']361 362 # Find JSON in the response363 json_start = generated_text.find('{')364 json_end = generated_text.rfind('}') + 1365 366 if json_start != -1 and json_end > json_start:367 json_str = generated_text[json_start:json_end]368 369 # Parse JSON370 import json371 data = json.loads(json_str)372 373 # Create InvoiceData object374 invoice_data = InvoiceData()375 invoice_data.supplier_name = str(data.get('supplier_name', '')).strip()376 invoice_data.buyer_name = str(data.get('buyer_name', '')).strip()377 invoice_data.invoice_number = str(data.get('invoice_number', '')).strip()378 invoice_data.date = self.parse_date(str(data.get('date', '')))379 380 # Parse amount381 try:382 amount_val = data.get('amount', 0)383 if isinstance(amount_val, str):384 amount_clean = re.sub(r'[^\d.]', '', amount_val)385 invoice_data.amount = float(amount_clean) if amount_clean else 0.0386 else:387 invoice_data.amount = float(amount_val)388 except:389 invoice_data.amount = 0.0390 391 # Parse quantity392 try:393 qty_val = data.get('quantity', 0)394 invoice_data.quantity = int(float(str(qty_val).replace(',', '')))395 except:396 invoice_data.quantity = 0397 398 invoice_data.product_description = str(data.get('product_description', '')).strip()399 invoice_data.extraction_confidence = 0.95 # High confidence for Mistral400 invoice_data.processing_method = "mistral_7b"401 402 return invoice_data403 else:404 st.warning("⚠️ Mistral response didn't contain valid JSON, falling back to regex")405 return self.extract_with_regex(text)406 407 except Exception as e:408 st.error(f"Mistral extraction failed: {e}")409 return self.extract_with_regex(text)410 411 def extract_with_ai(self, text: str) -> InvoiceData:412 """Extract invoice data using available AI method"""413 if self.use_mistral:414 st.info("🤖 Using Mistral 7B for extraction...")415 return self.extract_with_mistral(text)416 elif self.use_transformers:417 st.info("🤖 Using NER model for extraction...")418 return self.extract_with_ner(text)419 else:420 st.info("🔧 Using regex extraction...")421 return self.extract_with_regex(text)422 423 def extract_with_ner(self, text: str) -> InvoiceData:424 """Extract using NER model (fallback method)"""425 try:426 # Use NER to extract entities427 entities = self.ner_pipeline(text[:512]) # Limit text length428 429 invoice_data = InvoiceData()430 invoice_data.processing_method = "ai_ner"431 432 # Extract specific entities433 for entity in entities:434 entity_text = entity['word'].replace('##', '')435 436 if entity['entity_group'] == 'ORG':437 if not invoice_data.supplier_name:438 invoice_data.supplier_name = entity_text439 elif not invoice_data.buyer_name:440 invoice_data.buyer_name = entity_text441 442 elif entity['entity_group'] == 'MISC':443 if not invoice_data.invoice_number and any(c.isdigit() for c in entity_text):444 invoice_data.invoice_number = entity_text445 446 # Fall back to regex for missing fields447 regex_data = self.extract_with_regex(text)448 449 # Combine results450 if not invoice_data.invoice_number:451 invoice_data.invoice_number = regex_data.invoice_number452 if not invoice_data.amount:453 invoice_data.amount = regex_data.amount454 if not invoice_data.date:455 invoice_data.date = regex_data.date456 if not invoice_data.quantity:457 invoice_data.quantity = regex_data.quantity458 459 invoice_data.extraction_confidence = 0.8460 461 return invoice_data462 463 except Exception as e:464 st.error(f"NER extraction failed: {e}")465 return self.extract_with_regex(text)466 467 def extract_with_regex(self, text: str) -> InvoiceData:468 """Enhanced regex extraction with better amount detection"""469 invoice_data = InvoiceData()470 invoice_data.processing_method = "regex"471 472 # Enhanced regex patterns with more comprehensive matching473 patterns = {474 'invoice_number': [475 r'invoice\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',476 r'bill\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',477 r'inv\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',478 r'ref\s*(?:no|number|#)?\s*:?\s*([A-Z0-9\-_/]+)',479 r'#\s*([A-Z0-9\-_/]{3,})',480 r'(?:^|\s)([A-Z]{2,}\d{3,}|\d{3,}[A-Z]{2,})', # Common patterns like ABC123 or 123ABC481 ],482 'amount': [483 # Currency symbols with amounts484 r'total\s*(?:amount)?\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',485 r'amount\s*(?:due|paid|total)?\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',486 r'grand\s*total\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',487 r'net\s*(?:amount|total)\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',488 r'sub\s*total\s*:?\s*[\$₹£€]?\s*([0-9,]+\.?\d*)',489 490 # Currency symbols at the beginning491 r'[\$₹£€]\s*([0-9,]+\.?\d*)',492 493 # Amounts at end of lines (common in invoices)494 r'([0-9,]+\.?\d*)\s*[\$₹£€]?\s*495 496 def parse_date(self, date_str: str) -> str:497 """Parse date to YYYY-MM-DD format"""498 if not date_str:499 return ""500 501 formats = ['%Y-%m-%d', '%m/%d/%Y', '%d/%m/%Y', '%m-%d-%Y', '%d-%m-%Y', '%Y/%m/%d']502 503 for fmt in formats:504 try:505 parsed_date = datetime.strptime(date_str, fmt)506 return parsed_date.strftime('%Y-%m-%d')507 except ValueError:508 continue509 510 return date_str511 512# ===============================================================================513# VECTOR STORE CLASS514# ===============================================================================515 516class VectorStore:517 """Simplified vector store for Hugging Face Spaces"""518 519 def __init__(self, embedding_model: str = "all-MiniLM-L6-v2"):520 self.embedding_model_name = embedding_model521 self.vector_store_path = os.path.join(HF_CONFIG["data_dir"], "vectors.pkl")522 self.metadata_path = os.path.join(HF_CONFIG["data_dir"], "metadata.pkl")523 self.embedding_model = None524 self.vectors = []525 self.document_metadata = []526 self.embedding_dimension = None527 528 self.setup_embedding_model()529 self.load_vector_store()530 531 def setup_embedding_model(self):532 """Initialize the sentence transformer model"""533 if not SENTENCE_TRANSFORMERS_AVAILABLE:534 st.warning("⚠️ Sentence Transformers not available. Vector search disabled.")535 return536 537 try:538 with st.spinner(f"Loading embedding model: {self.embedding_model_name}..."):539 self.embedding_model = SentenceTransformer(540 self.embedding_model_name,541 cache_folder=HF_CONFIG["cache_dir"]542 )543 544 # Get embedding dimension545 test_embedding = self.embedding_model.encode(["test"])546 self.embedding_dimension = test_embedding.shape[0]547 548 st.success(f"✅ Embedding model loaded: {self.embedding_model_name}")549 550 except Exception as e:551 st.error(f"❌ Failed to load embedding model: {e}")552 self.embedding_model = None553 554 def load_vector_store(self):555 """Load existing vector store"""556 try:557 if os.path.exists(self.vector_store_path) and os.path.exists(self.metadata_path):558 with open(self.vector_store_path, 'rb') as f:559 self.vectors = pickle.load(f)560 561 with open(self.metadata_path, 'rb') as f:562 self.document_metadata = pickle.load(f)563 564 st.success(f"✅ Vector store loaded: {len(self.document_metadata)} documents")565 else:566 self.vectors = []567 self.document_metadata = []568 st.info("📄 New vector store initialized")569 570 except Exception as e:571 st.error(f"❌ Error loading vector store: {e}")572 self.vectors = []573 self.document_metadata = []574 575 def save_vector_store(self):576 """Save vector store to disk"""577 try:578 with open(self.vector_store_path, 'wb') as f:579 pickle.dump(self.vectors, f)580 581 with open(self.metadata_path, 'wb') as f:582 pickle.dump(self.document_metadata, f)583 584 return True585 except Exception as e:586 st.error(f"Error saving vector store: {e}")587 return False588 589 def create_document_text(self, invoice_data: dict, raw_text: str = "") -> str:590 """Create searchable text from invoice data"""591 text_parts = []592 593 for field, value in invoice_data.items():594 if value and field != 'id':595 text_parts.append(f"{field}: {value}")596 597 if raw_text:598 text_parts.append(f"content: {raw_text[:300]}")599 600 return " | ".join(text_parts)601 602 def add_document(self, invoice_data: dict, raw_text: str = "") -> bool:603 """Add a document to the vector store"""604 if not self.embedding_model:605 return False606 607 try:608 document_text = self.create_document_text(invoice_data, raw_text)609 610 # Generate embedding611 embedding = self.embedding_model.encode(document_text, normalize_embeddings=True)612 613 # Create metadata614 metadata = {615 'invoice_id': invoice_data.get('id', ''),616 'invoice_number': invoice_data.get('invoice_number', ''),617 'supplier_name': invoice_data.get('supplier_name', ''),618 'buyer_name': invoice_data.get('buyer_name', ''),619 'amount': invoice_data.get('amount', 0),620 'date': invoice_data.get('date', ''),621 'file_name': invoice_data.get('file_info', {}).get('file_name', ''),622 'document_text': document_text[:200],623 'timestamp': datetime.now().isoformat()624 }625 626 # Add to store627 self.vectors.append(embedding)628 self.document_metadata.append(metadata)629 630 return True631 632 except Exception as e:633 st.error(f"Error adding document to vector store: {e}")634 return False635 636 def semantic_search(self, query: str, top_k: int = 5) -> List[VectorSearchResult]:637 """Perform semantic search using cosine similarity"""638 if not self.embedding_model or not self.vectors:639 return []640 641 try:642 # Generate query embedding643 query_embedding = self.embedding_model.encode(query, normalize_embeddings=True)644 645 # Calculate similarities646 similarities = []647 for i, doc_embedding in enumerate(self.vectors):648 similarity = np.dot(query_embedding, doc_embedding)649 similarities.append((similarity, i))650 651 # Sort by similarity652 similarities.sort(reverse=True)653 654 # Return top results655 results = []656 for similarity, idx in similarities[:top_k]:657 if similarity > 0.1: # Relevance threshold658 metadata = self.document_metadata[idx]659 result = VectorSearchResult(660 invoice_id=metadata.get('invoice_id', ''),661 invoice_number=metadata.get('invoice_number', ''),662 supplier_name=metadata.get('supplier_name', ''),663 similarity_score=float(similarity),664 content_preview=metadata.get('document_text', ''),665 metadata=metadata666 )667 results.append(result)668 669 return results670 671 except Exception as e:672 st.error(f"Error in semantic search: {e}")673 return []674 675# ===============================================================================676# MAIN PROCESSOR CLASS677# ===============================================================================678 679class InvoiceProcessor:680 """Main invoice processor for Hugging Face Spaces"""681 682 def __init__(self):683 self.setup_storage()684 self.document_processor = DocumentProcessor()685 self.ai_extractor = AIExtractor()686 self.vector_store = VectorStore() if SENTENCE_TRANSFORMERS_AVAILABLE else None687 688 # Initialize stats689 self.processing_stats = {690 'total_processed': 0,691 'successful': 0,692 'failed': 0,693 'start_time': datetime.now()694 }695 696 def setup_storage(self):697 """Setup storage paths"""698 self.data_dir = HF_CONFIG["data_dir"]699 self.json_path = os.path.join(self.data_dir, "invoices.json")700 701 # Initialize JSON storage702 if not os.path.exists(self.json_path):703 initial_data = {704 "metadata": {705 "created_at": datetime.now().isoformat(),706 "version": "hf_v1.0",707 "total_invoices": 0708 },709 "invoices": [],710 "summary": {711 "total_amount": 0.0,712 "unique_suppliers": [],713 "processing_stats": {"successful": 0, "failed": 0}714 }715 }716 self.save_json_data(initial_data)717 718 def load_json_data(self) -> dict:719 """Load invoice data from JSON"""720 try:721 with open(self.json_path, 'r', encoding='utf-8') as f:722 return json.load(f)723 except (FileNotFoundError, json.JSONDecodeError):724 self.setup_storage()725 return self.load_json_data()726 727 def save_json_data(self, data: dict):728 """Save invoice data to JSON"""729 try:730 with open(self.json_path, 'w', encoding='utf-8') as f:731 json.dump(data, f, indent=2, ensure_ascii=False)732 except Exception as e:733 st.error(f"Error saving data: {e}")734 735 def process_uploaded_file(self, uploaded_file) -> InvoiceData:736 """Process a single uploaded file with enhanced debugging"""737 self.processing_stats['total_processed'] += 1738 739 try:740 # Debug file info741 file_size = len(uploaded_file.getvalue())742 file_extension = uploaded_file.name.split('.')[-1].lower() if '.' in uploaded_file.name else 'unknown'743 744 st.info(f"📄 Processing: {uploaded_file.name} ({file_size/1024:.1f} KB, .{file_extension})")745 746 # Check file size747 if file_size > HF_CONFIG["max_file_size_mb"] * 1024 * 1024:748 error_msg = f"File too large: {file_size / 1024 / 1024:.2f}MB > {HF_CONFIG['max_file_size_mb']}MB"749 st.error(error_msg)750 self.processing_stats['failed'] += 1751 return InvoiceData()752 753 # Check file type754 if file_extension not in ['pdf', 'txt']:755 error_msg = f"Unsupported file type: .{file_extension} (supported: PDF, TXT)"756 st.warning(error_msg)757 self.processing_stats['failed'] += 1758 return InvoiceData()759 760 # Save temporarily761 with tempfile.NamedTemporaryFile(delete=False, suffix=f".{file_extension}") as tmp_file:762 file_content = uploaded_file.getvalue()763 tmp_file.write(file_content)764 tmp_file_path = tmp_file.name765 766 st.info(f"💾 Saved temporarily to: {tmp_file_path}")767 768 try:769 # Extract text770 st.info("🔍 Extracting text from document...")771 text = self.document_processor.extract_text_from_document(tmp_file_path)772 773 if not text or not text.strip():774 st.warning(f"❌ No text extracted from {uploaded_file.name}")775 self.processing_stats['failed'] += 1776 return InvoiceData()777 778 text_length = len(text)779 st.info(f"📝 Extracted {text_length} characters of text")780 781 # Show text preview782 if text_length > 0:783 with st.expander("📄 Text Preview (First 500 characters)", expanded=False):784 st.text(text[:500] + "..." if len(text) > 500 else text)785 786 # Extract invoice data787 st.info("🤖 Extracting invoice data using AI/Regex...")788 invoice_data = self.ai_extractor.extract_with_ai(text)789 invoice_data.file_path = uploaded_file.name790 791 # Show extraction results792 st.info(f"📊 Extraction completed with {invoice_data.extraction_confidence:.1%} confidence")793 794 # Save to storage795 st.info("💾 Saving extracted data...")796 self.save_invoice_data(invoice_data, text, file_size)797 798 self.processing_stats['successful'] += 1799 st.success(f"✅ Successfully processed {uploaded_file.name}")800 801 return invoice_data802 803 finally:804 # Cleanup805 try:806 os.unlink(tmp_file_path)807 st.info("🧹 Cleaned up temporary file")808 except:809 pass810 811 except Exception as e:812 error_msg = f"Error processing {uploaded_file.name}: {str(e)}"813 st.error(error_msg)814 self.processing_stats['failed'] += 1815 816 # Show detailed error for debugging817 with st.expander("🔍 Error Details", expanded=False):818 st.code(str(e))819 import traceback820 st.code(traceback.format_exc())821 822 return InvoiceData()823 824 def save_invoice_data(self, invoice_data: InvoiceData, raw_text: str, file_size: int):825 """Save invoice data to JSON and vector store"""826 try:827 # Load existing data828 data = self.load_json_data()829 830 # Create invoice record831 invoice_record = {832 "id": len(data["invoices"]) + 1,833 "invoice_number": invoice_data.invoice_number,834 "supplier_name": invoice_data.supplier_name,835 "buyer_name": invoice_data.buyer_name,836 "date": invoice_data.date,837 "amount": invoice_data.amount,838 "quantity": invoice_data.quantity,839 "product_description": invoice_data.product_description,840 "file_info": {841 "file_name": invoice_data.file_path,842 "file_size": file_size843 },844 "extraction_info": {845 "confidence": invoice_data.extraction_confidence,846 "method": invoice_data.processing_method,847 "raw_text_preview": raw_text[:300]848 },849 "timestamps": {850 "created_at": datetime.now().isoformat()851 }852 }853 854 # Add to invoices855 data["invoices"].append(invoice_record)856 857 # Update summary858 self.update_summary(data)859 860 # Save JSON861 self.save_json_data(data)862 863 # Add to vector store864 if self.vector_store:865 self.vector_store.add_document(invoice_record, raw_text)866 self.vector_store.save_vector_store()867 868 except Exception as e:869 st.error(f"Error saving invoice data: {e}")870 871 def update_summary(self, data: dict):872 """Update summary statistics"""873 invoices = data["invoices"]874 875 total_amount = sum(inv.get("amount", 0) for inv in invoices)876 unique_suppliers = list(set(inv.get("supplier_name", "") for inv in invoices if inv.get("supplier_name")))877 878 data["summary"] = {879 "total_amount": total_amount,880 "unique_suppliers": unique_suppliers,881 "processing_stats": {882 "successful": self.processing_stats['successful'],883 "failed": self.processing_stats['failed'],884 "total_processed": self.processing_stats['total_processed']885 }886 }887 888 data["metadata"]["last_updated"] = datetime.now().isoformat()889 data["metadata"]["total_invoices"] = len(invoices)890 891# ===============================================================================892# CHATBOT CLASS893# ===============================================================================894 895class ChatBot:896 """Chatbot for invoice queries"""897 898 def __init__(self, processor: InvoiceProcessor):899 self.processor = processor900 901 def query_database(self, query: str) -> str:902 """Process user query and return response"""903 try:904 data = self.processor.load_json_data()905 invoices = data.get("invoices", [])906 907 if not invoices:908 return "No invoice data found. Please upload some invoices first."909 910 query_lower = query.lower()911 912 # Handle different query types913 if any(phrase in query_lower for phrase in ["summary", "overview", "total"]):914 return self.generate_summary(data)915 916 elif "count" in query_lower or "how many" in query_lower:917 return self.handle_count_query(data)918 919 elif any(phrase in query_lower for phrase in ["amount", "value", "money", "cost"]):920 return self.handle_amount_query(data)921 922 elif any(phrase in query_lower for phrase in ["supplier", "vendor", "company"]):923 return self.handle_supplier_query(data, query)924 925 elif self.processor.vector_store:926 return self.handle_semantic_search(query)927 928 else:929 return self.handle_general_query(data, query)930 931 except Exception as e:932 return f"Error processing query: {e}"933 934 def generate_summary(self, data: dict) -> str:935 """Generate comprehensive summary"""936 invoices = data.get("invoices", [])937 summary = data.get("summary", {})938 939 if not invoices:940 return "No invoices found in the system."941 942 total_amount = summary.get("total_amount", 0)943 avg_amount = total_amount / len(invoices) if invoices else 0944 unique_suppliers = len(summary.get("unique_suppliers", []))945 946 response = f"""947**📊 Invoice System Summary**948 949• **Total Invoices**: {len(invoices):,}950• **Total Value**: ₹{total_amount:,.2f}951• **Average Invoice**: ₹{avg_amount:,.2f}952• **Unique Suppliers**: {unique_suppliers}953 954**📈 Processing Stats**955• **Successful**: {summary.get('processing_stats', {}).get('successful', 0)}956• **Failed**: {summary.get('processing_stats', {}).get('failed', 0)}957 958**🔍 Recent Invoices**959"""960 961 # Show recent invoices962 recent = sorted(invoices, key=lambda x: x.get('timestamps', {}).get('created_at', ''), reverse=True)[:5]963 for i, inv in enumerate(recent, 1):964 response += f"\n{i}. **{inv.get('invoice_number', 'N/A')}** - {inv.get('supplier_name', 'Unknown')} (₹{inv.get('amount', 0):,.2f})"965 966 return response967 968 def handle_count_query(self, data: dict) -> str:969 """Handle count-related queries"""970 invoices = data.get("invoices", [])971 total = len(invoices)972 unique_numbers = len(set(inv.get('invoice_number', '') for inv in invoices if inv.get('invoice_number')))973 974 return f"""975**📊 Invoice Count Summary**976 977• **Total Records**: {total}978• **Unique Invoice Numbers**: {unique_numbers}979• **Duplicates**: {total - unique_numbers if total > unique_numbers else 0}980 981**📅 Processing Timeline**982• **First Invoice**: {invoices[0].get('timestamps', {}).get('created_at', 'N/A')[:10] if invoices else 'N/A'}983• **Latest Invoice**: {invoices[-1].get('timestamps', {}).get('created_at', 'N/A')[:10] if invoices else 'N/A'}984"""985 986 def handle_amount_query(self, data: dict) -> str:987 """Handle amount-related queries"""988 invoices = data.get("invoices", [])989 amounts = [inv.get('amount', 0) for inv in invoices if inv.get('amount', 0) > 0]990 991 if not amounts:992 return "No amount information found in invoices."993 994 total_amount = sum(amounts)995 avg_amount = total_amount / len(amounts)996 max_amount = max(amounts)997 min_amount = min(amounts)998 999 # Find high-value invoices1000 high_value_threshold = sorted(amounts, reverse=True)[min(4, len(amounts)-1)] if len(amounts) > 5 else max_amount1001 high_value_invoices = [inv for inv in invoices if inv.get('amount', 0) >= high_value_threshold]1002 1003 response = f"""1004**💰 Financial Analysis**1005 1006• **Total Amount**: ₹{total_amount:,.2f}1007• **Average Amount**: ₹{avg_amount:,.2f}1008• **Highest Invoice**: ₹{max_amount:,.2f}1009• **Lowest Invoice**: ₹{min_amount:,.2f}1010 1011**🎯 High-Value Invoices (₹{high_value_threshold:,.2f}+)**1012"""1013 1014 for i, inv in enumerate(high_value_invoices[:5], 1):1015 response += f"\n{i}. **{inv.get('invoice_number', 'N/A')}** - {inv.get('supplier_name', 'Unknown')} (₹{inv.get('amount', 0):,.2f})"1016 1017 return response1018 1019 def handle_supplier_query(self, data: dict, query: str) -> str:1020 """Handle supplier-related queries"""1021 invoices = data.get("invoices", [])1022 1023 # Count invoices by supplier1024 supplier_counts = {}1025 supplier_amounts = {}1026 1027 for inv in invoices:1028 supplier = inv.get('supplier_name', '').strip()1029 if supplier:1030 supplier_counts[supplier] = supplier_counts.get(supplier, 0) + 11031 supplier_amounts[supplier] = supplier_amounts.get(supplier, 0) + inv.get('amount', 0)1032 1033 if not supplier_counts:1034 return "No supplier information found in invoices."1035 1036 # Sort suppliers by amount1037 top_suppliers = sorted(supplier_amounts.items(), key=lambda x: x[1], reverse=True)[:10]1038 1039 response = f"""1040**🏢 Supplier Analysis**1041 1042• **Total Unique Suppliers**: {len(supplier_counts)}1043• **Most Active**: {max(supplier_counts, key=supplier_counts.get)} ({supplier_counts[max(supplier_counts, key=supplier_counts.get)]} invoices)1044 1045**💰 Top Suppliers by Amount**1046"""1047 1048 for i, (supplier, amount) in enumerate(top_suppliers, 1):1049 count = supplier_counts[supplier]1050 avg = amount / count if count > 0 else 01051 response += f"\n{i}. **{supplier}** - ₹{amount:,.2f} ({count} invoices, avg: ₹{avg:,.2f})"1052 1053 return response1054 1055 def handle_semantic_search(self, query: str) -> str:1056 """Handle semantic search queries"""1057 try:1058 results = self.processor.vector_store.semantic_search(query, top_k=5)1059 1060 if not results:1061 return f"No relevant results found for '{query}'. Try different keywords."1062 1063 response = f"🔍 **Semantic Search Results for '{query}'**\n\n"1064 1065 for i, result in enumerate(results, 1):1066 response += f"{i}. **{result.invoice_number}** - {result.supplier_name}\n"1067 response += f" • Similarity: {result.similarity_score:.3f}\n"1068 response += f" • Amount: ₹{result.metadata.get('amount', 0):,.2f}\n"1069 response += f" • Preview: {result.content_preview[:100]}...\n\n"1070 1071 return response1072 1073 except Exception as e:1074 return f"Semantic search error: {e}"1075 1076 def handle_general_query(self, data: dict, query: str) -> str:1077 """Handle general queries with keyword search"""1078 invoices = data.get("invoices", [])1079 query_words = query.lower().split()1080 1081 # Simple keyword matching1082 matching_invoices = []1083 for inv in invoices:1084 text_to_search = (1085 inv.get('supplier_name', '') + ' ' +1086 inv.get('buyer_name', '') + ' ' +1087 inv.get('product_description', '') + ' ' +1088 inv.get('extraction_info', {}).get('raw_text_preview', '')1089 ).lower()1090 1091 if any(word in text_to_search for word in query_words):1092 matching_invoices.append(inv)1093 1094 if not matching_invoices:1095 return f"No invoices found matching '{query}'. Try different keywords or check the summary."1096 1097 response = f"🔍 **Found {len(matching_invoices)} invoices matching '{query}'**\n\n"1098 1099 for i, inv in enumerate(matching_invoices[:5], 1):1100 response += f"{i}. **{inv.get('invoice_number', 'N/A')}** - {inv.get('supplier_name', 'Unknown')}\n"1101 response += f" • Amount: ₹{inv.get('amount', 0):,.2f}\n"1102 response += f" • Date: {inv.get('date', 'N/A')}\n\n"1103 1104 if len(matching_invoices) > 5:1105 response += f"... and {len(matching_invoices) - 5} more results."1106 1107 return response1108 1109# ===============================================================================1110# STREAMLIT APPLICATION1111# ===============================================================================1112 1113def create_app():1114 """Main Streamlit application"""1115 1116 # Generate unique session ID for this run1117 if 'session_id' not in st.session_state:1118 st.session_state.session_id = str(uuid.uuid4())[:8]1119 1120 session_id = st.session_state.session_id1121 1122 # Custom CSS1123 st.markdown("""1124 <style>1125 .main-header {1126 font-size: 2.5rem;1127 font-weight: bold;1128 text-align: center;1129 color: #FF6B35;1130 margin-bottom: 1rem;1131 }1132 .feature-box {1133 background: linear-gradient(135deg, #667eea 0%, #764ba2 100%);1134 padding: 1rem;1135 border-radius: 10px;1136 color: white;1137 margin: 0.5rem 0;1138 text-align: center;1139 }1140 .status-ok { color: #28a745; font-weight: bold; }1141 .status-warning { color: #ffc107; font-weight: bold; }1142 .status-error { color: #dc3545; font-weight: bold; }1143 </style>1144 """, unsafe_allow_html=True)1145 1146 # Header1147 st.markdown('<h1 class="main-header">📄 AI Invoice Processing System</h1>', unsafe_allow_html=True)1148 st.markdown("""1149 <div style="text-align: center; margin-bottom: 2rem;">1150 <p style="font-size: 1.1rem; color: #666;">1151 AI-Powered Document Processing • Semantic Search • Smart Analytics • Hugging Face Spaces1152 </p>1153 </div>1154 """, unsafe_allow_html=True)1155 1156 # Initialize processor1157 if 'processor' not in st.session_state:1158 with st.spinner("🔧 Initializing AI Invoice Processor..."):1159 try:1160 st.session_state.processor = InvoiceProcessor()1161 st.session_state.chatbot = ChatBot(st.session_state.processor)1162 st.session_state.chat_history = []1163 st.success("✅ System initialized successfully!")1164 except Exception as e:1165 st.error(f"❌ Initialization failed: {e}")1166 st.stop()1167 1168 # Sidebar1169 with st.sidebar:1170 st.header("🎛️ System Status")1171 1172 processor = st.session_state.processor1173 1174 # Component status1175 if processor.document_processor.processors:1176 st.markdown('<span class="status-ok">✅ Document Processing</span>', unsafe_allow_html=True)1177 else:1178 st.markdown('<span class="status-error">❌ Document Processing</span>', unsafe_allow_html=True)1179 1180 if processor.ai_extractor.use_transformers:1181 st.markdown('<span class="status-ok">✅ AI Extraction</span>', unsafe_allow_html=True)1182 else:1183 st.markdown('<span class="status-warning">⚠️ Regex Extraction</span>', unsafe_allow_html=True)1184 1185 if processor.vector_store and processor.vector_store.embedding_model:1186 st.markdown('<span class="status-ok">✅ Semantic Search</span>', unsafe_allow_html=True)1187 else:1188 st.markdown('<span class="status-warning">⚠️ Keyword Search Only</span>', unsafe_allow_html=True)1189 1190 # Quick stats1191 st.header("📊 Quick Stats")1192 try:1193 data = processor.load_json_data()1194 total_invoices = len(data.get("invoices", []))1195 total_amount = data.get("summary", {}).get("total_amount", 0)1196 1197 st.metric("Total Invoices", total_invoices)1198 st.metric("Total Value", f"₹{total_amount:,.2f}")1199 st.metric("Success Rate", f"{processor.processing_stats['successful']}/{processor.processing_stats['total_processed']}")1200 