sinhapiyush86/convAI
0
1#!/usr/bin/env python32"""3# Simplified PDF Processor for Hugging Face Spaces4 5This module provides comprehensive PDF processing functionality for the RAG system.6 7## Overview8 9The PDF processor handles the complete pipeline from raw PDF files to structured,10searchable document chunks. It includes:11 12- **Text Extraction**: Robust PDF text extraction with error handling13- **Text Cleaning**: Intelligent preprocessing and normalization14- **Metadata Extraction**: Document title, author, and file information15- **Smart Chunking**: Multiple chunk sizes for optimal retrieval16- **Query Preprocessing**: Text normalization for search queries17 18## Key Features19 20- ๐ **Multi-format Support**: Handles various PDF structures and layouts21- ๐งน **Intelligent Cleaning**: Removes noise while preserving important content22- ๐ **Flexible Chunking**: Multiple chunk sizes for different use cases23- ๐ **Search Optimization**: Preprocessing for better retrieval performance24- ๐ก๏ธ **Error Handling**: Graceful handling of corrupted or problematic files25 26## Architecture27 28The processor follows a modular design:291. **Text Extraction**: Raw PDF to text conversion302. **Text Cleaning**: Noise removal and normalization313. **Metadata Extraction**: Document information extraction324. **Chunking**: Intelligent text segmentation335. **Query Processing**: Search query optimization34 35## Usage Example36 37```python38processor = SimplePDFProcessor()39processed_doc = processor.process_document("document.pdf", [100, 400])40print(f"Processed {len(processed_doc.chunks)} chunks")41```42"""43 44import os45import re46import uuid47from typing import List, Dict, Optional48from dataclasses import dataclass49from pathlib import Path50import pypdf51from loguru import logger52 53 54# =============================================================================55# DATA STRUCTURES56# =============================================================================57 58 59@dataclass60class DocumentChunk:61 """62 Represents a processed document chunk with metadata63 64 Attributes:65 text: The cleaned and processed text content66 doc_id: Unique identifier for the source document67 filename: Name of the source PDF file68 chunk_id: Unique identifier for this specific chunk69 chunk_size: Target size used for chunking (in tokens)70 """71 72 text: str73 doc_id: str74 filename: str75 chunk_id: str76 chunk_size: int77 78 79@dataclass80class ProcessedDocument:81 """82 Represents a completely processed PDF document83 84 Attributes:85 filename: Name of the PDF file86 title: Extracted or inferred document title87 author: Extracted or inferred document author88 chunks: List of processed document chunks89 """90 91 filename: str92 title: str93 author: str94 chunks: List[DocumentChunk]95 96 97# =============================================================================98# MAIN PDF PROCESSOR CLASS99# =============================================================================100 101 102class SimplePDFProcessor:103 """104 Simplified PDF processor for Hugging Face Spaces105 106 This class provides comprehensive PDF processing capabilities including:107 - Text extraction and cleaning108 - Metadata extraction109 - Intelligent chunking110 - Query preprocessing111 - Error handling and logging112 """113 114 def __init__(self):115 """116 Initialize the PDF processor with default settings117 118 Sets up stop words and processing parameters for optimal119 document processing and search performance.120 """121 # Common English stop words for query preprocessing122 self.stop_words = {123 "the",124 "a",125 "an",126 "and",127 "or",128 "but",129 "in",130 "on",131 "at",132 "to",133 "for",134 "of",135 "with",136 "by",137 "is",138 "are",139 "was",140 "were",141 "be",142 "been",143 "being",144 "have",145 "has",146 "had",147 "do",148 "does",149 "did",150 "will",151 "would",152 "could",153 "should",154 "may",155 "might",156 "can",157 "this",158 "that",159 "these",160 "those",161 }162 163 def process_document(164 self, file_path: str, chunk_sizes: List[int] = None165 ) -> ProcessedDocument:166 """167 Process a PDF document through the complete pipeline168 169 This method orchestrates the entire PDF processing workflow:170 1. Extracts text from the PDF file171 2. Cleans and normalizes the text172 3. Extracts document metadata173 4. Creates chunks of different sizes174 5. Returns a structured document object175 176 Args:177 file_path: Path to the PDF file to process178 chunk_sizes: List of chunk sizes to create (in tokens)179 180 Returns:181 ProcessedDocument object with metadata and chunks182 183 Raises:184 Exception: If document processing fails185 """186 if chunk_sizes is None:187 chunk_sizes = [100, 400] # Default chunk sizes188 189 try:190 # Step 1: Extract raw text from PDF191 text = self._extract_text(file_path)192 193 # Step 2: Clean and normalize the text194 cleaned_text = self._clean_text(text)195 196 # Step 3: Extract document metadata197 metadata = self._extract_metadata(file_path)198 199 # Step 4: Create chunks of different sizes200 chunks = []201 doc_id = str(uuid.uuid4()) # Generate unique document ID202 203 for chunk_size in chunk_sizes:204 chunk_list = self._create_chunks(205 cleaned_text, chunk_size, doc_id, metadata["filename"]206 )207 chunks.extend(chunk_list)208 209 # Step 5: Return processed document210 return ProcessedDocument(211 filename=metadata["filename"],212 title=metadata["title"],213 author=metadata["author"],214 chunks=chunks,215 )216 217 except Exception as e:218 logger.error(f"Error processing document {file_path}: {e}")219 raise220 221 def _extract_text(self, file_path: str) -> str:222 """223 Extract text content from a PDF file224 225 This method:226 1. Opens the PDF file safely227 2. Iterates through all pages228 3. Extracts text from each page229 4. Combines all text with proper spacing230 5. Handles extraction errors gracefully231 232 Args:233 file_path: Path to the PDF file234 235 Returns:236 Extracted text content as a string237 238 Raises:239 Exception: If text extraction fails240 """241 try:242 with open(file_path, "rb") as file:243 # Create PDF reader object244 pdf_reader = pypdf.PdfReader(file)245 text = ""246 247 # Extract text from each page248 for page in pdf_reader.pages:249 page_text = page.extract_text()250 if page_text:251 text += page_text + "\n"252 253 return text254 255 except Exception as e:256 logger.error(f"Error extracting text from {file_path}: {e}")257 raise258 259 def _clean_text(self, text: str) -> str:260 """261 Clean and normalize extracted text262 263 This method performs comprehensive text cleaning:264 1. Removes excessive whitespace and newlines265 2. Normalizes special characters while preserving punctuation266 3. Removes page numbers and headers/footers267 4. Ensures consistent formatting268 269 Args:270 text: Raw extracted text from PDF271 272 Returns:273 Cleaned and normalized text274 """275 # Remove excessive whitespace (multiple spaces, tabs, etc.)276 text = re.sub(r"\s+", " ", text)277 278 # Remove special characters but preserve important punctuation279 # This keeps: letters, numbers, spaces, and common punctuation280 text = re.sub(r"[^\w\s\.\,\!\?\;\:\-\(\)\[\]\{\}]", "", text)281 282 # Remove standalone page numbers at line ends283 # These are often artifacts from PDF extraction284 text = re.sub(r"\b\d+\b(?=\s*\n)", "", text)285 286 # Normalize excessive newlines to consistent paragraph breaks287 text = re.sub(r"\n\s*\n\s*\n+", "\n\n", text)288 289 return text.strip()290 291 def _extract_metadata(self, file_path: str) -> Dict[str, str]:292 """293 Extract metadata from PDF file294 295 This method attempts to extract:296 1. Document title from PDF metadata297 2. Author information from PDF metadata298 3. Falls back to filename if metadata is unavailable299 300 Args:301 file_path: Path to the PDF file302 303 Returns:304 Dictionary containing filename, title, and author305 """306 try:307 with open(file_path, "rb") as file:308 pdf_reader = pypdf.PdfReader(file)309 info = pdf_reader.metadata310 311 return {312 "filename": Path(file_path).name,313 "title": (314 info.get("/Title", Path(file_path).stem)315 if info316 else Path(file_path).stem317 ),318 "author": info.get("/Author", "Unknown") if info else "Unknown",319 }320 321 except Exception as e:322 logger.warning(f"Error extracting metadata from {file_path}: {e}")323 # Fallback to basic information324 return {325 "filename": Path(file_path).name,326 "title": Path(file_path).stem,327 "author": "Unknown",328 }329 330 def _create_chunks(331 self, text: str, chunk_size: int, doc_id: str, filename: str332 ) -> List[DocumentChunk]:333 """334 Create text chunks of specified size335 336 This method implements intelligent chunking:337 1. Splits text into sentences for natural boundaries338 2. Groups sentences into chunks of target size339 3. Ensures chunks don't exceed the specified token limit340 4. Creates unique identifiers for each chunk341 342 Args:343 text: Clean text to chunk344 chunk_size: Target chunk size in tokens345 doc_id: Unique document identifier346 filename: Source filename347 348 Returns:349 List of DocumentChunk objects350 """351 chunks = []352 353 # Split text into sentences for natural chunking354 sentences = self._split_into_sentences(text)355 356 current_chunk = ""357 chunk_id = 0358 359 for sentence in sentences:360 # Estimate token count (rough approximation using word count)361 estimated_tokens = len(sentence.split())362 363 # Add sentence to current chunk if it fits364 if len(current_chunk.split()) + estimated_tokens <= chunk_size:365 current_chunk += sentence + " "366 else:367 # Save current chunk if not empty368 if current_chunk.strip():369 chunks.append(370 DocumentChunk(371 text=current_chunk.strip(),372 doc_id=doc_id,373 filename=filename,374 chunk_id=f"{doc_id}_{chunk_id}",375 chunk_size=chunk_size,376 )377 )378 chunk_id += 1379 380 # Start new chunk with current sentence381 current_chunk = sentence + " "382 383 # Add the last chunk if not empty384 if current_chunk.strip():385 chunks.append(386 DocumentChunk(387 text=current_chunk.strip(),388 doc_id=doc_id,389 filename=filename,390 chunk_id=f"{doc_id}_{chunk_id}",391 chunk_size=chunk_size,392 )393 )394 395 return chunks396 397 def _split_into_sentences(self, text: str) -> List[str]:398 """399 Split text into sentences for intelligent chunking400 401 This method:402 1. Uses regex patterns to identify sentence boundaries403 2. Filters out very short sentences (likely noise)404 3. Ensures minimum sentence quality405 406 Args:407 text: Text to split into sentences408 409 Returns:410 List of sentence strings411 """412 # Split on sentence-ending punctuation413 sentences = re.split(r"[.!?]+", text)414 415 # Clean and filter sentences416 cleaned_sentences = []417 for sentence in sentences:418 sentence = sentence.strip()419 # Only include sentences with meaningful content (minimum 3 words)420 if sentence and len(sentence.split()) > 3:421 cleaned_sentences.append(sentence)422 423 return cleaned_sentences424 425 def preprocess_query(self, query: str) -> str:426 """427 Preprocess query text for better search performance428 429 This method applies text normalization techniques:430 1. Converts to lowercase for case-insensitive matching431 2. Removes punctuation that might interfere with search432 3. Filters out common stop words433 4. Returns normalized query string434 435 Args:436 query: Raw query string from user437 438 Returns:439 Preprocessed query string optimized for search440 """441 # Convert to lowercase for consistent matching442 query = query.lower()443 444 # Remove punctuation that might interfere with search445 query = re.sub(r"[^\w\s]", "", query)446 447 # Remove stop words to focus on meaningful terms448 words = query.split()449 filtered_words = [word for word in words if word not in self.stop_words]450 451 return " ".join(filtered_words)452 