kernelmind/Resume-Screener-API
0
1"""2PDF parsing service with magic byte validation.3 4Why pdfplumber instead of PyPDF2?51. PyPDF2 chokes on multi-column layouts (very common in modern resumes)62. pdfplumber preserves reading order better73. Better handling of tables (experience sections often use tables)84. PyPDF2's text extraction is notoriously unreliable9 10Security note: We validate PDF magic bytes, not just file extension.11A renamed .exe or .html file won't pass validation.12"""13 14import io15import re16from typing import BinaryIO17 18import pdfplumber19 20from src.core.exceptions import InvalidFileError, PDFParsingError21from src.core.logging import logger22 23 24# PDF magic bytes - all PDFs start with this signature25# Fun fact: this translates to "%PDF" in ASCII26PDF_MAGIC_BYTES = b"%PDF"27 28# Max file size we'll process (10MB) - larger files are usually scanned images29# that won't parse well anyway30MAX_FILE_SIZE_BYTES = 10 * 1024 * 102431 32 33def validate_pdf_magic_bytes(file_content: bytes) -> bool:34 """35 Check if file starts with PDF magic bytes.36 37 This catches the classic "rename malware.exe to resume.pdf" attack.38 Most tutorials skip this, which is why AI-generated code often has this hole.39 40 Args:41 file_content: Raw bytes of the uploaded file42 43 Returns:44 True if valid PDF signature, False otherwise45 """46 if len(file_content) < 4:47 return False48 49 return file_content[:4] == PDF_MAGIC_BYTES50 51 52def clean_extracted_text(raw_text: str) -> str:53 """54 Clean and normalize extracted PDF text.55 56 PDFs have all kinds of weird whitespace artifacts.57 This function normalizes them for better LLM processing.58 """59 if not raw_text:60 return ""61 62 # Replace multiple spaces/tabs with single space63 text = re.sub(r"[ \t]+", " ", raw_text)64 65 # Replace 3+ newlines with double newline (preserve paragraph breaks)66 text = re.sub(r"\n{3,}", "\n\n", text)67 68 # Remove leading/trailing whitespace from each line69 lines = [line.strip() for line in text.split("\n")]70 text = "\n".join(lines)71 72 # Remove any null bytes (sometimes appear in corrupted PDFs)73 text = text.replace("\x00", "")74 75 # Final trim76 text = text.strip()77 78 return text79 80 81def parse_pdf(file_content: bytes, filename: str = "resume.pdf") -> str:82 """83 Parse PDF and extract cleaned text content.84 85 Args:86 file_content: Raw PDF bytes87 filename: Original filename (for error messages)88 89 Returns:90 Cleaned text content from PDF91 92 Raises:93 InvalidFileError: If file is not a valid PDF94 PDFParsingError: If PDF cannot be parsed95 """96 logger.info(f"Parsing PDF: {filename} ({len(file_content)} bytes)")97 98 # === Validation Layer ===99 100 # Check file size first (fast check)101 if len(file_content) > MAX_FILE_SIZE_BYTES:102 raise InvalidFileError(103 f"File too large: {len(file_content)} bytes. Max allowed: {MAX_FILE_SIZE_BYTES} bytes.",104 details={"filename": filename, "size": len(file_content)}105 )106 107 # Check magic bytes (security check)108 if not validate_pdf_magic_bytes(file_content):109 raise InvalidFileError(110 "File is not a valid PDF. Magic bytes mismatch.",111 details={112 "filename": filename,113 "expected_magic": PDF_MAGIC_BYTES.hex(),114 "actual_magic": file_content[:4].hex() if len(file_content) >= 4 else "too short"115 }116 )117 118 # === Extraction Layer ===119 120 try:121 # Wrap bytes in file-like object for pdfplumber122 pdf_file = io.BytesIO(file_content)123 124 all_text_parts: list[str] = []125 126 with pdfplumber.open(pdf_file) as pdf:127 logger.debug(f"PDF has {len(pdf.pages)} pages")128 129 for page_num, page in enumerate(pdf.pages, start=1):130 page_text = page.extract_text()131 132 if page_text:133 all_text_parts.append(page_text)134 logger.debug(f"Page {page_num}: extracted {len(page_text)} chars")135 else:136 # Scanned page with no text layer - common issue137 logger.warning(f"Page {page_num}: no text extracted (possibly scanned image)")138 139 if not all_text_parts:140 raise PDFParsingError(141 "No text could be extracted from PDF. It may be a scanned image without OCR.",142 details={"filename": filename}143 )144 145 # Join pages with double newline146 raw_text = "\n\n".join(all_text_parts)147 cleaned_text = clean_extracted_text(raw_text)148 149 logger.info(f"Successfully extracted {len(cleaned_text)} chars from {filename}")150 151 return cleaned_text152 153 except InvalidFileError:154 raise # Re-raise our own exceptions155 except PDFParsingError:156 raise157 except Exception as e:158 # Catch pdfplumber errors and wrap them159 logger.error(f"PDF parsing failed for {filename}: {e}")160 raise PDFParsingError(161 f"Failed to parse PDF: {str(e)}",162 details={"filename": filename, "error_type": type(e).__name__}163 )164 