Team Ai
Apppublic

kernelmind/Resume-Screener-API

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
pdf_parser.py164 linesDownload Raw Back to services
1"""2PDF parsing service with magic byte validation.3 4Why pdfplumber instead of PyPDF2?51. PyPDF2 chokes on multi-column layouts (very common in modern resumes)62. pdfplumber preserves reading order better73. Better handling of tables (experience sections often use tables)84. PyPDF2's text extraction is notoriously unreliable9 10Security note: We validate PDF magic bytes, not just file extension.11A renamed .exe or .html file won't pass validation.12"""13 14import io15import re16from typing import BinaryIO17 18import pdfplumber19 20from src.core.exceptions import InvalidFileError, PDFParsingError21from src.core.logging import logger22 23 24# PDF magic bytes - all PDFs start with this signature25# Fun fact: this translates to "%PDF" in ASCII26PDF_MAGIC_BYTES = b"%PDF"27 28# Max file size we'll process (10MB) - larger files are usually scanned images29# that won't parse well anyway30MAX_FILE_SIZE_BYTES = 10 * 1024 * 102431 32 33def validate_pdf_magic_bytes(file_content: bytes) -> bool:34    """35    Check if file starts with PDF magic bytes.36    37    This catches the classic "rename malware.exe to resume.pdf" attack.38    Most tutorials skip this, which is why AI-generated code often has this hole.39    40    Args:41        file_content: Raw bytes of the uploaded file42        43    Returns:44        True if valid PDF signature, False otherwise45    """46    if len(file_content) < 4:47        return False48    49    return file_content[:4] == PDF_MAGIC_BYTES50 51 52def clean_extracted_text(raw_text: str) -> str:53    """54    Clean and normalize extracted PDF text.55    56    PDFs have all kinds of weird whitespace artifacts.57    This function normalizes them for better LLM processing.58    """59    if not raw_text:60        return ""61    62    # Replace multiple spaces/tabs with single space63    text = re.sub(r"[ \t]+", " ", raw_text)64    65    # Replace 3+ newlines with double newline (preserve paragraph breaks)66    text = re.sub(r"\n{3,}", "\n\n", text)67    68    # Remove leading/trailing whitespace from each line69    lines = [line.strip() for line in text.split("\n")]70    text = "\n".join(lines)71    72    # Remove any null bytes (sometimes appear in corrupted PDFs)73    text = text.replace("\x00", "")74    75    # Final trim76    text = text.strip()77    78    return text79 80 81def parse_pdf(file_content: bytes, filename: str = "resume.pdf") -> str:82    """83    Parse PDF and extract cleaned text content.84    85    Args:86        file_content: Raw PDF bytes87        filename: Original filename (for error messages)88        89    Returns:90        Cleaned text content from PDF91        92    Raises:93        InvalidFileError: If file is not a valid PDF94        PDFParsingError: If PDF cannot be parsed95    """96    logger.info(f"Parsing PDF: {filename} ({len(file_content)} bytes)")97    98    # === Validation Layer ===99    100    # Check file size first (fast check)101    if len(file_content) > MAX_FILE_SIZE_BYTES:102        raise InvalidFileError(103            f"File too large: {len(file_content)} bytes. Max allowed: {MAX_FILE_SIZE_BYTES} bytes.",104            details={"filename": filename, "size": len(file_content)}105        )106    107    # Check magic bytes (security check)108    if not validate_pdf_magic_bytes(file_content):109        raise InvalidFileError(110            "File is not a valid PDF. Magic bytes mismatch.",111            details={112                "filename": filename,113                "expected_magic": PDF_MAGIC_BYTES.hex(),114                "actual_magic": file_content[:4].hex() if len(file_content) >= 4 else "too short"115            }116        )117    118    # === Extraction Layer ===119    120    try:121        # Wrap bytes in file-like object for pdfplumber122        pdf_file = io.BytesIO(file_content)123        124        all_text_parts: list[str] = []125        126        with pdfplumber.open(pdf_file) as pdf:127            logger.debug(f"PDF has {len(pdf.pages)} pages")128            129            for page_num, page in enumerate(pdf.pages, start=1):130                page_text = page.extract_text()131                132                if page_text:133                    all_text_parts.append(page_text)134                    logger.debug(f"Page {page_num}: extracted {len(page_text)} chars")135                else:136                    # Scanned page with no text layer - common issue137                    logger.warning(f"Page {page_num}: no text extracted (possibly scanned image)")138        139        if not all_text_parts:140            raise PDFParsingError(141                "No text could be extracted from PDF. It may be a scanned image without OCR.",142                details={"filename": filename}143            )144        145        # Join pages with double newline146        raw_text = "\n\n".join(all_text_parts)147        cleaned_text = clean_extracted_text(raw_text)148        149        logger.info(f"Successfully extracted {len(cleaned_text)} chars from {filename}")150        151        return cleaned_text152        153    except InvalidFileError:154        raise  # Re-raise our own exceptions155    except PDFParsingError:156        raise157    except Exception as e:158        # Catch pdfplumber errors and wrap them159        logger.error(f"PDF parsing failed for {filename}: {e}")160        raise PDFParsingError(161            f"Failed to parse PDF: {str(e)}",162            details={"filename": filename, "error_type": type(e).__name__}163        )164