Team Ai
Apppublic

documentExtractionag051/ExtractDocument

sourceHugging Faceupdated 9mo agoView on Hugging Face
0likes
pdf_utils.py526 linesDownload Raw Back to root
1"""2PDF Text Extraction Utility Module3 4Extracts text and bounding box coordinates from digital/searchable PDFs.5Uses pdfplumber as primary library with pypdf as fallback.6Converts word-level extractions into properly formatted line segments.7 8Output JSON Schema:9{10    "version": "1.0",11    "metadata": {...},12    "pages": [{"pageNum": 1, "width": ..., "height": ...}],13    "ocrBlocks": [...]14}15"""16 17import uuid18import io19from typing import Dict, List, Any, Set20 21# Primary library22try:23    import pdfplumber24    PDFPLUMBER_AVAILABLE = True25except ImportError:26    PDFPLUMBER_AVAILABLE = False27 28# Fallback library29try:30    from pypdf import PdfReader31    PYPDF_AVAILABLE = True32except ImportError:33    PYPDF_AVAILABLE = False34 35 36# =============================================================================37# Constants38# =============================================================================39 40DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD = 0.03  # 3% of page width41DEFAULT_SCALE_WIDTH = 2479  # Target width for consistent scaling42 43 44# =============================================================================45# Helper Functions46# =============================================================================47 48def _generate_uuid() -> str:49    """Generate a unique uppercase UUID for block identification."""50    return str(uuid.uuid4()).upper()51 52 53def _intersection_pct(rect1: Dict[str, Any], rect2: Dict[str, Any]) -> int:54    """55    Calculate the percentage of intersection between two rectangles.56    57    Args:58        rect1: First rectangle with x1, y1, x2, y259        rect2: Second rectangle with x1, y1, x2, y260        61    Returns:62        Intersection percentage (0-100)63    """64    x_left = max(rect1['x1'], rect2['x1'])65    y_top = max(rect1['y1'], rect2['y1'])66    x_right = min(rect1['x2'], rect2['x2'])67    y_bottom = min(rect1['y2'], rect2['y2'])68    69    if x_right < x_left or y_bottom < y_top:70        return 071    72    intersection_area = (x_right - x_left) * (y_bottom - y_top)73    rect1_area = (rect1['x2'] - rect1['x1']) * (rect1['y2'] - rect1['y1'])74    75    if rect1_area == 0:76        return 077    78    return int((intersection_area / rect1_area) * 100)79 80 81# =============================================================================82# PDF Digital Check83# =============================================================================84 85def is_pdf_digital(pdf_path: str) -> bool:86    """Check if PDF has embedded text (searchable/digital PDF)."""87    if PDFPLUMBER_AVAILABLE:88        try:89            with pdfplumber.open(pdf_path) as pdf:90                for page in pdf.pages[:min(3, len(pdf.pages))]:91                    text = page.extract_text()92                    if text and text.strip():93                        return True94            return False95        except Exception:96            pass97    98    if PYPDF_AVAILABLE:99        try:100            reader = PdfReader(pdf_path)101            for page in reader.pages[:min(3, len(reader.pages))]:102                text = page.extract_text()103                if text and text.strip():104                    return True105            return False106        except Exception:107            pass108    109    return False110 111 112# =============================================================================113# Line Detection & Processing114# =============================================================================115 116def _merge_lines(all_lines: List[List[Dict[str, Any]]]) -> List[List[Dict[str, Any]]]:117    """Merge lines that may overlap vertically."""118    used_ids: Set[int] = set()119    new_page_lines: List[List[Dict[str, Any]]] = []120    121    for i, page_line in enumerate(all_lines):122        if i not in used_ids and page_line:123            new_line = page_line.copy()124            y1 = page_line[-1]['geometry']['y1']125            y2 = page_line[-1]['geometry']['y2']126            127            for j, page_line_subsequent in enumerate(all_lines):128                if page_line_subsequent:129                    last_block = page_line_subsequent[-1]130                    if (i != j and131                        y1 <= last_block['geometry']['y1'] and132                        y2 >= last_block['geometry']['y2'] and133                        j not in used_ids):134                        used_ids.add(j)135                        new_line.extend(page_line_subsequent)136            137            new_line.sort(key=lambda block: block['geometry']['x1'])138            used_ids.add(i)139            new_page_lines.append(new_line)140    141    return new_page_lines142 143 144def _sort_words_in_reading_order(145    blocks: List[Dict[str, Any]],146    height: int,147    width: int148) -> List[Dict[str, Any]]:149    """Sort words in reading order (top to bottom, left to right)."""150    if not blocks:151        return []152    153    max_words = 1000154    y_axis_intersection_thresh = 65155    alternate_thresh = 90156    157    if len(blocks) >= max_words:158        return sorted(blocks, key=lambda b: (b['geometry']['y1'], b['geometry']['x1']))159    160    all_lines: List[List[Dict[str, Any]]] = []161    used_ids: Set[str] = set()162    163    for seed_block in blocks:164        if seed_block['id'] not in used_ids:165            line: List[Dict[str, Any]] = []166            for block in blocks:167                if block['id'] not in used_ids:168                    new_rect_y_axis = {169                        'x1': seed_block['geometry']['x1'],170                        'x2': seed_block['geometry']['x2'],171                        'y1': block['geometry']['y1'],172                        'y2': block['geometry']['y2']173                    }174                    175                    thresh1 = _intersection_pct(seed_block['geometry'], new_rect_y_axis)176                    thresh2 = _intersection_pct(new_rect_y_axis, seed_block['geometry'])177                    178                    if thresh1 >= y_axis_intersection_thresh and thresh2 >= y_axis_intersection_thresh:179                        line.append(block)180                        used_ids.add(block['id'])181                    else:182                        seed_center_x = (seed_block['geometry']['x1'] + seed_block['geometry']['x2']) / (2 * width)183                        seed_center_y = (seed_block['geometry']['y1'] + seed_block['geometry']['y2']) / (2 * height)184                        block_center_x = (block['geometry']['x1'] + block['geometry']['x2']) / (2 * width)185                        block_center_y = (block['geometry']['y1'] + block['geometry']['y2']) / (2 * height)186                        187                        dist = ((seed_center_x - block_center_x)**2 + (seed_center_y - block_center_y)**2)**0.5188                        189                        if dist <= 0.2 and (thresh1 >= alternate_thresh or thresh2 >= alternate_thresh):190                            line.append(block)191                            used_ids.add(block['id'])192            193            if line:194                all_lines.append(line)195    196    if all_lines:197        all_lines = _merge_lines(all_lines)198        all_lines.sort(key=lambda line: line[0]['geometry']['y1'] if line else 0)199    200    all_sorted_lines: List[Dict[str, Any]] = []201    for line in all_lines:202        all_sorted_lines.extend(line)203    204    return all_sorted_lines205 206 207def _identify_line_segments(208    blocks: List[Dict[str, Any]],209    width: int,210    line_segment_merge_threshold: float = 0.03211) -> List[List[Dict[str, Any]]]:212    """Identify line segments as groups of related words."""213    added_blocks: Set[str] = set()214    line_segments: List[List[Dict[str, Any]]] = []215    216    for seed_block in blocks:217        if seed_block.get('blockType') != 'WORD':218            continue219        220        if seed_block['id'] not in added_blocks:221            line_segment = [seed_block]222            added_blocks.add(seed_block['id'])223            224            for block in blocks:225                if block.get('blockType') != 'WORD' or block['id'] == seed_block['id'] or block['id'] in added_blocks:226                    continue227                228                vert_mid_point = (block['geometry']['y1'] + (block['geometry']['y2'] - block['geometry']['y1']) / 2)229                230                if seed_block['geometry']['y1'] < vert_mid_point < seed_block['geometry']['y2']:231                    dist_block = line_segment[-1]232                    horizontal_gap = block['geometry']['x1'] - dist_block['geometry']['x2']233                    gap_ratio = horizontal_gap / width if width > 0 else 0234                    235                    if gap_ratio < line_segment_merge_threshold and block['geometry']['x1'] > dist_block['geometry']['x2']:236                        line_segment.append(block)237                        added_blocks.add(block['id'])238            239            if line_segment:240                line_segments.append(line_segment)241    242    return line_segments243 244 245def _format_line_segments(246    line_segments: List[List[Dict[str, Any]]],247    page_num: int248) -> List[Dict[str, Any]]:249    """Format line segments into line blocks."""250    formatted_segments: List[Dict[str, Any]] = []251    252    for line_segment in line_segments:253        if not line_segment:254            continue255        256        segment_text = ' '.join([block['text'] for block in line_segment]).strip()257        if not segment_text:258            continue259        260        confidences = [block.get('confidence', 1.0) for block in line_segment]261        confidence = sum(confidences) / len(confidences) if confidences else 1.0262        263        x1 = min(block['geometry']['x1'] for block in line_segment)264        y1 = min(block['geometry']['y1'] for block in line_segment)265        x2 = max(block['geometry']['x2'] for block in line_segment)266        y2 = max(block['geometry']['y2'] for block in line_segment)267        268        formatted_segments.append({269            "id": _generate_uuid(),270            "pageNum": page_num,271            "blockType": "LINE",272            "text": segment_text,273            "confidence": confidence,274            "geometry": {"x1": x1, "y1": y1, "x2": x2, "y2": y2}275        })276    277    return formatted_segments278 279 280def _post_process_words_to_lines(281    word_blocks: List[Dict[str, Any]],282    page_width: int,283    page_height: int,284    page_num: int,285    line_segment_merge_threshold: float = 0.03286) -> List[Dict[str, Any]]:287    """Post-process word blocks to create properly separated line blocks."""288    if not word_blocks:289        return []290    291    for block in word_blocks:292        if 'id' not in block:293            block['id'] = _generate_uuid()294    295    sorted_blocks = _sort_words_in_reading_order(word_blocks, page_height, page_width)296    line_segments = _identify_line_segments(sorted_blocks, page_width, line_segment_merge_threshold)297    line_blocks = _format_line_segments(line_segments, page_num)298    299    return line_blocks300 301 302# =============================================================================303# Main Extraction Function304# =============================================================================305 306def extract_text_from_pdf(307    pdf_path: str,308    line_segment_merge_threshold: float = DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD309) -> Dict[str, Any]:310    """311    Main entry point for PDF text extraction.312    313    Args:314        pdf_path: Path to the PDF file315        line_segment_merge_threshold: Gap threshold for line merging (default: 0.03)316        317    Returns:318        Structured output dict with metadata and pages containing ocrBlocks319    """320    import os321    filename = os.path.basename(pdf_path)322    323    # Check if PDF is digital324    if not is_pdf_digital(pdf_path):325        return _create_empty_output(filename, line_segment_merge_threshold)326    327    if not PDFPLUMBER_AVAILABLE:328        raise RuntimeError("pdfplumber is required. Install with: pip install pdfplumber")329    330    # Initialize result structure with ocrBlocks at root level331    result: Dict[str, Any] = {332        "version": "1.0",333        "metadata": {334            "documentId": "pdf-extracted",335            "documentName": filename,336            "source": "pdfplumber",337            "numberOfPages": 0,338            "lineSegmentMergeThreshold": line_segment_merge_threshold339        },340        "pages": [],341        "ocrBlocks": []342    }343    344    with pdfplumber.open(pdf_path) as pdf:345        result["metadata"]["numberOfPages"] = len(pdf.pages)346        347        for page_num, page in enumerate(pdf.pages, start=1):348            # Scale to consistent dimensions349            scale_factor = DEFAULT_SCALE_WIDTH / page.width if page.width > 0 else 1350            page_width = int(page.width * scale_factor)351            page_height = int(page.height * scale_factor)352            353            # Add page info (without ocrBlocks)354            result["pages"].append({355                "pageNum": page_num,356                "width": page_width,357                "height": page_height358            })359            360            # Extract words with tighter tolerances361            words = page.extract_words(362                keep_blank_chars=False,363                x_tolerance=2,364                y_tolerance=2365            )366            367            # Convert to word blocks368            word_blocks: List[Dict[str, Any]] = []369            for word in words:370                word_block = {371                    "id": _generate_uuid(),372                    "pageNum": page_num,373                    "blockType": "WORD",374                    "text": word["text"],375                    "confidence": 1.0,376                    "geometry": {377                        "x1": int(word["x0"] * scale_factor),378                        "y1": int(word["top"] * scale_factor),379                        "x2": int(word["x1"] * scale_factor),380                        "y2": int(word["bottom"] * scale_factor)381                    }382                }383                word_blocks.append(word_block)384                result["ocrBlocks"].append(word_block)385            386            # Create line blocks387            line_blocks = _post_process_words_to_lines(388                word_blocks,389                page_width,390                page_height,391                page_num,392                line_segment_merge_threshold393            )394            395            # Add line blocks to root ocrBlocks396            result["ocrBlocks"].extend(line_blocks)397    398    return result399 400 401def extract_text_from_pdf_bytes(402    pdf_bytes: bytes,403    file_name: str,404    line_segment_merge_threshold: float = DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD405) -> Dict[str, Any]:406    """407    Extract text from in-memory PDF bytes (for file uploads).408    409    Args:410        pdf_bytes: PDF file content as bytes411        file_name: Name of the file (for metadata)412        line_segment_merge_threshold: Gap threshold for line merging413        414    Returns:415        Structured output dict with metadata and pages containing ocrBlocks416    """417    if not PDFPLUMBER_AVAILABLE:418        raise RuntimeError("pdfplumber is required. Install with: pip install pdfplumber")419    420    result: Dict[str, Any] = {421        "version": "1.0",422        "metadata": {423            "documentId": "pdf-extracted",424            "documentName": file_name,425            "source": "pdfplumber",426            "numberOfPages": 0,427            "lineSegmentMergeThreshold": line_segment_merge_threshold428        },429        "pages": [],430        "ocrBlocks": []431    }432    433    pdf_stream = io.BytesIO(pdf_bytes)434    435    with pdfplumber.open(pdf_stream) as pdf:436        result["metadata"]["numberOfPages"] = len(pdf.pages)437        438        for page_num, page in enumerate(pdf.pages, start=1):439            scale_factor = DEFAULT_SCALE_WIDTH / page.width if page.width > 0 else 1440            page_width = int(page.width * scale_factor)441            page_height = int(page.height * scale_factor)442            443            # Add page info (without ocrBlocks)444            result["pages"].append({445                "pageNum": page_num,446                "width": page_width,447                "height": page_height448            })449            450            words = page.extract_words(451                keep_blank_chars=False,452                x_tolerance=2,453                y_tolerance=2454            )455            456            word_blocks: List[Dict[str, Any]] = []457            for word in words:458                word_block = {459                    "id": _generate_uuid(),460                    "pageNum": page_num,461                    "blockType": "WORD",462                    "text": word["text"],463                    "confidence": 1.0,464                    "geometry": {465                        "x1": int(word["x0"] * scale_factor),466                        "y1": int(word["top"] * scale_factor),467                        "x2": int(word["x1"] * scale_factor),468                        "y2": int(word["bottom"] * scale_factor)469                    }470                }471                word_blocks.append(word_block)472                result["ocrBlocks"].append(word_block)473            474            line_blocks = _post_process_words_to_lines(475                word_blocks,476                page_width,477                page_height,478                page_num,479                line_segment_merge_threshold480            )481            482            # Add line blocks to root ocrBlocks483            result["ocrBlocks"].extend(line_blocks)484    485    return result486 487 488def _create_empty_output(filename: str, line_segment_merge_threshold: float) -> Dict[str, Any]:489    """Create empty output structure for non-digital PDFs."""490    return {491        "version": "1.0",492        "metadata": {493            "documentId": "pdf-extracted",494            "documentName": filename,495            "source": "none",496            "numberOfPages": 0,497            "lineSegmentMergeThreshold": line_segment_merge_threshold498        },499        "pages": [],500        "ocrBlocks": []501    }502 503 504# =============================================================================505# CLI Entry Point506# =============================================================================507 508if __name__ == '__main__':509    import sys510    import json511    512    if len(sys.argv) < 2:513        print("Usage: python pdf_utils.py <pdf_path> [threshold]")514        print("  threshold: line segment merge threshold (default: 0.03)")515        sys.exit(1)516    517    pdf_path = sys.argv[1]518    threshold = float(sys.argv[2]) if len(sys.argv) > 2 else DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD519    520    print(f"Extracting text from: {pdf_path}")521    print(f"Line segment merge threshold: {threshold}")522    print("-" * 50)523    524    result = extract_text_from_pdf(pdf_path, threshold)525    print(json.dumps(result, indent=2))526