documentExtractionag051/ExtractDocument
0
1"""2PDF Text Extraction Utility Module3 4Extracts text and bounding box coordinates from digital/searchable PDFs.5Uses pdfplumber as primary library with pypdf as fallback.6Converts word-level extractions into properly formatted line segments.7 8Output JSON Schema:9{10 "version": "1.0",11 "metadata": {...},12 "pages": [{"pageNum": 1, "width": ..., "height": ...}],13 "ocrBlocks": [...]14}15"""16 17import uuid18import io19from typing import Dict, List, Any, Set20 21# Primary library22try:23 import pdfplumber24 PDFPLUMBER_AVAILABLE = True25except ImportError:26 PDFPLUMBER_AVAILABLE = False27 28# Fallback library29try:30 from pypdf import PdfReader31 PYPDF_AVAILABLE = True32except ImportError:33 PYPDF_AVAILABLE = False34 35 36# =============================================================================37# Constants38# =============================================================================39 40DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD = 0.03 # 3% of page width41DEFAULT_SCALE_WIDTH = 2479 # Target width for consistent scaling42 43 44# =============================================================================45# Helper Functions46# =============================================================================47 48def _generate_uuid() -> str:49 """Generate a unique uppercase UUID for block identification."""50 return str(uuid.uuid4()).upper()51 52 53def _intersection_pct(rect1: Dict[str, Any], rect2: Dict[str, Any]) -> int:54 """55 Calculate the percentage of intersection between two rectangles.56 57 Args:58 rect1: First rectangle with x1, y1, x2, y259 rect2: Second rectangle with x1, y1, x2, y260 61 Returns:62 Intersection percentage (0-100)63 """64 x_left = max(rect1['x1'], rect2['x1'])65 y_top = max(rect1['y1'], rect2['y1'])66 x_right = min(rect1['x2'], rect2['x2'])67 y_bottom = min(rect1['y2'], rect2['y2'])68 69 if x_right < x_left or y_bottom < y_top:70 return 071 72 intersection_area = (x_right - x_left) * (y_bottom - y_top)73 rect1_area = (rect1['x2'] - rect1['x1']) * (rect1['y2'] - rect1['y1'])74 75 if rect1_area == 0:76 return 077 78 return int((intersection_area / rect1_area) * 100)79 80 81# =============================================================================82# PDF Digital Check83# =============================================================================84 85def is_pdf_digital(pdf_path: str) -> bool:86 """Check if PDF has embedded text (searchable/digital PDF)."""87 if PDFPLUMBER_AVAILABLE:88 try:89 with pdfplumber.open(pdf_path) as pdf:90 for page in pdf.pages[:min(3, len(pdf.pages))]:91 text = page.extract_text()92 if text and text.strip():93 return True94 return False95 except Exception:96 pass97 98 if PYPDF_AVAILABLE:99 try:100 reader = PdfReader(pdf_path)101 for page in reader.pages[:min(3, len(reader.pages))]:102 text = page.extract_text()103 if text and text.strip():104 return True105 return False106 except Exception:107 pass108 109 return False110 111 112# =============================================================================113# Line Detection & Processing114# =============================================================================115 116def _merge_lines(all_lines: List[List[Dict[str, Any]]]) -> List[List[Dict[str, Any]]]:117 """Merge lines that may overlap vertically."""118 used_ids: Set[int] = set()119 new_page_lines: List[List[Dict[str, Any]]] = []120 121 for i, page_line in enumerate(all_lines):122 if i not in used_ids and page_line:123 new_line = page_line.copy()124 y1 = page_line[-1]['geometry']['y1']125 y2 = page_line[-1]['geometry']['y2']126 127 for j, page_line_subsequent in enumerate(all_lines):128 if page_line_subsequent:129 last_block = page_line_subsequent[-1]130 if (i != j and131 y1 <= last_block['geometry']['y1'] and132 y2 >= last_block['geometry']['y2'] and133 j not in used_ids):134 used_ids.add(j)135 new_line.extend(page_line_subsequent)136 137 new_line.sort(key=lambda block: block['geometry']['x1'])138 used_ids.add(i)139 new_page_lines.append(new_line)140 141 return new_page_lines142 143 144def _sort_words_in_reading_order(145 blocks: List[Dict[str, Any]],146 height: int,147 width: int148) -> List[Dict[str, Any]]:149 """Sort words in reading order (top to bottom, left to right)."""150 if not blocks:151 return []152 153 max_words = 1000154 y_axis_intersection_thresh = 65155 alternate_thresh = 90156 157 if len(blocks) >= max_words:158 return sorted(blocks, key=lambda b: (b['geometry']['y1'], b['geometry']['x1']))159 160 all_lines: List[List[Dict[str, Any]]] = []161 used_ids: Set[str] = set()162 163 for seed_block in blocks:164 if seed_block['id'] not in used_ids:165 line: List[Dict[str, Any]] = []166 for block in blocks:167 if block['id'] not in used_ids:168 new_rect_y_axis = {169 'x1': seed_block['geometry']['x1'],170 'x2': seed_block['geometry']['x2'],171 'y1': block['geometry']['y1'],172 'y2': block['geometry']['y2']173 }174 175 thresh1 = _intersection_pct(seed_block['geometry'], new_rect_y_axis)176 thresh2 = _intersection_pct(new_rect_y_axis, seed_block['geometry'])177 178 if thresh1 >= y_axis_intersection_thresh and thresh2 >= y_axis_intersection_thresh:179 line.append(block)180 used_ids.add(block['id'])181 else:182 seed_center_x = (seed_block['geometry']['x1'] + seed_block['geometry']['x2']) / (2 * width)183 seed_center_y = (seed_block['geometry']['y1'] + seed_block['geometry']['y2']) / (2 * height)184 block_center_x = (block['geometry']['x1'] + block['geometry']['x2']) / (2 * width)185 block_center_y = (block['geometry']['y1'] + block['geometry']['y2']) / (2 * height)186 187 dist = ((seed_center_x - block_center_x)**2 + (seed_center_y - block_center_y)**2)**0.5188 189 if dist <= 0.2 and (thresh1 >= alternate_thresh or thresh2 >= alternate_thresh):190 line.append(block)191 used_ids.add(block['id'])192 193 if line:194 all_lines.append(line)195 196 if all_lines:197 all_lines = _merge_lines(all_lines)198 all_lines.sort(key=lambda line: line[0]['geometry']['y1'] if line else 0)199 200 all_sorted_lines: List[Dict[str, Any]] = []201 for line in all_lines:202 all_sorted_lines.extend(line)203 204 return all_sorted_lines205 206 207def _identify_line_segments(208 blocks: List[Dict[str, Any]],209 width: int,210 line_segment_merge_threshold: float = 0.03211) -> List[List[Dict[str, Any]]]:212 """Identify line segments as groups of related words."""213 added_blocks: Set[str] = set()214 line_segments: List[List[Dict[str, Any]]] = []215 216 for seed_block in blocks:217 if seed_block.get('blockType') != 'WORD':218 continue219 220 if seed_block['id'] not in added_blocks:221 line_segment = [seed_block]222 added_blocks.add(seed_block['id'])223 224 for block in blocks:225 if block.get('blockType') != 'WORD' or block['id'] == seed_block['id'] or block['id'] in added_blocks:226 continue227 228 vert_mid_point = (block['geometry']['y1'] + (block['geometry']['y2'] - block['geometry']['y1']) / 2)229 230 if seed_block['geometry']['y1'] < vert_mid_point < seed_block['geometry']['y2']:231 dist_block = line_segment[-1]232 horizontal_gap = block['geometry']['x1'] - dist_block['geometry']['x2']233 gap_ratio = horizontal_gap / width if width > 0 else 0234 235 if gap_ratio < line_segment_merge_threshold and block['geometry']['x1'] > dist_block['geometry']['x2']:236 line_segment.append(block)237 added_blocks.add(block['id'])238 239 if line_segment:240 line_segments.append(line_segment)241 242 return line_segments243 244 245def _format_line_segments(246 line_segments: List[List[Dict[str, Any]]],247 page_num: int248) -> List[Dict[str, Any]]:249 """Format line segments into line blocks."""250 formatted_segments: List[Dict[str, Any]] = []251 252 for line_segment in line_segments:253 if not line_segment:254 continue255 256 segment_text = ' '.join([block['text'] for block in line_segment]).strip()257 if not segment_text:258 continue259 260 confidences = [block.get('confidence', 1.0) for block in line_segment]261 confidence = sum(confidences) / len(confidences) if confidences else 1.0262 263 x1 = min(block['geometry']['x1'] for block in line_segment)264 y1 = min(block['geometry']['y1'] for block in line_segment)265 x2 = max(block['geometry']['x2'] for block in line_segment)266 y2 = max(block['geometry']['y2'] for block in line_segment)267 268 formatted_segments.append({269 "id": _generate_uuid(),270 "pageNum": page_num,271 "blockType": "LINE",272 "text": segment_text,273 "confidence": confidence,274 "geometry": {"x1": x1, "y1": y1, "x2": x2, "y2": y2}275 })276 277 return formatted_segments278 279 280def _post_process_words_to_lines(281 word_blocks: List[Dict[str, Any]],282 page_width: int,283 page_height: int,284 page_num: int,285 line_segment_merge_threshold: float = 0.03286) -> List[Dict[str, Any]]:287 """Post-process word blocks to create properly separated line blocks."""288 if not word_blocks:289 return []290 291 for block in word_blocks:292 if 'id' not in block:293 block['id'] = _generate_uuid()294 295 sorted_blocks = _sort_words_in_reading_order(word_blocks, page_height, page_width)296 line_segments = _identify_line_segments(sorted_blocks, page_width, line_segment_merge_threshold)297 line_blocks = _format_line_segments(line_segments, page_num)298 299 return line_blocks300 301 302# =============================================================================303# Main Extraction Function304# =============================================================================305 306def extract_text_from_pdf(307 pdf_path: str,308 line_segment_merge_threshold: float = DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD309) -> Dict[str, Any]:310 """311 Main entry point for PDF text extraction.312 313 Args:314 pdf_path: Path to the PDF file315 line_segment_merge_threshold: Gap threshold for line merging (default: 0.03)316 317 Returns:318 Structured output dict with metadata and pages containing ocrBlocks319 """320 import os321 filename = os.path.basename(pdf_path)322 323 # Check if PDF is digital324 if not is_pdf_digital(pdf_path):325 return _create_empty_output(filename, line_segment_merge_threshold)326 327 if not PDFPLUMBER_AVAILABLE:328 raise RuntimeError("pdfplumber is required. Install with: pip install pdfplumber")329 330 # Initialize result structure with ocrBlocks at root level331 result: Dict[str, Any] = {332 "version": "1.0",333 "metadata": {334 "documentId": "pdf-extracted",335 "documentName": filename,336 "source": "pdfplumber",337 "numberOfPages": 0,338 "lineSegmentMergeThreshold": line_segment_merge_threshold339 },340 "pages": [],341 "ocrBlocks": []342 }343 344 with pdfplumber.open(pdf_path) as pdf:345 result["metadata"]["numberOfPages"] = len(pdf.pages)346 347 for page_num, page in enumerate(pdf.pages, start=1):348 # Scale to consistent dimensions349 scale_factor = DEFAULT_SCALE_WIDTH / page.width if page.width > 0 else 1350 page_width = int(page.width * scale_factor)351 page_height = int(page.height * scale_factor)352 353 # Add page info (without ocrBlocks)354 result["pages"].append({355 "pageNum": page_num,356 "width": page_width,357 "height": page_height358 })359 360 # Extract words with tighter tolerances361 words = page.extract_words(362 keep_blank_chars=False,363 x_tolerance=2,364 y_tolerance=2365 )366 367 # Convert to word blocks368 word_blocks: List[Dict[str, Any]] = []369 for word in words:370 word_block = {371 "id": _generate_uuid(),372 "pageNum": page_num,373 "blockType": "WORD",374 "text": word["text"],375 "confidence": 1.0,376 "geometry": {377 "x1": int(word["x0"] * scale_factor),378 "y1": int(word["top"] * scale_factor),379 "x2": int(word["x1"] * scale_factor),380 "y2": int(word["bottom"] * scale_factor)381 }382 }383 word_blocks.append(word_block)384 result["ocrBlocks"].append(word_block)385 386 # Create line blocks387 line_blocks = _post_process_words_to_lines(388 word_blocks,389 page_width,390 page_height,391 page_num,392 line_segment_merge_threshold393 )394 395 # Add line blocks to root ocrBlocks396 result["ocrBlocks"].extend(line_blocks)397 398 return result399 400 401def extract_text_from_pdf_bytes(402 pdf_bytes: bytes,403 file_name: str,404 line_segment_merge_threshold: float = DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD405) -> Dict[str, Any]:406 """407 Extract text from in-memory PDF bytes (for file uploads).408 409 Args:410 pdf_bytes: PDF file content as bytes411 file_name: Name of the file (for metadata)412 line_segment_merge_threshold: Gap threshold for line merging413 414 Returns:415 Structured output dict with metadata and pages containing ocrBlocks416 """417 if not PDFPLUMBER_AVAILABLE:418 raise RuntimeError("pdfplumber is required. Install with: pip install pdfplumber")419 420 result: Dict[str, Any] = {421 "version": "1.0",422 "metadata": {423 "documentId": "pdf-extracted",424 "documentName": file_name,425 "source": "pdfplumber",426 "numberOfPages": 0,427 "lineSegmentMergeThreshold": line_segment_merge_threshold428 },429 "pages": [],430 "ocrBlocks": []431 }432 433 pdf_stream = io.BytesIO(pdf_bytes)434 435 with pdfplumber.open(pdf_stream) as pdf:436 result["metadata"]["numberOfPages"] = len(pdf.pages)437 438 for page_num, page in enumerate(pdf.pages, start=1):439 scale_factor = DEFAULT_SCALE_WIDTH / page.width if page.width > 0 else 1440 page_width = int(page.width * scale_factor)441 page_height = int(page.height * scale_factor)442 443 # Add page info (without ocrBlocks)444 result["pages"].append({445 "pageNum": page_num,446 "width": page_width,447 "height": page_height448 })449 450 words = page.extract_words(451 keep_blank_chars=False,452 x_tolerance=2,453 y_tolerance=2454 )455 456 word_blocks: List[Dict[str, Any]] = []457 for word in words:458 word_block = {459 "id": _generate_uuid(),460 "pageNum": page_num,461 "blockType": "WORD",462 "text": word["text"],463 "confidence": 1.0,464 "geometry": {465 "x1": int(word["x0"] * scale_factor),466 "y1": int(word["top"] * scale_factor),467 "x2": int(word["x1"] * scale_factor),468 "y2": int(word["bottom"] * scale_factor)469 }470 }471 word_blocks.append(word_block)472 result["ocrBlocks"].append(word_block)473 474 line_blocks = _post_process_words_to_lines(475 word_blocks,476 page_width,477 page_height,478 page_num,479 line_segment_merge_threshold480 )481 482 # Add line blocks to root ocrBlocks483 result["ocrBlocks"].extend(line_blocks)484 485 return result486 487 488def _create_empty_output(filename: str, line_segment_merge_threshold: float) -> Dict[str, Any]:489 """Create empty output structure for non-digital PDFs."""490 return {491 "version": "1.0",492 "metadata": {493 "documentId": "pdf-extracted",494 "documentName": filename,495 "source": "none",496 "numberOfPages": 0,497 "lineSegmentMergeThreshold": line_segment_merge_threshold498 },499 "pages": [],500 "ocrBlocks": []501 }502 503 504# =============================================================================505# CLI Entry Point506# =============================================================================507 508if __name__ == '__main__':509 import sys510 import json511 512 if len(sys.argv) < 2:513 print("Usage: python pdf_utils.py <pdf_path> [threshold]")514 print(" threshold: line segment merge threshold (default: 0.03)")515 sys.exit(1)516 517 pdf_path = sys.argv[1]518 threshold = float(sys.argv[2]) if len(sys.argv) > 2 else DEFAULT_LINE_SEGMENT_MERGE_THRESHOLD519 520 print(f"Extracting text from: {pdf_path}")521 print(f"Line segment merge threshold: {threshold}")522 print("-" * 50)523 524 result = extract_text_from_pdf(pdf_path, threshold)525 print(json.dumps(result, indent=2))526 