documentExtractionag051/ExtractDocument
0
1# custom > invoice_post_processing.py > extract_invoice_tables2import logging3import re4from typing import List, Dict, Any, Optional5from utils.geometry_utils import compute_global_bounds, Rect, TextRect6 7# Set up logging - use INFO level by default8logging.basicConfig(level=logging.INFO, format='%(levelname)s:%(name)s:%(message)s')9logger = logging.getLogger(__name__)10 11 12def _infer_anchor_pattern(anchor_column_lines, min_samples=3):13 """14 Automatically infer a regex pattern from anchor column values.15 16 Analyzes sample values to detect common patterns like:17 - Pure digits: "12345678" โ r"^\d{8}$"18 - Alphanumeric: "ABC123" โ r"^[A-Z]{3}\d{3}$"19 - With separators: "INV-2024-001" โ r"^[A-Z]+-\d+-\d+$"20 21 Args:22 anchor_column_lines: List of TextRect elements from the anchor column23 min_samples: Minimum samples needed for reliable pattern detection24 25 Returns:26 Compiled regex pattern, or None if detection fails27 """28 if not anchor_column_lines:29 logger.warning("No anchor column lines provided for pattern inference")30 return None31 32 # Collect unique non-empty values33 values = []34 for line in anchor_column_lines:35 text = line.text.strip()36 if text and len(text) >= 2: # Skip very short values37 values.append(text)38 39 if len(values) < min_samples:40 logger.warning(f"Not enough samples ({len(values)}) for reliable pattern detection")41 return None42 43 # Analyze character patterns in each value44 def _analyze_value(value):45 """Convert value to a pattern signature."""46 pattern_parts = []47 current_type = None48 current_count = 049 50 for char in value:51 if char.isdigit():52 char_type = 'D' # Digit53 elif char.isalpha():54 if char.isupper():55 char_type = 'U' # Uppercase56 else:57 char_type = 'L' # Lowercase58 elif char in '-_/\\.:':59 char_type = char # Separator (literal)60 elif char == ' ':61 char_type = 'S' # Space62 else:63 char_type = 'X' # Other64 65 if char_type == current_type:66 current_count += 167 else:68 if current_type is not None:69 pattern_parts.append((current_type, current_count))70 current_type = char_type71 current_count = 172 73 if current_type is not None:74 pattern_parts.append((current_type, current_count))75 76 return pattern_parts77 78 # Get pattern signatures for all values79 signatures = [_analyze_value(v) for v in values]80 81 # Find the most common signature structure82 sig_structures = {}83 for sig in signatures:84 # Create a structure key (types only, not counts)85 struct_key = tuple(t for t, c in sig)86 sig_structures.setdefault(struct_key, []).append(sig)87 88 # Get the most common structure89 best_structure = max(sig_structures.items(), key=lambda x: len(x[1]))90 structure_types, matching_sigs = best_structure91 92 if len(matching_sigs) < min_samples:93 logger.warning(f"No consistent pattern found across samples")94 return None95 96 # Build regex from the most common structure97 # Calculate min/max counts for each position98 position_counts = []99 for pos_idx in range(len(structure_types)):100 counts = [sig[pos_idx][1] for sig in matching_sigs if pos_idx < len(sig)]101 if counts:102 position_counts.append((min(counts), max(counts)))103 else:104 position_counts.append((1, 1))105 106 # Generate regex pattern107 regex_parts = ['^']108 for idx, char_type in enumerate(structure_types):109 min_count, max_count = position_counts[idx]110 111 if char_type == 'D':112 char_class = r'\d'113 elif char_type == 'U':114 char_class = r'[A-Z]'115 elif char_type == 'L':116 char_class = r'[a-z]'117 elif char_type == 'S':118 char_class = r'\s'119 elif char_type == 'X':120 char_class = r'.'121 elif char_type in '-_/\\.:':122 # Escape special regex characters123 char_class = re.escape(char_type)124 else:125 char_class = re.escape(char_type)126 127 # Add quantifier128 if min_count == max_count:129 if min_count == 1:130 regex_parts.append(char_class)131 else:132 regex_parts.append(f'{char_class}{{{min_count}}}')133 else:134 regex_parts.append(f'{char_class}{{{min_count},{max_count}}}')135 136 regex_parts.append('$')137 pattern_str = ''.join(regex_parts)138 139 # Validate: check how many values match the generated pattern140 try:141 compiled = re.compile(pattern_str)142 match_count = sum(1 for v in values if compiled.match(v))143 match_pct = (match_count / len(values)) * 100144 145 logger.info(f"=== Auto-detected Anchor Pattern ===")146 logger.info(f"Generated pattern: {pattern_str}")147 logger.info(f"Sample values: {values[:5]}{'...' if len(values) > 5 else ''}")148 logger.info(f"Match rate: {match_count}/{len(values)} ({match_pct:.1f}%)")149 150 if match_pct >= 60: # Accept if at least 60% match151 return compiled152 else:153 logger.warning(f"Pattern match rate too low ({match_pct:.1f}%), pattern may be unreliable")154 # Still return it but warn user155 return compiled156 157 except re.error as e:158 logger.error(f"Failed to compile inferred pattern '{pattern_str}': {e}")159 return None160 161 162def extract_invoice_tables(163 ocr_elems,164 validation_data,165 field_name='custom_field_name',166 anchor_keywords=None,167 column_headers=None,168 anchor_column=None,169 anchor_pattern=None,170 advanced_options=None,171 page_dimensions=None,172 **kwargs173):174 """175 Specialized post-processing pipeline for Invoice documents.176 177 Fixes applied:178 - Multi-line cell text is now ordered top-to-bottom, then left-to-right.179 - Rows with invalid Material codes are removed based on multiple regex opt.180 - Auto-detects anchor pattern from column values if not provided.181 182 Args:183 column_headers: Dict mapping field names to header text in document184 anchor_column: Column name used for row detection185 anchor_pattern: Regex pattern string for validating anchor column values.186 If not provided, the pattern will be AUTO-DETECTED from187 the values in the anchor column (e.g., if values are188 "12345678", "87654321", pattern becomes r"^\d{8}$")189 advanced_options: Dict for special logic options:190 - stop_text: Text marking end of table on last page191 - enable_continuation: Whether to enable continuation column192 - continuation_marker: Text after which continuation starts193 """194 logger.info("Running invoice specialized post-processing...")195 196 # Parse advanced options197 if advanced_options is None:198 advanced_options = {}199 200 stop_text = advanced_options.get("stop_text", "Total Amount:")201 enable_continuation = advanced_options.get("enable_continuation", False)202 continuation_column = advanced_options.get("continuation_column", "Description")203 continuation_marker = advanced_options.get("continuation_marker", "Ref.:")204 205 # Add table end markers for footer detection206 table_end_markers = advanced_options.get("table_end_markers", [])207 208 # Use provided column headers or default209 COLUMN_HEADERS = column_headers if column_headers else {210 "Material Code": "Material Code",211 "Description": "Description",212 "Quantity": "Qty.",213 "unit_price": "Unit Price",214 "total_price": "Total Net Value"215 }216 217 # Use provided anchor column or default218 ANCHOR_COLUMN = anchor_column if anchor_column else "Material Code"219 220 # Calculate cumulative heights for multi-page documents221 cumulative_heights = _calculate_cumulative_heights(page_dimensions)222 223 # Step 1: Create column spaces based on header positions224 column_spaces = _create_column_spaces(ocr_elems, page_dimensions, COLUMN_HEADERS)225 226 # Step 2: Find table boundaries (footer detection)227 table_boundaries = _find_table_boundaries(ocr_elems, page_dimensions, table_end_markers)228 229 # Step 3: Assign OCR lines to column spaces230 space_to_lines = _assign_lines_to_spaces(ocr_elems, column_spaces)231 232 # Step 4: Determine anchor pattern (user-provided or auto-detected)233 if anchor_pattern:234 # User provided a pattern - use it directly235 MATERIAL_CODE_PATTERNS = [re.compile(anchor_pattern)]236 logger.info(f"Using user-provided anchor pattern: {anchor_pattern}")237 else:238 # Auto-detect pattern from anchor column values239 anchor_col_lines = space_to_lines.get(ANCHOR_COLUMN, [])240 auto_pattern = _infer_anchor_pattern(anchor_col_lines)241 242 if auto_pattern:243 MATERIAL_CODE_PATTERNS = [auto_pattern]244 logger.info(f"Using auto-detected anchor pattern: {auto_pattern.pattern}")245 else:246 # Fallback: accept any non-empty value247 logger.warning("Could not auto-detect pattern, using fallback (any non-empty text)")248 MATERIAL_CODE_PATTERNS = [re.compile(r"^.+$")]249 250 # Step 5: Get lines from anchor column and identify rows251 anchor_col_lines = space_to_lines.get(ANCHOR_COLUMN, [])252 rows = _identify_rows(253 anchor_col_lines, 254 page_dimensions, 255 MATERIAL_CODE_PATTERNS[0],256 table_boundaries257 )258 259 logger.info(f"Identified {len(rows)} rows based on anchor column '{ANCHOR_COLUMN}'")260 261 # Step 5: Extract table data from rows262 table_rows = _extract_table_rows(263 rows=rows,264 column_spaces=column_spaces,265 space_to_lines=space_to_lines,266 cumulative_heights=cumulative_heights,267 MATERIAL_CODE_PATTERNS=MATERIAL_CODE_PATTERNS,268 ocr_elems=ocr_elems,269 anchor_column=ANCHOR_COLUMN,270 stop_text=stop_text,271 enable_continuation=enable_continuation,272 continuation_column=continuation_column,273 continuation_marker=continuation_marker,274 table_boundaries=table_boundaries275 )276 277 logger.info(f"Extracted {len(table_rows)} table rows")278 279 # Step 6: Update validation_data with table results280 validation_data["tables"] = {"table": table_rows}281 282 return validation_data283 284 285def _calculate_cumulative_heights(page_dimensions):286 """Calculate cumulative page heights for multi-page documents."""287 cumulative_heights = {}288 cumulative_height = 0289 290 for page_num in sorted(page_dimensions.keys()):291 cumulative_heights[page_num] = cumulative_height292 height = page_dimensions[page_num][1]293 cumulative_height += height294 295 return cumulative_heights296 297 298def _create_column_spaces(ocr_elems, page_dimensions, column_headers):299 """300 Create vertical column spaces based on header text positions.301 302 Uses a multi-pass matching strategy:303 1. Exact match (case-sensitive)304 2. Case-insensitive match305 3. Partial/contains match (header text in element or element in header)306 4. Normalized match (strip whitespace, lowercase)307 308 Args:309 ocr_elems: List of OCR TextRect elements310 page_dimensions: Dict of {page_num: (width, height)}311 column_headers: Dict mapping column keys to header text patterns312 313 Returns:314 Dict mapping column keys to Rect objects representing column spaces315 """316 column_spaces = {key: None for key in column_headers.keys()}317 max_page_height = max(height for width, height in page_dimensions.values()) if page_dimensions else 0318 319 # Get all LINE elements from page 1 for header matching320 page1_lines = [321 elem for elem in ocr_elems322 if elem.page_num == 1 and elem.block_type == "LINE"323 ]324 325 logger.info("=== Column Header Detection ===")326 logger.info(f"Looking for headers: {column_headers}")327 logger.info(f"Found {len(page1_lines)} LINE elements on page 1")328 329 # Log all page 1 LINE elements for debugging330 logger.debug("Page 1 LINE elements:")331 for elem in page1_lines:332 logger.debug(f" '{elem.text.strip()}' at x={elem.x1}-{elem.x2}, y={elem.y1}-{elem.y2}")333 334 def _normalize_text(text):335 """Normalize text for comparison: lowercase, strip, collapse whitespace."""336 return ' '.join(text.lower().strip().split())337 338 def _match_header(elem_text, header_text):339 """340 Multi-strategy header matching.341 342 Returns: (match_type, confidence) where confidence is 0-100343 """344 elem_clean = elem_text.strip()345 header_clean = header_text.strip()346 347 # Pass 1: Exact match348 if elem_clean == header_clean:349 return ("exact", 100)350 351 # Pass 2: Case-insensitive exact match352 if elem_clean.lower() == header_clean.lower():353 return ("case_insensitive", 95)354 355 # Pass 3: Normalized match (strip + lowercase + collapse whitespace)356 elem_norm = _normalize_text(elem_clean)357 header_norm = _normalize_text(header_clean)358 if elem_norm == header_norm:359 return ("normalized", 90)360 361 # Pass 4: Contains match (element contains header or vice versa)362 if header_norm in elem_norm:363 return ("contains_header", 80)364 if elem_norm in header_norm:365 return ("contains_elem", 75)366 367 # Pass 5: Starts-with match368 if elem_norm.startswith(header_norm):369 return ("starts_with", 70)370 if header_norm.startswith(elem_norm):371 return ("header_starts", 65)372 373 return (None, 0)374 375 # Match each header using multi-pass strategy376 matched_headers = {}377 378 for col_key, header_text in column_headers.items():379 best_match = None380 best_confidence = 0381 best_elem = None382 383 for elem in page1_lines:384 match_type, confidence = _match_header(elem.text, header_text)385 if confidence > best_confidence:386 best_confidence = confidence387 best_match = match_type388 best_elem = elem389 390 if best_match and best_elem:391 column_spaces[col_key] = Rect(392 x1=best_elem.x1,393 y1=0,394 x2=best_elem.x2,395 y2=max_page_height396 )397 matched_headers[col_key] = {398 "header_text": header_text,399 "matched_text": best_elem.text.strip(),400 "match_type": best_match,401 "confidence": best_confidence,402 "x_range": f"{best_elem.x1}-{best_elem.x2}"403 }404 logger.info(405 f"โ Matched '{col_key}' -> '{best_elem.text.strip()}' "406 f"(type={best_match}, conf={best_confidence}%, x={best_elem.x1}-{best_elem.x2})"407 )408 else:409 logger.warning(f"โ No match found for column '{col_key}' with header '{header_text}'")410 411 # Log summary412 found_count = sum(1 for space in column_spaces.values() if space is not None)413 missing_cols = [key for key, space in column_spaces.items() if space is None]414 415 logger.info(f"=== Header Detection Summary ===")416 logger.info(f"Found {found_count}/{len(column_headers)} columns")417 418 if missing_cols:419 logger.warning(f"Missing columns: {', '.join(missing_cols)}")420 logger.info("Available LINE texts on page 1 (for debugging):")421 for elem in page1_lines:422 logger.info(f" - '{elem.text.strip()}'")423 424 return column_spaces425 426 427def _find_table_boundaries(ocr_elems, page_dimensions, table_end_markers):428 """429 Find the table content boundaries for each page.430 Returns dict: {page_num: {'top': y1, 'bottom': y2}}431 432 The bottom boundary is determined by finding footer markers.433 """434 boundaries = {}435 436 for page_num, (page_width, page_height) in page_dimensions.items():437 # Default: full page438 boundaries[page_num] = {439 'top': 0,440 'bottom': page_height441 }442 443 if not table_end_markers:444 continue445 446 # Find the earliest footer marker on this page447 footer_top = page_height448 for elem in ocr_elems:449 if elem.page_num == page_num and elem.block_type == "LINE":450 for marker in table_end_markers:451 if marker in elem.text:452 # Use the top of the footer line as the table bottom453 if elem.y1 < footer_top:454 footer_top = elem.y1 - 10 # Small buffer above footer455 logger.info(f"Page {page_num}: Found footer marker '{marker}' "456 f"at y={elem.y1}, setting table bottom to {footer_top}")457 break458 459 boundaries[page_num]['bottom'] = footer_top460 461 return boundaries462 463 464def _assign_lines_to_spaces(ocr_elems, column_spaces):465 """Assign each LINE element to its best-matching column space."""466 space_to_lines = {key: [] for key in column_spaces.keys()}467 468 for line_elem in ocr_elems:469 if line_elem.block_type != "LINE":470 continue471 472 best_space, best_intersection = None, 0473 474 for col_key, space_rect in column_spaces.items():475 if space_rect is None:476 continue477 intersection = space_rect.intersection_pct(line_elem)478 if intersection > best_intersection:479 best_intersection, best_space = intersection, col_key480 481 if best_space and best_intersection > 0:482 space_to_lines[best_space].append(line_elem)483 484 return space_to_lines485 486 487def _identify_rows(material_code_lines, page_dimensions, pattern, table_boundaries=None):488 """489 Identify row boundaries based on material code positions.490 491 Args:492 table_boundaries: Dict with table content boundaries per page493 """494 # Log all lines being tested495 logger.info(f"Testing {len(material_code_lines)} lines against anchor pattern: {pattern.pattern}")496 for line in material_code_lines:497 text = line.text.strip()498 match = pattern.match(text)499 logger.debug(f" Line '{text}' -> {'MATCH' if match else 'no match'}")500 501 valid_material_lines = [502 line for line in material_code_lines if pattern.match(line.text.strip())503 ]504 505 logger.info(f"Found {len(valid_material_lines)} valid anchor column values")506 if valid_material_lines:507 logger.info(f"Valid material lines: {[line.text for line in valid_material_lines]}")508 else:509 # Show what values ARE in the anchor column for debugging510 logger.warning(f"No lines matched pattern '{pattern.pattern}'")511 logger.warning(f"Available anchor column values:")512 for line in material_code_lines[:20]: # Show first 20513 logger.info(f" - '{line.text.strip()}' (page {line.page_num}, y={line.y1})")514 515 page_to_lines, rows = {}, []516 for line in valid_material_lines:517 page_to_lines.setdefault(line.page_num, []).append(line)518 519 for page_num, lines in sorted(page_to_lines.items()):520 lines_sorted = sorted(lines, key=lambda l: l.y1)521 page_width, page_height = page_dimensions[page_num]522 523 # Get table bottom boundary for this page (excludes footer)524 if table_boundaries and page_num in table_boundaries:525 table_bottom = table_boundaries[page_num]['bottom']526 else:527 table_bottom = page_height528 529 for i, line in enumerate(lines_sorted):530 y1 = line.y1531 if i + 1 < len(lines_sorted):532 y2 = lines_sorted[i + 1].y1533 else:534 # Last row on this page - use table bottom (not page height)535 y2 = table_bottom536 537 row_rect = Rect(x1=0, y1=y1, x2=page_width, y2=y2, block_type="ROW", confidence=1.0)538 logger.debug(f"Created row rect: page={page_num}, x1={row_rect.x1}, y1={row_rect.y1}")539 rows.append({"page_num": page_num, "row_rect": row_rect})540 541 return rows542 543 544def _extract_table_rows(545 rows,546 column_spaces,547 space_to_lines,548 cumulative_heights,549 MATERIAL_CODE_PATTERNS,550 ocr_elems,551 anchor_column="Material Code",552 stop_text=None,553 enable_continuation=False,554 continuation_column=None,555 continuation_marker=None,556 table_boundaries=None557):558 """559 Extract cell values for each table row by finding line elements at row-column intersections.560 561 Args:562 stop_text: Text that marks end of table on last page563 enable_continuation: Whether to look for continuation on next page564 continuation_column: Which column may continue to next page565 continuation_marker: Text after which continuation starts566 table_boundaries: Dict with table content boundaries per page (to filter out footer)567 """568 # print(f"Rows to process: {rows}")569 # print(f"column_spaces: {column_spaces}")570 # print(f"space_to_lines: {list(space_to_lines)}")571 # print(f"cumulative_heights: {cumulative_heights}")572 # print(f"material_code_patterns: {MATERIAL_CODE_PATTERNS}")573 #print(f"ocr_elems: {ocr_elems}")574 table_rows = []575 576 if not rows:577 return table_rows578 579 highest_page_row_number = max(row["page_num"] for row in rows)580 logger.info(f"Highest page row number: {highest_page_row_number}")581 582 # Sort rows by page_num and y1 for easier next-row lookup583 sorted_rows = sorted(rows, key=lambda r: (r["page_num"], r["row_rect"].y1))584 585 for row_idx, row in enumerate(sorted_rows):586 page_num = row["page_num"]587 row_rect = row["row_rect"]588 height_offset = cumulative_heights.get(page_num, 0)589 row_data = {}590 591 for col_key, col_rect in column_spaces.items():592 #print(f"Processing column: {col_key}")593 if col_rect is None:594 continue595 596 # ---------------------------597 # GET TABLE BOUNDARY FOR THIS PAGE598 # ---------------------------599 page_table_bottom = None600 if table_boundaries and page_num in table_boundaries:601 page_table_bottom = table_boundaries[page_num]['bottom']602 603 # ---------------------------604 # CONFIGURABLE LAST-PAGE LOGIC605 # ---------------------------606 effective_col_rect_y2 = col_rect.y2607 if page_num == highest_page_row_number and stop_text:608 # Look for stop text to limit column boundary609 anchor_lines = sorted(610 space_to_lines.get(anchor_column, []),611 key=lambda l: l.y1612 )613 for line in anchor_lines:614 if stop_text in line.text:615 if line.page_num == page_num:616 effective_col_rect_y2 = line.y1 - 10617 break618 619 # ---------------------------620 # FIND BEST MATCHING LINES (filtered by table boundary)621 # ---------------------------622 candidate_lines = []623 for l in space_to_lines.get(col_key, []):624 if l.page_num != page_num:625 continue626 if col_rect.intersection_pct(l) <= 0:627 continue628 # Filter out lines below table boundary (footer lines)629 if page_table_bottom and l.y1 >= page_table_bottom:630 continue631 candidate_lines.append(l)632 633 # Score each line by its intersection with the row634 scored_lines = []635 for line in candidate_lines:636 row_intersection = row_rect.intersection_pct(line)637 if row_intersection > 0:638 scored_lines.append((line, row_intersection))639 640 # Filter to lines where this row has the best claim641 cell_lines = []642 for line, score in scored_lines:643 best_row_for_line = True644 for other_row in sorted_rows:645 if other_row is row:646 continue647 if other_row["page_num"] != page_num:648 continue649 other_intersection = other_row["row_rect"].intersection_pct(line)650 if other_intersection > score:651 best_row_for_line = False652 break653 654 if best_row_for_line:655 cell_lines.append(line)656 657 # ---------------------------658 # CONFIGURABLE CONTINUATION LOGIC659 # ---------------------------660 continuation_lines = []661 if enable_continuation and col_key == continuation_column:662 # Check if this is the last row on current page663 is_last_row_on_page = True664 for future_row in sorted_rows[row_idx + 1:]:665 if future_row["page_num"] == page_num:666 is_last_row_on_page = False667 break668 669 if is_last_row_on_page:670 next_page_num = page_num + 1671 672 # Get table boundaries for next page673 next_page_table_top = 0674 next_page_table_bottom = None675 if table_boundaries and next_page_num in table_boundaries:676 next_page_table_top = table_boundaries[next_page_num].get('top', 0)677 next_page_table_bottom = table_boundaries[next_page_num].get('bottom')678 679 # Find the first row on the next page (if any)680 next_row = None681 for future_row in sorted_rows[row_idx + 1:]:682 if future_row["page_num"] == next_page_num:683 next_row_y1 = future_row["row_rect"].y1684 break685 686 # Find the continuation marker line on the next page (optional)687 marker_line_y2 = 0688 if continuation_marker:689 for elem in ocr_elems:690 if elem.page_num == next_page_num and elem.block_type == "LINE":691 if continuation_marker in elem.text:692 marker_line_y2 = elem.y2693 break694 695 # If there's a next row on next page, get content above it696 if next_row_y1 is not None:697 # Get lines from next page that are above the next row698 for l in space_to_lines.get(col_key, []):699 if l.page_num != next_page_num:700 continue701 if col_rect.intersection_pct(l) <= 0:702 continue703 # Must be above the next row's anchor704 if l.y1 >= next_row_y1:705 continue706 # Must be within table area (not in footer)707 if next_page_table_bottom and l.y1 >= next_page_table_bottom:708 continue709 # Must be after continuation marker (if specified)710 if continuation_marker and marker_line_y2 and l.y1 <= marker_line_y2:711 continue712 713 continuation_lines.append(l)714 715 if continuation_lines:716 logger.info(717 f"Found {len(continuation_lines)} continuation lines for {col_key} "718 f"from page {page_num} to page {next_page_num} (above next row at y={next_row_y1})"719 )720 else:721 # No more rows on next page - check if content continues there722 for l in space_to_lines.get(col_key, []):723 if l.page_num != next_page_num:724 continue725 if col_rect.intersection_pct(l) <= 0:726 continue727 # Must be within table area (not in footer)728 if next_page_table_bottom and l.y1 >= next_page_table_bottom:729 continue730 # Must be after continuation marker (if specified)731 if continuation_marker and marker_line_y2 and l.y1 <= marker_line_y2:732 continue733 734 continuation_lines.append(l)735 736 if continuation_lines:737 logger.info(738 f"Found {len(continuation_lines)} continuation lines for {col_key} "739 f"from page {page_num} to page {next_page_num} (no more rows on next page)"740 )741 742 # ---------------------------743 # SORT + MERGE TEXT744 # ---------------------------745 # Combine cell_lines and continuation_lines for text extraction746 all_lines_for_text = cell_lines + continuation_lines747 all_lines_sorted = sorted(748 all_lines_for_text,749 key=lambda l: (l.page_num, round(l.y1, 1), l.x1)750 )751 752 cell_text = " ".join(l.text.strip() for l in all_lines_sorted)753 754 # Determine bounding box (only from original page cell_lines)755 cell_lines_sorted = sorted(756 cell_lines,757 key=lambda l: (round(l.y1, 1), l.x1)758 )759 760 if cell_lines_sorted:761 x1 = min(l.x1 for l in cell_lines_sorted)762 y1 = min(l.y1 for l in cell_lines_sorted) + height_offset763 x2 = max(l.x2 for l in cell_lines_sorted)764 y2 = max(l.y2 for l in cell_lines_sorted) + height_offset765 bounds = f"{x1}, {y1}, {x2 - x1}, {y2 - y1}"766 else:767 bounds = "0, 0, -1, -1"768 769 row_data[col_key] = {770 "value": cell_text,771 "bounds": bounds772 }773 774 # ---------------------------775 # ANCHOR COLUMN VALIDATION776 # ---------------------------777 anchor_value = row_data.get(anchor_column, {}).get("value", "").strip()778 logger.info(f"Validating anchor column '{anchor_column}' value: '{anchor_value}'")779 780 # if not any(p.match(anchor_value) for p in MATERIAL_CODE_PATTERNS):781 # logger.info(f"Skipping invalid {anchor_column} row: '{anchor_value}'")782 # continue783 784 table_rows.append(row_data)785 786 logger.info(f"Final table row count after filtering: {len(table_rows)}")787 # for row in table_rows:788 # print(row['description'])789 790 return table_rows791 