Team Ai
Apppublic

documentExtractionag051/ExtractDocument

sourceHugging Faceupdated 9mo agoView on Hugging Face
0likes
invoice_post_processing.py791 linesDownload Raw Back to root
1# custom > invoice_post_processing.py > extract_invoice_tables2import logging3import re4from typing import List, Dict, Any, Optional5from utils.geometry_utils import compute_global_bounds, Rect, TextRect6 7# Set up logging - use INFO level by default8logging.basicConfig(level=logging.INFO, format='%(levelname)s:%(name)s:%(message)s')9logger = logging.getLogger(__name__)10 11 12def _infer_anchor_pattern(anchor_column_lines, min_samples=3):13    """14    Automatically infer a regex pattern from anchor column values.15    16    Analyzes sample values to detect common patterns like:17    - Pure digits: "12345678" โ†’ r"^\d{8}$"18    - Alphanumeric: "ABC123" โ†’ r"^[A-Z]{3}\d{3}$"19    - With separators: "INV-2024-001" โ†’ r"^[A-Z]+-\d+-\d+$"20    21    Args:22        anchor_column_lines: List of TextRect elements from the anchor column23        min_samples: Minimum samples needed for reliable pattern detection24    25    Returns:26        Compiled regex pattern, or None if detection fails27    """28    if not anchor_column_lines:29        logger.warning("No anchor column lines provided for pattern inference")30        return None31    32    # Collect unique non-empty values33    values = []34    for line in anchor_column_lines:35        text = line.text.strip()36        if text and len(text) >= 2:  # Skip very short values37            values.append(text)38    39    if len(values) < min_samples:40        logger.warning(f"Not enough samples ({len(values)}) for reliable pattern detection")41        return None42    43    # Analyze character patterns in each value44    def _analyze_value(value):45        """Convert value to a pattern signature."""46        pattern_parts = []47        current_type = None48        current_count = 049        50        for char in value:51            if char.isdigit():52                char_type = 'D'  # Digit53            elif char.isalpha():54                if char.isupper():55                    char_type = 'U'  # Uppercase56                else:57                    char_type = 'L'  # Lowercase58            elif char in '-_/\\.:':59                char_type = char  # Separator (literal)60            elif char == ' ':61                char_type = 'S'  # Space62            else:63                char_type = 'X'  # Other64            65            if char_type == current_type:66                current_count += 167            else:68                if current_type is not None:69                    pattern_parts.append((current_type, current_count))70                current_type = char_type71                current_count = 172        73        if current_type is not None:74            pattern_parts.append((current_type, current_count))75        76        return pattern_parts77    78    # Get pattern signatures for all values79    signatures = [_analyze_value(v) for v in values]80    81    # Find the most common signature structure82    sig_structures = {}83    for sig in signatures:84        # Create a structure key (types only, not counts)85        struct_key = tuple(t for t, c in sig)86        sig_structures.setdefault(struct_key, []).append(sig)87    88    # Get the most common structure89    best_structure = max(sig_structures.items(), key=lambda x: len(x[1]))90    structure_types, matching_sigs = best_structure91    92    if len(matching_sigs) < min_samples:93        logger.warning(f"No consistent pattern found across samples")94        return None95    96    # Build regex from the most common structure97    # Calculate min/max counts for each position98    position_counts = []99    for pos_idx in range(len(structure_types)):100        counts = [sig[pos_idx][1] for sig in matching_sigs if pos_idx < len(sig)]101        if counts:102            position_counts.append((min(counts), max(counts)))103        else:104            position_counts.append((1, 1))105    106    # Generate regex pattern107    regex_parts = ['^']108    for idx, char_type in enumerate(structure_types):109        min_count, max_count = position_counts[idx]110        111        if char_type == 'D':112            char_class = r'\d'113        elif char_type == 'U':114            char_class = r'[A-Z]'115        elif char_type == 'L':116            char_class = r'[a-z]'117        elif char_type == 'S':118            char_class = r'\s'119        elif char_type == 'X':120            char_class = r'.'121        elif char_type in '-_/\\.:':122            # Escape special regex characters123            char_class = re.escape(char_type)124        else:125            char_class = re.escape(char_type)126        127        # Add quantifier128        if min_count == max_count:129            if min_count == 1:130                regex_parts.append(char_class)131            else:132                regex_parts.append(f'{char_class}{{{min_count}}}')133        else:134            regex_parts.append(f'{char_class}{{{min_count},{max_count}}}')135    136    regex_parts.append('$')137    pattern_str = ''.join(regex_parts)138    139    # Validate: check how many values match the generated pattern140    try:141        compiled = re.compile(pattern_str)142        match_count = sum(1 for v in values if compiled.match(v))143        match_pct = (match_count / len(values)) * 100144        145        logger.info(f"=== Auto-detected Anchor Pattern ===")146        logger.info(f"Generated pattern: {pattern_str}")147        logger.info(f"Sample values: {values[:5]}{'...' if len(values) > 5 else ''}")148        logger.info(f"Match rate: {match_count}/{len(values)} ({match_pct:.1f}%)")149        150        if match_pct >= 60:  # Accept if at least 60% match151            return compiled152        else:153            logger.warning(f"Pattern match rate too low ({match_pct:.1f}%), pattern may be unreliable")154            # Still return it but warn user155            return compiled156            157    except re.error as e:158        logger.error(f"Failed to compile inferred pattern '{pattern_str}': {e}")159        return None160 161 162def extract_invoice_tables(163    ocr_elems,164    validation_data,165    field_name='custom_field_name',166    anchor_keywords=None,167    column_headers=None,168    anchor_column=None,169    anchor_pattern=None,170    advanced_options=None,171    page_dimensions=None,172    **kwargs173):174    """175    Specialized post-processing pipeline for Invoice documents.176    177    Fixes applied:178    - Multi-line cell text is now ordered top-to-bottom, then left-to-right.179    - Rows with invalid Material codes are removed based on multiple regex opt.180    - Auto-detects anchor pattern from column values if not provided.181    182    Args:183        column_headers: Dict mapping field names to header text in document184        anchor_column: Column name used for row detection185        anchor_pattern: Regex pattern string for validating anchor column values.186                       If not provided, the pattern will be AUTO-DETECTED from187                       the values in the anchor column (e.g., if values are188                       "12345678", "87654321", pattern becomes r"^\d{8}$")189        advanced_options: Dict for special logic options:190            - stop_text: Text marking end of table on last page191            - enable_continuation: Whether to enable continuation column192            - continuation_marker: Text after which continuation starts193    """194    logger.info("Running invoice specialized post-processing...")195    196    # Parse advanced options197    if advanced_options is None:198        advanced_options = {}199    200    stop_text = advanced_options.get("stop_text", "Total Amount:")201    enable_continuation = advanced_options.get("enable_continuation", False)202    continuation_column = advanced_options.get("continuation_column", "Description")203    continuation_marker = advanced_options.get("continuation_marker", "Ref.:")204    205    # Add table end markers for footer detection206    table_end_markers = advanced_options.get("table_end_markers", [])207    208    # Use provided column headers or default209    COLUMN_HEADERS = column_headers if column_headers else {210        "Material Code": "Material Code",211        "Description": "Description",212        "Quantity": "Qty.",213        "unit_price": "Unit Price",214        "total_price": "Total Net Value"215    }216    217    # Use provided anchor column or default218    ANCHOR_COLUMN = anchor_column if anchor_column else "Material Code"219    220    # Calculate cumulative heights for multi-page documents221    cumulative_heights = _calculate_cumulative_heights(page_dimensions)222    223    # Step 1: Create column spaces based on header positions224    column_spaces = _create_column_spaces(ocr_elems, page_dimensions, COLUMN_HEADERS)225    226    # Step 2: Find table boundaries (footer detection)227    table_boundaries = _find_table_boundaries(ocr_elems, page_dimensions, table_end_markers)228    229    # Step 3: Assign OCR lines to column spaces230    space_to_lines = _assign_lines_to_spaces(ocr_elems, column_spaces)231    232    # Step 4: Determine anchor pattern (user-provided or auto-detected)233    if anchor_pattern:234        # User provided a pattern - use it directly235        MATERIAL_CODE_PATTERNS = [re.compile(anchor_pattern)]236        logger.info(f"Using user-provided anchor pattern: {anchor_pattern}")237    else:238        # Auto-detect pattern from anchor column values239        anchor_col_lines = space_to_lines.get(ANCHOR_COLUMN, [])240        auto_pattern = _infer_anchor_pattern(anchor_col_lines)241        242        if auto_pattern:243            MATERIAL_CODE_PATTERNS = [auto_pattern]244            logger.info(f"Using auto-detected anchor pattern: {auto_pattern.pattern}")245        else:246            # Fallback: accept any non-empty value247            logger.warning("Could not auto-detect pattern, using fallback (any non-empty text)")248            MATERIAL_CODE_PATTERNS = [re.compile(r"^.+$")]249    250    # Step 5: Get lines from anchor column and identify rows251    anchor_col_lines = space_to_lines.get(ANCHOR_COLUMN, [])252    rows = _identify_rows(253        anchor_col_lines, 254        page_dimensions, 255        MATERIAL_CODE_PATTERNS[0],256        table_boundaries257    )258    259    logger.info(f"Identified {len(rows)} rows based on anchor column '{ANCHOR_COLUMN}'")260    261    # Step 5: Extract table data from rows262    table_rows = _extract_table_rows(263        rows=rows,264        column_spaces=column_spaces,265        space_to_lines=space_to_lines,266        cumulative_heights=cumulative_heights,267        MATERIAL_CODE_PATTERNS=MATERIAL_CODE_PATTERNS,268        ocr_elems=ocr_elems,269        anchor_column=ANCHOR_COLUMN,270        stop_text=stop_text,271        enable_continuation=enable_continuation,272        continuation_column=continuation_column,273        continuation_marker=continuation_marker,274        table_boundaries=table_boundaries275    )276    277    logger.info(f"Extracted {len(table_rows)} table rows")278    279    # Step 6: Update validation_data with table results280    validation_data["tables"] = {"table": table_rows}281    282    return validation_data283 284 285def _calculate_cumulative_heights(page_dimensions):286    """Calculate cumulative page heights for multi-page documents."""287    cumulative_heights = {}288    cumulative_height = 0289    290    for page_num in sorted(page_dimensions.keys()):291        cumulative_heights[page_num] = cumulative_height292        height = page_dimensions[page_num][1]293        cumulative_height += height294    295    return cumulative_heights296 297 298def _create_column_spaces(ocr_elems, page_dimensions, column_headers):299    """300    Create vertical column spaces based on header text positions.301    302    Uses a multi-pass matching strategy:303    1. Exact match (case-sensitive)304    2. Case-insensitive match305    3. Partial/contains match (header text in element or element in header)306    4. Normalized match (strip whitespace, lowercase)307    308    Args:309        ocr_elems: List of OCR TextRect elements310        page_dimensions: Dict of {page_num: (width, height)}311        column_headers: Dict mapping column keys to header text patterns312    313    Returns:314        Dict mapping column keys to Rect objects representing column spaces315    """316    column_spaces = {key: None for key in column_headers.keys()}317    max_page_height = max(height for width, height in page_dimensions.values()) if page_dimensions else 0318    319    # Get all LINE elements from page 1 for header matching320    page1_lines = [321        elem for elem in ocr_elems322        if elem.page_num == 1 and elem.block_type == "LINE"323    ]324    325    logger.info("=== Column Header Detection ===")326    logger.info(f"Looking for headers: {column_headers}")327    logger.info(f"Found {len(page1_lines)} LINE elements on page 1")328    329    # Log all page 1 LINE elements for debugging330    logger.debug("Page 1 LINE elements:")331    for elem in page1_lines:332        logger.debug(f"  '{elem.text.strip()}' at x={elem.x1}-{elem.x2}, y={elem.y1}-{elem.y2}")333    334    def _normalize_text(text):335        """Normalize text for comparison: lowercase, strip, collapse whitespace."""336        return ' '.join(text.lower().strip().split())337    338    def _match_header(elem_text, header_text):339        """340        Multi-strategy header matching.341        342        Returns: (match_type, confidence) where confidence is 0-100343        """344        elem_clean = elem_text.strip()345        header_clean = header_text.strip()346        347        # Pass 1: Exact match348        if elem_clean == header_clean:349            return ("exact", 100)350        351        # Pass 2: Case-insensitive exact match352        if elem_clean.lower() == header_clean.lower():353            return ("case_insensitive", 95)354        355        # Pass 3: Normalized match (strip + lowercase + collapse whitespace)356        elem_norm = _normalize_text(elem_clean)357        header_norm = _normalize_text(header_clean)358        if elem_norm == header_norm:359            return ("normalized", 90)360        361        # Pass 4: Contains match (element contains header or vice versa)362        if header_norm in elem_norm:363            return ("contains_header", 80)364        if elem_norm in header_norm:365            return ("contains_elem", 75)366        367        # Pass 5: Starts-with match368        if elem_norm.startswith(header_norm):369            return ("starts_with", 70)370        if header_norm.startswith(elem_norm):371            return ("header_starts", 65)372        373        return (None, 0)374    375    # Match each header using multi-pass strategy376    matched_headers = {}377    378    for col_key, header_text in column_headers.items():379        best_match = None380        best_confidence = 0381        best_elem = None382        383        for elem in page1_lines:384            match_type, confidence = _match_header(elem.text, header_text)385            if confidence > best_confidence:386                best_confidence = confidence387                best_match = match_type388                best_elem = elem389        390        if best_match and best_elem:391            column_spaces[col_key] = Rect(392                x1=best_elem.x1,393                y1=0,394                x2=best_elem.x2,395                y2=max_page_height396            )397            matched_headers[col_key] = {398                "header_text": header_text,399                "matched_text": best_elem.text.strip(),400                "match_type": best_match,401                "confidence": best_confidence,402                "x_range": f"{best_elem.x1}-{best_elem.x2}"403            }404            logger.info(405                f"โœ“ Matched '{col_key}' -> '{best_elem.text.strip()}' "406                f"(type={best_match}, conf={best_confidence}%, x={best_elem.x1}-{best_elem.x2})"407            )408        else:409            logger.warning(f"โœ— No match found for column '{col_key}' with header '{header_text}'")410    411    # Log summary412    found_count = sum(1 for space in column_spaces.values() if space is not None)413    missing_cols = [key for key, space in column_spaces.items() if space is None]414    415    logger.info(f"=== Header Detection Summary ===")416    logger.info(f"Found {found_count}/{len(column_headers)} columns")417    418    if missing_cols:419        logger.warning(f"Missing columns: {', '.join(missing_cols)}")420        logger.info("Available LINE texts on page 1 (for debugging):")421        for elem in page1_lines:422            logger.info(f"  - '{elem.text.strip()}'")423    424    return column_spaces425 426 427def _find_table_boundaries(ocr_elems, page_dimensions, table_end_markers):428    """429    Find the table content boundaries for each page.430    Returns dict: {page_num: {'top': y1, 'bottom': y2}}431    432    The bottom boundary is determined by finding footer markers.433    """434    boundaries = {}435    436    for page_num, (page_width, page_height) in page_dimensions.items():437        # Default: full page438        boundaries[page_num] = {439            'top': 0,440            'bottom': page_height441        }442        443        if not table_end_markers:444            continue445        446        # Find the earliest footer marker on this page447        footer_top = page_height448        for elem in ocr_elems:449            if elem.page_num == page_num and elem.block_type == "LINE":450                for marker in table_end_markers:451                    if marker in elem.text:452                        # Use the top of the footer line as the table bottom453                        if elem.y1 < footer_top:454                            footer_top = elem.y1 - 10  # Small buffer above footer455                            logger.info(f"Page {page_num}: Found footer marker '{marker}' "456                                      f"at y={elem.y1}, setting table bottom to {footer_top}")457                        break458        459        boundaries[page_num]['bottom'] = footer_top460    461    return boundaries462 463 464def _assign_lines_to_spaces(ocr_elems, column_spaces):465    """Assign each LINE element to its best-matching column space."""466    space_to_lines = {key: [] for key in column_spaces.keys()}467    468    for line_elem in ocr_elems:469        if line_elem.block_type != "LINE":470            continue471        472        best_space, best_intersection = None, 0473        474        for col_key, space_rect in column_spaces.items():475            if space_rect is None:476                continue477            intersection = space_rect.intersection_pct(line_elem)478            if intersection > best_intersection:479                best_intersection, best_space = intersection, col_key480        481        if best_space and best_intersection > 0:482            space_to_lines[best_space].append(line_elem)483    484    return space_to_lines485 486 487def _identify_rows(material_code_lines, page_dimensions, pattern, table_boundaries=None):488    """489    Identify row boundaries based on material code positions.490    491    Args:492        table_boundaries: Dict with table content boundaries per page493    """494    # Log all lines being tested495    logger.info(f"Testing {len(material_code_lines)} lines against anchor pattern: {pattern.pattern}")496    for line in material_code_lines:497        text = line.text.strip()498        match = pattern.match(text)499        logger.debug(f"  Line '{text}' -> {'MATCH' if match else 'no match'}")500    501    valid_material_lines = [502        line for line in material_code_lines if pattern.match(line.text.strip())503    ]504    505    logger.info(f"Found {len(valid_material_lines)} valid anchor column values")506    if valid_material_lines:507        logger.info(f"Valid material lines: {[line.text for line in valid_material_lines]}")508    else:509        # Show what values ARE in the anchor column for debugging510        logger.warning(f"No lines matched pattern '{pattern.pattern}'")511        logger.warning(f"Available anchor column values:")512        for line in material_code_lines[:20]:  # Show first 20513            logger.info(f"  - '{line.text.strip()}' (page {line.page_num}, y={line.y1})")514    515    page_to_lines, rows = {}, []516    for line in valid_material_lines:517        page_to_lines.setdefault(line.page_num, []).append(line)518    519    for page_num, lines in sorted(page_to_lines.items()):520        lines_sorted = sorted(lines, key=lambda l: l.y1)521        page_width, page_height = page_dimensions[page_num]522        523        # Get table bottom boundary for this page (excludes footer)524        if table_boundaries and page_num in table_boundaries:525            table_bottom = table_boundaries[page_num]['bottom']526        else:527            table_bottom = page_height528        529        for i, line in enumerate(lines_sorted):530            y1 = line.y1531            if i + 1 < len(lines_sorted):532                y2 = lines_sorted[i + 1].y1533            else:534                # Last row on this page - use table bottom (not page height)535                y2 = table_bottom536            537            row_rect = Rect(x1=0, y1=y1, x2=page_width, y2=y2, block_type="ROW", confidence=1.0)538            logger.debug(f"Created row rect: page={page_num}, x1={row_rect.x1}, y1={row_rect.y1}")539            rows.append({"page_num": page_num, "row_rect": row_rect})540    541    return rows542 543 544def _extract_table_rows(545    rows,546    column_spaces,547    space_to_lines,548    cumulative_heights,549    MATERIAL_CODE_PATTERNS,550    ocr_elems,551    anchor_column="Material Code",552    stop_text=None,553    enable_continuation=False,554    continuation_column=None,555    continuation_marker=None,556    table_boundaries=None557):558    """559    Extract cell values for each table row by finding line elements at row-column intersections.560    561    Args:562        stop_text: Text that marks end of table on last page563        enable_continuation: Whether to look for continuation on next page564        continuation_column: Which column may continue to next page565        continuation_marker: Text after which continuation starts566        table_boundaries: Dict with table content boundaries per page (to filter out footer)567    """568    # print(f"Rows to process: {rows}")569    # print(f"column_spaces: {column_spaces}")570    # print(f"space_to_lines: {list(space_to_lines)}")571    # print(f"cumulative_heights: {cumulative_heights}")572    # print(f"material_code_patterns: {MATERIAL_CODE_PATTERNS}")573    #print(f"ocr_elems: {ocr_elems}")574    table_rows = []575    576    if not rows:577        return table_rows578    579    highest_page_row_number = max(row["page_num"] for row in rows)580    logger.info(f"Highest page row number: {highest_page_row_number}")581    582    # Sort rows by page_num and y1 for easier next-row lookup583    sorted_rows = sorted(rows, key=lambda r: (r["page_num"], r["row_rect"].y1))584    585    for row_idx, row in enumerate(sorted_rows):586        page_num = row["page_num"]587        row_rect = row["row_rect"]588        height_offset = cumulative_heights.get(page_num, 0)589        row_data = {}590        591        for col_key, col_rect in column_spaces.items():592            #print(f"Processing column: {col_key}")593            if col_rect is None:594                continue595            596            # ---------------------------597            # GET TABLE BOUNDARY FOR THIS PAGE598            # ---------------------------599            page_table_bottom = None600            if table_boundaries and page_num in table_boundaries:601                page_table_bottom = table_boundaries[page_num]['bottom']602            603            # ---------------------------604            # CONFIGURABLE LAST-PAGE LOGIC605            # ---------------------------606            effective_col_rect_y2 = col_rect.y2607            if page_num == highest_page_row_number and stop_text:608                # Look for stop text to limit column boundary609                anchor_lines = sorted(610                    space_to_lines.get(anchor_column, []),611                    key=lambda l: l.y1612                )613                for line in anchor_lines:614                    if stop_text in line.text:615                        if line.page_num == page_num:616                            effective_col_rect_y2 = line.y1 - 10617                            break618            619            # ---------------------------620            # FIND BEST MATCHING LINES (filtered by table boundary)621            # ---------------------------622            candidate_lines = []623            for l in space_to_lines.get(col_key, []):624                if l.page_num != page_num:625                    continue626                if col_rect.intersection_pct(l) <= 0:627                    continue628                # Filter out lines below table boundary (footer lines)629                if page_table_bottom and l.y1 >= page_table_bottom:630                    continue631                candidate_lines.append(l)632            633            # Score each line by its intersection with the row634            scored_lines = []635            for line in candidate_lines:636                row_intersection = row_rect.intersection_pct(line)637                if row_intersection > 0:638                    scored_lines.append((line, row_intersection))639            640            # Filter to lines where this row has the best claim641            cell_lines = []642            for line, score in scored_lines:643                best_row_for_line = True644                for other_row in sorted_rows:645                    if other_row is row:646                        continue647                    if other_row["page_num"] != page_num:648                        continue649                    other_intersection = other_row["row_rect"].intersection_pct(line)650                    if other_intersection > score:651                        best_row_for_line = False652                        break653                654                if best_row_for_line:655                    cell_lines.append(line)656            657            # ---------------------------658            # CONFIGURABLE CONTINUATION LOGIC659            # ---------------------------660            continuation_lines = []661            if enable_continuation and col_key == continuation_column:662                # Check if this is the last row on current page663                is_last_row_on_page = True664                for future_row in sorted_rows[row_idx + 1:]:665                    if future_row["page_num"] == page_num:666                        is_last_row_on_page = False667                        break668                669                if is_last_row_on_page:670                    next_page_num = page_num + 1671                    672                    # Get table boundaries for next page673                    next_page_table_top = 0674                    next_page_table_bottom = None675                    if table_boundaries and next_page_num in table_boundaries:676                        next_page_table_top = table_boundaries[next_page_num].get('top', 0)677                        next_page_table_bottom = table_boundaries[next_page_num].get('bottom')678                    679                    # Find the first row on the next page (if any)680                    next_row = None681                    for future_row in sorted_rows[row_idx + 1:]:682                        if future_row["page_num"] == next_page_num:683                            next_row_y1 = future_row["row_rect"].y1684                            break685                    686                    # Find the continuation marker line on the next page (optional)687                    marker_line_y2 = 0688                    if continuation_marker:689                        for elem in ocr_elems:690                            if elem.page_num == next_page_num and elem.block_type == "LINE":691                                if continuation_marker in elem.text:692                                    marker_line_y2 = elem.y2693                                    break694                    695                    # If there's a next row on next page, get content above it696                    if next_row_y1 is not None:697                        # Get lines from next page that are above the next row698                        for l in space_to_lines.get(col_key, []):699                            if l.page_num != next_page_num:700                                continue701                            if col_rect.intersection_pct(l) <= 0:702                                continue703                            # Must be above the next row's anchor704                            if l.y1 >= next_row_y1:705                                continue706                            # Must be within table area (not in footer)707                            if next_page_table_bottom and l.y1 >= next_page_table_bottom:708                                continue709                            # Must be after continuation marker (if specified)710                            if continuation_marker and marker_line_y2 and l.y1 <= marker_line_y2:711                                continue712                            713                            continuation_lines.append(l)714                    715                    if continuation_lines:716                        logger.info(717                            f"Found {len(continuation_lines)} continuation lines for {col_key} "718                            f"from page {page_num} to page {next_page_num} (above next row at y={next_row_y1})"719                        )720                else:721                    # No more rows on next page - check if content continues there722                    for l in space_to_lines.get(col_key, []):723                        if l.page_num != next_page_num:724                            continue725                        if col_rect.intersection_pct(l) <= 0:726                            continue727                        # Must be within table area (not in footer)728                        if next_page_table_bottom and l.y1 >= next_page_table_bottom:729                            continue730                        # Must be after continuation marker (if specified)731                        if continuation_marker and marker_line_y2 and l.y1 <= marker_line_y2:732                            continue733                        734                        continuation_lines.append(l)735                    736                    if continuation_lines:737                        logger.info(738                            f"Found {len(continuation_lines)} continuation lines for {col_key} "739                            f"from page {page_num} to page {next_page_num} (no more rows on next page)"740                        )741            742            # ---------------------------743            # SORT + MERGE TEXT744            # ---------------------------745            # Combine cell_lines and continuation_lines for text extraction746            all_lines_for_text = cell_lines + continuation_lines747            all_lines_sorted = sorted(748                all_lines_for_text,749                key=lambda l: (l.page_num, round(l.y1, 1), l.x1)750            )751            752            cell_text = " ".join(l.text.strip() for l in all_lines_sorted)753            754            # Determine bounding box (only from original page cell_lines)755            cell_lines_sorted = sorted(756                cell_lines,757                key=lambda l: (round(l.y1, 1), l.x1)758            )759            760            if cell_lines_sorted:761                x1 = min(l.x1 for l in cell_lines_sorted)762                y1 = min(l.y1 for l in cell_lines_sorted) + height_offset763                x2 = max(l.x2 for l in cell_lines_sorted)764                y2 = max(l.y2 for l in cell_lines_sorted) + height_offset765                bounds = f"{x1}, {y1}, {x2 - x1}, {y2 - y1}"766            else:767                bounds = "0, 0, -1, -1"768            769            row_data[col_key] = {770                "value": cell_text,771                "bounds": bounds772            }773        774        # ---------------------------775        # ANCHOR COLUMN VALIDATION776        # ---------------------------777        anchor_value = row_data.get(anchor_column, {}).get("value", "").strip()778        logger.info(f"Validating anchor column '{anchor_column}' value: '{anchor_value}'")779        780        # if not any(p.match(anchor_value) for p in MATERIAL_CODE_PATTERNS):781        #     logger.info(f"Skipping invalid {anchor_column} row: '{anchor_value}'")782        #     continue783        784        table_rows.append(row_data)785    786    logger.info(f"Final table row count after filtering: {len(table_rows)}")787    # for row in table_rows:788    #     print(row['description'])789    790    return table_rows791