documentExtractionag051/ExtractDocument
0
1from collections import defaultdict2import logging3 4logger = logging.getLogger(__name__)5 6 7def identify_vendor_type_based_on_ocr_search(8 ocr_elems,9 layout_mapping,10 match_full_words=False, # Optional: avoid substring false positives11):12 """13 Identify vendor by searching OCR text using keyword mappings.14 15 Args:16 ocr_elems: iterable of OCR objects with .text attribute.17 layout_mapping: dict {vendor_name: [keyword1, keyword2, ...]}18 match_full_words: if True, match exact tokens instead of substring.19 20 Returns:21 Best-matching vendor name (str) or None.22 """23 24 if not layout_mapping:25 logger.info("No vendor keyword mapping found.")26 return None27 28 vendor_scores = defaultdict(int)29 30 # Pre-normalize all OCR text into a list31 all_text_lines = []32 for elem in ocr_elems:33 if hasattr(elem, "text") and elem.text:34 line = str(elem.text).upper().strip()35 if line:36 all_text_lines.append(line)37 38 if not all_text_lines:39 logger.info("No OCR text available to perform vendor identification.")40 return None41 42 # Tokenize if full-word matching enabled43 tokenized_lines = (44 [set(line.replace(",", " ").replace(".", " ").split()) for line in all_text_lines]45 if match_full_words46 else None47 )48 49 # Evaluate each vendor50 for vendor, keywords in layout_mapping.items():51 vendor_score = 052 keywords_upper = [kw.upper() for kw in keywords]53 54 for kw in keywords_upper:55 for idx, line in enumerate(all_text_lines):56 57 if match_full_words:58 if kw in tokenized_lines[idx]:59 vendor_score += 160 else:61 if kw in line:62 vendor_score += 163 64 vendor_scores[vendor] = vendor_score65 66 # No matches67 if not any(vendor_scores.values()):68 logger.info("No vendor match found.")69 return None70 71 # Pick vendor with highest score72 best_vendor = max(vendor_scores, key=vendor_scores.get)73 logger.info(f"Identified vendor '{best_vendor}' with score {vendor_scores[best_vendor]}.")74 return best_vendor75 