documentExtractionag051/ExtractDocument
0
1"""2pp_utils.py3===========4 5Utility functions for post-processing operations.6"""7 8import json9from .geometry_utils import TextRect10 11 12def open_json(file_name):13 """14 Open and load a JSON file.15 16 Args:17 file_name (str): Path to JSON file18 19 Returns:20 dict: Parsed JSON data21 """22 with open(file_name, "r", encoding="utf-8-sig") as f:23 content = f.read()24 25 # Handle potential duplicated content (legacy issue)26 repeat_index = content.find('{"pages":', content.find('{"pages":') + 1)27 if repeat_index != -1:28 content = content[:repeat_index]29 30 try:31 data = json.loads(content)32 except json.JSONDecodeError as e:33 print(f"Error decoding JSON: {e}")34 data = None35 36 return data37 38 39def create_text_rects(elements):40 """41 Create a list of TextRect objects from JSON elements.42 43 Args:44 elements (list): List of JSON elements containing geometry and other attributes.45 46 Returns:47 list: List of TextRect objects.48 """49 return [50 TextRect(51 x1=elem['geometry']['x1'],52 y1=elem['geometry']['y1'],53 x2=elem['geometry']['x2'],54 y2=elem['geometry']['y2'],55 confidence=elem.get('confidence', 0.0),56 block_type=elem.get('blockType', 'LINE'),57 text=elem.get('text', ''),58 page_num=elem.get('pageNum', 1)59 )60 for elem in elements61 ]62 