seanpedrickcase/document_redaction
10
1import json2import os3 4import boto35from dotenv import load_dotenv6 7# Import the main function from your CLI script8from cli_redact import main as cli_main9from tools.config import (10 AWS_LLM_PII_OPTION,11 AWS_REGION,12 AZURE_OPENAI_API_KEY,13 AZURE_OPENAI_INFERENCE_ENDPOINT,14 CHOSEN_LLM_ENTITIES,15 CHOSEN_LLM_PII_INFERENCE_METHOD,16 CLOUD_LLM_PII_MODEL_CHOICE,17 CLOUD_VLM_MODEL_CHOICE,18 DEFAULT_DUPLICATE_DETECTION_THRESHOLD,19 DEFAULT_FUZZY_SPELLING_MISTAKES_NUM,20 DEFAULT_INFERENCE_SERVER_PII_MODEL,21 DEFAULT_INFERENCE_SERVER_VLM_MODEL,22 DEFAULT_MIN_CONSECUTIVE_PAGES,23 DEFAULT_MIN_WORD_COUNT,24 DEFAULT_PAGE_MAX,25 DEFAULT_PAGE_MIN,26 EFFICIENT_OCR,27 EFFICIENT_OCR_MIN_EMBEDDED_IMAGE_PX,28 EFFICIENT_OCR_MIN_IMAGE_COVERAGE_FRACTION,29 EFFICIENT_OCR_MIN_WORDS,30 GEMINI_API_KEY,31 HYBRID_TEXTRACT_BEDROCK_VLM,32 IMAGES_DPI,33 INFERENCE_SERVER_API_URL,34 LAMBDA_DEFAULT_USERNAME,35 LAMBDA_EXTRACT_SIGNATURES,36 LAMBDA_MAX_POLL_ATTEMPTS,37 LAMBDA_POLL_INTERVAL,38 LAMBDA_PREPARE_IMAGES,39 LLM_MAX_NEW_TOKENS,40 LLM_TEMPERATURE,41 OCR_FIRST_PASS_MAX_WORKERS,42 SUMMARY_PAGE_GROUP_MAX_WORKERS,43)44 45 46def _get_env_list(env_var_name: str | list[str] | None) -> list[str]:47 """Parses a comma-separated environment variable into a list of strings."""48 if isinstance(env_var_name, list):49 return env_var_name50 if env_var_name is None:51 return []52 53 # Handle string input54 value = str(env_var_name).strip()55 if not value or value == "[]":56 return []57 58 # Remove brackets if present (e.g., "[item1, item2]" -> "item1, item2")59 if value.startswith("[") and value.endswith("]"):60 value = value[1:-1]61 62 # Remove quotes and split by comma63 value = value.replace('"', "").replace("'", "")64 if not value:65 return []66 67 # Split by comma and filter out any empty strings68 return [s.strip() for s in value.split(",") if s.strip()]69 70 71def convert_string_to_boolean(value: str) -> bool:72 """Convert string to boolean, handling various formats."""73 if isinstance(value, bool):74 return value75 elif value in ["True", "1", "true", "TRUE"]:76 return True77 elif value in ["False", "0", "false", "FALSE"]:78 return False79 else:80 raise ValueError(f"Invalid boolean value: {value}")81 82 83print("Lambda entrypoint loading...")84 85# Initialize S3 client outside the handler for connection reuse86s3_client = boto3.client("s3", region_name=os.getenv("AWS_REGION", AWS_REGION))87print("S3 client initialised")88 89# Lambda's only writable directory is /tmp. Ensure that all temporary files are stored in this directory.90TMP_DIR = "/tmp"91INPUT_DIR = os.path.join(TMP_DIR, "input")92OUTPUT_DIR = os.path.join(TMP_DIR, "output")93os.environ["TESSERACT_DATA_FOLDER"] = os.path.join(TMP_DIR, "share/tessdata")94os.environ["MPLCONFIGDIR"] = os.path.join(TMP_DIR, "matplotlib_cache")95os.environ["GRADIO_TEMP_DIR"] = os.path.join(TMP_DIR, "gradio_tmp")96os.environ["FEEDBACK_LOGS_FOLDER"] = os.path.join(TMP_DIR, "feedback")97os.environ["ACCESS_LOGS_FOLDER"] = os.path.join(TMP_DIR, "logs")98os.environ["USAGE_LOGS_FOLDER"] = os.path.join(TMP_DIR, "usage")99os.environ["PADDLE_MODEL_PATH"] = os.path.join(TMP_DIR, "paddle_models")100os.environ["SPACY_MODEL_PATH"] = os.path.join(TMP_DIR, "spacy_models")101 102# Define compatible file types for processing103COMPATIBLE_FILE_TYPES = {104 ".pdf",105 ".xlsx",106 ".xls",107 ".png",108 ".jpeg",109 ".csv",110 ".parquet",111 ".txt",112 ".jpg",113}114 115 116def download_file_from_s3(bucket_name, key, download_path):117 """Download a file from S3 to the local filesystem."""118 try:119 s3_client.download_file(bucket_name, key, download_path)120 print("Successfully downloaded file from S3")121 except Exception as e:122 print(f"Error downloading from S3: {e}")123 raise124 125 126def upload_directory_to_s3(local_directory, bucket_name, s3_prefix):127 """Upload all files from a local directory to an S3 prefix."""128 for root, _, files in os.walk(local_directory):129 for file_name in files:130 local_file_path = os.path.join(root, file_name)131 # Create a relative path to maintain directory structure if needed132 relative_path = os.path.relpath(local_file_path, local_directory)133 output_key = os.path.join(s3_prefix, relative_path)134 135 try:136 s3_client.upload_file(local_file_path, bucket_name, output_key)137 print(138 f"Successfully uploaded {local_file_path} to s3://{bucket_name}/{output_key}"139 )140 except Exception as e:141 print(f"Error uploading to S3: {e}")142 raise143 144 145def lambda_handler(event, context):146 print(f"Received event: {json.dumps(event)}")147 148 # 1. Setup temporary directories149 os.makedirs(INPUT_DIR, exist_ok=True)150 os.makedirs(OUTPUT_DIR, exist_ok=True)151 152 # 2. Extract information from the event153 # Assumes the event is triggered by S3 and may contain an 'arguments' payload154 try:155 record = event["Records"][0]156 bucket_name = record["s3"]["bucket"]["name"]157 input_key = record["s3"]["object"]["key"]158 159 # The user metadata can be used to pass arguments160 # This is more robust than embedding them in the main event body161 try:162 response = s3_client.head_object(Bucket=bucket_name, Key=input_key)163 metadata = response.get("Metadata", dict())164 print(f"S3 object metadata: {metadata}")165 166 # Arguments can be passed as a JSON string in metadata167 arguments_str = metadata.get("arguments", "{}")168 print(f"Arguments string from metadata: '{arguments_str}'")169 170 if arguments_str and arguments_str != "{}":171 arguments = json.loads(arguments_str)172 print(f"Successfully parsed arguments from metadata: {arguments}")173 else:174 arguments = dict()175 print("No arguments found in metadata, using empty dictionary")176 except Exception as e:177 print(f"Warning: Could not parse metadata arguments: {e}")178 print("Using empty arguments dictionary")179 arguments = dict()180 181 except (KeyError, IndexError) as e:182 print(183 f"Could not parse S3 event record: {e}. Checking for direct invocation payload."184 )185 # Fallback for direct invocation (e.g., from Step Functions or manual test)186 bucket_name = event.get("bucket_name")187 input_key = event.get("input_key")188 arguments = event.get("arguments", dict())189 if not all([bucket_name, input_key]):190 raise ValueError(191 "Missing 'bucket_name' or 'input_key' in direct invocation event."192 )193 194 # print(f"Processing s3://{bucket_name}/{input_key}")195 # print(f"With arguments: {arguments}")196 # print(f"Arguments type: {type(arguments)}")197 198 # Log file type information199 file_extension = os.path.splitext(input_key)[1].lower()200 print(f"Detected file extension: '{file_extension}'")201 202 # 3. Download the main input file203 input_file_path = os.path.join(INPUT_DIR, os.path.basename(input_key))204 download_file_from_s3(bucket_name, input_key, input_file_path)205 206 # 3.1. Validate file type compatibility207 is_env_file = input_key.lower().endswith(".env")208 209 if not is_env_file and file_extension not in COMPATIBLE_FILE_TYPES:210 error_message = f"File type '{file_extension}' is not supported for processing. Compatible file types are: {', '.join(sorted(COMPATIBLE_FILE_TYPES))}"211 print(f"ERROR: {error_message}")212 print(f"File was not processed due to unsupported file type: {file_extension}")213 return {214 "statusCode": 400,215 "body": json.dumps(216 {217 "error": "Unsupported file type",218 "message": error_message,219 "supported_types": list(COMPATIBLE_FILE_TYPES),220 "received_type": file_extension,221 "file_processed": False,222 }223 ),224 }225 226 print(f"File type '{file_extension}' is compatible for processing")227 if is_env_file:228 print("Processing .env file for configuration")229 else:230 print(f"Processing {file_extension} file for redaction/anonymization")231 232 # 3.5. Check if the downloaded file is a .env file and handle accordingly233 actual_input_file_path = input_file_path234 if input_key.lower().endswith(".env"):235 print("Detected .env file, loading environment variables...")236 237 # Load environment variables from the .env file238 print(f"Loading .env file from: {input_file_path}")239 240 # Check if file exists and is readable241 if os.path.exists(input_file_path):242 print(".env file exists and is readable")243 with open(input_file_path, "r") as f:244 content = f.read()245 print(f".env file content preview: {content[:200]}...")246 else:247 print(f"ERROR: .env file does not exist at {input_file_path}")248 249 load_dotenv(input_file_path, override=True)250 print("Environment variables loaded from .env file")251 252 # Extract the actual input file path from environment variables253 # Look for common environment variable names that might contain the input file path254 env_input_file = os.getenv(255 "INPUT_FILE"256 ) # This needs to be the full S3 path to the input file, e.g.INPUT_FILE=s3://my-processing-bucket/documents/sensitive-data.pdf257 258 if env_input_file:259 print(f"Found input file path in environment: {env_input_file}")260 261 # If the path is an S3 path, download it262 if env_input_file.startswith("s3://"):263 # Parse S3 path: s3://bucket/key264 s3_path_parts = env_input_file[5:].split("/", 1)265 if len(s3_path_parts) == 2:266 env_bucket = s3_path_parts[0]267 env_key = s3_path_parts[1]268 actual_input_file_path = os.path.join(269 INPUT_DIR, os.path.basename(env_key)270 )271 print(272 f"Downloading actual input file from s3://{env_bucket}/{env_key}"273 )274 download_file_from_s3(env_bucket, env_key, actual_input_file_path)275 else:276 print("Warning: Invalid S3 path format in environment variable")277 actual_input_file_path = input_file_path278 else:279 # Assume it's a local path or relative path280 actual_input_file_path = env_input_file281 print(282 f"Using input file path from environment: {actual_input_file_path}"283 )284 else:285 print("Warning: No input file path found in environment variables")286 print(287 "Available environment variables:",288 [289 k290 for k in os.environ.keys()291 if k.startswith(("INPUT", "FILE", "DOCUMENT", "DIRECT"))292 ],293 )294 # Fall back to using the .env file itself (though this might not be what we want)295 actual_input_file_path = input_file_path296 else:297 print("File is not a .env file, proceeding with normal processing")298 299 # 4. Prepare arguments for the CLI function300 # This dictionary should mirror the one in your app.py's "direct mode"301 # If we loaded a .env file, use environment variables as defaults302 303 # Note: For task "combine_review_pdfs" the CLI expects multiple input file paths.304 # Lambda currently passes a single file (the S3-triggered object). To support305 # combine_review_pdfs here, the event would need to supply multiple keys or306 # arguments["input_files"] (list of S3 URIs) and download each before calling cli_main.307 # For task "summarise", PDF input is supported: the CLI extracts text via ocr_method308 # then summarises (same as direct mode and CLI --task summarise --input_file file.pdf).309 cli_args = {310 # Task Selection311 "task": arguments.get("task", os.getenv("DIRECT_MODE_TASK", "redact")),312 # General Arguments (apply to all file types)313 "input_file": actual_input_file_path,314 "output_dir": OUTPUT_DIR,315 "input_dir": INPUT_DIR,316 "language": arguments.get("language", os.getenv("DEFAULT_LANGUAGE", "en")),317 "allow_list": arguments.get("allow_list", os.getenv("ALLOW_LIST_PATH", "")),318 "pii_detector": arguments.get(319 "pii_detector", os.getenv("LOCAL_PII_OPTION", "Local")320 ),321 "username": arguments.get(322 "username", os.getenv("DIRECT_MODE_DEFAULT_USER", LAMBDA_DEFAULT_USERNAME)323 ),324 "save_to_user_folders": convert_string_to_boolean(325 arguments.get(326 "save_to_user_folders", os.getenv("SESSION_OUTPUT_FOLDER", "False")327 )328 ),329 "local_redact_entities": _get_env_list(330 arguments.get(331 "local_redact_entities", os.getenv("CHOSEN_REDACT_ENTITIES", list())332 )333 ),334 "aws_redact_entities": _get_env_list(335 arguments.get(336 "aws_redact_entities", os.getenv("CHOSEN_COMPREHEND_ENTITIES", list())337 )338 ),339 "aws_access_key": None, # Use IAM Role instead of keys340 "aws_secret_key": None, # Use IAM Role instead of keys341 "cost_code": arguments.get("cost_code", os.getenv("DEFAULT_COST_CODE", "")),342 "aws_region": os.getenv("AWS_REGION", ""),343 "s3_bucket": bucket_name,344 "do_initial_clean": arguments.get(345 "do_initial_clean",346 convert_string_to_boolean(347 os.getenv("DO_INITIAL_TABULAR_DATA_CLEAN", "False")348 ),349 ),350 "save_logs_to_csv": convert_string_to_boolean(351 arguments.get("save_logs_to_csv", os.getenv("SAVE_LOGS_TO_CSV", "True"))352 ),353 "save_logs_to_dynamodb": arguments.get(354 "save_logs_to_dynamodb",355 convert_string_to_boolean(os.getenv("SAVE_LOGS_TO_DYNAMODB", "False")),356 ),357 "display_file_names_in_logs": convert_string_to_boolean(358 arguments.get(359 "display_file_names_in_logs",360 os.getenv("DISPLAY_FILE_NAMES_IN_LOGS", "True"),361 )362 ),363 "upload_logs_to_s3": convert_string_to_boolean(364 arguments.get("upload_logs_to_s3", os.getenv("RUN_AWS_FUNCTIONS", "False"))365 ),366 "s3_logs_prefix": arguments.get(367 "s3_logs_prefix", os.getenv("S3_USAGE_LOGS_FOLDER", "")368 ),369 "feedback_logs_folder": arguments.get(370 "feedback_logs_folder",371 os.getenv("FEEDBACK_LOGS_FOLDER", os.environ["FEEDBACK_LOGS_FOLDER"]),372 ),373 "access_logs_folder": arguments.get(374 "access_logs_folder",375 os.getenv("ACCESS_LOGS_FOLDER", os.environ["ACCESS_LOGS_FOLDER"]),376 ),377 "usage_logs_folder": arguments.get(378 "usage_logs_folder",379 os.getenv("USAGE_LOGS_FOLDER", os.environ["USAGE_LOGS_FOLDER"]),380 ),381 "paddle_model_path": arguments.get(382 "paddle_model_path",383 os.getenv("PADDLE_MODEL_PATH", os.environ["PADDLE_MODEL_PATH"]),384 ),385 "spacy_model_path": arguments.get(386 "spacy_model_path",387 os.getenv("SPACY_MODEL_PATH", os.environ["SPACY_MODEL_PATH"]),388 ),389 # PDF/Image Redaction Arguments390 "ocr_method": arguments.get("ocr_method", os.getenv("OCR_METHOD", "Local OCR")),391 "page_min": int(392 arguments.get("page_min", os.getenv("DEFAULT_PAGE_MIN", DEFAULT_PAGE_MIN))393 ),394 "page_max": int(395 arguments.get("page_max", os.getenv("DEFAULT_PAGE_MAX", DEFAULT_PAGE_MAX))396 ),397 "images_dpi": float(398 arguments.get("images_dpi", os.getenv("IMAGES_DPI", IMAGES_DPI))399 ),400 "chosen_local_ocr_model": arguments.get(401 "chosen_local_ocr_model", os.getenv("DEFAULT_LOCAL_OCR_MODEL", "tesseract")402 ),403 "preprocess_local_ocr_images": convert_string_to_boolean(404 arguments.get(405 "preprocess_local_ocr_images",406 os.getenv("PREPROCESS_LOCAL_OCR_IMAGES", "True"),407 )408 ),409 "compress_redacted_pdf": convert_string_to_boolean(410 arguments.get(411 "compress_redacted_pdf", os.getenv("COMPRESS_REDACTED_PDF", "True")412 )413 ),414 "return_pdf_end_of_redaction": convert_string_to_boolean(415 arguments.get(416 "return_pdf_end_of_redaction", os.getenv("RETURN_REDACTED_PDF", "True")417 )418 ),419 "deny_list_file": arguments.get(420 "deny_list_file", os.getenv("DENY_LIST_PATH", "")421 ),422 "allow_list_file": arguments.get(423 "allow_list_file", os.getenv("ALLOW_LIST_PATH", "")424 ),425 "redact_whole_page_file": arguments.get(426 "redact_whole_page_file", os.getenv("WHOLE_PAGE_REDACTION_LIST_PATH", "")427 ),428 "handwrite_signature_extraction": _get_env_list(429 arguments.get(430 "handwrite_signature_extraction",431 os.getenv(432 "DEFAULT_HANDWRITE_SIGNATURE_CHECKBOX",433 ["Extract handwriting", "Extract signatures"],434 ),435 )436 ),437 "extract_forms": convert_string_to_boolean(438 arguments.get(439 "extract_forms",440 os.getenv("INCLUDE_FORM_EXTRACTION_TEXTRACT_OPTION", "False"),441 )442 ),443 "extract_tables": convert_string_to_boolean(444 arguments.get(445 "extract_tables",446 os.getenv("INCLUDE_TABLE_EXTRACTION_TEXTRACT_OPTION", "False"),447 )448 ),449 "extract_layout": convert_string_to_boolean(450 arguments.get(451 "extract_layout",452 os.getenv("INCLUDE_LAYOUT_EXTRACTION_TEXTRACT_OPTION", "False"),453 )454 ),455 # VLM OCR Arguments456 "vlm_model_choice": arguments.get(457 "vlm_model_choice",458 os.getenv("CLOUD_VLM_MODEL_CHOICE", CLOUD_VLM_MODEL_CHOICE),459 ),460 "inference_server_vlm_model": arguments.get(461 "inference_server_vlm_model",462 os.getenv(463 "DEFAULT_INFERENCE_SERVER_VLM_MODEL", DEFAULT_INFERENCE_SERVER_VLM_MODEL464 ),465 ),466 "inference_server_api_url": arguments.get(467 "inference_server_api_url",468 os.getenv("INFERENCE_SERVER_API_URL", INFERENCE_SERVER_API_URL),469 ),470 "gemini_api_key": arguments.get(471 "gemini_api_key", os.getenv("GEMINI_API_KEY", GEMINI_API_KEY)472 ),473 "azure_openai_api_key": arguments.get(474 "azure_openai_api_key",475 os.getenv("AZURE_OPENAI_API_KEY", AZURE_OPENAI_API_KEY),476 ),477 "azure_openai_endpoint": arguments.get(478 "azure_openai_endpoint",479 os.getenv(480 "AZURE_OPENAI_INFERENCE_ENDPOINT", AZURE_OPENAI_INFERENCE_ENDPOINT481 ),482 ),483 "ocr_first_pass_max_workers": int(484 arguments.get(485 "ocr_first_pass_max_workers",486 os.getenv(487 "OCR_FIRST_PASS_MAX_WORKERS", str(OCR_FIRST_PASS_MAX_WORKERS)488 ),489 )490 ),491 "efficient_ocr": convert_string_to_boolean(492 arguments.get(493 "efficient_ocr", os.getenv("EFFICIENT_OCR", str(EFFICIENT_OCR))494 )495 ),496 "efficient_ocr_min_words": int(497 arguments.get(498 "efficient_ocr_min_words",499 os.getenv("EFFICIENT_OCR_MIN_WORDS", str(EFFICIENT_OCR_MIN_WORDS)),500 )501 ),502 "efficient_ocr_min_image_coverage_fraction": float(503 arguments.get(504 "efficient_ocr_min_image_coverage_fraction",505 os.getenv(506 "EFFICIENT_OCR_MIN_IMAGE_COVERAGE_FRACTION",507 str(EFFICIENT_OCR_MIN_IMAGE_COVERAGE_FRACTION),508 ),509 )510 ),511 "efficient_ocr_min_embedded_image_px": int(512 arguments.get(513 "efficient_ocr_min_embedded_image_px",514 os.getenv(515 "EFFICIENT_OCR_MIN_EMBEDDED_IMAGE_PX",516 str(EFFICIENT_OCR_MIN_EMBEDDED_IMAGE_PX),517 ),518 )519 ),520 "hybrid_textract_bedrock_vlm": convert_string_to_boolean(521 arguments.get(522 "hybrid_textract_bedrock_vlm",523 os.getenv(524 "HYBRID_TEXTRACT_BEDROCK_VLM", str(HYBRID_TEXTRACT_BEDROCK_VLM)525 ),526 )527 ),528 # LLM PII Detection Arguments529 # Note: The actual model used is determined by pii_identification_method in the downstream code530 # This is just the default - it will be overridden based on the selected PII method531 "llm_model_choice": arguments.get(532 "llm_model_choice",533 os.getenv("CLOUD_LLM_PII_MODEL_CHOICE", CLOUD_LLM_PII_MODEL_CHOICE),534 ),535 "llm_inference_method": arguments.get(536 "llm_inference_method",537 os.getenv(538 "CHOSEN_LLM_PII_INFERENCE_METHOD", CHOSEN_LLM_PII_INFERENCE_METHOD539 ),540 ),541 "inference_server_pii_model": arguments.get(542 "inference_server_pii_model",543 os.getenv(544 "DEFAULT_INFERENCE_SERVER_PII_MODEL", DEFAULT_INFERENCE_SERVER_PII_MODEL545 ),546 ),547 "llm_temperature": float(548 arguments.get(549 "llm_temperature",550 os.getenv("LLM_TEMPERATURE", LLM_TEMPERATURE),551 )552 ),553 "llm_max_tokens": int(554 arguments.get(555 "llm_max_tokens",556 os.getenv("LLM_MAX_NEW_TOKENS", LLM_MAX_NEW_TOKENS),557 )558 ),559 "llm_redact_entities": _get_env_list(560 arguments.get(561 "llm_redact_entities",562 os.getenv("CHOSEN_LLM_ENTITIES", CHOSEN_LLM_ENTITIES),563 )564 ),565 "custom_llm_instructions": arguments.get(566 "custom_llm_instructions", os.getenv("CUSTOM_LLM_INSTRUCTIONS", "")567 ),568 # Document Summarisation Arguments (used when task is summarise)569 "summarisation_inference_method": arguments.get(570 "summarisation_inference_method",571 os.getenv("SUMMARISATION_INFERENCE_METHOD", AWS_LLM_PII_OPTION),572 ),573 "summarisation_temperature": float(574 arguments.get(575 "summarisation_temperature",576 os.getenv("SUMMARISATION_TEMPERATURE", "0.6"),577 )578 ),579 "summarisation_max_pages_per_group": int(580 arguments.get(581 "summarisation_max_pages_per_group",582 os.getenv("SUMMARISATION_MAX_PAGES_PER_GROUP", "30"),583 )584 ),585 "summary_page_group_max_workers": int(586 arguments.get(587 "summary_page_group_max_workers",588 os.getenv(589 "SUMMARY_PAGE_GROUP_MAX_WORKERS",590 str(SUMMARY_PAGE_GROUP_MAX_WORKERS),591 ),592 )593 ),594 "summarisation_api_key": arguments.get(595 "summarisation_api_key", os.getenv("SUMMARISATION_API_KEY", "")596 ),597 "summarisation_context": arguments.get(598 "summarisation_context", os.getenv("SUMMARISATION_CONTEXT", "")599 ),600 "summarisation_format": arguments.get(601 "summarisation_format", os.getenv("SUMMARISATION_FORMAT", "detailed")602 ),603 "summarisation_additional_instructions": arguments.get(604 "summarisation_additional_instructions",605 os.getenv("SUMMARISATION_ADDITIONAL_INSTRUCTIONS", ""),606 ),607 # Word/Tabular Anonymisation Arguments608 "anon_strategy": arguments.get(609 "anon_strategy",610 os.getenv("DEFAULT_TABULAR_ANONYMISATION_STRATEGY", "redact completely"),611 ),612 "text_columns": arguments.get(613 "text_columns", _get_env_list(os.getenv("DEFAULT_TEXT_COLUMNS", list()))614 ),615 "excel_sheets": arguments.get(616 "excel_sheets", _get_env_list(os.getenv("DEFAULT_EXCEL_SHEETS", list()))617 ),618 "fuzzy_mistakes": int(619 arguments.get(620 "fuzzy_mistakes",621 os.getenv(622 "DEFAULT_FUZZY_SPELLING_MISTAKES_NUM",623 DEFAULT_FUZZY_SPELLING_MISTAKES_NUM,624 ),625 )626 ),627 "match_fuzzy_whole_phrase_bool": convert_string_to_boolean(628 arguments.get(629 "match_fuzzy_whole_phrase_bool",630 os.getenv("MATCH_FUZZY_WHOLE_PHRASE_BOOL", "True"),631 )632 ),633 # Duplicate Detection Arguments634 "duplicate_type": arguments.get(635 "duplicate_type", os.getenv("DIRECT_MODE_DUPLICATE_TYPE", "pages")636 ),637 "similarity_threshold": float(638 arguments.get(639 "similarity_threshold",640 os.getenv(641 "DEFAULT_DUPLICATE_DETECTION_THRESHOLD",642 DEFAULT_DUPLICATE_DETECTION_THRESHOLD,643 ),644 )645 ),646 "min_word_count": int(647 arguments.get(648 "min_word_count",649 os.getenv("DEFAULT_MIN_WORD_COUNT", DEFAULT_MIN_WORD_COUNT),650 )651 ),652 "min_consecutive_pages": int(653 arguments.get(654 "min_consecutive_pages",655 os.getenv(656 "DEFAULT_MIN_CONSECUTIVE_PAGES", DEFAULT_MIN_CONSECUTIVE_PAGES657 ),658 )659 ),660 "greedy_match": convert_string_to_boolean(661 arguments.get(662 "greedy_match", os.getenv("USE_GREEDY_DUPLICATE_DETECTION", "False")663 )664 ),665 "combine_pages": convert_string_to_boolean(666 arguments.get("combine_pages", os.getenv("DEFAULT_COMBINE_PAGES", "True"))667 ),668 "remove_duplicate_rows": convert_string_to_boolean(669 arguments.get(670 "remove_duplicate_rows", os.getenv("REMOVE_DUPLICATE_ROWS", "False")671 )672 ),673 # Textract Batch Operations Arguments674 "textract_action": arguments.get("textract_action", ""),675 "job_id": arguments.get("job_id", ""),676 "extract_signatures": convert_string_to_boolean(677 arguments.get("extract_signatures", str(LAMBDA_EXTRACT_SIGNATURES))678 ),679 "textract_bucket": arguments.get(680 "textract_bucket", os.getenv("TEXTRACT_WHOLE_DOCUMENT_ANALYSIS_BUCKET", "")681 ),682 "textract_input_prefix": arguments.get(683 "textract_input_prefix",684 os.getenv("TEXTRACT_WHOLE_DOCUMENT_ANALYSIS_INPUT_SUBFOLDER", ""),685 ),686 "textract_output_prefix": arguments.get(687 "textract_output_prefix",688 os.getenv("TEXTRACT_WHOLE_DOCUMENT_ANALYSIS_OUTPUT_SUBFOLDER", ""),689 ),690 "s3_textract_document_logs_subfolder": arguments.get(691 "s3_textract_document_logs_subfolder", os.getenv("TEXTRACT_JOBS_S3_LOC", "")692 ),693 "local_textract_document_logs_subfolder": arguments.get(694 "local_textract_document_logs_subfolder",695 os.getenv("TEXTRACT_JOBS_LOCAL_LOC", ""),696 ),697 "poll_interval": int(arguments.get("poll_interval", LAMBDA_POLL_INTERVAL)),698 "max_poll_attempts": int(699 arguments.get("max_poll_attempts", LAMBDA_MAX_POLL_ATTEMPTS)700 ),701 # Additional arguments that were missing702 "search_query": arguments.get(703 "search_query", os.getenv("DEFAULT_SEARCH_QUERY", "")704 ),705 "prepare_images": convert_string_to_boolean(706 arguments.get("prepare_images", str(LAMBDA_PREPARE_IMAGES))707 ),708 }709 710 # Combine extraction options711 extraction_options = (712 _get_env_list(cli_args["handwrite_signature_extraction"])713 if cli_args["handwrite_signature_extraction"]714 else list()715 )716 if cli_args["extract_forms"]:717 extraction_options.append("Extract forms")718 if cli_args["extract_tables"]:719 extraction_options.append("Extract tables")720 if cli_args["extract_layout"]:721 extraction_options.append("Extract layout")722 cli_args["handwrite_signature_extraction"] = extraction_options723 724 # Download optional files if they are specified725 # Note: These can be S3 keys (relative to bucket) or full s3:// paths726 # If they're full s3:// paths, the CLI will handle them automatically727 # If they're S3 keys (not starting with s3:// and not existing locally), download them here728 allow_list_file = arguments.get("allow_list_file") or cli_args.get(729 "allow_list_file"730 )731 if allow_list_file:732 # Check if it's a full S3 path (s3://bucket/key)733 if allow_list_file.startswith("s3://"):734 # Let the CLI handle it - don't download here735 cli_args["allow_list_file"] = allow_list_file736 elif os.path.exists(allow_list_file) or os.path.isabs(allow_list_file):737 # It's already a local absolute path or exists - use it as-is738 cli_args["allow_list_file"] = allow_list_file739 else:740 # Assume it's an S3 key (relative to bucket) - download it741 allow_list_path = os.path.join(INPUT_DIR, "allow_list.csv")742 download_file_from_s3(bucket_name, allow_list_file, allow_list_path)743 cli_args["allow_list_file"] = allow_list_path744 745 deny_list_file = arguments.get("deny_list_file") or cli_args.get("deny_list_file")746 if deny_list_file:747 # Check if it's a full S3 path (s3://bucket/key)748 if deny_list_file.startswith("s3://"):749 # Let the CLI handle it - don't download here750 cli_args["deny_list_file"] = deny_list_file751 elif os.path.exists(deny_list_file) or os.path.isabs(deny_list_file):752 # It's already a local absolute path or exists - use it as-is753 cli_args["deny_list_file"] = deny_list_file754 else:755 # Assume it's an S3 key (relative to bucket) - download it756 deny_list_path = os.path.join(INPUT_DIR, "deny_list.csv")757 download_file_from_s3(bucket_name, deny_list_file, deny_list_path)758 cli_args["deny_list_file"] = deny_list_path759 760 redact_whole_page_file = arguments.get("redact_whole_page_file") or cli_args.get(761 "redact_whole_page_file"762 )763 if redact_whole_page_file:764 # Check if it's a full S3 path (s3://bucket/key)765 if redact_whole_page_file.startswith("s3://"):766 # Let the CLI handle it - don't download here767 cli_args["redact_whole_page_file"] = redact_whole_page_file768 elif os.path.exists(redact_whole_page_file) or os.path.isabs(769 redact_whole_page_file770 ):771 # It's already a local absolute path or exists - use it as-is772 cli_args["redact_whole_page_file"] = redact_whole_page_file773 else:774 # Assume it's an S3 key (relative to bucket) - download it775 redact_whole_page_path = os.path.join(INPUT_DIR, "redact_whole_page.csv")776 download_file_from_s3(777 bucket_name, redact_whole_page_file, redact_whole_page_path778 )779 cli_args["redact_whole_page_file"] = redact_whole_page_path780 781 # 5. Execute the main application logic782 try:783 print("--- Starting CLI Redact Main Function ---")784 cli_main(direct_mode_args=cli_args)785 print("--- CLI Redact Main Function Finished ---")786 except Exception as e:787 print(f"An error occurred during CLI execution: {e}")788 # Optionally, re-raise the exception to make the Lambda fail789 raise790 791 # 6. Upload results back to S3792 output_s3_prefix = f"output/{os.path.splitext(os.path.basename(input_key))[0]}"793 print(794 f"Uploading contents of {OUTPUT_DIR} to s3://{bucket_name}/{output_s3_prefix}/"795 )796 upload_directory_to_s3(OUTPUT_DIR, bucket_name, output_s3_prefix)797 798 return {799 "statusCode": 200,800 "body": json.dumps(801 f"Processing complete for {input_key}. Output saved to s3://{bucket_name}/{output_s3_prefix}/"802 ),803 }804 