Team Ai
Apppublic

seanpedrickcase/document_redaction

sourceHugging Faceagpl-3.0updated 12d agoView on Hugging Face
10likes
lambda_entrypoint.py804 linesDownload Raw Back to root
1import json2import os3 4import boto35from dotenv import load_dotenv6 7# Import the main function from your CLI script8from cli_redact import main as cli_main9from tools.config import (10    AWS_LLM_PII_OPTION,11    AWS_REGION,12    AZURE_OPENAI_API_KEY,13    AZURE_OPENAI_INFERENCE_ENDPOINT,14    CHOSEN_LLM_ENTITIES,15    CHOSEN_LLM_PII_INFERENCE_METHOD,16    CLOUD_LLM_PII_MODEL_CHOICE,17    CLOUD_VLM_MODEL_CHOICE,18    DEFAULT_DUPLICATE_DETECTION_THRESHOLD,19    DEFAULT_FUZZY_SPELLING_MISTAKES_NUM,20    DEFAULT_INFERENCE_SERVER_PII_MODEL,21    DEFAULT_INFERENCE_SERVER_VLM_MODEL,22    DEFAULT_MIN_CONSECUTIVE_PAGES,23    DEFAULT_MIN_WORD_COUNT,24    DEFAULT_PAGE_MAX,25    DEFAULT_PAGE_MIN,26    EFFICIENT_OCR,27    EFFICIENT_OCR_MIN_EMBEDDED_IMAGE_PX,28    EFFICIENT_OCR_MIN_IMAGE_COVERAGE_FRACTION,29    EFFICIENT_OCR_MIN_WORDS,30    GEMINI_API_KEY,31    HYBRID_TEXTRACT_BEDROCK_VLM,32    IMAGES_DPI,33    INFERENCE_SERVER_API_URL,34    LAMBDA_DEFAULT_USERNAME,35    LAMBDA_EXTRACT_SIGNATURES,36    LAMBDA_MAX_POLL_ATTEMPTS,37    LAMBDA_POLL_INTERVAL,38    LAMBDA_PREPARE_IMAGES,39    LLM_MAX_NEW_TOKENS,40    LLM_TEMPERATURE,41    OCR_FIRST_PASS_MAX_WORKERS,42    SUMMARY_PAGE_GROUP_MAX_WORKERS,43)44 45 46def _get_env_list(env_var_name: str | list[str] | None) -> list[str]:47    """Parses a comma-separated environment variable into a list of strings."""48    if isinstance(env_var_name, list):49        return env_var_name50    if env_var_name is None:51        return []52 53    # Handle string input54    value = str(env_var_name).strip()55    if not value or value == "[]":56        return []57 58    # Remove brackets if present (e.g., "[item1, item2]" -> "item1, item2")59    if value.startswith("[") and value.endswith("]"):60        value = value[1:-1]61 62    # Remove quotes and split by comma63    value = value.replace('"', "").replace("'", "")64    if not value:65        return []66 67    # Split by comma and filter out any empty strings68    return [s.strip() for s in value.split(",") if s.strip()]69 70 71def convert_string_to_boolean(value: str) -> bool:72    """Convert string to boolean, handling various formats."""73    if isinstance(value, bool):74        return value75    elif value in ["True", "1", "true", "TRUE"]:76        return True77    elif value in ["False", "0", "false", "FALSE"]:78        return False79    else:80        raise ValueError(f"Invalid boolean value: {value}")81 82 83print("Lambda entrypoint loading...")84 85# Initialize S3 client outside the handler for connection reuse86s3_client = boto3.client("s3", region_name=os.getenv("AWS_REGION", AWS_REGION))87print("S3 client initialised")88 89# Lambda's only writable directory is /tmp. Ensure that all temporary files are stored in this directory.90TMP_DIR = "/tmp"91INPUT_DIR = os.path.join(TMP_DIR, "input")92OUTPUT_DIR = os.path.join(TMP_DIR, "output")93os.environ["TESSERACT_DATA_FOLDER"] = os.path.join(TMP_DIR, "share/tessdata")94os.environ["MPLCONFIGDIR"] = os.path.join(TMP_DIR, "matplotlib_cache")95os.environ["GRADIO_TEMP_DIR"] = os.path.join(TMP_DIR, "gradio_tmp")96os.environ["FEEDBACK_LOGS_FOLDER"] = os.path.join(TMP_DIR, "feedback")97os.environ["ACCESS_LOGS_FOLDER"] = os.path.join(TMP_DIR, "logs")98os.environ["USAGE_LOGS_FOLDER"] = os.path.join(TMP_DIR, "usage")99os.environ["PADDLE_MODEL_PATH"] = os.path.join(TMP_DIR, "paddle_models")100os.environ["SPACY_MODEL_PATH"] = os.path.join(TMP_DIR, "spacy_models")101 102# Define compatible file types for processing103COMPATIBLE_FILE_TYPES = {104    ".pdf",105    ".xlsx",106    ".xls",107    ".png",108    ".jpeg",109    ".csv",110    ".parquet",111    ".txt",112    ".jpg",113}114 115 116def download_file_from_s3(bucket_name, key, download_path):117    """Download a file from S3 to the local filesystem."""118    try:119        s3_client.download_file(bucket_name, key, download_path)120        print("Successfully downloaded file from S3")121    except Exception as e:122        print(f"Error downloading from S3: {e}")123        raise124 125 126def upload_directory_to_s3(local_directory, bucket_name, s3_prefix):127    """Upload all files from a local directory to an S3 prefix."""128    for root, _, files in os.walk(local_directory):129        for file_name in files:130            local_file_path = os.path.join(root, file_name)131            # Create a relative path to maintain directory structure if needed132            relative_path = os.path.relpath(local_file_path, local_directory)133            output_key = os.path.join(s3_prefix, relative_path)134 135            try:136                s3_client.upload_file(local_file_path, bucket_name, output_key)137                print(138                    f"Successfully uploaded {local_file_path} to s3://{bucket_name}/{output_key}"139                )140            except Exception as e:141                print(f"Error uploading to S3: {e}")142                raise143 144 145def lambda_handler(event, context):146    print(f"Received event: {json.dumps(event)}")147 148    # 1. Setup temporary directories149    os.makedirs(INPUT_DIR, exist_ok=True)150    os.makedirs(OUTPUT_DIR, exist_ok=True)151 152    # 2. Extract information from the event153    # Assumes the event is triggered by S3 and may contain an 'arguments' payload154    try:155        record = event["Records"][0]156        bucket_name = record["s3"]["bucket"]["name"]157        input_key = record["s3"]["object"]["key"]158 159        # The user metadata can be used to pass arguments160        # This is more robust than embedding them in the main event body161        try:162            response = s3_client.head_object(Bucket=bucket_name, Key=input_key)163            metadata = response.get("Metadata", dict())164            print(f"S3 object metadata: {metadata}")165 166            # Arguments can be passed as a JSON string in metadata167            arguments_str = metadata.get("arguments", "{}")168            print(f"Arguments string from metadata: '{arguments_str}'")169 170            if arguments_str and arguments_str != "{}":171                arguments = json.loads(arguments_str)172                print(f"Successfully parsed arguments from metadata: {arguments}")173            else:174                arguments = dict()175                print("No arguments found in metadata, using empty dictionary")176        except Exception as e:177            print(f"Warning: Could not parse metadata arguments: {e}")178            print("Using empty arguments dictionary")179            arguments = dict()180 181    except (KeyError, IndexError) as e:182        print(183            f"Could not parse S3 event record: {e}. Checking for direct invocation payload."184        )185        # Fallback for direct invocation (e.g., from Step Functions or manual test)186        bucket_name = event.get("bucket_name")187        input_key = event.get("input_key")188        arguments = event.get("arguments", dict())189        if not all([bucket_name, input_key]):190            raise ValueError(191                "Missing 'bucket_name' or 'input_key' in direct invocation event."192            )193 194    # print(f"Processing s3://{bucket_name}/{input_key}")195    # print(f"With arguments: {arguments}")196    # print(f"Arguments type: {type(arguments)}")197 198    # Log file type information199    file_extension = os.path.splitext(input_key)[1].lower()200    print(f"Detected file extension: '{file_extension}'")201 202    # 3. Download the main input file203    input_file_path = os.path.join(INPUT_DIR, os.path.basename(input_key))204    download_file_from_s3(bucket_name, input_key, input_file_path)205 206    # 3.1. Validate file type compatibility207    is_env_file = input_key.lower().endswith(".env")208 209    if not is_env_file and file_extension not in COMPATIBLE_FILE_TYPES:210        error_message = f"File type '{file_extension}' is not supported for processing. Compatible file types are: {', '.join(sorted(COMPATIBLE_FILE_TYPES))}"211        print(f"ERROR: {error_message}")212        print(f"File was not processed due to unsupported file type: {file_extension}")213        return {214            "statusCode": 400,215            "body": json.dumps(216                {217                    "error": "Unsupported file type",218                    "message": error_message,219                    "supported_types": list(COMPATIBLE_FILE_TYPES),220                    "received_type": file_extension,221                    "file_processed": False,222                }223            ),224        }225 226    print(f"File type '{file_extension}' is compatible for processing")227    if is_env_file:228        print("Processing .env file for configuration")229    else:230        print(f"Processing {file_extension} file for redaction/anonymization")231 232    # 3.5. Check if the downloaded file is a .env file and handle accordingly233    actual_input_file_path = input_file_path234    if input_key.lower().endswith(".env"):235        print("Detected .env file, loading environment variables...")236 237        # Load environment variables from the .env file238        print(f"Loading .env file from: {input_file_path}")239 240        # Check if file exists and is readable241        if os.path.exists(input_file_path):242            print(".env file exists and is readable")243            with open(input_file_path, "r") as f:244                content = f.read()245                print(f".env file content preview: {content[:200]}...")246        else:247            print(f"ERROR: .env file does not exist at {input_file_path}")248 249        load_dotenv(input_file_path, override=True)250        print("Environment variables loaded from .env file")251 252        # Extract the actual input file path from environment variables253        # Look for common environment variable names that might contain the input file path254        env_input_file = os.getenv(255            "INPUT_FILE"256        )  # This needs to be the full S3 path to the input file, e.g.INPUT_FILE=s3://my-processing-bucket/documents/sensitive-data.pdf257 258        if env_input_file:259            print(f"Found input file path in environment: {env_input_file}")260 261            # If the path is an S3 path, download it262            if env_input_file.startswith("s3://"):263                # Parse S3 path: s3://bucket/key264                s3_path_parts = env_input_file[5:].split("/", 1)265                if len(s3_path_parts) == 2:266                    env_bucket = s3_path_parts[0]267                    env_key = s3_path_parts[1]268                    actual_input_file_path = os.path.join(269                        INPUT_DIR, os.path.basename(env_key)270                    )271                    print(272                        f"Downloading actual input file from s3://{env_bucket}/{env_key}"273                    )274                    download_file_from_s3(env_bucket, env_key, actual_input_file_path)275                else:276                    print("Warning: Invalid S3 path format in environment variable")277                    actual_input_file_path = input_file_path278            else:279                # Assume it's a local path or relative path280                actual_input_file_path = env_input_file281                print(282                    f"Using input file path from environment: {actual_input_file_path}"283                )284        else:285            print("Warning: No input file path found in environment variables")286            print(287                "Available environment variables:",288                [289                    k290                    for k in os.environ.keys()291                    if k.startswith(("INPUT", "FILE", "DOCUMENT", "DIRECT"))292                ],293            )294            # Fall back to using the .env file itself (though this might not be what we want)295            actual_input_file_path = input_file_path296    else:297        print("File is not a .env file, proceeding with normal processing")298 299    # 4. Prepare arguments for the CLI function300    # This dictionary should mirror the one in your app.py's "direct mode"301    # If we loaded a .env file, use environment variables as defaults302 303    # Note: For task "combine_review_pdfs" the CLI expects multiple input file paths.304    # Lambda currently passes a single file (the S3-triggered object). To support305    # combine_review_pdfs here, the event would need to supply multiple keys or306    # arguments["input_files"] (list of S3 URIs) and download each before calling cli_main.307    # For task "summarise", PDF input is supported: the CLI extracts text via ocr_method308    # then summarises (same as direct mode and CLI --task summarise --input_file file.pdf).309    cli_args = {310        # Task Selection311        "task": arguments.get("task", os.getenv("DIRECT_MODE_TASK", "redact")),312        # General Arguments (apply to all file types)313        "input_file": actual_input_file_path,314        "output_dir": OUTPUT_DIR,315        "input_dir": INPUT_DIR,316        "language": arguments.get("language", os.getenv("DEFAULT_LANGUAGE", "en")),317        "allow_list": arguments.get("allow_list", os.getenv("ALLOW_LIST_PATH", "")),318        "pii_detector": arguments.get(319            "pii_detector", os.getenv("LOCAL_PII_OPTION", "Local")320        ),321        "username": arguments.get(322            "username", os.getenv("DIRECT_MODE_DEFAULT_USER", LAMBDA_DEFAULT_USERNAME)323        ),324        "save_to_user_folders": convert_string_to_boolean(325            arguments.get(326                "save_to_user_folders", os.getenv("SESSION_OUTPUT_FOLDER", "False")327            )328        ),329        "local_redact_entities": _get_env_list(330            arguments.get(331                "local_redact_entities", os.getenv("CHOSEN_REDACT_ENTITIES", list())332            )333        ),334        "aws_redact_entities": _get_env_list(335            arguments.get(336                "aws_redact_entities", os.getenv("CHOSEN_COMPREHEND_ENTITIES", list())337            )338        ),339        "aws_access_key": None,  # Use IAM Role instead of keys340        "aws_secret_key": None,  # Use IAM Role instead of keys341        "cost_code": arguments.get("cost_code", os.getenv("DEFAULT_COST_CODE", "")),342        "aws_region": os.getenv("AWS_REGION", ""),343        "s3_bucket": bucket_name,344        "do_initial_clean": arguments.get(345            "do_initial_clean",346            convert_string_to_boolean(347                os.getenv("DO_INITIAL_TABULAR_DATA_CLEAN", "False")348            ),349        ),350        "save_logs_to_csv": convert_string_to_boolean(351            arguments.get("save_logs_to_csv", os.getenv("SAVE_LOGS_TO_CSV", "True"))352        ),353        "save_logs_to_dynamodb": arguments.get(354            "save_logs_to_dynamodb",355            convert_string_to_boolean(os.getenv("SAVE_LOGS_TO_DYNAMODB", "False")),356        ),357        "display_file_names_in_logs": convert_string_to_boolean(358            arguments.get(359                "display_file_names_in_logs",360                os.getenv("DISPLAY_FILE_NAMES_IN_LOGS", "True"),361            )362        ),363        "upload_logs_to_s3": convert_string_to_boolean(364            arguments.get("upload_logs_to_s3", os.getenv("RUN_AWS_FUNCTIONS", "False"))365        ),366        "s3_logs_prefix": arguments.get(367            "s3_logs_prefix", os.getenv("S3_USAGE_LOGS_FOLDER", "")368        ),369        "feedback_logs_folder": arguments.get(370            "feedback_logs_folder",371            os.getenv("FEEDBACK_LOGS_FOLDER", os.environ["FEEDBACK_LOGS_FOLDER"]),372        ),373        "access_logs_folder": arguments.get(374            "access_logs_folder",375            os.getenv("ACCESS_LOGS_FOLDER", os.environ["ACCESS_LOGS_FOLDER"]),376        ),377        "usage_logs_folder": arguments.get(378            "usage_logs_folder",379            os.getenv("USAGE_LOGS_FOLDER", os.environ["USAGE_LOGS_FOLDER"]),380        ),381        "paddle_model_path": arguments.get(382            "paddle_model_path",383            os.getenv("PADDLE_MODEL_PATH", os.environ["PADDLE_MODEL_PATH"]),384        ),385        "spacy_model_path": arguments.get(386            "spacy_model_path",387            os.getenv("SPACY_MODEL_PATH", os.environ["SPACY_MODEL_PATH"]),388        ),389        # PDF/Image Redaction Arguments390        "ocr_method": arguments.get("ocr_method", os.getenv("OCR_METHOD", "Local OCR")),391        "page_min": int(392            arguments.get("page_min", os.getenv("DEFAULT_PAGE_MIN", DEFAULT_PAGE_MIN))393        ),394        "page_max": int(395            arguments.get("page_max", os.getenv("DEFAULT_PAGE_MAX", DEFAULT_PAGE_MAX))396        ),397        "images_dpi": float(398            arguments.get("images_dpi", os.getenv("IMAGES_DPI", IMAGES_DPI))399        ),400        "chosen_local_ocr_model": arguments.get(401            "chosen_local_ocr_model", os.getenv("DEFAULT_LOCAL_OCR_MODEL", "tesseract")402        ),403        "preprocess_local_ocr_images": convert_string_to_boolean(404            arguments.get(405                "preprocess_local_ocr_images",406                os.getenv("PREPROCESS_LOCAL_OCR_IMAGES", "True"),407            )408        ),409        "compress_redacted_pdf": convert_string_to_boolean(410            arguments.get(411                "compress_redacted_pdf", os.getenv("COMPRESS_REDACTED_PDF", "True")412            )413        ),414        "return_pdf_end_of_redaction": convert_string_to_boolean(415            arguments.get(416                "return_pdf_end_of_redaction", os.getenv("RETURN_REDACTED_PDF", "True")417            )418        ),419        "deny_list_file": arguments.get(420            "deny_list_file", os.getenv("DENY_LIST_PATH", "")421        ),422        "allow_list_file": arguments.get(423            "allow_list_file", os.getenv("ALLOW_LIST_PATH", "")424        ),425        "redact_whole_page_file": arguments.get(426            "redact_whole_page_file", os.getenv("WHOLE_PAGE_REDACTION_LIST_PATH", "")427        ),428        "handwrite_signature_extraction": _get_env_list(429            arguments.get(430                "handwrite_signature_extraction",431                os.getenv(432                    "DEFAULT_HANDWRITE_SIGNATURE_CHECKBOX",433                    ["Extract handwriting", "Extract signatures"],434                ),435            )436        ),437        "extract_forms": convert_string_to_boolean(438            arguments.get(439                "extract_forms",440                os.getenv("INCLUDE_FORM_EXTRACTION_TEXTRACT_OPTION", "False"),441            )442        ),443        "extract_tables": convert_string_to_boolean(444            arguments.get(445                "extract_tables",446                os.getenv("INCLUDE_TABLE_EXTRACTION_TEXTRACT_OPTION", "False"),447            )448        ),449        "extract_layout": convert_string_to_boolean(450            arguments.get(451                "extract_layout",452                os.getenv("INCLUDE_LAYOUT_EXTRACTION_TEXTRACT_OPTION", "False"),453            )454        ),455        # VLM OCR Arguments456        "vlm_model_choice": arguments.get(457            "vlm_model_choice",458            os.getenv("CLOUD_VLM_MODEL_CHOICE", CLOUD_VLM_MODEL_CHOICE),459        ),460        "inference_server_vlm_model": arguments.get(461            "inference_server_vlm_model",462            os.getenv(463                "DEFAULT_INFERENCE_SERVER_VLM_MODEL", DEFAULT_INFERENCE_SERVER_VLM_MODEL464            ),465        ),466        "inference_server_api_url": arguments.get(467            "inference_server_api_url",468            os.getenv("INFERENCE_SERVER_API_URL", INFERENCE_SERVER_API_URL),469        ),470        "gemini_api_key": arguments.get(471            "gemini_api_key", os.getenv("GEMINI_API_KEY", GEMINI_API_KEY)472        ),473        "azure_openai_api_key": arguments.get(474            "azure_openai_api_key",475            os.getenv("AZURE_OPENAI_API_KEY", AZURE_OPENAI_API_KEY),476        ),477        "azure_openai_endpoint": arguments.get(478            "azure_openai_endpoint",479            os.getenv(480                "AZURE_OPENAI_INFERENCE_ENDPOINT", AZURE_OPENAI_INFERENCE_ENDPOINT481            ),482        ),483        "ocr_first_pass_max_workers": int(484            arguments.get(485                "ocr_first_pass_max_workers",486                os.getenv(487                    "OCR_FIRST_PASS_MAX_WORKERS", str(OCR_FIRST_PASS_MAX_WORKERS)488                ),489            )490        ),491        "efficient_ocr": convert_string_to_boolean(492            arguments.get(493                "efficient_ocr", os.getenv("EFFICIENT_OCR", str(EFFICIENT_OCR))494            )495        ),496        "efficient_ocr_min_words": int(497            arguments.get(498                "efficient_ocr_min_words",499                os.getenv("EFFICIENT_OCR_MIN_WORDS", str(EFFICIENT_OCR_MIN_WORDS)),500            )501        ),502        "efficient_ocr_min_image_coverage_fraction": float(503            arguments.get(504                "efficient_ocr_min_image_coverage_fraction",505                os.getenv(506                    "EFFICIENT_OCR_MIN_IMAGE_COVERAGE_FRACTION",507                    str(EFFICIENT_OCR_MIN_IMAGE_COVERAGE_FRACTION),508                ),509            )510        ),511        "efficient_ocr_min_embedded_image_px": int(512            arguments.get(513                "efficient_ocr_min_embedded_image_px",514                os.getenv(515                    "EFFICIENT_OCR_MIN_EMBEDDED_IMAGE_PX",516                    str(EFFICIENT_OCR_MIN_EMBEDDED_IMAGE_PX),517                ),518            )519        ),520        "hybrid_textract_bedrock_vlm": convert_string_to_boolean(521            arguments.get(522                "hybrid_textract_bedrock_vlm",523                os.getenv(524                    "HYBRID_TEXTRACT_BEDROCK_VLM", str(HYBRID_TEXTRACT_BEDROCK_VLM)525                ),526            )527        ),528        # LLM PII Detection Arguments529        # Note: The actual model used is determined by pii_identification_method in the downstream code530        # This is just the default - it will be overridden based on the selected PII method531        "llm_model_choice": arguments.get(532            "llm_model_choice",533            os.getenv("CLOUD_LLM_PII_MODEL_CHOICE", CLOUD_LLM_PII_MODEL_CHOICE),534        ),535        "llm_inference_method": arguments.get(536            "llm_inference_method",537            os.getenv(538                "CHOSEN_LLM_PII_INFERENCE_METHOD", CHOSEN_LLM_PII_INFERENCE_METHOD539            ),540        ),541        "inference_server_pii_model": arguments.get(542            "inference_server_pii_model",543            os.getenv(544                "DEFAULT_INFERENCE_SERVER_PII_MODEL", DEFAULT_INFERENCE_SERVER_PII_MODEL545            ),546        ),547        "llm_temperature": float(548            arguments.get(549                "llm_temperature",550                os.getenv("LLM_TEMPERATURE", LLM_TEMPERATURE),551            )552        ),553        "llm_max_tokens": int(554            arguments.get(555                "llm_max_tokens",556                os.getenv("LLM_MAX_NEW_TOKENS", LLM_MAX_NEW_TOKENS),557            )558        ),559        "llm_redact_entities": _get_env_list(560            arguments.get(561                "llm_redact_entities",562                os.getenv("CHOSEN_LLM_ENTITIES", CHOSEN_LLM_ENTITIES),563            )564        ),565        "custom_llm_instructions": arguments.get(566            "custom_llm_instructions", os.getenv("CUSTOM_LLM_INSTRUCTIONS", "")567        ),568        # Document Summarisation Arguments (used when task is summarise)569        "summarisation_inference_method": arguments.get(570            "summarisation_inference_method",571            os.getenv("SUMMARISATION_INFERENCE_METHOD", AWS_LLM_PII_OPTION),572        ),573        "summarisation_temperature": float(574            arguments.get(575                "summarisation_temperature",576                os.getenv("SUMMARISATION_TEMPERATURE", "0.6"),577            )578        ),579        "summarisation_max_pages_per_group": int(580            arguments.get(581                "summarisation_max_pages_per_group",582                os.getenv("SUMMARISATION_MAX_PAGES_PER_GROUP", "30"),583            )584        ),585        "summary_page_group_max_workers": int(586            arguments.get(587                "summary_page_group_max_workers",588                os.getenv(589                    "SUMMARY_PAGE_GROUP_MAX_WORKERS",590                    str(SUMMARY_PAGE_GROUP_MAX_WORKERS),591                ),592            )593        ),594        "summarisation_api_key": arguments.get(595            "summarisation_api_key", os.getenv("SUMMARISATION_API_KEY", "")596        ),597        "summarisation_context": arguments.get(598            "summarisation_context", os.getenv("SUMMARISATION_CONTEXT", "")599        ),600        "summarisation_format": arguments.get(601            "summarisation_format", os.getenv("SUMMARISATION_FORMAT", "detailed")602        ),603        "summarisation_additional_instructions": arguments.get(604            "summarisation_additional_instructions",605            os.getenv("SUMMARISATION_ADDITIONAL_INSTRUCTIONS", ""),606        ),607        # Word/Tabular Anonymisation Arguments608        "anon_strategy": arguments.get(609            "anon_strategy",610            os.getenv("DEFAULT_TABULAR_ANONYMISATION_STRATEGY", "redact completely"),611        ),612        "text_columns": arguments.get(613            "text_columns", _get_env_list(os.getenv("DEFAULT_TEXT_COLUMNS", list()))614        ),615        "excel_sheets": arguments.get(616            "excel_sheets", _get_env_list(os.getenv("DEFAULT_EXCEL_SHEETS", list()))617        ),618        "fuzzy_mistakes": int(619            arguments.get(620                "fuzzy_mistakes",621                os.getenv(622                    "DEFAULT_FUZZY_SPELLING_MISTAKES_NUM",623                    DEFAULT_FUZZY_SPELLING_MISTAKES_NUM,624                ),625            )626        ),627        "match_fuzzy_whole_phrase_bool": convert_string_to_boolean(628            arguments.get(629                "match_fuzzy_whole_phrase_bool",630                os.getenv("MATCH_FUZZY_WHOLE_PHRASE_BOOL", "True"),631            )632        ),633        # Duplicate Detection Arguments634        "duplicate_type": arguments.get(635            "duplicate_type", os.getenv("DIRECT_MODE_DUPLICATE_TYPE", "pages")636        ),637        "similarity_threshold": float(638            arguments.get(639                "similarity_threshold",640                os.getenv(641                    "DEFAULT_DUPLICATE_DETECTION_THRESHOLD",642                    DEFAULT_DUPLICATE_DETECTION_THRESHOLD,643                ),644            )645        ),646        "min_word_count": int(647            arguments.get(648                "min_word_count",649                os.getenv("DEFAULT_MIN_WORD_COUNT", DEFAULT_MIN_WORD_COUNT),650            )651        ),652        "min_consecutive_pages": int(653            arguments.get(654                "min_consecutive_pages",655                os.getenv(656                    "DEFAULT_MIN_CONSECUTIVE_PAGES", DEFAULT_MIN_CONSECUTIVE_PAGES657                ),658            )659        ),660        "greedy_match": convert_string_to_boolean(661            arguments.get(662                "greedy_match", os.getenv("USE_GREEDY_DUPLICATE_DETECTION", "False")663            )664        ),665        "combine_pages": convert_string_to_boolean(666            arguments.get("combine_pages", os.getenv("DEFAULT_COMBINE_PAGES", "True"))667        ),668        "remove_duplicate_rows": convert_string_to_boolean(669            arguments.get(670                "remove_duplicate_rows", os.getenv("REMOVE_DUPLICATE_ROWS", "False")671            )672        ),673        # Textract Batch Operations Arguments674        "textract_action": arguments.get("textract_action", ""),675        "job_id": arguments.get("job_id", ""),676        "extract_signatures": convert_string_to_boolean(677            arguments.get("extract_signatures", str(LAMBDA_EXTRACT_SIGNATURES))678        ),679        "textract_bucket": arguments.get(680            "textract_bucket", os.getenv("TEXTRACT_WHOLE_DOCUMENT_ANALYSIS_BUCKET", "")681        ),682        "textract_input_prefix": arguments.get(683            "textract_input_prefix",684            os.getenv("TEXTRACT_WHOLE_DOCUMENT_ANALYSIS_INPUT_SUBFOLDER", ""),685        ),686        "textract_output_prefix": arguments.get(687            "textract_output_prefix",688            os.getenv("TEXTRACT_WHOLE_DOCUMENT_ANALYSIS_OUTPUT_SUBFOLDER", ""),689        ),690        "s3_textract_document_logs_subfolder": arguments.get(691            "s3_textract_document_logs_subfolder", os.getenv("TEXTRACT_JOBS_S3_LOC", "")692        ),693        "local_textract_document_logs_subfolder": arguments.get(694            "local_textract_document_logs_subfolder",695            os.getenv("TEXTRACT_JOBS_LOCAL_LOC", ""),696        ),697        "poll_interval": int(arguments.get("poll_interval", LAMBDA_POLL_INTERVAL)),698        "max_poll_attempts": int(699            arguments.get("max_poll_attempts", LAMBDA_MAX_POLL_ATTEMPTS)700        ),701        # Additional arguments that were missing702        "search_query": arguments.get(703            "search_query", os.getenv("DEFAULT_SEARCH_QUERY", "")704        ),705        "prepare_images": convert_string_to_boolean(706            arguments.get("prepare_images", str(LAMBDA_PREPARE_IMAGES))707        ),708    }709 710    # Combine extraction options711    extraction_options = (712        _get_env_list(cli_args["handwrite_signature_extraction"])713        if cli_args["handwrite_signature_extraction"]714        else list()715    )716    if cli_args["extract_forms"]:717        extraction_options.append("Extract forms")718    if cli_args["extract_tables"]:719        extraction_options.append("Extract tables")720    if cli_args["extract_layout"]:721        extraction_options.append("Extract layout")722    cli_args["handwrite_signature_extraction"] = extraction_options723 724    # Download optional files if they are specified725    # Note: These can be S3 keys (relative to bucket) or full s3:// paths726    # If they're full s3:// paths, the CLI will handle them automatically727    # If they're S3 keys (not starting with s3:// and not existing locally), download them here728    allow_list_file = arguments.get("allow_list_file") or cli_args.get(729        "allow_list_file"730    )731    if allow_list_file:732        # Check if it's a full S3 path (s3://bucket/key)733        if allow_list_file.startswith("s3://"):734            # Let the CLI handle it - don't download here735            cli_args["allow_list_file"] = allow_list_file736        elif os.path.exists(allow_list_file) or os.path.isabs(allow_list_file):737            # It's already a local absolute path or exists - use it as-is738            cli_args["allow_list_file"] = allow_list_file739        else:740            # Assume it's an S3 key (relative to bucket) - download it741            allow_list_path = os.path.join(INPUT_DIR, "allow_list.csv")742            download_file_from_s3(bucket_name, allow_list_file, allow_list_path)743            cli_args["allow_list_file"] = allow_list_path744 745    deny_list_file = arguments.get("deny_list_file") or cli_args.get("deny_list_file")746    if deny_list_file:747        # Check if it's a full S3 path (s3://bucket/key)748        if deny_list_file.startswith("s3://"):749            # Let the CLI handle it - don't download here750            cli_args["deny_list_file"] = deny_list_file751        elif os.path.exists(deny_list_file) or os.path.isabs(deny_list_file):752            # It's already a local absolute path or exists - use it as-is753            cli_args["deny_list_file"] = deny_list_file754        else:755            # Assume it's an S3 key (relative to bucket) - download it756            deny_list_path = os.path.join(INPUT_DIR, "deny_list.csv")757            download_file_from_s3(bucket_name, deny_list_file, deny_list_path)758            cli_args["deny_list_file"] = deny_list_path759 760    redact_whole_page_file = arguments.get("redact_whole_page_file") or cli_args.get(761        "redact_whole_page_file"762    )763    if redact_whole_page_file:764        # Check if it's a full S3 path (s3://bucket/key)765        if redact_whole_page_file.startswith("s3://"):766            # Let the CLI handle it - don't download here767            cli_args["redact_whole_page_file"] = redact_whole_page_file768        elif os.path.exists(redact_whole_page_file) or os.path.isabs(769            redact_whole_page_file770        ):771            # It's already a local absolute path or exists - use it as-is772            cli_args["redact_whole_page_file"] = redact_whole_page_file773        else:774            # Assume it's an S3 key (relative to bucket) - download it775            redact_whole_page_path = os.path.join(INPUT_DIR, "redact_whole_page.csv")776            download_file_from_s3(777                bucket_name, redact_whole_page_file, redact_whole_page_path778            )779            cli_args["redact_whole_page_file"] = redact_whole_page_path780 781    # 5. Execute the main application logic782    try:783        print("--- Starting CLI Redact Main Function ---")784        cli_main(direct_mode_args=cli_args)785        print("--- CLI Redact Main Function Finished ---")786    except Exception as e:787        print(f"An error occurred during CLI execution: {e}")788        # Optionally, re-raise the exception to make the Lambda fail789        raise790 791    # 6. Upload results back to S3792    output_s3_prefix = f"output/{os.path.splitext(os.path.basename(input_key))[0]}"793    print(794        f"Uploading contents of {OUTPUT_DIR} to s3://{bucket_name}/{output_s3_prefix}/"795    )796    upload_directory_to_s3(OUTPUT_DIR, bucket_name, output_s3_prefix)797 798    return {799        "statusCode": 200,800        "body": json.dumps(801            f"Processing complete for {input_key}. Output saved to s3://{bucket_name}/{output_s3_prefix}/"802        ),803    }804