From b260c23ebe0d479f1d19c3a638c910763dfca7ea Mon Sep 17 00:00:00 2001 From: Pankaj Katariya Date: Wed, 13 Mar 2024 13:11:19 +0530 Subject: [PATCH] Added validation lambda code --- lambda/pdf-validation/lambda_function.py | 143 +++++++++++++++++++++++ 1 file changed, 143 insertions(+) create mode 100644 lambda/pdf-validation/lambda_function.py diff --git a/lambda/pdf-validation/lambda_function.py b/lambda/pdf-validation/lambda_function.py new file mode 100644 index 0000000..88eb546 --- /dev/null +++ b/lambda/pdf-validation/lambda_function.py @@ -0,0 +1,143 @@ +import boto3 +import logging +import botocore +from PyPDF2 import PdfReader +from io import BytesIO +import os +from configparser import ConfigParser +import urllib.parse +from urllib.parse import urlparse + +# Configure logging +logger = logging.getLogger() +logger.setLevel(logging.INFO) + +# Initialize S3 & Textract clients +s3_client = boto3.client('s3') + +# Function to load configuration from S3 +def load_config_from_s3(bucket_name, file_key): + + # Download the config file from S3 + response = s3_client.get_object(Bucket=bucket_name, Key=file_key) + config_content = response['Body'].read().decode('utf-8') + + # Parse the config file + config_parser = ConfigParser() + config_parser.read_string(config_content) + + # Convert the configuration to a dictionary + config_dict = {} + for section in config_parser.sections(): + config_dict[section] = {key.upper(): value for key, value in config_parser.items(section)} + + return config_dict + +def lambda_handler(event, context): + + for record in event['Records']: + + # Retrieve the S3 bucket and key from the event + bucket = event['Records'][0]['s3']['bucket']['name'] + key = urllib.parse.unquote_plus(event['Records'][0]['s3']['object']['key']) + region = record['awsRegion'] + + logger.info(f"Processing S3 file - Bucket: {bucket}, Key: {key}, Region: {region}") + + # Read environment variables + property_file_path = os.environ.get('PROPERTY_FILE_S3_PATH', '') + batch_id = os.environ.get('BATCH_ID', '') + logger.info(f"Batch ID: {batch_id}") + file_path_array = property_file_path.split("/") + # Extract BUCKET_NAME and config_file_path + S3_BUCKET_NAME = file_path_array[0] + CONFIG_FILE_PATH = "/".join(file_path_array[1:]) + logger.info(f"Using S3 bucket: {S3_BUCKET_NAME} and config file path: {CONFIG_FILE_PATH}") + + # Load config file + config_dict = load_config_from_s3(bucket, CONFIG_FILE_PATH) + INVALID_PDF_FILE_LOCATION = config_dict['FOLDER_LOCATIONS']['INVALID_PDF_FILE_LOCATION'].format(batch_id) + SOURCE_LOCATION = config_dict['FOLDER_LOCATIONS']['SOURCE_LOCATION'].format(batch_id) + logger.info("Config file loaded successfully.") + + logger.info(f"Processing file: s3://{bucket}/{key}") + + + # Check if the file has a '.filepart' extension + if key.lower().endswith('.filepart'): + new_key = key[:-9] + '.pdf' # Rename to have a '.pdf' extension + s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=new_key) + s3_client.delete_object(Bucket=bucket, Key=key) + key = new_key # Update key to the new filename + + + + # Validate number of pages + try: + pdf_reader = PdfReader(BytesIO(s3_client.get_object(Bucket=bucket, Key=key)['Body'].read())) + if is_password_protected(pdf_reader): + logger.info("File is password protected. Moving to 'unprocessed' folder.") + move_to_unprocessed(bucket, key,INVALID_PDF_FILE_LOCATION) + # Validate file size + file_size = s3_client.head_object(Bucket=bucket, Key=key)['ContentLength'] + logger.info(f"File size: {file_size} bytes") + if file_size > 500 * 1024 * 1024: # 500MB + logger.info("File size exceeds 500MB. Moving to 'unprocessed' folder.") + move_to_unprocessed(bucket, key,INVALID_PDF_FILE_LOCATION) + return + + num_pages = len(pdf_reader.pages) + logger.info(f"Number of pages: {num_pages}") + if num_pages > 3000: + logger.info("Number of pages exceeds 3000. Moving to 'unprocessed' folder.") + move_to_unprocessed(bucket,key,INVALID_PDF_FILE_LOCATION) + return + except botocore.exceptions.ClientError as e: + if e.response['Error']['Code'] == '404': + logger.error(f"File not found: s3://{bucket}/{key}") + # Handle the case where the file doesn't exist + return + else: + logger.error(f"Error checking file size: {str(e)}. Moving to 'unprocessed' folder.") + move_to_unprocessed(bucket, key, INVALID_PDF_FILE_LOCATION) + return + except Exception as e: + logger.error(f"Error checking file size: {str(e)}. Moving to 'unprocessed' folder.") + move_to_unprocessed(bucket, key, INVALID_PDF_FILE_LOCATION) + return + + # Validate password protection and resolution + + if not is_resolution_valid(pdf_reader): + logger.info("File has invalid resolution. Moving to 'unprocessed' folder.") + move_to_unprocessed(bucket, key,INVALID_PDF_FILE_LOCATION) + else: + logger.info("File passed all conditions. Moving to 'processed' folder.") + move_to_processed(bucket, key,SOURCE_LOCATION) + +def is_password_protected(pdf_reader): + return pdf_reader.is_encrypted + +def is_resolution_valid(pdf_reader, max_resolution=3000*4000): + for page_num in range(len(pdf_reader.pages)): + page = pdf_reader.pages[page_num] + page_width, page_height = page.mediabox.upper_right + if page_width > max_resolution or page_height > max_resolution: + return False + return True + + +def move_to_processed(bucket, key,valid_file_location): + #replace processed with valid_file_location like in unprocessed method + #s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=f'processed/{key}') + s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=f'{valid_file_location}/{get_filename_from_path(key)}') + s3_client.delete_object(Bucket=bucket, Key=key) + logger.info(f"File moved to 'processed' folder: s3://{bucket}/{get_filename_from_path(key)}") + +def move_to_unprocessed(bucket, key,invalid_file_location): + s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=f'{invalid_file_location}{get_filename_from_path(key)}') + s3_client.delete_object(Bucket=bucket, Key=key) + logger.info(f"File moved to 'unprocessed' folder: s3://{bucket}/{invalid_file_location}{get_filename_from_path(key)}") + +def get_filename_from_path(full_path): + return os.path.basename(full_path)