144 lines
6.2 KiB
Python
144 lines
6.2 KiB
Python
|
|
import boto3
|
||
|
|
import logging
|
||
|
|
import botocore
|
||
|
|
from PyPDF2 import PdfReader
|
||
|
|
from io import BytesIO
|
||
|
|
import os
|
||
|
|
from configparser import ConfigParser
|
||
|
|
import urllib.parse
|
||
|
|
from urllib.parse import urlparse
|
||
|
|
|
||
|
|
# Configure logging
|
||
|
|
logger = logging.getLogger()
|
||
|
|
logger.setLevel(logging.INFO)
|
||
|
|
|
||
|
|
# Initialize S3 & Textract clients
|
||
|
|
s3_client = boto3.client('s3')
|
||
|
|
|
||
|
|
# Function to load configuration from S3
|
||
|
|
def load_config_from_s3(bucket_name, file_key):
|
||
|
|
|
||
|
|
# Download the config file from S3
|
||
|
|
response = s3_client.get_object(Bucket=bucket_name, Key=file_key)
|
||
|
|
config_content = response['Body'].read().decode('utf-8')
|
||
|
|
|
||
|
|
# Parse the config file
|
||
|
|
config_parser = ConfigParser()
|
||
|
|
config_parser.read_string(config_content)
|
||
|
|
|
||
|
|
# Convert the configuration to a dictionary
|
||
|
|
config_dict = {}
|
||
|
|
for section in config_parser.sections():
|
||
|
|
config_dict[section] = {key.upper(): value for key, value in config_parser.items(section)}
|
||
|
|
|
||
|
|
return config_dict
|
||
|
|
|
||
|
|
def lambda_handler(event, context):
|
||
|
|
|
||
|
|
for record in event['Records']:
|
||
|
|
|
||
|
|
# Retrieve the S3 bucket and key from the event
|
||
|
|
bucket = event['Records'][0]['s3']['bucket']['name']
|
||
|
|
key = urllib.parse.unquote_plus(event['Records'][0]['s3']['object']['key'])
|
||
|
|
region = record['awsRegion']
|
||
|
|
|
||
|
|
logger.info(f"Processing S3 file - Bucket: {bucket}, Key: {key}, Region: {region}")
|
||
|
|
|
||
|
|
# Read environment variables
|
||
|
|
property_file_path = os.environ.get('PROPERTY_FILE_S3_PATH', '')
|
||
|
|
batch_id = os.environ.get('BATCH_ID', '')
|
||
|
|
logger.info(f"Batch ID: {batch_id}")
|
||
|
|
file_path_array = property_file_path.split("/")
|
||
|
|
# Extract BUCKET_NAME and config_file_path
|
||
|
|
S3_BUCKET_NAME = file_path_array[0]
|
||
|
|
CONFIG_FILE_PATH = "/".join(file_path_array[1:])
|
||
|
|
logger.info(f"Using S3 bucket: {S3_BUCKET_NAME} and config file path: {CONFIG_FILE_PATH}")
|
||
|
|
|
||
|
|
# Load config file
|
||
|
|
config_dict = load_config_from_s3(bucket, CONFIG_FILE_PATH)
|
||
|
|
INVALID_PDF_FILE_LOCATION = config_dict['FOLDER_LOCATIONS']['INVALID_PDF_FILE_LOCATION'].format(batch_id)
|
||
|
|
SOURCE_LOCATION = config_dict['FOLDER_LOCATIONS']['SOURCE_LOCATION'].format(batch_id)
|
||
|
|
logger.info("Config file loaded successfully.")
|
||
|
|
|
||
|
|
logger.info(f"Processing file: s3://{bucket}/{key}")
|
||
|
|
|
||
|
|
|
||
|
|
# Check if the file has a '.filepart' extension
|
||
|
|
if key.lower().endswith('.filepart'):
|
||
|
|
new_key = key[:-9] + '.pdf' # Rename to have a '.pdf' extension
|
||
|
|
s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=new_key)
|
||
|
|
s3_client.delete_object(Bucket=bucket, Key=key)
|
||
|
|
key = new_key # Update key to the new filename
|
||
|
|
|
||
|
|
|
||
|
|
|
||
|
|
# Validate number of pages
|
||
|
|
try:
|
||
|
|
pdf_reader = PdfReader(BytesIO(s3_client.get_object(Bucket=bucket, Key=key)['Body'].read()))
|
||
|
|
if is_password_protected(pdf_reader):
|
||
|
|
logger.info("File is password protected. Moving to 'unprocessed' folder.")
|
||
|
|
move_to_unprocessed(bucket, key,INVALID_PDF_FILE_LOCATION)
|
||
|
|
# Validate file size
|
||
|
|
file_size = s3_client.head_object(Bucket=bucket, Key=key)['ContentLength']
|
||
|
|
logger.info(f"File size: {file_size} bytes")
|
||
|
|
if file_size > 500 * 1024 * 1024: # 500MB
|
||
|
|
logger.info("File size exceeds 500MB. Moving to 'unprocessed' folder.")
|
||
|
|
move_to_unprocessed(bucket, key,INVALID_PDF_FILE_LOCATION)
|
||
|
|
return
|
||
|
|
|
||
|
|
num_pages = len(pdf_reader.pages)
|
||
|
|
logger.info(f"Number of pages: {num_pages}")
|
||
|
|
if num_pages > 3000:
|
||
|
|
logger.info("Number of pages exceeds 3000. Moving to 'unprocessed' folder.")
|
||
|
|
move_to_unprocessed(bucket,key,INVALID_PDF_FILE_LOCATION)
|
||
|
|
return
|
||
|
|
except botocore.exceptions.ClientError as e:
|
||
|
|
if e.response['Error']['Code'] == '404':
|
||
|
|
logger.error(f"File not found: s3://{bucket}/{key}")
|
||
|
|
# Handle the case where the file doesn't exist
|
||
|
|
return
|
||
|
|
else:
|
||
|
|
logger.error(f"Error checking file size: {str(e)}. Moving to 'unprocessed' folder.")
|
||
|
|
move_to_unprocessed(bucket, key, INVALID_PDF_FILE_LOCATION)
|
||
|
|
return
|
||
|
|
except Exception as e:
|
||
|
|
logger.error(f"Error checking file size: {str(e)}. Moving to 'unprocessed' folder.")
|
||
|
|
move_to_unprocessed(bucket, key, INVALID_PDF_FILE_LOCATION)
|
||
|
|
return
|
||
|
|
|
||
|
|
# Validate password protection and resolution
|
||
|
|
|
||
|
|
if not is_resolution_valid(pdf_reader):
|
||
|
|
logger.info("File has invalid resolution. Moving to 'unprocessed' folder.")
|
||
|
|
move_to_unprocessed(bucket, key,INVALID_PDF_FILE_LOCATION)
|
||
|
|
else:
|
||
|
|
logger.info("File passed all conditions. Moving to 'processed' folder.")
|
||
|
|
move_to_processed(bucket, key,SOURCE_LOCATION)
|
||
|
|
|
||
|
|
def is_password_protected(pdf_reader):
|
||
|
|
return pdf_reader.is_encrypted
|
||
|
|
|
||
|
|
def is_resolution_valid(pdf_reader, max_resolution=3000*4000):
|
||
|
|
for page_num in range(len(pdf_reader.pages)):
|
||
|
|
page = pdf_reader.pages[page_num]
|
||
|
|
page_width, page_height = page.mediabox.upper_right
|
||
|
|
if page_width > max_resolution or page_height > max_resolution:
|
||
|
|
return False
|
||
|
|
return True
|
||
|
|
|
||
|
|
|
||
|
|
def move_to_processed(bucket, key,valid_file_location):
|
||
|
|
#replace processed with valid_file_location like in unprocessed method
|
||
|
|
#s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=f'processed/{key}')
|
||
|
|
s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=f'{valid_file_location}/{get_filename_from_path(key)}')
|
||
|
|
s3_client.delete_object(Bucket=bucket, Key=key)
|
||
|
|
logger.info(f"File moved to 'processed' folder: s3://{bucket}/{get_filename_from_path(key)}")
|
||
|
|
|
||
|
|
def move_to_unprocessed(bucket, key,invalid_file_location):
|
||
|
|
s3_client.copy_object(Bucket=bucket, CopySource={'Bucket': bucket, 'Key': key}, Key=f'{invalid_file_location}{get_filename_from_path(key)}')
|
||
|
|
s3_client.delete_object(Bucket=bucket, Key=key)
|
||
|
|
logger.info(f"File moved to 'unprocessed' folder: s3://{bucket}/{invalid_file_location}{get_filename_from_path(key)}")
|
||
|
|
|
||
|
|
def get_filename_from_path(full_path):
|
||
|
|
return os.path.basename(full_path)
|