import boto3 import pandas as pd from concurrent.futures import ThreadPoolExecutor, as_completed """ This script is used to copy complex contract text files to a different folder in the same S3 bucket. The script reads a CSV file that contains the filenames and their table counts. It filters the rows where the table count is greater than or equal to 1. This script is NOT CLIENT SPECIFIC. It is a generic script that can be used for any client. """ # Function to copy a single file def copy_file(bucket_name, source_prefix, dest_prefix, file_name, profile_name): try: # Initialize S3 client session = boto3.Session(profile_name=profile_name) s3 = session.client('s3') copy_source = {'Bucket': bucket_name, 'Key': f'{source_prefix}/{file_name}'} dest_key = f'{dest_prefix}/{file_name}' s3.copy_object(CopySource=copy_source, Bucket=bucket_name, Key=dest_key) print(f'Successfully copied from {copy_source} to {dest_key}') except Exception as e: # print(f'Failed to copy {file_name}: {e}') print('File not in bucket. Skipping') pass # Function to process copying in parallel def copy_files_in_parallel(bucket_name, source_prefix, dest_prefix, file_list,profile_name ,max_workers=10): with ThreadPoolExecutor(max_workers=max_workers) as executor: futures = [ executor.submit(copy_file, bucket_name, source_prefix, dest_prefix, file_name, profile_name=profile_name) for file_name in file_list ] for future in as_completed(futures): future.result() # This will raise any exceptions that occurred during execution # Main function def main(): # Client bucket where the text files are stored bucket_name = 'centene-national-contracting-files' # This is used to copy complex contract text files to a different folder source_prefix = 'batch6_16_file/txt_files' dest_prefix_3a = 'batch6_16_file/complex_contract_files/txt_files' max_workers = 50 profile_name = 'default' # Set this as per your AWS profile # Read the table analysis file (from DS code) and read the filenames and their table counts csv_file = 'C:\\Doczy\\National contracting\\Somefolder\\CNC-6to16-Table-Analysis.csv' # Replace with your CSV file path file_df = pd.read_csv(csv_file) # Filter rows where 'tables' value is greater than or equal to 1 complex_table_files = file_df[file_df['Table Count'] >= 1] # Assuming the CSV has a column 'file_name' with the list of file names file_list = complex_table_files['Filename_pdf'].tolist() copy_files_in_parallel(bucket_name, source_prefix, dest_prefix_3a, file_list, profile_name = profile_name, max_workers=max_workers) if __name__ == '__main__': main()