2024-12-11 15:46:49 +00:00
import boto3
import pandas as pd
from concurrent . futures import ThreadPoolExecutor , as_completed
2024-12-17 11:18:41 +00:00
"""
This script is used to search for files in a specific S3 bucket with a specific prefix and copy them to another S3 bucket with a different prefix.
This is similar to the script s3_search_and_copy.py, but this script is designed to be used with a CSV file that contains a list of file names to search for in the prefix to limit the scope.
"""
2024-12-11 15:46:49 +00:00
s3_client = boto3 . Session ( profile_name = ' temp_cred ' ) . client ( ' s3 ' )
def get_filenames_from_s3 ( bucket , prefix ) :
paginator = s3_client . get_paginator ( ' list_objects_v2 ' )
page_iterator = paginator . paginate ( Bucket = bucket , Prefix = prefix )
filenames = set ( )
for page in page_iterator :
if ' Contents ' in page :
for obj in page [ ' Contents ' ] :
filenames . add ( ( obj [ ' Key ' ] . split ( ' / ' ) [ - 1 ] ) [ : - 4 ] )
return filenames
def copy_file_if_exists ( file_name , source_bucket , source_prefix , destination_bucket , destination_prefix , existing_files ) :
if file_name in existing_files :
source_key = f " { source_prefix } / { file_name } " . strip ( ' / ' ) + " .pdf "
destination_key = f " { destination_prefix } / { file_name } " . strip ( ' / ' ) + " .pdf "
# s3_client.copy_object(
# CopySource={'Bucket': source_bucket, 'Key': source_key},
# Bucket=destination_bucket,
# Key=destination_key
# )
response = s3_client . get_object ( Bucket = source_bucket , Key = source_key )
file_content = response [ ' Body ' ] . read ( )
# print(f"Downloaded {source_key} from s3://{source_bucket}")
s3_client . put_object ( Bucket = destination_bucket , Key = destination_key , Body = file_content )
print ( f " Copied: { file_name } to { destination_prefix } " )
else :
print ( f " File not found: { file_name } in { source_prefix } " )
rerun = pd . concat ( [ rerun , df [ df [ ' File name Without Extension ' ] == file_name ] ] )
def copy_files_multithreaded ( file_list , source_bucket , source_prefix , destination_bucket , destination_prefix , existing_files , max_workers = 20 ) :
with ThreadPoolExecutor ( max_workers = max_workers ) as executor :
futures = [
executor . submit ( copy_file_if_exists , file_name , source_bucket , source_prefix , destination_bucket , destination_prefix , existing_files )
for file_name in file_list
]
for future in as_completed ( futures ) :
try :
future . result ( )
except Exception as e :
print ( f " Error copying file: { e } " )
csv_path = ' new_duplicates_in_batch3.csv '
file_name_column = ' File Name Without Extension '
source_bucket = ' centene-national-contracting-files '
search_prefix = ' batch_3_priority_files/txt_files '
source_prefix = ' batch_3_priority_files/pdf_files '
destination_bucket = ' centene-national-contracting-files '
destination_prefix = ' batch_3_priority_files/re_run_files '
df = pd . read_csv ( csv_path )
rerun = pd . DataFrame ( columns = df . columns )
file_names = df [ file_name_column ] . dropna ( ) . tolist ( )
print ( f " Loaded { len ( file_names ) } file names from CSV. " )
existing_files = get_filenames_from_s3 ( source_bucket , search_prefix )
print ( f " Found { len ( existing_files ) } files in S3 source folder ' { source_prefix } ' . " )
copy_files_multithreaded ( file_names , source_bucket , source_prefix , destination_bucket , destination_prefix , existing_files )
rerun . reset_index ( drop = True , inplace = True )
rerun . to_csv ( ' batch3_rerun.csv ' )