Files
doczyai-pipelines/archive/ops_scripts/generic/complex_tables_files_movement.py
T

65 lines
2.7 KiB
Python
Raw Normal View History

import boto3
import pandas as pd
from concurrent.futures import ThreadPoolExecutor, as_completed
"""
This script is used to copy complex contract text files to a different folder in the same S3 bucket.
The script reads a CSV file that contains the filenames and their table counts. It filters the rows where the table count is greater than or equal to 1.
This script is NOT CLIENT SPECIFIC. It is a generic script that can be used for any client.
"""
# Function to copy a single file
def copy_file(bucket_name, source_prefix, dest_prefix, file_name, profile_name):
try:
# Initialize S3 client
session = boto3.Session(profile_name=profile_name)
s3 = session.client('s3')
copy_source = {'Bucket': bucket_name, 'Key': f'{source_prefix}/{file_name}'}
dest_key = f'{dest_prefix}/{file_name}'
s3.copy_object(CopySource=copy_source, Bucket=bucket_name, Key=dest_key)
print(f'Successfully copied from {copy_source} to {dest_key}')
except Exception as e:
# print(f'Failed to copy {file_name}: {e}')
print('File not in bucket. Skipping')
pass
# Function to process copying in parallel
def copy_files_in_parallel(bucket_name, source_prefix, dest_prefix, file_list,profile_name ,max_workers=10):
with ThreadPoolExecutor(max_workers=max_workers) as executor:
futures = [
executor.submit(copy_file, bucket_name, source_prefix, dest_prefix, file_name, profile_name=profile_name)
for file_name in file_list
]
for future in as_completed(futures):
future.result() # This will raise any exceptions that occurred during execution
# Main function
def main():
# Client bucket where the text files are stored
bucket_name = 'centene-national-contracting-files'
# This is used to copy complex contract text files to a different folder
source_prefix = 'batch6_16_file/txt_files'
dest_prefix_3a = 'batch6_16_file/complex_contract_files/txt_files'
max_workers = 50
profile_name = 'default' # Set this as per your AWS profile
# Read the table analysis file (from DS code) and read the filenames and their table counts
csv_file = 'C:\\Doczy\\National contracting\\Somefolder\\CNC-6to16-Table-Analysis.csv' # Replace with your CSV file path
file_df = pd.read_csv(csv_file)
# Filter rows where 'tables' value is greater than or equal to 1
complex_table_files = file_df[file_df['Table Count'] >= 1]
# Assuming the CSV has a column 'file_name' with the list of file names
file_list = complex_table_files['Filename_pdf'].tolist()
copy_files_in_parallel(bucket_name, source_prefix, dest_prefix_3a, file_list, profile_name = profile_name, max_workers=max_workers)
if __name__ == '__main__':
main()