Files
doczyai-pipelines/ops_scripts/CNC/local_search_files_from_csv.py
T
Umang Shailesh Mistry c210052952 Merged in feature/ops_scripts (pull request #333)
Feature/ops scripts

* Added comments to Aryan's script and added some more scripts

* Search and Copy Script uploaded as a Python Notebook - with comments and markdown

* Merged main into feature/ops_scripts

* Merged main into feature/ops_scripts


Approved-by: Michael McGuinness
Approved-by: Chris Stobie
2024-12-17 11:18:41 +00:00

73 lines
2.8 KiB
Python

import os
import pandas as pd
from concurrent.futures import ThreadPoolExecutor, as_completed
from collections import defaultdict
import csv
"""
This script searches for files in a given directory based on a list of filenames in a CSV file.
The CSV file should have a column named 'filename' containing the filenames to search for.
This is maily used to find files that are not present in the s3 or cannot be located.
In that case, we use the list of missing files (in the CSV) to search for them in the T drive.
This script may not be needed as all CNC batches have been staged for execution.
"""
def build_file_cache(base_directory):
file_cache = defaultdict(list)
for root, dirs, files in os.walk(base_directory):
for file in files:
file_cache[file].append(os.path.join(root, file))
return file_cache
def find_files(file_cache, filenames):
found_files = {}
for filename in filenames:
if filename in file_cache:
found_files[filename] = file_cache[filename][0]
else:
found_files[filename] = "Not Found"
return found_files
def parallel_search(base_directory, filenames, max_workers=50):
print("Building file cache...")
file_cache = build_file_cache(base_directory)
with open('all_file_paths.csv', mode='w', newline='', encoding='utf-8') as csv_file:
writer = csv.writer(csv_file)
for key, values in file_cache.items():
row = [key] + values if isinstance(values, list) else [key, values]
writer.writerow(row)
print(f"Dictionary has been successfully written to 'all_file_paths.csv'.")
print(f"File cache built with {len(file_cache)} unique files.")
found_files = {}
with ThreadPoolExecutor(max_workers=max_workers) as executor:
futures = {executor.submit(find_files, file_cache, chunk): chunk
for chunk in chunked(filenames, len(filenames) // max_workers)}
for future in as_completed(futures):
found_files.update(future.result())
return found_files
def chunked(iterable, n):
for i in range(0, len(iterable), n):
yield iterable[i:i + n]
def search_files_from_csv(csv_file, base_directory, output_csv):
df = pd.read_csv(csv_file)
df['filename'] += ".Pdf"
filenames = df['filename'].tolist()
print(f"Searching for {len(filenames)} files in '{base_directory}'...")
file_paths = parallel_search(base_directory, filenames)
df['file_path'] = df['filename'].apply(lambda x: file_paths.get(x, "Not Found"))
df.to_csv(output_csv, index=False)
print(f"Results saved to '{output_csv}'.")
input_csv = "missing_files_renaming.csv"
base_dir = "T:/AArete Client Work/Doczy-Production/Restricted/2024-06-28-pdf/"
output_csv = "missing_file_paths_2.csv"
search_files_from_csv(input_csv, base_dir, output_csv)