2024-12-11 15:46:49 +00:00
|
|
|
import os
|
|
|
|
|
import pandas as pd
|
|
|
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
|
|
|
from collections import defaultdict
|
|
|
|
|
import csv
|
|
|
|
|
|
2024-12-17 11:18:41 +00:00
|
|
|
"""
|
|
|
|
|
This script searches for files in a given directory based on a list of filenames in a CSV file.
|
|
|
|
|
The CSV file should have a column named 'filename' containing the filenames to search for.
|
|
|
|
|
|
|
|
|
|
This is maily used to find files that are not present in the s3 or cannot be located.
|
|
|
|
|
In that case, we use the list of missing files (in the CSV) to search for them in the T drive.
|
|
|
|
|
|
|
|
|
|
This script may not be needed as all CNC batches have been staged for execution.
|
|
|
|
|
|
|
|
|
|
"""
|
|
|
|
|
|
2024-12-11 15:46:49 +00:00
|
|
|
def build_file_cache(base_directory):
|
|
|
|
|
file_cache = defaultdict(list)
|
|
|
|
|
|
|
|
|
|
for root, dirs, files in os.walk(base_directory):
|
|
|
|
|
for file in files:
|
|
|
|
|
file_cache[file].append(os.path.join(root, file))
|
|
|
|
|
|
|
|
|
|
return file_cache
|
|
|
|
|
|
|
|
|
|
def find_files(file_cache, filenames):
|
|
|
|
|
found_files = {}
|
|
|
|
|
for filename in filenames:
|
|
|
|
|
if filename in file_cache:
|
|
|
|
|
found_files[filename] = file_cache[filename][0]
|
|
|
|
|
else:
|
|
|
|
|
found_files[filename] = "Not Found"
|
|
|
|
|
return found_files
|
|
|
|
|
|
|
|
|
|
def parallel_search(base_directory, filenames, max_workers=50):
|
|
|
|
|
print("Building file cache...")
|
|
|
|
|
file_cache = build_file_cache(base_directory)
|
|
|
|
|
with open('all_file_paths.csv', mode='w', newline='', encoding='utf-8') as csv_file:
|
|
|
|
|
writer = csv.writer(csv_file)
|
|
|
|
|
for key, values in file_cache.items():
|
|
|
|
|
row = [key] + values if isinstance(values, list) else [key, values]
|
|
|
|
|
writer.writerow(row)
|
|
|
|
|
print(f"Dictionary has been successfully written to 'all_file_paths.csv'.")
|
|
|
|
|
print(f"File cache built with {len(file_cache)} unique files.")
|
|
|
|
|
found_files = {}
|
|
|
|
|
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
|
|
|
futures = {executor.submit(find_files, file_cache, chunk): chunk
|
|
|
|
|
for chunk in chunked(filenames, len(filenames) // max_workers)}
|
|
|
|
|
|
|
|
|
|
for future in as_completed(futures):
|
|
|
|
|
found_files.update(future.result())
|
|
|
|
|
|
|
|
|
|
return found_files
|
|
|
|
|
|
|
|
|
|
def chunked(iterable, n):
|
|
|
|
|
for i in range(0, len(iterable), n):
|
|
|
|
|
yield iterable[i:i + n]
|
|
|
|
|
|
|
|
|
|
def search_files_from_csv(csv_file, base_directory, output_csv):
|
|
|
|
|
df = pd.read_csv(csv_file)
|
|
|
|
|
df['filename'] += ".Pdf"
|
|
|
|
|
filenames = df['filename'].tolist()
|
|
|
|
|
print(f"Searching for {len(filenames)} files in '{base_directory}'...")
|
|
|
|
|
file_paths = parallel_search(base_directory, filenames)
|
|
|
|
|
df['file_path'] = df['filename'].apply(lambda x: file_paths.get(x, "Not Found"))
|
|
|
|
|
df.to_csv(output_csv, index=False)
|
|
|
|
|
print(f"Results saved to '{output_csv}'.")
|
|
|
|
|
|
|
|
|
|
input_csv = "missing_files_renaming.csv"
|
|
|
|
|
base_dir = "T:/AArete Client Work/Doczy-Production/Restricted/2024-06-28-pdf/"
|
|
|
|
|
output_csv = "missing_file_paths_2.csv"
|
|
|
|
|
search_files_from_csv(input_csv, base_dir, output_csv)
|