6c028ef892
DRAFT: Feature/adhoc ops scripts * added adhoc ops scripts for pre-doczy work * Added first draft of cost automation script * Fixed dir name * Added Scheduler lambda * Added TX adhoc script for rerun + Updated cost automation script * Added some misc scripts * Merged main into feature/adhoc-ops-scripts * Aryan Ad Hoc Scripts Pushed * De Duplication Script added * Merged main into feature/adhoc-ops-scripts Approved-by: Umang Shailesh Mistry
62 lines
2.3 KiB
Python
62 lines
2.3 KiB
Python
import os
|
|
import pandas as pd
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from collections import defaultdict
|
|
import csv
|
|
|
|
def build_file_cache(base_directory):
|
|
file_cache = defaultdict(list)
|
|
|
|
for root, dirs, files in os.walk(base_directory):
|
|
for file in files:
|
|
file_cache[file].append(os.path.join(root, file))
|
|
|
|
return file_cache
|
|
|
|
def find_files(file_cache, filenames):
|
|
found_files = {}
|
|
for filename in filenames:
|
|
if filename in file_cache:
|
|
found_files[filename] = file_cache[filename][0]
|
|
else:
|
|
found_files[filename] = "Not Found"
|
|
return found_files
|
|
|
|
def parallel_search(base_directory, filenames, max_workers=50):
|
|
print("Building file cache...")
|
|
file_cache = build_file_cache(base_directory)
|
|
with open('all_file_paths.csv', mode='w', newline='', encoding='utf-8') as csv_file:
|
|
writer = csv.writer(csv_file)
|
|
for key, values in file_cache.items():
|
|
row = [key] + values if isinstance(values, list) else [key, values]
|
|
writer.writerow(row)
|
|
print(f"Dictionary has been successfully written to 'all_file_paths.csv'.")
|
|
print(f"File cache built with {len(file_cache)} unique files.")
|
|
found_files = {}
|
|
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
futures = {executor.submit(find_files, file_cache, chunk): chunk
|
|
for chunk in chunked(filenames, len(filenames) // max_workers)}
|
|
|
|
for future in as_completed(futures):
|
|
found_files.update(future.result())
|
|
|
|
return found_files
|
|
|
|
def chunked(iterable, n):
|
|
for i in range(0, len(iterable), n):
|
|
yield iterable[i:i + n]
|
|
|
|
def search_files_from_csv(csv_file, base_directory, output_csv):
|
|
df = pd.read_csv(csv_file)
|
|
df['filename'] += ".Pdf"
|
|
filenames = df['filename'].tolist()
|
|
print(f"Searching for {len(filenames)} files in '{base_directory}'...")
|
|
file_paths = parallel_search(base_directory, filenames)
|
|
df['file_path'] = df['filename'].apply(lambda x: file_paths.get(x, "Not Found"))
|
|
df.to_csv(output_csv, index=False)
|
|
print(f"Results saved to '{output_csv}'.")
|
|
|
|
input_csv = "missing_files_renaming.csv"
|
|
base_dir = "T:/AArete Client Work/Doczy-Production/Restricted/2024-06-28-pdf/"
|
|
output_csv = "missing_file_paths_2.csv"
|
|
search_files_from_csv(input_csv, base_dir, output_csv) |