2da639e59b
Feature/DAIP2-2314 DAIP2 1687 hybrid * remove -files from s3 prefix requirements * Resolve input paths * fix: VendorProcessor.process_file returns (df, None) tuple runner.safe_process_file unpacks the result as (cc_df, dashboard_df), so returning a single DataFrame caused every vendor/generic file to fail with "too many values to unpack (expected 2)" — Python iterates DataFrame columns during unpacking. Vendor pipelines have no dashboard variant; second slot is None and the existing `dashboard_result is not None` guard in runner.py already handles it. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com> * DAIP2-2314 + DAIP2-1687: pad DYNAMIC_PRIMARY + DYNAMIC_PRIMARY_ENTITY_CLASSIFICATION over 1024-token cache floor - Pad DYNAMIC_PRIMARY_INSTRUCTION with three new sections: [SCOPE BOUNDARIES], [SOURCE TEXT INTERPRETATION], [REASONING DISCIPLINE], plus a [WORKED EXAMPLES] block. Estimated tokens: 447 -> 1117 (Sonnet 4.5 1024-min, +93 margin). All additions reinforce existing rules (sibling-field separation, alias mapping, pricing-vs-LOB distinction, contrastive-clause exclusion, exhibit-header binding) — no new directives that could bias extraction. - Pad DYNAMIC_PRIMARY_ENTITY_CLASSIFICATION_INSTRUCTION with a [FINAL CHECKLIST BEFORE OUTPUT] block. Estimated tokens: 956 -> 1101 (Sonnet 4.5 1024-min, +77 margin). Reinforces the existing 4-step anti-duplication protocol and JSON shape requirements. - Register both new entries in cache_registry: DYNAMIC_PRIMARY_ENTITY_CLASSIFICATION as INSTRUCTION_PLUS_CONTEXT (caches at warm-up), DYNAMIC_PRIMARY_ENTITIES as CONTEXT (instruction is intentionally short; CONTEXT c… * black format fix * Merged dev into feature/DAIP2-2314-DAIP2-1687-hybrid * fixed raw lob values in base lob field mapping and composite entities fix * black format fix * fixed LOB Program output issues * issue fixes * remove debugging code * Updated prompts * updated additional instructions * Update Program-->LOB * LLM-based AD-Program/Product mapping to LOB even when there is a crosswalk * black format fix * Merged dev into feature/DAIP2-2314-DAIP2-1687-hybrid * added logging in prompt call tracking * added updated logging in prompt call tracking * aaded min cache token per usage label * added cache registry for dynamic primary mapping prompt calls * reolved mapping prompts ambiguities * black format fix * Phase 2 modifications added * reverted phase 2 modifications Approved-by: Katon Minhas
124 lines
4.0 KiB
Python
Executable File
124 lines
4.0 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
Script to merge multiple text files into a single file.
|
|
|
|
Usage:
|
|
python merge_files.py <output_file> <input_file1> <input_file2> ...
|
|
python merge_files.py <output_file> --dir <input_directory> [--pattern <file_pattern>]
|
|
|
|
This script concatenates the contents of the input files into the output file,
|
|
separating each file's content with a newline.
|
|
For CSV files, it properly merges by keeping only the header from the first file.
|
|
"""
|
|
|
|
import sys
|
|
import os
|
|
import glob
|
|
import csv
|
|
|
|
def merge_files(file_list, output_file):
|
|
"""
|
|
Merge a list of files into a single output file.
|
|
|
|
Args:
|
|
file_list (list): List of file paths to merge.
|
|
output_file (str): Path to the output file.
|
|
"""
|
|
if not file_list:
|
|
return
|
|
|
|
# Check if files are CSV based on extension
|
|
is_csv = any(fname.lower().endswith('.csv') for fname in file_list)
|
|
|
|
if is_csv:
|
|
merge_csv_files(file_list, output_file)
|
|
else:
|
|
merge_text_files(file_list, output_file)
|
|
|
|
def merge_text_files(file_list, output_file):
|
|
"""
|
|
Merge text files by concatenating their contents.
|
|
"""
|
|
with open(output_file, 'w', encoding='utf-8') as outfile:
|
|
for fname in file_list:
|
|
if not os.path.isfile(fname):
|
|
print(f"Warning: {fname} is not a file or does not exist. Skipping.")
|
|
continue
|
|
try:
|
|
with open(fname, 'r', encoding='utf-8') as infile:
|
|
content = infile.read()
|
|
outfile.write(content)
|
|
outfile.write('\n') # Add a newline separator between files
|
|
except Exception as e:
|
|
print(f"Error reading {fname}: {e}")
|
|
|
|
def merge_csv_files(file_list, output_file):
|
|
"""
|
|
Merge CSV files by keeping header from first file and appending data from others.
|
|
"""
|
|
first_file = True
|
|
|
|
with open(output_file, 'w', newline='', encoding='utf-8') as outfile:
|
|
writer = None
|
|
|
|
for fname in file_list:
|
|
if not os.path.isfile(fname):
|
|
print(f"Warning: {fname} is not a file or does not exist. Skipping.")
|
|
continue
|
|
|
|
try:
|
|
with open(fname, 'r', encoding='utf-8') as infile:
|
|
reader = csv.reader(infile)
|
|
|
|
for row_num, row in enumerate(reader):
|
|
if first_file or row_num > 0: # Skip header for subsequent files
|
|
if writer is None:
|
|
writer = csv.writer(outfile)
|
|
writer.writerow(row)
|
|
|
|
first_file = False
|
|
|
|
except Exception as e:
|
|
print(f"Error reading {fname}: {e}")
|
|
|
|
def get_files_from_dir(directory, pattern='*'):
|
|
"""
|
|
Get all files from a directory matching a pattern.
|
|
|
|
Args:
|
|
directory (str): Directory path.
|
|
pattern (str): Glob pattern for files.
|
|
|
|
Returns:
|
|
list: List of file paths.
|
|
"""
|
|
if not os.path.isdir(directory):
|
|
print(f"Error: {directory} is not a directory.")
|
|
return []
|
|
return glob.glob(os.path.join(directory, pattern))
|
|
|
|
if __name__ == "__main__":
|
|
if len(sys.argv) < 3:
|
|
print("Usage: python merge_files.py <output_file> <input_file1> <input_file2> ...")
|
|
print(" or: python merge_files.py <output_file> --dir <input_directory> [--pattern <file_pattern>]")
|
|
sys.exit(1)
|
|
|
|
output_file = sys.argv[1]
|
|
|
|
if sys.argv[2] == '--dir':
|
|
if len(sys.argv) < 4:
|
|
print("Usage: python merge_files.py <output_file> --dir <input_directory> [--pattern <file_pattern>]")
|
|
sys.exit(1)
|
|
directory = sys.argv[3]
|
|
pattern = sys.argv[5] if len(sys.argv) > 5 and sys.argv[4] == '--pattern' else '*'
|
|
input_files = get_files_from_dir(directory, pattern)
|
|
else:
|
|
input_files = sys.argv[2:]
|
|
|
|
if not input_files:
|
|
print("No input files found.")
|
|
sys.exit(1)
|
|
|
|
merge_files(input_files, output_file)
|
|
print(f"Merged {len(input_files)} files into {output_file}")
|