Merged in bugfix/qc_qa_filetype (pull request #810)
Bugfix/qc qa filetype * WIP: support multi-filetype * Add graceful error handling to QC/QA standalone CLI Wrap validation pipeline in try/except to ensure no partial results are created/uploaded if the pipeline fails mid-execution. * Add comprehensive QC/QA unit tests, improve coverage to 95% - Add CLI integration tests for __main__.py (6 tests) - Add pipeline validation tests for all validation steps (20+ tests) - Test multi-filetype I/O (CSV, XLSX input/output) - Test graceful error handling - Coverage: 43% → 95% * Merge remote-tracking branch 'origin/main' into bugfix/qc_qa_filetype Approved-by: Katon Minhas
This commit is contained in:
committed by
Katon Minhas
parent
9b89b0a10a
commit
c302c1023a
@@ -8,9 +8,9 @@ Usage:
|
|||||||
stats_file=/path/to/stats.csv
|
stats_file=/path/to/stats.csv
|
||||||
|
|
||||||
Arguments:
|
Arguments:
|
||||||
input_file - Path to input file (.xlsx, .xls, .csv)
|
input_file - Path to input file (.xlsx, .xls, .xlsm, .xlsb, .csv)
|
||||||
output_file - Path to output file (.xlsx)
|
output_file - Path to output file (.xlsx, .xls, .csv) - stats file will use same extension
|
||||||
stats_file - (Optional) Path to statistics CSV for outlier detection
|
stats_file - (Optional) Path to statistics file (.xlsx, .xls, .xlsm, .xlsb, .csv) for outlier detection
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import os
|
import os
|
||||||
@@ -24,8 +24,10 @@ from src.config import get_arg_value
|
|||||||
from src.qc_qa.pipeline import (
|
from src.qc_qa.pipeline import (
|
||||||
convert_excel_to_csv,
|
convert_excel_to_csv,
|
||||||
generate_statistics,
|
generate_statistics,
|
||||||
|
read_file_by_extension,
|
||||||
run_qc_qa_pipeline,
|
run_qc_qa_pipeline,
|
||||||
save_qc_qa_outputs,
|
save_qc_qa_outputs,
|
||||||
|
write_file_by_extension,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -63,44 +65,50 @@ def main():
|
|||||||
df = pd.read_csv(csv_file)
|
df = pd.read_csv(csv_file)
|
||||||
logging.info(f"Loaded DataFrame with {len(df)} rows and {len(df.columns)} columns")
|
logging.info(f"Loaded DataFrame with {len(df)} rows and {len(df.columns)} columns")
|
||||||
|
|
||||||
# Load stats if provided
|
# Load stats if provided (supports CSV and Excel)
|
||||||
stats_df = None
|
stats_df = None
|
||||||
if stats_file:
|
if stats_file:
|
||||||
if not os.path.exists(stats_file):
|
if not os.path.exists(stats_file):
|
||||||
logging.warning(f"Warning: Stats file not found: {stats_file}")
|
logging.warning(f"Warning: Stats file not found: {stats_file}")
|
||||||
else:
|
else:
|
||||||
stats_df = pd.read_csv(stats_file, encoding="utf-8-sig")
|
stats_df = read_file_by_extension(stats_file)
|
||||||
stats_df.columns = stats_df.columns.str.strip().str.upper()
|
stats_df.columns = stats_df.columns.str.strip().str.upper()
|
||||||
logging.info(f"Loaded statistics DataFrame")
|
logging.info(f"Loaded statistics DataFrame")
|
||||||
|
|
||||||
# Run validation
|
# Run validation pipeline with graceful error handling
|
||||||
validated_df = run_qc_qa_pipeline(df, stats_df)
|
try:
|
||||||
|
validated_df = run_qc_qa_pipeline(df, stats_df)
|
||||||
|
|
||||||
# Generate statistics
|
# Generate statistics
|
||||||
logging.info("Generating statistics...")
|
logging.info("Generating statistics...")
|
||||||
output_stats_df = generate_statistics(validated_df)
|
output_stats_df = generate_statistics(validated_df)
|
||||||
|
|
||||||
# Save to qa_qc_output folder (local) and S3 using centralized function
|
# Save to qa_qc_output folder (local) and S3 using centralized function
|
||||||
logging.info("Saving outputs to qa_qc_output and S3...")
|
logging.info("Saving outputs to qa_qc_output and S3...")
|
||||||
|
|
||||||
save_qc_qa_outputs(
|
save_qc_qa_outputs(
|
||||||
validated_df=validated_df,
|
validated_df=validated_df,
|
||||||
stats_df=output_stats_df,
|
stats_df=output_stats_df,
|
||||||
run_timestamp=run_timestamp,
|
run_timestamp=run_timestamp,
|
||||||
write_to_s3=config.WRITE_TO_S3,
|
write_to_s3=config.WRITE_TO_S3,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Legacy: Save to user-specified output file if provided
|
# Legacy: Save to user-specified output file if provided (supports CSV and Excel)
|
||||||
if output_file:
|
if output_file:
|
||||||
logging.info(f"\nLegacy output mode - saving to specified path...")
|
logging.info(f"\nLegacy output mode - saving to specified path...")
|
||||||
base, _ = os.path.splitext(output_file)
|
base, ext = os.path.splitext(output_file)
|
||||||
stats_output = f"{base}_stats.xlsx"
|
stats_output = f"{base}_stats{ext}"
|
||||||
|
|
||||||
logging.info(f"Saving validated data to: {output_file}")
|
logging.info(f"Saving validated data to: {output_file}")
|
||||||
validated_df.to_excel(output_file, index=False)
|
write_file_by_extension(validated_df, output_file)
|
||||||
|
|
||||||
logging.info(f"Saving statistics to: {stats_output}")
|
logging.info(f"Saving statistics to: {stats_output}")
|
||||||
output_stats_df.to_excel(stats_output, index=False)
|
write_file_by_extension(output_stats_df, stats_output)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(f"QC/QA validation failed: {e}")
|
||||||
|
logging.error("No results were created or uploaded due to the failure.")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
# Clean up temporary CSV if it was converted
|
# Clean up temporary CSV if it was converted
|
||||||
if csv_file != input_file and os.path.exists(csv_file):
|
if csv_file != input_file and os.path.exists(csv_file):
|
||||||
|
|||||||
@@ -29,6 +29,47 @@ warnings.filterwarnings("ignore", message="Setting an item of incompatible dtype
|
|||||||
warnings.filterwarnings("ignore", message="The behavior of DataFrame concatenation")
|
warnings.filterwarnings("ignore", message="The behavior of DataFrame concatenation")
|
||||||
|
|
||||||
|
|
||||||
|
def read_file_by_extension(file_path: str) -> pd.DataFrame:
|
||||||
|
"""
|
||||||
|
Read a file based on its extension (CSV or Excel).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_path: Path to input file (.csv, .xlsx, .xls, .xlsm, .xlsb)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DataFrame with file contents
|
||||||
|
"""
|
||||||
|
file_ext = os.path.splitext(file_path)[1].lower()
|
||||||
|
|
||||||
|
if file_ext == ".csv":
|
||||||
|
return pd.read_csv(file_path, encoding="utf-8-sig")
|
||||||
|
elif file_ext in [".xlsx", ".xls", ".xlsm", ".xlsb"]:
|
||||||
|
if file_ext == ".xlsb":
|
||||||
|
return pd.read_excel(file_path, engine="pyxlsb")
|
||||||
|
else:
|
||||||
|
return pd.read_excel(file_path)
|
||||||
|
else:
|
||||||
|
raise ValueError(f"Unsupported file extension: {file_ext}. Supported: .csv, .xlsx, .xls, .xlsm, .xlsb")
|
||||||
|
|
||||||
|
|
||||||
|
def write_file_by_extension(df: pd.DataFrame, file_path: str) -> None:
|
||||||
|
"""
|
||||||
|
Write a DataFrame to a file based on its extension (CSV or Excel).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
df: DataFrame to write
|
||||||
|
file_path: Path to output file (.csv, .xlsx, .xls)
|
||||||
|
"""
|
||||||
|
file_ext = os.path.splitext(file_path)[1].lower()
|
||||||
|
|
||||||
|
if file_ext == ".csv":
|
||||||
|
df.to_csv(file_path, index=False)
|
||||||
|
elif file_ext in [".xlsx", ".xls"]:
|
||||||
|
df.to_excel(file_path, index=False)
|
||||||
|
else:
|
||||||
|
raise ValueError(f"Unsupported file extension: {file_ext}. Supported: .csv, .xlsx, .xls")
|
||||||
|
|
||||||
|
|
||||||
def convert_excel_to_csv(input_file: str) -> str:
|
def convert_excel_to_csv(input_file: str) -> str:
|
||||||
"""
|
"""
|
||||||
Convert Excel file to CSV and return the CSV file path.
|
Convert Excel file to CSV and return the CSV file path.
|
||||||
|
|||||||
@@ -1052,7 +1052,7 @@ def format_date_series(series: pd.Series, col_name: str) -> Tuple[pd.Series, pd.
|
|||||||
# Outlier = parsed but absurd year (<1900 or > today)
|
# Outlier = parsed but absurd year (<1900 or > today)
|
||||||
today = datetime.now().date()
|
today = datetime.now().date()
|
||||||
outlier_mask = (~parsed_dates.isna()) & (
|
outlier_mask = (~parsed_dates.isna()) & (
|
||||||
parsed_dates.apply(lambda d: d.year if d else 0) < 1900 | (parsed_dates > today)
|
(parsed_dates.apply(lambda d: d.year if d else 0) < 1900) | (parsed_dates > today)
|
||||||
)
|
)
|
||||||
|
|
||||||
flags = pd.Series("", index=series.index, dtype="object")
|
flags = pd.Series("", index=series.index, dtype="object")
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
Comprehensive tests for QC/QA validation pipeline.
|
Comprehensive tests for QC/QA validation pipeline.
|
||||||
|
|
||||||
Tests cover:
|
Tests cover:
|
||||||
- pipeline.py: convert_excel_to_csv, get_stats, run_qc_qa_pipeline, save_qc_qa_outputs, generate_statistics
|
- pipeline.py: read_file_by_extension, write_file_by_extension, convert_excel_to_csv, get_stats, run_qc_qa_pipeline, save_qc_qa_outputs, generate_statistics
|
||||||
- qa_qc_utils.py: Helper functions for validation
|
- qa_qc_utils.py: Helper functions for validation
|
||||||
- __main__.py: CLI integration
|
- __main__.py: CLI integration
|
||||||
"""
|
"""
|
||||||
@@ -19,8 +19,10 @@ from src.qc_qa.pipeline import (
|
|||||||
convert_excel_to_csv,
|
convert_excel_to_csv,
|
||||||
generate_statistics,
|
generate_statistics,
|
||||||
get_stats,
|
get_stats,
|
||||||
|
read_file_by_extension,
|
||||||
run_qc_qa_pipeline,
|
run_qc_qa_pipeline,
|
||||||
save_qc_qa_outputs,
|
save_qc_qa_outputs,
|
||||||
|
write_file_by_extension,
|
||||||
)
|
)
|
||||||
from src.utils.qa_qc_utils import (
|
from src.utils.qa_qc_utils import (
|
||||||
append_flag,
|
append_flag,
|
||||||
@@ -140,6 +142,89 @@ class TestPipelineFunctions:
|
|||||||
with pytest.raises(Exception, match="Read error"):
|
with pytest.raises(Exception, match="Read error"):
|
||||||
convert_excel_to_csv(input_file)
|
convert_excel_to_csv(input_file)
|
||||||
|
|
||||||
|
# Tests for read_file_by_extension
|
||||||
|
def test_read_file_by_extension_csv(self, tmp_path):
|
||||||
|
"""Test reading CSV file."""
|
||||||
|
csv_file = tmp_path / "test.csv"
|
||||||
|
csv_file.write_text("col1,col2\n1,2\n3,4")
|
||||||
|
|
||||||
|
result = read_file_by_extension(str(csv_file))
|
||||||
|
|
||||||
|
assert isinstance(result, pd.DataFrame)
|
||||||
|
assert len(result) == 2
|
||||||
|
assert list(result.columns) == ["col1", "col2"]
|
||||||
|
|
||||||
|
def test_read_file_by_extension_xlsx(self, mocker, tmp_path):
|
||||||
|
"""Test reading Excel .xlsx file."""
|
||||||
|
input_file = str(tmp_path / "test.xlsx")
|
||||||
|
mock_df = pd.DataFrame({"col1": [1, 2], "col2": [3, 4]})
|
||||||
|
mocker.patch("pandas.read_excel", return_value=mock_df)
|
||||||
|
|
||||||
|
result = read_file_by_extension(input_file)
|
||||||
|
|
||||||
|
assert isinstance(result, pd.DataFrame)
|
||||||
|
assert len(result) == 2
|
||||||
|
pd.read_excel.assert_called_once_with(input_file)
|
||||||
|
|
||||||
|
def test_read_file_by_extension_xlsb(self, mocker, tmp_path):
|
||||||
|
"""Test reading Excel .xlsb file with pyxlsb engine."""
|
||||||
|
input_file = str(tmp_path / "test.xlsb")
|
||||||
|
mock_df = pd.DataFrame({"col1": [1, 2]})
|
||||||
|
mocker.patch("pandas.read_excel", return_value=mock_df)
|
||||||
|
|
||||||
|
result = read_file_by_extension(input_file)
|
||||||
|
|
||||||
|
assert isinstance(result, pd.DataFrame)
|
||||||
|
pd.read_excel.assert_called_once_with(input_file, engine="pyxlsb")
|
||||||
|
|
||||||
|
def test_read_file_by_extension_unsupported(self, tmp_path):
|
||||||
|
"""Test error for unsupported file extension."""
|
||||||
|
input_file = str(tmp_path / "test.txt")
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="Unsupported file extension"):
|
||||||
|
read_file_by_extension(input_file)
|
||||||
|
|
||||||
|
# Tests for write_file_by_extension
|
||||||
|
def test_write_file_by_extension_csv(self, tmp_path):
|
||||||
|
"""Test writing CSV file."""
|
||||||
|
output_file = str(tmp_path / "output.csv")
|
||||||
|
df = pd.DataFrame({"col1": [1, 2], "col2": [3, 4]})
|
||||||
|
|
||||||
|
write_file_by_extension(df, output_file)
|
||||||
|
|
||||||
|
# Verify file was created and contains expected data
|
||||||
|
result = pd.read_csv(output_file)
|
||||||
|
assert len(result) == 2
|
||||||
|
assert list(result.columns) == ["col1", "col2"]
|
||||||
|
|
||||||
|
def test_write_file_by_extension_xlsx(self, mocker, tmp_path):
|
||||||
|
"""Test writing Excel .xlsx file."""
|
||||||
|
output_file = str(tmp_path / "output.xlsx")
|
||||||
|
df = pd.DataFrame({"col1": [1, 2], "col2": [3, 4]})
|
||||||
|
mock_to_excel = mocker.patch.object(pd.DataFrame, "to_excel")
|
||||||
|
|
||||||
|
write_file_by_extension(df, output_file)
|
||||||
|
|
||||||
|
mock_to_excel.assert_called_once_with(output_file, index=False)
|
||||||
|
|
||||||
|
def test_write_file_by_extension_xls(self, mocker, tmp_path):
|
||||||
|
"""Test writing Excel .xls file."""
|
||||||
|
output_file = str(tmp_path / "output.xls")
|
||||||
|
df = pd.DataFrame({"col1": [1, 2]})
|
||||||
|
mock_to_excel = mocker.patch.object(pd.DataFrame, "to_excel")
|
||||||
|
|
||||||
|
write_file_by_extension(df, output_file)
|
||||||
|
|
||||||
|
mock_to_excel.assert_called_once_with(output_file, index=False)
|
||||||
|
|
||||||
|
def test_write_file_by_extension_unsupported(self, tmp_path):
|
||||||
|
"""Test error for unsupported output file extension."""
|
||||||
|
output_file = str(tmp_path / "output.txt")
|
||||||
|
df = pd.DataFrame({"col1": [1, 2]})
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="Unsupported file extension"):
|
||||||
|
write_file_by_extension(df, output_file)
|
||||||
|
|
||||||
# Tests for get_stats
|
# Tests for get_stats
|
||||||
def test_get_stats_found(self, sample_stats_df):
|
def test_get_stats_found(self, sample_stats_df):
|
||||||
"""Test retrieving statistics for an existing column."""
|
"""Test retrieving statistics for an existing column."""
|
||||||
@@ -224,6 +309,224 @@ class TestPipelineFunctions:
|
|||||||
result = run_qc_qa_pipeline(df, stats_df=None)
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
assert "Manual Review Flag (Invalid)" in result.columns
|
assert "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_cpt_code_cleaning(self):
|
||||||
|
"""Test CPT code parsing and cleaning."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"CPT4_PROC_CD": ["99213, 99214, INVALID"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "CPT4_PROC_CD_CLEANED" in result.columns
|
||||||
|
assert "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_modifier_validation(self):
|
||||||
|
"""Test CPT modifier validation."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"CPT4_PROC_MOD": ["25, 59, TOOLONG"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "CPT4_PROC_MOD_CLEANED" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_revenue_code_cleaning(self):
|
||||||
|
"""Test revenue code parsing and cleaning."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"REVENUE_CD": ["0450, 0360, INVALID"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "REVENUE_CD_CLEANED" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_diagnosis_code_cleaning(self):
|
||||||
|
"""Test diagnosis code parsing and cleaning."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"DIAG_CD": ["A00.0, B01.1"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "DIAG_CD_CLEANED" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_tin_validation(self):
|
||||||
|
"""Test TIN formatting and validation."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"TIN": ["123456789"],
|
||||||
|
"PROV_GROUP_TIN": ["12-3456789"],
|
||||||
|
"REIMB_PROV_TIN": ["invalid"],
|
||||||
|
"PROV_OTHER_TIN": ["987654321"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_npi_cleaning(self):
|
||||||
|
"""Test NPI cleaning and validation."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"PROV_GROUP_NPI": ["1234567890"],
|
||||||
|
"REIMB_PROV_NPI": ["1234567890|9876543210"],
|
||||||
|
"PROV_OTHER_NPI": ["invalid"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
# Should clean and validate NPIs
|
||||||
|
assert "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_lob_cleaning(self):
|
||||||
|
"""Test LOB standardization."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"LOB": ["Medicare Advantage"],
|
||||||
|
"AARETE_DERIVED_LINE_OF_BUSINESS": ["commercial"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
# LOB values should be standardized
|
||||||
|
assert "LOB" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_value_based_flag(self):
|
||||||
|
"""Test flagging value-based contracts."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"CONTRACT_NAME": ["Value Based Care Agreement"],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Other)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_duplicate_detection(self):
|
||||||
|
"""Test duplicate row detection."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf", "test.pdf"],
|
||||||
|
"RATE": [100, 100],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
# Should flag duplicates
|
||||||
|
flag_col = result["Manual Review Flag (Other)"]
|
||||||
|
assert any("Duplicate" in str(val) for val in flag_col.values)
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_json_validation(self):
|
||||||
|
"""Test JSON format validation."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf", "test2.pdf"],
|
||||||
|
"PROV_INFO_JSON": ['[{"key": "value"}]', 'invalid json'],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_negative_rate_flag(self):
|
||||||
|
"""Test flagging negative rates."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"REIMB_PCT_RATE": [-10.5],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
# Should flag negative rates
|
||||||
|
assert "Manual Review Flag (NegRate)" in result.columns or "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_mandatory_blank_check(self):
|
||||||
|
"""Test mandatory field blank check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"PROVIDER_INFO_JSON": [""],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": [""],
|
||||||
|
"CARVEOUT_CD": [""],
|
||||||
|
"REIMB_LESSER_OF_ID": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_fee_schedule_mandatory(self):
|
||||||
|
"""Test Fee Schedule mandatory fields check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": ["Fee Schedule"],
|
||||||
|
"AARETE_DERIVED_FEE_SCHEDULE": [""],
|
||||||
|
"REIMB_PCT_RATE": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_flat_rate_mandatory(self):
|
||||||
|
"""Test Flat Rate mandatory fields check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": ["Flat Rate"],
|
||||||
|
"REIMB_FEE_RATE": [""],
|
||||||
|
"CPT4_PROC_CD": [""],
|
||||||
|
"REVENUE_CD": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_per_diem_mandatory(self):
|
||||||
|
"""Test Per Diem mandatory fields check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": ["Per Diem"],
|
||||||
|
"REIMB_FEE_RATE": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_grouper_mandatory(self):
|
||||||
|
"""Test Grouper mandatory fields check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": ["Grouper"],
|
||||||
|
"GROUPER_TYPE": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_billed_charges_mandatory(self):
|
||||||
|
"""Test Billed Charges mandatory fields check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": ["Billed Charges"],
|
||||||
|
"REIMB_PCT_RATE": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_stats_outlier_detection(self):
|
||||||
|
"""Test outlier detection with stats DataFrame."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf", "test2.pdf"],
|
||||||
|
"REIMB_TERM": ["short", "This is an extremely long reimbursement term that exceeds normal length bounds"],
|
||||||
|
})
|
||||||
|
stats_df = pd.DataFrame({
|
||||||
|
"COL_NAME": ["REIMB_TERM"],
|
||||||
|
"MEAN_VALUE": [10.0],
|
||||||
|
"STD_VALUE": [2.0],
|
||||||
|
"MAX_LENGTH": [50],
|
||||||
|
"MIN_LENGTH": [5],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=stats_df)
|
||||||
|
# Should detect outliers based on stats
|
||||||
|
assert "Manual Review Flag (Invalid)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_lob_derivation_check(self):
|
||||||
|
"""Test LOB derivation missing check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"LOB": ["Commercial"],
|
||||||
|
"PROGRAM": ["HMO"],
|
||||||
|
"PRODUCT": ["Gold"],
|
||||||
|
"AARETE_DERIVED_LOB": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
|
def test_run_qc_qa_pipeline_reimb_fields_all_blank(self):
|
||||||
|
"""Test all reimbursement fields blank check."""
|
||||||
|
df = pd.DataFrame({
|
||||||
|
"FILE_NAME": ["test.pdf"],
|
||||||
|
"AARETE_DERIVED_REIMB_METHOD": [""],
|
||||||
|
"REIMB_TERM": [""],
|
||||||
|
"LESSER_OF_IND": [""],
|
||||||
|
"REIMB_PCT_RATE": [""],
|
||||||
|
"REIMB_FEE_RATE": [""],
|
||||||
|
})
|
||||||
|
result = run_qc_qa_pipeline(df, stats_df=None)
|
||||||
|
assert "Manual Review Flag (Blank)" in result.columns
|
||||||
|
|
||||||
# Tests for save_qc_qa_outputs
|
# Tests for save_qc_qa_outputs
|
||||||
def test_save_qc_qa_outputs_local_only(self, sample_dataframe, mocker):
|
def test_save_qc_qa_outputs_local_only(self, sample_dataframe, mocker):
|
||||||
"""Test saving outputs locally without S3."""
|
"""Test saving outputs locally without S3."""
|
||||||
@@ -606,24 +909,139 @@ class TestCLIIntegration:
|
|||||||
|
|
||||||
def test_main_missing_input_file(self, mocker):
|
def test_main_missing_input_file(self, mocker):
|
||||||
"""Test CLI exits when input_file is missing."""
|
"""Test CLI exits when input_file is missing."""
|
||||||
# Skip - mocking get_arg_value is complex due to module import
|
mocker.patch("src.qc_qa.__main__.get_arg_value", return_value=None)
|
||||||
pytest.skip("CLI tests require complex mocking - tested manually")
|
|
||||||
|
|
||||||
def test_main_file_not_found(self, mocker):
|
with pytest.raises(SystemExit) as exc_info:
|
||||||
|
from src.qc_qa.__main__ import main
|
||||||
|
main()
|
||||||
|
|
||||||
|
assert exc_info.value.code == 1
|
||||||
|
|
||||||
|
def test_main_file_not_found(self, mocker, tmp_path):
|
||||||
"""Test CLI exits when input file doesn't exist."""
|
"""Test CLI exits when input file doesn't exist."""
|
||||||
pytest.skip("CLI tests require complex mocking - tested manually")
|
non_existent = str(tmp_path / "does_not_exist.csv")
|
||||||
|
|
||||||
|
def mock_get_arg(key, default=None):
|
||||||
|
return {"input_file": non_existent, "output_file": None, "stats_file": None}.get(key, default)
|
||||||
|
|
||||||
|
mocker.patch("src.qc_qa.__main__.get_arg_value", side_effect=mock_get_arg)
|
||||||
|
|
||||||
|
with pytest.raises(SystemExit) as exc_info:
|
||||||
|
from src.qc_qa.__main__ import main
|
||||||
|
main()
|
||||||
|
|
||||||
|
assert exc_info.value.code == 1
|
||||||
|
|
||||||
def test_main_success_csv_input(self, mocker, tmp_path):
|
def test_main_success_csv_input(self, mocker, tmp_path):
|
||||||
"""Test successful CLI execution with CSV input."""
|
"""Test successful CLI execution with CSV input."""
|
||||||
pytest.skip("CLI tests require complex mocking - tested manually")
|
# Create test input file
|
||||||
|
input_file = tmp_path / "input.csv"
|
||||||
|
input_file.write_text("FILE_NAME,EFFECTIVE_DATE\ntest.pdf,01/01/2024\n")
|
||||||
|
|
||||||
|
output_file = tmp_path / "output.csv"
|
||||||
|
|
||||||
|
def mock_get_arg(key, default=None):
|
||||||
|
values = {
|
||||||
|
"input_file": str(input_file),
|
||||||
|
"output_file": str(output_file),
|
||||||
|
"stats_file": None
|
||||||
|
}
|
||||||
|
return values.get(key, default)
|
||||||
|
|
||||||
|
mocker.patch("src.qc_qa.__main__.get_arg_value", side_effect=mock_get_arg)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.WRITE_TO_S3", False)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.BATCH_ID", "test_batch")
|
||||||
|
mocker.patch("src.qc_qa.__main__.save_qc_qa_outputs")
|
||||||
|
|
||||||
|
from src.qc_qa.__main__ import main
|
||||||
|
main() # Should complete without error
|
||||||
|
|
||||||
|
# Verify output was created
|
||||||
|
assert output_file.exists()
|
||||||
|
|
||||||
def test_main_with_stats_file(self, mocker, tmp_path):
|
def test_main_with_stats_file(self, mocker, tmp_path):
|
||||||
"""Test CLI with statistics file."""
|
"""Test CLI with statistics file for outlier detection."""
|
||||||
pytest.skip("CLI tests require complex mocking - tested manually")
|
input_file = tmp_path / "input.csv"
|
||||||
|
input_file.write_text("FILE_NAME,REIMB_TERM\ntest.pdf,some term\n")
|
||||||
|
|
||||||
def test_main_with_legacy_output(self, mocker, tmp_path):
|
stats_file = tmp_path / "stats.csv"
|
||||||
"""Test CLI with legacy Excel output."""
|
stats_file.write_text("COL_NAME,MEAN_VALUE,STD_VALUE,MAX_LENGTH,MIN_LENGTH\nREIMB_TERM,10,2,50,5\n")
|
||||||
pytest.skip("CLI tests require complex mocking - tested manually")
|
|
||||||
|
output_file = tmp_path / "output.csv"
|
||||||
|
|
||||||
|
def mock_get_arg(key, default=None):
|
||||||
|
values = {
|
||||||
|
"input_file": str(input_file),
|
||||||
|
"output_file": str(output_file),
|
||||||
|
"stats_file": str(stats_file)
|
||||||
|
}
|
||||||
|
return values.get(key, default)
|
||||||
|
|
||||||
|
mocker.patch("src.qc_qa.__main__.get_arg_value", side_effect=mock_get_arg)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.WRITE_TO_S3", False)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.BATCH_ID", "test_batch")
|
||||||
|
mocker.patch("src.qc_qa.__main__.save_qc_qa_outputs")
|
||||||
|
|
||||||
|
from src.qc_qa.__main__ import main
|
||||||
|
main()
|
||||||
|
|
||||||
|
assert output_file.exists()
|
||||||
|
|
||||||
|
def test_main_with_excel_output(self, mocker, tmp_path):
|
||||||
|
"""Test CLI with Excel output file."""
|
||||||
|
input_file = tmp_path / "input.csv"
|
||||||
|
input_file.write_text("FILE_NAME,EFFECTIVE_DATE\ntest.pdf,01/01/2024\n")
|
||||||
|
|
||||||
|
output_file = tmp_path / "output.xlsx"
|
||||||
|
|
||||||
|
def mock_get_arg(key, default=None):
|
||||||
|
values = {
|
||||||
|
"input_file": str(input_file),
|
||||||
|
"output_file": str(output_file),
|
||||||
|
"stats_file": None
|
||||||
|
}
|
||||||
|
return values.get(key, default)
|
||||||
|
|
||||||
|
mocker.patch("src.qc_qa.__main__.get_arg_value", side_effect=mock_get_arg)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.WRITE_TO_S3", False)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.BATCH_ID", "test_batch")
|
||||||
|
mocker.patch("src.qc_qa.__main__.save_qc_qa_outputs")
|
||||||
|
|
||||||
|
from src.qc_qa.__main__ import main
|
||||||
|
main()
|
||||||
|
|
||||||
|
assert output_file.exists()
|
||||||
|
# Verify stats file also created with same extension
|
||||||
|
stats_output = tmp_path / "output_stats.xlsx"
|
||||||
|
assert stats_output.exists()
|
||||||
|
|
||||||
|
def test_main_pipeline_error_graceful_exit(self, mocker, tmp_path):
|
||||||
|
"""Test CLI exits gracefully when pipeline raises an error."""
|
||||||
|
input_file = tmp_path / "input.csv"
|
||||||
|
input_file.write_text("FILE_NAME\ntest.pdf\n")
|
||||||
|
|
||||||
|
output_file = tmp_path / "output.csv"
|
||||||
|
|
||||||
|
def mock_get_arg(key, default=None):
|
||||||
|
values = {
|
||||||
|
"input_file": str(input_file),
|
||||||
|
"output_file": str(output_file),
|
||||||
|
"stats_file": None
|
||||||
|
}
|
||||||
|
return values.get(key, default)
|
||||||
|
|
||||||
|
mocker.patch("src.qc_qa.__main__.get_arg_value", side_effect=mock_get_arg)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.WRITE_TO_S3", False)
|
||||||
|
mocker.patch("src.qc_qa.__main__.config.BATCH_ID", "test_batch")
|
||||||
|
mocker.patch("src.qc_qa.__main__.run_qc_qa_pipeline", side_effect=Exception("Pipeline error"))
|
||||||
|
|
||||||
|
with pytest.raises(SystemExit) as exc_info:
|
||||||
|
from src.qc_qa.__main__ import main
|
||||||
|
main()
|
||||||
|
|
||||||
|
assert exc_info.value.code == 1
|
||||||
|
# Output should NOT exist due to graceful failure
|
||||||
|
assert not output_file.exists()
|
||||||
|
|
||||||
|
|
||||||
class TestInvestmentMainIntegration:
|
class TestInvestmentMainIntegration:
|
||||||
|
|||||||
Reference in New Issue
Block a user