Merged in feature/streamlit_multipageApp (pull request #125)
Basic error handling for Client ID
This commit is contained in:
+165
-164
@@ -81,197 +81,198 @@ with client_row[0]:
|
||||
with client_row[1]:
|
||||
client = st.selectbox('Client Name',(client_list), label_visibility = "collapsed", index = None)
|
||||
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
if client:
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
|
||||
# to be deleted when buckets for different clients are ready; below line is added only for testing the corresponding DAG
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
# to be deleted when buckets for different clients are ready; below line is added only for testing the corresponding DAG
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/", Delimiter='/')
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/", Delimiter='/')
|
||||
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
|
||||
# Hardcoded batch_list for testing purposes
|
||||
# batch_list = ['batch_020524103737', 'batch_090524131433', 'batch_090524131607', 'batch_100524123000', 'batch_130524064322',
|
||||
# 'batch_160524071331', 'batch_200524213550', 'batch_250424112237', 'batch_280524120530', 'batch_280524121721', 'batch_280524144222',
|
||||
# 'batch_290524123926', 'batch_290524164044', 'batch_310524102029', 'batch_310524124050', 'batch_310524162346', 'batch_310524162631']
|
||||
# Hardcoded batch_list for testing purposes
|
||||
# batch_list = ['batch_020524103737', 'batch_090524131433', 'batch_090524131607', 'batch_100524123000', 'batch_130524064322',
|
||||
# 'batch_160524071331', 'batch_200524213550', 'batch_250424112237', 'batch_280524120530', 'batch_280524121721', 'batch_280524144222',
|
||||
# 'batch_290524123926', 'batch_290524164044', 'batch_310524102029', 'batch_310524124050', 'batch_310524162346', 'batch_310524162631']
|
||||
|
||||
if 'sorted_list' not in st.session_state:
|
||||
st.session_state.sorted_list = batch_list
|
||||
if 'sorted_list' not in st.session_state:
|
||||
st.session_state.sorted_list = batch_list
|
||||
|
||||
def sort_list(ex_list, sort_by, order):
|
||||
if sort_by == 'Alphabetical':
|
||||
ex_list = sorted(ex_list, reverse=(order == 'Descending'))
|
||||
elif sort_by == 'Create Date':
|
||||
ex_list = ex_list if order == 'Ascending' else list(reversed(ex_list))
|
||||
def sort_list(ex_list, sort_by, order):
|
||||
if sort_by == 'Alphabetical':
|
||||
ex_list = sorted(ex_list, reverse=(order == 'Descending'))
|
||||
elif sort_by == 'Create Date':
|
||||
ex_list = ex_list if order == 'Ascending' else list(reversed(ex_list))
|
||||
|
||||
return ex_list
|
||||
return ex_list
|
||||
|
||||
col1, col2, col3, col4 = st.columns([0.5, 0.5, 0.5, 0.5])
|
||||
col1, col2, col3, col4 = st.columns([0.5, 0.5, 0.5, 0.5])
|
||||
|
||||
|
||||
with col1:
|
||||
sort_by = st.radio("**Sort Batch_IDs**", ('Alphabetical', 'Create Date'))
|
||||
with col1:
|
||||
sort_by = st.radio("**Sort Batch_IDs**", ('Alphabetical', 'Create Date'))
|
||||
|
||||
with col2:
|
||||
order = st.radio('', ('Ascending','Descending'))
|
||||
with col2:
|
||||
order = st.radio('', ('Ascending','Descending'))
|
||||
|
||||
with col3:
|
||||
add_vertical_space(2)
|
||||
if st.button('Apply'):
|
||||
st.session_state.sorted_list = sort_list(batch_list, sort_by, order)
|
||||
with col3:
|
||||
add_vertical_space(2)
|
||||
if st.button('Apply'):
|
||||
st.session_state.sorted_list = sort_list(batch_list, sort_by, order)
|
||||
|
||||
path_row = st.columns([0.1, 0.8])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', st.session_state.sorted_list, label_visibility = "collapsed", index = None)
|
||||
path_row = st.columns([0.1, 0.8])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', st.session_state.sorted_list, label_visibility = "collapsed", index = None)
|
||||
|
||||
if not batch_id:
|
||||
batch_id = "None"
|
||||
if not batch_id:
|
||||
batch_id = "None"
|
||||
|
||||
checks = st.columns([0.1, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12])
|
||||
with checks[0]:
|
||||
st.write("**Group No.**")
|
||||
checks = st.columns([0.1, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12])
|
||||
with checks[0]:
|
||||
st.write("**Group No.**")
|
||||
|
||||
with checks[1]:
|
||||
a = st.checkbox('Unique Key', key = str(1), args="Unique")
|
||||
with checks[2]:
|
||||
b = st.checkbox('Pricing Before Carveouts', key = str(2))
|
||||
with checks[3]:
|
||||
c = st.checkbox('Contract Related', key = str(3))
|
||||
with checks[4]:
|
||||
d = st.checkbox('Provider', key = str(4))
|
||||
with checks[5]:
|
||||
e = st.checkbox('Timeline', key = str(5))
|
||||
with checks[6]:
|
||||
f = st.checkbox('Carveout Indicator', key = str(6))
|
||||
with checks[7]:
|
||||
g = st.checkbox('Carveout Methodology', key = str(7))
|
||||
with checks[1]:
|
||||
a = st.checkbox('Unique Key', key = str(1), args="Unique")
|
||||
with checks[2]:
|
||||
b = st.checkbox('Pricing Before Carveouts', key = str(2))
|
||||
with checks[3]:
|
||||
c = st.checkbox('Contract Related', key = str(3))
|
||||
with checks[4]:
|
||||
d = st.checkbox('Provider', key = str(4))
|
||||
with checks[5]:
|
||||
e = st.checkbox('Timeline', key = str(5))
|
||||
with checks[6]:
|
||||
f = st.checkbox('Carveout Indicator', key = str(6))
|
||||
with checks[7]:
|
||||
g = st.checkbox('Carveout Methodology', key = str(7))
|
||||
|
||||
add_vertical_space(1)
|
||||
add_vertical_space(1)
|
||||
|
||||
df = pd.DataFrame(columns=['Contract Name', 'Unique Key','Pricing Before Carveouts'
|
||||
, 'Contract Related', 'Provider', 'Timeline', 'Carveout Indicator', 'Carveout Methodology'])
|
||||
file_list = []
|
||||
file_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/"+batch_id+"/", Delimiter='/')
|
||||
df = pd.DataFrame(columns=['Contract Name', 'Unique Key','Pricing Before Carveouts'
|
||||
, 'Contract Related', 'Provider', 'Timeline', 'Carveout Indicator', 'Carveout Methodology'])
|
||||
file_list = []
|
||||
file_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/"+batch_id+"/", Delimiter='/')
|
||||
|
||||
# Hardcoded file_list for testing purposes
|
||||
# file_list = ['Boilerplate_TX Amendment Mission Health Network effective_040114 MU.pdf', 'Custom_TX - MP AMENDMENT - MISSION HEALTH NETWORK - MU.pdf',
|
||||
# 'Delaware First Health_First State Homecare Agency_212260_7 MU.pdf', 'Molina Healthcare of Texas, Inc. Amendment 4 - HIX ACA__EFF 01012016_MU.pdf']
|
||||
# Hardcoded file_list for testing purposes
|
||||
# file_list = ['Boilerplate_TX Amendment Mission Health Network effective_040114 MU.pdf', 'Custom_TX - MP AMENDMENT - MISSION HEALTH NETWORK - MU.pdf',
|
||||
# 'Delaware First Health_First State Homecare Agency_212260_7 MU.pdf', 'Molina Healthcare of Texas, Inc. Amendment 4 - HIX ACA__EFF 01012016_MU.pdf']
|
||||
|
||||
if st.button("Read the contracts from Path"):
|
||||
for obj in file_objects.get('Contents',[]):
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'].split('/')[-1])
|
||||
|
||||
df['Contract Name'] = file_list
|
||||
# df['Request ID'] = range(len(file_list))
|
||||
# df['Contract ID'] = file_list
|
||||
df['Unique Key'] = a
|
||||
df['Pricing Before Carveouts'] = b
|
||||
df['Contract Related'] = c
|
||||
df['Provider'] = d
|
||||
df['Timeline'] = e
|
||||
df['Carveout Indicator'] = f
|
||||
df['Carveout Methodology'] = g
|
||||
dir_path = os.path.dirname(os.path.realpath(__file__))
|
||||
print(f'DEBUGGING: PWD= {dir_path}')
|
||||
df.to_csv('temp1.csv', index=False)
|
||||
if st.button("Read the contracts from Path"):
|
||||
for obj in file_objects.get('Contents',[]):
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'].split('/')[-1])
|
||||
|
||||
df['Contract Name'] = file_list
|
||||
# df['Request ID'] = range(len(file_list))
|
||||
# df['Contract ID'] = file_list
|
||||
df['Unique Key'] = a
|
||||
df['Pricing Before Carveouts'] = b
|
||||
df['Contract Related'] = c
|
||||
df['Provider'] = d
|
||||
df['Timeline'] = e
|
||||
df['Carveout Indicator'] = f
|
||||
df['Carveout Methodology'] = g
|
||||
dir_path = os.path.dirname(os.path.realpath(__file__))
|
||||
print(f'DEBUGGING: PWD= {dir_path}')
|
||||
df.to_csv('temp1.csv', index=False)
|
||||
|
||||
add_vertical_space(1)
|
||||
df2 = pd.read_csv('temp1.csv')
|
||||
edited_df = st.data_editor(df2)
|
||||
add_vertical_space(1)
|
||||
df2 = pd.read_csv('temp1.csv')
|
||||
edited_df = st.data_editor(df2)
|
||||
|
||||
edited_df['REQUEST_USER'] = user_mail
|
||||
edited_df['LATEST_FLAG BOOLEAN'] = True
|
||||
edited_df['PIPELINE_KICKOFF_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
edited_df['REQUEST_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
edited_df['REQUEST_USER'] = user_mail
|
||||
edited_df['LATEST_FLAG BOOLEAN'] = True
|
||||
edited_df['PIPELINE_KICKOFF_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
edited_df['REQUEST_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
|
||||
csv = convert_df(edited_df)
|
||||
# edited_df = edited_df.reset_index() # make sure indexes pair with number of rows
|
||||
csv = convert_df(edited_df)
|
||||
# edited_df = edited_df.reset_index() # make sure indexes pair with number of rows
|
||||
|
||||
# additional_info = pd.DataFrame(columns=['REQUEST_ID','T_DRIVE_PATH','CLIENT_NAME'
|
||||
# , 'GROUP_NAME', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info = pd.DataFrame(columns=['CLIENT_NAME', 'BATCH_ID', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info.loc[0] = [client, batch_id, st.session_state.user_info['mail'], datetime.now().strftime("%Y-%m-%d %H:%M:%S")]
|
||||
st.write(additional_info)
|
||||
# additional_info = pd.DataFrame(columns=['REQUEST_ID','T_DRIVE_PATH','CLIENT_NAME'
|
||||
# , 'GROUP_NAME', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info = pd.DataFrame(columns=['CLIENT_NAME', 'BATCH_ID', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info.loc[0] = [client, batch_id, st.session_state.user_info['mail'], datetime.now().strftime("%Y-%m-%d %H:%M:%S")]
|
||||
st.write(additional_info)
|
||||
|
||||
st.session_state.contract_count = 0
|
||||
contract_list = []
|
||||
for index, row in edited_df.iterrows():
|
||||
allow_run_for_contract = False
|
||||
group_list = []
|
||||
if row['Unique Key']:
|
||||
group_list.append('A')
|
||||
allow_run_for_contract = True
|
||||
if row['Pricing Before Carveouts']:
|
||||
group_list.append('B')
|
||||
allow_run_for_contract = True
|
||||
if row['Contract Related']:
|
||||
group_list.append('C')
|
||||
allow_run_for_contract = True
|
||||
if row['Provider']:
|
||||
group_list.append('D')
|
||||
allow_run_for_contract = True
|
||||
if row['Timeline']:
|
||||
group_list.append('E')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Indicator']:
|
||||
group_list.append('F')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Methodology']:
|
||||
group_list.append('G')
|
||||
allow_run_for_contract = True
|
||||
if allow_run_for_contract: st.session_state.contract_count += 1
|
||||
entry_dict = {
|
||||
"contract_name": row['Contract Name'],
|
||||
"groups": group_list,
|
||||
"contract_source_path": "contracts-landing-zone/"+batch_id+"/"+row['Contract Name']
|
||||
st.session_state.contract_count = 0
|
||||
contract_list = []
|
||||
for index, row in edited_df.iterrows():
|
||||
allow_run_for_contract = False
|
||||
group_list = []
|
||||
if row['Unique Key']:
|
||||
group_list.append('A')
|
||||
allow_run_for_contract = True
|
||||
if row['Pricing Before Carveouts']:
|
||||
group_list.append('B')
|
||||
allow_run_for_contract = True
|
||||
if row['Contract Related']:
|
||||
group_list.append('C')
|
||||
allow_run_for_contract = True
|
||||
if row['Provider']:
|
||||
group_list.append('D')
|
||||
allow_run_for_contract = True
|
||||
if row['Timeline']:
|
||||
group_list.append('E')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Indicator']:
|
||||
group_list.append('F')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Methodology']:
|
||||
group_list.append('G')
|
||||
allow_run_for_contract = True
|
||||
if allow_run_for_contract: st.session_state.contract_count += 1
|
||||
entry_dict = {
|
||||
"contract_name": row['Contract Name'],
|
||||
"groups": group_list,
|
||||
"contract_source_path": "contracts-landing-zone/"+batch_id+"/"+row['Contract Name']
|
||||
}
|
||||
contract_list.append(entry_dict)
|
||||
|
||||
myobj = {
|
||||
"s3_bucket": client_bucket,
|
||||
"batch_id": batch_id,
|
||||
"client_name": client,
|
||||
"username": user_mail,
|
||||
"contract_list": contract_list
|
||||
}
|
||||
contract_list.append(entry_dict)
|
||||
|
||||
myobj = {
|
||||
"s3_bucket": client_bucket,
|
||||
"batch_id": batch_id,
|
||||
"client_name": client,
|
||||
"username": user_mail,
|
||||
"contract_list": contract_list
|
||||
}
|
||||
|
||||
buttons = st.columns([0.8, 0.2])
|
||||
with buttons[0]:
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
if st.button("Run Doczy.AI Pipeline"):
|
||||
if not st.session_state.contract_count == len(edited_df):
|
||||
st.error("Select at least one Group No. for every Contract")
|
||||
else:
|
||||
with st.spinner('Running...'):
|
||||
# csv_buf = StringIO()
|
||||
# additional_info.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/request_submission.csv')
|
||||
# csv_buf = StringIO()
|
||||
# edited_df.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/contract_config.csv')
|
||||
# try:
|
||||
# save_to_sf('load_request_and_contract_submissions', request_submission_file_name = "request_submission.csv", contract_config_file_name = "contract_config.csv")
|
||||
# except Exception as e:
|
||||
# st.write(e)
|
||||
response = requests.post(doczy_pipeline, json = myobj)
|
||||
if response.status_code >= 200 and response.status_code < 300:
|
||||
st.write("Success")
|
||||
# st.write(myobj)
|
||||
buttons = st.columns([0.8, 0.2])
|
||||
with buttons[0]:
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
if st.button("Run Doczy.AI Pipeline"):
|
||||
if not st.session_state.contract_count == len(edited_df):
|
||||
st.error("Select at least one Group No. for every Contract")
|
||||
else:
|
||||
st.write("Failed")
|
||||
# st.write(response.text)
|
||||
with st.spinner('Running...'):
|
||||
# csv_buf = StringIO()
|
||||
# additional_info.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/request_submission.csv')
|
||||
# csv_buf = StringIO()
|
||||
# edited_df.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/contract_config.csv')
|
||||
# try:
|
||||
# save_to_sf('load_request_and_contract_submissions', request_submission_file_name = "request_submission.csv", contract_config_file_name = "contract_config.csv")
|
||||
# except Exception as e:
|
||||
# st.write(e)
|
||||
response = requests.post(doczy_pipeline, json = myobj)
|
||||
if response.status_code >= 200 and response.status_code < 300:
|
||||
st.write("Success")
|
||||
# st.write(myobj)
|
||||
else:
|
||||
st.write("Failed")
|
||||
# st.write(response.text)
|
||||
|
||||
|
||||
+142
-141
@@ -113,155 +113,156 @@ with client_row[0]:
|
||||
with client_row[1]:
|
||||
client = st.selectbox('Client Name',(client_list), label_visibility = "collapsed", index= None)
|
||||
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
if client:
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="textract-receiver-processed-pdfs/", Delimiter='/')
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="textract-receiver-processed-pdfs/", Delimiter='/')
|
||||
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
|
||||
path_row = st.columns([0.2, 0.7, 0.1])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', batch_list, label_visibility = "collapsed", index = None)
|
||||
path_row = st.columns([0.2, 0.7, 0.1])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', batch_list, label_visibility = "collapsed", index = None)
|
||||
|
||||
if batch_id:
|
||||
if batch_id:
|
||||
|
||||
objects = s3_client.list_objects_v2(Bucket=client_bucket, Prefix="textract-receiver-processed-pdfs/"+batch_id+"/")
|
||||
file_list = []
|
||||
if 'Contents' in objects:
|
||||
for obj in objects['Contents']:
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'])
|
||||
else:
|
||||
st.error('This batch_id is empty.')
|
||||
|
||||
contract_list = sorted(file_list)
|
||||
|
||||
file_row = st.columns([0.2, 0.7, 0.1])
|
||||
with file_row[0]:
|
||||
st.write("**Contract Name**")
|
||||
with file_row[1]:
|
||||
file_name = st.selectbox('Select a file', ['All'] + contract_list, label_visibility = "collapsed", index= None)
|
||||
|
||||
field_row = st.columns([0.2, 0.7, 0.1])
|
||||
with field_row[0]:
|
||||
st.write("**Field Group**")
|
||||
with field_row[1]:
|
||||
field_group = st.selectbox('Field Group',('Unique Key', 'Contract Related', 'Pricing Before Carveouts - I'
|
||||
, 'Pricing Before Carveouts - II', 'Carveout Indicator, Code Type and Code #s - I'
|
||||
, 'Carveout Indicator, Code Type and Code #s - II', 'Carveout Indicator, Code Type and Code #s - III'
|
||||
, 'Optimize Carving Indic.', 'Carveout Method - I', 'Carveout Method - II', 'Provider'
|
||||
, 'Timeline'), label_visibility = "collapsed", index = None)
|
||||
|
||||
if field_group == 'Unique Key':
|
||||
fields = fields[fields['PRIORITY'] == 'A']
|
||||
elif field_group == 'Contract Related':
|
||||
fields = fields[fields['PRIORITY'] == 'C']
|
||||
elif field_group == 'Pricing Before Carveouts - I':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[0]
|
||||
elif field_group == 'Pricing Before Carveouts - II':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - I':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[0]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - II':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - III':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[2]
|
||||
elif field_group == 'Carveout Methodology - I':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[0]
|
||||
elif field_group == 'Carveout Methodology - II':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[1]
|
||||
elif field_group == 'Carveout Method - III':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[2]
|
||||
elif field_group == 'Carveout Method - IV':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[3]
|
||||
elif field_group == 'Provider':
|
||||
fields = fields[fields['PRIORITY'] == 'D']
|
||||
elif field_group == 'Timeline':
|
||||
fields = fields[fields['PRIORITY'] == 'E']
|
||||
|
||||
if st.button("Show Results"):
|
||||
query = 'select * from "DOCZY_PIPELINE_RAW_OUTPUT"'
|
||||
cur.execute(query)
|
||||
df2 = pd.DataFrame.from_records(iter(cur), columns=[x[0] for x in cur.description])
|
||||
|
||||
# get this dataframe from snowflake table
|
||||
# df2 = pd.DataFrame(columns=['Contract Name','Field Name', 'SF_DB_COL_NAME', 'Snippet','Page Number'
|
||||
# , 'Field Extracted Value', 'Actual Value','Imputed Value'])
|
||||
df2.to_csv('temp2.csv', index=False)
|
||||
|
||||
if st.button("Show PDF"):
|
||||
if file_name == None or file_name == "All":
|
||||
st.error("Choose one specific file.")
|
||||
objects = s3_client.list_objects_v2(Bucket=client_bucket, Prefix="textract-receiver-processed-pdfs/"+batch_id+"/")
|
||||
file_list = []
|
||||
if 'Contents' in objects:
|
||||
for obj in objects['Contents']:
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'])
|
||||
else:
|
||||
with st.sidebar:
|
||||
st.markdown(
|
||||
"""
|
||||
<style>
|
||||
section[data-testid="stSidebar"] {
|
||||
width: 550px !important; # Set the width to your desired value
|
||||
}
|
||||
</style>
|
||||
""",
|
||||
unsafe_allow_html=True,
|
||||
)
|
||||
s3_obj = s3_client.get_object(Bucket = client_bucket, Key = file_name)
|
||||
data=s3_obj['Body'].read()
|
||||
pdf_viewer(data, width=1500)
|
||||
st.error('This batch_id is empty.')
|
||||
|
||||
# if st.button("Show PDF"):
|
||||
# if file_name == None or file_name == "All":
|
||||
# st.error("Choose one specific file.")
|
||||
# else:
|
||||
# with st.sidebar:
|
||||
# with open(file_name, "rb") as f:
|
||||
# base64_pdf = base64.b64encode(f.read()).decode('utf-8')
|
||||
# # Embedding PDF in HTML
|
||||
# pdf_display = F'<iframe src="data:application/pdf;base64,{base64_pdf}" width="500" height="1000" type="application/pdf"></iframe>'
|
||||
# # Displaying File
|
||||
# st.markdown(
|
||||
# """
|
||||
# <style>
|
||||
# section[data-testid="stSidebar"] {
|
||||
# width: 600px !important; # Set the width to your desired value
|
||||
# }
|
||||
# </style>
|
||||
# """,
|
||||
# unsafe_allow_html=True,
|
||||
# )
|
||||
# st.markdown(pdf_display, unsafe_allow_html=True)
|
||||
contract_list = sorted(file_list)
|
||||
|
||||
df2 = pd.read_csv('temp2.csv')
|
||||
df2['Imputed Value'] = ''
|
||||
edited_df = st.data_editor(df2)
|
||||
file_row = st.columns([0.2, 0.7, 0.1])
|
||||
with file_row[0]:
|
||||
st.write("**Contract Name**")
|
||||
with file_row[1]:
|
||||
file_name = st.selectbox('Select a file', ['All'] + contract_list, label_visibility = "collapsed", index= None)
|
||||
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
field_row = st.columns([0.2, 0.7, 0.1])
|
||||
with field_row[0]:
|
||||
st.write("**Field Group**")
|
||||
with field_row[1]:
|
||||
field_group = st.selectbox('Field Group',('Unique Key', 'Contract Related', 'Pricing Before Carveouts - I'
|
||||
, 'Pricing Before Carveouts - II', 'Carveout Indicator, Code Type and Code #s - I'
|
||||
, 'Carveout Indicator, Code Type and Code #s - II', 'Carveout Indicator, Code Type and Code #s - III'
|
||||
, 'Optimize Carving Indic.', 'Carveout Method - I', 'Carveout Method - II', 'Provider'
|
||||
, 'Timeline'), label_visibility = "collapsed", index = None)
|
||||
|
||||
csv = convert_df(edited_df)
|
||||
if field_group == 'Unique Key':
|
||||
fields = fields[fields['PRIORITY'] == 'A']
|
||||
elif field_group == 'Contract Related':
|
||||
fields = fields[fields['PRIORITY'] == 'C']
|
||||
elif field_group == 'Pricing Before Carveouts - I':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[0]
|
||||
elif field_group == 'Pricing Before Carveouts - II':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - I':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[0]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - II':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - III':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[2]
|
||||
elif field_group == 'Carveout Methodology - I':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[0]
|
||||
elif field_group == 'Carveout Methodology - II':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[1]
|
||||
elif field_group == 'Carveout Method - III':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[2]
|
||||
elif field_group == 'Carveout Method - IV':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[3]
|
||||
elif field_group == 'Provider':
|
||||
fields = fields[fields['PRIORITY'] == 'D']
|
||||
elif field_group == 'Timeline':
|
||||
fields = fields[fields['PRIORITY'] == 'E']
|
||||
|
||||
buttons = st.columns(3)
|
||||
with buttons[0]:
|
||||
# st.button("Save All Imputations")
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
# st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
st.write("")
|
||||
with buttons[2]:
|
||||
if st.button("Kickoff Database Integration"):
|
||||
st.write("Stored in DB")
|
||||
if st.button("Show Results"):
|
||||
query = 'select * from "DOCZY_PIPELINE_RAW_OUTPUT"'
|
||||
cur.execute(query)
|
||||
df2 = pd.DataFrame.from_records(iter(cur), columns=[x[0] for x in cur.description])
|
||||
|
||||
# get this dataframe from snowflake table
|
||||
# df2 = pd.DataFrame(columns=['Contract Name','Field Name', 'SF_DB_COL_NAME', 'Snippet','Page Number'
|
||||
# , 'Field Extracted Value', 'Actual Value','Imputed Value'])
|
||||
df2.to_csv('temp2.csv', index=False)
|
||||
|
||||
if st.button("Show PDF"):
|
||||
if file_name == None or file_name == "All":
|
||||
st.error("Choose one specific file.")
|
||||
else:
|
||||
with st.sidebar:
|
||||
st.markdown(
|
||||
"""
|
||||
<style>
|
||||
section[data-testid="stSidebar"] {
|
||||
width: 550px !important; # Set the width to your desired value
|
||||
}
|
||||
</style>
|
||||
""",
|
||||
unsafe_allow_html=True,
|
||||
)
|
||||
s3_obj = s3_client.get_object(Bucket = client_bucket, Key = file_name)
|
||||
data=s3_obj['Body'].read()
|
||||
pdf_viewer(data, width=1500)
|
||||
|
||||
# if st.button("Show PDF"):
|
||||
# if file_name == None or file_name == "All":
|
||||
# st.error("Choose one specific file.")
|
||||
# else:
|
||||
# with st.sidebar:
|
||||
# with open(file_name, "rb") as f:
|
||||
# base64_pdf = base64.b64encode(f.read()).decode('utf-8')
|
||||
# # Embedding PDF in HTML
|
||||
# pdf_display = F'<iframe src="data:application/pdf;base64,{base64_pdf}" width="500" height="1000" type="application/pdf"></iframe>'
|
||||
# # Displaying File
|
||||
# st.markdown(
|
||||
# """
|
||||
# <style>
|
||||
# section[data-testid="stSidebar"] {
|
||||
# width: 600px !important; # Set the width to your desired value
|
||||
# }
|
||||
# </style>
|
||||
# """,
|
||||
# unsafe_allow_html=True,
|
||||
# )
|
||||
# st.markdown(pdf_display, unsafe_allow_html=True)
|
||||
|
||||
df2 = pd.read_csv('temp2.csv')
|
||||
df2['Imputed Value'] = ''
|
||||
edited_df = st.data_editor(df2)
|
||||
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
|
||||
csv = convert_df(edited_df)
|
||||
|
||||
buttons = st.columns(3)
|
||||
with buttons[0]:
|
||||
# st.button("Save All Imputations")
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
# st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
st.write("")
|
||||
with buttons[2]:
|
||||
if st.button("Kickoff Database Integration"):
|
||||
st.write("Stored in DB")
|
||||
@@ -81,197 +81,198 @@ with client_row[0]:
|
||||
with client_row[1]:
|
||||
client = st.selectbox('Client Name',(client_list), label_visibility = "collapsed", index = None)
|
||||
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
if client:
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
|
||||
# to be deleted when buckets for different clients are ready; below line is added only for testing the corresponding DAG
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
# to be deleted when buckets for different clients are ready; below line is added only for testing the corresponding DAG
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/", Delimiter='/')
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/", Delimiter='/')
|
||||
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
|
||||
# Hardcoded batch_list for testing purposes
|
||||
# batch_list = ['batch_020524103737', 'batch_090524131433', 'batch_090524131607', 'batch_100524123000', 'batch_130524064322',
|
||||
# 'batch_160524071331', 'batch_200524213550', 'batch_250424112237', 'batch_280524120530', 'batch_280524121721', 'batch_280524144222',
|
||||
# 'batch_290524123926', 'batch_290524164044', 'batch_310524102029', 'batch_310524124050', 'batch_310524162346', 'batch_310524162631']
|
||||
# Hardcoded batch_list for testing purposes
|
||||
# batch_list = ['batch_020524103737', 'batch_090524131433', 'batch_090524131607', 'batch_100524123000', 'batch_130524064322',
|
||||
# 'batch_160524071331', 'batch_200524213550', 'batch_250424112237', 'batch_280524120530', 'batch_280524121721', 'batch_280524144222',
|
||||
# 'batch_290524123926', 'batch_290524164044', 'batch_310524102029', 'batch_310524124050', 'batch_310524162346', 'batch_310524162631']
|
||||
|
||||
if 'sorted_list' not in st.session_state:
|
||||
st.session_state.sorted_list = batch_list
|
||||
if 'sorted_list' not in st.session_state:
|
||||
st.session_state.sorted_list = batch_list
|
||||
|
||||
def sort_list(ex_list, sort_by, order):
|
||||
if sort_by == 'Alphabetical':
|
||||
ex_list = sorted(ex_list, reverse=(order == 'Descending'))
|
||||
elif sort_by == 'Create Date':
|
||||
ex_list = ex_list if order == 'Ascending' else list(reversed(ex_list))
|
||||
def sort_list(ex_list, sort_by, order):
|
||||
if sort_by == 'Alphabetical':
|
||||
ex_list = sorted(ex_list, reverse=(order == 'Descending'))
|
||||
elif sort_by == 'Create Date':
|
||||
ex_list = ex_list if order == 'Ascending' else list(reversed(ex_list))
|
||||
|
||||
return ex_list
|
||||
return ex_list
|
||||
|
||||
col1, col2, col3, col4 = st.columns([0.5, 0.5, 0.5, 0.5])
|
||||
col1, col2, col3, col4 = st.columns([0.5, 0.5, 0.5, 0.5])
|
||||
|
||||
|
||||
with col1:
|
||||
sort_by = st.radio("**Sort Batch_IDs**", ('Alphabetical', 'Create Date'))
|
||||
with col1:
|
||||
sort_by = st.radio("**Sort Batch_IDs**", ('Alphabetical', 'Create Date'))
|
||||
|
||||
with col2:
|
||||
order = st.radio('', ('Ascending','Descending'))
|
||||
with col2:
|
||||
order = st.radio('', ('Ascending','Descending'))
|
||||
|
||||
with col3:
|
||||
add_vertical_space(2)
|
||||
if st.button('Apply'):
|
||||
st.session_state.sorted_list = sort_list(batch_list, sort_by, order)
|
||||
with col3:
|
||||
add_vertical_space(2)
|
||||
if st.button('Apply'):
|
||||
st.session_state.sorted_list = sort_list(batch_list, sort_by, order)
|
||||
|
||||
path_row = st.columns([0.1, 0.8])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', st.session_state.sorted_list, label_visibility = "collapsed", index = None)
|
||||
path_row = st.columns([0.1, 0.8])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', st.session_state.sorted_list, label_visibility = "collapsed", index = None)
|
||||
|
||||
if not batch_id:
|
||||
batch_id = "None"
|
||||
if not batch_id:
|
||||
batch_id = "None"
|
||||
|
||||
checks = st.columns([0.1, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12])
|
||||
with checks[0]:
|
||||
st.write("**Group No.**")
|
||||
checks = st.columns([0.1, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12, 0.12])
|
||||
with checks[0]:
|
||||
st.write("**Group No.**")
|
||||
|
||||
with checks[1]:
|
||||
a = st.checkbox('Unique Key', key = str(1), args="Unique")
|
||||
with checks[2]:
|
||||
b = st.checkbox('Pricing Before Carveouts', key = str(2))
|
||||
with checks[3]:
|
||||
c = st.checkbox('Contract Related', key = str(3))
|
||||
with checks[4]:
|
||||
d = st.checkbox('Provider', key = str(4))
|
||||
with checks[5]:
|
||||
e = st.checkbox('Timeline', key = str(5))
|
||||
with checks[6]:
|
||||
f = st.checkbox('Carveout Indicator', key = str(6))
|
||||
with checks[7]:
|
||||
g = st.checkbox('Carveout Methodology', key = str(7))
|
||||
with checks[1]:
|
||||
a = st.checkbox('Unique Key', key = str(1), args="Unique")
|
||||
with checks[2]:
|
||||
b = st.checkbox('Pricing Before Carveouts', key = str(2))
|
||||
with checks[3]:
|
||||
c = st.checkbox('Contract Related', key = str(3))
|
||||
with checks[4]:
|
||||
d = st.checkbox('Provider', key = str(4))
|
||||
with checks[5]:
|
||||
e = st.checkbox('Timeline', key = str(5))
|
||||
with checks[6]:
|
||||
f = st.checkbox('Carveout Indicator', key = str(6))
|
||||
with checks[7]:
|
||||
g = st.checkbox('Carveout Methodology', key = str(7))
|
||||
|
||||
add_vertical_space(1)
|
||||
add_vertical_space(1)
|
||||
|
||||
df = pd.DataFrame(columns=['Contract Name', 'Unique Key','Pricing Before Carveouts'
|
||||
, 'Contract Related', 'Provider', 'Timeline', 'Carveout Indicator', 'Carveout Methodology'])
|
||||
file_list = []
|
||||
file_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/"+batch_id+"/", Delimiter='/')
|
||||
df = pd.DataFrame(columns=['Contract Name', 'Unique Key','Pricing Before Carveouts'
|
||||
, 'Contract Related', 'Provider', 'Timeline', 'Carveout Indicator', 'Carveout Methodology'])
|
||||
file_list = []
|
||||
file_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="contracts-landing-zone/"+batch_id+"/", Delimiter='/')
|
||||
|
||||
# Hardcoded file_list for testing purposes
|
||||
# file_list = ['Boilerplate_TX Amendment Mission Health Network effective_040114 MU.pdf', 'Custom_TX - MP AMENDMENT - MISSION HEALTH NETWORK - MU.pdf',
|
||||
# 'Delaware First Health_First State Homecare Agency_212260_7 MU.pdf', 'Molina Healthcare of Texas, Inc. Amendment 4 - HIX ACA__EFF 01012016_MU.pdf']
|
||||
# Hardcoded file_list for testing purposes
|
||||
# file_list = ['Boilerplate_TX Amendment Mission Health Network effective_040114 MU.pdf', 'Custom_TX - MP AMENDMENT - MISSION HEALTH NETWORK - MU.pdf',
|
||||
# 'Delaware First Health_First State Homecare Agency_212260_7 MU.pdf', 'Molina Healthcare of Texas, Inc. Amendment 4 - HIX ACA__EFF 01012016_MU.pdf']
|
||||
|
||||
if st.button("Read the contracts from Path"):
|
||||
for obj in file_objects.get('Contents',[]):
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'].split('/')[-1])
|
||||
|
||||
df['Contract Name'] = file_list
|
||||
# df['Request ID'] = range(len(file_list))
|
||||
# df['Contract ID'] = file_list
|
||||
df['Unique Key'] = a
|
||||
df['Pricing Before Carveouts'] = b
|
||||
df['Contract Related'] = c
|
||||
df['Provider'] = d
|
||||
df['Timeline'] = e
|
||||
df['Carveout Indicator'] = f
|
||||
df['Carveout Methodology'] = g
|
||||
dir_path = os.path.dirname(os.path.realpath(__file__))
|
||||
print(f'DEBUGGING: PWD= {dir_path}')
|
||||
df.to_csv('temp1.csv', index=False)
|
||||
if st.button("Read the contracts from Path"):
|
||||
for obj in file_objects.get('Contents',[]):
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'].split('/')[-1])
|
||||
|
||||
df['Contract Name'] = file_list
|
||||
# df['Request ID'] = range(len(file_list))
|
||||
# df['Contract ID'] = file_list
|
||||
df['Unique Key'] = a
|
||||
df['Pricing Before Carveouts'] = b
|
||||
df['Contract Related'] = c
|
||||
df['Provider'] = d
|
||||
df['Timeline'] = e
|
||||
df['Carveout Indicator'] = f
|
||||
df['Carveout Methodology'] = g
|
||||
dir_path = os.path.dirname(os.path.realpath(__file__))
|
||||
print(f'DEBUGGING: PWD= {dir_path}')
|
||||
df.to_csv('temp1.csv', index=False)
|
||||
|
||||
add_vertical_space(1)
|
||||
df2 = pd.read_csv('temp1.csv')
|
||||
edited_df = st.data_editor(df2)
|
||||
add_vertical_space(1)
|
||||
df2 = pd.read_csv('temp1.csv')
|
||||
edited_df = st.data_editor(df2)
|
||||
|
||||
edited_df['REQUEST_USER'] = user_mail
|
||||
edited_df['LATEST_FLAG BOOLEAN'] = True
|
||||
edited_df['PIPELINE_KICKOFF_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
edited_df['REQUEST_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
edited_df['REQUEST_USER'] = user_mail
|
||||
edited_df['LATEST_FLAG BOOLEAN'] = True
|
||||
edited_df['PIPELINE_KICKOFF_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
edited_df['REQUEST_DATETIME'] = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
|
||||
csv = convert_df(edited_df)
|
||||
# edited_df = edited_df.reset_index() # make sure indexes pair with number of rows
|
||||
csv = convert_df(edited_df)
|
||||
# edited_df = edited_df.reset_index() # make sure indexes pair with number of rows
|
||||
|
||||
# additional_info = pd.DataFrame(columns=['REQUEST_ID','T_DRIVE_PATH','CLIENT_NAME'
|
||||
# , 'GROUP_NAME', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info = pd.DataFrame(columns=['CLIENT_NAME', 'BATCH_ID', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info.loc[0] = [client, batch_id, st.session_state.user_info['mail'], datetime.now().strftime("%Y-%m-%d %H:%M:%S")]
|
||||
st.write(additional_info)
|
||||
# additional_info = pd.DataFrame(columns=['REQUEST_ID','T_DRIVE_PATH','CLIENT_NAME'
|
||||
# , 'GROUP_NAME', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info = pd.DataFrame(columns=['CLIENT_NAME', 'BATCH_ID', 'REQUEST_USERNAME', 'REQUEST_DATETIME'])
|
||||
additional_info.loc[0] = [client, batch_id, st.session_state.user_info['mail'], datetime.now().strftime("%Y-%m-%d %H:%M:%S")]
|
||||
st.write(additional_info)
|
||||
|
||||
st.session_state.contract_count = 0
|
||||
contract_list = []
|
||||
for index, row in edited_df.iterrows():
|
||||
allow_run_for_contract = False
|
||||
group_list = []
|
||||
if row['Unique Key']:
|
||||
group_list.append('A')
|
||||
allow_run_for_contract = True
|
||||
if row['Pricing Before Carveouts']:
|
||||
group_list.append('B')
|
||||
allow_run_for_contract = True
|
||||
if row['Contract Related']:
|
||||
group_list.append('C')
|
||||
allow_run_for_contract = True
|
||||
if row['Provider']:
|
||||
group_list.append('D')
|
||||
allow_run_for_contract = True
|
||||
if row['Timeline']:
|
||||
group_list.append('E')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Indicator']:
|
||||
group_list.append('F')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Methodology']:
|
||||
group_list.append('G')
|
||||
allow_run_for_contract = True
|
||||
if allow_run_for_contract: st.session_state.contract_count += 1
|
||||
entry_dict = {
|
||||
"contract_name": row['Contract Name'],
|
||||
"groups": group_list,
|
||||
"contract_source_path": "contracts-landing-zone/"+batch_id+"/"+row['Contract Name']
|
||||
st.session_state.contract_count = 0
|
||||
contract_list = []
|
||||
for index, row in edited_df.iterrows():
|
||||
allow_run_for_contract = False
|
||||
group_list = []
|
||||
if row['Unique Key']:
|
||||
group_list.append('A')
|
||||
allow_run_for_contract = True
|
||||
if row['Pricing Before Carveouts']:
|
||||
group_list.append('B')
|
||||
allow_run_for_contract = True
|
||||
if row['Contract Related']:
|
||||
group_list.append('C')
|
||||
allow_run_for_contract = True
|
||||
if row['Provider']:
|
||||
group_list.append('D')
|
||||
allow_run_for_contract = True
|
||||
if row['Timeline']:
|
||||
group_list.append('E')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Indicator']:
|
||||
group_list.append('F')
|
||||
allow_run_for_contract = True
|
||||
if row['Carveout Methodology']:
|
||||
group_list.append('G')
|
||||
allow_run_for_contract = True
|
||||
if allow_run_for_contract: st.session_state.contract_count += 1
|
||||
entry_dict = {
|
||||
"contract_name": row['Contract Name'],
|
||||
"groups": group_list,
|
||||
"contract_source_path": "contracts-landing-zone/"+batch_id+"/"+row['Contract Name']
|
||||
}
|
||||
contract_list.append(entry_dict)
|
||||
|
||||
myobj = {
|
||||
"s3_bucket": client_bucket,
|
||||
"batch_id": batch_id,
|
||||
"client_name": client,
|
||||
"username": user_mail,
|
||||
"contract_list": contract_list
|
||||
}
|
||||
contract_list.append(entry_dict)
|
||||
|
||||
myobj = {
|
||||
"s3_bucket": client_bucket,
|
||||
"batch_id": batch_id,
|
||||
"client_name": client,
|
||||
"username": user_mail,
|
||||
"contract_list": contract_list
|
||||
}
|
||||
|
||||
buttons = st.columns([0.8, 0.2])
|
||||
with buttons[0]:
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
if st.button("Run Doczy.AI Pipeline"):
|
||||
if not st.session_state.contract_count == len(edited_df):
|
||||
st.error("Select at least one Group No. for every Contract")
|
||||
else:
|
||||
with st.spinner('Running...'):
|
||||
# csv_buf = StringIO()
|
||||
# additional_info.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/request_submission.csv')
|
||||
# csv_buf = StringIO()
|
||||
# edited_df.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/contract_config.csv')
|
||||
# try:
|
||||
# save_to_sf('load_request_and_contract_submissions', request_submission_file_name = "request_submission.csv", contract_config_file_name = "contract_config.csv")
|
||||
# except Exception as e:
|
||||
# st.write(e)
|
||||
response = requests.post(doczy_pipeline, json = myobj)
|
||||
if response.status_code >= 200 and response.status_code < 300:
|
||||
st.write("Success")
|
||||
# st.write(myobj)
|
||||
buttons = st.columns([0.8, 0.2])
|
||||
with buttons[0]:
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
if st.button("Run Doczy.AI Pipeline"):
|
||||
if not st.session_state.contract_count == len(edited_df):
|
||||
st.error("Select at least one Group No. for every Contract")
|
||||
else:
|
||||
st.write("Failed")
|
||||
# st.write(response.text)
|
||||
with st.spinner('Running...'):
|
||||
# csv_buf = StringIO()
|
||||
# additional_info.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/request_submission.csv')
|
||||
# csv_buf = StringIO()
|
||||
# edited_df.to_csv(csv_buf, header=True, index=False)
|
||||
# csv_buf.seek(0)
|
||||
# s3_client.put_object(Bucket='doczy-dev-infra-raw-data-ingestion', Body=csv_buf.getvalue(), Key='training_interface/contract_config.csv')
|
||||
# try:
|
||||
# save_to_sf('load_request_and_contract_submissions', request_submission_file_name = "request_submission.csv", contract_config_file_name = "contract_config.csv")
|
||||
# except Exception as e:
|
||||
# st.write(e)
|
||||
response = requests.post(doczy_pipeline, json = myobj)
|
||||
if response.status_code >= 200 and response.status_code < 300:
|
||||
st.write("Success")
|
||||
# st.write(myobj)
|
||||
else:
|
||||
st.write("Failed")
|
||||
# st.write(response.text)
|
||||
|
||||
|
||||
@@ -113,155 +113,156 @@ with client_row[0]:
|
||||
with client_row[1]:
|
||||
client = st.selectbox('Client Name',(client_list), label_visibility = "collapsed", index= None)
|
||||
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
if client:
|
||||
client_bucket = client_s3_paths.get(client)
|
||||
# client_bucket = 'doczyai-use2-d-cn1-s3-textract-processing-001'
|
||||
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="textract-receiver-processed-pdfs/", Delimiter='/')
|
||||
batch_objects = s3_client.list_objects_v2(Bucket=client_bucket
|
||||
, Prefix="textract-receiver-processed-pdfs/", Delimiter='/')
|
||||
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
batch_list = []
|
||||
for prefix in batch_objects['CommonPrefixes']:
|
||||
batch_list.append(prefix['Prefix'][:-1].split('/')[-1])
|
||||
|
||||
path_row = st.columns([0.2, 0.7, 0.1])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', batch_list, label_visibility = "collapsed", index = None)
|
||||
path_row = st.columns([0.2, 0.7, 0.1])
|
||||
with path_row[0]:
|
||||
st.write("**Batch ID**")
|
||||
with path_row[1]:
|
||||
batch_id = st.selectbox('**Batch ID**', batch_list, label_visibility = "collapsed", index = None)
|
||||
|
||||
if batch_id:
|
||||
if batch_id:
|
||||
|
||||
objects = s3_client.list_objects_v2(Bucket=client_bucket, Prefix="textract-receiver-processed-pdfs/"+batch_id+"/")
|
||||
file_list = []
|
||||
if 'Contents' in objects:
|
||||
for obj in objects['Contents']:
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'])
|
||||
else:
|
||||
st.error('This batch_id is empty.')
|
||||
|
||||
contract_list = sorted(file_list)
|
||||
|
||||
file_row = st.columns([0.2, 0.7, 0.1])
|
||||
with file_row[0]:
|
||||
st.write("**Contract Name**")
|
||||
with file_row[1]:
|
||||
file_name = st.selectbox('Select a file', ['All'] + contract_list, label_visibility = "collapsed", index= None)
|
||||
|
||||
field_row = st.columns([0.2, 0.7, 0.1])
|
||||
with field_row[0]:
|
||||
st.write("**Field Group**")
|
||||
with field_row[1]:
|
||||
field_group = st.selectbox('Field Group',('Unique Key', 'Contract Related', 'Pricing Before Carveouts - I'
|
||||
, 'Pricing Before Carveouts - II', 'Carveout Indicator, Code Type and Code #s - I'
|
||||
, 'Carveout Indicator, Code Type and Code #s - II', 'Carveout Indicator, Code Type and Code #s - III'
|
||||
, 'Optimize Carving Indic.', 'Carveout Method - I', 'Carveout Method - II', 'Provider'
|
||||
, 'Timeline'), label_visibility = "collapsed", index = None)
|
||||
|
||||
if field_group == 'Unique Key':
|
||||
fields = fields[fields['PRIORITY'] == 'A']
|
||||
elif field_group == 'Contract Related':
|
||||
fields = fields[fields['PRIORITY'] == 'C']
|
||||
elif field_group == 'Pricing Before Carveouts - I':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[0]
|
||||
elif field_group == 'Pricing Before Carveouts - II':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - I':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[0]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - II':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - III':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[2]
|
||||
elif field_group == 'Carveout Methodology - I':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[0]
|
||||
elif field_group == 'Carveout Methodology - II':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[1]
|
||||
elif field_group == 'Carveout Method - III':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[2]
|
||||
elif field_group == 'Carveout Method - IV':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[3]
|
||||
elif field_group == 'Provider':
|
||||
fields = fields[fields['PRIORITY'] == 'D']
|
||||
elif field_group == 'Timeline':
|
||||
fields = fields[fields['PRIORITY'] == 'E']
|
||||
|
||||
if st.button("Show Results"):
|
||||
query = 'select * from "DOCZY_PIPELINE_RAW_OUTPUT"'
|
||||
cur.execute(query)
|
||||
df2 = pd.DataFrame.from_records(iter(cur), columns=[x[0] for x in cur.description])
|
||||
|
||||
# get this dataframe from snowflake table
|
||||
# df2 = pd.DataFrame(columns=['Contract Name','Field Name', 'SF_DB_COL_NAME', 'Snippet','Page Number'
|
||||
# , 'Field Extracted Value', 'Actual Value','Imputed Value'])
|
||||
df2.to_csv('temp2.csv', index=False)
|
||||
|
||||
if st.button("Show PDF"):
|
||||
if file_name == None or file_name == "All":
|
||||
st.error("Choose one specific file.")
|
||||
objects = s3_client.list_objects_v2(Bucket=client_bucket, Prefix="textract-receiver-processed-pdfs/"+batch_id+"/")
|
||||
file_list = []
|
||||
if 'Contents' in objects:
|
||||
for obj in objects['Contents']:
|
||||
if not obj['Key'].endswith('/'):
|
||||
file_list.append(obj['Key'])
|
||||
else:
|
||||
with st.sidebar:
|
||||
st.markdown(
|
||||
"""
|
||||
<style>
|
||||
section[data-testid="stSidebar"] {
|
||||
width: 550px !important; # Set the width to your desired value
|
||||
}
|
||||
</style>
|
||||
""",
|
||||
unsafe_allow_html=True,
|
||||
)
|
||||
s3_obj = s3_client.get_object(Bucket = client_bucket, Key = file_name)
|
||||
data=s3_obj['Body'].read()
|
||||
pdf_viewer(data, width=1500)
|
||||
st.error('This batch_id is empty.')
|
||||
|
||||
# if st.button("Show PDF"):
|
||||
# if file_name == None or file_name == "All":
|
||||
# st.error("Choose one specific file.")
|
||||
# else:
|
||||
# with st.sidebar:
|
||||
# with open(file_name, "rb") as f:
|
||||
# base64_pdf = base64.b64encode(f.read()).decode('utf-8')
|
||||
# # Embedding PDF in HTML
|
||||
# pdf_display = F'<iframe src="data:application/pdf;base64,{base64_pdf}" width="500" height="1000" type="application/pdf"></iframe>'
|
||||
# # Displaying File
|
||||
# st.markdown(
|
||||
# """
|
||||
# <style>
|
||||
# section[data-testid="stSidebar"] {
|
||||
# width: 600px !important; # Set the width to your desired value
|
||||
# }
|
||||
# </style>
|
||||
# """,
|
||||
# unsafe_allow_html=True,
|
||||
# )
|
||||
# st.markdown(pdf_display, unsafe_allow_html=True)
|
||||
contract_list = sorted(file_list)
|
||||
|
||||
df2 = pd.read_csv('temp2.csv')
|
||||
df2['Imputed Value'] = ''
|
||||
edited_df = st.data_editor(df2)
|
||||
file_row = st.columns([0.2, 0.7, 0.1])
|
||||
with file_row[0]:
|
||||
st.write("**Contract Name**")
|
||||
with file_row[1]:
|
||||
file_name = st.selectbox('Select a file', ['All'] + contract_list, label_visibility = "collapsed", index= None)
|
||||
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
field_row = st.columns([0.2, 0.7, 0.1])
|
||||
with field_row[0]:
|
||||
st.write("**Field Group**")
|
||||
with field_row[1]:
|
||||
field_group = st.selectbox('Field Group',('Unique Key', 'Contract Related', 'Pricing Before Carveouts - I'
|
||||
, 'Pricing Before Carveouts - II', 'Carveout Indicator, Code Type and Code #s - I'
|
||||
, 'Carveout Indicator, Code Type and Code #s - II', 'Carveout Indicator, Code Type and Code #s - III'
|
||||
, 'Optimize Carving Indic.', 'Carveout Method - I', 'Carveout Method - II', 'Provider'
|
||||
, 'Timeline'), label_visibility = "collapsed", index = None)
|
||||
|
||||
csv = convert_df(edited_df)
|
||||
if field_group == 'Unique Key':
|
||||
fields = fields[fields['PRIORITY'] == 'A']
|
||||
elif field_group == 'Contract Related':
|
||||
fields = fields[fields['PRIORITY'] == 'C']
|
||||
elif field_group == 'Pricing Before Carveouts - I':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[0]
|
||||
elif field_group == 'Pricing Before Carveouts - II':
|
||||
fields = fields[fields['PRIORITY'] == 'B']
|
||||
fields = np.array_split(fields, 2)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - I':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[0]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - II':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[1]
|
||||
elif field_group == 'Carveout Indicator, Code Type and Code #s - III':
|
||||
fields = fields[fields['PRIORITY'] == 'F']
|
||||
fields = np.array_split(fields, 3)[2]
|
||||
elif field_group == 'Carveout Methodology - I':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[0]
|
||||
elif field_group == 'Carveout Methodology - II':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[1]
|
||||
elif field_group == 'Carveout Method - III':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[2]
|
||||
elif field_group == 'Carveout Method - IV':
|
||||
fields = fields[fields['PRIORITY'] == 'G']
|
||||
fields = np.array_split(fields, 4)[3]
|
||||
elif field_group == 'Provider':
|
||||
fields = fields[fields['PRIORITY'] == 'D']
|
||||
elif field_group == 'Timeline':
|
||||
fields = fields[fields['PRIORITY'] == 'E']
|
||||
|
||||
buttons = st.columns(3)
|
||||
with buttons[0]:
|
||||
# st.button("Save All Imputations")
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
# st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
st.write("")
|
||||
with buttons[2]:
|
||||
if st.button("Kickoff Database Integration"):
|
||||
st.write("Stored in DB")
|
||||
if st.button("Show Results"):
|
||||
query = 'select * from "DOCZY_PIPELINE_RAW_OUTPUT"'
|
||||
cur.execute(query)
|
||||
df2 = pd.DataFrame.from_records(iter(cur), columns=[x[0] for x in cur.description])
|
||||
|
||||
# get this dataframe from snowflake table
|
||||
# df2 = pd.DataFrame(columns=['Contract Name','Field Name', 'SF_DB_COL_NAME', 'Snippet','Page Number'
|
||||
# , 'Field Extracted Value', 'Actual Value','Imputed Value'])
|
||||
df2.to_csv('temp2.csv', index=False)
|
||||
|
||||
if st.button("Show PDF"):
|
||||
if file_name == None or file_name == "All":
|
||||
st.error("Choose one specific file.")
|
||||
else:
|
||||
with st.sidebar:
|
||||
st.markdown(
|
||||
"""
|
||||
<style>
|
||||
section[data-testid="stSidebar"] {
|
||||
width: 550px !important; # Set the width to your desired value
|
||||
}
|
||||
</style>
|
||||
""",
|
||||
unsafe_allow_html=True,
|
||||
)
|
||||
s3_obj = s3_client.get_object(Bucket = client_bucket, Key = file_name)
|
||||
data=s3_obj['Body'].read()
|
||||
pdf_viewer(data, width=1500)
|
||||
|
||||
# if st.button("Show PDF"):
|
||||
# if file_name == None or file_name == "All":
|
||||
# st.error("Choose one specific file.")
|
||||
# else:
|
||||
# with st.sidebar:
|
||||
# with open(file_name, "rb") as f:
|
||||
# base64_pdf = base64.b64encode(f.read()).decode('utf-8')
|
||||
# # Embedding PDF in HTML
|
||||
# pdf_display = F'<iframe src="data:application/pdf;base64,{base64_pdf}" width="500" height="1000" type="application/pdf"></iframe>'
|
||||
# # Displaying File
|
||||
# st.markdown(
|
||||
# """
|
||||
# <style>
|
||||
# section[data-testid="stSidebar"] {
|
||||
# width: 600px !important; # Set the width to your desired value
|
||||
# }
|
||||
# </style>
|
||||
# """,
|
||||
# unsafe_allow_html=True,
|
||||
# )
|
||||
# st.markdown(pdf_display, unsafe_allow_html=True)
|
||||
|
||||
df2 = pd.read_csv('temp2.csv')
|
||||
df2['Imputed Value'] = ''
|
||||
edited_df = st.data_editor(df2)
|
||||
|
||||
@st.cache_data
|
||||
def convert_df(df):
|
||||
return df.to_csv(index=False).encode('utf-8')
|
||||
|
||||
csv = convert_df(edited_df)
|
||||
|
||||
buttons = st.columns(3)
|
||||
with buttons[0]:
|
||||
# st.button("Save All Imputations")
|
||||
st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
with buttons[1]:
|
||||
# st.download_button("Download Table", csv, "file.csv", "text/csv", key='download-csv')
|
||||
st.write("")
|
||||
with buttons[2]:
|
||||
if st.button("Kickoff Database Integration"):
|
||||
st.write("Stored in DB")
|
||||
Reference in New Issue
Block a user