| import streamlit as st
|
| import pandas as pd
|
| import numpy as np
|
| import joblib
|
|
|
|
|
| st.set_page_config(
|
| page_title='Cancer Risk Predictor',
|
| layout='wide',
|
| initial_sidebar_state='collapsed'
|
| )
|
|
|
| @st.cache_resource
|
| def load_artifacts():
|
| model = joblib.load('Models/model_xgb_new.pkl')
|
| le = joblib.load('Models/Label_encoder.pkl')
|
| feature_name = joblib.load('Models/Feature_names.pkl')
|
| return model, le, feature_name
|
|
|
| model, le, FEATURE_NAMES = load_artifacts()
|
|
|
|
|
| st.title("🧬 Cancer Risk Level Predictor")
|
| st.markdown("Assess patient risk levels using advanced machine learning diagnostics. Choose a processing mode below.")
|
|
|
|
|
| tab_batch, tab_manual = st.tabs(["📁 Batch Processing (CSV Upload)", "👤 Individual Patient Intake"])
|
|
|
| def preprocess_input(df):
|
| missing = [c for c in FEATURE_NAMES if c not in df.columns]
|
| if missing:
|
| st.warning(f"⚠️ Missing Columns in input - Filling {len(missing)} fields with zero : {missing}")
|
| for c in missing:
|
| df[c] = 0
|
| df = df[FEATURE_NAMES].copy()
|
| df = df.apply(pd.to_numeric, errors='coerce').fillna(0)
|
| return df
|
|
|
|
|
|
|
|
|
| with tab_batch:
|
| st.subheader("Batch Dataset Analysis")
|
| st.markdown("Upload a patient population roster to parse risk profiles simultaneously.")
|
|
|
| uploaded_file = st.file_uploader("Drop your dataset here", type=['csv'], label_visibility="collapsed")
|
|
|
| if uploaded_file is not None:
|
| input_df = pd.read_csv(uploaded_file)
|
|
|
| with st.spinner("Processing framework algorithms..."):
|
| x = preprocess_input(input_df)
|
| pred_enc = model.predict(x)
|
| probs = model.predict_proba(x)
|
| preds = le.inverse_transform(pred_enc)
|
|
|
| result = x.copy()
|
| result['Predicted_Risk_Level'] = preds
|
|
|
| for i, cls in enumerate(le.classes_):
|
| result[f'prob_{cls}'] = probs[:, i]
|
|
|
| st.success("✅ Analytics Engine Execution Complete")
|
|
|
|
|
| st.dataframe(result, use_container_width=True)
|
|
|
| st.download_button(
|
| label="📥 Download Structured Predictions CSV",
|
| data=result.to_csv(index=False),
|
| file_name='predictions.csv',
|
| mime='text/csv',
|
| type='secondary'
|
| )
|
|
|
|
|
|
|
|
|
| with tab_manual:
|
| st.subheader("Manual Clinical Intake Form")
|
|
|
|
|
| with st.container(border=True):
|
| st.markdown("#### 🩺 Demographics & Administrative Meta")
|
| c_meta1, c_meta2, c_meta3, c_meta4 = st.columns(4)
|
| input_data = {}
|
|
|
| with c_meta1:
|
| input_data['Patient_ID'] = st.text_input("Patient Identification ID", value="LU0005")
|
| with c_meta2:
|
| input_data['Cancer_Type'] = st.selectbox("Suspected Target System", ['Breast', 'Prostate', 'Skin', 'Colon', 'Lung'])
|
| with c_meta3:
|
| input_data['Age'] = st.number_input("Biological Age (Years)", min_value=0, max_value=120, value=50)
|
| with c_meta4:
|
| input_data['Gender'] = st.selectbox("Biological Sex Assigned", [0, 1], format_func=lambda x: "Female" if x == 0 else "Male")
|
|
|
| with st.container(border=True):
|
| st.markdown("#### ⚖️ Physiological Vitals & Laboratory Scoring")
|
| c_phys1, c_phys2, c_phys3 = st.columns(3)
|
|
|
| with c_phys1:
|
| input_data['BMI'] = st.number_input("Body Mass Index (BMI)", min_value=10.0, max_value=60.0, value=25.0, step=0.1)
|
| with c_phys2:
|
| input_data['Overall_Risk_Score'] = st.number_input("Pre-evaluation Baseline Risk Score", min_value=0.0, max_value=1.0, value=0.4, step=0.01)
|
| with c_phys3:
|
| input_data['Calcium_Intake'] = st.slider("Dietary Calcium Scaling (0-10)", 0, 10, 5)
|
|
|
| with st.container(border=True):
|
| st.markdown("#### 🧬 Genetic & Pathogenic Pre-dispositions")
|
| c_gen1, c_gen2, c_gen3 = st.columns(3)
|
|
|
| with c_gen1:
|
| input_data['Family_History'] = st.selectbox("Family Oncological History", [0, 1], format_func=lambda x: "No History" if x == 0 else "Positive History")
|
| with c_gen2:
|
| input_data['BRCA_Mutation'] = st.selectbox("BRCA Variant Mutation Status", [0, 1], format_func=lambda x: "Negative / Normal" if x == 0 else "Positive Marker Present")
|
| with c_gen3:
|
| input_data['H_Pylori_Infection'] = st.selectbox("Active H. Pylori Infection Status", [0, 1], format_func=lambda x: "Negative" if x == 0 else "Positive Exposure")
|
|
|
| with st.container(border=True):
|
| st.markdown("#### 🚬 Behavioral Habits & Environmental Exposures")
|
| c_beh1, c_beh2 = st.columns(2)
|
|
|
| with c_beh1:
|
| input_data['Smoking'] = st.slider("Tobacco / Smoking Exposure Severity (0-10)", 0, 10, 0)
|
| input_data['Alcohol_Use'] = st.slider("Alcohol Consumption Volume Scale (0-10)", 0, 10, 0)
|
| input_data['Obesity'] = st.slider("Clinical Obesity Stratification Index (0-10)", 0, 10, 0)
|
| input_data['Physical_Activity'] = st.slider("Physical Activity Engagement Metric (0-10)", 0, 10, 5)
|
| with c_beh2:
|
| input_data['Diet_Red_Meat'] = st.slider("Red Meat Consumption Frequency (0-10)", 0, 10, 0)
|
| input_data['Diet_Salted_Processed'] = st.slider("Processed / Sodium Food Prevalence (0-10)", 0, 10, 0)
|
| input_data['Fruit_Veg_Intake'] = st.slider("Nutritional Fruit & Vegetable Density (0-10)", 0, 10, 5)
|
| input_data['Physical_Activity_Level'] = st.slider("Cardio / Physical Exertion Intensity Level (0-10)", 0, 10, 5)
|
|
|
| with st.container(border=True):
|
| st.markdown("#### 🏭 Ecological & Occupational Conditions")
|
| c_eco1, c_eco2 = st.columns(2)
|
| with c_eco1:
|
| input_data['Air_Pollution'] = st.slider("Particulate Air Pollution Hazard Scale (0-10)", 0, 10, 2)
|
| with c_eco2:
|
| input_data['Occupational_Hazards'] = st.slider("Workplace Carcinogen Exposure Scale (0-10)", 0, 10, 2)
|
|
|
|
|
| _, btn_col, _ = st.columns([2, 1, 2])
|
| with btn_col:
|
| predict_triggered = st.button("🔬 Compute Comprehensive Risk Profile", type="primary", use_container_width=True)
|
|
|
| if predict_triggered:
|
| x_single = pd.DataFrame([input_data])
|
| x_proc = preprocess_input(x_single)
|
|
|
| pred_enc = model.predict(x_proc)[0]
|
| probs = model.predict_proba(x_proc)[0]
|
| pred = le.inverse_transform([pred_enc])[0]
|
|
|
| st.markdown("---")
|
| st.subheader("📋 Core Diagnostic Output Results")
|
|
|
|
|
| c_res1, c_res2 = st.columns([2, 3])
|
|
|
| with c_res1:
|
| st.metric(label="Calculated Stratified Risk Target Level", value=f"{pred}")
|
|
|
|
|
| high_prob_val = probs[list(le.classes_).index('High')] if 'High' in le.classes_ else 0.0
|
|
|
| st.metric(label="Absolute High-Risk Probability Node Index", value=f"{high_prob_val:.2%}")
|
|
|
| if high_prob_val >= 0.5:
|
| st.warning("⚠️ **ALERT:** Patient assessment metrics exceed the preset danger threshold parameters. Immediate triage or secondary clinical review is strongly advised.")
|
| else:
|
| st.success("✅ **STABLE:** Calculated biomarkers indicate the individual falls within normalized operating action baselines.")
|
|
|
| with c_res2:
|
| st.markdown("**Categorical Confidence Vector Matrix**")
|
| prob_df = pd.DataFrame({
|
| 'Risk Classification Stratum': list(le.classes_),
|
| 'Confidence Interval Probability': probs
|
| }).sort_values('Confidence Interval Probability', ascending=False).reset_index(drop=True)
|
|
|
|
|
| prob_df['Confidence Interval Probability'] = prob_df['Confidence Interval Probability'].map(lambda x: f"{x:.2%}")
|
| st.table(prob_df)
|
|
|
| st.markdown("---")
|
| st.caption("⚙️ Diagnostic Engine: Class-Weighted Extreme Gradient Boosting (XGBoost Architecture optimized via Optuna framework). Check your input array configurations against standard evaluation models before deploying to clinical trials.")
|
|
|