ML_assessment / app.py
Dave67350's picture
Update app.py
5fc67e1 verified
Raw
History Blame Contribute Delete
14.6 kB
import gradio as gr
import giskard
import pandas as pd
import tempfile
import traceback
import os
import joblib
import zipfile
import onnxruntime as ort
import torch
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from reportlab.lib.pagesizes import letter
from reportlab.pdfgen import canvas
from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, mean_squared_error, r2_score
from nltk.translate.bleu_score import sentence_bleu
try:
from transformers import pipeline, AutoModelForCausalLM, AutoTokenizer
except ImportError:
pipeline = None
try:
import tensorflow as tf
except ImportError:
tf = None
try:
import xgboost as xgb
except ImportError:
xgb = None
try:
import lightgbm as lgb
except ImportError:
lgb = None
def detect_model_type(model):
try:
# Sklearn classification
if hasattr(model, "predict_proba") and hasattr(model, "classes_"):
return "classification"
# Sklearn regression
if hasattr(model, "predict") and hasattr(model, "n_features_in_") and not hasattr(model, "classes_"):
return "regression"
# Text generation (Hugging Face pipelines or models)
if hasattr(model, "generate") or "transformers.pipelines" in str(type(model)):
return "text_generation"
# TensorFlow model
if tf and isinstance(model, tf.keras.Model):
return "tensorflow"
# XGBoost
if xgb and isinstance(model, xgb.Booster):
# Try to infer from attributes if possible (e.g., check for 'multi:softprob' param)
params = model.attributes() if hasattr(model, "attributes") else {}
booster_type = params.get("objective", "")
if "binary" in booster_type or "softprob" in booster_type:
return "classification"
else:
return "regression"
# LightGBM
if lgb and isinstance(model, lgb.Booster):
objective = model.params.get("objective", "")
if "binary" in objective or "multiclass" in objective:
return "classification"
else:
return "regression"
# ONNX
if isinstance(model, str) and model.endswith(".onnx"):
return "onnx"
except Exception as e:
print(f"[detect_model_type] Warning: {str(e)}")
return "unknown"
def create_giskard_model(model, model_type, name="", description=""):
feature_names = []
if hasattr(model, "feature_names_in_"):
feature_names = list(model.feature_names_in_)
elif hasattr(model, "n_features_in_"):
feature_names = [f"feature{i+1}" for i in range(model.n_features_in_)]
else:
feature_names = ["feature1", "feature2"]
if model_type == "text_generation":
return giskard.Model(
model=model,
model_type="text_generation",
name=name or "Text Generator",
description=description or "Text generation model",
feature_names=["text"]
)
elif model_type == "classification":
labels = list(getattr(model, "classes_", []))
if not labels:
raise ValueError("Cannot infer classification labels; model must have `classes_` attribute.")
def predict_proba_safe(df):
predictions = model.predict_proba(df.values)
return predictions.astype(np.float32)
return giskard.Model(
model=predict_proba_safe,
model_type="classification",
feature_names=feature_names,
classification_labels=labels
)
elif model_type == "regression":
def predict_safe(df):
predictions = model.predict(df.values)
return predictions.astype(np.float32)
return giskard.Model(
model=predict_safe,
model_type="regression",
feature_names=feature_names
)
elif model_type == "tensorflow":
return giskard.Model(
model=model.predict,
model_type="regression",
name=name,
description=description
)
elif model_type == "onnx":
def onnx_predict(input_data):
ort_session = ort.InferenceSession(model)
inputs = {ort_session.get_inputs()[0].name: input_data.to_numpy().astype('float32')}
result = ort_session.run(None, inputs)[0]
return result.astype(np.float32)
return giskard.Model(
model=onnx_predict,
model_type="regression",
feature_names=feature_names
)
else:
raise ValueError("Unsupported model type")
def extract_if_zip(path):
if zipfile.is_zipfile(path):
with zipfile.ZipFile(path, 'r') as zip_ref:
extracted_path = tempfile.mkdtemp()
zip_ref.extractall(extracted_path)
for file in os.listdir(extracted_path):
full_path = os.path.join(extracted_path, file)
if os.path.isfile(full_path):
return full_path
return path
def save_issues_pdf(issues_df, performance_metrics, deviation_info, output_path):
c = canvas.Canvas(output_path, pagesize=letter)
width, height = letter
text = c.beginText(40, height - 40)
text.setFont("Helvetica", 10)
text.textLine("Performance Metrics:")
for metric, value in performance_metrics.items():
text.textLine(f"{metric}: {value:.4f}" if isinstance(value, (int, float)) else f"{metric}: {value}")
text.textLine("\nDeviation from Expected Outcomes:")
for dev in deviation_info:
text.textLine(dev)
if 'bias' in issues_df.columns:
text.textLine("\nBias/Fairness Issues:")
bias_issues = issues_df[issues_df['bias'] == True]
if not bias_issues.empty:
for issue in bias_issues['issue_description']:
text.textLine(issue)
else:
text.textLine("No bias or fairness issues found.")
else:
text.textLine("No bias or fairness issues found.")
text.textLine("\nIssues Detected:")
if issues_df.empty:
text.textLine("No issues detected.")
else:
lines = issues_df.to_string().split('\n')
for line in lines:
text.textLine(line)
c.drawText(text)
c.save()
def assess_uploaded_model(model_file, name, description, manual_type):
try:
model = None
model_type = "unknown"
with tempfile.NamedTemporaryFile(delete=False) as tmp_file:
with open(model_file.name, "rb") as f:
tmp_file.write(f.read())
tmp_file_path = tmp_file.name
model_path = extract_if_zip(tmp_file_path)
try:
model = joblib.load(model_path)
model_type = detect_model_type(model)
except Exception:
pass
if model is None and tf is not None:
try:
model = tf.keras.models.load_model(model_path)
model_type = detect_model_type(model)
except Exception:
pass
if model is None and xgb is not None:
try:
model = xgb.Booster()
model.load_model(model_path)
model_type = detect_model_type(model)
except Exception:
pass
if model is None and lgb is not None:
try:
model = lgb.Booster(model_file=model_path)
model_type = detect_model_type(model)
except Exception:
pass
if model is None and model_path.endswith(".onnx"):
model = model_path
model_type = detect_model_type(model)
elif model is None and name.lower().startswith("gpt") and pipeline is not None:
tokenizer = AutoTokenizer.from_pretrained(name)
model = AutoModelForCausalLM.from_pretrained(name)
model = pipeline("text-generation", model=model, tokenizer=tokenizer)
model_type = detect_model_type(model)
if model is None:
return gr.update(value="❌ Error: Could not load model."), gr.update(value=""), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
giskard_model = create_giskard_model(model, model_type, name, description)
issues_df = pd.DataFrame()
if hasattr(model, "n_features_in_"):
n_features = model.n_features_in_
feature_names = [f"feature{i+1}" for i in range(n_features)]
elif hasattr(model, "feature_names_in_"):
feature_names = list(model.feature_names_in_)
else:
n_features = 3 # fallback default
feature_names = [f"feature{i+1}" for i in range(n_features)]
row_count = 100
rng = np.random.default_rng(seed=42)
df_base = pd.DataFrame(rng.normal(loc=0, scale=5, size=(row_count, len(feature_names))), columns=feature_names)
df_base.iloc[0, 0] = np.nan
df_base.iloc[1, 1] = 9999
df_base.iloc[2, 2] = -9999
df_scan = df_base.fillna(-999)
target = rng.integers(0, 2, size=row_count) if model_type == "classification" else rng.normal(size=row_count)
df_scan["__target__"] = target
column_types = {col: "numeric" for col in feature_names}
dataset = giskard.Dataset(df=df_scan, target="__target__", column_types=column_types)
try:
scan_results = giskard.scan(giskard_model, dataset)
issues_df = scan_results.to_dataframe()
except Exception as scan_error:
return gr.update(value=f"❌ Error: {str(scan_error)}"), gr.update(value=traceback.format_exc()), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
performance_metrics = {}
try:
if model_type in ["classification", "regression"]:
preds = model.predict(df_scan[feature_names])
if model_type == "classification":
if hasattr(model, "predict_proba"):
preds = np.argmax(model.predict_proba(df_scan[feature_names]), axis=1)
performance_metrics = {
"Accuracy": accuracy_score(target, preds),
"Precision": precision_score(target, preds, average="binary", zero_division=0),
"Recall": recall_score(target, preds, average="binary", zero_division=0),
"F1 Score": f1_score(target, preds, average="binary", zero_division=0)
}
elif model_type == "regression":
performance_metrics = {
"MSE": mean_squared_error(target, preds),
"R²": r2_score(target, preds)
}
elif model_type == "tensorflow":
preds = model.predict(df_scan[feature_names])
performance_metrics = {
"MSE": mean_squared_error(target, preds),
"R²": r2_score(target, preds)
}
elif model_type == "onnx":
ort_session = ort.InferenceSession(model_path)
input_name = ort_session.get_inputs()[0].name
onnx_input = df_scan[feature_names].to_numpy().astype("float32")
preds = ort_session.run(None, {input_name: onnx_input})[0]
performance_metrics = {
"MSE": mean_squared_error(target, preds),
"R²": r2_score(target, preds)
}
elif model_type == "text_generation":
reference = "This is a reference sentence."
input_text = "This is"
result = model(input_text)[0]['generated_text']
bleu_score = sentence_bleu([reference.split()], result.split())
performance_metrics = {
"BLEU Score": bleu_score
}
else:
performance_metrics = {
"Warning": "Unsupported model type for auto-metrics"
}
except Exception as e:
performance_metrics = {
"Error": f"Could not calculate metrics: {str(e)}"
}
deviation_info = ["Feature `feature1` has fail rate due to perturbation: 5%"]
pdf_path = os.path.join(tempfile.gettempdir(), "issues_report.pdf")
save_issues_pdf(issues_df, performance_metrics, deviation_info, pdf_path)
status_msg = f"✅ Scan complete.\nModel type: {model_type}.\n[Download PDF]({pdf_path})"
return (
gr.update(value=status_msg),
gr.update(value="Check PDF report"),
gr.update(value=pdf_path, visible=True),
gr.update(value=None, visible=False),
gr.update(value=None, visible=False)
)
except Exception as e:
return (
gr.update(value=f"❌ Error: {str(e)}"),
gr.update(value=traceback.format_exc()),
gr.update(visible=False),
gr.update(visible=False),
gr.update(visible=False)
)
def main():
with gr.Blocks() as demo:
gr.Markdown("## 🧠 Universal ML Model Assessment (with Giskard)")
gr.Markdown("Upload a `.pkl`, `.joblib`, `.onnx`, `.zip`, `.h5`, `.pb` or HuggingFace Transformers model (like `gpt2`).")
model_file = gr.File(label="Upload Model File", file_types=[".pkl", ".joblib", ".h5", ".pb", ".onnx", ".zip"])
name = gr.Textbox(label="Model Name (e.g. gpt2)")
description = gr.Textbox(label="Model Description")
manual_type = gr.Dropdown(label="Optional Model Type", choices=["", "classification", "regression", "text_generation", "tensorflow", "onnx"], value="")
assess_btn = gr.Button("🩺 Assess Model")
status = gr.Textbox(label="Status")
issues = gr.Textbox(label="Assessment Results", lines=10)
with gr.Row():
download_pdf = gr.File(label="📄 Download PDF Report", visible=False)
download_csv = gr.File(label="📊 Download CSV Report", visible=False)
download_plot = gr.File(label="🖼️ View Plot", visible=False)
assess_btn.click(
assess_uploaded_model,
inputs=[model_file, name, description, manual_type],
outputs=[status, issues, download_pdf, download_csv, download_plot]
)
demo.launch()
if __name__ == "__main__":
main()