SmartWareHouseAI / build_artifacts.py
Pro-Coder's picture
Upload 34 files
8f9afd8 verified
Raw
History Blame Contribute Delete
15.5 kB
"""
build_artifacts.py
-------------------
One-shot build script: generates synthetic datasets, trains the intent
classifier and the anomaly detector, evaluates the retrieval pipeline, and
saves every model/plot/metric the app needs to `models/`, `data/`, and
`assets/`. Run this once locally (or in CI) before deploying -- the Gradio
app itself only *loads* these pre-built artifacts, so the Space starts up
in a couple of seconds instead of retraining on every boot.
Usage:
python build_artifacts.py
"""
import json
import os
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
import numpy as np
from sklearn.metrics import (
accuracy_score,
classification_report,
confusion_matrix,
f1_score,
precision_score,
recall_score,
roc_auc_score,
roc_curve,
)
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from src.data_generation import (
RETRIEVAL_EVAL_SET,
generate_intent_dataset,
generate_inventory_db,
generate_orders_db,
generate_sensor_dataset,
)
from src.intent_model import build_pipeline, save_pipeline
from src.anomaly_model import FEATURES, build_model as build_anomaly_model, save_artifacts as save_anomaly_artifacts
from src.retriever import KBRetriever
ROOT = os.path.dirname(os.path.abspath(__file__))
MODELS_DIR = os.path.join(ROOT, "models")
DATA_DIR = os.path.join(ROOT, "data")
ASSETS_DIR = os.path.join(ROOT, "assets")
for d in (MODELS_DIR, DATA_DIR, ASSETS_DIR):
os.makedirs(d, exist_ok=True)
SEED = 42
def build_intent_classifier():
print("== Intent classifier ==")
df = generate_intent_dataset(n_per_intent=60, seed=SEED)
df.to_csv(os.path.join(DATA_DIR, "intent_dataset.csv"), index=False)
X_train, X_test, y_train, y_test = train_test_split(
df["text"], df["intent"], test_size=0.25, random_state=SEED, stratify=df["intent"]
)
pipeline = build_pipeline()
pipeline.fit(X_train, y_train)
y_pred = pipeline.predict(X_test)
acc = accuracy_score(y_test, y_pred)
macro_f1 = f1_score(y_test, y_pred, average="macro")
report = classification_report(y_test, y_pred, output_dict=True)
labels = sorted(df["intent"].unique())
cm = confusion_matrix(y_test, y_pred, labels=labels)
print(f"accuracy={acc:.4f} macro_f1={macro_f1:.4f}")
# Confusion matrix plot
fig, ax = plt.subplots(figsize=(7.5, 6.5))
im = ax.imshow(cm, cmap="Blues")
ax.set_xticks(range(len(labels)))
ax.set_yticks(range(len(labels)))
ax.set_xticklabels(labels, rotation=45, ha="right", fontsize=8)
ax.set_yticklabels(labels, fontsize=8)
ax.set_xlabel("Predicted intent")
ax.set_ylabel("True intent")
ax.set_title(f"Intent Classifier Confusion Matrix (acc={acc:.1%})")
for i in range(len(labels)):
for j in range(len(labels)):
ax.text(j, i, cm[i, j], ha="center", va="center",
color="white" if cm[i, j] > cm.max() / 2 else "black", fontsize=8)
fig.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "intent_confusion_matrix.png"), dpi=150)
plt.close(fig)
# Retrain on FULL data for the deployed model (more data = better generalisation)
pipeline_full = build_pipeline()
pipeline_full.fit(df["text"], df["intent"])
save_pipeline(pipeline_full, os.path.join(MODELS_DIR, "intent_pipeline.joblib"))
# Per-class precision/recall/F1 bar chart (clearer at a glance than the table alone)
fig, ax = plt.subplots(figsize=(9, 5))
x = np.arange(len(labels))
width = 0.25
precisions = [report[l]["precision"] for l in labels]
recalls = [report[l]["recall"] for l in labels]
f1s = [report[l]["f1-score"] for l in labels]
ax.bar(x - width, precisions, width, label="Precision", color="#3b82f6")
ax.bar(x, recalls, width, label="Recall", color="#10b981")
ax.bar(x + width, f1s, width, label="F1", color="#f59e0b")
ax.set_xticks(x)
ax.set_xticklabels(labels, rotation=35, ha="right", fontsize=8)
ax.set_ylim(0, 1.15)
ax.set_ylabel("Score")
ax.set_title("Intent Classifier: Per-Class Precision / Recall / F1")
ax.legend(loc="lower right", ncol=3)
ax.grid(axis="y", alpha=0.3)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "intent_per_class_bar.png"), dpi=150)
plt.close(fig)
metrics = {
"accuracy": acc,
"macro_f1": macro_f1,
"n_train": len(X_train),
"n_test": len(X_test),
"n_classes": len(labels),
"classes": labels,
"classification_report": report,
}
with open(os.path.join(DATA_DIR, "intent_eval.json"), "w") as f:
json.dump(metrics, f, indent=2)
# Dataset composition chart (helps a reader understand what the model was trained on)
counts = df["intent"].value_counts().reindex(labels)
fig, ax = plt.subplots(figsize=(8, 4.5))
ax.barh(labels, counts.values, color="#6366f1")
ax.set_xlabel("Number of examples")
ax.set_title(f"Intent Dataset Composition (n={len(df)}, synthetic, templated)")
ax.grid(axis="x", alpha=0.3)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "intent_dataset_composition.png"), dpi=150)
plt.close(fig)
return metrics
def build_anomaly_detector():
print("== Anomaly detector ==")
df = generate_sensor_dataset(n_normal=900, n_anomaly=100, seed=SEED)
df.to_csv(os.path.join(DATA_DIR, "sensor_dataset.csv"), index=False)
X = df[FEATURES].values
y = df["label"].values # ground truth, used only for evaluation (model itself is unsupervised)
X_train, X_test, y_train, y_test = train_test_split(
X, y, test_size=0.3, random_state=SEED, stratify=y
)
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
X_test_scaled = scaler.transform(X_test)
# Contamination set close to the true training-set anomaly rate
contamination = float(np.clip(y_train.mean(), 0.01, 0.4))
model = build_anomaly_model(contamination=contamination, seed=SEED)
model.fit(X_train_scaled)
raw_scores = model.decision_function(X_test_scaled) # higher = more normal
anomaly_scores = 0.5 - raw_scores # higher = more anomalous
preds = model.predict(X_test_scaled)
preds_binary = (preds == -1).astype(int)
precision = precision_score(y_test, preds_binary, zero_division=0)
recall = recall_score(y_test, preds_binary, zero_division=0)
f1 = f1_score(y_test, preds_binary, zero_division=0)
try:
roc_auc = roc_auc_score(y_test, anomaly_scores)
except ValueError:
roc_auc = float("nan")
acc = accuracy_score(y_test, preds_binary)
cm = confusion_matrix(y_test, preds_binary)
print(f"precision={precision:.4f} recall={recall:.4f} f1={f1:.4f} roc_auc={roc_auc:.4f}")
# Confusion matrix plot
fig, ax = plt.subplots(figsize=(4.5, 4))
im = ax.imshow(cm, cmap="Oranges")
ax.set_xticks([0, 1]); ax.set_yticks([0, 1])
ax.set_xticklabels(["Normal", "Anomaly"])
ax.set_yticklabels(["Normal", "Anomaly"])
ax.set_xlabel("Predicted"); ax.set_ylabel("Actual")
ax.set_title(f"Anomaly Detector Confusion Matrix\n(F1={f1:.2f})")
for i in range(2):
for j in range(2):
ax.text(j, i, cm[i, j], ha="center", va="center",
color="white" if cm[i, j] > cm.max() / 2 else "black")
fig.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "anomaly_confusion_matrix.png"), dpi=150)
plt.close(fig)
# ROC curve plot
fpr, tpr, _ = roc_curve(y_test, anomaly_scores)
fig, ax = plt.subplots(figsize=(5, 4.5))
ax.plot(fpr, tpr, label=f"ROC-AUC = {roc_auc:.3f}", color="#2563eb", linewidth=2)
ax.plot([0, 1], [0, 1], linestyle="--", color="gray", linewidth=1)
ax.set_xlabel("False Positive Rate")
ax.set_ylabel("True Positive Rate")
ax.set_title("Anomaly Detector ROC Curve")
ax.legend(loc="lower right")
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "anomaly_roc_curve.png"), dpi=150)
plt.close(fig)
# Retrain on full data for the deployed model
scaler_full = StandardScaler()
X_full_scaled = scaler_full.fit_transform(X)
contamination_full = float(np.clip(y.mean(), 0.01, 0.4))
model_full = build_anomaly_model(contamination=contamination_full, seed=SEED)
model_full.fit(X_full_scaled)
save_anomaly_artifacts(
model_full, scaler_full,
os.path.join(MODELS_DIR, "anomaly_iforest.joblib"),
os.path.join(MODELS_DIR, "anomaly_scaler.joblib"),
)
metrics = {
"precision": precision,
"recall": recall,
"f1": f1,
"roc_auc": roc_auc,
"accuracy": acc,
"n_test": len(y_test),
"test_anomaly_rate": float(y_test.mean()),
"contamination_used": contamination,
}
with open(os.path.join(DATA_DIR, "anomaly_eval.json"), "w") as f:
json.dump(metrics, f, indent=2)
# Metrics bar chart
fig, ax = plt.subplots(figsize=(6.5, 4.5))
metric_names = ["Precision", "Recall", "F1", "ROC-AUC", "Accuracy"]
metric_vals = [precision, recall, f1, roc_auc, acc]
bars = ax.bar(metric_names, metric_vals, color=["#3b82f6", "#10b981", "#f59e0b", "#8b5cf6", "#ef4444"])
ax.set_ylim(0, 1.15)
ax.set_ylabel("Score")
ax.set_title("Anomaly Detector: Evaluation Metrics")
ax.grid(axis="y", alpha=0.3)
for bar, val in zip(bars, metric_vals):
ax.text(bar.get_x() + bar.get_width() / 2, val + 0.03, f"{val:.2f}", ha="center", fontsize=9)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "anomaly_metrics_bar.png"), dpi=150)
plt.close(fig)
# Sensor feature distributions: normal vs anomaly (helps a reader see *why*
# the model flags what it flags -- directly supports the Predictive
# Maintenance tab's sliders)
fig, axes = plt.subplots(2, 2, figsize=(10, 7))
titles = {
"motor_temp_c": "Motor Temperature (°C)",
"vibration_mm_s": "Vibration (mm/s)",
"current_amps": "Motor Current (A)",
"belt_speed_mps": "Belt Speed (m/s)",
}
for ax, feat in zip(axes.flat, FEATURES):
normal_vals = df.loc[df["label"] == 0, feat]
anomaly_vals = df.loc[df["label"] == 1, feat]
ax.hist(normal_vals, bins=25, alpha=0.6, label="Normal", color="#10b981")
ax.hist(anomaly_vals, bins=25, alpha=0.6, label="Anomaly", color="#ef4444")
ax.set_title(titles[feat], fontsize=10)
ax.legend(fontsize=8)
ax.grid(alpha=0.3)
fig.suptitle("Sensor Feature Distributions: Normal vs. Anomaly (synthetic training data)", fontsize=11)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "sensor_distributions.png"), dpi=150)
plt.close(fig)
return metrics
def build_retrieval_eval():
print("== Retrieval (RAG) evaluation ==")
retriever = KBRetriever()
hits_at_1, hits_at_2 = 0, 0
rows = []
for query, expected_id in RETRIEVAL_EVAL_SET:
results = retriever.retrieve(query, k=2)
top_ids = [r.id for r in results]
hit1 = top_ids[0] == expected_id
hit2 = expected_id in top_ids
hits_at_1 += int(hit1)
hits_at_2 += int(hit2)
rows.append({
"query": query,
"expected": expected_id,
"retrieved_top1": top_ids[0],
"hit@1": hit1,
"hit@2": hit2,
"top1_score": round(results[0].score, 4),
})
n = len(RETRIEVAL_EVAL_SET)
metrics = {
"hit_rate_at_1": hits_at_1 / n,
"hit_rate_at_2": hits_at_2 / n,
"n_queries": n,
"rows": rows,
}
print(f"hit@1={metrics['hit_rate_at_1']:.2f} hit@2={metrics['hit_rate_at_2']:.2f}")
with open(os.path.join(DATA_DIR, "retrieval_eval.json"), "w") as f:
json.dump(metrics, f, indent=2)
fig, ax = plt.subplots(figsize=(4.5, 4))
bars = ax.bar(["Hit Rate @ 1", "Hit Rate @ 2"],
[metrics["hit_rate_at_1"], metrics["hit_rate_at_2"]],
color=["#3b82f6", "#10b981"])
ax.set_ylim(0, 1.15)
ax.set_ylabel("Hit rate")
ax.set_title(f"RAG Retriever Hit Rate (n={n} labelled queries)")
ax.grid(axis="y", alpha=0.3)
for bar, val in zip(bars, [metrics["hit_rate_at_1"], metrics["hit_rate_at_2"]]):
ax.text(bar.get_x() + bar.get_width() / 2, val + 0.03, f"{val:.0%}", ha="center", fontsize=10)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "retrieval_hitrate_bar.png"), dpi=150)
plt.close(fig)
return metrics
def build_inventory_and_orders():
print("== Inventory & Orders synthetic DB ==")
inv = generate_inventory_db(seed=SEED)
orders = generate_orders_db(seed=SEED)
inv.to_csv(os.path.join(DATA_DIR, "inventory.csv"), index=False)
orders.to_csv(os.path.join(DATA_DIR, "orders.csv"), index=False)
print(f"inventory rows={len(inv)} orders rows={len(orders)}")
def build_latency_benchmark(intent_metrics, anomaly_metrics):
print("== Latency benchmark ==")
import time
from src.intent_model import load_pipeline, predict as intent_predict
from src.anomaly_model import load_artifacts, score_reading
pipeline = load_pipeline(os.path.join(MODELS_DIR, "intent_pipeline.joblib"))
model, scaler = load_artifacts(
os.path.join(MODELS_DIR, "anomaly_iforest.joblib"),
os.path.join(MODELS_DIR, "anomaly_scaler.joblib"),
)
retriever = KBRetriever()
sample_query = "The conveyor belt in Zone C is making noise"
sample_reading = {"motor_temp_c": 82.0, "vibration_mm_s": 6.1, "current_amps": 20.5, "belt_speed_mps": 0.7}
def timeit(fn, n=50):
start = time.perf_counter()
for _ in range(n):
fn()
return (time.perf_counter() - start) / n * 1000 # ms
intent_ms = timeit(lambda: intent_predict(pipeline, sample_query))
anomaly_ms = timeit(lambda: score_reading(model, scaler, sample_reading))
retrieval_ms = timeit(lambda: retriever.retrieve(sample_query, k=2))
latency = {
"intent_classifier_ms": round(intent_ms, 3),
"anomaly_detector_ms": round(anomaly_ms, 3),
"kb_retrieval_ms": round(retrieval_ms, 3),
"note": "LLM generation latency depends on the external Inference API "
"call and is measured live in the app, not benchmarked here.",
}
with open(os.path.join(DATA_DIR, "latency_eval.json"), "w") as f:
json.dump(latency, f, indent=2)
print(latency)
fig, ax = plt.subplots(figsize=(6, 4))
components = ["Intent\nclassifier", "Anomaly\ndetector", "KB\nretrieval"]
values = [intent_ms, anomaly_ms, retrieval_ms]
bars = ax.bar(components, values, color=["#3b82f6", "#f59e0b", "#10b981"])
ax.set_ylabel("Latency (ms, avg of 50 runs)")
ax.set_title("Local Component Latency (CPU)")
ax.grid(axis="y", alpha=0.3)
for bar, val in zip(bars, values):
ax.text(bar.get_x() + bar.get_width() / 2, val, f"{val:.2f} ms", ha="center", va="bottom", fontsize=9)
fig.tight_layout()
fig.savefig(os.path.join(ASSETS_DIR, "latency_bar.png"), dpi=150)
plt.close(fig)
if __name__ == "__main__":
intent_metrics = build_intent_classifier()
anomaly_metrics = build_anomaly_detector()
build_retrieval_eval()
build_inventory_and_orders()
build_latency_benchmark(intent_metrics, anomaly_metrics)
print("\nAll artifacts built successfully.")