import os os.environ["CUDA_VISIBLE_DEVICES"] = "-1" os.environ["TF_CPP_MIN_LOG_LEVEL"] = "2" import numpy as np import librosa import scipy.signal from scipy.stats import kurtosis import gradio as gr import tensorflow as tf import zipfile import tempfile import warnings warnings.filterwarnings('ignore', category=UserWarning) MAX_TIME_FRAMES = 313 N_MELS = 128 NUM_CLASSES = 6 LABEL_MAP_INVERSE = { 0: 'Machine 1_Normal', 1: 'Machine 1_Abnormal', 2: 'Machine 2_Normal', 3: 'Machine 2_Abnormal', 4: 'Machine 3_Normal', 5: 'Machine 3_Abnormal' } # ========================================== # PREPROCESSOR # ========================================== class MachineListenerPreprocessor: def __init__(self, target_sr=16000, n_fft=2048, hop_length=512, n_mels=N_MELS, n_mfcc=20): self.sr = target_sr self.n_fft = n_fft self.hop_length = hop_length self.n_mels = n_mels self.n_mfcc = n_mfcc def _apply_highpass_filter(self, y, cutoff=60.0): nyquist = 0.5 * self.sr normal_cutoff = cutoff / nyquist if normal_cutoff >= 1.0: return y b, a = scipy.signal.butter(4, normal_cutoff, btype='high', analog=False) return scipy.signal.filtfilt(b, a, y) def _truncate_silence(self, y, top_db=25): y_trimmed, _ = librosa.effects.trim(y, top_db=top_db, frame_length=self.n_fft, hop_length=self.hop_length) return y_trimmed def _mean_variance_normalize(self, y): return (y - np.mean(y)) / (np.std(y) + 1e-8) def process_audio(self, file_path): y, _ = librosa.load(file_path, sr=self.sr) y = self._apply_highpass_filter(y) y = self._truncate_silence(y, top_db=25) if len(y) == 0: raise ValueError(f"Silence only: {file_path}") y = self._mean_variance_normalize(y) mel_spec = librosa.feature.melspectrogram( y=y, sr=self.sr, n_fft=self.n_fft, hop_length=self.hop_length, n_mels=self.n_mels ) log_mel_spec = librosa.power_to_db(mel_spec, ref=np.max) return log_mel_spec def pad_or_truncate(spectrogram, max_frames): if spectrogram.shape[1] > max_frames: return spectrogram[:, :max_frames] elif spectrogram.shape[1] < max_frames: pad_width = max_frames - spectrogram.shape[1] return np.pad(spectrogram, pad_width=((0, 0), (0, pad_width)), mode='constant') return spectrogram # ========================================== # CUSTOM LAYERS (replace Lambda layers from training) # ========================================== class NormLayer(tf.keras.layers.Layer): """Replaces Lambda(lambda t: t / 80.0) used for spectrogram normalization.""" def call(self, x): return x / 80.0 class FreqReduceLayer(tf.keras.layers.Layer): """Replaces Lambda(lambda t: tf.reduce_mean(t, axis=1)) used to collapse frequency.""" def call(self, x): return tf.reduce_mean(x, axis=1) class SpecAugmentLayer(tf.keras.layers.Layer): """Training-only augmentation. Passes through during inference.""" def __init__(self, freq_mask_param=15, time_mask_param=30, **kwargs): super().__init__(**kwargs) self.freq_mask_param = freq_mask_param self.time_mask_param = time_mask_param def call(self, inputs, training=None): if not training: return inputs freq_max = tf.shape(inputs)[1] time_max = tf.shape(inputs)[2] f = tf.random.uniform([], minval=0, maxval=self.freq_mask_param, dtype=tf.int32) f0 = tf.random.uniform([], minval=0, maxval=freq_max - f, dtype=tf.int32) freq_indices = tf.range(freq_max) freq_mask = tf.logical_or(freq_indices < f0, freq_indices >= f0 + f) freq_mask = tf.cast(freq_mask, inputs.dtype) freq_mask = tf.reshape(freq_mask, [1, -1, 1, 1]) inputs = inputs * freq_mask t = tf.random.uniform([], minval=0, maxval=self.time_mask_param, dtype=tf.int32) t0 = tf.random.uniform([], minval=0, maxval=time_max - t, dtype=tf.int32) time_indices = tf.range(time_max) time_mask = tf.logical_or(time_indices < t0, time_indices >= t0 + t) time_mask = tf.cast(time_mask, inputs.dtype) time_mask = tf.reshape(time_mask, [1, 1, -1, 1]) inputs = inputs * time_mask return inputs def get_config(self): config = super().get_config() config.update({ "freq_mask_param": self.freq_mask_param, "time_mask_param": self.time_mask_param, }) return config # ========================================== # MODEL ARCHITECTURE (exact replica of V2-F) # ========================================== def se_block(x, filters, ratio=8): se = tf.keras.layers.GlobalAveragePooling2D()(x) se = tf.keras.layers.Dense(filters // ratio, activation='relu')(se) se = tf.keras.layers.Dense(filters, activation='sigmoid')(se) se = tf.keras.layers.Reshape([1, 1, filters])(se) return x * se def build_v2f(): inp = tf.keras.Input(shape=(N_MELS, MAX_TIME_FRAMES, 1)) x = SpecAugmentLayer(freq_mask_param=15, time_mask_param=30)(inp) x = NormLayer()(x) # Block 1 x = tf.keras.layers.Conv2D(32, (3, 3), padding='same', kernel_regularizer=tf.keras.regularizers.l2(1e-4))(x) x = tf.keras.layers.BatchNormalization()(x) x = tf.keras.layers.Activation('relu')(x) x = se_block(x, 32) x = tf.keras.layers.MaxPooling2D(pool_size=(2, 1))(x) x = tf.keras.layers.Dropout(0.2)(x) # Block 2 x = tf.keras.layers.Conv2D(64, (3, 3), padding='same', kernel_regularizer=tf.keras.regularizers.l2(1e-4))(x) x = tf.keras.layers.BatchNormalization()(x) x = tf.keras.layers.Activation('relu')(x) x = se_block(x, 64) x = tf.keras.layers.MaxPooling2D(pool_size=(2, 1))(x) x = tf.keras.layers.Dropout(0.2)(x) # Block 3 x = tf.keras.layers.Conv2D(128, (3, 3), padding='same', kernel_regularizer=tf.keras.regularizers.l2(1e-4))(x) x = tf.keras.layers.BatchNormalization()(x) x = tf.keras.layers.Activation('relu')(x) x = se_block(x, 128) x = tf.keras.layers.MaxPooling2D(pool_size=(2, 1))(x) x = tf.keras.layers.Dropout(0.2)(x) # Bridge: reduce channels and collapse frequency x = tf.keras.layers.Conv2D(64, (1, 1), padding='same', activation='relu')(x) x = FreqReduceLayer()(x) # BiLSTM x = tf.keras.layers.Bidirectional( tf.keras.layers.LSTM(64, return_sequences=True) )(x) x = tf.keras.layers.Dropout(0.3)(x) # MultiHead Attention x = tf.keras.layers.MaxPooling1D(pool_size=4)(x) x = tf.keras.layers.MultiHeadAttention(num_heads=4, key_dim=32)(x, x) x = tf.keras.layers.GlobalAveragePooling1D()(x) x = tf.keras.layers.Dropout(0.4)(x) out = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x) return tf.keras.Model(inputs=inp, outputs=out) # ========================================== # LOAD MODEL (rebuild + weights only) # ========================================== def load_model(): """ Rebuild the V2-F architecture in code, then load ONLY the weights from the .keras file. This completely bypasses Lambda deserialization and Python bytecode compatibility issues. """ keras_path = 'best_v2f_generalist.keras' try: model = build_v2f() # .keras file is a zip; extract the weights h5 and load with tempfile.TemporaryDirectory() as tmpdir: with zipfile.ZipFile(keras_path, 'r') as z: z.extract('model.weights.h5', tmpdir) model.load_weights(os.path.join(tmpdir, 'model.weights.h5')) print("Model rebuilt and weights loaded successfully.") return model, "" except Exception as e: return None, str(e) # ========================================== # STARTUP # ========================================== model, model_error = load_model() if model is None: print(f"Warning: Could not load model. Error: {model_error}") preprocessor = MachineListenerPreprocessor() # ========================================== # PREDICTION # ========================================== def predict(audio_filepath): if model is None: return f"Model not loaded properly. Error: {model_error}" if audio_filepath is None: return "Please upload an audio file." try: log_mel_spec = preprocessor.process_audio(audio_filepath) spec = pad_or_truncate(log_mel_spec, MAX_TIME_FRAMES) # Shape: (1, 128, 313, 1) — single channel spectrogram spec_batch = spec[np.newaxis, ..., np.newaxis].astype(np.float32) predictions = model(spec_batch, training=False).numpy() predicted_class_idx = np.argmax(predictions, axis=-1)[0] predicted_label = LABEL_MAP_INVERSE.get(predicted_class_idx, "Unknown") confidence = float(np.max(predictions)) return f"Prediction: {predicted_label} (Confidence: {confidence:.2f})" except Exception as e: return f"Error processing file: {str(e)}" # ========================================== # GRADIO UI # ========================================== iface = gr.Interface( fn=predict, inputs=gr.Audio(type="filepath", label="Upload Machine Audio"), outputs="text", title="Machine Listener Diagnosis", description="Upload a sound from a machine to predict whether it is Normal or Abnormal." ) if __name__ == "__main__": iface.launch()