| import gradio as gr | |
| import librosa | |
| import numpy as np | |
| import tensorflow as tf | |
| from tensorflow.keras.models import load_model | |
| # Load your pre-trained model (make sure it's in the same directory or provide a path) | |
| model = load_model('voice_authentication_model.keras') # Replace with your model's filename | |
| # Function to extract MFCC features and make a prediction | |
| def predict_user_or_non_user(audio): | |
| try: | |
| # Load the audio file using librosa | |
| y, sr = librosa.load(audio, sr=16000) # sr=None to keep the original sampling rate | |
| # Optional: Normalize the audio volume (if necessary) | |
| y = librosa.util.normalize(y) | |
| # Extract MFCC features from the audio | |
| mfccs = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13) # You can adjust n_mfcc as needed | |
| mfccs = np.mean(mfccs.T, axis=0) # Take the mean of MFCCs over time to reduce dimension | |
| # Reshape the MFCCs to match the input shape expected by the model | |
| mfccs = mfccs.reshape(1, -1) # Reshape to 1 sample, with the number of features | |
| # Predict the class (user or non-user) | |
| prediction = model.predict(mfccs) | |
| # Debugging: print raw model output | |
| print(f"Raw Prediction Output: {prediction}") | |
| # Apply thresholding based on the raw prediction value | |
| # If the model outputs a probability, try adjusting the threshold (e.g., 0.6 instead of 0.5) | |
| if prediction[0] > 0.5: | |
| return "User" | |
| else: | |
| return "Non-User" | |
| except Exception as e: | |
| print(f"Error in prediction: {e}") | |
| return "Error during prediction" | |
| # Define the Gradio interface | |
| iface = gr.Interface( | |
| fn=predict_user_or_non_user, # The function to call when an audio input is given | |
| inputs=gr.Audio(type="filepath"), # Corrected audio input setup | |
| outputs="text", # Output will be text (User or Non-user) | |
| live=True # Live mode so the interface updates in real-time | |
| ) | |
| # Launch the interface | |
| iface.launch() |