Spaces:
Sleeping
Sleeping
| import cv2 | |
| import mediapipe as mp | |
| import numpy as np | |
| import torch | |
| import torch.nn as nn | |
| import gradio as gr | |
| # ---------------------------- | |
| # Labels | |
| # ---------------------------- | |
| GESTURE_LABELS = { | |
| 0: "A", 1: "B", 2: "L", 3: "U", 4: "V", 5: "W", | |
| 6: "Z", 7: "F", 8: "five", 9: "one", 10: "three", | |
| 11: "two", 12: "six", 13: "seven", 14: "eight", | |
| 15: "nine", 16: "ten", 17: "E", 18: "four", | |
| 19: "i", 20: "k", 21: "r", 22: "zero", | |
| 23: "m", 24: "s" | |
| } | |
| CONF_THRESHOLD = 0.6 | |
| # ---------------------------- | |
| # Model | |
| # ---------------------------- | |
| class GestureNet(nn.Module): | |
| def __init__(self, input_size=126, num_classes=len(GESTURE_LABELS)): | |
| super().__init__() | |
| self.fc1 = nn.Linear(input_size, 256) | |
| self.fc2 = nn.Linear(256, 128) | |
| self.fc3 = nn.Linear(128, num_classes) | |
| self.relu = nn.ReLU() | |
| self.dropout = nn.Dropout(0.3) | |
| def forward(self, x): | |
| x = self.relu(self.fc1(x)) | |
| x = self.dropout(x) | |
| x = self.relu(self.fc2(x)) | |
| x = self.dropout(x) | |
| return self.fc3(x) | |
| model = GestureNet() | |
| model.load_state_dict(torch.load("gesture_model1.pth", map_location="cpu")) | |
| model.eval() | |
| # ---------------------------- | |
| # MediaPipe | |
| # ---------------------------- | |
| mp_hands = mp.solutions.hands | |
| # ---------------------------- | |
| # Predict | |
| # ---------------------------- | |
| def predict(image): | |
| # Gradio sends RGB, convert to BGR for OpenCV then back to RGB for MediaPipe | |
| image_bgr = cv2.cvtColor(image, cv2.COLOR_RGB2BGR) | |
| image_rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB) | |
| with mp_hands.Hands(static_image_mode=True, max_num_hands=2) as hands: | |
| results = hands.process(image_rgb) | |
| coords = [] | |
| if results.multi_hand_landmarks: | |
| for hand_landmarks in results.multi_hand_landmarks: | |
| hand_coords = np.array([[lm.x, lm.y, lm.z] for lm in hand_landmarks.landmark]) | |
| hand_coords -= hand_coords[0] | |
| max_val = np.max(np.linalg.norm(hand_coords, axis=1)) | |
| if max_val > 0: | |
| hand_coords /= max_val | |
| coords.extend(hand_coords.flatten()) | |
| if len(coords) < 126: | |
| coords.extend([0.0] * (126 - len(coords))) | |
| elif len(coords) > 126: | |
| coords = coords[:126] | |
| if len(coords) == 126: | |
| input_tensor = torch.tensor(coords, dtype=torch.float32).unsqueeze(0) | |
| with torch.no_grad(): | |
| outputs = model(input_tensor) | |
| probs = torch.softmax(outputs, dim=1) | |
| pred_class = torch.argmax(probs, dim=1).item() | |
| confidence = probs[0][pred_class].item() | |
| if confidence >= CONF_THRESHOLD: | |
| return f"{GESTURE_LABELS[pred_class]} ({confidence*100:.2f}%)" | |
| return "Unknown" | |
| # ---------------------------- | |
| # Gradio UI | |
| # ---------------------------- | |
| app = gr.Interface( | |
| fn=predict, | |
| inputs=gr.Image(type="numpy"), | |
| outputs="text", | |
| title="Hand Gesture Recognition", | |
| description="Upload an image of a hand gesture to recognize it." | |
| ) | |
| app.launch() |