| import gradio as gr |
| import cv2 |
| import numpy as np |
| import tempfile |
| import json |
| from modules import mediapipe_generator |
| from modules.mediapipe_generator import extract_to_array |
| from modules.sentence_guesser import SentenceGuesser |
|
|
| guesser = SentenceGuesser(model="fold1") |
|
|
| EXTRACTION_PATH = "metadata/extraction_order.json" |
| SIGN_DICT_PATH = "metadata/sign_dict.json" |
|
|
| with open(EXTRACTION_PATH, 'r') as f: |
| extraction_order = json.load(f) |
| with open(SIGN_DICT_PATH, 'r') as f: |
| sign_dict = {int(k): v for k, v in json.load(f).items()} |
|
|
| def retry_tts(cached_words, cached_emotions): |
| if not cached_words or not cached_emotions: |
| gr.Warning("No translation available to synthesize yet. Please analyze a sequence first.") |
| return None |
| |
| print("🔄 Retrying TTS Generation...") |
| |
| return None |
|
|
| def get_sign_label(idx): |
| if idx == 80: |
| return "" |
| else: |
| return sign_dict.get(idx, f"ID_{idx}") |
|
|
| def render_normalized_component(landmarks, canvas_w, canvas_h, padding=0.1, color=(0, 255, 0)): |
| canvas = np.zeros((canvas_h, canvas_w, 3), dtype=np.uint8) |
| |
| if landmarks.shape[0] == 0 or np.isnan(landmarks).any(): |
| return canvas |
| |
| min_x, max_x = np.min(landmarks[:, 0]), np.max(landmarks[:, 0]) |
| min_y, max_y = np.min(landmarks[:, 1]), np.max(landmarks[:, 1]) |
| |
| data_w = max_x - min_x |
| data_h = max_y - min_y |
| |
| if data_w == 0 or data_h == 0: |
| return canvas |
| |
| usable_w = canvas_w * (1 - padding * 2) |
| usable_h = canvas_h * (1 - padding * 2) |
| scale = min(usable_w / data_w, usable_h / data_h) |
| |
| cx, cy = (min_x + max_x) / 2, (min_y + max_y) / 2 |
| canvas_cx, canvas_cy = canvas_w / 2, canvas_h / 2 |
| |
| for x, y in landmarks: |
| pix_x = int((x - cx) * scale + canvas_cx) |
| pix_y = int((y - cy) * scale + canvas_cy) |
| cv2.circle(canvas, (pix_x, pix_y), 2, color, -1) |
| |
| return canvas |
|
|
| def render_diagnostic_matrix(orig_frame, raw_landmarks, norm_pose, norm_lhand, norm_rhand, norm_face, current_label): |
| cell_w, cell_h = 320, 240 |
| |
| raw_rgb = cv2.resize(orig_frame, (cell_w, cell_h)) |
| raw_overlaid = raw_rgb.copy() |
| |
| for lm in raw_landmarks: |
| x_pix, y_pix = int(lm[0] * cell_w), int(lm[1] * cell_h) |
| cv2.circle(raw_overlaid, (x_pix, y_pix), 2, (0, 0, 255), -1) |
| |
| canvas_pose = render_normalized_component(norm_pose, cell_w, cell_h, color=(255, 255, 0)) |
| canvas_face = render_normalized_component(norm_face, cell_w, cell_h, color=(0, 255, 255)) |
| canvas_lhand = render_normalized_component(norm_lhand, cell_w, cell_h, color=(255, 0, 255)) |
| canvas_rhand = render_normalized_component(norm_rhand, cell_w, cell_h, color=(0, 255, 0)) |
|
|
| def add_title(img, text): |
| cv2.putText(img, text, (10, 25), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255, 255, 255), 1) |
| return img |
|
|
| add_title(raw_rgb, "Raw RGB") |
| cv2.putText(raw_rgb, f"PRED: {current_label}", (10, cell_h - 15), cv2.FONT_HERSHEY_DUPLEX, 0.7, (0, 255, 0), 2) |
| |
| add_title(raw_overlaid, "Raw + Landmarks") |
| add_title(canvas_pose, "Norm: Pose") |
| add_title(canvas_lhand, "Norm: L-Hand") |
| add_title(canvas_rhand, "Norm: R-Hand") |
| add_title(canvas_face, "Norm: Face") |
|
|
| row1 = np.hstack((raw_rgb, raw_overlaid)) |
| row2 = np.hstack((canvas_pose, canvas_face)) |
| row3 = np.hstack((canvas_lhand, canvas_rhand)) |
| |
| return np.vstack((row1, row2, row3)) |
|
|
| def render_diagnostic_video(original_path, raw_data, norm_data, frame_indices, fps=30): |
| temp_file = tempfile.NamedTemporaryFile(delete=False, suffix='.mp4') |
| output_path = temp_file.name |
| |
| w, h = 640, 720 |
| fourcc = cv2.VideoWriter_fourcc(*'mp4v') |
| out = cv2.VideoWriter(output_path, fourcc, fps, (w, h)) |
| cap = cv2.VideoCapture(original_path) |
|
|
| for t in range(raw_data.shape[0]): |
| ret, orig_frame = cap.read() |
| if not ret: |
| orig_frame = np.zeros((480, 640, 3), dtype=np.uint8) |
|
|
| norm_pose = norm_data[t, :7, :] |
| norm_lhand = norm_data[t, 7:28, :] |
| norm_rhand = norm_data[t, 28:49, :] |
| norm_face = norm_data[t, 49:, :] |
| |
| current_label = get_sign_label(int(frame_indices[t])) |
|
|
| matrix_frame = render_diagnostic_matrix( |
| orig_frame, |
| raw_data[t], |
| norm_pose, |
| norm_lhand, |
| norm_rhand, |
| norm_face, |
| current_label |
| ) |
| out.write(matrix_frame) |
|
|
| cap.release() |
| out.release() |
| return output_path |
|
|
| def predict_sign(video_path): |
| mediapipe_results = mediapipe_generator.generate_mediapipe_gradio(video_path) |
| pose_seq = np.array([extract_to_array(r.pose_landmarks, 33, 4) for r in mediapipe_results]) |
| face_seq = np.array([extract_to_array(r.face_landmarks, 468, 3) for r in mediapipe_results]) |
| lh_seq = np.array([extract_to_array(r.left_hand_landmarks, 21, 3) for r in mediapipe_results]) |
| rh_seq = np.array([extract_to_array(r.right_hand_landmarks, 21, 3) for r in mediapipe_results]) |
|
|
| result = guesser.predict(pose_seq, face_seq, lh_seq, rh_seq) |
|
|
| words = result['prediction_string'].split() |
| emotions = result['prediction_emotions'] |
| |
| mapped_output = "\n".join([f"{w} -> [{e}]" for w, e in zip(words, emotions)]) |
|
|
| display_text = ( |
| f"Raw Window IDs: {result['raw_ids']}\n\n" |
| f"Translation: {result['prediction_string']}\n\n" |
| f"Emotion Mapping:\n{mapped_output}" |
| ) |
| viz_path = render_diagnostic_video( |
| video_path, |
| result['raw_data'], |
| result['norm_data'], |
| result['frame_indices'] |
| ) |
|
|
| |
| audio_path = None |
| |
| return display_text, audio_path, viz_path, words, emotions |
|
|
| with gr.Blocks() as demo: |
| current_words = gr.State([]) |
| current_emotions = gr.State([]) |
|
|
| gr.Markdown("# Filipino Sign Language Recognition") |
|
|
| with gr.Row(): |
| with gr.Column(): |
| video_input = gr.Video(label="Input: Upload or Record") |
| submit_btn = gr.Button("Analyze Sequence", variant="primary") |
| |
| with gr.Column(): |
| output_text = gr.Textbox(label="Model Predictions") |
|
|
| viz_output = gr.Video(label="Detected and Normalized Landmarks") |
|
|
| with gr.Row(): |
| |
| retry_audio_btn = gr.Button("🔄 Retry Audio", size="sm", scale=1) |
|
|
| submit_btn.click( |
| fn=predict_sign, |
| inputs=video_input, |
| outputs=[output_text, audio_output, viz_output, current_words, current_emotions] |
| ) |
| |
| retry_audio_btn.click( |
| fn=retry_tts, |
| inputs=[current_words, current_emotions], |
| outputs=[audio_output] |
| ) |
|
|
| demo.launch() |
|
|