import pandas as pd import numpy as np import torch import json from tqdm import tqdm import os import h5py import csv from os import listdir from pathlib import Path from modules.normalized_video_visualizer import normalize_megalist_frame def fsl105_merge_metadata(root_dir, trim_indices): final_dict = {} with open(root_dir + "/labels.csv", mode="r") as infile: with open(trim_indices, "r") as trimfile: reader = csv.reader(infile) trim_dict = json.load(trimfile) next(reader) for row in reader: id_metadata = {"label": row[1], "category": row[2]} final_dict.update({row[0]:id_metadata}) sign_video_list = listdir(root_dir + "/clips/" + row[0]) final_video_list = [] for x in sign_video_list: video_instance = dict() video_instance["filename"] = x video_instance["filepath"] = root_dir + "/clips/" + str(row[0]) + "/" + x remove = None startsAt = None endsAt = None if row[0] in trim_dict: if x in trim_dict[row[0]]: remove = trim_dict[row[0]][x].get("remove") startsAt = trim_dict[row[0]][x].get("start") endsAt = trim_dict[row[0]][x].get("end") video_instance["remove"] = remove video_instance["startsAt"] = startsAt video_instance["endsAt"] = endsAt final_video_list.append(video_instance) final_dict[row[0]].update({"instances":final_video_list}) with open("metadata/fsl105-labels.json", "w") as f: f.write(json.dumps(final_dict, indent=2)) def get_valid_instances(filepath: str): with open(filepath, "r") as file: labels_dict = json.load(file) for x in labels_dict.keys(): for y in labels_dict[x]["instances"]: if y["remove"] != None: labels_dict[x]["instances"].remove(y) return labels_dict def stream_npy_to_hdf5(source_dir, output_file): source_path = Path(source_dir) with h5py.File(output_file, "a") as hf: class_dirs = [d for d in source_path.iterdir() if d.is_dir()] for class_dir in class_dirs: class_id = class_dir.name grp = hf.require_group(class_id) npy_files = list(class_dir.glob("*.npy")) print(f"Processing class: {class_id} ({len(npy_files)} files)...") for npy_path in npy_files: file_name = npy_path.name try: features = np.load(npy_path) arr = np.array(features, dtype="float32") if file_name in grp: del grp[file_name] file_name = file_name.replace(".npy", "") grp.create_dataset(file_name, data=arr, compression="gzip") except Exception as e: print(f"Error processing {npy_path}: {e}") print(f"Final HDF5 file saved and closed at: {output_file}") def process_and_save_normalized_hdf5(input_path, output_path): if not os.path.exists(input_path): print(f"Error: Source HDF5 not found at {input_path}") return with h5py.File(input_path, 'r') as source_hf, h5py.File(output_path, 'w') as target_hf: for class_id in tqdm(source_hf.keys(), desc="Processing Classes"): target_group = target_hf.create_group(class_id) for file_key in source_hf[class_id].keys(): landmarks_raw = np.array(source_hf[class_id][file_key])[:, :, :2] normalized_sequence = [] for frame_data in landmarks_raw: b_norm, lh_norm, rh_norm = normalize_megalist_frame(frame_data) combined_frame = torch.cat([b_norm, lh_norm, rh_norm], dim=0) normalized_sequence.append(combined_frame.numpy()) final_data = np.array(normalized_sequence, dtype="float32") target_group.create_dataset( file_key, data=final_data, compression="gzip", compression_opts=4 ) def flatten_samples_preserve_hierarchy(source_h5_path, target_h5_path, extraction_order): mapping_logic = {} for ds_name in ['pose', 'left_hand', 'right_hand', 'face']: req = extraction_order[ds_name] sorted_idx = sorted(list(set(req))) idx_map = {idx: i for i, idx in enumerate(sorted_idx)} reorder = np.array([idx_map[idx] for idx in req]) mapping_logic[ds_name] = { 'sorted': sorted_idx, 'reorder': reorder } with h5py.File(source_h5_path, 'r') as src, h5py.File(target_h5_path, 'w') as dst: all_classes = list(src.keys()) for class_id in tqdm(all_classes, desc="Flattening Classes"): class_group_dst = dst.create_group(class_id) class_group_src = src[class_id] for sample_name in class_group_src.keys(): sample_block = class_group_src[sample_name] try: parts = [] for ds_name in ['pose', 'left_hand', 'right_hand', 'face']: logic = mapping_logic[ds_name] data = sample_block[ds_name][:, logic['sorted'], :3] data = data[:, logic['reorder'], :] parts.append(data) combined_tensor = np.concatenate(parts, axis=1).astype(np.float32) class_group_dst.create_dataset( sample_name, data=combined_tensor, compression="gzip", chunks=True ) except Exception as e: print(f"Error processing {class_id}/{sample_name}: {e}") print(f"\nProcessing complete. New hierarchy saved to: {target_h5_path}") def apply_signbart_normalization(data): pose_len = len(extraction_order['pose']) lh_len = len(extraction_order['left_hand']) rh_len = len(extraction_order['right_hand']) part_indices = { "body": (0, pose_len), "lh": (pose_len, pose_len + lh_len), "rh": (pose_len + lh_len, pose_len + lh_len + rh_len) } normalized = data.copy() for name, (start, end) in part_indices.items(): part_data = normalized[:, start:end, :2] mask = (part_data != 0).any(axis=-1) if not np.any(mask): continue points = part_data[mask] p_min = points.min(axis=0) p_max = points.max(axis=0) margin = (p_max - p_min) * 0.05 p_min -= margin p_max += margin range_val = p_max - p_min range_val[range_val == 0] = 1.0 normalized[:, start:end, :2] = np.where( normalized[:, start:end, :2] != 0, (normalized[:, start:end, :2] - p_min) / range_val, 0 ) return normalized def extract_video_data(pose_seq, face_seq, lh_seq, rh_seq, extraction_order): def get_ordered_indices(indices, sequence): requested_indices = extraction_order[indices] sorted_indices = sorted(list(set(requested_indices))) data_subset = sequence[:, sorted_indices, :] index_map = {idx: i for i, idx in enumerate(sorted_indices)} reorder_map = [index_map[idx] for idx in requested_indices] return data_subset[:, reorder_map, :] pose = get_ordered_indices("pose", pose_seq)[:, :, :2] lh = get_ordered_indices("left_hand", lh_seq)[:, :, :2] rh = get_ordered_indices("right_hand", rh_seq)[:, :, :2] face = get_ordered_indices("face", face_seq)[:, :, :2] combined = np.concatenate([pose, lh, rh, face], axis=1) return combined def flatten_samples_preserve_hierarchy(source_h5_path, target_h5_path, extraction_order): mapping_logic = {} for ds_name in ['pose', 'left_hand', 'right_hand', 'face']: req = extraction_order[ds_name] sorted_idx = sorted(list(set(req))) idx_map = {idx: i for i, idx in enumerate(sorted_idx)} reorder = np.array([idx_map[idx] for idx in req]) mapping_logic[ds_name] = { 'sorted': sorted_idx, 'reorder': reorder } with h5py.File(source_h5_path, 'r') as src, h5py.File(target_h5_path, 'w') as dst: all_classes = list(src.keys()) for class_id in tqdm(all_classes, desc="Flattening Classes"): class_group_dst = dst.create_group(class_id) class_group_src = src[class_id] for sample_name in class_group_src.keys(): sample_block = class_group_src[sample_name] try: parts = [] for ds_name in ['pose', 'left_hand', 'right_hand', 'face']: logic = mapping_logic[ds_name] data = sample_block[ds_name][:, logic['sorted'], :3] data = data[:, logic['reorder'], :] parts.append(data) combined_tensor = np.concatenate(parts, axis=1).astype(np.float32) class_group_dst.create_dataset( sample_name, data=combined_tensor, compression="gzip", chunks=True ) except Exception as e: print(f"Error processing {class_id}/{sample_name}: {e}") print(f"\nProcessing complete. New hierarchy saved to: {target_h5_path}") def apply_hybrid_normalization(data, extraction_order): pose_len = len(extraction_order['pose']) lh_len = len(extraction_order['left_hand']) rh_len = len(extraction_order['right_hand']) pose_start = 0 lh_start = pose_len rh_start = lh_start + lh_len face_start = rh_start + rh_len FACE_REGIONS = { "left_eye": [46, 52, 53, 65, 7, 159, 155, 145, 70, 107, 105, 22, 23, 24, 110, 157, 158], "right_eye": [295, 283, 282, 276, 382, 386, 249, 374, 336, 300, 285, 252, 253, 254, 339, 384, 385], "mouth": [324, 13, 78, 14, 61, 291, 37, 0, 267, 84, 17, 314, 308, 318, 402, 312, 178, 88, 95] } region_map = {} for name, ids in FACE_REGIONS.items(): indices = [face_start + extraction_order['face'].index(lm_id) for lm_id in ids if lm_id in extraction_order['face']] region_map[name] = indices normalized = data.copy() for f in range(normalized.shape[0]): frame = normalized[f] try: NOSE_I = pose_start + extraction_order['pose'].index(0) L_SHOULDER_I = pose_start + extraction_order['pose'].index(11) R_SHOULDER_I = pose_start + extraction_order['pose'].index(12) L_EYE_I = face_start + extraction_order['face'].index(386) shoulder_dist = np.linalg.norm(frame[L_SHOULDER_I, :2] - frame[R_SHOULDER_I, :2]) head_unit = shoulder_dist / 2.0 if head_unit > 1e-8: nose_x = frame[NOSE_I, 0] l_eye_y = frame[L_EYE_I, 1] box_left = nose_x - (3 * head_unit) box_right = nose_x + (3 * head_unit) box_top = l_eye_y + (0.5 * head_unit) box_bottom = l_eye_y - (6 * head_unit) width = max(box_right - box_left, 1e-8) height = max(box_top - box_bottom, 1e-8) pose_indices = range(pose_start, lh_start) for idx in pose_indices: if not np.all(frame[idx, :2] == 0): frame[idx, 0] = (frame[idx, 0] - box_left) / width - 0.5 frame[idx, 1] = (frame[idx, 1] - box_bottom) / height - 0.5 except ValueError: pass for start, end in [(lh_start, rh_start), (rh_start, face_start)]: pts = frame[start:end, :2] mask = (pts != 0).any(axis=-1) if np.sum(mask) >= 2: h_min, h_max = pts[mask].min(axis=0), pts[mask].max(axis=0) side = np.max(h_max - h_min) * 1.2 if side > 1e-8: center = (h_min + h_max) / 2.0 box_min = center - (side / 2.0) frame[start:end, :2] = np.where( frame[start:end, :2] != 0, ((frame[start:end, :2] - box_min) / side) - 0.5, 0 ) for region_name, indices in region_map.items(): pts = frame[indices, :2] mask = (pts != 0).any(axis=-1) if np.sum(mask) >= 2: p_min, p_max = pts[mask].min(axis=0), pts[mask].max(axis=0) side = np.max(p_max - p_min) * 1.2 if side > 1e-8: center = (p_min + p_max) / 2.0 box_min = center - (side / 2.0) frame[indices, :2] = np.where( frame[indices, :2] != 0, ((frame[indices, :2] - box_min) / side) - 0.5, 0 ) normalized[f] = frame return normalized