# Copyright (c) OpenMMLab. All rights reserved. from typing import List, Tuple, Dict import numpy as np from models.base.onnx_base import ONNX_Base from models.base.trt_base import TRT_Base from models.engine.utils import * class RTMPose(): def __init__(self, use_torch: bool=False) -> None: self.use_torch = use_torch def preprocess(self, input_data: np.ndarray, input_size: Tuple[int, int] = (192, 256)) -> Tuple[np.ndarray, np.ndarray, np.ndarray]: """Do preprocessing for RTMPose model inference. Args: img (np.ndarray): Input image in shape. input_size (tuple): Input image size in shape (w, h). Returns: tuple: - resized_img (np.ndarray): Preprocessed image. - center (np.ndarray): Center of image. - scale (np.ndarray): Scale of image. """ tensor_data = [] # get shape of image scales =[] centers = [] for i in range(len(input_data)): img = input_data[i] img_shape = img.shape[:2] bbox = np.array([0, 0, img_shape[1], img_shape[0]]) # get center and scale center, scale = bbox_xyxy2cs(bbox, padding=1.25) # do affine transformation resized_img, scale = top_down_affine(input_size, scale, center, img) # normalize image mean = np.array([123.675, 116.28, 103.53]) std = np.array([58.395, 57.12, 57.375]) resized_img = (resized_img - mean) / std centers.append(center) scales.append(scale) if self.use_torch: tensor_data.append(torch.from_numpy(resized_img).to(self.device)) else: tensor_data.append(resized_img.transpose(2, 0, 1)) if self.use_torch: tensor_data = torch.stack(tensor_data, dim=0)[:, :, :, [2, 1, 0]].permute(0, 3, 1, 2).float().contiguous() else: tensor_data = np.stack(tensor_data, axis=0) return tensor_data, centers, scales def postprocess(self, outputs: List[np.ndarray], model_input_size: Tuple[int, int], centers: List[np.ndarray], scales: List[np.ndarray], simcc_split_ratio: float = 2.0, use_torch=False ) -> Tuple[np.ndarray, np.ndarray]: """Postprocess for RTMPose model output. Args: outputs (np.ndarray): Output of RTMPose model. model_input_size (tuple): RTMPose model Input image size. center List[tuple(int,int)]: List of Center of bbox in shape (x, y). scale List[tuple(int,int)]: List of Scales of bbox in shape (w, h). simcc_split_ratio (float): Split ratio of simcc. Returns: tuple: - keypoints (np.ndarray): Rescaled keypoints. - scores (np.ndarray): Model predict scores. """ # use simcc to decode simcc_x, simcc_y = outputs tensor_keypoints = [] tensor_scores = [] assert simcc_x.shape[0] == simcc_y.shape[0] for i in range(simcc_x.shape[0]): simcc_x_3d = simcc_x[i][np.newaxis, :, :] simcc_y_3d = simcc_y[i][np.newaxis, :, :] keypoints, scores = decode(simcc_x_3d, simcc_y_3d, simcc_split_ratio, use_torch=use_torch) # rescale keypoints keypoints = keypoints / model_input_size * scales[i] + centers[i] - scales[i] / 2 tensor_keypoints.append(keypoints) tensor_scores.append(scores) tensor_keypoints = np.vstack(tensor_keypoints) tensor_scores = np.vstack(tensor_scores) return tensor_keypoints, tensor_scores def crop_objects(self, image: np.ndarray, bounding_boxes: np.ndarray): """ Function to crop objects in input image. Args: image (np.ndarray): input image with shape (H, W, C). bounding_boxes (np.ndarray): Array with shape Nx4 with N is the number of objects. """ max_h, max_w = image.shape[:2] cropped_images = [] for box in bounding_boxes: x_top, y_top, x_bottom, y_bottom, _ = box.astype(int).tolist() x_top = max(0, x_top) y_top = max(0, y_top) x_bottom = min(x_bottom, max_w) y_bottom = min(y_bottom, max_h) cropped_image = image[y_top:y_bottom, x_top:x_bottom] cropped_images.append(cropped_image) return cropped_images class RTMPoseONNX(ONNX_Base, RTMPose): def __init__(self, use_torch, img_shape: Tuple[int, int, int]=(3, 256, 192), batch_size: int=32, model_path: str="", device: str='0'): #/home/ccvn/Workspace/haimd/CC-Demo-Collection/end2end.onnx """_summary_ RTMPose ONNX class for inference, which is base on ONNX_BASE and RTMPose Args: use_torch (_type_): use torch tensor or numpy array in preprocess and postprocess function. img_shape (Tuple[int, int], optional): _description_. Defaults to (640, 640). batch_size (int, optional): _description_. Defaults to 32. model_path (str, optional): _description_. Defaults to "". device (str, optional): _description_. Defaults to '0'. """ self.img_shape = img_shape self.batch_size = batch_size input_shape = (self.batch_size, *self.img_shape) super().__init__(input_shape, model_path, device) RTMPose.__init__(self, use_torch=use_torch) def infer_batch(self, image_batch: np.ndarray): h, w = self.session.get_inputs()[0].shape[2:] model_input_size = (w, h) numpy_array_data, centers, scales = self.preprocess(image_batch, model_input_size) numpy_array_data = numpy_array_data.astype(np.float32) results = super().infer_batch(numpy_array_data) keypoints, scores = self.postprocess(results, model_input_size, centers, scales) return {'keypoints':keypoints, 'scores': scores} class RTMPoseTRT(TRT_Base, RTMPose): def __init__(self, use_torch, img_shape: Tuple[int, int, int]=(3, 256, 192), batch_size: int=1, model_path: str="", device: str='0',): """ RTMPoseTRT class for inference, which is based on TRT_Base and RTMPose. """ self.img_shape = img_shape self.batch_size = batch_size input_shape = (self.batch_size, *self.img_shape) super().__init__(input_shape, model_path, device) RTMPose.__init__(self, use_torch=use_torch) def infer_batch(self, image_batch: np.ndarray): model_input_size = (self.img_shape[-1], self.img_shape[1]) tensor_data, centers, scales = self.preprocess(image_batch, model_input_size) self.change_runtime_dimension(input_shape=(len(tensor_data), 3, model_input_size[1], model_input_size[0])) self.model['binding_addrs']['input'] = int(tensor_data.data_ptr()) self.model['context'].execute_v2(list(self.model['binding_addrs'].values())) simcc_x = self.model['bindings']['simcc_x'].data.cpu() simcc_y = self.model['bindings']['simcc_y'].data.cpu() results = (simcc_x, simcc_y) keypoints, scores = self.postprocess(results, model_input_size, centers, scales, use_torch=self.use_torch) return {'keypoints':keypoints, 'scores': scores}