program(1.0) [buildInfo = dict, tensor>({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}, {"coremltools-component-torch", "2.7.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0b1"}})] { func main(tensor attention_mask, tensor cross_attention_mask, tensor encoder_hidden_states, tensor input_id, tensor k_cache_0, tensor k_cache_1, tensor k_cache_2, tensor k_cache_3, tensor k_cache_4, tensor k_cache_5, tensor k_cache_6, tensor k_cache_7, tensor position_id, tensor v_cache_0, tensor v_cache_1, tensor v_cache_2, tensor v_cache_3, tensor v_cache_4, tensor v_cache_5, tensor v_cache_6, tensor v_cache_7) { tensor var_282 = const()[name = tensor("op_282"), val = tensor([1, 1, 1, 1])]; tensor var_283 = reshape(shape = var_282, x = position_id)[name = tensor("op_283")]; tensor pos_idx_reps_0 = const()[name = tensor("pos_idx_reps_0"), val = tensor([1, 8, 1, 128])]; tensor pos_idx = tile(reps = pos_idx_reps_0, x = var_283)[name = tensor("pos_idx")]; tensor var_295 = const()[name = tensor("op_295"), val = tensor(0)]; tensor var_303_batch_dims_0 = const()[name = tensor("op_303_batch_dims_0"), val = tensor(0)]; tensor var_303_validate_indices_0 = const()[name = tensor("op_303_validate_indices_0"), val = tensor(false)]; tensor embedding_token_embedding_weight_to_fp16 = const()[name = tensor("embedding_token_embedding_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(64)))]; tensor input_id_to_int16_dtype_0 = const()[name = tensor("input_id_to_int16_dtype_0"), val = tensor("int16")]; tensor cast_68_dtype_0 = const()[name = tensor("cast_68_dtype_0"), val = tensor("int32")]; tensor greater_equal_0_y_0 = const()[name = tensor("greater_equal_0_y_0"), val = tensor(0)]; tensor input_id_to_int16 = cast(dtype = input_id_to_int16_dtype_0, x = input_id)[name = tensor("cast_108")]; tensor cast_68 = cast(dtype = cast_68_dtype_0, x = input_id_to_int16)[name = tensor("cast_107")]; tensor greater_equal_0 = greater_equal(x = cast_68, y = greater_equal_0_y_0)[name = tensor("greater_equal_0")]; tensor slice_by_index_0 = const()[name = tensor("slice_by_index_0"), val = tensor(16384)]; tensor add_16 = add(x = cast_68, y = slice_by_index_0)[name = tensor("add_16")]; tensor select_0 = select(a = cast_68, b = add_16, cond = greater_equal_0)[name = tensor("select_0")]; tensor var_303_cast_fp16_cast_uint16_axis_0 = const()[name = tensor("op_303_cast_fp16_cast_uint16_axis_0"), val = tensor(0)]; tensor select_0_to_int16_dtype_0 = const()[name = tensor("select_0_to_int16_dtype_0"), val = tensor("int16")]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = tensor("cast_106")]; tensor var_303_cast_fp16_cast_uint16_cast_uint16 = gather(axis = var_303_cast_fp16_cast_uint16_axis_0, batch_dims = var_303_batch_dims_0, indices = select_0_to_int16, validate_indices = var_303_validate_indices_0, x = embedding_token_embedding_weight_to_fp16)[name = tensor("op_303_cast_fp16_cast_uint16_cast_uint16")]; tensor var_305 = const()[name = tensor("op_305"), val = tensor([-1])]; tensor var_306 = reshape(shape = var_305, x = position_id)[name = tensor("op_306")]; tensor var_307_batch_dims_0 = const()[name = tensor("op_307_batch_dims_0"), val = tensor(0)]; tensor var_307_validate_indices_0 = const()[name = tensor("op_307_validate_indices_0"), val = tensor(false)]; tensor embedding_position_embedding_pos_enc_to_fp16 = const()[name = tensor("embedding_position_embedding_pos_enc_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(33554560)))]; tensor var_306_to_uint16_dtype_0 = const()[name = tensor("op_306_to_uint16_dtype_0"), val = tensor("uint16")]; tensor var_306_to_uint16 = cast(dtype = var_306_to_uint16_dtype_0, x = var_306)[name = tensor("cast_105")]; tensor var_307_cast_fp16_cast_uint16 = gather(axis = var_295, batch_dims = var_307_batch_dims_0, indices = var_306_to_uint16, validate_indices = var_307_validate_indices_0, x = embedding_position_embedding_pos_enc_to_fp16)[name = tensor("op_307_cast_fp16_cast_uint16")]; tensor var_310 = const()[name = tensor("op_310"), val = tensor([1, 1, -1])]; tensor var_311_cast_fp16 = reshape(shape = var_310, x = var_307_cast_fp16_cast_uint16)[name = tensor("op_311_cast_fp16")]; tensor input_1_cast_fp16 = add(x = var_303_cast_fp16_cast_uint16_cast_uint16, y = var_311_cast_fp16)[name = tensor("input_1_cast_fp16")]; tensor input_3_axes_0 = const()[name = tensor("input_3_axes_0"), val = tensor([-1])]; tensor embedding_layer_norm_weight_to_fp16 = const()[name = tensor("embedding_layer_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35651776)))]; tensor embedding_layer_norm_bias_to_fp16 = const()[name = tensor("embedding_layer_norm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35653888)))]; tensor var_292_to_fp16 = const()[name = tensor("op_292_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_3_cast_fp16 = layer_norm(axes = input_3_axes_0, beta = embedding_layer_norm_bias_to_fp16, epsilon = var_292_to_fp16, gamma = embedding_layer_norm_weight_to_fp16, x = input_1_cast_fp16)[name = tensor("input_3_cast_fp16")]; tensor input_5_axes_0 = const()[name = tensor("input_5_axes_0"), val = tensor([-1])]; tensor layers_0_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_0_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35656000)))]; tensor layers_0_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_0_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35658112)))]; tensor var_318_to_fp16 = const()[name = tensor("op_318_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_5_cast_fp16 = layer_norm(axes = input_5_axes_0, beta = layers_0_layer_norm_1_bias_to_fp16, epsilon = var_318_to_fp16, gamma = layers_0_layer_norm_1_weight_to_fp16, x = input_3_cast_fp16)[name = tensor("input_5_cast_fp16")]; tensor layers_0_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35660224)))]; tensor layers_0_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(37757440)))]; tensor linear_0_cast_fp16 = linear(bias = layers_0_first_sub_layer_query_net_bias_to_fp16, weight = layers_0_first_sub_layer_query_net_weight_to_fp16, x = input_5_cast_fp16)[name = tensor("linear_0_cast_fp16")]; tensor layers_0_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(37759552)))]; tensor layers_0_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(39856768)))]; tensor linear_1_cast_fp16 = linear(bias = layers_0_first_sub_layer_key_net_bias_to_fp16, weight = layers_0_first_sub_layer_key_net_weight_to_fp16, x = input_5_cast_fp16)[name = tensor("linear_1_cast_fp16")]; tensor layers_0_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(39858880)))]; tensor layers_0_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41956096)))]; tensor linear_2_cast_fp16 = linear(bias = layers_0_first_sub_layer_value_net_bias_to_fp16, weight = layers_0_first_sub_layer_value_net_weight_to_fp16, x = input_5_cast_fp16)[name = tensor("linear_2_cast_fp16")]; tensor var_343 = const()[name = tensor("op_343"), val = tensor([1, 1, 8, 128])]; tensor var_344_cast_fp16 = reshape(shape = var_343, x = linear_0_cast_fp16)[name = tensor("op_344_cast_fp16")]; tensor query_1_perm_0 = const()[name = tensor("query_1_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_358 = const()[name = tensor("op_358"), val = tensor([1, 1, 8, 128])]; tensor var_359_cast_fp16 = reshape(shape = var_358, x = linear_1_cast_fp16)[name = tensor("op_359_cast_fp16")]; tensor key_1_perm_0 = const()[name = tensor("key_1_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_373 = const()[name = tensor("op_373"), val = tensor([1, 1, 8, 128])]; tensor var_374_cast_fp16 = reshape(shape = var_373, x = linear_2_cast_fp16)[name = tensor("op_374_cast_fp16")]; tensor value_1_perm_0 = const()[name = tensor("value_1_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_1_axis_0 = const()[name = tensor("k_cache_new_1_axis_0"), val = tensor(2)]; tensor k_cache_new_1_mode_0 = const()[name = tensor("k_cache_new_1_mode_0"), val = tensor("update")]; tensor k_cache_new_1_validate_indices_0 = const()[name = tensor("k_cache_new_1_validate_indices_0"), val = tensor(false)]; tensor k_cache_0_to_fp16_dtype_0 = const()[name = tensor("k_cache_0_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_0_to_fp16 = cast(dtype = k_cache_0_to_fp16_dtype_0, x = k_cache_0)[name = tensor("cast_104")]; tensor key_1_cast_fp16 = transpose(perm = key_1_perm_0, x = var_359_cast_fp16)[name = tensor("transpose_110")]; tensor k_cache_new_1_cast_fp16 = scatter_along_axis(axis = k_cache_new_1_axis_0, data = k_cache_0_to_fp16, indices = pos_idx, mode = k_cache_new_1_mode_0, updates = key_1_cast_fp16, validate_indices = k_cache_new_1_validate_indices_0)[name = tensor("k_cache_new_1_cast_fp16")]; tensor k_cache_new_1_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_1_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_1_axis_0 = const()[name = tensor("v_cache_new_1_axis_0"), val = tensor(2)]; tensor v_cache_new_1_mode_0 = const()[name = tensor("v_cache_new_1_mode_0"), val = tensor("update")]; tensor v_cache_new_1_validate_indices_0 = const()[name = tensor("v_cache_new_1_validate_indices_0"), val = tensor(false)]; tensor v_cache_0_to_fp16_dtype_0 = const()[name = tensor("v_cache_0_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_0_to_fp16 = cast(dtype = v_cache_0_to_fp16_dtype_0, x = v_cache_0)[name = tensor("cast_102")]; tensor value_1_cast_fp16 = transpose(perm = value_1_perm_0, x = var_374_cast_fp16)[name = tensor("transpose_109")]; tensor v_cache_new_1_cast_fp16 = scatter_along_axis(axis = v_cache_new_1_axis_0, data = v_cache_0_to_fp16, indices = pos_idx, mode = v_cache_new_1_mode_0, updates = value_1_cast_fp16, validate_indices = v_cache_new_1_validate_indices_0)[name = tensor("v_cache_new_1_cast_fp16")]; tensor v_cache_new_1_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_1_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_384_to_fp16 = const()[name = tensor("op_384_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_1_cast_fp16 = transpose(perm = query_1_perm_0, x = var_344_cast_fp16)[name = tensor("transpose_111")]; tensor mul_0_cast_fp16 = mul(x = query_1_cast_fp16, y = var_384_to_fp16)[name = tensor("mul_0_cast_fp16")]; tensor matmul_0_transpose_y_0 = const()[name = tensor("matmul_0_transpose_y_0"), val = tensor(true)]; tensor matmul_0_transpose_x_0 = const()[name = tensor("matmul_0_transpose_x_0"), val = tensor(false)]; tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = mul_0_cast_fp16, y = k_cache_new_1_cast_fp16)[name = tensor("matmul_0_cast_fp16")]; tensor attention_mask_to_fp16_dtype_0 = const()[name = tensor("attention_mask_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attention_mask_to_fp16 = cast(dtype = attention_mask_to_fp16_dtype_0, x = attention_mask)[name = tensor("cast_100")]; tensor add_0_cast_fp16 = add(x = matmul_0_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_0_cast_fp16")]; tensor softmax_0_axis_0 = const()[name = tensor("softmax_0_axis_0"), val = tensor(-1)]; tensor softmax_0_cast_fp16 = softmax(axis = softmax_0_axis_0, x = add_0_cast_fp16)[name = tensor("softmax_0_cast_fp16")]; tensor attn_output_1_transpose_x_0 = const()[name = tensor("attn_output_1_transpose_x_0"), val = tensor(false)]; tensor attn_output_1_transpose_y_0 = const()[name = tensor("attn_output_1_transpose_y_0"), val = tensor(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = softmax_0_cast_fp16, y = v_cache_new_1_cast_fp16)[name = tensor("attn_output_1_cast_fp16")]; tensor var_389_perm_0 = const()[name = tensor("op_389_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_395 = const()[name = tensor("op_395"), val = tensor([1, 1, 1024])]; tensor var_389_cast_fp16 = transpose(perm = var_389_perm_0, x = attn_output_1_cast_fp16)[name = tensor("transpose_108")]; tensor input_7_cast_fp16 = reshape(shape = var_395, x = var_389_cast_fp16)[name = tensor("input_7_cast_fp16")]; tensor layers_0_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41958208)))]; tensor layers_0_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_0_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(44055424)))]; tensor linear_3_cast_fp16 = linear(bias = layers_0_first_sub_layer_out_projection_bias_to_fp16, weight = layers_0_first_sub_layer_out_projection_weight_to_fp16, x = input_7_cast_fp16)[name = tensor("linear_3_cast_fp16")]; tensor input_9_cast_fp16 = add(x = input_3_cast_fp16, y = linear_3_cast_fp16)[name = tensor("input_9_cast_fp16")]; tensor input_11_axes_0 = const()[name = tensor("input_11_axes_0"), val = tensor([-1])]; tensor layers_0_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_0_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(44057536)))]; tensor layers_0_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_0_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(44059648)))]; tensor var_403_to_fp16 = const()[name = tensor("op_403_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_11_cast_fp16 = layer_norm(axes = input_11_axes_0, beta = layers_0_layer_norm_2_bias_to_fp16, epsilon = var_403_to_fp16, gamma = layers_0_layer_norm_2_weight_to_fp16, x = input_9_cast_fp16)[name = tensor("input_11_cast_fp16")]; tensor layers_0_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(44061760)))]; tensor layers_0_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(46158976)))]; tensor linear_4_cast_fp16 = linear(bias = layers_0_second_sub_layer_query_net_bias_to_fp16, weight = layers_0_second_sub_layer_query_net_weight_to_fp16, x = input_11_cast_fp16)[name = tensor("linear_4_cast_fp16")]; tensor var_427 = const()[name = tensor("op_427"), val = tensor([1, 1, 8, 128])]; tensor var_428_cast_fp16 = reshape(shape = var_427, x = linear_4_cast_fp16)[name = tensor("op_428_cast_fp16")]; tensor encoder_hidden_states_to_fp16_dtype_0 = const()[name = tensor("encoder_hidden_states_to_fp16_dtype_0"), val = tensor("fp16")]; tensor layers_0_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(46161088)))]; tensor layers_0_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(48258304)))]; tensor encoder_hidden_states_to_fp16 = cast(dtype = encoder_hidden_states_to_fp16_dtype_0, x = encoder_hidden_states)[name = tensor("cast_99")]; tensor linear_5_cast_fp16 = linear(bias = layers_0_second_sub_layer_key_net_bias_to_fp16, weight = layers_0_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_5_cast_fp16")]; tensor var_435 = const()[name = tensor("op_435"), val = tensor([1, 438, 8, 128])]; tensor var_436_cast_fp16 = reshape(shape = var_435, x = linear_5_cast_fp16)[name = tensor("op_436_cast_fp16")]; tensor layers_0_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(48260416)))]; tensor layers_0_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(50357632)))]; tensor linear_6_cast_fp16 = linear(bias = layers_0_second_sub_layer_value_net_bias_to_fp16, weight = layers_0_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_6_cast_fp16")]; tensor var_443 = const()[name = tensor("op_443"), val = tensor([1, 438, 8, 128])]; tensor var_444_cast_fp16 = reshape(shape = var_443, x = linear_6_cast_fp16)[name = tensor("op_444_cast_fp16")]; tensor value_3_perm_0 = const()[name = tensor("value_3_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_410_to_fp16 = const()[name = tensor("op_410_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_1_cast_fp16 = mul(x = var_428_cast_fp16, y = var_410_to_fp16)[name = tensor("mul_1_cast_fp16")]; tensor matmul_1_transpose_y_0 = const()[name = tensor("matmul_1_transpose_y_0"), val = tensor(true)]; tensor matmul_1_transpose_x_0 = const()[name = tensor("matmul_1_transpose_x_0"), val = tensor(false)]; tensor transpose_32_perm_0 = const()[name = tensor("transpose_32_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_33_perm_0 = const()[name = tensor("transpose_33_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_33 = transpose(perm = transpose_33_perm_0, x = var_436_cast_fp16)[name = tensor("transpose_105")]; tensor transpose_32 = transpose(perm = transpose_32_perm_0, x = mul_1_cast_fp16)[name = tensor("transpose_106")]; tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = transpose_32, y = transpose_33)[name = tensor("matmul_1_cast_fp16")]; tensor cross_attention_mask_to_fp16_dtype_0 = const()[name = tensor("cross_attention_mask_to_fp16_dtype_0"), val = tensor("fp16")]; tensor cross_attention_mask_to_fp16 = cast(dtype = cross_attention_mask_to_fp16_dtype_0, x = cross_attention_mask)[name = tensor("cast_98")]; tensor add_1_cast_fp16 = add(x = matmul_1_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_1_cast_fp16")]; tensor softmax_1_axis_0 = const()[name = tensor("softmax_1_axis_0"), val = tensor(-1)]; tensor softmax_1_cast_fp16 = softmax(axis = softmax_1_axis_0, x = add_1_cast_fp16)[name = tensor("softmax_1_cast_fp16")]; tensor attn_output_5_transpose_x_0 = const()[name = tensor("attn_output_5_transpose_x_0"), val = tensor(false)]; tensor attn_output_5_transpose_y_0 = const()[name = tensor("attn_output_5_transpose_y_0"), val = tensor(false)]; tensor value_3_cast_fp16 = transpose(perm = value_3_perm_0, x = var_444_cast_fp16)[name = tensor("transpose_107")]; tensor attn_output_5_cast_fp16 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = softmax_1_cast_fp16, y = value_3_cast_fp16)[name = tensor("attn_output_5_cast_fp16")]; tensor var_447_perm_0 = const()[name = tensor("op_447_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_451 = const()[name = tensor("op_451"), val = tensor([1, 1, 1024])]; tensor var_447_cast_fp16 = transpose(perm = var_447_perm_0, x = attn_output_5_cast_fp16)[name = tensor("transpose_104")]; tensor input_13_cast_fp16 = reshape(shape = var_451, x = var_447_cast_fp16)[name = tensor("input_13_cast_fp16")]; tensor layers_0_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(50359744)))]; tensor layers_0_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_0_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(52456960)))]; tensor linear_7_cast_fp16 = linear(bias = layers_0_second_sub_layer_out_projection_bias_to_fp16, weight = layers_0_second_sub_layer_out_projection_weight_to_fp16, x = input_13_cast_fp16)[name = tensor("linear_7_cast_fp16")]; tensor input_15_cast_fp16 = add(x = input_9_cast_fp16, y = linear_7_cast_fp16)[name = tensor("input_15_cast_fp16")]; tensor input_17_axes_0 = const()[name = tensor("input_17_axes_0"), val = tensor([-1])]; tensor layers_0_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_0_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(52459072)))]; tensor layers_0_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_0_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(52461184)))]; tensor var_459_to_fp16 = const()[name = tensor("op_459_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_17_cast_fp16 = layer_norm(axes = input_17_axes_0, beta = layers_0_layer_norm_3_bias_to_fp16, epsilon = var_459_to_fp16, gamma = layers_0_layer_norm_3_weight_to_fp16, x = input_15_cast_fp16)[name = tensor("input_17_cast_fp16")]; tensor layers_0_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_0_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(52463296)))]; tensor layers_0_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_0_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(60851968)))]; tensor linear_8_cast_fp16 = linear(bias = layers_0_third_sub_layer_dense_in_bias_to_fp16, weight = layers_0_third_sub_layer_dense_in_weight_to_fp16, x = input_17_cast_fp16)[name = tensor("linear_8_cast_fp16")]; tensor input_21_cast_fp16 = relu(x = linear_8_cast_fp16)[name = tensor("input_21_cast_fp16")]; tensor layers_0_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_0_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(60860224)))]; tensor layers_0_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_0_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(69248896)))]; tensor linear_9_cast_fp16 = linear(bias = layers_0_third_sub_layer_dense_out_bias_to_fp16, weight = layers_0_third_sub_layer_dense_out_weight_to_fp16, x = input_21_cast_fp16)[name = tensor("linear_9_cast_fp16")]; tensor input_23_cast_fp16 = add(x = input_15_cast_fp16, y = linear_9_cast_fp16)[name = tensor("input_23_cast_fp16")]; tensor input_25_axes_0 = const()[name = tensor("input_25_axes_0"), val = tensor([-1])]; tensor layers_1_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_1_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(69251008)))]; tensor layers_1_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_1_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(69253120)))]; tensor var_477_to_fp16 = const()[name = tensor("op_477_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_25_cast_fp16 = layer_norm(axes = input_25_axes_0, beta = layers_1_layer_norm_1_bias_to_fp16, epsilon = var_477_to_fp16, gamma = layers_1_layer_norm_1_weight_to_fp16, x = input_23_cast_fp16)[name = tensor("input_25_cast_fp16")]; tensor layers_1_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(69255232)))]; tensor layers_1_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(71352448)))]; tensor linear_10_cast_fp16 = linear(bias = layers_1_first_sub_layer_query_net_bias_to_fp16, weight = layers_1_first_sub_layer_query_net_weight_to_fp16, x = input_25_cast_fp16)[name = tensor("linear_10_cast_fp16")]; tensor layers_1_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(71354560)))]; tensor layers_1_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(73451776)))]; tensor linear_11_cast_fp16 = linear(bias = layers_1_first_sub_layer_key_net_bias_to_fp16, weight = layers_1_first_sub_layer_key_net_weight_to_fp16, x = input_25_cast_fp16)[name = tensor("linear_11_cast_fp16")]; tensor layers_1_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(73453888)))]; tensor layers_1_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(75551104)))]; tensor linear_12_cast_fp16 = linear(bias = layers_1_first_sub_layer_value_net_bias_to_fp16, weight = layers_1_first_sub_layer_value_net_weight_to_fp16, x = input_25_cast_fp16)[name = tensor("linear_12_cast_fp16")]; tensor var_502 = const()[name = tensor("op_502"), val = tensor([1, 1, 8, 128])]; tensor var_503_cast_fp16 = reshape(shape = var_502, x = linear_10_cast_fp16)[name = tensor("op_503_cast_fp16")]; tensor query_5_perm_0 = const()[name = tensor("query_5_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_517 = const()[name = tensor("op_517"), val = tensor([1, 1, 8, 128])]; tensor var_518_cast_fp16 = reshape(shape = var_517, x = linear_11_cast_fp16)[name = tensor("op_518_cast_fp16")]; tensor key_5_perm_0 = const()[name = tensor("key_5_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_532 = const()[name = tensor("op_532"), val = tensor([1, 1, 8, 128])]; tensor var_533_cast_fp16 = reshape(shape = var_532, x = linear_12_cast_fp16)[name = tensor("op_533_cast_fp16")]; tensor value_5_perm_0 = const()[name = tensor("value_5_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_3_axis_0 = const()[name = tensor("k_cache_new_3_axis_0"), val = tensor(2)]; tensor k_cache_new_3_mode_0 = const()[name = tensor("k_cache_new_3_mode_0"), val = tensor("update")]; tensor k_cache_new_3_validate_indices_0 = const()[name = tensor("k_cache_new_3_validate_indices_0"), val = tensor(false)]; tensor k_cache_1_to_fp16_dtype_0 = const()[name = tensor("k_cache_1_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_1_to_fp16 = cast(dtype = k_cache_1_to_fp16_dtype_0, x = k_cache_1)[name = tensor("cast_97")]; tensor key_5_cast_fp16 = transpose(perm = key_5_perm_0, x = var_518_cast_fp16)[name = tensor("transpose_102")]; tensor k_cache_new_3_cast_fp16 = scatter_along_axis(axis = k_cache_new_3_axis_0, data = k_cache_1_to_fp16, indices = pos_idx, mode = k_cache_new_3_mode_0, updates = key_5_cast_fp16, validate_indices = k_cache_new_3_validate_indices_0)[name = tensor("k_cache_new_3_cast_fp16")]; tensor k_cache_new_3_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_3_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_3_axis_0 = const()[name = tensor("v_cache_new_3_axis_0"), val = tensor(2)]; tensor v_cache_new_3_mode_0 = const()[name = tensor("v_cache_new_3_mode_0"), val = tensor("update")]; tensor v_cache_new_3_validate_indices_0 = const()[name = tensor("v_cache_new_3_validate_indices_0"), val = tensor(false)]; tensor v_cache_1_to_fp16_dtype_0 = const()[name = tensor("v_cache_1_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_1_to_fp16 = cast(dtype = v_cache_1_to_fp16_dtype_0, x = v_cache_1)[name = tensor("cast_95")]; tensor value_5_cast_fp16 = transpose(perm = value_5_perm_0, x = var_533_cast_fp16)[name = tensor("transpose_101")]; tensor v_cache_new_3_cast_fp16 = scatter_along_axis(axis = v_cache_new_3_axis_0, data = v_cache_1_to_fp16, indices = pos_idx, mode = v_cache_new_3_mode_0, updates = value_5_cast_fp16, validate_indices = v_cache_new_3_validate_indices_0)[name = tensor("v_cache_new_3_cast_fp16")]; tensor v_cache_new_3_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_3_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_543_to_fp16 = const()[name = tensor("op_543_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_5_cast_fp16 = transpose(perm = query_5_perm_0, x = var_503_cast_fp16)[name = tensor("transpose_103")]; tensor mul_2_cast_fp16 = mul(x = query_5_cast_fp16, y = var_543_to_fp16)[name = tensor("mul_2_cast_fp16")]; tensor matmul_2_transpose_y_0 = const()[name = tensor("matmul_2_transpose_y_0"), val = tensor(true)]; tensor matmul_2_transpose_x_0 = const()[name = tensor("matmul_2_transpose_x_0"), val = tensor(false)]; tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = mul_2_cast_fp16, y = k_cache_new_3_cast_fp16)[name = tensor("matmul_2_cast_fp16")]; tensor add_2_cast_fp16 = add(x = matmul_2_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_2_cast_fp16")]; tensor softmax_2_axis_0 = const()[name = tensor("softmax_2_axis_0"), val = tensor(-1)]; tensor softmax_2_cast_fp16 = softmax(axis = softmax_2_axis_0, x = add_2_cast_fp16)[name = tensor("softmax_2_cast_fp16")]; tensor attn_output_7_transpose_x_0 = const()[name = tensor("attn_output_7_transpose_x_0"), val = tensor(false)]; tensor attn_output_7_transpose_y_0 = const()[name = tensor("attn_output_7_transpose_y_0"), val = tensor(false)]; tensor attn_output_7_cast_fp16 = matmul(transpose_x = attn_output_7_transpose_x_0, transpose_y = attn_output_7_transpose_y_0, x = softmax_2_cast_fp16, y = v_cache_new_3_cast_fp16)[name = tensor("attn_output_7_cast_fp16")]; tensor var_548_perm_0 = const()[name = tensor("op_548_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_554 = const()[name = tensor("op_554"), val = tensor([1, 1, 1024])]; tensor var_548_cast_fp16 = transpose(perm = var_548_perm_0, x = attn_output_7_cast_fp16)[name = tensor("transpose_100")]; tensor input_27_cast_fp16 = reshape(shape = var_554, x = var_548_cast_fp16)[name = tensor("input_27_cast_fp16")]; tensor layers_1_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(75553216)))]; tensor layers_1_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_1_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(77650432)))]; tensor linear_13_cast_fp16 = linear(bias = layers_1_first_sub_layer_out_projection_bias_to_fp16, weight = layers_1_first_sub_layer_out_projection_weight_to_fp16, x = input_27_cast_fp16)[name = tensor("linear_13_cast_fp16")]; tensor input_29_cast_fp16 = add(x = input_23_cast_fp16, y = linear_13_cast_fp16)[name = tensor("input_29_cast_fp16")]; tensor input_31_axes_0 = const()[name = tensor("input_31_axes_0"), val = tensor([-1])]; tensor layers_1_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_1_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(77652544)))]; tensor layers_1_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_1_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(77654656)))]; tensor var_562_to_fp16 = const()[name = tensor("op_562_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_31_cast_fp16 = layer_norm(axes = input_31_axes_0, beta = layers_1_layer_norm_2_bias_to_fp16, epsilon = var_562_to_fp16, gamma = layers_1_layer_norm_2_weight_to_fp16, x = input_29_cast_fp16)[name = tensor("input_31_cast_fp16")]; tensor layers_1_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(77656768)))]; tensor layers_1_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(79753984)))]; tensor linear_14_cast_fp16 = linear(bias = layers_1_second_sub_layer_query_net_bias_to_fp16, weight = layers_1_second_sub_layer_query_net_weight_to_fp16, x = input_31_cast_fp16)[name = tensor("linear_14_cast_fp16")]; tensor var_586 = const()[name = tensor("op_586"), val = tensor([1, 1, 8, 128])]; tensor var_587_cast_fp16 = reshape(shape = var_586, x = linear_14_cast_fp16)[name = tensor("op_587_cast_fp16")]; tensor layers_1_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(79756096)))]; tensor layers_1_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(81853312)))]; tensor linear_15_cast_fp16 = linear(bias = layers_1_second_sub_layer_key_net_bias_to_fp16, weight = layers_1_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_15_cast_fp16")]; tensor var_594 = const()[name = tensor("op_594"), val = tensor([1, 438, 8, 128])]; tensor var_595_cast_fp16 = reshape(shape = var_594, x = linear_15_cast_fp16)[name = tensor("op_595_cast_fp16")]; tensor layers_1_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(81855424)))]; tensor layers_1_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(83952640)))]; tensor linear_16_cast_fp16 = linear(bias = layers_1_second_sub_layer_value_net_bias_to_fp16, weight = layers_1_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_16_cast_fp16")]; tensor var_602 = const()[name = tensor("op_602"), val = tensor([1, 438, 8, 128])]; tensor var_603_cast_fp16 = reshape(shape = var_602, x = linear_16_cast_fp16)[name = tensor("op_603_cast_fp16")]; tensor value_7_perm_0 = const()[name = tensor("value_7_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_569_to_fp16 = const()[name = tensor("op_569_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_3_cast_fp16 = mul(x = var_587_cast_fp16, y = var_569_to_fp16)[name = tensor("mul_3_cast_fp16")]; tensor matmul_3_transpose_y_0 = const()[name = tensor("matmul_3_transpose_y_0"), val = tensor(true)]; tensor matmul_3_transpose_x_0 = const()[name = tensor("matmul_3_transpose_x_0"), val = tensor(false)]; tensor transpose_34_perm_0 = const()[name = tensor("transpose_34_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_35_perm_0 = const()[name = tensor("transpose_35_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_35 = transpose(perm = transpose_35_perm_0, x = var_595_cast_fp16)[name = tensor("transpose_97")]; tensor transpose_34 = transpose(perm = transpose_34_perm_0, x = mul_3_cast_fp16)[name = tensor("transpose_98")]; tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = transpose_34, y = transpose_35)[name = tensor("matmul_3_cast_fp16")]; tensor add_3_cast_fp16 = add(x = matmul_3_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_3_cast_fp16")]; tensor softmax_3_axis_0 = const()[name = tensor("softmax_3_axis_0"), val = tensor(-1)]; tensor softmax_3_cast_fp16 = softmax(axis = softmax_3_axis_0, x = add_3_cast_fp16)[name = tensor("softmax_3_cast_fp16")]; tensor attn_output_11_transpose_x_0 = const()[name = tensor("attn_output_11_transpose_x_0"), val = tensor(false)]; tensor attn_output_11_transpose_y_0 = const()[name = tensor("attn_output_11_transpose_y_0"), val = tensor(false)]; tensor value_7_cast_fp16 = transpose(perm = value_7_perm_0, x = var_603_cast_fp16)[name = tensor("transpose_99")]; tensor attn_output_11_cast_fp16 = matmul(transpose_x = attn_output_11_transpose_x_0, transpose_y = attn_output_11_transpose_y_0, x = softmax_3_cast_fp16, y = value_7_cast_fp16)[name = tensor("attn_output_11_cast_fp16")]; tensor var_606_perm_0 = const()[name = tensor("op_606_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_610 = const()[name = tensor("op_610"), val = tensor([1, 1, 1024])]; tensor var_606_cast_fp16 = transpose(perm = var_606_perm_0, x = attn_output_11_cast_fp16)[name = tensor("transpose_96")]; tensor input_33_cast_fp16 = reshape(shape = var_610, x = var_606_cast_fp16)[name = tensor("input_33_cast_fp16")]; tensor layers_1_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(83954752)))]; tensor layers_1_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_1_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(86051968)))]; tensor linear_17_cast_fp16 = linear(bias = layers_1_second_sub_layer_out_projection_bias_to_fp16, weight = layers_1_second_sub_layer_out_projection_weight_to_fp16, x = input_33_cast_fp16)[name = tensor("linear_17_cast_fp16")]; tensor input_35_cast_fp16 = add(x = input_29_cast_fp16, y = linear_17_cast_fp16)[name = tensor("input_35_cast_fp16")]; tensor input_37_axes_0 = const()[name = tensor("input_37_axes_0"), val = tensor([-1])]; tensor layers_1_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_1_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(86054080)))]; tensor layers_1_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_1_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(86056192)))]; tensor var_618_to_fp16 = const()[name = tensor("op_618_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_37_cast_fp16 = layer_norm(axes = input_37_axes_0, beta = layers_1_layer_norm_3_bias_to_fp16, epsilon = var_618_to_fp16, gamma = layers_1_layer_norm_3_weight_to_fp16, x = input_35_cast_fp16)[name = tensor("input_37_cast_fp16")]; tensor layers_1_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_1_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(86058304)))]; tensor layers_1_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_1_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(94446976)))]; tensor linear_18_cast_fp16 = linear(bias = layers_1_third_sub_layer_dense_in_bias_to_fp16, weight = layers_1_third_sub_layer_dense_in_weight_to_fp16, x = input_37_cast_fp16)[name = tensor("linear_18_cast_fp16")]; tensor input_41_cast_fp16 = relu(x = linear_18_cast_fp16)[name = tensor("input_41_cast_fp16")]; tensor layers_1_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_1_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(94455232)))]; tensor layers_1_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_1_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(102843904)))]; tensor linear_19_cast_fp16 = linear(bias = layers_1_third_sub_layer_dense_out_bias_to_fp16, weight = layers_1_third_sub_layer_dense_out_weight_to_fp16, x = input_41_cast_fp16)[name = tensor("linear_19_cast_fp16")]; tensor input_43_cast_fp16 = add(x = input_35_cast_fp16, y = linear_19_cast_fp16)[name = tensor("input_43_cast_fp16")]; tensor input_45_axes_0 = const()[name = tensor("input_45_axes_0"), val = tensor([-1])]; tensor layers_2_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_2_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(102846016)))]; tensor layers_2_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_2_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(102848128)))]; tensor var_636_to_fp16 = const()[name = tensor("op_636_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_45_cast_fp16 = layer_norm(axes = input_45_axes_0, beta = layers_2_layer_norm_1_bias_to_fp16, epsilon = var_636_to_fp16, gamma = layers_2_layer_norm_1_weight_to_fp16, x = input_43_cast_fp16)[name = tensor("input_45_cast_fp16")]; tensor layers_2_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(102850240)))]; tensor layers_2_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(104947456)))]; tensor linear_20_cast_fp16 = linear(bias = layers_2_first_sub_layer_query_net_bias_to_fp16, weight = layers_2_first_sub_layer_query_net_weight_to_fp16, x = input_45_cast_fp16)[name = tensor("linear_20_cast_fp16")]; tensor layers_2_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(104949568)))]; tensor layers_2_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(107046784)))]; tensor linear_21_cast_fp16 = linear(bias = layers_2_first_sub_layer_key_net_bias_to_fp16, weight = layers_2_first_sub_layer_key_net_weight_to_fp16, x = input_45_cast_fp16)[name = tensor("linear_21_cast_fp16")]; tensor layers_2_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(107048896)))]; tensor layers_2_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(109146112)))]; tensor linear_22_cast_fp16 = linear(bias = layers_2_first_sub_layer_value_net_bias_to_fp16, weight = layers_2_first_sub_layer_value_net_weight_to_fp16, x = input_45_cast_fp16)[name = tensor("linear_22_cast_fp16")]; tensor var_661 = const()[name = tensor("op_661"), val = tensor([1, 1, 8, 128])]; tensor var_662_cast_fp16 = reshape(shape = var_661, x = linear_20_cast_fp16)[name = tensor("op_662_cast_fp16")]; tensor query_9_perm_0 = const()[name = tensor("query_9_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_676 = const()[name = tensor("op_676"), val = tensor([1, 1, 8, 128])]; tensor var_677_cast_fp16 = reshape(shape = var_676, x = linear_21_cast_fp16)[name = tensor("op_677_cast_fp16")]; tensor key_9_perm_0 = const()[name = tensor("key_9_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_691 = const()[name = tensor("op_691"), val = tensor([1, 1, 8, 128])]; tensor var_692_cast_fp16 = reshape(shape = var_691, x = linear_22_cast_fp16)[name = tensor("op_692_cast_fp16")]; tensor value_9_perm_0 = const()[name = tensor("value_9_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_5_axis_0 = const()[name = tensor("k_cache_new_5_axis_0"), val = tensor(2)]; tensor k_cache_new_5_mode_0 = const()[name = tensor("k_cache_new_5_mode_0"), val = tensor("update")]; tensor k_cache_new_5_validate_indices_0 = const()[name = tensor("k_cache_new_5_validate_indices_0"), val = tensor(false)]; tensor k_cache_2_to_fp16_dtype_0 = const()[name = tensor("k_cache_2_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_2_to_fp16 = cast(dtype = k_cache_2_to_fp16_dtype_0, x = k_cache_2)[name = tensor("cast_93")]; tensor key_9_cast_fp16 = transpose(perm = key_9_perm_0, x = var_677_cast_fp16)[name = tensor("transpose_94")]; tensor k_cache_new_5_cast_fp16 = scatter_along_axis(axis = k_cache_new_5_axis_0, data = k_cache_2_to_fp16, indices = pos_idx, mode = k_cache_new_5_mode_0, updates = key_9_cast_fp16, validate_indices = k_cache_new_5_validate_indices_0)[name = tensor("k_cache_new_5_cast_fp16")]; tensor k_cache_new_5_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_5_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_5_axis_0 = const()[name = tensor("v_cache_new_5_axis_0"), val = tensor(2)]; tensor v_cache_new_5_mode_0 = const()[name = tensor("v_cache_new_5_mode_0"), val = tensor("update")]; tensor v_cache_new_5_validate_indices_0 = const()[name = tensor("v_cache_new_5_validate_indices_0"), val = tensor(false)]; tensor v_cache_2_to_fp16_dtype_0 = const()[name = tensor("v_cache_2_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_2_to_fp16 = cast(dtype = v_cache_2_to_fp16_dtype_0, x = v_cache_2)[name = tensor("cast_91")]; tensor value_9_cast_fp16 = transpose(perm = value_9_perm_0, x = var_692_cast_fp16)[name = tensor("transpose_93")]; tensor v_cache_new_5_cast_fp16 = scatter_along_axis(axis = v_cache_new_5_axis_0, data = v_cache_2_to_fp16, indices = pos_idx, mode = v_cache_new_5_mode_0, updates = value_9_cast_fp16, validate_indices = v_cache_new_5_validate_indices_0)[name = tensor("v_cache_new_5_cast_fp16")]; tensor v_cache_new_5_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_5_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_702_to_fp16 = const()[name = tensor("op_702_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_9_cast_fp16 = transpose(perm = query_9_perm_0, x = var_662_cast_fp16)[name = tensor("transpose_95")]; tensor mul_4_cast_fp16 = mul(x = query_9_cast_fp16, y = var_702_to_fp16)[name = tensor("mul_4_cast_fp16")]; tensor matmul_4_transpose_y_0 = const()[name = tensor("matmul_4_transpose_y_0"), val = tensor(true)]; tensor matmul_4_transpose_x_0 = const()[name = tensor("matmul_4_transpose_x_0"), val = tensor(false)]; tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = mul_4_cast_fp16, y = k_cache_new_5_cast_fp16)[name = tensor("matmul_4_cast_fp16")]; tensor add_4_cast_fp16 = add(x = matmul_4_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_4_cast_fp16")]; tensor softmax_4_axis_0 = const()[name = tensor("softmax_4_axis_0"), val = tensor(-1)]; tensor softmax_4_cast_fp16 = softmax(axis = softmax_4_axis_0, x = add_4_cast_fp16)[name = tensor("softmax_4_cast_fp16")]; tensor attn_output_13_transpose_x_0 = const()[name = tensor("attn_output_13_transpose_x_0"), val = tensor(false)]; tensor attn_output_13_transpose_y_0 = const()[name = tensor("attn_output_13_transpose_y_0"), val = tensor(false)]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = softmax_4_cast_fp16, y = v_cache_new_5_cast_fp16)[name = tensor("attn_output_13_cast_fp16")]; tensor var_707_perm_0 = const()[name = tensor("op_707_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_713 = const()[name = tensor("op_713"), val = tensor([1, 1, 1024])]; tensor var_707_cast_fp16 = transpose(perm = var_707_perm_0, x = attn_output_13_cast_fp16)[name = tensor("transpose_92")]; tensor input_47_cast_fp16 = reshape(shape = var_713, x = var_707_cast_fp16)[name = tensor("input_47_cast_fp16")]; tensor layers_2_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(109148224)))]; tensor layers_2_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_2_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(111245440)))]; tensor linear_23_cast_fp16 = linear(bias = layers_2_first_sub_layer_out_projection_bias_to_fp16, weight = layers_2_first_sub_layer_out_projection_weight_to_fp16, x = input_47_cast_fp16)[name = tensor("linear_23_cast_fp16")]; tensor input_49_cast_fp16 = add(x = input_43_cast_fp16, y = linear_23_cast_fp16)[name = tensor("input_49_cast_fp16")]; tensor input_51_axes_0 = const()[name = tensor("input_51_axes_0"), val = tensor([-1])]; tensor layers_2_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_2_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(111247552)))]; tensor layers_2_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_2_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(111249664)))]; tensor var_721_to_fp16 = const()[name = tensor("op_721_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_51_cast_fp16 = layer_norm(axes = input_51_axes_0, beta = layers_2_layer_norm_2_bias_to_fp16, epsilon = var_721_to_fp16, gamma = layers_2_layer_norm_2_weight_to_fp16, x = input_49_cast_fp16)[name = tensor("input_51_cast_fp16")]; tensor layers_2_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(111251776)))]; tensor layers_2_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(113348992)))]; tensor linear_24_cast_fp16 = linear(bias = layers_2_second_sub_layer_query_net_bias_to_fp16, weight = layers_2_second_sub_layer_query_net_weight_to_fp16, x = input_51_cast_fp16)[name = tensor("linear_24_cast_fp16")]; tensor var_745 = const()[name = tensor("op_745"), val = tensor([1, 1, 8, 128])]; tensor var_746_cast_fp16 = reshape(shape = var_745, x = linear_24_cast_fp16)[name = tensor("op_746_cast_fp16")]; tensor layers_2_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(113351104)))]; tensor layers_2_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(115448320)))]; tensor linear_25_cast_fp16 = linear(bias = layers_2_second_sub_layer_key_net_bias_to_fp16, weight = layers_2_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_25_cast_fp16")]; tensor var_753 = const()[name = tensor("op_753"), val = tensor([1, 438, 8, 128])]; tensor var_754_cast_fp16 = reshape(shape = var_753, x = linear_25_cast_fp16)[name = tensor("op_754_cast_fp16")]; tensor layers_2_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(115450432)))]; tensor layers_2_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(117547648)))]; tensor linear_26_cast_fp16 = linear(bias = layers_2_second_sub_layer_value_net_bias_to_fp16, weight = layers_2_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_26_cast_fp16")]; tensor var_761 = const()[name = tensor("op_761"), val = tensor([1, 438, 8, 128])]; tensor var_762_cast_fp16 = reshape(shape = var_761, x = linear_26_cast_fp16)[name = tensor("op_762_cast_fp16")]; tensor value_11_perm_0 = const()[name = tensor("value_11_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_728_to_fp16 = const()[name = tensor("op_728_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_5_cast_fp16 = mul(x = var_746_cast_fp16, y = var_728_to_fp16)[name = tensor("mul_5_cast_fp16")]; tensor matmul_5_transpose_y_0 = const()[name = tensor("matmul_5_transpose_y_0"), val = tensor(true)]; tensor matmul_5_transpose_x_0 = const()[name = tensor("matmul_5_transpose_x_0"), val = tensor(false)]; tensor transpose_36_perm_0 = const()[name = tensor("transpose_36_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_37_perm_0 = const()[name = tensor("transpose_37_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_37 = transpose(perm = transpose_37_perm_0, x = var_754_cast_fp16)[name = tensor("transpose_89")]; tensor transpose_36 = transpose(perm = transpose_36_perm_0, x = mul_5_cast_fp16)[name = tensor("transpose_90")]; tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = transpose_36, y = transpose_37)[name = tensor("matmul_5_cast_fp16")]; tensor add_5_cast_fp16 = add(x = matmul_5_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_5_cast_fp16")]; tensor softmax_5_axis_0 = const()[name = tensor("softmax_5_axis_0"), val = tensor(-1)]; tensor softmax_5_cast_fp16 = softmax(axis = softmax_5_axis_0, x = add_5_cast_fp16)[name = tensor("softmax_5_cast_fp16")]; tensor attn_output_17_transpose_x_0 = const()[name = tensor("attn_output_17_transpose_x_0"), val = tensor(false)]; tensor attn_output_17_transpose_y_0 = const()[name = tensor("attn_output_17_transpose_y_0"), val = tensor(false)]; tensor value_11_cast_fp16 = transpose(perm = value_11_perm_0, x = var_762_cast_fp16)[name = tensor("transpose_91")]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = softmax_5_cast_fp16, y = value_11_cast_fp16)[name = tensor("attn_output_17_cast_fp16")]; tensor var_765_perm_0 = const()[name = tensor("op_765_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_769 = const()[name = tensor("op_769"), val = tensor([1, 1, 1024])]; tensor var_765_cast_fp16 = transpose(perm = var_765_perm_0, x = attn_output_17_cast_fp16)[name = tensor("transpose_88")]; tensor input_53_cast_fp16 = reshape(shape = var_769, x = var_765_cast_fp16)[name = tensor("input_53_cast_fp16")]; tensor layers_2_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(117549760)))]; tensor layers_2_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_2_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(119646976)))]; tensor linear_27_cast_fp16 = linear(bias = layers_2_second_sub_layer_out_projection_bias_to_fp16, weight = layers_2_second_sub_layer_out_projection_weight_to_fp16, x = input_53_cast_fp16)[name = tensor("linear_27_cast_fp16")]; tensor input_55_cast_fp16 = add(x = input_49_cast_fp16, y = linear_27_cast_fp16)[name = tensor("input_55_cast_fp16")]; tensor input_57_axes_0 = const()[name = tensor("input_57_axes_0"), val = tensor([-1])]; tensor layers_2_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_2_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(119649088)))]; tensor layers_2_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_2_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(119651200)))]; tensor var_777_to_fp16 = const()[name = tensor("op_777_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_57_cast_fp16 = layer_norm(axes = input_57_axes_0, beta = layers_2_layer_norm_3_bias_to_fp16, epsilon = var_777_to_fp16, gamma = layers_2_layer_norm_3_weight_to_fp16, x = input_55_cast_fp16)[name = tensor("input_57_cast_fp16")]; tensor layers_2_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_2_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(119653312)))]; tensor layers_2_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_2_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(128041984)))]; tensor linear_28_cast_fp16 = linear(bias = layers_2_third_sub_layer_dense_in_bias_to_fp16, weight = layers_2_third_sub_layer_dense_in_weight_to_fp16, x = input_57_cast_fp16)[name = tensor("linear_28_cast_fp16")]; tensor input_61_cast_fp16 = relu(x = linear_28_cast_fp16)[name = tensor("input_61_cast_fp16")]; tensor layers_2_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_2_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(128050240)))]; tensor layers_2_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_2_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(136438912)))]; tensor linear_29_cast_fp16 = linear(bias = layers_2_third_sub_layer_dense_out_bias_to_fp16, weight = layers_2_third_sub_layer_dense_out_weight_to_fp16, x = input_61_cast_fp16)[name = tensor("linear_29_cast_fp16")]; tensor input_63_cast_fp16 = add(x = input_55_cast_fp16, y = linear_29_cast_fp16)[name = tensor("input_63_cast_fp16")]; tensor input_65_axes_0 = const()[name = tensor("input_65_axes_0"), val = tensor([-1])]; tensor layers_3_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_3_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(136441024)))]; tensor layers_3_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_3_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(136443136)))]; tensor var_795_to_fp16 = const()[name = tensor("op_795_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_65_cast_fp16 = layer_norm(axes = input_65_axes_0, beta = layers_3_layer_norm_1_bias_to_fp16, epsilon = var_795_to_fp16, gamma = layers_3_layer_norm_1_weight_to_fp16, x = input_63_cast_fp16)[name = tensor("input_65_cast_fp16")]; tensor layers_3_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(136445248)))]; tensor layers_3_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(138542464)))]; tensor linear_30_cast_fp16 = linear(bias = layers_3_first_sub_layer_query_net_bias_to_fp16, weight = layers_3_first_sub_layer_query_net_weight_to_fp16, x = input_65_cast_fp16)[name = tensor("linear_30_cast_fp16")]; tensor layers_3_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(138544576)))]; tensor layers_3_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(140641792)))]; tensor linear_31_cast_fp16 = linear(bias = layers_3_first_sub_layer_key_net_bias_to_fp16, weight = layers_3_first_sub_layer_key_net_weight_to_fp16, x = input_65_cast_fp16)[name = tensor("linear_31_cast_fp16")]; tensor layers_3_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(140643904)))]; tensor layers_3_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(142741120)))]; tensor linear_32_cast_fp16 = linear(bias = layers_3_first_sub_layer_value_net_bias_to_fp16, weight = layers_3_first_sub_layer_value_net_weight_to_fp16, x = input_65_cast_fp16)[name = tensor("linear_32_cast_fp16")]; tensor var_820 = const()[name = tensor("op_820"), val = tensor([1, 1, 8, 128])]; tensor var_821_cast_fp16 = reshape(shape = var_820, x = linear_30_cast_fp16)[name = tensor("op_821_cast_fp16")]; tensor query_13_perm_0 = const()[name = tensor("query_13_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_835 = const()[name = tensor("op_835"), val = tensor([1, 1, 8, 128])]; tensor var_836_cast_fp16 = reshape(shape = var_835, x = linear_31_cast_fp16)[name = tensor("op_836_cast_fp16")]; tensor key_13_perm_0 = const()[name = tensor("key_13_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_850 = const()[name = tensor("op_850"), val = tensor([1, 1, 8, 128])]; tensor var_851_cast_fp16 = reshape(shape = var_850, x = linear_32_cast_fp16)[name = tensor("op_851_cast_fp16")]; tensor value_13_perm_0 = const()[name = tensor("value_13_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_7_axis_0 = const()[name = tensor("k_cache_new_7_axis_0"), val = tensor(2)]; tensor k_cache_new_7_mode_0 = const()[name = tensor("k_cache_new_7_mode_0"), val = tensor("update")]; tensor k_cache_new_7_validate_indices_0 = const()[name = tensor("k_cache_new_7_validate_indices_0"), val = tensor(false)]; tensor k_cache_3_to_fp16_dtype_0 = const()[name = tensor("k_cache_3_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_3_to_fp16 = cast(dtype = k_cache_3_to_fp16_dtype_0, x = k_cache_3)[name = tensor("cast_89")]; tensor key_13_cast_fp16 = transpose(perm = key_13_perm_0, x = var_836_cast_fp16)[name = tensor("transpose_86")]; tensor k_cache_new_7_cast_fp16 = scatter_along_axis(axis = k_cache_new_7_axis_0, data = k_cache_3_to_fp16, indices = pos_idx, mode = k_cache_new_7_mode_0, updates = key_13_cast_fp16, validate_indices = k_cache_new_7_validate_indices_0)[name = tensor("k_cache_new_7_cast_fp16")]; tensor k_cache_new_7_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_7_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_7_axis_0 = const()[name = tensor("v_cache_new_7_axis_0"), val = tensor(2)]; tensor v_cache_new_7_mode_0 = const()[name = tensor("v_cache_new_7_mode_0"), val = tensor("update")]; tensor v_cache_new_7_validate_indices_0 = const()[name = tensor("v_cache_new_7_validate_indices_0"), val = tensor(false)]; tensor v_cache_3_to_fp16_dtype_0 = const()[name = tensor("v_cache_3_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_3_to_fp16 = cast(dtype = v_cache_3_to_fp16_dtype_0, x = v_cache_3)[name = tensor("cast_87")]; tensor value_13_cast_fp16 = transpose(perm = value_13_perm_0, x = var_851_cast_fp16)[name = tensor("transpose_85")]; tensor v_cache_new_7_cast_fp16 = scatter_along_axis(axis = v_cache_new_7_axis_0, data = v_cache_3_to_fp16, indices = pos_idx, mode = v_cache_new_7_mode_0, updates = value_13_cast_fp16, validate_indices = v_cache_new_7_validate_indices_0)[name = tensor("v_cache_new_7_cast_fp16")]; tensor v_cache_new_7_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_7_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_861_to_fp16 = const()[name = tensor("op_861_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_13_cast_fp16 = transpose(perm = query_13_perm_0, x = var_821_cast_fp16)[name = tensor("transpose_87")]; tensor mul_6_cast_fp16 = mul(x = query_13_cast_fp16, y = var_861_to_fp16)[name = tensor("mul_6_cast_fp16")]; tensor matmul_6_transpose_y_0 = const()[name = tensor("matmul_6_transpose_y_0"), val = tensor(true)]; tensor matmul_6_transpose_x_0 = const()[name = tensor("matmul_6_transpose_x_0"), val = tensor(false)]; tensor matmul_6_cast_fp16 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = mul_6_cast_fp16, y = k_cache_new_7_cast_fp16)[name = tensor("matmul_6_cast_fp16")]; tensor add_6_cast_fp16 = add(x = matmul_6_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_6_cast_fp16")]; tensor softmax_6_axis_0 = const()[name = tensor("softmax_6_axis_0"), val = tensor(-1)]; tensor softmax_6_cast_fp16 = softmax(axis = softmax_6_axis_0, x = add_6_cast_fp16)[name = tensor("softmax_6_cast_fp16")]; tensor attn_output_19_transpose_x_0 = const()[name = tensor("attn_output_19_transpose_x_0"), val = tensor(false)]; tensor attn_output_19_transpose_y_0 = const()[name = tensor("attn_output_19_transpose_y_0"), val = tensor(false)]; tensor attn_output_19_cast_fp16 = matmul(transpose_x = attn_output_19_transpose_x_0, transpose_y = attn_output_19_transpose_y_0, x = softmax_6_cast_fp16, y = v_cache_new_7_cast_fp16)[name = tensor("attn_output_19_cast_fp16")]; tensor var_866_perm_0 = const()[name = tensor("op_866_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_872 = const()[name = tensor("op_872"), val = tensor([1, 1, 1024])]; tensor var_866_cast_fp16 = transpose(perm = var_866_perm_0, x = attn_output_19_cast_fp16)[name = tensor("transpose_84")]; tensor input_67_cast_fp16 = reshape(shape = var_872, x = var_866_cast_fp16)[name = tensor("input_67_cast_fp16")]; tensor layers_3_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(142743232)))]; tensor layers_3_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_3_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(144840448)))]; tensor linear_33_cast_fp16 = linear(bias = layers_3_first_sub_layer_out_projection_bias_to_fp16, weight = layers_3_first_sub_layer_out_projection_weight_to_fp16, x = input_67_cast_fp16)[name = tensor("linear_33_cast_fp16")]; tensor input_69_cast_fp16 = add(x = input_63_cast_fp16, y = linear_33_cast_fp16)[name = tensor("input_69_cast_fp16")]; tensor input_71_axes_0 = const()[name = tensor("input_71_axes_0"), val = tensor([-1])]; tensor layers_3_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_3_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(144842560)))]; tensor layers_3_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_3_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(144844672)))]; tensor var_880_to_fp16 = const()[name = tensor("op_880_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_71_cast_fp16 = layer_norm(axes = input_71_axes_0, beta = layers_3_layer_norm_2_bias_to_fp16, epsilon = var_880_to_fp16, gamma = layers_3_layer_norm_2_weight_to_fp16, x = input_69_cast_fp16)[name = tensor("input_71_cast_fp16")]; tensor layers_3_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(144846784)))]; tensor layers_3_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(146944000)))]; tensor linear_34_cast_fp16 = linear(bias = layers_3_second_sub_layer_query_net_bias_to_fp16, weight = layers_3_second_sub_layer_query_net_weight_to_fp16, x = input_71_cast_fp16)[name = tensor("linear_34_cast_fp16")]; tensor var_904 = const()[name = tensor("op_904"), val = tensor([1, 1, 8, 128])]; tensor var_905_cast_fp16 = reshape(shape = var_904, x = linear_34_cast_fp16)[name = tensor("op_905_cast_fp16")]; tensor layers_3_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(146946112)))]; tensor layers_3_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(149043328)))]; tensor linear_35_cast_fp16 = linear(bias = layers_3_second_sub_layer_key_net_bias_to_fp16, weight = layers_3_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_35_cast_fp16")]; tensor var_912 = const()[name = tensor("op_912"), val = tensor([1, 438, 8, 128])]; tensor var_913_cast_fp16 = reshape(shape = var_912, x = linear_35_cast_fp16)[name = tensor("op_913_cast_fp16")]; tensor layers_3_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(149045440)))]; tensor layers_3_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(151142656)))]; tensor linear_36_cast_fp16 = linear(bias = layers_3_second_sub_layer_value_net_bias_to_fp16, weight = layers_3_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_36_cast_fp16")]; tensor var_920 = const()[name = tensor("op_920"), val = tensor([1, 438, 8, 128])]; tensor var_921_cast_fp16 = reshape(shape = var_920, x = linear_36_cast_fp16)[name = tensor("op_921_cast_fp16")]; tensor value_15_perm_0 = const()[name = tensor("value_15_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_887_to_fp16 = const()[name = tensor("op_887_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_7_cast_fp16 = mul(x = var_905_cast_fp16, y = var_887_to_fp16)[name = tensor("mul_7_cast_fp16")]; tensor matmul_7_transpose_y_0 = const()[name = tensor("matmul_7_transpose_y_0"), val = tensor(true)]; tensor matmul_7_transpose_x_0 = const()[name = tensor("matmul_7_transpose_x_0"), val = tensor(false)]; tensor transpose_38_perm_0 = const()[name = tensor("transpose_38_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_39_perm_0 = const()[name = tensor("transpose_39_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_39 = transpose(perm = transpose_39_perm_0, x = var_913_cast_fp16)[name = tensor("transpose_81")]; tensor transpose_38 = transpose(perm = transpose_38_perm_0, x = mul_7_cast_fp16)[name = tensor("transpose_82")]; tensor matmul_7_cast_fp16 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = transpose_38, y = transpose_39)[name = tensor("matmul_7_cast_fp16")]; tensor add_7_cast_fp16 = add(x = matmul_7_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_7_cast_fp16")]; tensor softmax_7_axis_0 = const()[name = tensor("softmax_7_axis_0"), val = tensor(-1)]; tensor softmax_7_cast_fp16 = softmax(axis = softmax_7_axis_0, x = add_7_cast_fp16)[name = tensor("softmax_7_cast_fp16")]; tensor attn_output_23_transpose_x_0 = const()[name = tensor("attn_output_23_transpose_x_0"), val = tensor(false)]; tensor attn_output_23_transpose_y_0 = const()[name = tensor("attn_output_23_transpose_y_0"), val = tensor(false)]; tensor value_15_cast_fp16 = transpose(perm = value_15_perm_0, x = var_921_cast_fp16)[name = tensor("transpose_83")]; tensor attn_output_23_cast_fp16 = matmul(transpose_x = attn_output_23_transpose_x_0, transpose_y = attn_output_23_transpose_y_0, x = softmax_7_cast_fp16, y = value_15_cast_fp16)[name = tensor("attn_output_23_cast_fp16")]; tensor var_924_perm_0 = const()[name = tensor("op_924_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_928 = const()[name = tensor("op_928"), val = tensor([1, 1, 1024])]; tensor var_924_cast_fp16 = transpose(perm = var_924_perm_0, x = attn_output_23_cast_fp16)[name = tensor("transpose_80")]; tensor input_73_cast_fp16 = reshape(shape = var_928, x = var_924_cast_fp16)[name = tensor("input_73_cast_fp16")]; tensor layers_3_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(151144768)))]; tensor layers_3_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_3_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(153241984)))]; tensor linear_37_cast_fp16 = linear(bias = layers_3_second_sub_layer_out_projection_bias_to_fp16, weight = layers_3_second_sub_layer_out_projection_weight_to_fp16, x = input_73_cast_fp16)[name = tensor("linear_37_cast_fp16")]; tensor input_75_cast_fp16 = add(x = input_69_cast_fp16, y = linear_37_cast_fp16)[name = tensor("input_75_cast_fp16")]; tensor input_77_axes_0 = const()[name = tensor("input_77_axes_0"), val = tensor([-1])]; tensor layers_3_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_3_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(153244096)))]; tensor layers_3_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_3_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(153246208)))]; tensor var_936_to_fp16 = const()[name = tensor("op_936_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_77_cast_fp16 = layer_norm(axes = input_77_axes_0, beta = layers_3_layer_norm_3_bias_to_fp16, epsilon = var_936_to_fp16, gamma = layers_3_layer_norm_3_weight_to_fp16, x = input_75_cast_fp16)[name = tensor("input_77_cast_fp16")]; tensor layers_3_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_3_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(153248320)))]; tensor layers_3_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_3_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(161636992)))]; tensor linear_38_cast_fp16 = linear(bias = layers_3_third_sub_layer_dense_in_bias_to_fp16, weight = layers_3_third_sub_layer_dense_in_weight_to_fp16, x = input_77_cast_fp16)[name = tensor("linear_38_cast_fp16")]; tensor input_81_cast_fp16 = relu(x = linear_38_cast_fp16)[name = tensor("input_81_cast_fp16")]; tensor layers_3_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_3_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(161645248)))]; tensor layers_3_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_3_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(170033920)))]; tensor linear_39_cast_fp16 = linear(bias = layers_3_third_sub_layer_dense_out_bias_to_fp16, weight = layers_3_third_sub_layer_dense_out_weight_to_fp16, x = input_81_cast_fp16)[name = tensor("linear_39_cast_fp16")]; tensor input_83_cast_fp16 = add(x = input_75_cast_fp16, y = linear_39_cast_fp16)[name = tensor("input_83_cast_fp16")]; tensor input_85_axes_0 = const()[name = tensor("input_85_axes_0"), val = tensor([-1])]; tensor layers_4_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_4_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(170036032)))]; tensor layers_4_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_4_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(170038144)))]; tensor var_954_to_fp16 = const()[name = tensor("op_954_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_85_cast_fp16 = layer_norm(axes = input_85_axes_0, beta = layers_4_layer_norm_1_bias_to_fp16, epsilon = var_954_to_fp16, gamma = layers_4_layer_norm_1_weight_to_fp16, x = input_83_cast_fp16)[name = tensor("input_85_cast_fp16")]; tensor layers_4_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(170040256)))]; tensor layers_4_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(172137472)))]; tensor linear_40_cast_fp16 = linear(bias = layers_4_first_sub_layer_query_net_bias_to_fp16, weight = layers_4_first_sub_layer_query_net_weight_to_fp16, x = input_85_cast_fp16)[name = tensor("linear_40_cast_fp16")]; tensor layers_4_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(172139584)))]; tensor layers_4_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(174236800)))]; tensor linear_41_cast_fp16 = linear(bias = layers_4_first_sub_layer_key_net_bias_to_fp16, weight = layers_4_first_sub_layer_key_net_weight_to_fp16, x = input_85_cast_fp16)[name = tensor("linear_41_cast_fp16")]; tensor layers_4_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(174238912)))]; tensor layers_4_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(176336128)))]; tensor linear_42_cast_fp16 = linear(bias = layers_4_first_sub_layer_value_net_bias_to_fp16, weight = layers_4_first_sub_layer_value_net_weight_to_fp16, x = input_85_cast_fp16)[name = tensor("linear_42_cast_fp16")]; tensor var_979 = const()[name = tensor("op_979"), val = tensor([1, 1, 8, 128])]; tensor var_980_cast_fp16 = reshape(shape = var_979, x = linear_40_cast_fp16)[name = tensor("op_980_cast_fp16")]; tensor query_17_perm_0 = const()[name = tensor("query_17_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_994 = const()[name = tensor("op_994"), val = tensor([1, 1, 8, 128])]; tensor var_995_cast_fp16 = reshape(shape = var_994, x = linear_41_cast_fp16)[name = tensor("op_995_cast_fp16")]; tensor key_17_perm_0 = const()[name = tensor("key_17_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1009 = const()[name = tensor("op_1009"), val = tensor([1, 1, 8, 128])]; tensor var_1010_cast_fp16 = reshape(shape = var_1009, x = linear_42_cast_fp16)[name = tensor("op_1010_cast_fp16")]; tensor value_17_perm_0 = const()[name = tensor("value_17_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_9_axis_0 = const()[name = tensor("k_cache_new_9_axis_0"), val = tensor(2)]; tensor k_cache_new_9_mode_0 = const()[name = tensor("k_cache_new_9_mode_0"), val = tensor("update")]; tensor k_cache_new_9_validate_indices_0 = const()[name = tensor("k_cache_new_9_validate_indices_0"), val = tensor(false)]; tensor k_cache_4_to_fp16_dtype_0 = const()[name = tensor("k_cache_4_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_4_to_fp16 = cast(dtype = k_cache_4_to_fp16_dtype_0, x = k_cache_4)[name = tensor("cast_85")]; tensor key_17_cast_fp16 = transpose(perm = key_17_perm_0, x = var_995_cast_fp16)[name = tensor("transpose_78")]; tensor k_cache_new_9_cast_fp16 = scatter_along_axis(axis = k_cache_new_9_axis_0, data = k_cache_4_to_fp16, indices = pos_idx, mode = k_cache_new_9_mode_0, updates = key_17_cast_fp16, validate_indices = k_cache_new_9_validate_indices_0)[name = tensor("k_cache_new_9_cast_fp16")]; tensor k_cache_new_9_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_9_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_9_axis_0 = const()[name = tensor("v_cache_new_9_axis_0"), val = tensor(2)]; tensor v_cache_new_9_mode_0 = const()[name = tensor("v_cache_new_9_mode_0"), val = tensor("update")]; tensor v_cache_new_9_validate_indices_0 = const()[name = tensor("v_cache_new_9_validate_indices_0"), val = tensor(false)]; tensor v_cache_4_to_fp16_dtype_0 = const()[name = tensor("v_cache_4_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_4_to_fp16 = cast(dtype = v_cache_4_to_fp16_dtype_0, x = v_cache_4)[name = tensor("cast_83")]; tensor value_17_cast_fp16 = transpose(perm = value_17_perm_0, x = var_1010_cast_fp16)[name = tensor("transpose_77")]; tensor v_cache_new_9_cast_fp16 = scatter_along_axis(axis = v_cache_new_9_axis_0, data = v_cache_4_to_fp16, indices = pos_idx, mode = v_cache_new_9_mode_0, updates = value_17_cast_fp16, validate_indices = v_cache_new_9_validate_indices_0)[name = tensor("v_cache_new_9_cast_fp16")]; tensor v_cache_new_9_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_9_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1020_to_fp16 = const()[name = tensor("op_1020_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_17_cast_fp16 = transpose(perm = query_17_perm_0, x = var_980_cast_fp16)[name = tensor("transpose_79")]; tensor mul_8_cast_fp16 = mul(x = query_17_cast_fp16, y = var_1020_to_fp16)[name = tensor("mul_8_cast_fp16")]; tensor matmul_8_transpose_y_0 = const()[name = tensor("matmul_8_transpose_y_0"), val = tensor(true)]; tensor matmul_8_transpose_x_0 = const()[name = tensor("matmul_8_transpose_x_0"), val = tensor(false)]; tensor matmul_8_cast_fp16 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = mul_8_cast_fp16, y = k_cache_new_9_cast_fp16)[name = tensor("matmul_8_cast_fp16")]; tensor add_8_cast_fp16 = add(x = matmul_8_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_8_cast_fp16")]; tensor softmax_8_axis_0 = const()[name = tensor("softmax_8_axis_0"), val = tensor(-1)]; tensor softmax_8_cast_fp16 = softmax(axis = softmax_8_axis_0, x = add_8_cast_fp16)[name = tensor("softmax_8_cast_fp16")]; tensor attn_output_25_transpose_x_0 = const()[name = tensor("attn_output_25_transpose_x_0"), val = tensor(false)]; tensor attn_output_25_transpose_y_0 = const()[name = tensor("attn_output_25_transpose_y_0"), val = tensor(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = softmax_8_cast_fp16, y = v_cache_new_9_cast_fp16)[name = tensor("attn_output_25_cast_fp16")]; tensor var_1025_perm_0 = const()[name = tensor("op_1025_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1031 = const()[name = tensor("op_1031"), val = tensor([1, 1, 1024])]; tensor var_1025_cast_fp16 = transpose(perm = var_1025_perm_0, x = attn_output_25_cast_fp16)[name = tensor("transpose_76")]; tensor input_87_cast_fp16 = reshape(shape = var_1031, x = var_1025_cast_fp16)[name = tensor("input_87_cast_fp16")]; tensor layers_4_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(176338240)))]; tensor layers_4_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_4_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(178435456)))]; tensor linear_43_cast_fp16 = linear(bias = layers_4_first_sub_layer_out_projection_bias_to_fp16, weight = layers_4_first_sub_layer_out_projection_weight_to_fp16, x = input_87_cast_fp16)[name = tensor("linear_43_cast_fp16")]; tensor input_89_cast_fp16 = add(x = input_83_cast_fp16, y = linear_43_cast_fp16)[name = tensor("input_89_cast_fp16")]; tensor input_91_axes_0 = const()[name = tensor("input_91_axes_0"), val = tensor([-1])]; tensor layers_4_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_4_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(178437568)))]; tensor layers_4_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_4_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(178439680)))]; tensor var_1039_to_fp16 = const()[name = tensor("op_1039_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_91_cast_fp16 = layer_norm(axes = input_91_axes_0, beta = layers_4_layer_norm_2_bias_to_fp16, epsilon = var_1039_to_fp16, gamma = layers_4_layer_norm_2_weight_to_fp16, x = input_89_cast_fp16)[name = tensor("input_91_cast_fp16")]; tensor layers_4_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(178441792)))]; tensor layers_4_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(180539008)))]; tensor linear_44_cast_fp16 = linear(bias = layers_4_second_sub_layer_query_net_bias_to_fp16, weight = layers_4_second_sub_layer_query_net_weight_to_fp16, x = input_91_cast_fp16)[name = tensor("linear_44_cast_fp16")]; tensor var_1063 = const()[name = tensor("op_1063"), val = tensor([1, 1, 8, 128])]; tensor var_1064_cast_fp16 = reshape(shape = var_1063, x = linear_44_cast_fp16)[name = tensor("op_1064_cast_fp16")]; tensor layers_4_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(180541120)))]; tensor layers_4_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(182638336)))]; tensor linear_45_cast_fp16 = linear(bias = layers_4_second_sub_layer_key_net_bias_to_fp16, weight = layers_4_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_45_cast_fp16")]; tensor var_1071 = const()[name = tensor("op_1071"), val = tensor([1, 438, 8, 128])]; tensor var_1072_cast_fp16 = reshape(shape = var_1071, x = linear_45_cast_fp16)[name = tensor("op_1072_cast_fp16")]; tensor layers_4_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(182640448)))]; tensor layers_4_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(184737664)))]; tensor linear_46_cast_fp16 = linear(bias = layers_4_second_sub_layer_value_net_bias_to_fp16, weight = layers_4_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_46_cast_fp16")]; tensor var_1079 = const()[name = tensor("op_1079"), val = tensor([1, 438, 8, 128])]; tensor var_1080_cast_fp16 = reshape(shape = var_1079, x = linear_46_cast_fp16)[name = tensor("op_1080_cast_fp16")]; tensor value_19_perm_0 = const()[name = tensor("value_19_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1046_to_fp16 = const()[name = tensor("op_1046_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_9_cast_fp16 = mul(x = var_1064_cast_fp16, y = var_1046_to_fp16)[name = tensor("mul_9_cast_fp16")]; tensor matmul_9_transpose_y_0 = const()[name = tensor("matmul_9_transpose_y_0"), val = tensor(true)]; tensor matmul_9_transpose_x_0 = const()[name = tensor("matmul_9_transpose_x_0"), val = tensor(false)]; tensor transpose_40_perm_0 = const()[name = tensor("transpose_40_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_41_perm_0 = const()[name = tensor("transpose_41_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_41 = transpose(perm = transpose_41_perm_0, x = var_1072_cast_fp16)[name = tensor("transpose_73")]; tensor transpose_40 = transpose(perm = transpose_40_perm_0, x = mul_9_cast_fp16)[name = tensor("transpose_74")]; tensor matmul_9_cast_fp16 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = transpose_40, y = transpose_41)[name = tensor("matmul_9_cast_fp16")]; tensor add_9_cast_fp16 = add(x = matmul_9_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_9_cast_fp16")]; tensor softmax_9_axis_0 = const()[name = tensor("softmax_9_axis_0"), val = tensor(-1)]; tensor softmax_9_cast_fp16 = softmax(axis = softmax_9_axis_0, x = add_9_cast_fp16)[name = tensor("softmax_9_cast_fp16")]; tensor attn_output_29_transpose_x_0 = const()[name = tensor("attn_output_29_transpose_x_0"), val = tensor(false)]; tensor attn_output_29_transpose_y_0 = const()[name = tensor("attn_output_29_transpose_y_0"), val = tensor(false)]; tensor value_19_cast_fp16 = transpose(perm = value_19_perm_0, x = var_1080_cast_fp16)[name = tensor("transpose_75")]; tensor attn_output_29_cast_fp16 = matmul(transpose_x = attn_output_29_transpose_x_0, transpose_y = attn_output_29_transpose_y_0, x = softmax_9_cast_fp16, y = value_19_cast_fp16)[name = tensor("attn_output_29_cast_fp16")]; tensor var_1083_perm_0 = const()[name = tensor("op_1083_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1087 = const()[name = tensor("op_1087"), val = tensor([1, 1, 1024])]; tensor var_1083_cast_fp16 = transpose(perm = var_1083_perm_0, x = attn_output_29_cast_fp16)[name = tensor("transpose_72")]; tensor input_93_cast_fp16 = reshape(shape = var_1087, x = var_1083_cast_fp16)[name = tensor("input_93_cast_fp16")]; tensor layers_4_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(184739776)))]; tensor layers_4_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_4_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(186836992)))]; tensor linear_47_cast_fp16 = linear(bias = layers_4_second_sub_layer_out_projection_bias_to_fp16, weight = layers_4_second_sub_layer_out_projection_weight_to_fp16, x = input_93_cast_fp16)[name = tensor("linear_47_cast_fp16")]; tensor input_95_cast_fp16 = add(x = input_89_cast_fp16, y = linear_47_cast_fp16)[name = tensor("input_95_cast_fp16")]; tensor input_97_axes_0 = const()[name = tensor("input_97_axes_0"), val = tensor([-1])]; tensor layers_4_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_4_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(186839104)))]; tensor layers_4_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_4_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(186841216)))]; tensor var_1095_to_fp16 = const()[name = tensor("op_1095_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_97_cast_fp16 = layer_norm(axes = input_97_axes_0, beta = layers_4_layer_norm_3_bias_to_fp16, epsilon = var_1095_to_fp16, gamma = layers_4_layer_norm_3_weight_to_fp16, x = input_95_cast_fp16)[name = tensor("input_97_cast_fp16")]; tensor layers_4_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_4_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(186843328)))]; tensor layers_4_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_4_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(195232000)))]; tensor linear_48_cast_fp16 = linear(bias = layers_4_third_sub_layer_dense_in_bias_to_fp16, weight = layers_4_third_sub_layer_dense_in_weight_to_fp16, x = input_97_cast_fp16)[name = tensor("linear_48_cast_fp16")]; tensor input_101_cast_fp16 = relu(x = linear_48_cast_fp16)[name = tensor("input_101_cast_fp16")]; tensor layers_4_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_4_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(195240256)))]; tensor layers_4_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_4_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(203628928)))]; tensor linear_49_cast_fp16 = linear(bias = layers_4_third_sub_layer_dense_out_bias_to_fp16, weight = layers_4_third_sub_layer_dense_out_weight_to_fp16, x = input_101_cast_fp16)[name = tensor("linear_49_cast_fp16")]; tensor input_103_cast_fp16 = add(x = input_95_cast_fp16, y = linear_49_cast_fp16)[name = tensor("input_103_cast_fp16")]; tensor input_105_axes_0 = const()[name = tensor("input_105_axes_0"), val = tensor([-1])]; tensor layers_5_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_5_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(203631040)))]; tensor layers_5_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_5_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(203633152)))]; tensor var_1113_to_fp16 = const()[name = tensor("op_1113_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_105_cast_fp16 = layer_norm(axes = input_105_axes_0, beta = layers_5_layer_norm_1_bias_to_fp16, epsilon = var_1113_to_fp16, gamma = layers_5_layer_norm_1_weight_to_fp16, x = input_103_cast_fp16)[name = tensor("input_105_cast_fp16")]; tensor layers_5_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(203635264)))]; tensor layers_5_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(205732480)))]; tensor linear_50_cast_fp16 = linear(bias = layers_5_first_sub_layer_query_net_bias_to_fp16, weight = layers_5_first_sub_layer_query_net_weight_to_fp16, x = input_105_cast_fp16)[name = tensor("linear_50_cast_fp16")]; tensor layers_5_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(205734592)))]; tensor layers_5_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(207831808)))]; tensor linear_51_cast_fp16 = linear(bias = layers_5_first_sub_layer_key_net_bias_to_fp16, weight = layers_5_first_sub_layer_key_net_weight_to_fp16, x = input_105_cast_fp16)[name = tensor("linear_51_cast_fp16")]; tensor layers_5_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(207833920)))]; tensor layers_5_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(209931136)))]; tensor linear_52_cast_fp16 = linear(bias = layers_5_first_sub_layer_value_net_bias_to_fp16, weight = layers_5_first_sub_layer_value_net_weight_to_fp16, x = input_105_cast_fp16)[name = tensor("linear_52_cast_fp16")]; tensor var_1138 = const()[name = tensor("op_1138"), val = tensor([1, 1, 8, 128])]; tensor var_1139_cast_fp16 = reshape(shape = var_1138, x = linear_50_cast_fp16)[name = tensor("op_1139_cast_fp16")]; tensor query_21_perm_0 = const()[name = tensor("query_21_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1153 = const()[name = tensor("op_1153"), val = tensor([1, 1, 8, 128])]; tensor var_1154_cast_fp16 = reshape(shape = var_1153, x = linear_51_cast_fp16)[name = tensor("op_1154_cast_fp16")]; tensor key_21_perm_0 = const()[name = tensor("key_21_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1168 = const()[name = tensor("op_1168"), val = tensor([1, 1, 8, 128])]; tensor var_1169_cast_fp16 = reshape(shape = var_1168, x = linear_52_cast_fp16)[name = tensor("op_1169_cast_fp16")]; tensor value_21_perm_0 = const()[name = tensor("value_21_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_11_axis_0 = const()[name = tensor("k_cache_new_11_axis_0"), val = tensor(2)]; tensor k_cache_new_11_mode_0 = const()[name = tensor("k_cache_new_11_mode_0"), val = tensor("update")]; tensor k_cache_new_11_validate_indices_0 = const()[name = tensor("k_cache_new_11_validate_indices_0"), val = tensor(false)]; tensor k_cache_5_to_fp16_dtype_0 = const()[name = tensor("k_cache_5_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_5_to_fp16 = cast(dtype = k_cache_5_to_fp16_dtype_0, x = k_cache_5)[name = tensor("cast_81")]; tensor key_21_cast_fp16 = transpose(perm = key_21_perm_0, x = var_1154_cast_fp16)[name = tensor("transpose_70")]; tensor k_cache_new_11_cast_fp16 = scatter_along_axis(axis = k_cache_new_11_axis_0, data = k_cache_5_to_fp16, indices = pos_idx, mode = k_cache_new_11_mode_0, updates = key_21_cast_fp16, validate_indices = k_cache_new_11_validate_indices_0)[name = tensor("k_cache_new_11_cast_fp16")]; tensor k_cache_new_11_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_11_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_11_axis_0 = const()[name = tensor("v_cache_new_11_axis_0"), val = tensor(2)]; tensor v_cache_new_11_mode_0 = const()[name = tensor("v_cache_new_11_mode_0"), val = tensor("update")]; tensor v_cache_new_11_validate_indices_0 = const()[name = tensor("v_cache_new_11_validate_indices_0"), val = tensor(false)]; tensor v_cache_5_to_fp16_dtype_0 = const()[name = tensor("v_cache_5_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_5_to_fp16 = cast(dtype = v_cache_5_to_fp16_dtype_0, x = v_cache_5)[name = tensor("cast_79")]; tensor value_21_cast_fp16 = transpose(perm = value_21_perm_0, x = var_1169_cast_fp16)[name = tensor("transpose_69")]; tensor v_cache_new_11_cast_fp16 = scatter_along_axis(axis = v_cache_new_11_axis_0, data = v_cache_5_to_fp16, indices = pos_idx, mode = v_cache_new_11_mode_0, updates = value_21_cast_fp16, validate_indices = v_cache_new_11_validate_indices_0)[name = tensor("v_cache_new_11_cast_fp16")]; tensor v_cache_new_11_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_11_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1179_to_fp16 = const()[name = tensor("op_1179_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_21_cast_fp16 = transpose(perm = query_21_perm_0, x = var_1139_cast_fp16)[name = tensor("transpose_71")]; tensor mul_10_cast_fp16 = mul(x = query_21_cast_fp16, y = var_1179_to_fp16)[name = tensor("mul_10_cast_fp16")]; tensor matmul_10_transpose_y_0 = const()[name = tensor("matmul_10_transpose_y_0"), val = tensor(true)]; tensor matmul_10_transpose_x_0 = const()[name = tensor("matmul_10_transpose_x_0"), val = tensor(false)]; tensor matmul_10_cast_fp16 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = mul_10_cast_fp16, y = k_cache_new_11_cast_fp16)[name = tensor("matmul_10_cast_fp16")]; tensor add_10_cast_fp16 = add(x = matmul_10_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_10_cast_fp16")]; tensor softmax_10_axis_0 = const()[name = tensor("softmax_10_axis_0"), val = tensor(-1)]; tensor softmax_10_cast_fp16 = softmax(axis = softmax_10_axis_0, x = add_10_cast_fp16)[name = tensor("softmax_10_cast_fp16")]; tensor attn_output_31_transpose_x_0 = const()[name = tensor("attn_output_31_transpose_x_0"), val = tensor(false)]; tensor attn_output_31_transpose_y_0 = const()[name = tensor("attn_output_31_transpose_y_0"), val = tensor(false)]; tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = softmax_10_cast_fp16, y = v_cache_new_11_cast_fp16)[name = tensor("attn_output_31_cast_fp16")]; tensor var_1184_perm_0 = const()[name = tensor("op_1184_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1190 = const()[name = tensor("op_1190"), val = tensor([1, 1, 1024])]; tensor var_1184_cast_fp16 = transpose(perm = var_1184_perm_0, x = attn_output_31_cast_fp16)[name = tensor("transpose_68")]; tensor input_107_cast_fp16 = reshape(shape = var_1190, x = var_1184_cast_fp16)[name = tensor("input_107_cast_fp16")]; tensor layers_5_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(209933248)))]; tensor layers_5_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_5_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(212030464)))]; tensor linear_53_cast_fp16 = linear(bias = layers_5_first_sub_layer_out_projection_bias_to_fp16, weight = layers_5_first_sub_layer_out_projection_weight_to_fp16, x = input_107_cast_fp16)[name = tensor("linear_53_cast_fp16")]; tensor input_109_cast_fp16 = add(x = input_103_cast_fp16, y = linear_53_cast_fp16)[name = tensor("input_109_cast_fp16")]; tensor input_111_axes_0 = const()[name = tensor("input_111_axes_0"), val = tensor([-1])]; tensor layers_5_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_5_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(212032576)))]; tensor layers_5_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_5_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(212034688)))]; tensor var_1198_to_fp16 = const()[name = tensor("op_1198_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_111_cast_fp16 = layer_norm(axes = input_111_axes_0, beta = layers_5_layer_norm_2_bias_to_fp16, epsilon = var_1198_to_fp16, gamma = layers_5_layer_norm_2_weight_to_fp16, x = input_109_cast_fp16)[name = tensor("input_111_cast_fp16")]; tensor layers_5_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(212036800)))]; tensor layers_5_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(214134016)))]; tensor linear_54_cast_fp16 = linear(bias = layers_5_second_sub_layer_query_net_bias_to_fp16, weight = layers_5_second_sub_layer_query_net_weight_to_fp16, x = input_111_cast_fp16)[name = tensor("linear_54_cast_fp16")]; tensor var_1222 = const()[name = tensor("op_1222"), val = tensor([1, 1, 8, 128])]; tensor var_1223_cast_fp16 = reshape(shape = var_1222, x = linear_54_cast_fp16)[name = tensor("op_1223_cast_fp16")]; tensor layers_5_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(214136128)))]; tensor layers_5_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(216233344)))]; tensor linear_55_cast_fp16 = linear(bias = layers_5_second_sub_layer_key_net_bias_to_fp16, weight = layers_5_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_55_cast_fp16")]; tensor var_1230 = const()[name = tensor("op_1230"), val = tensor([1, 438, 8, 128])]; tensor var_1231_cast_fp16 = reshape(shape = var_1230, x = linear_55_cast_fp16)[name = tensor("op_1231_cast_fp16")]; tensor layers_5_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(216235456)))]; tensor layers_5_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(218332672)))]; tensor linear_56_cast_fp16 = linear(bias = layers_5_second_sub_layer_value_net_bias_to_fp16, weight = layers_5_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_56_cast_fp16")]; tensor var_1238 = const()[name = tensor("op_1238"), val = tensor([1, 438, 8, 128])]; tensor var_1239_cast_fp16 = reshape(shape = var_1238, x = linear_56_cast_fp16)[name = tensor("op_1239_cast_fp16")]; tensor value_23_perm_0 = const()[name = tensor("value_23_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1205_to_fp16 = const()[name = tensor("op_1205_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_11_cast_fp16 = mul(x = var_1223_cast_fp16, y = var_1205_to_fp16)[name = tensor("mul_11_cast_fp16")]; tensor matmul_11_transpose_y_0 = const()[name = tensor("matmul_11_transpose_y_0"), val = tensor(true)]; tensor matmul_11_transpose_x_0 = const()[name = tensor("matmul_11_transpose_x_0"), val = tensor(false)]; tensor transpose_42_perm_0 = const()[name = tensor("transpose_42_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_43_perm_0 = const()[name = tensor("transpose_43_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_43 = transpose(perm = transpose_43_perm_0, x = var_1231_cast_fp16)[name = tensor("transpose_65")]; tensor transpose_42 = transpose(perm = transpose_42_perm_0, x = mul_11_cast_fp16)[name = tensor("transpose_66")]; tensor matmul_11_cast_fp16 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = transpose_42, y = transpose_43)[name = tensor("matmul_11_cast_fp16")]; tensor add_11_cast_fp16 = add(x = matmul_11_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_11_cast_fp16")]; tensor softmax_11_axis_0 = const()[name = tensor("softmax_11_axis_0"), val = tensor(-1)]; tensor softmax_11_cast_fp16 = softmax(axis = softmax_11_axis_0, x = add_11_cast_fp16)[name = tensor("softmax_11_cast_fp16")]; tensor attn_output_35_transpose_x_0 = const()[name = tensor("attn_output_35_transpose_x_0"), val = tensor(false)]; tensor attn_output_35_transpose_y_0 = const()[name = tensor("attn_output_35_transpose_y_0"), val = tensor(false)]; tensor value_23_cast_fp16 = transpose(perm = value_23_perm_0, x = var_1239_cast_fp16)[name = tensor("transpose_67")]; tensor attn_output_35_cast_fp16 = matmul(transpose_x = attn_output_35_transpose_x_0, transpose_y = attn_output_35_transpose_y_0, x = softmax_11_cast_fp16, y = value_23_cast_fp16)[name = tensor("attn_output_35_cast_fp16")]; tensor var_1242_perm_0 = const()[name = tensor("op_1242_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1246 = const()[name = tensor("op_1246"), val = tensor([1, 1, 1024])]; tensor var_1242_cast_fp16 = transpose(perm = var_1242_perm_0, x = attn_output_35_cast_fp16)[name = tensor("transpose_64")]; tensor input_113_cast_fp16 = reshape(shape = var_1246, x = var_1242_cast_fp16)[name = tensor("input_113_cast_fp16")]; tensor layers_5_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(218334784)))]; tensor layers_5_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_5_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(220432000)))]; tensor linear_57_cast_fp16 = linear(bias = layers_5_second_sub_layer_out_projection_bias_to_fp16, weight = layers_5_second_sub_layer_out_projection_weight_to_fp16, x = input_113_cast_fp16)[name = tensor("linear_57_cast_fp16")]; tensor input_115_cast_fp16 = add(x = input_109_cast_fp16, y = linear_57_cast_fp16)[name = tensor("input_115_cast_fp16")]; tensor input_117_axes_0 = const()[name = tensor("input_117_axes_0"), val = tensor([-1])]; tensor layers_5_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_5_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(220434112)))]; tensor layers_5_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_5_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(220436224)))]; tensor var_1254_to_fp16 = const()[name = tensor("op_1254_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_117_cast_fp16 = layer_norm(axes = input_117_axes_0, beta = layers_5_layer_norm_3_bias_to_fp16, epsilon = var_1254_to_fp16, gamma = layers_5_layer_norm_3_weight_to_fp16, x = input_115_cast_fp16)[name = tensor("input_117_cast_fp16")]; tensor layers_5_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_5_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(220438336)))]; tensor layers_5_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_5_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(228827008)))]; tensor linear_58_cast_fp16 = linear(bias = layers_5_third_sub_layer_dense_in_bias_to_fp16, weight = layers_5_third_sub_layer_dense_in_weight_to_fp16, x = input_117_cast_fp16)[name = tensor("linear_58_cast_fp16")]; tensor input_121_cast_fp16 = relu(x = linear_58_cast_fp16)[name = tensor("input_121_cast_fp16")]; tensor layers_5_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_5_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(228835264)))]; tensor layers_5_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_5_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(237223936)))]; tensor linear_59_cast_fp16 = linear(bias = layers_5_third_sub_layer_dense_out_bias_to_fp16, weight = layers_5_third_sub_layer_dense_out_weight_to_fp16, x = input_121_cast_fp16)[name = tensor("linear_59_cast_fp16")]; tensor input_123_cast_fp16 = add(x = input_115_cast_fp16, y = linear_59_cast_fp16)[name = tensor("input_123_cast_fp16")]; tensor input_125_axes_0 = const()[name = tensor("input_125_axes_0"), val = tensor([-1])]; tensor layers_6_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_6_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(237226048)))]; tensor layers_6_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_6_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(237228160)))]; tensor var_1272_to_fp16 = const()[name = tensor("op_1272_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_125_cast_fp16 = layer_norm(axes = input_125_axes_0, beta = layers_6_layer_norm_1_bias_to_fp16, epsilon = var_1272_to_fp16, gamma = layers_6_layer_norm_1_weight_to_fp16, x = input_123_cast_fp16)[name = tensor("input_125_cast_fp16")]; tensor layers_6_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(237230272)))]; tensor layers_6_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(239327488)))]; tensor linear_60_cast_fp16 = linear(bias = layers_6_first_sub_layer_query_net_bias_to_fp16, weight = layers_6_first_sub_layer_query_net_weight_to_fp16, x = input_125_cast_fp16)[name = tensor("linear_60_cast_fp16")]; tensor layers_6_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(239329600)))]; tensor layers_6_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(241426816)))]; tensor linear_61_cast_fp16 = linear(bias = layers_6_first_sub_layer_key_net_bias_to_fp16, weight = layers_6_first_sub_layer_key_net_weight_to_fp16, x = input_125_cast_fp16)[name = tensor("linear_61_cast_fp16")]; tensor layers_6_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(241428928)))]; tensor layers_6_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(243526144)))]; tensor linear_62_cast_fp16 = linear(bias = layers_6_first_sub_layer_value_net_bias_to_fp16, weight = layers_6_first_sub_layer_value_net_weight_to_fp16, x = input_125_cast_fp16)[name = tensor("linear_62_cast_fp16")]; tensor var_1297 = const()[name = tensor("op_1297"), val = tensor([1, 1, 8, 128])]; tensor var_1298_cast_fp16 = reshape(shape = var_1297, x = linear_60_cast_fp16)[name = tensor("op_1298_cast_fp16")]; tensor query_25_perm_0 = const()[name = tensor("query_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1312 = const()[name = tensor("op_1312"), val = tensor([1, 1, 8, 128])]; tensor var_1313_cast_fp16 = reshape(shape = var_1312, x = linear_61_cast_fp16)[name = tensor("op_1313_cast_fp16")]; tensor key_25_perm_0 = const()[name = tensor("key_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1327 = const()[name = tensor("op_1327"), val = tensor([1, 1, 8, 128])]; tensor var_1328_cast_fp16 = reshape(shape = var_1327, x = linear_62_cast_fp16)[name = tensor("op_1328_cast_fp16")]; tensor value_25_perm_0 = const()[name = tensor("value_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_13_axis_0 = const()[name = tensor("k_cache_new_13_axis_0"), val = tensor(2)]; tensor k_cache_new_13_mode_0 = const()[name = tensor("k_cache_new_13_mode_0"), val = tensor("update")]; tensor k_cache_new_13_validate_indices_0 = const()[name = tensor("k_cache_new_13_validate_indices_0"), val = tensor(false)]; tensor k_cache_6_to_fp16_dtype_0 = const()[name = tensor("k_cache_6_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_6_to_fp16 = cast(dtype = k_cache_6_to_fp16_dtype_0, x = k_cache_6)[name = tensor("cast_77")]; tensor key_25_cast_fp16 = transpose(perm = key_25_perm_0, x = var_1313_cast_fp16)[name = tensor("transpose_62")]; tensor k_cache_new_13_cast_fp16 = scatter_along_axis(axis = k_cache_new_13_axis_0, data = k_cache_6_to_fp16, indices = pos_idx, mode = k_cache_new_13_mode_0, updates = key_25_cast_fp16, validate_indices = k_cache_new_13_validate_indices_0)[name = tensor("k_cache_new_13_cast_fp16")]; tensor k_cache_new_13_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_13_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_13_axis_0 = const()[name = tensor("v_cache_new_13_axis_0"), val = tensor(2)]; tensor v_cache_new_13_mode_0 = const()[name = tensor("v_cache_new_13_mode_0"), val = tensor("update")]; tensor v_cache_new_13_validate_indices_0 = const()[name = tensor("v_cache_new_13_validate_indices_0"), val = tensor(false)]; tensor v_cache_6_to_fp16_dtype_0 = const()[name = tensor("v_cache_6_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_6_to_fp16 = cast(dtype = v_cache_6_to_fp16_dtype_0, x = v_cache_6)[name = tensor("cast_75")]; tensor value_25_cast_fp16 = transpose(perm = value_25_perm_0, x = var_1328_cast_fp16)[name = tensor("transpose_61")]; tensor v_cache_new_13_cast_fp16 = scatter_along_axis(axis = v_cache_new_13_axis_0, data = v_cache_6_to_fp16, indices = pos_idx, mode = v_cache_new_13_mode_0, updates = value_25_cast_fp16, validate_indices = v_cache_new_13_validate_indices_0)[name = tensor("v_cache_new_13_cast_fp16")]; tensor v_cache_new_13_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_13_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1338_to_fp16 = const()[name = tensor("op_1338_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_25_cast_fp16 = transpose(perm = query_25_perm_0, x = var_1298_cast_fp16)[name = tensor("transpose_63")]; tensor mul_12_cast_fp16 = mul(x = query_25_cast_fp16, y = var_1338_to_fp16)[name = tensor("mul_12_cast_fp16")]; tensor matmul_12_transpose_y_0 = const()[name = tensor("matmul_12_transpose_y_0"), val = tensor(true)]; tensor matmul_12_transpose_x_0 = const()[name = tensor("matmul_12_transpose_x_0"), val = tensor(false)]; tensor matmul_12_cast_fp16 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = mul_12_cast_fp16, y = k_cache_new_13_cast_fp16)[name = tensor("matmul_12_cast_fp16")]; tensor add_12_cast_fp16 = add(x = matmul_12_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_12_cast_fp16")]; tensor softmax_12_axis_0 = const()[name = tensor("softmax_12_axis_0"), val = tensor(-1)]; tensor softmax_12_cast_fp16 = softmax(axis = softmax_12_axis_0, x = add_12_cast_fp16)[name = tensor("softmax_12_cast_fp16")]; tensor attn_output_37_transpose_x_0 = const()[name = tensor("attn_output_37_transpose_x_0"), val = tensor(false)]; tensor attn_output_37_transpose_y_0 = const()[name = tensor("attn_output_37_transpose_y_0"), val = tensor(false)]; tensor attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = softmax_12_cast_fp16, y = v_cache_new_13_cast_fp16)[name = tensor("attn_output_37_cast_fp16")]; tensor var_1343_perm_0 = const()[name = tensor("op_1343_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1349 = const()[name = tensor("op_1349"), val = tensor([1, 1, 1024])]; tensor var_1343_cast_fp16 = transpose(perm = var_1343_perm_0, x = attn_output_37_cast_fp16)[name = tensor("transpose_60")]; tensor input_127_cast_fp16 = reshape(shape = var_1349, x = var_1343_cast_fp16)[name = tensor("input_127_cast_fp16")]; tensor layers_6_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(243528256)))]; tensor layers_6_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_6_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(245625472)))]; tensor linear_63_cast_fp16 = linear(bias = layers_6_first_sub_layer_out_projection_bias_to_fp16, weight = layers_6_first_sub_layer_out_projection_weight_to_fp16, x = input_127_cast_fp16)[name = tensor("linear_63_cast_fp16")]; tensor input_129_cast_fp16 = add(x = input_123_cast_fp16, y = linear_63_cast_fp16)[name = tensor("input_129_cast_fp16")]; tensor input_131_axes_0 = const()[name = tensor("input_131_axes_0"), val = tensor([-1])]; tensor layers_6_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_6_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(245627584)))]; tensor layers_6_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_6_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(245629696)))]; tensor var_1357_to_fp16 = const()[name = tensor("op_1357_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_131_cast_fp16 = layer_norm(axes = input_131_axes_0, beta = layers_6_layer_norm_2_bias_to_fp16, epsilon = var_1357_to_fp16, gamma = layers_6_layer_norm_2_weight_to_fp16, x = input_129_cast_fp16)[name = tensor("input_131_cast_fp16")]; tensor layers_6_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(245631808)))]; tensor layers_6_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(247729024)))]; tensor linear_64_cast_fp16 = linear(bias = layers_6_second_sub_layer_query_net_bias_to_fp16, weight = layers_6_second_sub_layer_query_net_weight_to_fp16, x = input_131_cast_fp16)[name = tensor("linear_64_cast_fp16")]; tensor var_1381 = const()[name = tensor("op_1381"), val = tensor([1, 1, 8, 128])]; tensor var_1382_cast_fp16 = reshape(shape = var_1381, x = linear_64_cast_fp16)[name = tensor("op_1382_cast_fp16")]; tensor layers_6_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(247731136)))]; tensor layers_6_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(249828352)))]; tensor linear_65_cast_fp16 = linear(bias = layers_6_second_sub_layer_key_net_bias_to_fp16, weight = layers_6_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_65_cast_fp16")]; tensor var_1389 = const()[name = tensor("op_1389"), val = tensor([1, 438, 8, 128])]; tensor var_1390_cast_fp16 = reshape(shape = var_1389, x = linear_65_cast_fp16)[name = tensor("op_1390_cast_fp16")]; tensor layers_6_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(249830464)))]; tensor layers_6_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(251927680)))]; tensor linear_66_cast_fp16 = linear(bias = layers_6_second_sub_layer_value_net_bias_to_fp16, weight = layers_6_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_66_cast_fp16")]; tensor var_1397 = const()[name = tensor("op_1397"), val = tensor([1, 438, 8, 128])]; tensor var_1398_cast_fp16 = reshape(shape = var_1397, x = linear_66_cast_fp16)[name = tensor("op_1398_cast_fp16")]; tensor value_27_perm_0 = const()[name = tensor("value_27_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1364_to_fp16 = const()[name = tensor("op_1364_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_13_cast_fp16 = mul(x = var_1382_cast_fp16, y = var_1364_to_fp16)[name = tensor("mul_13_cast_fp16")]; tensor matmul_13_transpose_y_0 = const()[name = tensor("matmul_13_transpose_y_0"), val = tensor(true)]; tensor matmul_13_transpose_x_0 = const()[name = tensor("matmul_13_transpose_x_0"), val = tensor(false)]; tensor transpose_44_perm_0 = const()[name = tensor("transpose_44_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_45_perm_0 = const()[name = tensor("transpose_45_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_45 = transpose(perm = transpose_45_perm_0, x = var_1390_cast_fp16)[name = tensor("transpose_57")]; tensor transpose_44 = transpose(perm = transpose_44_perm_0, x = mul_13_cast_fp16)[name = tensor("transpose_58")]; tensor matmul_13_cast_fp16 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = transpose_44, y = transpose_45)[name = tensor("matmul_13_cast_fp16")]; tensor add_13_cast_fp16 = add(x = matmul_13_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_13_cast_fp16")]; tensor softmax_13_axis_0 = const()[name = tensor("softmax_13_axis_0"), val = tensor(-1)]; tensor softmax_13_cast_fp16 = softmax(axis = softmax_13_axis_0, x = add_13_cast_fp16)[name = tensor("softmax_13_cast_fp16")]; tensor attn_output_41_transpose_x_0 = const()[name = tensor("attn_output_41_transpose_x_0"), val = tensor(false)]; tensor attn_output_41_transpose_y_0 = const()[name = tensor("attn_output_41_transpose_y_0"), val = tensor(false)]; tensor value_27_cast_fp16 = transpose(perm = value_27_perm_0, x = var_1398_cast_fp16)[name = tensor("transpose_59")]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_0, transpose_y = attn_output_41_transpose_y_0, x = softmax_13_cast_fp16, y = value_27_cast_fp16)[name = tensor("attn_output_41_cast_fp16")]; tensor var_1401_perm_0 = const()[name = tensor("op_1401_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1405 = const()[name = tensor("op_1405"), val = tensor([1, 1, 1024])]; tensor var_1401_cast_fp16 = transpose(perm = var_1401_perm_0, x = attn_output_41_cast_fp16)[name = tensor("transpose_56")]; tensor input_133_cast_fp16 = reshape(shape = var_1405, x = var_1401_cast_fp16)[name = tensor("input_133_cast_fp16")]; tensor layers_6_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(251929792)))]; tensor layers_6_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_6_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(254027008)))]; tensor linear_67_cast_fp16 = linear(bias = layers_6_second_sub_layer_out_projection_bias_to_fp16, weight = layers_6_second_sub_layer_out_projection_weight_to_fp16, x = input_133_cast_fp16)[name = tensor("linear_67_cast_fp16")]; tensor input_135_cast_fp16 = add(x = input_129_cast_fp16, y = linear_67_cast_fp16)[name = tensor("input_135_cast_fp16")]; tensor input_137_axes_0 = const()[name = tensor("input_137_axes_0"), val = tensor([-1])]; tensor layers_6_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_6_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(254029120)))]; tensor layers_6_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_6_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(254031232)))]; tensor var_1413_to_fp16 = const()[name = tensor("op_1413_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_137_cast_fp16 = layer_norm(axes = input_137_axes_0, beta = layers_6_layer_norm_3_bias_to_fp16, epsilon = var_1413_to_fp16, gamma = layers_6_layer_norm_3_weight_to_fp16, x = input_135_cast_fp16)[name = tensor("input_137_cast_fp16")]; tensor layers_6_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_6_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(254033344)))]; tensor layers_6_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_6_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262422016)))]; tensor linear_68_cast_fp16 = linear(bias = layers_6_third_sub_layer_dense_in_bias_to_fp16, weight = layers_6_third_sub_layer_dense_in_weight_to_fp16, x = input_137_cast_fp16)[name = tensor("linear_68_cast_fp16")]; tensor input_141_cast_fp16 = relu(x = linear_68_cast_fp16)[name = tensor("input_141_cast_fp16")]; tensor layers_6_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_6_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262430272)))]; tensor layers_6_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_6_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270818944)))]; tensor linear_69_cast_fp16 = linear(bias = layers_6_third_sub_layer_dense_out_bias_to_fp16, weight = layers_6_third_sub_layer_dense_out_weight_to_fp16, x = input_141_cast_fp16)[name = tensor("linear_69_cast_fp16")]; tensor input_143_cast_fp16 = add(x = input_135_cast_fp16, y = linear_69_cast_fp16)[name = tensor("input_143_cast_fp16")]; tensor input_145_axes_0 = const()[name = tensor("input_145_axes_0"), val = tensor([-1])]; tensor layers_7_layer_norm_1_weight_to_fp16 = const()[name = tensor("layers_7_layer_norm_1_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270821056)))]; tensor layers_7_layer_norm_1_bias_to_fp16 = const()[name = tensor("layers_7_layer_norm_1_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270823168)))]; tensor var_1431_to_fp16 = const()[name = tensor("op_1431_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_145_cast_fp16 = layer_norm(axes = input_145_axes_0, beta = layers_7_layer_norm_1_bias_to_fp16, epsilon = var_1431_to_fp16, gamma = layers_7_layer_norm_1_weight_to_fp16, x = input_143_cast_fp16)[name = tensor("input_145_cast_fp16")]; tensor layers_7_first_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270825280)))]; tensor layers_7_first_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(272922496)))]; tensor linear_70_cast_fp16 = linear(bias = layers_7_first_sub_layer_query_net_bias_to_fp16, weight = layers_7_first_sub_layer_query_net_weight_to_fp16, x = input_145_cast_fp16)[name = tensor("linear_70_cast_fp16")]; tensor layers_7_first_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(272924608)))]; tensor layers_7_first_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(275021824)))]; tensor linear_71_cast_fp16 = linear(bias = layers_7_first_sub_layer_key_net_bias_to_fp16, weight = layers_7_first_sub_layer_key_net_weight_to_fp16, x = input_145_cast_fp16)[name = tensor("linear_71_cast_fp16")]; tensor layers_7_first_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(275023936)))]; tensor layers_7_first_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(277121152)))]; tensor linear_72_cast_fp16 = linear(bias = layers_7_first_sub_layer_value_net_bias_to_fp16, weight = layers_7_first_sub_layer_value_net_weight_to_fp16, x = input_145_cast_fp16)[name = tensor("linear_72_cast_fp16")]; tensor var_1456 = const()[name = tensor("op_1456"), val = tensor([1, 1, 8, 128])]; tensor var_1457_cast_fp16 = reshape(shape = var_1456, x = linear_70_cast_fp16)[name = tensor("op_1457_cast_fp16")]; tensor query_29_perm_0 = const()[name = tensor("query_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1471 = const()[name = tensor("op_1471"), val = tensor([1, 1, 8, 128])]; tensor var_1472_cast_fp16 = reshape(shape = var_1471, x = linear_71_cast_fp16)[name = tensor("op_1472_cast_fp16")]; tensor key_29_perm_0 = const()[name = tensor("key_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1486 = const()[name = tensor("op_1486"), val = tensor([1, 1, 8, 128])]; tensor var_1487_cast_fp16 = reshape(shape = var_1486, x = linear_72_cast_fp16)[name = tensor("op_1487_cast_fp16")]; tensor value_29_perm_0 = const()[name = tensor("value_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor k_cache_new_axis_0 = const()[name = tensor("k_cache_new_axis_0"), val = tensor(2)]; tensor k_cache_new_mode_0 = const()[name = tensor("k_cache_new_mode_0"), val = tensor("update")]; tensor k_cache_new_validate_indices_0 = const()[name = tensor("k_cache_new_validate_indices_0"), val = tensor(false)]; tensor k_cache_7_to_fp16_dtype_0 = const()[name = tensor("k_cache_7_to_fp16_dtype_0"), val = tensor("fp16")]; tensor k_cache_7_to_fp16 = cast(dtype = k_cache_7_to_fp16_dtype_0, x = k_cache_7)[name = tensor("cast_73")]; tensor key_29_cast_fp16 = transpose(perm = key_29_perm_0, x = var_1472_cast_fp16)[name = tensor("transpose_54")]; tensor k_cache_new_cast_fp16 = scatter_along_axis(axis = k_cache_new_axis_0, data = k_cache_7_to_fp16, indices = pos_idx, mode = k_cache_new_mode_0, updates = key_29_cast_fp16, validate_indices = k_cache_new_validate_indices_0)[name = tensor("k_cache_new_cast_fp16")]; tensor k_cache_new_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("k_cache_new_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor v_cache_new_axis_0 = const()[name = tensor("v_cache_new_axis_0"), val = tensor(2)]; tensor v_cache_new_mode_0 = const()[name = tensor("v_cache_new_mode_0"), val = tensor("update")]; tensor v_cache_new_validate_indices_0 = const()[name = tensor("v_cache_new_validate_indices_0"), val = tensor(false)]; tensor v_cache_7_to_fp16_dtype_0 = const()[name = tensor("v_cache_7_to_fp16_dtype_0"), val = tensor("fp16")]; tensor v_cache_7_to_fp16 = cast(dtype = v_cache_7_to_fp16_dtype_0, x = v_cache_7)[name = tensor("cast_71")]; tensor value_29_cast_fp16 = transpose(perm = value_29_perm_0, x = var_1487_cast_fp16)[name = tensor("transpose_53")]; tensor v_cache_new_cast_fp16 = scatter_along_axis(axis = v_cache_new_axis_0, data = v_cache_7_to_fp16, indices = pos_idx, mode = v_cache_new_mode_0, updates = value_29_cast_fp16, validate_indices = v_cache_new_validate_indices_0)[name = tensor("v_cache_new_cast_fp16")]; tensor v_cache_new_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("v_cache_new_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1497_to_fp16 = const()[name = tensor("op_1497_to_fp16"), val = tensor(0x1.6ap-4)]; tensor query_29_cast_fp16 = transpose(perm = query_29_perm_0, x = var_1457_cast_fp16)[name = tensor("transpose_55")]; tensor mul_14_cast_fp16 = mul(x = query_29_cast_fp16, y = var_1497_to_fp16)[name = tensor("mul_14_cast_fp16")]; tensor matmul_14_transpose_y_0 = const()[name = tensor("matmul_14_transpose_y_0"), val = tensor(true)]; tensor matmul_14_transpose_x_0 = const()[name = tensor("matmul_14_transpose_x_0"), val = tensor(false)]; tensor matmul_14_cast_fp16 = matmul(transpose_x = matmul_14_transpose_x_0, transpose_y = matmul_14_transpose_y_0, x = mul_14_cast_fp16, y = k_cache_new_cast_fp16)[name = tensor("matmul_14_cast_fp16")]; tensor add_14_cast_fp16 = add(x = matmul_14_cast_fp16, y = attention_mask_to_fp16)[name = tensor("add_14_cast_fp16")]; tensor softmax_14_axis_0 = const()[name = tensor("softmax_14_axis_0"), val = tensor(-1)]; tensor softmax_14_cast_fp16 = softmax(axis = softmax_14_axis_0, x = add_14_cast_fp16)[name = tensor("softmax_14_cast_fp16")]; tensor attn_output_43_transpose_x_0 = const()[name = tensor("attn_output_43_transpose_x_0"), val = tensor(false)]; tensor attn_output_43_transpose_y_0 = const()[name = tensor("attn_output_43_transpose_y_0"), val = tensor(false)]; tensor attn_output_43_cast_fp16 = matmul(transpose_x = attn_output_43_transpose_x_0, transpose_y = attn_output_43_transpose_y_0, x = softmax_14_cast_fp16, y = v_cache_new_cast_fp16)[name = tensor("attn_output_43_cast_fp16")]; tensor var_1502_perm_0 = const()[name = tensor("op_1502_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1508 = const()[name = tensor("op_1508"), val = tensor([1, 1, 1024])]; tensor var_1502_cast_fp16 = transpose(perm = var_1502_perm_0, x = attn_output_43_cast_fp16)[name = tensor("transpose_52")]; tensor input_147_cast_fp16 = reshape(shape = var_1508, x = var_1502_cast_fp16)[name = tensor("input_147_cast_fp16")]; tensor layers_7_first_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(277123264)))]; tensor layers_7_first_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_7_first_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279220480)))]; tensor linear_73_cast_fp16 = linear(bias = layers_7_first_sub_layer_out_projection_bias_to_fp16, weight = layers_7_first_sub_layer_out_projection_weight_to_fp16, x = input_147_cast_fp16)[name = tensor("linear_73_cast_fp16")]; tensor input_149_cast_fp16 = add(x = input_143_cast_fp16, y = linear_73_cast_fp16)[name = tensor("input_149_cast_fp16")]; tensor input_151_axes_0 = const()[name = tensor("input_151_axes_0"), val = tensor([-1])]; tensor layers_7_layer_norm_2_weight_to_fp16 = const()[name = tensor("layers_7_layer_norm_2_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279222592)))]; tensor layers_7_layer_norm_2_bias_to_fp16 = const()[name = tensor("layers_7_layer_norm_2_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279224704)))]; tensor var_1516_to_fp16 = const()[name = tensor("op_1516_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_151_cast_fp16 = layer_norm(axes = input_151_axes_0, beta = layers_7_layer_norm_2_bias_to_fp16, epsilon = var_1516_to_fp16, gamma = layers_7_layer_norm_2_weight_to_fp16, x = input_149_cast_fp16)[name = tensor("input_151_cast_fp16")]; tensor layers_7_second_sub_layer_query_net_weight_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_query_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279226816)))]; tensor layers_7_second_sub_layer_query_net_bias_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_query_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281324032)))]; tensor linear_74_cast_fp16 = linear(bias = layers_7_second_sub_layer_query_net_bias_to_fp16, weight = layers_7_second_sub_layer_query_net_weight_to_fp16, x = input_151_cast_fp16)[name = tensor("linear_74_cast_fp16")]; tensor var_1540 = const()[name = tensor("op_1540"), val = tensor([1, 1, 8, 128])]; tensor var_1541_cast_fp16 = reshape(shape = var_1540, x = linear_74_cast_fp16)[name = tensor("op_1541_cast_fp16")]; tensor layers_7_second_sub_layer_key_net_weight_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_key_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281326144)))]; tensor layers_7_second_sub_layer_key_net_bias_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_key_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283423360)))]; tensor linear_75_cast_fp16 = linear(bias = layers_7_second_sub_layer_key_net_bias_to_fp16, weight = layers_7_second_sub_layer_key_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_75_cast_fp16")]; tensor var_1548 = const()[name = tensor("op_1548"), val = tensor([1, 438, 8, 128])]; tensor var_1549_cast_fp16 = reshape(shape = var_1548, x = linear_75_cast_fp16)[name = tensor("op_1549_cast_fp16")]; tensor layers_7_second_sub_layer_value_net_weight_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_value_net_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283425472)))]; tensor layers_7_second_sub_layer_value_net_bias_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_value_net_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(285522688)))]; tensor linear_76_cast_fp16 = linear(bias = layers_7_second_sub_layer_value_net_bias_to_fp16, weight = layers_7_second_sub_layer_value_net_weight_to_fp16, x = encoder_hidden_states_to_fp16)[name = tensor("linear_76_cast_fp16")]; tensor var_1556 = const()[name = tensor("op_1556"), val = tensor([1, 438, 8, 128])]; tensor var_1557_cast_fp16 = reshape(shape = var_1556, x = linear_76_cast_fp16)[name = tensor("op_1557_cast_fp16")]; tensor value_perm_0 = const()[name = tensor("value_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1523_to_fp16 = const()[name = tensor("op_1523_to_fp16"), val = tensor(0x1.6ap-4)]; tensor mul_15_cast_fp16 = mul(x = var_1541_cast_fp16, y = var_1523_to_fp16)[name = tensor("mul_15_cast_fp16")]; tensor matmul_15_transpose_y_0 = const()[name = tensor("matmul_15_transpose_y_0"), val = tensor(true)]; tensor matmul_15_transpose_x_0 = const()[name = tensor("matmul_15_transpose_x_0"), val = tensor(false)]; tensor transpose_46_perm_0 = const()[name = tensor("transpose_46_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_47_perm_0 = const()[name = tensor("transpose_47_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_47 = transpose(perm = transpose_47_perm_0, x = var_1549_cast_fp16)[name = tensor("transpose_49")]; tensor transpose_46 = transpose(perm = transpose_46_perm_0, x = mul_15_cast_fp16)[name = tensor("transpose_50")]; tensor matmul_15_cast_fp16 = matmul(transpose_x = matmul_15_transpose_x_0, transpose_y = matmul_15_transpose_y_0, x = transpose_46, y = transpose_47)[name = tensor("matmul_15_cast_fp16")]; tensor add_15_cast_fp16 = add(x = matmul_15_cast_fp16, y = cross_attention_mask_to_fp16)[name = tensor("add_15_cast_fp16")]; tensor softmax_15_axis_0 = const()[name = tensor("softmax_15_axis_0"), val = tensor(-1)]; tensor softmax_15_cast_fp16 = softmax(axis = softmax_15_axis_0, x = add_15_cast_fp16)[name = tensor("softmax_15_cast_fp16")]; tensor attn_output_transpose_x_0 = const()[name = tensor("attn_output_transpose_x_0"), val = tensor(false)]; tensor attn_output_transpose_y_0 = const()[name = tensor("attn_output_transpose_y_0"), val = tensor(false)]; tensor value_cast_fp16 = transpose(perm = value_perm_0, x = var_1557_cast_fp16)[name = tensor("transpose_51")]; tensor attn_output_cast_fp16 = matmul(transpose_x = attn_output_transpose_x_0, transpose_y = attn_output_transpose_y_0, x = softmax_15_cast_fp16, y = value_cast_fp16)[name = tensor("attn_output_cast_fp16")]; tensor var_1560_perm_0 = const()[name = tensor("op_1560_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1564 = const()[name = tensor("op_1564"), val = tensor([1, 1, 1024])]; tensor var_1560_cast_fp16 = transpose(perm = var_1560_perm_0, x = attn_output_cast_fp16)[name = tensor("transpose_48")]; tensor input_153_cast_fp16 = reshape(shape = var_1564, x = var_1560_cast_fp16)[name = tensor("input_153_cast_fp16")]; tensor layers_7_second_sub_layer_out_projection_weight_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_out_projection_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(285524800)))]; tensor layers_7_second_sub_layer_out_projection_bias_to_fp16 = const()[name = tensor("layers_7_second_sub_layer_out_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287622016)))]; tensor linear_77_cast_fp16 = linear(bias = layers_7_second_sub_layer_out_projection_bias_to_fp16, weight = layers_7_second_sub_layer_out_projection_weight_to_fp16, x = input_153_cast_fp16)[name = tensor("linear_77_cast_fp16")]; tensor input_155_cast_fp16 = add(x = input_149_cast_fp16, y = linear_77_cast_fp16)[name = tensor("input_155_cast_fp16")]; tensor input_157_axes_0 = const()[name = tensor("input_157_axes_0"), val = tensor([-1])]; tensor layers_7_layer_norm_3_weight_to_fp16 = const()[name = tensor("layers_7_layer_norm_3_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287624128)))]; tensor layers_7_layer_norm_3_bias_to_fp16 = const()[name = tensor("layers_7_layer_norm_3_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287626240)))]; tensor var_1572_to_fp16 = const()[name = tensor("op_1572_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_157_cast_fp16 = layer_norm(axes = input_157_axes_0, beta = layers_7_layer_norm_3_bias_to_fp16, epsilon = var_1572_to_fp16, gamma = layers_7_layer_norm_3_weight_to_fp16, x = input_155_cast_fp16)[name = tensor("input_157_cast_fp16")]; tensor layers_7_third_sub_layer_dense_in_weight_to_fp16 = const()[name = tensor("layers_7_third_sub_layer_dense_in_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287628352)))]; tensor layers_7_third_sub_layer_dense_in_bias_to_fp16 = const()[name = tensor("layers_7_third_sub_layer_dense_in_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(296017024)))]; tensor linear_78_cast_fp16 = linear(bias = layers_7_third_sub_layer_dense_in_bias_to_fp16, weight = layers_7_third_sub_layer_dense_in_weight_to_fp16, x = input_157_cast_fp16)[name = tensor("linear_78_cast_fp16")]; tensor input_161_cast_fp16 = relu(x = linear_78_cast_fp16)[name = tensor("input_161_cast_fp16")]; tensor layers_7_third_sub_layer_dense_out_weight_to_fp16 = const()[name = tensor("layers_7_third_sub_layer_dense_out_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(296025280)))]; tensor layers_7_third_sub_layer_dense_out_bias_to_fp16 = const()[name = tensor("layers_7_third_sub_layer_dense_out_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304413952)))]; tensor linear_79_cast_fp16 = linear(bias = layers_7_third_sub_layer_dense_out_bias_to_fp16, weight = layers_7_third_sub_layer_dense_out_weight_to_fp16, x = input_161_cast_fp16)[name = tensor("linear_79_cast_fp16")]; tensor input_163_cast_fp16 = add(x = input_155_cast_fp16, y = linear_79_cast_fp16)[name = tensor("input_163_cast_fp16")]; tensor input_axes_0 = const()[name = tensor("input_axes_0"), val = tensor([-1])]; tensor final_norm_weight_to_fp16 = const()[name = tensor("final_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304416064)))]; tensor final_norm_bias_to_fp16 = const()[name = tensor("final_norm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304418176)))]; tensor var_1590_to_fp16 = const()[name = tensor("op_1590_to_fp16"), val = tensor(0x1.5p-17)]; tensor input_cast_fp16 = layer_norm(axes = input_axes_0, beta = final_norm_bias_to_fp16, epsilon = var_1590_to_fp16, gamma = final_norm_weight_to_fp16, x = input_163_cast_fp16)[name = tensor("input_cast_fp16")]; tensor lm_head_bias_to_fp16 = const()[name = tensor("lm_head_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304420288)))]; tensor linear_80_cast_fp16 = linear(bias = lm_head_bias_to_fp16, weight = embedding_token_embedding_weight_to_fp16, x = input_cast_fp16)[name = tensor("linear_80_cast_fp16")]; tensor var_1600_axes_0 = const()[name = tensor("op_1600_axes_0"), val = tensor([1])]; tensor var_1600_cast_fp16 = squeeze(axes = var_1600_axes_0, x = linear_80_cast_fp16)[name = tensor("op_1600_cast_fp16")]; tensor var_1600_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1600_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor logits = cast(dtype = var_1600_cast_fp16_to_fp32_dtype_0, x = var_1600_cast_fp16)[name = tensor("cast_69")]; tensor v_cache_7_out = cast(dtype = v_cache_new_cast_fp16_to_fp32_dtype_0, x = v_cache_new_cast_fp16)[name = tensor("cast_70")]; tensor k_cache_7_out = cast(dtype = k_cache_new_cast_fp16_to_fp32_dtype_0, x = k_cache_new_cast_fp16)[name = tensor("cast_72")]; tensor v_cache_6_out = cast(dtype = v_cache_new_13_cast_fp16_to_fp32_dtype_0, x = v_cache_new_13_cast_fp16)[name = tensor("cast_74")]; tensor k_cache_6_out = cast(dtype = k_cache_new_13_cast_fp16_to_fp32_dtype_0, x = k_cache_new_13_cast_fp16)[name = tensor("cast_76")]; tensor v_cache_5_out = cast(dtype = v_cache_new_11_cast_fp16_to_fp32_dtype_0, x = v_cache_new_11_cast_fp16)[name = tensor("cast_78")]; tensor k_cache_5_out = cast(dtype = k_cache_new_11_cast_fp16_to_fp32_dtype_0, x = k_cache_new_11_cast_fp16)[name = tensor("cast_80")]; tensor v_cache_4_out = cast(dtype = v_cache_new_9_cast_fp16_to_fp32_dtype_0, x = v_cache_new_9_cast_fp16)[name = tensor("cast_82")]; tensor k_cache_4_out = cast(dtype = k_cache_new_9_cast_fp16_to_fp32_dtype_0, x = k_cache_new_9_cast_fp16)[name = tensor("cast_84")]; tensor v_cache_3_out = cast(dtype = v_cache_new_7_cast_fp16_to_fp32_dtype_0, x = v_cache_new_7_cast_fp16)[name = tensor("cast_86")]; tensor k_cache_3_out = cast(dtype = k_cache_new_7_cast_fp16_to_fp32_dtype_0, x = k_cache_new_7_cast_fp16)[name = tensor("cast_88")]; tensor v_cache_2_out = cast(dtype = v_cache_new_5_cast_fp16_to_fp32_dtype_0, x = v_cache_new_5_cast_fp16)[name = tensor("cast_90")]; tensor k_cache_2_out = cast(dtype = k_cache_new_5_cast_fp16_to_fp32_dtype_0, x = k_cache_new_5_cast_fp16)[name = tensor("cast_92")]; tensor v_cache_1_out = cast(dtype = v_cache_new_3_cast_fp16_to_fp32_dtype_0, x = v_cache_new_3_cast_fp16)[name = tensor("cast_94")]; tensor k_cache_1_out = cast(dtype = k_cache_new_3_cast_fp16_to_fp32_dtype_0, x = k_cache_new_3_cast_fp16)[name = tensor("cast_96")]; tensor v_cache_0_out = cast(dtype = v_cache_new_1_cast_fp16_to_fp32_dtype_0, x = v_cache_new_1_cast_fp16)[name = tensor("cast_101")]; tensor k_cache_0_out = cast(dtype = k_cache_new_1_cast_fp16_to_fp32_dtype_0, x = k_cache_new_1_cast_fp16)[name = tensor("cast_103")]; } -> (logits, k_cache_0_out, v_cache_0_out, k_cache_1_out, v_cache_1_out, k_cache_2_out, v_cache_2_out, k_cache_3_out, v_cache_3_out, k_cache_4_out, v_cache_4_out, k_cache_5_out, v_cache_5_out, k_cache_6_out, v_cache_6_out, k_cache_7_out, v_cache_7_out); }