program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}})] { func main(tensor attention_mask, tensor input_ids) { int32 embedding_batch_dims_0 = const()[name = string("embedding_batch_dims_0"), val = int32(0)]; bool embedding_validate_indices_0 = const()[name = string("embedding_validate_indices_0"), val = bool(false)]; tensor p_st_0_model_embed_tokens_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201326720))))[name = string("p_st_0_model_embed_tokens_weight_to_fp16_quantized")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = input_ids, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(262144)]; tensor add_0 = add(x = input_ids, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = input_ids, b = add_0, cond = greater_equal_0)[name = string("select_0")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor greater_equal_0_1 = greater_equal(x = select_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(262144)]; tensor add_0_1 = add(x = select_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = select_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 embedding_cast_fp16_axis_0 = const()[name = string("embedding_cast_fp16_axis_0"), val = int32(0)]; tensor embedding_cast_fp16 = gather(axis = embedding_cast_fp16_axis_0, batch_dims = embedding_batch_dims_0, indices = select_0_1, validate_indices = embedding_validate_indices_0, x = p_st_0_model_embed_tokens_weight_to_fp16_quantized)[name = string("embedding_cast_fp16")]; fp16 const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = fp16(0x1.bb8p+4)]; tensor mul_cast_fp16 = mul(x = embedding_cast_fp16, y = const_6_to_fp16)[name = string("mul_cast_fp16")]; tensor unsqueeze_1_axes_0 = const()[name = string("unsqueeze_1_axes_0"), val = tensor([1])]; tensor unsqueeze_1 = expand_dims(axes = unsqueeze_1_axes_0, x = attention_mask)[name = string("unsqueeze_1")]; tensor unsqueeze_2_axes_0 = const()[name = string("unsqueeze_2_axes_0"), val = tensor([2])]; tensor unsqueeze_2 = expand_dims(axes = unsqueeze_2_axes_0, x = unsqueeze_1)[name = string("unsqueeze_2")]; fp16 const_21_to_fp16 = const()[name = string("const_21_to_fp16"), val = fp16(0x1p+0)]; string _to_copy_3_to_fp16_dtype_0 = const()[name = string("_to_copy_3_to_fp16_dtype_0"), val = string("fp16")]; tensor unsqueeze_2_to_fp16 = cast(dtype = _to_copy_3_to_fp16_dtype_0, x = unsqueeze_2)[name = string("cast_1")]; tensor rsub_cast_fp16 = sub(x = const_21_to_fp16, y = unsqueeze_2_to_fp16)[name = string("rsub_cast_fp16")]; fp16 const_22_to_fp16 = const()[name = string("const_22_to_fp16"), val = fp16(-inf)]; tensor mul_1_cast_fp16 = mul(x = rsub_cast_fp16, y = const_22_to_fp16)[name = string("mul_1_cast_fp16")]; tensor expand_reps_0 = const()[name = string("expand_reps_0"), val = tensor([1, 1, 512, 1])]; tensor expand_cast_fp16 = tile(reps = expand_reps_0, x = mul_1_cast_fp16)[name = string("expand_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_1_cast_fp16 = pow(x = mul_cast_fp16, y = const_280_promoted_to_fp16)[name = string("pow_1_cast_fp16")]; tensor mean_axes_0 = const()[name = string("mean_axes_0"), val = tensor([-1])]; bool mean_keep_dims_0 = const()[name = string("mean_keep_dims_0"), val = bool(true)]; tensor mean_cast_fp16 = reduce_mean(axes = mean_axes_0, keep_dims = mean_keep_dims_0, x = pow_1_cast_fp16)[name = string("mean_cast_fp16")]; fp16 const_283_to_fp16 = const()[name = string("const_283_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_cast_fp16 = add(x = mean_cast_fp16, y = const_283_to_fp16)[name = string("add_cast_fp16")]; fp32 rsqrt_epsilon_0 = const()[name = string("rsqrt_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_cast_fp16 = rsqrt(epsilon = rsqrt_epsilon_0, x = add_cast_fp16)[name = string("rsqrt_cast_fp16")]; tensor mul_51_cast_fp16 = mul(x = mul_cast_fp16, y = rsqrt_cast_fp16)[name = string("mul_51_cast_fp16")]; tensor add_1_to_fp16 = const()[name = string("add_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201851072)))]; tensor mul_52_cast_fp16 = mul(x = mul_51_cast_fp16, y = add_1_to_fp16)[name = string("mul_52_cast_fp16")]; tensor p_st_0_model_layers_0_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201852672))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202442560))))[name = string("p_st_0_model_layers_0_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_0_bias_0_to_fp16 = const()[name = string("linear_0_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202444160)))]; tensor linear_0_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_0_self_attn_q_proj_weight_to_fp16_quantized, x = mul_52_cast_fp16)[name = string("linear_0_cast_fp16")]; tensor const_287 = const()[name = string("const_287"), val = tensor([1, 512, -1, 256])]; tensor view_cast_fp16 = reshape(shape = const_287, x = linear_0_cast_fp16)[name = string("view_cast_fp16")]; tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_0_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202445760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202642432))))[name = string("p_st_0_model_layers_0_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_1_bias_0_to_fp16 = const()[name = string("linear_1_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202643008)))]; tensor linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_0_self_attn_k_proj_weight_to_fp16_quantized, x = mul_52_cast_fp16)[name = string("linear_1_cast_fp16")]; tensor const_290 = const()[name = string("const_290"), val = tensor([1, 512, -1, 256])]; tensor view_1_cast_fp16 = reshape(shape = const_290, x = linear_1_cast_fp16)[name = string("view_1_cast_fp16")]; tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_0_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202643584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202840256))))[name = string("p_st_0_model_layers_0_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_2_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_0_self_attn_v_proj_weight_to_fp16_quantized, x = mul_52_cast_fp16)[name = string("linear_2_cast_fp16")]; tensor const_293 = const()[name = string("const_293"), val = tensor([1, 512, -1, 256])]; tensor view_2_cast_fp16 = reshape(shape = const_293, x = linear_2_cast_fp16)[name = string("view_2_cast_fp16")]; tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_297_promoted_to_fp16 = const()[name = string("const_297_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = view_cast_fp16)[name = string("transpose_95")]; tensor pow_2_cast_fp16 = pow(x = transpose_24_cast_fp16, y = const_297_promoted_to_fp16)[name = string("pow_2_cast_fp16")]; tensor mean_1_axes_0 = const()[name = string("mean_1_axes_0"), val = tensor([-1])]; bool mean_1_keep_dims_0 = const()[name = string("mean_1_keep_dims_0"), val = bool(true)]; tensor mean_1_cast_fp16 = reduce_mean(axes = mean_1_axes_0, keep_dims = mean_1_keep_dims_0, x = pow_2_cast_fp16)[name = string("mean_1_cast_fp16")]; fp16 const_300_to_fp16 = const()[name = string("const_300_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_2_cast_fp16 = add(x = mean_1_cast_fp16, y = const_300_to_fp16)[name = string("add_2_cast_fp16")]; fp32 rsqrt_1_epsilon_0 = const()[name = string("rsqrt_1_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_1_cast_fp16 = rsqrt(epsilon = rsqrt_1_epsilon_0, x = add_2_cast_fp16)[name = string("rsqrt_1_cast_fp16")]; tensor mul_53_cast_fp16 = mul(x = transpose_24_cast_fp16, y = rsqrt_1_cast_fp16)[name = string("mul_53_cast_fp16")]; tensor add_3_to_fp16 = const()[name = string("add_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202840832)))]; tensor mul_54_cast_fp16 = mul(x = mul_53_cast_fp16, y = add_3_to_fp16)[name = string("mul_54_cast_fp16")]; fp16 const_305_promoted_to_fp16 = const()[name = string("const_305_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = view_1_cast_fp16)[name = string("transpose_94")]; tensor pow_3_cast_fp16 = pow(x = transpose_25_cast_fp16, y = const_305_promoted_to_fp16)[name = string("pow_3_cast_fp16")]; tensor mean_2_axes_0 = const()[name = string("mean_2_axes_0"), val = tensor([-1])]; bool mean_2_keep_dims_0 = const()[name = string("mean_2_keep_dims_0"), val = bool(true)]; tensor mean_2_cast_fp16 = reduce_mean(axes = mean_2_axes_0, keep_dims = mean_2_keep_dims_0, x = pow_3_cast_fp16)[name = string("mean_2_cast_fp16")]; fp16 const_308_to_fp16 = const()[name = string("const_308_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_4_cast_fp16 = add(x = mean_2_cast_fp16, y = const_308_to_fp16)[name = string("add_4_cast_fp16")]; fp32 rsqrt_2_epsilon_0 = const()[name = string("rsqrt_2_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_2_cast_fp16 = rsqrt(epsilon = rsqrt_2_epsilon_0, x = add_4_cast_fp16)[name = string("rsqrt_2_cast_fp16")]; tensor mul_55_cast_fp16 = mul(x = transpose_25_cast_fp16, y = rsqrt_2_cast_fp16)[name = string("mul_55_cast_fp16")]; tensor add_5_to_fp16 = const()[name = string("add_5_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202841408)))]; tensor mul_56_cast_fp16 = mul(x = mul_55_cast_fp16, y = add_5_to_fp16)[name = string("mul_56_cast_fp16")]; tensor unsqueeze_77_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202841984))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("unsqueeze_77_to_fp16_quantized")]; tensor mul_57_cast_fp16 = mul(x = mul_54_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_57_cast_fp16")]; tensor slice_77_begin_0 = const()[name = string("slice_77_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_77_end_0 = const()[name = string("slice_77_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_77_end_mask_0 = const()[name = string("slice_77_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_77_cast_fp16 = slice_by_index(begin = slice_77_begin_0, end = slice_77_end_0, end_mask = slice_77_end_mask_0, x = mul_54_cast_fp16)[name = string("slice_77_cast_fp16")]; tensor slice_78_begin_0 = const()[name = string("slice_78_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_78_end_0 = const()[name = string("slice_78_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_78_end_mask_0 = const()[name = string("slice_78_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_78_cast_fp16 = slice_by_index(begin = slice_78_begin_0, end = slice_78_end_0, end_mask = slice_78_end_mask_0, x = mul_54_cast_fp16)[name = string("slice_78_cast_fp16")]; fp16 const_320_promoted_to_fp16 = const()[name = string("const_320_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_cast_fp16 = mul(x = slice_78_cast_fp16, y = const_320_promoted_to_fp16)[name = string("neg_cast_fp16")]; int32 const_321 = const()[name = string("const_321"), val = int32(-1)]; bool cat_24_interleave_0 = const()[name = string("cat_24_interleave_0"), val = bool(false)]; tensor cat_24_cast_fp16 = concat(axis = const_321, interleave = cat_24_interleave_0, values = (neg_cast_fp16, slice_77_cast_fp16))[name = string("cat_24_cast_fp16")]; tensor unsqueeze_78_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202973120))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("unsqueeze_78_to_fp16_quantized")]; tensor mul_58_cast_fp16 = mul(x = cat_24_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_58_cast_fp16")]; tensor add_6_cast_fp16 = add(x = mul_57_cast_fp16, y = mul_58_cast_fp16)[name = string("add_6_cast_fp16")]; tensor mul_59_cast_fp16 = mul(x = mul_56_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_59_cast_fp16")]; tensor slice_79_begin_0 = const()[name = string("slice_79_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_79_end_0 = const()[name = string("slice_79_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_79_end_mask_0 = const()[name = string("slice_79_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_79_cast_fp16 = slice_by_index(begin = slice_79_begin_0, end = slice_79_end_0, end_mask = slice_79_end_mask_0, x = mul_56_cast_fp16)[name = string("slice_79_cast_fp16")]; tensor slice_80_begin_0 = const()[name = string("slice_80_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_80_end_0 = const()[name = string("slice_80_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_80_end_mask_0 = const()[name = string("slice_80_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_80_cast_fp16 = slice_by_index(begin = slice_80_begin_0, end = slice_80_end_0, end_mask = slice_80_end_mask_0, x = mul_56_cast_fp16)[name = string("slice_80_cast_fp16")]; fp16 const_328_promoted_to_fp16 = const()[name = string("const_328_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_1_cast_fp16 = mul(x = slice_80_cast_fp16, y = const_328_promoted_to_fp16)[name = string("neg_1_cast_fp16")]; int32 const_329 = const()[name = string("const_329"), val = int32(-1)]; bool cat_25_interleave_0 = const()[name = string("cat_25_interleave_0"), val = bool(false)]; tensor cat_25_cast_fp16 = concat(axis = const_329, interleave = cat_25_interleave_0, values = (neg_1_cast_fp16, slice_79_cast_fp16))[name = string("cat_25_cast_fp16")]; tensor mul_60_cast_fp16 = mul(x = cat_25_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_60_cast_fp16")]; tensor add_7_cast_fp16 = add(x = mul_59_cast_fp16, y = mul_60_cast_fp16)[name = string("add_7_cast_fp16")]; int32 const_330 = const()[name = string("const_330"), val = int32(-2)]; bool cat_26_interleave_0 = const()[name = string("cat_26_interleave_0"), val = bool(false)]; tensor cat_26_cast_fp16 = concat(axis = const_330, interleave = cat_26_interleave_0, values = add_7_cast_fp16)[name = string("cat_26_cast_fp16")]; int32 const_331 = const()[name = string("const_331"), val = int32(-2)]; bool cat_27_interleave_0 = const()[name = string("cat_27_interleave_0"), val = bool(false)]; tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = view_2_cast_fp16)[name = string("transpose_93")]; tensor cat_27_cast_fp16 = concat(axis = const_331, interleave = cat_27_interleave_0, values = transpose_26_cast_fp16)[name = string("cat_27_cast_fp16")]; tensor unsqueeze_79_axes_0 = const()[name = string("unsqueeze_79_axes_0"), val = tensor([2])]; tensor unsqueeze_79_cast_fp16 = expand_dims(axes = unsqueeze_79_axes_0, x = cat_26_cast_fp16)[name = string("unsqueeze_79_cast_fp16")]; tensor expand_26_reps_0 = const()[name = string("expand_26_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_26_cast_fp16 = tile(reps = expand_26_reps_0, x = unsqueeze_79_cast_fp16)[name = string("expand_26_cast_fp16")]; tensor const_346 = const()[name = string("const_346"), val = tensor([1, 3, 512, 256])]; tensor view_3_cast_fp16 = reshape(shape = const_346, x = expand_26_cast_fp16)[name = string("view_3_cast_fp16")]; tensor unsqueeze_80_axes_0 = const()[name = string("unsqueeze_80_axes_0"), val = tensor([2])]; tensor unsqueeze_80_cast_fp16 = expand_dims(axes = unsqueeze_80_axes_0, x = cat_27_cast_fp16)[name = string("unsqueeze_80_cast_fp16")]; tensor expand_27_reps_0 = const()[name = string("expand_27_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_27_cast_fp16 = tile(reps = expand_27_reps_0, x = unsqueeze_80_cast_fp16)[name = string("expand_27_cast_fp16")]; tensor const_361 = const()[name = string("const_361"), val = tensor([1, 3, 512, 256])]; tensor view_4_cast_fp16 = reshape(shape = const_361, x = expand_27_cast_fp16)[name = string("view_4_cast_fp16")]; bool matmul_24_transpose_x_1 = const()[name = string("matmul_24_transpose_x_1"), val = bool(false)]; bool matmul_24_transpose_y_1 = const()[name = string("matmul_24_transpose_y_1"), val = bool(true)]; tensor matmul_24_cast_fp16 = matmul(transpose_x = matmul_24_transpose_x_1, transpose_y = matmul_24_transpose_y_1, x = add_6_cast_fp16, y = view_3_cast_fp16)[name = string("matmul_24_cast_fp16")]; fp16 const_364_to_fp16 = const()[name = string("const_364_to_fp16"), val = fp16(0x1p-4)]; tensor mul_61_cast_fp16 = mul(x = matmul_24_cast_fp16, y = const_364_to_fp16)[name = string("mul_61_cast_fp16")]; tensor add_8_cast_fp16 = add(x = mul_61_cast_fp16, y = expand_cast_fp16)[name = string("add_8_cast_fp16")]; int32 const_374 = const()[name = string("const_374"), val = int32(-1)]; tensor softmax_cast_fp16 = softmax(axis = const_374, x = add_8_cast_fp16)[name = string("softmax_cast_fp16")]; bool matmul_25_transpose_x_0 = const()[name = string("matmul_25_transpose_x_0"), val = bool(false)]; bool matmul_25_transpose_y_0 = const()[name = string("matmul_25_transpose_y_0"), val = bool(false)]; tensor matmul_25_cast_fp16 = matmul(transpose_x = matmul_25_transpose_x_0, transpose_y = matmul_25_transpose_y_0, x = softmax_cast_fp16, y = view_4_cast_fp16)[name = string("matmul_25_cast_fp16")]; tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_379 = const()[name = string("const_379"), val = tensor([1, 512, -1])]; tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = matmul_25_cast_fp16)[name = string("transpose_92")]; tensor view_5_cast_fp16 = reshape(shape = const_379, x = transpose_28_cast_fp16)[name = string("view_5_cast_fp16")]; tensor p_st_0_model_layers_0_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203104256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203694144))))[name = string("p_st_0_model_layers_0_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_3_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_0_self_attn_o_proj_weight_to_fp16_quantized, x = view_5_cast_fp16)[name = string("linear_3_cast_fp16")]; fp16 const_381_promoted_to_fp16 = const()[name = string("const_381_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_4_cast_fp16 = pow(x = linear_3_cast_fp16, y = const_381_promoted_to_fp16)[name = string("pow_4_cast_fp16")]; tensor mean_3_axes_0 = const()[name = string("mean_3_axes_0"), val = tensor([-1])]; bool mean_3_keep_dims_0 = const()[name = string("mean_3_keep_dims_0"), val = bool(true)]; tensor mean_3_cast_fp16 = reduce_mean(axes = mean_3_axes_0, keep_dims = mean_3_keep_dims_0, x = pow_4_cast_fp16)[name = string("mean_3_cast_fp16")]; fp16 const_384_to_fp16 = const()[name = string("const_384_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_9_cast_fp16 = add(x = mean_3_cast_fp16, y = const_384_to_fp16)[name = string("add_9_cast_fp16")]; fp32 rsqrt_3_epsilon_0 = const()[name = string("rsqrt_3_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_3_cast_fp16 = rsqrt(epsilon = rsqrt_3_epsilon_0, x = add_9_cast_fp16)[name = string("rsqrt_3_cast_fp16")]; tensor mul_62_cast_fp16 = mul(x = linear_3_cast_fp16, y = rsqrt_3_cast_fp16)[name = string("mul_62_cast_fp16")]; tensor add_10_to_fp16 = const()[name = string("add_10_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203695744)))]; tensor mul_63_cast_fp16 = mul(x = mul_62_cast_fp16, y = add_10_to_fp16)[name = string("mul_63_cast_fp16")]; tensor add_11_cast_fp16 = add(x = mul_cast_fp16, y = mul_63_cast_fp16)[name = string("add_11_cast_fp16")]; fp16 const_389_promoted_to_fp16 = const()[name = string("const_389_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_5_cast_fp16 = pow(x = add_11_cast_fp16, y = const_389_promoted_to_fp16)[name = string("pow_5_cast_fp16")]; tensor mean_4_axes_0 = const()[name = string("mean_4_axes_0"), val = tensor([-1])]; bool mean_4_keep_dims_0 = const()[name = string("mean_4_keep_dims_0"), val = bool(true)]; tensor mean_4_cast_fp16 = reduce_mean(axes = mean_4_axes_0, keep_dims = mean_4_keep_dims_0, x = pow_5_cast_fp16)[name = string("mean_4_cast_fp16")]; fp16 const_392_to_fp16 = const()[name = string("const_392_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_12_cast_fp16 = add(x = mean_4_cast_fp16, y = const_392_to_fp16)[name = string("add_12_cast_fp16")]; fp32 rsqrt_4_epsilon_0 = const()[name = string("rsqrt_4_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_4_cast_fp16 = rsqrt(epsilon = rsqrt_4_epsilon_0, x = add_12_cast_fp16)[name = string("rsqrt_4_cast_fp16")]; tensor mul_64_cast_fp16 = mul(x = add_11_cast_fp16, y = rsqrt_4_cast_fp16)[name = string("mul_64_cast_fp16")]; tensor add_13_to_fp16 = const()[name = string("add_13_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203697344)))]; tensor mul_65_cast_fp16 = mul(x = mul_64_cast_fp16, y = add_13_to_fp16)[name = string("mul_65_cast_fp16")]; tensor p_st_0_model_layers_0_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203698944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204583744))))[name = string("p_st_0_model_layers_0_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_4_bias_0_to_fp16 = const()[name = string("linear_4_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204586112)))]; tensor linear_4_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_0_mlp_gate_proj_weight_to_fp16_quantized, x = mul_65_cast_fp16)[name = string("linear_4_cast_fp16")]; string gelu_mode_0 = const()[name = string("gelu_mode_0"), val = string("EXACT")]; tensor gelu_cast_fp16 = gelu(mode = gelu_mode_0, x = linear_4_cast_fp16)[name = string("gelu_cast_fp16")]; tensor p_st_0_model_layers_0_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204588480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205473280))))[name = string("p_st_0_model_layers_0_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_5_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_0_mlp_up_proj_weight_to_fp16_quantized, x = mul_65_cast_fp16)[name = string("linear_5_cast_fp16")]; tensor mul_66_cast_fp16 = mul(x = gelu_cast_fp16, y = linear_5_cast_fp16)[name = string("mul_66_cast_fp16")]; tensor p_st_0_model_layers_0_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205475648))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206360448))))[name = string("p_st_0_model_layers_0_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_6_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_0_mlp_down_proj_weight_to_fp16_quantized, x = mul_66_cast_fp16)[name = string("linear_6_cast_fp16")]; fp16 const_397_promoted_to_fp16 = const()[name = string("const_397_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_6_cast_fp16 = pow(x = linear_6_cast_fp16, y = const_397_promoted_to_fp16)[name = string("pow_6_cast_fp16")]; tensor mean_5_axes_0 = const()[name = string("mean_5_axes_0"), val = tensor([-1])]; bool mean_5_keep_dims_0 = const()[name = string("mean_5_keep_dims_0"), val = bool(true)]; tensor mean_5_cast_fp16 = reduce_mean(axes = mean_5_axes_0, keep_dims = mean_5_keep_dims_0, x = pow_6_cast_fp16)[name = string("mean_5_cast_fp16")]; fp16 const_400_to_fp16 = const()[name = string("const_400_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_14_cast_fp16 = add(x = mean_5_cast_fp16, y = const_400_to_fp16)[name = string("add_14_cast_fp16")]; fp32 rsqrt_5_epsilon_0 = const()[name = string("rsqrt_5_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_5_cast_fp16 = rsqrt(epsilon = rsqrt_5_epsilon_0, x = add_14_cast_fp16)[name = string("rsqrt_5_cast_fp16")]; tensor mul_67_cast_fp16 = mul(x = linear_6_cast_fp16, y = rsqrt_5_cast_fp16)[name = string("mul_67_cast_fp16")]; tensor add_15_to_fp16 = const()[name = string("add_15_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206362048)))]; tensor mul_68_cast_fp16 = mul(x = mul_67_cast_fp16, y = add_15_to_fp16)[name = string("mul_68_cast_fp16")]; tensor add_16_cast_fp16 = add(x = add_11_cast_fp16, y = mul_68_cast_fp16)[name = string("add_16_cast_fp16")]; fp16 const_405_promoted_to_fp16 = const()[name = string("const_405_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_7_cast_fp16 = pow(x = add_16_cast_fp16, y = const_405_promoted_to_fp16)[name = string("pow_7_cast_fp16")]; tensor mean_6_axes_0 = const()[name = string("mean_6_axes_0"), val = tensor([-1])]; bool mean_6_keep_dims_0 = const()[name = string("mean_6_keep_dims_0"), val = bool(true)]; tensor mean_6_cast_fp16 = reduce_mean(axes = mean_6_axes_0, keep_dims = mean_6_keep_dims_0, x = pow_7_cast_fp16)[name = string("mean_6_cast_fp16")]; fp16 const_408_to_fp16 = const()[name = string("const_408_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_17_cast_fp16 = add(x = mean_6_cast_fp16, y = const_408_to_fp16)[name = string("add_17_cast_fp16")]; fp32 rsqrt_6_epsilon_0 = const()[name = string("rsqrt_6_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_6_cast_fp16 = rsqrt(epsilon = rsqrt_6_epsilon_0, x = add_17_cast_fp16)[name = string("rsqrt_6_cast_fp16")]; tensor mul_69_cast_fp16 = mul(x = add_16_cast_fp16, y = rsqrt_6_cast_fp16)[name = string("mul_69_cast_fp16")]; tensor add_18_to_fp16 = const()[name = string("add_18_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206363648)))]; tensor mul_70_cast_fp16 = mul(x = mul_69_cast_fp16, y = add_18_to_fp16)[name = string("mul_70_cast_fp16")]; tensor p_st_0_model_layers_1_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206365248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206955136))))[name = string("p_st_0_model_layers_1_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_7_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_1_self_attn_q_proj_weight_to_fp16_quantized, x = mul_70_cast_fp16)[name = string("linear_7_cast_fp16")]; tensor const_412 = const()[name = string("const_412"), val = tensor([1, 512, -1, 256])]; tensor view_6_cast_fp16 = reshape(shape = const_412, x = linear_7_cast_fp16)[name = string("view_6_cast_fp16")]; tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_1_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206956736))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207153408))))[name = string("p_st_0_model_layers_1_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_8_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_1_self_attn_k_proj_weight_to_fp16_quantized, x = mul_70_cast_fp16)[name = string("linear_8_cast_fp16")]; tensor const_415 = const()[name = string("const_415"), val = tensor([1, 512, -1, 256])]; tensor view_7_cast_fp16 = reshape(shape = const_415, x = linear_8_cast_fp16)[name = string("view_7_cast_fp16")]; tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_1_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207153984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207350656))))[name = string("p_st_0_model_layers_1_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_9_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_1_self_attn_v_proj_weight_to_fp16_quantized, x = mul_70_cast_fp16)[name = string("linear_9_cast_fp16")]; tensor const_418 = const()[name = string("const_418"), val = tensor([1, 512, -1, 256])]; tensor view_8_cast_fp16 = reshape(shape = const_418, x = linear_9_cast_fp16)[name = string("view_8_cast_fp16")]; tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_422_promoted_to_fp16 = const()[name = string("const_422_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = view_6_cast_fp16)[name = string("transpose_91")]; tensor pow_8_cast_fp16 = pow(x = transpose_29_cast_fp16, y = const_422_promoted_to_fp16)[name = string("pow_8_cast_fp16")]; tensor mean_7_axes_0 = const()[name = string("mean_7_axes_0"), val = tensor([-1])]; bool mean_7_keep_dims_0 = const()[name = string("mean_7_keep_dims_0"), val = bool(true)]; tensor mean_7_cast_fp16 = reduce_mean(axes = mean_7_axes_0, keep_dims = mean_7_keep_dims_0, x = pow_8_cast_fp16)[name = string("mean_7_cast_fp16")]; fp16 const_425_to_fp16 = const()[name = string("const_425_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_19_cast_fp16 = add(x = mean_7_cast_fp16, y = const_425_to_fp16)[name = string("add_19_cast_fp16")]; fp32 rsqrt_7_epsilon_0 = const()[name = string("rsqrt_7_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_7_cast_fp16 = rsqrt(epsilon = rsqrt_7_epsilon_0, x = add_19_cast_fp16)[name = string("rsqrt_7_cast_fp16")]; tensor mul_71_cast_fp16 = mul(x = transpose_29_cast_fp16, y = rsqrt_7_cast_fp16)[name = string("mul_71_cast_fp16")]; tensor add_20_to_fp16 = const()[name = string("add_20_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207351232)))]; tensor mul_72_cast_fp16 = mul(x = mul_71_cast_fp16, y = add_20_to_fp16)[name = string("mul_72_cast_fp16")]; fp16 const_430_promoted_to_fp16 = const()[name = string("const_430_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = view_7_cast_fp16)[name = string("transpose_90")]; tensor pow_9_cast_fp16 = pow(x = transpose_30_cast_fp16, y = const_430_promoted_to_fp16)[name = string("pow_9_cast_fp16")]; tensor mean_8_axes_0 = const()[name = string("mean_8_axes_0"), val = tensor([-1])]; bool mean_8_keep_dims_0 = const()[name = string("mean_8_keep_dims_0"), val = bool(true)]; tensor mean_8_cast_fp16 = reduce_mean(axes = mean_8_axes_0, keep_dims = mean_8_keep_dims_0, x = pow_9_cast_fp16)[name = string("mean_8_cast_fp16")]; fp16 const_433_to_fp16 = const()[name = string("const_433_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_21_cast_fp16 = add(x = mean_8_cast_fp16, y = const_433_to_fp16)[name = string("add_21_cast_fp16")]; fp32 rsqrt_8_epsilon_0 = const()[name = string("rsqrt_8_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_8_cast_fp16 = rsqrt(epsilon = rsqrt_8_epsilon_0, x = add_21_cast_fp16)[name = string("rsqrt_8_cast_fp16")]; tensor mul_73_cast_fp16 = mul(x = transpose_30_cast_fp16, y = rsqrt_8_cast_fp16)[name = string("mul_73_cast_fp16")]; tensor add_22_to_fp16 = const()[name = string("add_22_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207351808)))]; tensor mul_74_cast_fp16 = mul(x = mul_73_cast_fp16, y = add_22_to_fp16)[name = string("mul_74_cast_fp16")]; tensor mul_75_cast_fp16 = mul(x = mul_72_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_75_cast_fp16")]; tensor slice_100_begin_0 = const()[name = string("slice_100_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_100_end_0 = const()[name = string("slice_100_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_100_end_mask_0 = const()[name = string("slice_100_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_100_cast_fp16 = slice_by_index(begin = slice_100_begin_0, end = slice_100_end_0, end_mask = slice_100_end_mask_0, x = mul_72_cast_fp16)[name = string("slice_100_cast_fp16")]; tensor slice_101_begin_0 = const()[name = string("slice_101_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_101_end_0 = const()[name = string("slice_101_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_101_end_mask_0 = const()[name = string("slice_101_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_101_cast_fp16 = slice_by_index(begin = slice_101_begin_0, end = slice_101_end_0, end_mask = slice_101_end_mask_0, x = mul_72_cast_fp16)[name = string("slice_101_cast_fp16")]; fp16 const_445_promoted_to_fp16 = const()[name = string("const_445_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_2_cast_fp16 = mul(x = slice_101_cast_fp16, y = const_445_promoted_to_fp16)[name = string("neg_2_cast_fp16")]; int32 const_446 = const()[name = string("const_446"), val = int32(-1)]; bool cat_28_interleave_0 = const()[name = string("cat_28_interleave_0"), val = bool(false)]; tensor cat_28_cast_fp16 = concat(axis = const_446, interleave = cat_28_interleave_0, values = (neg_2_cast_fp16, slice_100_cast_fp16))[name = string("cat_28_cast_fp16")]; tensor mul_76_cast_fp16 = mul(x = cat_28_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_76_cast_fp16")]; tensor add_23_cast_fp16 = add(x = mul_75_cast_fp16, y = mul_76_cast_fp16)[name = string("add_23_cast_fp16")]; tensor mul_77_cast_fp16 = mul(x = mul_74_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_77_cast_fp16")]; tensor slice_102_begin_0 = const()[name = string("slice_102_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_102_end_0 = const()[name = string("slice_102_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_102_end_mask_0 = const()[name = string("slice_102_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_102_cast_fp16 = slice_by_index(begin = slice_102_begin_0, end = slice_102_end_0, end_mask = slice_102_end_mask_0, x = mul_74_cast_fp16)[name = string("slice_102_cast_fp16")]; tensor slice_103_begin_0 = const()[name = string("slice_103_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_103_end_0 = const()[name = string("slice_103_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_103_end_mask_0 = const()[name = string("slice_103_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_103_cast_fp16 = slice_by_index(begin = slice_103_begin_0, end = slice_103_end_0, end_mask = slice_103_end_mask_0, x = mul_74_cast_fp16)[name = string("slice_103_cast_fp16")]; fp16 const_453_promoted_to_fp16 = const()[name = string("const_453_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_3_cast_fp16 = mul(x = slice_103_cast_fp16, y = const_453_promoted_to_fp16)[name = string("neg_3_cast_fp16")]; int32 const_454 = const()[name = string("const_454"), val = int32(-1)]; bool cat_29_interleave_0 = const()[name = string("cat_29_interleave_0"), val = bool(false)]; tensor cat_29_cast_fp16 = concat(axis = const_454, interleave = cat_29_interleave_0, values = (neg_3_cast_fp16, slice_102_cast_fp16))[name = string("cat_29_cast_fp16")]; tensor mul_78_cast_fp16 = mul(x = cat_29_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_78_cast_fp16")]; tensor add_24_cast_fp16 = add(x = mul_77_cast_fp16, y = mul_78_cast_fp16)[name = string("add_24_cast_fp16")]; int32 const_455 = const()[name = string("const_455"), val = int32(-2)]; bool cat_30_interleave_0 = const()[name = string("cat_30_interleave_0"), val = bool(false)]; tensor cat_30_cast_fp16 = concat(axis = const_455, interleave = cat_30_interleave_0, values = add_24_cast_fp16)[name = string("cat_30_cast_fp16")]; int32 const_456 = const()[name = string("const_456"), val = int32(-2)]; bool cat_31_interleave_0 = const()[name = string("cat_31_interleave_0"), val = bool(false)]; tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = view_8_cast_fp16)[name = string("transpose_89")]; tensor cat_31_cast_fp16 = concat(axis = const_456, interleave = cat_31_interleave_0, values = transpose_31_cast_fp16)[name = string("cat_31_cast_fp16")]; tensor unsqueeze_83_axes_0 = const()[name = string("unsqueeze_83_axes_0"), val = tensor([2])]; tensor unsqueeze_83_cast_fp16 = expand_dims(axes = unsqueeze_83_axes_0, x = cat_30_cast_fp16)[name = string("unsqueeze_83_cast_fp16")]; tensor expand_28_reps_0 = const()[name = string("expand_28_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_28_cast_fp16 = tile(reps = expand_28_reps_0, x = unsqueeze_83_cast_fp16)[name = string("expand_28_cast_fp16")]; tensor const_471 = const()[name = string("const_471"), val = tensor([1, 3, 512, 256])]; tensor view_9_cast_fp16 = reshape(shape = const_471, x = expand_28_cast_fp16)[name = string("view_9_cast_fp16")]; tensor unsqueeze_84_axes_0 = const()[name = string("unsqueeze_84_axes_0"), val = tensor([2])]; tensor unsqueeze_84_cast_fp16 = expand_dims(axes = unsqueeze_84_axes_0, x = cat_31_cast_fp16)[name = string("unsqueeze_84_cast_fp16")]; tensor expand_29_reps_0 = const()[name = string("expand_29_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_29_cast_fp16 = tile(reps = expand_29_reps_0, x = unsqueeze_84_cast_fp16)[name = string("expand_29_cast_fp16")]; tensor const_486 = const()[name = string("const_486"), val = tensor([1, 3, 512, 256])]; tensor view_10_cast_fp16 = reshape(shape = const_486, x = expand_29_cast_fp16)[name = string("view_10_cast_fp16")]; bool matmul_26_transpose_x_1 = const()[name = string("matmul_26_transpose_x_1"), val = bool(false)]; bool matmul_26_transpose_y_1 = const()[name = string("matmul_26_transpose_y_1"), val = bool(true)]; tensor matmul_26_cast_fp16 = matmul(transpose_x = matmul_26_transpose_x_1, transpose_y = matmul_26_transpose_y_1, x = add_23_cast_fp16, y = view_9_cast_fp16)[name = string("matmul_26_cast_fp16")]; fp16 const_489_to_fp16 = const()[name = string("const_489_to_fp16"), val = fp16(0x1p-4)]; tensor mul_79_cast_fp16 = mul(x = matmul_26_cast_fp16, y = const_489_to_fp16)[name = string("mul_79_cast_fp16")]; tensor add_25_cast_fp16 = add(x = mul_79_cast_fp16, y = expand_cast_fp16)[name = string("add_25_cast_fp16")]; int32 const_499 = const()[name = string("const_499"), val = int32(-1)]; tensor softmax_1_cast_fp16 = softmax(axis = const_499, x = add_25_cast_fp16)[name = string("softmax_1_cast_fp16")]; bool matmul_27_transpose_x_0 = const()[name = string("matmul_27_transpose_x_0"), val = bool(false)]; bool matmul_27_transpose_y_0 = const()[name = string("matmul_27_transpose_y_0"), val = bool(false)]; tensor matmul_27_cast_fp16 = matmul(transpose_x = matmul_27_transpose_x_0, transpose_y = matmul_27_transpose_y_0, x = softmax_1_cast_fp16, y = view_10_cast_fp16)[name = string("matmul_27_cast_fp16")]; tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_504 = const()[name = string("const_504"), val = tensor([1, 512, -1])]; tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = matmul_27_cast_fp16)[name = string("transpose_88")]; tensor view_11_cast_fp16 = reshape(shape = const_504, x = transpose_33_cast_fp16)[name = string("view_11_cast_fp16")]; tensor p_st_0_model_layers_1_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207352384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207942272))))[name = string("p_st_0_model_layers_1_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_10_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_1_self_attn_o_proj_weight_to_fp16_quantized, x = view_11_cast_fp16)[name = string("linear_10_cast_fp16")]; fp16 const_506_promoted_to_fp16 = const()[name = string("const_506_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_10_cast_fp16 = pow(x = linear_10_cast_fp16, y = const_506_promoted_to_fp16)[name = string("pow_10_cast_fp16")]; tensor mean_9_axes_0 = const()[name = string("mean_9_axes_0"), val = tensor([-1])]; bool mean_9_keep_dims_0 = const()[name = string("mean_9_keep_dims_0"), val = bool(true)]; tensor mean_9_cast_fp16 = reduce_mean(axes = mean_9_axes_0, keep_dims = mean_9_keep_dims_0, x = pow_10_cast_fp16)[name = string("mean_9_cast_fp16")]; fp16 const_509_to_fp16 = const()[name = string("const_509_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_26_cast_fp16 = add(x = mean_9_cast_fp16, y = const_509_to_fp16)[name = string("add_26_cast_fp16")]; fp32 rsqrt_9_epsilon_0 = const()[name = string("rsqrt_9_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_9_cast_fp16 = rsqrt(epsilon = rsqrt_9_epsilon_0, x = add_26_cast_fp16)[name = string("rsqrt_9_cast_fp16")]; tensor mul_80_cast_fp16 = mul(x = linear_10_cast_fp16, y = rsqrt_9_cast_fp16)[name = string("mul_80_cast_fp16")]; tensor add_27_to_fp16 = const()[name = string("add_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207943872)))]; tensor mul_81_cast_fp16 = mul(x = mul_80_cast_fp16, y = add_27_to_fp16)[name = string("mul_81_cast_fp16")]; tensor add_28_cast_fp16 = add(x = add_16_cast_fp16, y = mul_81_cast_fp16)[name = string("add_28_cast_fp16")]; fp16 const_514_promoted_to_fp16 = const()[name = string("const_514_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_11_cast_fp16 = pow(x = add_28_cast_fp16, y = const_514_promoted_to_fp16)[name = string("pow_11_cast_fp16")]; tensor mean_10_axes_0 = const()[name = string("mean_10_axes_0"), val = tensor([-1])]; bool mean_10_keep_dims_0 = const()[name = string("mean_10_keep_dims_0"), val = bool(true)]; tensor mean_10_cast_fp16 = reduce_mean(axes = mean_10_axes_0, keep_dims = mean_10_keep_dims_0, x = pow_11_cast_fp16)[name = string("mean_10_cast_fp16")]; fp16 const_517_to_fp16 = const()[name = string("const_517_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_29_cast_fp16 = add(x = mean_10_cast_fp16, y = const_517_to_fp16)[name = string("add_29_cast_fp16")]; fp32 rsqrt_10_epsilon_0 = const()[name = string("rsqrt_10_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_10_cast_fp16 = rsqrt(epsilon = rsqrt_10_epsilon_0, x = add_29_cast_fp16)[name = string("rsqrt_10_cast_fp16")]; tensor mul_82_cast_fp16 = mul(x = add_28_cast_fp16, y = rsqrt_10_cast_fp16)[name = string("mul_82_cast_fp16")]; tensor add_30_to_fp16 = const()[name = string("add_30_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207945472)))]; tensor mul_83_cast_fp16 = mul(x = mul_82_cast_fp16, y = add_30_to_fp16)[name = string("mul_83_cast_fp16")]; tensor p_st_0_model_layers_1_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207947072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(208831872))))[name = string("p_st_0_model_layers_1_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_11_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_1_mlp_gate_proj_weight_to_fp16_quantized, x = mul_83_cast_fp16)[name = string("linear_11_cast_fp16")]; string gelu_1_mode_0 = const()[name = string("gelu_1_mode_0"), val = string("EXACT")]; tensor gelu_1_cast_fp16 = gelu(mode = gelu_1_mode_0, x = linear_11_cast_fp16)[name = string("gelu_1_cast_fp16")]; tensor p_st_0_model_layers_1_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(208834240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(209719040))))[name = string("p_st_0_model_layers_1_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_12_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_1_mlp_up_proj_weight_to_fp16_quantized, x = mul_83_cast_fp16)[name = string("linear_12_cast_fp16")]; tensor mul_84_cast_fp16 = mul(x = gelu_1_cast_fp16, y = linear_12_cast_fp16)[name = string("mul_84_cast_fp16")]; tensor p_st_0_model_layers_1_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(209721408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210606208))))[name = string("p_st_0_model_layers_1_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_13_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_1_mlp_down_proj_weight_to_fp16_quantized, x = mul_84_cast_fp16)[name = string("linear_13_cast_fp16")]; fp16 const_522_promoted_to_fp16 = const()[name = string("const_522_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_12_cast_fp16 = pow(x = linear_13_cast_fp16, y = const_522_promoted_to_fp16)[name = string("pow_12_cast_fp16")]; tensor mean_11_axes_0 = const()[name = string("mean_11_axes_0"), val = tensor([-1])]; bool mean_11_keep_dims_0 = const()[name = string("mean_11_keep_dims_0"), val = bool(true)]; tensor mean_11_cast_fp16 = reduce_mean(axes = mean_11_axes_0, keep_dims = mean_11_keep_dims_0, x = pow_12_cast_fp16)[name = string("mean_11_cast_fp16")]; fp16 const_525_to_fp16 = const()[name = string("const_525_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_31_cast_fp16 = add(x = mean_11_cast_fp16, y = const_525_to_fp16)[name = string("add_31_cast_fp16")]; fp32 rsqrt_11_epsilon_0 = const()[name = string("rsqrt_11_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_11_cast_fp16 = rsqrt(epsilon = rsqrt_11_epsilon_0, x = add_31_cast_fp16)[name = string("rsqrt_11_cast_fp16")]; tensor mul_85_cast_fp16 = mul(x = linear_13_cast_fp16, y = rsqrt_11_cast_fp16)[name = string("mul_85_cast_fp16")]; tensor add_32_to_fp16 = const()[name = string("add_32_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210607808)))]; tensor mul_86_cast_fp16 = mul(x = mul_85_cast_fp16, y = add_32_to_fp16)[name = string("mul_86_cast_fp16")]; tensor add_33_cast_fp16 = add(x = add_28_cast_fp16, y = mul_86_cast_fp16)[name = string("add_33_cast_fp16")]; fp16 const_530_promoted_to_fp16 = const()[name = string("const_530_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_13_cast_fp16 = pow(x = add_33_cast_fp16, y = const_530_promoted_to_fp16)[name = string("pow_13_cast_fp16")]; tensor mean_12_axes_0 = const()[name = string("mean_12_axes_0"), val = tensor([-1])]; bool mean_12_keep_dims_0 = const()[name = string("mean_12_keep_dims_0"), val = bool(true)]; tensor mean_12_cast_fp16 = reduce_mean(axes = mean_12_axes_0, keep_dims = mean_12_keep_dims_0, x = pow_13_cast_fp16)[name = string("mean_12_cast_fp16")]; fp16 const_533_to_fp16 = const()[name = string("const_533_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_34_cast_fp16 = add(x = mean_12_cast_fp16, y = const_533_to_fp16)[name = string("add_34_cast_fp16")]; fp32 rsqrt_12_epsilon_0 = const()[name = string("rsqrt_12_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_12_cast_fp16 = rsqrt(epsilon = rsqrt_12_epsilon_0, x = add_34_cast_fp16)[name = string("rsqrt_12_cast_fp16")]; tensor mul_87_cast_fp16 = mul(x = add_33_cast_fp16, y = rsqrt_12_cast_fp16)[name = string("mul_87_cast_fp16")]; tensor add_35_to_fp16 = const()[name = string("add_35_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210609408)))]; tensor mul_88_cast_fp16 = mul(x = mul_87_cast_fp16, y = add_35_to_fp16)[name = string("mul_88_cast_fp16")]; tensor p_st_0_model_layers_2_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210611008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211200896))))[name = string("p_st_0_model_layers_2_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_14_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_2_self_attn_q_proj_weight_to_fp16_quantized, x = mul_88_cast_fp16)[name = string("linear_14_cast_fp16")]; tensor const_537 = const()[name = string("const_537"), val = tensor([1, 512, -1, 256])]; tensor view_12_cast_fp16 = reshape(shape = const_537, x = linear_14_cast_fp16)[name = string("view_12_cast_fp16")]; tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_2_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211202496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211399168))))[name = string("p_st_0_model_layers_2_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_15_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_2_self_attn_k_proj_weight_to_fp16_quantized, x = mul_88_cast_fp16)[name = string("linear_15_cast_fp16")]; tensor const_540 = const()[name = string("const_540"), val = tensor([1, 512, -1, 256])]; tensor view_13_cast_fp16 = reshape(shape = const_540, x = linear_15_cast_fp16)[name = string("view_13_cast_fp16")]; tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_2_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211399744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211596416))))[name = string("p_st_0_model_layers_2_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_16_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_2_self_attn_v_proj_weight_to_fp16_quantized, x = mul_88_cast_fp16)[name = string("linear_16_cast_fp16")]; tensor const_543 = const()[name = string("const_543"), val = tensor([1, 512, -1, 256])]; tensor view_14_cast_fp16 = reshape(shape = const_543, x = linear_16_cast_fp16)[name = string("view_14_cast_fp16")]; tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_547_promoted_to_fp16 = const()[name = string("const_547_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = view_12_cast_fp16)[name = string("transpose_87")]; tensor pow_14_cast_fp16 = pow(x = transpose_34_cast_fp16, y = const_547_promoted_to_fp16)[name = string("pow_14_cast_fp16")]; tensor mean_13_axes_0 = const()[name = string("mean_13_axes_0"), val = tensor([-1])]; bool mean_13_keep_dims_0 = const()[name = string("mean_13_keep_dims_0"), val = bool(true)]; tensor mean_13_cast_fp16 = reduce_mean(axes = mean_13_axes_0, keep_dims = mean_13_keep_dims_0, x = pow_14_cast_fp16)[name = string("mean_13_cast_fp16")]; fp16 const_550_to_fp16 = const()[name = string("const_550_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_36_cast_fp16 = add(x = mean_13_cast_fp16, y = const_550_to_fp16)[name = string("add_36_cast_fp16")]; fp32 rsqrt_13_epsilon_0 = const()[name = string("rsqrt_13_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_13_cast_fp16 = rsqrt(epsilon = rsqrt_13_epsilon_0, x = add_36_cast_fp16)[name = string("rsqrt_13_cast_fp16")]; tensor mul_89_cast_fp16 = mul(x = transpose_34_cast_fp16, y = rsqrt_13_cast_fp16)[name = string("mul_89_cast_fp16")]; tensor add_37_to_fp16 = const()[name = string("add_37_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211596992)))]; tensor mul_90_cast_fp16 = mul(x = mul_89_cast_fp16, y = add_37_to_fp16)[name = string("mul_90_cast_fp16")]; fp16 const_555_promoted_to_fp16 = const()[name = string("const_555_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = view_13_cast_fp16)[name = string("transpose_86")]; tensor pow_15_cast_fp16 = pow(x = transpose_35_cast_fp16, y = const_555_promoted_to_fp16)[name = string("pow_15_cast_fp16")]; tensor mean_14_axes_0 = const()[name = string("mean_14_axes_0"), val = tensor([-1])]; bool mean_14_keep_dims_0 = const()[name = string("mean_14_keep_dims_0"), val = bool(true)]; tensor mean_14_cast_fp16 = reduce_mean(axes = mean_14_axes_0, keep_dims = mean_14_keep_dims_0, x = pow_15_cast_fp16)[name = string("mean_14_cast_fp16")]; fp16 const_558_to_fp16 = const()[name = string("const_558_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_38_cast_fp16 = add(x = mean_14_cast_fp16, y = const_558_to_fp16)[name = string("add_38_cast_fp16")]; fp32 rsqrt_14_epsilon_0 = const()[name = string("rsqrt_14_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_14_cast_fp16 = rsqrt(epsilon = rsqrt_14_epsilon_0, x = add_38_cast_fp16)[name = string("rsqrt_14_cast_fp16")]; tensor mul_91_cast_fp16 = mul(x = transpose_35_cast_fp16, y = rsqrt_14_cast_fp16)[name = string("mul_91_cast_fp16")]; tensor add_39_to_fp16 = const()[name = string("add_39_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211597568)))]; tensor mul_92_cast_fp16 = mul(x = mul_91_cast_fp16, y = add_39_to_fp16)[name = string("mul_92_cast_fp16")]; tensor mul_93_cast_fp16 = mul(x = mul_90_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_93_cast_fp16")]; tensor slice_123_begin_0 = const()[name = string("slice_123_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_123_end_0 = const()[name = string("slice_123_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_123_end_mask_0 = const()[name = string("slice_123_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_123_cast_fp16 = slice_by_index(begin = slice_123_begin_0, end = slice_123_end_0, end_mask = slice_123_end_mask_0, x = mul_90_cast_fp16)[name = string("slice_123_cast_fp16")]; tensor slice_124_begin_0 = const()[name = string("slice_124_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_124_end_0 = const()[name = string("slice_124_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_124_end_mask_0 = const()[name = string("slice_124_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_124_cast_fp16 = slice_by_index(begin = slice_124_begin_0, end = slice_124_end_0, end_mask = slice_124_end_mask_0, x = mul_90_cast_fp16)[name = string("slice_124_cast_fp16")]; fp16 const_570_promoted_to_fp16 = const()[name = string("const_570_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_4_cast_fp16 = mul(x = slice_124_cast_fp16, y = const_570_promoted_to_fp16)[name = string("neg_4_cast_fp16")]; int32 const_571 = const()[name = string("const_571"), val = int32(-1)]; bool cat_32_interleave_0 = const()[name = string("cat_32_interleave_0"), val = bool(false)]; tensor cat_32_cast_fp16 = concat(axis = const_571, interleave = cat_32_interleave_0, values = (neg_4_cast_fp16, slice_123_cast_fp16))[name = string("cat_32_cast_fp16")]; tensor mul_94_cast_fp16 = mul(x = cat_32_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_94_cast_fp16")]; tensor add_40_cast_fp16 = add(x = mul_93_cast_fp16, y = mul_94_cast_fp16)[name = string("add_40_cast_fp16")]; tensor mul_95_cast_fp16 = mul(x = mul_92_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_95_cast_fp16")]; tensor slice_125_begin_0 = const()[name = string("slice_125_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_125_end_0 = const()[name = string("slice_125_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_125_end_mask_0 = const()[name = string("slice_125_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_125_cast_fp16 = slice_by_index(begin = slice_125_begin_0, end = slice_125_end_0, end_mask = slice_125_end_mask_0, x = mul_92_cast_fp16)[name = string("slice_125_cast_fp16")]; tensor slice_126_begin_0 = const()[name = string("slice_126_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_126_end_0 = const()[name = string("slice_126_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_126_end_mask_0 = const()[name = string("slice_126_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_126_cast_fp16 = slice_by_index(begin = slice_126_begin_0, end = slice_126_end_0, end_mask = slice_126_end_mask_0, x = mul_92_cast_fp16)[name = string("slice_126_cast_fp16")]; fp16 const_578_promoted_to_fp16 = const()[name = string("const_578_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_5_cast_fp16 = mul(x = slice_126_cast_fp16, y = const_578_promoted_to_fp16)[name = string("neg_5_cast_fp16")]; int32 const_579 = const()[name = string("const_579"), val = int32(-1)]; bool cat_33_interleave_0 = const()[name = string("cat_33_interleave_0"), val = bool(false)]; tensor cat_33_cast_fp16 = concat(axis = const_579, interleave = cat_33_interleave_0, values = (neg_5_cast_fp16, slice_125_cast_fp16))[name = string("cat_33_cast_fp16")]; tensor mul_96_cast_fp16 = mul(x = cat_33_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_96_cast_fp16")]; tensor add_41_cast_fp16 = add(x = mul_95_cast_fp16, y = mul_96_cast_fp16)[name = string("add_41_cast_fp16")]; int32 const_580 = const()[name = string("const_580"), val = int32(-2)]; bool cat_34_interleave_0 = const()[name = string("cat_34_interleave_0"), val = bool(false)]; tensor cat_34_cast_fp16 = concat(axis = const_580, interleave = cat_34_interleave_0, values = add_41_cast_fp16)[name = string("cat_34_cast_fp16")]; int32 const_581 = const()[name = string("const_581"), val = int32(-2)]; bool cat_35_interleave_0 = const()[name = string("cat_35_interleave_0"), val = bool(false)]; tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = view_14_cast_fp16)[name = string("transpose_85")]; tensor cat_35_cast_fp16 = concat(axis = const_581, interleave = cat_35_interleave_0, values = transpose_36_cast_fp16)[name = string("cat_35_cast_fp16")]; tensor unsqueeze_87_axes_0 = const()[name = string("unsqueeze_87_axes_0"), val = tensor([2])]; tensor unsqueeze_87_cast_fp16 = expand_dims(axes = unsqueeze_87_axes_0, x = cat_34_cast_fp16)[name = string("unsqueeze_87_cast_fp16")]; tensor expand_30_reps_0 = const()[name = string("expand_30_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_30_cast_fp16 = tile(reps = expand_30_reps_0, x = unsqueeze_87_cast_fp16)[name = string("expand_30_cast_fp16")]; tensor const_596 = const()[name = string("const_596"), val = tensor([1, 3, 512, 256])]; tensor view_15_cast_fp16 = reshape(shape = const_596, x = expand_30_cast_fp16)[name = string("view_15_cast_fp16")]; tensor unsqueeze_88_axes_0 = const()[name = string("unsqueeze_88_axes_0"), val = tensor([2])]; tensor unsqueeze_88_cast_fp16 = expand_dims(axes = unsqueeze_88_axes_0, x = cat_35_cast_fp16)[name = string("unsqueeze_88_cast_fp16")]; tensor expand_31_reps_0 = const()[name = string("expand_31_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_31_cast_fp16 = tile(reps = expand_31_reps_0, x = unsqueeze_88_cast_fp16)[name = string("expand_31_cast_fp16")]; tensor const_611 = const()[name = string("const_611"), val = tensor([1, 3, 512, 256])]; tensor view_16_cast_fp16 = reshape(shape = const_611, x = expand_31_cast_fp16)[name = string("view_16_cast_fp16")]; bool matmul_28_transpose_x_1 = const()[name = string("matmul_28_transpose_x_1"), val = bool(false)]; bool matmul_28_transpose_y_1 = const()[name = string("matmul_28_transpose_y_1"), val = bool(true)]; tensor matmul_28_cast_fp16 = matmul(transpose_x = matmul_28_transpose_x_1, transpose_y = matmul_28_transpose_y_1, x = add_40_cast_fp16, y = view_15_cast_fp16)[name = string("matmul_28_cast_fp16")]; fp16 const_614_to_fp16 = const()[name = string("const_614_to_fp16"), val = fp16(0x1p-4)]; tensor mul_97_cast_fp16 = mul(x = matmul_28_cast_fp16, y = const_614_to_fp16)[name = string("mul_97_cast_fp16")]; tensor add_42_cast_fp16 = add(x = mul_97_cast_fp16, y = expand_cast_fp16)[name = string("add_42_cast_fp16")]; int32 const_624 = const()[name = string("const_624"), val = int32(-1)]; tensor softmax_2_cast_fp16 = softmax(axis = const_624, x = add_42_cast_fp16)[name = string("softmax_2_cast_fp16")]; bool matmul_29_transpose_x_0 = const()[name = string("matmul_29_transpose_x_0"), val = bool(false)]; bool matmul_29_transpose_y_0 = const()[name = string("matmul_29_transpose_y_0"), val = bool(false)]; tensor matmul_29_cast_fp16 = matmul(transpose_x = matmul_29_transpose_x_0, transpose_y = matmul_29_transpose_y_0, x = softmax_2_cast_fp16, y = view_16_cast_fp16)[name = string("matmul_29_cast_fp16")]; tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_629 = const()[name = string("const_629"), val = tensor([1, 512, -1])]; tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = matmul_29_cast_fp16)[name = string("transpose_84")]; tensor view_17_cast_fp16 = reshape(shape = const_629, x = transpose_38_cast_fp16)[name = string("view_17_cast_fp16")]; tensor p_st_0_model_layers_2_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211598144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(212188032))))[name = string("p_st_0_model_layers_2_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_17_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_2_self_attn_o_proj_weight_to_fp16_quantized, x = view_17_cast_fp16)[name = string("linear_17_cast_fp16")]; fp16 const_631_promoted_to_fp16 = const()[name = string("const_631_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_16_cast_fp16 = pow(x = linear_17_cast_fp16, y = const_631_promoted_to_fp16)[name = string("pow_16_cast_fp16")]; tensor mean_15_axes_0 = const()[name = string("mean_15_axes_0"), val = tensor([-1])]; bool mean_15_keep_dims_0 = const()[name = string("mean_15_keep_dims_0"), val = bool(true)]; tensor mean_15_cast_fp16 = reduce_mean(axes = mean_15_axes_0, keep_dims = mean_15_keep_dims_0, x = pow_16_cast_fp16)[name = string("mean_15_cast_fp16")]; fp16 const_634_to_fp16 = const()[name = string("const_634_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_43_cast_fp16 = add(x = mean_15_cast_fp16, y = const_634_to_fp16)[name = string("add_43_cast_fp16")]; fp32 rsqrt_15_epsilon_0 = const()[name = string("rsqrt_15_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_15_cast_fp16 = rsqrt(epsilon = rsqrt_15_epsilon_0, x = add_43_cast_fp16)[name = string("rsqrt_15_cast_fp16")]; tensor mul_98_cast_fp16 = mul(x = linear_17_cast_fp16, y = rsqrt_15_cast_fp16)[name = string("mul_98_cast_fp16")]; tensor add_44_to_fp16 = const()[name = string("add_44_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(212189632)))]; tensor mul_99_cast_fp16 = mul(x = mul_98_cast_fp16, y = add_44_to_fp16)[name = string("mul_99_cast_fp16")]; tensor add_45_cast_fp16 = add(x = add_33_cast_fp16, y = mul_99_cast_fp16)[name = string("add_45_cast_fp16")]; fp16 const_639_promoted_to_fp16 = const()[name = string("const_639_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_17_cast_fp16 = pow(x = add_45_cast_fp16, y = const_639_promoted_to_fp16)[name = string("pow_17_cast_fp16")]; tensor mean_16_axes_0 = const()[name = string("mean_16_axes_0"), val = tensor([-1])]; bool mean_16_keep_dims_0 = const()[name = string("mean_16_keep_dims_0"), val = bool(true)]; tensor mean_16_cast_fp16 = reduce_mean(axes = mean_16_axes_0, keep_dims = mean_16_keep_dims_0, x = pow_17_cast_fp16)[name = string("mean_16_cast_fp16")]; fp16 const_642_to_fp16 = const()[name = string("const_642_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_46_cast_fp16 = add(x = mean_16_cast_fp16, y = const_642_to_fp16)[name = string("add_46_cast_fp16")]; fp32 rsqrt_16_epsilon_0 = const()[name = string("rsqrt_16_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_16_cast_fp16 = rsqrt(epsilon = rsqrt_16_epsilon_0, x = add_46_cast_fp16)[name = string("rsqrt_16_cast_fp16")]; tensor mul_100_cast_fp16 = mul(x = add_45_cast_fp16, y = rsqrt_16_cast_fp16)[name = string("mul_100_cast_fp16")]; tensor add_47_to_fp16 = const()[name = string("add_47_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(212191232)))]; tensor mul_101_cast_fp16 = mul(x = mul_100_cast_fp16, y = add_47_to_fp16)[name = string("mul_101_cast_fp16")]; tensor p_st_0_model_layers_2_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(212192832))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213077632))))[name = string("p_st_0_model_layers_2_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_18_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_2_mlp_gate_proj_weight_to_fp16_quantized, x = mul_101_cast_fp16)[name = string("linear_18_cast_fp16")]; string gelu_2_mode_0 = const()[name = string("gelu_2_mode_0"), val = string("EXACT")]; tensor gelu_2_cast_fp16 = gelu(mode = gelu_2_mode_0, x = linear_18_cast_fp16)[name = string("gelu_2_cast_fp16")]; tensor p_st_0_model_layers_2_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213080000))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213964800))))[name = string("p_st_0_model_layers_2_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_19_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_2_mlp_up_proj_weight_to_fp16_quantized, x = mul_101_cast_fp16)[name = string("linear_19_cast_fp16")]; tensor mul_102_cast_fp16 = mul(x = gelu_2_cast_fp16, y = linear_19_cast_fp16)[name = string("mul_102_cast_fp16")]; tensor p_st_0_model_layers_2_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213967168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(214851968))))[name = string("p_st_0_model_layers_2_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_20_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_2_mlp_down_proj_weight_to_fp16_quantized, x = mul_102_cast_fp16)[name = string("linear_20_cast_fp16")]; fp16 const_647_promoted_to_fp16 = const()[name = string("const_647_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_18_cast_fp16 = pow(x = linear_20_cast_fp16, y = const_647_promoted_to_fp16)[name = string("pow_18_cast_fp16")]; tensor mean_17_axes_0 = const()[name = string("mean_17_axes_0"), val = tensor([-1])]; bool mean_17_keep_dims_0 = const()[name = string("mean_17_keep_dims_0"), val = bool(true)]; tensor mean_17_cast_fp16 = reduce_mean(axes = mean_17_axes_0, keep_dims = mean_17_keep_dims_0, x = pow_18_cast_fp16)[name = string("mean_17_cast_fp16")]; fp16 const_650_to_fp16 = const()[name = string("const_650_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_48_cast_fp16 = add(x = mean_17_cast_fp16, y = const_650_to_fp16)[name = string("add_48_cast_fp16")]; fp32 rsqrt_17_epsilon_0 = const()[name = string("rsqrt_17_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_17_cast_fp16 = rsqrt(epsilon = rsqrt_17_epsilon_0, x = add_48_cast_fp16)[name = string("rsqrt_17_cast_fp16")]; tensor mul_103_cast_fp16 = mul(x = linear_20_cast_fp16, y = rsqrt_17_cast_fp16)[name = string("mul_103_cast_fp16")]; tensor add_49_to_fp16 = const()[name = string("add_49_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(214853568)))]; tensor mul_104_cast_fp16 = mul(x = mul_103_cast_fp16, y = add_49_to_fp16)[name = string("mul_104_cast_fp16")]; tensor add_50_cast_fp16 = add(x = add_45_cast_fp16, y = mul_104_cast_fp16)[name = string("add_50_cast_fp16")]; fp16 const_655_promoted_to_fp16 = const()[name = string("const_655_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_19_cast_fp16 = pow(x = add_50_cast_fp16, y = const_655_promoted_to_fp16)[name = string("pow_19_cast_fp16")]; tensor mean_18_axes_0 = const()[name = string("mean_18_axes_0"), val = tensor([-1])]; bool mean_18_keep_dims_0 = const()[name = string("mean_18_keep_dims_0"), val = bool(true)]; tensor mean_18_cast_fp16 = reduce_mean(axes = mean_18_axes_0, keep_dims = mean_18_keep_dims_0, x = pow_19_cast_fp16)[name = string("mean_18_cast_fp16")]; fp16 const_658_to_fp16 = const()[name = string("const_658_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_51_cast_fp16 = add(x = mean_18_cast_fp16, y = const_658_to_fp16)[name = string("add_51_cast_fp16")]; fp32 rsqrt_18_epsilon_0 = const()[name = string("rsqrt_18_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_18_cast_fp16 = rsqrt(epsilon = rsqrt_18_epsilon_0, x = add_51_cast_fp16)[name = string("rsqrt_18_cast_fp16")]; tensor mul_105_cast_fp16 = mul(x = add_50_cast_fp16, y = rsqrt_18_cast_fp16)[name = string("mul_105_cast_fp16")]; tensor add_52_to_fp16 = const()[name = string("add_52_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(214855168)))]; tensor mul_106_cast_fp16 = mul(x = mul_105_cast_fp16, y = add_52_to_fp16)[name = string("mul_106_cast_fp16")]; tensor p_st_0_model_layers_3_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(214856768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215446656))))[name = string("p_st_0_model_layers_3_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_21_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_3_self_attn_q_proj_weight_to_fp16_quantized, x = mul_106_cast_fp16)[name = string("linear_21_cast_fp16")]; tensor const_662 = const()[name = string("const_662"), val = tensor([1, 512, -1, 256])]; tensor view_18_cast_fp16 = reshape(shape = const_662, x = linear_21_cast_fp16)[name = string("view_18_cast_fp16")]; tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_3_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215448256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215644928))))[name = string("p_st_0_model_layers_3_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_22_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_3_self_attn_k_proj_weight_to_fp16_quantized, x = mul_106_cast_fp16)[name = string("linear_22_cast_fp16")]; tensor const_665 = const()[name = string("const_665"), val = tensor([1, 512, -1, 256])]; tensor view_19_cast_fp16 = reshape(shape = const_665, x = linear_22_cast_fp16)[name = string("view_19_cast_fp16")]; tensor transpose_40_perm_0 = const()[name = string("transpose_40_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_3_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215645504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215842176))))[name = string("p_st_0_model_layers_3_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_23_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_3_self_attn_v_proj_weight_to_fp16_quantized, x = mul_106_cast_fp16)[name = string("linear_23_cast_fp16")]; tensor const_668 = const()[name = string("const_668"), val = tensor([1, 512, -1, 256])]; tensor view_20_cast_fp16 = reshape(shape = const_668, x = linear_23_cast_fp16)[name = string("view_20_cast_fp16")]; tensor transpose_41_perm_0 = const()[name = string("transpose_41_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_672_promoted_to_fp16 = const()[name = string("const_672_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = view_18_cast_fp16)[name = string("transpose_83")]; tensor pow_20_cast_fp16 = pow(x = transpose_39_cast_fp16, y = const_672_promoted_to_fp16)[name = string("pow_20_cast_fp16")]; tensor mean_19_axes_0 = const()[name = string("mean_19_axes_0"), val = tensor([-1])]; bool mean_19_keep_dims_0 = const()[name = string("mean_19_keep_dims_0"), val = bool(true)]; tensor mean_19_cast_fp16 = reduce_mean(axes = mean_19_axes_0, keep_dims = mean_19_keep_dims_0, x = pow_20_cast_fp16)[name = string("mean_19_cast_fp16")]; fp16 const_675_to_fp16 = const()[name = string("const_675_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_53_cast_fp16 = add(x = mean_19_cast_fp16, y = const_675_to_fp16)[name = string("add_53_cast_fp16")]; fp32 rsqrt_19_epsilon_0 = const()[name = string("rsqrt_19_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_19_cast_fp16 = rsqrt(epsilon = rsqrt_19_epsilon_0, x = add_53_cast_fp16)[name = string("rsqrt_19_cast_fp16")]; tensor mul_107_cast_fp16 = mul(x = transpose_39_cast_fp16, y = rsqrt_19_cast_fp16)[name = string("mul_107_cast_fp16")]; tensor add_54_to_fp16 = const()[name = string("add_54_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215842752)))]; tensor mul_108_cast_fp16 = mul(x = mul_107_cast_fp16, y = add_54_to_fp16)[name = string("mul_108_cast_fp16")]; fp16 const_680_promoted_to_fp16 = const()[name = string("const_680_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_40_cast_fp16 = transpose(perm = transpose_40_perm_0, x = view_19_cast_fp16)[name = string("transpose_82")]; tensor pow_21_cast_fp16 = pow(x = transpose_40_cast_fp16, y = const_680_promoted_to_fp16)[name = string("pow_21_cast_fp16")]; tensor mean_20_axes_0 = const()[name = string("mean_20_axes_0"), val = tensor([-1])]; bool mean_20_keep_dims_0 = const()[name = string("mean_20_keep_dims_0"), val = bool(true)]; tensor mean_20_cast_fp16 = reduce_mean(axes = mean_20_axes_0, keep_dims = mean_20_keep_dims_0, x = pow_21_cast_fp16)[name = string("mean_20_cast_fp16")]; fp16 const_683_to_fp16 = const()[name = string("const_683_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_55_cast_fp16 = add(x = mean_20_cast_fp16, y = const_683_to_fp16)[name = string("add_55_cast_fp16")]; fp32 rsqrt_20_epsilon_0 = const()[name = string("rsqrt_20_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_20_cast_fp16 = rsqrt(epsilon = rsqrt_20_epsilon_0, x = add_55_cast_fp16)[name = string("rsqrt_20_cast_fp16")]; tensor mul_109_cast_fp16 = mul(x = transpose_40_cast_fp16, y = rsqrt_20_cast_fp16)[name = string("mul_109_cast_fp16")]; tensor add_56_to_fp16 = const()[name = string("add_56_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215843328)))]; tensor mul_110_cast_fp16 = mul(x = mul_109_cast_fp16, y = add_56_to_fp16)[name = string("mul_110_cast_fp16")]; tensor mul_111_cast_fp16 = mul(x = mul_108_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_111_cast_fp16")]; tensor slice_146_begin_0 = const()[name = string("slice_146_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_146_end_0 = const()[name = string("slice_146_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_146_end_mask_0 = const()[name = string("slice_146_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_146_cast_fp16 = slice_by_index(begin = slice_146_begin_0, end = slice_146_end_0, end_mask = slice_146_end_mask_0, x = mul_108_cast_fp16)[name = string("slice_146_cast_fp16")]; tensor slice_147_begin_0 = const()[name = string("slice_147_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_147_end_0 = const()[name = string("slice_147_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_147_end_mask_0 = const()[name = string("slice_147_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_147_cast_fp16 = slice_by_index(begin = slice_147_begin_0, end = slice_147_end_0, end_mask = slice_147_end_mask_0, x = mul_108_cast_fp16)[name = string("slice_147_cast_fp16")]; fp16 const_695_promoted_to_fp16 = const()[name = string("const_695_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_6_cast_fp16 = mul(x = slice_147_cast_fp16, y = const_695_promoted_to_fp16)[name = string("neg_6_cast_fp16")]; int32 const_696 = const()[name = string("const_696"), val = int32(-1)]; bool cat_36_interleave_0 = const()[name = string("cat_36_interleave_0"), val = bool(false)]; tensor cat_36_cast_fp16 = concat(axis = const_696, interleave = cat_36_interleave_0, values = (neg_6_cast_fp16, slice_146_cast_fp16))[name = string("cat_36_cast_fp16")]; tensor mul_112_cast_fp16 = mul(x = cat_36_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_112_cast_fp16")]; tensor add_57_cast_fp16 = add(x = mul_111_cast_fp16, y = mul_112_cast_fp16)[name = string("add_57_cast_fp16")]; tensor mul_113_cast_fp16 = mul(x = mul_110_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_113_cast_fp16")]; tensor slice_148_begin_0 = const()[name = string("slice_148_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_148_end_0 = const()[name = string("slice_148_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_148_end_mask_0 = const()[name = string("slice_148_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_148_cast_fp16 = slice_by_index(begin = slice_148_begin_0, end = slice_148_end_0, end_mask = slice_148_end_mask_0, x = mul_110_cast_fp16)[name = string("slice_148_cast_fp16")]; tensor slice_149_begin_0 = const()[name = string("slice_149_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_149_end_0 = const()[name = string("slice_149_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_149_end_mask_0 = const()[name = string("slice_149_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_149_cast_fp16 = slice_by_index(begin = slice_149_begin_0, end = slice_149_end_0, end_mask = slice_149_end_mask_0, x = mul_110_cast_fp16)[name = string("slice_149_cast_fp16")]; fp16 const_703_promoted_to_fp16 = const()[name = string("const_703_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_7_cast_fp16 = mul(x = slice_149_cast_fp16, y = const_703_promoted_to_fp16)[name = string("neg_7_cast_fp16")]; int32 const_704 = const()[name = string("const_704"), val = int32(-1)]; bool cat_37_interleave_0 = const()[name = string("cat_37_interleave_0"), val = bool(false)]; tensor cat_37_cast_fp16 = concat(axis = const_704, interleave = cat_37_interleave_0, values = (neg_7_cast_fp16, slice_148_cast_fp16))[name = string("cat_37_cast_fp16")]; tensor mul_114_cast_fp16 = mul(x = cat_37_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_114_cast_fp16")]; tensor add_58_cast_fp16 = add(x = mul_113_cast_fp16, y = mul_114_cast_fp16)[name = string("add_58_cast_fp16")]; int32 const_705 = const()[name = string("const_705"), val = int32(-2)]; bool cat_38_interleave_0 = const()[name = string("cat_38_interleave_0"), val = bool(false)]; tensor cat_38_cast_fp16 = concat(axis = const_705, interleave = cat_38_interleave_0, values = add_58_cast_fp16)[name = string("cat_38_cast_fp16")]; int32 const_706 = const()[name = string("const_706"), val = int32(-2)]; bool cat_39_interleave_0 = const()[name = string("cat_39_interleave_0"), val = bool(false)]; tensor transpose_41_cast_fp16 = transpose(perm = transpose_41_perm_0, x = view_20_cast_fp16)[name = string("transpose_81")]; tensor cat_39_cast_fp16 = concat(axis = const_706, interleave = cat_39_interleave_0, values = transpose_41_cast_fp16)[name = string("cat_39_cast_fp16")]; tensor unsqueeze_91_axes_0 = const()[name = string("unsqueeze_91_axes_0"), val = tensor([2])]; tensor unsqueeze_91_cast_fp16 = expand_dims(axes = unsqueeze_91_axes_0, x = cat_38_cast_fp16)[name = string("unsqueeze_91_cast_fp16")]; tensor expand_32_reps_0 = const()[name = string("expand_32_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_32_cast_fp16 = tile(reps = expand_32_reps_0, x = unsqueeze_91_cast_fp16)[name = string("expand_32_cast_fp16")]; tensor const_721 = const()[name = string("const_721"), val = tensor([1, 3, 512, 256])]; tensor view_21_cast_fp16 = reshape(shape = const_721, x = expand_32_cast_fp16)[name = string("view_21_cast_fp16")]; tensor unsqueeze_92_axes_0 = const()[name = string("unsqueeze_92_axes_0"), val = tensor([2])]; tensor unsqueeze_92_cast_fp16 = expand_dims(axes = unsqueeze_92_axes_0, x = cat_39_cast_fp16)[name = string("unsqueeze_92_cast_fp16")]; tensor expand_33_reps_0 = const()[name = string("expand_33_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_33_cast_fp16 = tile(reps = expand_33_reps_0, x = unsqueeze_92_cast_fp16)[name = string("expand_33_cast_fp16")]; tensor const_736 = const()[name = string("const_736"), val = tensor([1, 3, 512, 256])]; tensor view_22_cast_fp16 = reshape(shape = const_736, x = expand_33_cast_fp16)[name = string("view_22_cast_fp16")]; bool matmul_30_transpose_x_1 = const()[name = string("matmul_30_transpose_x_1"), val = bool(false)]; bool matmul_30_transpose_y_1 = const()[name = string("matmul_30_transpose_y_1"), val = bool(true)]; tensor matmul_30_cast_fp16 = matmul(transpose_x = matmul_30_transpose_x_1, transpose_y = matmul_30_transpose_y_1, x = add_57_cast_fp16, y = view_21_cast_fp16)[name = string("matmul_30_cast_fp16")]; fp16 const_739_to_fp16 = const()[name = string("const_739_to_fp16"), val = fp16(0x1p-4)]; tensor mul_115_cast_fp16 = mul(x = matmul_30_cast_fp16, y = const_739_to_fp16)[name = string("mul_115_cast_fp16")]; tensor add_59_cast_fp16 = add(x = mul_115_cast_fp16, y = expand_cast_fp16)[name = string("add_59_cast_fp16")]; int32 const_749 = const()[name = string("const_749"), val = int32(-1)]; tensor softmax_3_cast_fp16 = softmax(axis = const_749, x = add_59_cast_fp16)[name = string("softmax_3_cast_fp16")]; bool matmul_31_transpose_x_0 = const()[name = string("matmul_31_transpose_x_0"), val = bool(false)]; bool matmul_31_transpose_y_0 = const()[name = string("matmul_31_transpose_y_0"), val = bool(false)]; tensor matmul_31_cast_fp16 = matmul(transpose_x = matmul_31_transpose_x_0, transpose_y = matmul_31_transpose_y_0, x = softmax_3_cast_fp16, y = view_22_cast_fp16)[name = string("matmul_31_cast_fp16")]; tensor transpose_43_perm_0 = const()[name = string("transpose_43_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_754 = const()[name = string("const_754"), val = tensor([1, 512, -1])]; tensor transpose_43_cast_fp16 = transpose(perm = transpose_43_perm_0, x = matmul_31_cast_fp16)[name = string("transpose_80")]; tensor view_23_cast_fp16 = reshape(shape = const_754, x = transpose_43_cast_fp16)[name = string("view_23_cast_fp16")]; tensor p_st_0_model_layers_3_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215843904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216433792))))[name = string("p_st_0_model_layers_3_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_24_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_3_self_attn_o_proj_weight_to_fp16_quantized, x = view_23_cast_fp16)[name = string("linear_24_cast_fp16")]; fp16 const_756_promoted_to_fp16 = const()[name = string("const_756_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_22_cast_fp16 = pow(x = linear_24_cast_fp16, y = const_756_promoted_to_fp16)[name = string("pow_22_cast_fp16")]; tensor mean_21_axes_0 = const()[name = string("mean_21_axes_0"), val = tensor([-1])]; bool mean_21_keep_dims_0 = const()[name = string("mean_21_keep_dims_0"), val = bool(true)]; tensor mean_21_cast_fp16 = reduce_mean(axes = mean_21_axes_0, keep_dims = mean_21_keep_dims_0, x = pow_22_cast_fp16)[name = string("mean_21_cast_fp16")]; fp16 const_759_to_fp16 = const()[name = string("const_759_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_60_cast_fp16 = add(x = mean_21_cast_fp16, y = const_759_to_fp16)[name = string("add_60_cast_fp16")]; fp32 rsqrt_21_epsilon_0 = const()[name = string("rsqrt_21_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_21_cast_fp16 = rsqrt(epsilon = rsqrt_21_epsilon_0, x = add_60_cast_fp16)[name = string("rsqrt_21_cast_fp16")]; tensor mul_116_cast_fp16 = mul(x = linear_24_cast_fp16, y = rsqrt_21_cast_fp16)[name = string("mul_116_cast_fp16")]; tensor add_61_to_fp16 = const()[name = string("add_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216435392)))]; tensor mul_117_cast_fp16 = mul(x = mul_116_cast_fp16, y = add_61_to_fp16)[name = string("mul_117_cast_fp16")]; tensor add_62_cast_fp16 = add(x = add_50_cast_fp16, y = mul_117_cast_fp16)[name = string("add_62_cast_fp16")]; fp16 const_764_promoted_to_fp16 = const()[name = string("const_764_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_23_cast_fp16 = pow(x = add_62_cast_fp16, y = const_764_promoted_to_fp16)[name = string("pow_23_cast_fp16")]; tensor mean_22_axes_0 = const()[name = string("mean_22_axes_0"), val = tensor([-1])]; bool mean_22_keep_dims_0 = const()[name = string("mean_22_keep_dims_0"), val = bool(true)]; tensor mean_22_cast_fp16 = reduce_mean(axes = mean_22_axes_0, keep_dims = mean_22_keep_dims_0, x = pow_23_cast_fp16)[name = string("mean_22_cast_fp16")]; fp16 const_767_to_fp16 = const()[name = string("const_767_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_63_cast_fp16 = add(x = mean_22_cast_fp16, y = const_767_to_fp16)[name = string("add_63_cast_fp16")]; fp32 rsqrt_22_epsilon_0 = const()[name = string("rsqrt_22_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_22_cast_fp16 = rsqrt(epsilon = rsqrt_22_epsilon_0, x = add_63_cast_fp16)[name = string("rsqrt_22_cast_fp16")]; tensor mul_118_cast_fp16 = mul(x = add_62_cast_fp16, y = rsqrt_22_cast_fp16)[name = string("mul_118_cast_fp16")]; tensor add_64_to_fp16 = const()[name = string("add_64_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216436992)))]; tensor mul_119_cast_fp16 = mul(x = mul_118_cast_fp16, y = add_64_to_fp16)[name = string("mul_119_cast_fp16")]; tensor p_st_0_model_layers_3_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216438592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217323392))))[name = string("p_st_0_model_layers_3_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_25_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_3_mlp_gate_proj_weight_to_fp16_quantized, x = mul_119_cast_fp16)[name = string("linear_25_cast_fp16")]; string gelu_3_mode_0 = const()[name = string("gelu_3_mode_0"), val = string("EXACT")]; tensor gelu_3_cast_fp16 = gelu(mode = gelu_3_mode_0, x = linear_25_cast_fp16)[name = string("gelu_3_cast_fp16")]; tensor p_st_0_model_layers_3_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217325760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218210560))))[name = string("p_st_0_model_layers_3_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_26_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_3_mlp_up_proj_weight_to_fp16_quantized, x = mul_119_cast_fp16)[name = string("linear_26_cast_fp16")]; tensor mul_120_cast_fp16 = mul(x = gelu_3_cast_fp16, y = linear_26_cast_fp16)[name = string("mul_120_cast_fp16")]; tensor p_st_0_model_layers_3_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218212928))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219097728))))[name = string("p_st_0_model_layers_3_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_27_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_3_mlp_down_proj_weight_to_fp16_quantized, x = mul_120_cast_fp16)[name = string("linear_27_cast_fp16")]; fp16 const_772_promoted_to_fp16 = const()[name = string("const_772_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_24_cast_fp16 = pow(x = linear_27_cast_fp16, y = const_772_promoted_to_fp16)[name = string("pow_24_cast_fp16")]; tensor mean_23_axes_0 = const()[name = string("mean_23_axes_0"), val = tensor([-1])]; bool mean_23_keep_dims_0 = const()[name = string("mean_23_keep_dims_0"), val = bool(true)]; tensor mean_23_cast_fp16 = reduce_mean(axes = mean_23_axes_0, keep_dims = mean_23_keep_dims_0, x = pow_24_cast_fp16)[name = string("mean_23_cast_fp16")]; fp16 const_775_to_fp16 = const()[name = string("const_775_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_65_cast_fp16 = add(x = mean_23_cast_fp16, y = const_775_to_fp16)[name = string("add_65_cast_fp16")]; fp32 rsqrt_23_epsilon_0 = const()[name = string("rsqrt_23_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_23_cast_fp16 = rsqrt(epsilon = rsqrt_23_epsilon_0, x = add_65_cast_fp16)[name = string("rsqrt_23_cast_fp16")]; tensor mul_121_cast_fp16 = mul(x = linear_27_cast_fp16, y = rsqrt_23_cast_fp16)[name = string("mul_121_cast_fp16")]; tensor add_66_to_fp16 = const()[name = string("add_66_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219099328)))]; tensor mul_122_cast_fp16 = mul(x = mul_121_cast_fp16, y = add_66_to_fp16)[name = string("mul_122_cast_fp16")]; tensor add_67_cast_fp16 = add(x = add_62_cast_fp16, y = mul_122_cast_fp16)[name = string("add_67_cast_fp16")]; fp16 const_780_promoted_to_fp16 = const()[name = string("const_780_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_25_cast_fp16 = pow(x = add_67_cast_fp16, y = const_780_promoted_to_fp16)[name = string("pow_25_cast_fp16")]; tensor mean_24_axes_0 = const()[name = string("mean_24_axes_0"), val = tensor([-1])]; bool mean_24_keep_dims_0 = const()[name = string("mean_24_keep_dims_0"), val = bool(true)]; tensor mean_24_cast_fp16 = reduce_mean(axes = mean_24_axes_0, keep_dims = mean_24_keep_dims_0, x = pow_25_cast_fp16)[name = string("mean_24_cast_fp16")]; fp16 const_783_to_fp16 = const()[name = string("const_783_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_68_cast_fp16 = add(x = mean_24_cast_fp16, y = const_783_to_fp16)[name = string("add_68_cast_fp16")]; fp32 rsqrt_24_epsilon_0 = const()[name = string("rsqrt_24_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_24_cast_fp16 = rsqrt(epsilon = rsqrt_24_epsilon_0, x = add_68_cast_fp16)[name = string("rsqrt_24_cast_fp16")]; tensor mul_123_cast_fp16 = mul(x = add_67_cast_fp16, y = rsqrt_24_cast_fp16)[name = string("mul_123_cast_fp16")]; tensor add_69_to_fp16 = const()[name = string("add_69_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219100928)))]; tensor mul_124_cast_fp16 = mul(x = mul_123_cast_fp16, y = add_69_to_fp16)[name = string("mul_124_cast_fp16")]; tensor p_st_0_model_layers_4_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219102528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219692416))))[name = string("p_st_0_model_layers_4_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_28_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_4_self_attn_q_proj_weight_to_fp16_quantized, x = mul_124_cast_fp16)[name = string("linear_28_cast_fp16")]; tensor const_787 = const()[name = string("const_787"), val = tensor([1, 512, -1, 256])]; tensor view_24_cast_fp16 = reshape(shape = const_787, x = linear_28_cast_fp16)[name = string("view_24_cast_fp16")]; tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_4_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219694016))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219890688))))[name = string("p_st_0_model_layers_4_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_29_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_4_self_attn_k_proj_weight_to_fp16_quantized, x = mul_124_cast_fp16)[name = string("linear_29_cast_fp16")]; tensor const_790 = const()[name = string("const_790"), val = tensor([1, 512, -1, 256])]; tensor view_25_cast_fp16 = reshape(shape = const_790, x = linear_29_cast_fp16)[name = string("view_25_cast_fp16")]; tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_4_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219891264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220087936))))[name = string("p_st_0_model_layers_4_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_30_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_4_self_attn_v_proj_weight_to_fp16_quantized, x = mul_124_cast_fp16)[name = string("linear_30_cast_fp16")]; tensor const_793 = const()[name = string("const_793"), val = tensor([1, 512, -1, 256])]; tensor view_26_cast_fp16 = reshape(shape = const_793, x = linear_30_cast_fp16)[name = string("view_26_cast_fp16")]; tensor transpose_46_perm_0 = const()[name = string("transpose_46_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_797_promoted_to_fp16 = const()[name = string("const_797_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = view_24_cast_fp16)[name = string("transpose_79")]; tensor pow_26_cast_fp16 = pow(x = transpose_44_cast_fp16, y = const_797_promoted_to_fp16)[name = string("pow_26_cast_fp16")]; tensor mean_25_axes_0 = const()[name = string("mean_25_axes_0"), val = tensor([-1])]; bool mean_25_keep_dims_0 = const()[name = string("mean_25_keep_dims_0"), val = bool(true)]; tensor mean_25_cast_fp16 = reduce_mean(axes = mean_25_axes_0, keep_dims = mean_25_keep_dims_0, x = pow_26_cast_fp16)[name = string("mean_25_cast_fp16")]; fp16 const_800_to_fp16 = const()[name = string("const_800_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_70_cast_fp16 = add(x = mean_25_cast_fp16, y = const_800_to_fp16)[name = string("add_70_cast_fp16")]; fp32 rsqrt_25_epsilon_0 = const()[name = string("rsqrt_25_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_25_cast_fp16 = rsqrt(epsilon = rsqrt_25_epsilon_0, x = add_70_cast_fp16)[name = string("rsqrt_25_cast_fp16")]; tensor mul_125_cast_fp16 = mul(x = transpose_44_cast_fp16, y = rsqrt_25_cast_fp16)[name = string("mul_125_cast_fp16")]; tensor add_71_to_fp16 = const()[name = string("add_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220088512)))]; tensor mul_126_cast_fp16 = mul(x = mul_125_cast_fp16, y = add_71_to_fp16)[name = string("mul_126_cast_fp16")]; fp16 const_805_promoted_to_fp16 = const()[name = string("const_805_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = view_25_cast_fp16)[name = string("transpose_78")]; tensor pow_27_cast_fp16 = pow(x = transpose_45_cast_fp16, y = const_805_promoted_to_fp16)[name = string("pow_27_cast_fp16")]; tensor mean_26_axes_0 = const()[name = string("mean_26_axes_0"), val = tensor([-1])]; bool mean_26_keep_dims_0 = const()[name = string("mean_26_keep_dims_0"), val = bool(true)]; tensor mean_26_cast_fp16 = reduce_mean(axes = mean_26_axes_0, keep_dims = mean_26_keep_dims_0, x = pow_27_cast_fp16)[name = string("mean_26_cast_fp16")]; fp16 const_808_to_fp16 = const()[name = string("const_808_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_72_cast_fp16 = add(x = mean_26_cast_fp16, y = const_808_to_fp16)[name = string("add_72_cast_fp16")]; fp32 rsqrt_26_epsilon_0 = const()[name = string("rsqrt_26_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_26_cast_fp16 = rsqrt(epsilon = rsqrt_26_epsilon_0, x = add_72_cast_fp16)[name = string("rsqrt_26_cast_fp16")]; tensor mul_127_cast_fp16 = mul(x = transpose_45_cast_fp16, y = rsqrt_26_cast_fp16)[name = string("mul_127_cast_fp16")]; tensor add_73_to_fp16 = const()[name = string("add_73_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220089088)))]; tensor mul_128_cast_fp16 = mul(x = mul_127_cast_fp16, y = add_73_to_fp16)[name = string("mul_128_cast_fp16")]; tensor mul_129_cast_fp16 = mul(x = mul_126_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_129_cast_fp16")]; tensor slice_169_begin_0 = const()[name = string("slice_169_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_169_end_0 = const()[name = string("slice_169_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_169_end_mask_0 = const()[name = string("slice_169_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_169_cast_fp16 = slice_by_index(begin = slice_169_begin_0, end = slice_169_end_0, end_mask = slice_169_end_mask_0, x = mul_126_cast_fp16)[name = string("slice_169_cast_fp16")]; tensor slice_170_begin_0 = const()[name = string("slice_170_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_170_end_0 = const()[name = string("slice_170_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_170_end_mask_0 = const()[name = string("slice_170_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_170_cast_fp16 = slice_by_index(begin = slice_170_begin_0, end = slice_170_end_0, end_mask = slice_170_end_mask_0, x = mul_126_cast_fp16)[name = string("slice_170_cast_fp16")]; fp16 const_820_promoted_to_fp16 = const()[name = string("const_820_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_8_cast_fp16 = mul(x = slice_170_cast_fp16, y = const_820_promoted_to_fp16)[name = string("neg_8_cast_fp16")]; int32 const_821 = const()[name = string("const_821"), val = int32(-1)]; bool cat_40_interleave_0 = const()[name = string("cat_40_interleave_0"), val = bool(false)]; tensor cat_40_cast_fp16 = concat(axis = const_821, interleave = cat_40_interleave_0, values = (neg_8_cast_fp16, slice_169_cast_fp16))[name = string("cat_40_cast_fp16")]; tensor mul_130_cast_fp16 = mul(x = cat_40_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_130_cast_fp16")]; tensor add_74_cast_fp16 = add(x = mul_129_cast_fp16, y = mul_130_cast_fp16)[name = string("add_74_cast_fp16")]; tensor mul_131_cast_fp16 = mul(x = mul_128_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_131_cast_fp16")]; tensor slice_171_begin_0 = const()[name = string("slice_171_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_171_end_0 = const()[name = string("slice_171_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_171_end_mask_0 = const()[name = string("slice_171_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_171_cast_fp16 = slice_by_index(begin = slice_171_begin_0, end = slice_171_end_0, end_mask = slice_171_end_mask_0, x = mul_128_cast_fp16)[name = string("slice_171_cast_fp16")]; tensor slice_172_begin_0 = const()[name = string("slice_172_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_172_end_0 = const()[name = string("slice_172_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_172_end_mask_0 = const()[name = string("slice_172_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_172_cast_fp16 = slice_by_index(begin = slice_172_begin_0, end = slice_172_end_0, end_mask = slice_172_end_mask_0, x = mul_128_cast_fp16)[name = string("slice_172_cast_fp16")]; fp16 const_828_promoted_to_fp16 = const()[name = string("const_828_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_9_cast_fp16 = mul(x = slice_172_cast_fp16, y = const_828_promoted_to_fp16)[name = string("neg_9_cast_fp16")]; int32 const_829 = const()[name = string("const_829"), val = int32(-1)]; bool cat_41_interleave_0 = const()[name = string("cat_41_interleave_0"), val = bool(false)]; tensor cat_41_cast_fp16 = concat(axis = const_829, interleave = cat_41_interleave_0, values = (neg_9_cast_fp16, slice_171_cast_fp16))[name = string("cat_41_cast_fp16")]; tensor mul_132_cast_fp16 = mul(x = cat_41_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_132_cast_fp16")]; tensor add_75_cast_fp16 = add(x = mul_131_cast_fp16, y = mul_132_cast_fp16)[name = string("add_75_cast_fp16")]; int32 const_830 = const()[name = string("const_830"), val = int32(-2)]; bool cat_42_interleave_0 = const()[name = string("cat_42_interleave_0"), val = bool(false)]; tensor cat_42_cast_fp16 = concat(axis = const_830, interleave = cat_42_interleave_0, values = add_75_cast_fp16)[name = string("cat_42_cast_fp16")]; int32 const_831 = const()[name = string("const_831"), val = int32(-2)]; bool cat_43_interleave_0 = const()[name = string("cat_43_interleave_0"), val = bool(false)]; tensor transpose_46_cast_fp16 = transpose(perm = transpose_46_perm_0, x = view_26_cast_fp16)[name = string("transpose_77")]; tensor cat_43_cast_fp16 = concat(axis = const_831, interleave = cat_43_interleave_0, values = transpose_46_cast_fp16)[name = string("cat_43_cast_fp16")]; tensor unsqueeze_95_axes_0 = const()[name = string("unsqueeze_95_axes_0"), val = tensor([2])]; tensor unsqueeze_95_cast_fp16 = expand_dims(axes = unsqueeze_95_axes_0, x = cat_42_cast_fp16)[name = string("unsqueeze_95_cast_fp16")]; tensor expand_34_reps_0 = const()[name = string("expand_34_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_34_cast_fp16 = tile(reps = expand_34_reps_0, x = unsqueeze_95_cast_fp16)[name = string("expand_34_cast_fp16")]; tensor const_846 = const()[name = string("const_846"), val = tensor([1, 3, 512, 256])]; tensor view_27_cast_fp16 = reshape(shape = const_846, x = expand_34_cast_fp16)[name = string("view_27_cast_fp16")]; tensor unsqueeze_96_axes_0 = const()[name = string("unsqueeze_96_axes_0"), val = tensor([2])]; tensor unsqueeze_96_cast_fp16 = expand_dims(axes = unsqueeze_96_axes_0, x = cat_43_cast_fp16)[name = string("unsqueeze_96_cast_fp16")]; tensor expand_35_reps_0 = const()[name = string("expand_35_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_35_cast_fp16 = tile(reps = expand_35_reps_0, x = unsqueeze_96_cast_fp16)[name = string("expand_35_cast_fp16")]; tensor const_861 = const()[name = string("const_861"), val = tensor([1, 3, 512, 256])]; tensor view_28_cast_fp16 = reshape(shape = const_861, x = expand_35_cast_fp16)[name = string("view_28_cast_fp16")]; bool matmul_32_transpose_x_1 = const()[name = string("matmul_32_transpose_x_1"), val = bool(false)]; bool matmul_32_transpose_y_1 = const()[name = string("matmul_32_transpose_y_1"), val = bool(true)]; tensor matmul_32_cast_fp16 = matmul(transpose_x = matmul_32_transpose_x_1, transpose_y = matmul_32_transpose_y_1, x = add_74_cast_fp16, y = view_27_cast_fp16)[name = string("matmul_32_cast_fp16")]; fp16 const_864_to_fp16 = const()[name = string("const_864_to_fp16"), val = fp16(0x1p-4)]; tensor mul_133_cast_fp16 = mul(x = matmul_32_cast_fp16, y = const_864_to_fp16)[name = string("mul_133_cast_fp16")]; tensor add_76_cast_fp16 = add(x = mul_133_cast_fp16, y = expand_cast_fp16)[name = string("add_76_cast_fp16")]; int32 const_874 = const()[name = string("const_874"), val = int32(-1)]; tensor softmax_4_cast_fp16 = softmax(axis = const_874, x = add_76_cast_fp16)[name = string("softmax_4_cast_fp16")]; bool matmul_33_transpose_x_0 = const()[name = string("matmul_33_transpose_x_0"), val = bool(false)]; bool matmul_33_transpose_y_0 = const()[name = string("matmul_33_transpose_y_0"), val = bool(false)]; tensor matmul_33_cast_fp16 = matmul(transpose_x = matmul_33_transpose_x_0, transpose_y = matmul_33_transpose_y_0, x = softmax_4_cast_fp16, y = view_28_cast_fp16)[name = string("matmul_33_cast_fp16")]; tensor transpose_48_perm_0 = const()[name = string("transpose_48_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_879 = const()[name = string("const_879"), val = tensor([1, 512, -1])]; tensor transpose_48_cast_fp16 = transpose(perm = transpose_48_perm_0, x = matmul_33_cast_fp16)[name = string("transpose_76")]; tensor view_29_cast_fp16 = reshape(shape = const_879, x = transpose_48_cast_fp16)[name = string("view_29_cast_fp16")]; tensor p_st_0_model_layers_4_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220089664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220679552))))[name = string("p_st_0_model_layers_4_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_31_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_4_self_attn_o_proj_weight_to_fp16_quantized, x = view_29_cast_fp16)[name = string("linear_31_cast_fp16")]; fp16 const_881_promoted_to_fp16 = const()[name = string("const_881_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_28_cast_fp16 = pow(x = linear_31_cast_fp16, y = const_881_promoted_to_fp16)[name = string("pow_28_cast_fp16")]; tensor mean_27_axes_0 = const()[name = string("mean_27_axes_0"), val = tensor([-1])]; bool mean_27_keep_dims_0 = const()[name = string("mean_27_keep_dims_0"), val = bool(true)]; tensor mean_27_cast_fp16 = reduce_mean(axes = mean_27_axes_0, keep_dims = mean_27_keep_dims_0, x = pow_28_cast_fp16)[name = string("mean_27_cast_fp16")]; fp16 const_884_to_fp16 = const()[name = string("const_884_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_77_cast_fp16 = add(x = mean_27_cast_fp16, y = const_884_to_fp16)[name = string("add_77_cast_fp16")]; fp32 rsqrt_27_epsilon_0 = const()[name = string("rsqrt_27_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_27_cast_fp16 = rsqrt(epsilon = rsqrt_27_epsilon_0, x = add_77_cast_fp16)[name = string("rsqrt_27_cast_fp16")]; tensor mul_134_cast_fp16 = mul(x = linear_31_cast_fp16, y = rsqrt_27_cast_fp16)[name = string("mul_134_cast_fp16")]; tensor add_78_to_fp16 = const()[name = string("add_78_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220681152)))]; tensor mul_135_cast_fp16 = mul(x = mul_134_cast_fp16, y = add_78_to_fp16)[name = string("mul_135_cast_fp16")]; tensor add_79_cast_fp16 = add(x = add_67_cast_fp16, y = mul_135_cast_fp16)[name = string("add_79_cast_fp16")]; fp16 const_889_promoted_to_fp16 = const()[name = string("const_889_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_29_cast_fp16 = pow(x = add_79_cast_fp16, y = const_889_promoted_to_fp16)[name = string("pow_29_cast_fp16")]; tensor mean_28_axes_0 = const()[name = string("mean_28_axes_0"), val = tensor([-1])]; bool mean_28_keep_dims_0 = const()[name = string("mean_28_keep_dims_0"), val = bool(true)]; tensor mean_28_cast_fp16 = reduce_mean(axes = mean_28_axes_0, keep_dims = mean_28_keep_dims_0, x = pow_29_cast_fp16)[name = string("mean_28_cast_fp16")]; fp16 const_892_to_fp16 = const()[name = string("const_892_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_80_cast_fp16 = add(x = mean_28_cast_fp16, y = const_892_to_fp16)[name = string("add_80_cast_fp16")]; fp32 rsqrt_28_epsilon_0 = const()[name = string("rsqrt_28_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_28_cast_fp16 = rsqrt(epsilon = rsqrt_28_epsilon_0, x = add_80_cast_fp16)[name = string("rsqrt_28_cast_fp16")]; tensor mul_136_cast_fp16 = mul(x = add_79_cast_fp16, y = rsqrt_28_cast_fp16)[name = string("mul_136_cast_fp16")]; tensor add_81_to_fp16 = const()[name = string("add_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220682752)))]; tensor mul_137_cast_fp16 = mul(x = mul_136_cast_fp16, y = add_81_to_fp16)[name = string("mul_137_cast_fp16")]; tensor p_st_0_model_layers_4_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220684352))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(221569152))))[name = string("p_st_0_model_layers_4_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_32_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_4_mlp_gate_proj_weight_to_fp16_quantized, x = mul_137_cast_fp16)[name = string("linear_32_cast_fp16")]; string gelu_4_mode_0 = const()[name = string("gelu_4_mode_0"), val = string("EXACT")]; tensor gelu_4_cast_fp16 = gelu(mode = gelu_4_mode_0, x = linear_32_cast_fp16)[name = string("gelu_4_cast_fp16")]; tensor p_st_0_model_layers_4_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(221571520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(222456320))))[name = string("p_st_0_model_layers_4_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_33_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_4_mlp_up_proj_weight_to_fp16_quantized, x = mul_137_cast_fp16)[name = string("linear_33_cast_fp16")]; tensor mul_138_cast_fp16 = mul(x = gelu_4_cast_fp16, y = linear_33_cast_fp16)[name = string("mul_138_cast_fp16")]; tensor p_st_0_model_layers_4_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(222458688))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223343488))))[name = string("p_st_0_model_layers_4_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_34_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_4_mlp_down_proj_weight_to_fp16_quantized, x = mul_138_cast_fp16)[name = string("linear_34_cast_fp16")]; fp16 const_897_promoted_to_fp16 = const()[name = string("const_897_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_30_cast_fp16 = pow(x = linear_34_cast_fp16, y = const_897_promoted_to_fp16)[name = string("pow_30_cast_fp16")]; tensor mean_29_axes_0 = const()[name = string("mean_29_axes_0"), val = tensor([-1])]; bool mean_29_keep_dims_0 = const()[name = string("mean_29_keep_dims_0"), val = bool(true)]; tensor mean_29_cast_fp16 = reduce_mean(axes = mean_29_axes_0, keep_dims = mean_29_keep_dims_0, x = pow_30_cast_fp16)[name = string("mean_29_cast_fp16")]; fp16 const_900_to_fp16 = const()[name = string("const_900_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_82_cast_fp16 = add(x = mean_29_cast_fp16, y = const_900_to_fp16)[name = string("add_82_cast_fp16")]; fp32 rsqrt_29_epsilon_0 = const()[name = string("rsqrt_29_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_29_cast_fp16 = rsqrt(epsilon = rsqrt_29_epsilon_0, x = add_82_cast_fp16)[name = string("rsqrt_29_cast_fp16")]; tensor mul_139_cast_fp16 = mul(x = linear_34_cast_fp16, y = rsqrt_29_cast_fp16)[name = string("mul_139_cast_fp16")]; tensor add_83_to_fp16 = const()[name = string("add_83_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223345088)))]; tensor mul_140_cast_fp16 = mul(x = mul_139_cast_fp16, y = add_83_to_fp16)[name = string("mul_140_cast_fp16")]; tensor add_84_cast_fp16 = add(x = add_79_cast_fp16, y = mul_140_cast_fp16)[name = string("add_84_cast_fp16")]; fp16 const_905_promoted_to_fp16 = const()[name = string("const_905_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_31_cast_fp16 = pow(x = add_84_cast_fp16, y = const_905_promoted_to_fp16)[name = string("pow_31_cast_fp16")]; tensor mean_30_axes_0 = const()[name = string("mean_30_axes_0"), val = tensor([-1])]; bool mean_30_keep_dims_0 = const()[name = string("mean_30_keep_dims_0"), val = bool(true)]; tensor mean_30_cast_fp16 = reduce_mean(axes = mean_30_axes_0, keep_dims = mean_30_keep_dims_0, x = pow_31_cast_fp16)[name = string("mean_30_cast_fp16")]; fp16 const_908_to_fp16 = const()[name = string("const_908_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_85_cast_fp16 = add(x = mean_30_cast_fp16, y = const_908_to_fp16)[name = string("add_85_cast_fp16")]; fp32 rsqrt_30_epsilon_0 = const()[name = string("rsqrt_30_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_30_cast_fp16 = rsqrt(epsilon = rsqrt_30_epsilon_0, x = add_85_cast_fp16)[name = string("rsqrt_30_cast_fp16")]; tensor mul_141_cast_fp16 = mul(x = add_84_cast_fp16, y = rsqrt_30_cast_fp16)[name = string("mul_141_cast_fp16")]; tensor add_86_to_fp16 = const()[name = string("add_86_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223346688)))]; tensor mul_142_cast_fp16 = mul(x = mul_141_cast_fp16, y = add_86_to_fp16)[name = string("mul_142_cast_fp16")]; tensor p_st_0_model_layers_5_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223348288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223938176))))[name = string("p_st_0_model_layers_5_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_35_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_5_self_attn_q_proj_weight_to_fp16_quantized, x = mul_142_cast_fp16)[name = string("linear_35_cast_fp16")]; tensor const_912 = const()[name = string("const_912"), val = tensor([1, 512, -1, 256])]; tensor view_30_cast_fp16 = reshape(shape = const_912, x = linear_35_cast_fp16)[name = string("view_30_cast_fp16")]; tensor transpose_49_perm_0 = const()[name = string("transpose_49_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_5_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223939776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224136448))))[name = string("p_st_0_model_layers_5_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_36_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_5_self_attn_k_proj_weight_to_fp16_quantized, x = mul_142_cast_fp16)[name = string("linear_36_cast_fp16")]; tensor const_915 = const()[name = string("const_915"), val = tensor([1, 512, -1, 256])]; tensor view_31_cast_fp16 = reshape(shape = const_915, x = linear_36_cast_fp16)[name = string("view_31_cast_fp16")]; tensor transpose_50_perm_0 = const()[name = string("transpose_50_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_5_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224137024))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224333696))))[name = string("p_st_0_model_layers_5_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_37_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_5_self_attn_v_proj_weight_to_fp16_quantized, x = mul_142_cast_fp16)[name = string("linear_37_cast_fp16")]; tensor const_918 = const()[name = string("const_918"), val = tensor([1, 512, -1, 256])]; tensor view_32_cast_fp16 = reshape(shape = const_918, x = linear_37_cast_fp16)[name = string("view_32_cast_fp16")]; tensor transpose_51_perm_0 = const()[name = string("transpose_51_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_922_promoted_to_fp16 = const()[name = string("const_922_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_49_cast_fp16 = transpose(perm = transpose_49_perm_0, x = view_30_cast_fp16)[name = string("transpose_75")]; tensor pow_32_cast_fp16 = pow(x = transpose_49_cast_fp16, y = const_922_promoted_to_fp16)[name = string("pow_32_cast_fp16")]; tensor mean_31_axes_0 = const()[name = string("mean_31_axes_0"), val = tensor([-1])]; bool mean_31_keep_dims_0 = const()[name = string("mean_31_keep_dims_0"), val = bool(true)]; tensor mean_31_cast_fp16 = reduce_mean(axes = mean_31_axes_0, keep_dims = mean_31_keep_dims_0, x = pow_32_cast_fp16)[name = string("mean_31_cast_fp16")]; fp16 const_925_to_fp16 = const()[name = string("const_925_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_87_cast_fp16 = add(x = mean_31_cast_fp16, y = const_925_to_fp16)[name = string("add_87_cast_fp16")]; fp32 rsqrt_31_epsilon_0 = const()[name = string("rsqrt_31_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_31_cast_fp16 = rsqrt(epsilon = rsqrt_31_epsilon_0, x = add_87_cast_fp16)[name = string("rsqrt_31_cast_fp16")]; tensor mul_143_cast_fp16 = mul(x = transpose_49_cast_fp16, y = rsqrt_31_cast_fp16)[name = string("mul_143_cast_fp16")]; tensor add_88_to_fp16 = const()[name = string("add_88_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224334272)))]; tensor mul_144_cast_fp16 = mul(x = mul_143_cast_fp16, y = add_88_to_fp16)[name = string("mul_144_cast_fp16")]; fp16 const_930_promoted_to_fp16 = const()[name = string("const_930_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_50_cast_fp16 = transpose(perm = transpose_50_perm_0, x = view_31_cast_fp16)[name = string("transpose_74")]; tensor pow_33_cast_fp16 = pow(x = transpose_50_cast_fp16, y = const_930_promoted_to_fp16)[name = string("pow_33_cast_fp16")]; tensor mean_32_axes_0 = const()[name = string("mean_32_axes_0"), val = tensor([-1])]; bool mean_32_keep_dims_0 = const()[name = string("mean_32_keep_dims_0"), val = bool(true)]; tensor mean_32_cast_fp16 = reduce_mean(axes = mean_32_axes_0, keep_dims = mean_32_keep_dims_0, x = pow_33_cast_fp16)[name = string("mean_32_cast_fp16")]; fp16 const_933_to_fp16 = const()[name = string("const_933_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_89_cast_fp16 = add(x = mean_32_cast_fp16, y = const_933_to_fp16)[name = string("add_89_cast_fp16")]; fp32 rsqrt_32_epsilon_0 = const()[name = string("rsqrt_32_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_32_cast_fp16 = rsqrt(epsilon = rsqrt_32_epsilon_0, x = add_89_cast_fp16)[name = string("rsqrt_32_cast_fp16")]; tensor mul_145_cast_fp16 = mul(x = transpose_50_cast_fp16, y = rsqrt_32_cast_fp16)[name = string("mul_145_cast_fp16")]; tensor add_90_to_fp16 = const()[name = string("add_90_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224334848)))]; tensor mul_146_cast_fp16 = mul(x = mul_145_cast_fp16, y = add_90_to_fp16)[name = string("mul_146_cast_fp16")]; tensor unsqueeze_97_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224335424))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("unsqueeze_97_to_fp16_quantized")]; tensor mul_147_cast_fp16 = mul(x = mul_144_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_147_cast_fp16")]; tensor slice_192_begin_0 = const()[name = string("slice_192_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_192_end_0 = const()[name = string("slice_192_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_192_end_mask_0 = const()[name = string("slice_192_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_192_cast_fp16 = slice_by_index(begin = slice_192_begin_0, end = slice_192_end_0, end_mask = slice_192_end_mask_0, x = mul_144_cast_fp16)[name = string("slice_192_cast_fp16")]; tensor slice_193_begin_0 = const()[name = string("slice_193_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_193_end_0 = const()[name = string("slice_193_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_193_end_mask_0 = const()[name = string("slice_193_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_193_cast_fp16 = slice_by_index(begin = slice_193_begin_0, end = slice_193_end_0, end_mask = slice_193_end_mask_0, x = mul_144_cast_fp16)[name = string("slice_193_cast_fp16")]; fp16 const_945_promoted_to_fp16 = const()[name = string("const_945_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_10_cast_fp16 = mul(x = slice_193_cast_fp16, y = const_945_promoted_to_fp16)[name = string("neg_10_cast_fp16")]; int32 const_946 = const()[name = string("const_946"), val = int32(-1)]; bool cat_44_interleave_0 = const()[name = string("cat_44_interleave_0"), val = bool(false)]; tensor cat_44_cast_fp16 = concat(axis = const_946, interleave = cat_44_interleave_0, values = (neg_10_cast_fp16, slice_192_cast_fp16))[name = string("cat_44_cast_fp16")]; tensor unsqueeze_98_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224466560))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("unsqueeze_98_to_fp16_quantized")]; tensor mul_148_cast_fp16 = mul(x = cat_44_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_148_cast_fp16")]; tensor add_91_cast_fp16 = add(x = mul_147_cast_fp16, y = mul_148_cast_fp16)[name = string("add_91_cast_fp16")]; tensor mul_149_cast_fp16 = mul(x = mul_146_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_149_cast_fp16")]; tensor slice_194_begin_0 = const()[name = string("slice_194_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_194_end_0 = const()[name = string("slice_194_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_194_end_mask_0 = const()[name = string("slice_194_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_194_cast_fp16 = slice_by_index(begin = slice_194_begin_0, end = slice_194_end_0, end_mask = slice_194_end_mask_0, x = mul_146_cast_fp16)[name = string("slice_194_cast_fp16")]; tensor slice_195_begin_0 = const()[name = string("slice_195_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_195_end_0 = const()[name = string("slice_195_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_195_end_mask_0 = const()[name = string("slice_195_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_195_cast_fp16 = slice_by_index(begin = slice_195_begin_0, end = slice_195_end_0, end_mask = slice_195_end_mask_0, x = mul_146_cast_fp16)[name = string("slice_195_cast_fp16")]; fp16 const_953_promoted_to_fp16 = const()[name = string("const_953_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_11_cast_fp16 = mul(x = slice_195_cast_fp16, y = const_953_promoted_to_fp16)[name = string("neg_11_cast_fp16")]; int32 const_954 = const()[name = string("const_954"), val = int32(-1)]; bool cat_45_interleave_0 = const()[name = string("cat_45_interleave_0"), val = bool(false)]; tensor cat_45_cast_fp16 = concat(axis = const_954, interleave = cat_45_interleave_0, values = (neg_11_cast_fp16, slice_194_cast_fp16))[name = string("cat_45_cast_fp16")]; tensor mul_150_cast_fp16 = mul(x = cat_45_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_150_cast_fp16")]; tensor add_92_cast_fp16 = add(x = mul_149_cast_fp16, y = mul_150_cast_fp16)[name = string("add_92_cast_fp16")]; int32 const_955 = const()[name = string("const_955"), val = int32(-2)]; bool cat_46_interleave_0 = const()[name = string("cat_46_interleave_0"), val = bool(false)]; tensor cat_46_cast_fp16 = concat(axis = const_955, interleave = cat_46_interleave_0, values = add_92_cast_fp16)[name = string("cat_46_cast_fp16")]; int32 const_956 = const()[name = string("const_956"), val = int32(-2)]; bool cat_47_interleave_0 = const()[name = string("cat_47_interleave_0"), val = bool(false)]; tensor transpose_51_cast_fp16 = transpose(perm = transpose_51_perm_0, x = view_32_cast_fp16)[name = string("transpose_73")]; tensor cat_47_cast_fp16 = concat(axis = const_956, interleave = cat_47_interleave_0, values = transpose_51_cast_fp16)[name = string("cat_47_cast_fp16")]; tensor unsqueeze_99_axes_0 = const()[name = string("unsqueeze_99_axes_0"), val = tensor([2])]; tensor unsqueeze_99_cast_fp16 = expand_dims(axes = unsqueeze_99_axes_0, x = cat_46_cast_fp16)[name = string("unsqueeze_99_cast_fp16")]; tensor expand_36_reps_0 = const()[name = string("expand_36_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_36_cast_fp16 = tile(reps = expand_36_reps_0, x = unsqueeze_99_cast_fp16)[name = string("expand_36_cast_fp16")]; tensor const_971 = const()[name = string("const_971"), val = tensor([1, 3, 512, 256])]; tensor view_33_cast_fp16 = reshape(shape = const_971, x = expand_36_cast_fp16)[name = string("view_33_cast_fp16")]; tensor unsqueeze_100_axes_0 = const()[name = string("unsqueeze_100_axes_0"), val = tensor([2])]; tensor unsqueeze_100_cast_fp16 = expand_dims(axes = unsqueeze_100_axes_0, x = cat_47_cast_fp16)[name = string("unsqueeze_100_cast_fp16")]; tensor expand_37_reps_0 = const()[name = string("expand_37_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_37_cast_fp16 = tile(reps = expand_37_reps_0, x = unsqueeze_100_cast_fp16)[name = string("expand_37_cast_fp16")]; tensor const_986 = const()[name = string("const_986"), val = tensor([1, 3, 512, 256])]; tensor view_34_cast_fp16 = reshape(shape = const_986, x = expand_37_cast_fp16)[name = string("view_34_cast_fp16")]; bool matmul_34_transpose_x_1 = const()[name = string("matmul_34_transpose_x_1"), val = bool(false)]; bool matmul_34_transpose_y_1 = const()[name = string("matmul_34_transpose_y_1"), val = bool(true)]; tensor matmul_34_cast_fp16 = matmul(transpose_x = matmul_34_transpose_x_1, transpose_y = matmul_34_transpose_y_1, x = add_91_cast_fp16, y = view_33_cast_fp16)[name = string("matmul_34_cast_fp16")]; fp16 const_989_to_fp16 = const()[name = string("const_989_to_fp16"), val = fp16(0x1p-4)]; tensor mul_151_cast_fp16 = mul(x = matmul_34_cast_fp16, y = const_989_to_fp16)[name = string("mul_151_cast_fp16")]; tensor add_93_cast_fp16 = add(x = mul_151_cast_fp16, y = expand_cast_fp16)[name = string("add_93_cast_fp16")]; int32 const_999 = const()[name = string("const_999"), val = int32(-1)]; tensor softmax_5_cast_fp16 = softmax(axis = const_999, x = add_93_cast_fp16)[name = string("softmax_5_cast_fp16")]; bool matmul_35_transpose_x_0 = const()[name = string("matmul_35_transpose_x_0"), val = bool(false)]; bool matmul_35_transpose_y_0 = const()[name = string("matmul_35_transpose_y_0"), val = bool(false)]; tensor matmul_35_cast_fp16 = matmul(transpose_x = matmul_35_transpose_x_0, transpose_y = matmul_35_transpose_y_0, x = softmax_5_cast_fp16, y = view_34_cast_fp16)[name = string("matmul_35_cast_fp16")]; tensor transpose_53_perm_0 = const()[name = string("transpose_53_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1004 = const()[name = string("const_1004"), val = tensor([1, 512, -1])]; tensor transpose_53_cast_fp16 = transpose(perm = transpose_53_perm_0, x = matmul_35_cast_fp16)[name = string("transpose_72")]; tensor view_35_cast_fp16 = reshape(shape = const_1004, x = transpose_53_cast_fp16)[name = string("view_35_cast_fp16")]; tensor p_st_0_model_layers_5_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224597696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225187584))))[name = string("p_st_0_model_layers_5_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_38_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_5_self_attn_o_proj_weight_to_fp16_quantized, x = view_35_cast_fp16)[name = string("linear_38_cast_fp16")]; fp16 const_1006_promoted_to_fp16 = const()[name = string("const_1006_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_34_cast_fp16 = pow(x = linear_38_cast_fp16, y = const_1006_promoted_to_fp16)[name = string("pow_34_cast_fp16")]; tensor mean_33_axes_0 = const()[name = string("mean_33_axes_0"), val = tensor([-1])]; bool mean_33_keep_dims_0 = const()[name = string("mean_33_keep_dims_0"), val = bool(true)]; tensor mean_33_cast_fp16 = reduce_mean(axes = mean_33_axes_0, keep_dims = mean_33_keep_dims_0, x = pow_34_cast_fp16)[name = string("mean_33_cast_fp16")]; fp16 const_1009_to_fp16 = const()[name = string("const_1009_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_94_cast_fp16 = add(x = mean_33_cast_fp16, y = const_1009_to_fp16)[name = string("add_94_cast_fp16")]; fp32 rsqrt_33_epsilon_0 = const()[name = string("rsqrt_33_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_33_cast_fp16 = rsqrt(epsilon = rsqrt_33_epsilon_0, x = add_94_cast_fp16)[name = string("rsqrt_33_cast_fp16")]; tensor mul_152_cast_fp16 = mul(x = linear_38_cast_fp16, y = rsqrt_33_cast_fp16)[name = string("mul_152_cast_fp16")]; tensor add_95_to_fp16 = const()[name = string("add_95_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225189184)))]; tensor mul_153_cast_fp16 = mul(x = mul_152_cast_fp16, y = add_95_to_fp16)[name = string("mul_153_cast_fp16")]; tensor add_96_cast_fp16 = add(x = add_84_cast_fp16, y = mul_153_cast_fp16)[name = string("add_96_cast_fp16")]; fp16 const_1014_promoted_to_fp16 = const()[name = string("const_1014_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_35_cast_fp16 = pow(x = add_96_cast_fp16, y = const_1014_promoted_to_fp16)[name = string("pow_35_cast_fp16")]; tensor mean_34_axes_0 = const()[name = string("mean_34_axes_0"), val = tensor([-1])]; bool mean_34_keep_dims_0 = const()[name = string("mean_34_keep_dims_0"), val = bool(true)]; tensor mean_34_cast_fp16 = reduce_mean(axes = mean_34_axes_0, keep_dims = mean_34_keep_dims_0, x = pow_35_cast_fp16)[name = string("mean_34_cast_fp16")]; fp16 const_1017_to_fp16 = const()[name = string("const_1017_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_97_cast_fp16 = add(x = mean_34_cast_fp16, y = const_1017_to_fp16)[name = string("add_97_cast_fp16")]; fp32 rsqrt_34_epsilon_0 = const()[name = string("rsqrt_34_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_34_cast_fp16 = rsqrt(epsilon = rsqrt_34_epsilon_0, x = add_97_cast_fp16)[name = string("rsqrt_34_cast_fp16")]; tensor mul_154_cast_fp16 = mul(x = add_96_cast_fp16, y = rsqrt_34_cast_fp16)[name = string("mul_154_cast_fp16")]; tensor add_98_to_fp16 = const()[name = string("add_98_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225190784)))]; tensor mul_155_cast_fp16 = mul(x = mul_154_cast_fp16, y = add_98_to_fp16)[name = string("mul_155_cast_fp16")]; tensor p_st_0_model_layers_5_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225192384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226077184))))[name = string("p_st_0_model_layers_5_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_39_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_5_mlp_gate_proj_weight_to_fp16_quantized, x = mul_155_cast_fp16)[name = string("linear_39_cast_fp16")]; string gelu_5_mode_0 = const()[name = string("gelu_5_mode_0"), val = string("EXACT")]; tensor gelu_5_cast_fp16 = gelu(mode = gelu_5_mode_0, x = linear_39_cast_fp16)[name = string("gelu_5_cast_fp16")]; tensor p_st_0_model_layers_5_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226079552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226964352))))[name = string("p_st_0_model_layers_5_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_40_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_5_mlp_up_proj_weight_to_fp16_quantized, x = mul_155_cast_fp16)[name = string("linear_40_cast_fp16")]; tensor mul_156_cast_fp16 = mul(x = gelu_5_cast_fp16, y = linear_40_cast_fp16)[name = string("mul_156_cast_fp16")]; tensor p_st_0_model_layers_5_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226966720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227851520))))[name = string("p_st_0_model_layers_5_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_41_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_5_mlp_down_proj_weight_to_fp16_quantized, x = mul_156_cast_fp16)[name = string("linear_41_cast_fp16")]; fp16 const_1022_promoted_to_fp16 = const()[name = string("const_1022_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_36_cast_fp16 = pow(x = linear_41_cast_fp16, y = const_1022_promoted_to_fp16)[name = string("pow_36_cast_fp16")]; tensor mean_35_axes_0 = const()[name = string("mean_35_axes_0"), val = tensor([-1])]; bool mean_35_keep_dims_0 = const()[name = string("mean_35_keep_dims_0"), val = bool(true)]; tensor mean_35_cast_fp16 = reduce_mean(axes = mean_35_axes_0, keep_dims = mean_35_keep_dims_0, x = pow_36_cast_fp16)[name = string("mean_35_cast_fp16")]; fp16 const_1025_to_fp16 = const()[name = string("const_1025_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_99_cast_fp16 = add(x = mean_35_cast_fp16, y = const_1025_to_fp16)[name = string("add_99_cast_fp16")]; fp32 rsqrt_35_epsilon_0 = const()[name = string("rsqrt_35_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_35_cast_fp16 = rsqrt(epsilon = rsqrt_35_epsilon_0, x = add_99_cast_fp16)[name = string("rsqrt_35_cast_fp16")]; tensor mul_157_cast_fp16 = mul(x = linear_41_cast_fp16, y = rsqrt_35_cast_fp16)[name = string("mul_157_cast_fp16")]; tensor add_100_to_fp16 = const()[name = string("add_100_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227853120)))]; tensor mul_158_cast_fp16 = mul(x = mul_157_cast_fp16, y = add_100_to_fp16)[name = string("mul_158_cast_fp16")]; tensor add_101_cast_fp16 = add(x = add_96_cast_fp16, y = mul_158_cast_fp16)[name = string("add_101_cast_fp16")]; fp16 const_1030_promoted_to_fp16 = const()[name = string("const_1030_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_37_cast_fp16 = pow(x = add_101_cast_fp16, y = const_1030_promoted_to_fp16)[name = string("pow_37_cast_fp16")]; tensor mean_36_axes_0 = const()[name = string("mean_36_axes_0"), val = tensor([-1])]; bool mean_36_keep_dims_0 = const()[name = string("mean_36_keep_dims_0"), val = bool(true)]; tensor mean_36_cast_fp16 = reduce_mean(axes = mean_36_axes_0, keep_dims = mean_36_keep_dims_0, x = pow_37_cast_fp16)[name = string("mean_36_cast_fp16")]; fp16 const_1033_to_fp16 = const()[name = string("const_1033_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_102_cast_fp16 = add(x = mean_36_cast_fp16, y = const_1033_to_fp16)[name = string("add_102_cast_fp16")]; fp32 rsqrt_36_epsilon_0 = const()[name = string("rsqrt_36_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_36_cast_fp16 = rsqrt(epsilon = rsqrt_36_epsilon_0, x = add_102_cast_fp16)[name = string("rsqrt_36_cast_fp16")]; tensor mul_159_cast_fp16 = mul(x = add_101_cast_fp16, y = rsqrt_36_cast_fp16)[name = string("mul_159_cast_fp16")]; tensor add_103_to_fp16 = const()[name = string("add_103_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227854720)))]; tensor mul_160_cast_fp16 = mul(x = mul_159_cast_fp16, y = add_103_to_fp16)[name = string("mul_160_cast_fp16")]; tensor p_st_0_model_layers_6_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227856320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228446208))))[name = string("p_st_0_model_layers_6_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_42_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_6_self_attn_q_proj_weight_to_fp16_quantized, x = mul_160_cast_fp16)[name = string("linear_42_cast_fp16")]; tensor const_1037 = const()[name = string("const_1037"), val = tensor([1, 512, -1, 256])]; tensor view_36_cast_fp16 = reshape(shape = const_1037, x = linear_42_cast_fp16)[name = string("view_36_cast_fp16")]; tensor transpose_54_perm_0 = const()[name = string("transpose_54_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_6_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228447808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228644480))))[name = string("p_st_0_model_layers_6_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_43_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_6_self_attn_k_proj_weight_to_fp16_quantized, x = mul_160_cast_fp16)[name = string("linear_43_cast_fp16")]; tensor const_1040 = const()[name = string("const_1040"), val = tensor([1, 512, -1, 256])]; tensor view_37_cast_fp16 = reshape(shape = const_1040, x = linear_43_cast_fp16)[name = string("view_37_cast_fp16")]; tensor transpose_55_perm_0 = const()[name = string("transpose_55_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_6_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228645056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228841728))))[name = string("p_st_0_model_layers_6_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_44_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_6_self_attn_v_proj_weight_to_fp16_quantized, x = mul_160_cast_fp16)[name = string("linear_44_cast_fp16")]; tensor const_1043 = const()[name = string("const_1043"), val = tensor([1, 512, -1, 256])]; tensor view_38_cast_fp16 = reshape(shape = const_1043, x = linear_44_cast_fp16)[name = string("view_38_cast_fp16")]; tensor transpose_56_perm_0 = const()[name = string("transpose_56_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1047_promoted_to_fp16 = const()[name = string("const_1047_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_54_cast_fp16 = transpose(perm = transpose_54_perm_0, x = view_36_cast_fp16)[name = string("transpose_71")]; tensor pow_38_cast_fp16 = pow(x = transpose_54_cast_fp16, y = const_1047_promoted_to_fp16)[name = string("pow_38_cast_fp16")]; tensor mean_37_axes_0 = const()[name = string("mean_37_axes_0"), val = tensor([-1])]; bool mean_37_keep_dims_0 = const()[name = string("mean_37_keep_dims_0"), val = bool(true)]; tensor mean_37_cast_fp16 = reduce_mean(axes = mean_37_axes_0, keep_dims = mean_37_keep_dims_0, x = pow_38_cast_fp16)[name = string("mean_37_cast_fp16")]; fp16 const_1050_to_fp16 = const()[name = string("const_1050_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_104_cast_fp16 = add(x = mean_37_cast_fp16, y = const_1050_to_fp16)[name = string("add_104_cast_fp16")]; fp32 rsqrt_37_epsilon_0 = const()[name = string("rsqrt_37_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_37_cast_fp16 = rsqrt(epsilon = rsqrt_37_epsilon_0, x = add_104_cast_fp16)[name = string("rsqrt_37_cast_fp16")]; tensor mul_161_cast_fp16 = mul(x = transpose_54_cast_fp16, y = rsqrt_37_cast_fp16)[name = string("mul_161_cast_fp16")]; tensor add_105_to_fp16 = const()[name = string("add_105_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228842304)))]; tensor mul_162_cast_fp16 = mul(x = mul_161_cast_fp16, y = add_105_to_fp16)[name = string("mul_162_cast_fp16")]; fp16 const_1055_promoted_to_fp16 = const()[name = string("const_1055_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_55_cast_fp16 = transpose(perm = transpose_55_perm_0, x = view_37_cast_fp16)[name = string("transpose_70")]; tensor pow_39_cast_fp16 = pow(x = transpose_55_cast_fp16, y = const_1055_promoted_to_fp16)[name = string("pow_39_cast_fp16")]; tensor mean_38_axes_0 = const()[name = string("mean_38_axes_0"), val = tensor([-1])]; bool mean_38_keep_dims_0 = const()[name = string("mean_38_keep_dims_0"), val = bool(true)]; tensor mean_38_cast_fp16 = reduce_mean(axes = mean_38_axes_0, keep_dims = mean_38_keep_dims_0, x = pow_39_cast_fp16)[name = string("mean_38_cast_fp16")]; fp16 const_1058_to_fp16 = const()[name = string("const_1058_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_106_cast_fp16 = add(x = mean_38_cast_fp16, y = const_1058_to_fp16)[name = string("add_106_cast_fp16")]; fp32 rsqrt_38_epsilon_0 = const()[name = string("rsqrt_38_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_38_cast_fp16 = rsqrt(epsilon = rsqrt_38_epsilon_0, x = add_106_cast_fp16)[name = string("rsqrt_38_cast_fp16")]; tensor mul_163_cast_fp16 = mul(x = transpose_55_cast_fp16, y = rsqrt_38_cast_fp16)[name = string("mul_163_cast_fp16")]; tensor add_107_to_fp16 = const()[name = string("add_107_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228842880)))]; tensor mul_164_cast_fp16 = mul(x = mul_163_cast_fp16, y = add_107_to_fp16)[name = string("mul_164_cast_fp16")]; tensor mul_165_cast_fp16 = mul(x = mul_162_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_165_cast_fp16")]; tensor slice_207_begin_0 = const()[name = string("slice_207_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_207_end_0 = const()[name = string("slice_207_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_207_end_mask_0 = const()[name = string("slice_207_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_207_cast_fp16 = slice_by_index(begin = slice_207_begin_0, end = slice_207_end_0, end_mask = slice_207_end_mask_0, x = mul_162_cast_fp16)[name = string("slice_207_cast_fp16")]; tensor slice_208_begin_0 = const()[name = string("slice_208_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_208_end_0 = const()[name = string("slice_208_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_208_end_mask_0 = const()[name = string("slice_208_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_208_cast_fp16 = slice_by_index(begin = slice_208_begin_0, end = slice_208_end_0, end_mask = slice_208_end_mask_0, x = mul_162_cast_fp16)[name = string("slice_208_cast_fp16")]; fp16 const_1070_promoted_to_fp16 = const()[name = string("const_1070_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_12_cast_fp16 = mul(x = slice_208_cast_fp16, y = const_1070_promoted_to_fp16)[name = string("neg_12_cast_fp16")]; int32 const_1071 = const()[name = string("const_1071"), val = int32(-1)]; bool cat_48_interleave_0 = const()[name = string("cat_48_interleave_0"), val = bool(false)]; tensor cat_48_cast_fp16 = concat(axis = const_1071, interleave = cat_48_interleave_0, values = (neg_12_cast_fp16, slice_207_cast_fp16))[name = string("cat_48_cast_fp16")]; tensor mul_166_cast_fp16 = mul(x = cat_48_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_166_cast_fp16")]; tensor add_108_cast_fp16 = add(x = mul_165_cast_fp16, y = mul_166_cast_fp16)[name = string("add_108_cast_fp16")]; tensor mul_167_cast_fp16 = mul(x = mul_164_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_167_cast_fp16")]; tensor slice_209_begin_0 = const()[name = string("slice_209_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_209_end_0 = const()[name = string("slice_209_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_209_end_mask_0 = const()[name = string("slice_209_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_209_cast_fp16 = slice_by_index(begin = slice_209_begin_0, end = slice_209_end_0, end_mask = slice_209_end_mask_0, x = mul_164_cast_fp16)[name = string("slice_209_cast_fp16")]; tensor slice_210_begin_0 = const()[name = string("slice_210_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_210_end_0 = const()[name = string("slice_210_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_210_end_mask_0 = const()[name = string("slice_210_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_210_cast_fp16 = slice_by_index(begin = slice_210_begin_0, end = slice_210_end_0, end_mask = slice_210_end_mask_0, x = mul_164_cast_fp16)[name = string("slice_210_cast_fp16")]; fp16 const_1078_promoted_to_fp16 = const()[name = string("const_1078_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_13_cast_fp16 = mul(x = slice_210_cast_fp16, y = const_1078_promoted_to_fp16)[name = string("neg_13_cast_fp16")]; int32 const_1079 = const()[name = string("const_1079"), val = int32(-1)]; bool cat_49_interleave_0 = const()[name = string("cat_49_interleave_0"), val = bool(false)]; tensor cat_49_cast_fp16 = concat(axis = const_1079, interleave = cat_49_interleave_0, values = (neg_13_cast_fp16, slice_209_cast_fp16))[name = string("cat_49_cast_fp16")]; tensor mul_168_cast_fp16 = mul(x = cat_49_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_168_cast_fp16")]; tensor add_109_cast_fp16 = add(x = mul_167_cast_fp16, y = mul_168_cast_fp16)[name = string("add_109_cast_fp16")]; int32 const_1080 = const()[name = string("const_1080"), val = int32(-2)]; bool cat_50_interleave_0 = const()[name = string("cat_50_interleave_0"), val = bool(false)]; tensor cat_50_cast_fp16 = concat(axis = const_1080, interleave = cat_50_interleave_0, values = add_109_cast_fp16)[name = string("cat_50_cast_fp16")]; int32 const_1081 = const()[name = string("const_1081"), val = int32(-2)]; bool cat_51_interleave_0 = const()[name = string("cat_51_interleave_0"), val = bool(false)]; tensor transpose_56_cast_fp16 = transpose(perm = transpose_56_perm_0, x = view_38_cast_fp16)[name = string("transpose_69")]; tensor cat_51_cast_fp16 = concat(axis = const_1081, interleave = cat_51_interleave_0, values = transpose_56_cast_fp16)[name = string("cat_51_cast_fp16")]; tensor unsqueeze_103_axes_0 = const()[name = string("unsqueeze_103_axes_0"), val = tensor([2])]; tensor unsqueeze_103_cast_fp16 = expand_dims(axes = unsqueeze_103_axes_0, x = cat_50_cast_fp16)[name = string("unsqueeze_103_cast_fp16")]; tensor expand_38_reps_0 = const()[name = string("expand_38_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_38_cast_fp16 = tile(reps = expand_38_reps_0, x = unsqueeze_103_cast_fp16)[name = string("expand_38_cast_fp16")]; tensor const_1096 = const()[name = string("const_1096"), val = tensor([1, 3, 512, 256])]; tensor view_39_cast_fp16 = reshape(shape = const_1096, x = expand_38_cast_fp16)[name = string("view_39_cast_fp16")]; tensor unsqueeze_104_axes_0 = const()[name = string("unsqueeze_104_axes_0"), val = tensor([2])]; tensor unsqueeze_104_cast_fp16 = expand_dims(axes = unsqueeze_104_axes_0, x = cat_51_cast_fp16)[name = string("unsqueeze_104_cast_fp16")]; tensor expand_39_reps_0 = const()[name = string("expand_39_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_39_cast_fp16 = tile(reps = expand_39_reps_0, x = unsqueeze_104_cast_fp16)[name = string("expand_39_cast_fp16")]; tensor const_1111 = const()[name = string("const_1111"), val = tensor([1, 3, 512, 256])]; tensor view_40_cast_fp16 = reshape(shape = const_1111, x = expand_39_cast_fp16)[name = string("view_40_cast_fp16")]; bool matmul_36_transpose_x_1 = const()[name = string("matmul_36_transpose_x_1"), val = bool(false)]; bool matmul_36_transpose_y_1 = const()[name = string("matmul_36_transpose_y_1"), val = bool(true)]; tensor matmul_36_cast_fp16 = matmul(transpose_x = matmul_36_transpose_x_1, transpose_y = matmul_36_transpose_y_1, x = add_108_cast_fp16, y = view_39_cast_fp16)[name = string("matmul_36_cast_fp16")]; fp16 const_1114_to_fp16 = const()[name = string("const_1114_to_fp16"), val = fp16(0x1p-4)]; tensor mul_169_cast_fp16 = mul(x = matmul_36_cast_fp16, y = const_1114_to_fp16)[name = string("mul_169_cast_fp16")]; tensor add_110_cast_fp16 = add(x = mul_169_cast_fp16, y = expand_cast_fp16)[name = string("add_110_cast_fp16")]; int32 const_1124 = const()[name = string("const_1124"), val = int32(-1)]; tensor softmax_6_cast_fp16 = softmax(axis = const_1124, x = add_110_cast_fp16)[name = string("softmax_6_cast_fp16")]; bool matmul_37_transpose_x_0 = const()[name = string("matmul_37_transpose_x_0"), val = bool(false)]; bool matmul_37_transpose_y_0 = const()[name = string("matmul_37_transpose_y_0"), val = bool(false)]; tensor matmul_37_cast_fp16 = matmul(transpose_x = matmul_37_transpose_x_0, transpose_y = matmul_37_transpose_y_0, x = softmax_6_cast_fp16, y = view_40_cast_fp16)[name = string("matmul_37_cast_fp16")]; tensor transpose_58_perm_0 = const()[name = string("transpose_58_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1129 = const()[name = string("const_1129"), val = tensor([1, 512, -1])]; tensor transpose_58_cast_fp16 = transpose(perm = transpose_58_perm_0, x = matmul_37_cast_fp16)[name = string("transpose_68")]; tensor view_41_cast_fp16 = reshape(shape = const_1129, x = transpose_58_cast_fp16)[name = string("view_41_cast_fp16")]; tensor p_st_0_model_layers_6_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228843456))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229433344))))[name = string("p_st_0_model_layers_6_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_45_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_6_self_attn_o_proj_weight_to_fp16_quantized, x = view_41_cast_fp16)[name = string("linear_45_cast_fp16")]; fp16 const_1131_promoted_to_fp16 = const()[name = string("const_1131_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_40_cast_fp16 = pow(x = linear_45_cast_fp16, y = const_1131_promoted_to_fp16)[name = string("pow_40_cast_fp16")]; tensor mean_39_axes_0 = const()[name = string("mean_39_axes_0"), val = tensor([-1])]; bool mean_39_keep_dims_0 = const()[name = string("mean_39_keep_dims_0"), val = bool(true)]; tensor mean_39_cast_fp16 = reduce_mean(axes = mean_39_axes_0, keep_dims = mean_39_keep_dims_0, x = pow_40_cast_fp16)[name = string("mean_39_cast_fp16")]; fp16 const_1134_to_fp16 = const()[name = string("const_1134_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_111_cast_fp16 = add(x = mean_39_cast_fp16, y = const_1134_to_fp16)[name = string("add_111_cast_fp16")]; fp32 rsqrt_39_epsilon_0 = const()[name = string("rsqrt_39_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_39_cast_fp16 = rsqrt(epsilon = rsqrt_39_epsilon_0, x = add_111_cast_fp16)[name = string("rsqrt_39_cast_fp16")]; tensor mul_170_cast_fp16 = mul(x = linear_45_cast_fp16, y = rsqrt_39_cast_fp16)[name = string("mul_170_cast_fp16")]; tensor add_112_to_fp16 = const()[name = string("add_112_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229434944)))]; tensor mul_171_cast_fp16 = mul(x = mul_170_cast_fp16, y = add_112_to_fp16)[name = string("mul_171_cast_fp16")]; tensor add_113_cast_fp16 = add(x = add_101_cast_fp16, y = mul_171_cast_fp16)[name = string("add_113_cast_fp16")]; fp16 const_1139_promoted_to_fp16 = const()[name = string("const_1139_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_41_cast_fp16 = pow(x = add_113_cast_fp16, y = const_1139_promoted_to_fp16)[name = string("pow_41_cast_fp16")]; tensor mean_40_axes_0 = const()[name = string("mean_40_axes_0"), val = tensor([-1])]; bool mean_40_keep_dims_0 = const()[name = string("mean_40_keep_dims_0"), val = bool(true)]; tensor mean_40_cast_fp16 = reduce_mean(axes = mean_40_axes_0, keep_dims = mean_40_keep_dims_0, x = pow_41_cast_fp16)[name = string("mean_40_cast_fp16")]; fp16 const_1142_to_fp16 = const()[name = string("const_1142_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_114_cast_fp16 = add(x = mean_40_cast_fp16, y = const_1142_to_fp16)[name = string("add_114_cast_fp16")]; fp32 rsqrt_40_epsilon_0 = const()[name = string("rsqrt_40_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_40_cast_fp16 = rsqrt(epsilon = rsqrt_40_epsilon_0, x = add_114_cast_fp16)[name = string("rsqrt_40_cast_fp16")]; tensor mul_172_cast_fp16 = mul(x = add_113_cast_fp16, y = rsqrt_40_cast_fp16)[name = string("mul_172_cast_fp16")]; tensor add_115_to_fp16 = const()[name = string("add_115_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229436544)))]; tensor mul_173_cast_fp16 = mul(x = mul_172_cast_fp16, y = add_115_to_fp16)[name = string("mul_173_cast_fp16")]; tensor p_st_0_model_layers_6_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229438144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230322944))))[name = string("p_st_0_model_layers_6_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_46_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_6_mlp_gate_proj_weight_to_fp16_quantized, x = mul_173_cast_fp16)[name = string("linear_46_cast_fp16")]; string gelu_6_mode_0 = const()[name = string("gelu_6_mode_0"), val = string("EXACT")]; tensor gelu_6_cast_fp16 = gelu(mode = gelu_6_mode_0, x = linear_46_cast_fp16)[name = string("gelu_6_cast_fp16")]; tensor p_st_0_model_layers_6_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230325312))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(231210112))))[name = string("p_st_0_model_layers_6_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_47_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_6_mlp_up_proj_weight_to_fp16_quantized, x = mul_173_cast_fp16)[name = string("linear_47_cast_fp16")]; tensor mul_174_cast_fp16 = mul(x = gelu_6_cast_fp16, y = linear_47_cast_fp16)[name = string("mul_174_cast_fp16")]; tensor p_st_0_model_layers_6_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(231212480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232097280))))[name = string("p_st_0_model_layers_6_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_48_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_6_mlp_down_proj_weight_to_fp16_quantized, x = mul_174_cast_fp16)[name = string("linear_48_cast_fp16")]; fp16 const_1147_promoted_to_fp16 = const()[name = string("const_1147_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_42_cast_fp16 = pow(x = linear_48_cast_fp16, y = const_1147_promoted_to_fp16)[name = string("pow_42_cast_fp16")]; tensor mean_41_axes_0 = const()[name = string("mean_41_axes_0"), val = tensor([-1])]; bool mean_41_keep_dims_0 = const()[name = string("mean_41_keep_dims_0"), val = bool(true)]; tensor mean_41_cast_fp16 = reduce_mean(axes = mean_41_axes_0, keep_dims = mean_41_keep_dims_0, x = pow_42_cast_fp16)[name = string("mean_41_cast_fp16")]; fp16 const_1150_to_fp16 = const()[name = string("const_1150_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_116_cast_fp16 = add(x = mean_41_cast_fp16, y = const_1150_to_fp16)[name = string("add_116_cast_fp16")]; fp32 rsqrt_41_epsilon_0 = const()[name = string("rsqrt_41_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_41_cast_fp16 = rsqrt(epsilon = rsqrt_41_epsilon_0, x = add_116_cast_fp16)[name = string("rsqrt_41_cast_fp16")]; tensor mul_175_cast_fp16 = mul(x = linear_48_cast_fp16, y = rsqrt_41_cast_fp16)[name = string("mul_175_cast_fp16")]; tensor add_117_to_fp16 = const()[name = string("add_117_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232098880)))]; tensor mul_176_cast_fp16 = mul(x = mul_175_cast_fp16, y = add_117_to_fp16)[name = string("mul_176_cast_fp16")]; tensor add_118_cast_fp16 = add(x = add_113_cast_fp16, y = mul_176_cast_fp16)[name = string("add_118_cast_fp16")]; fp16 const_1155_promoted_to_fp16 = const()[name = string("const_1155_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_43_cast_fp16 = pow(x = add_118_cast_fp16, y = const_1155_promoted_to_fp16)[name = string("pow_43_cast_fp16")]; tensor mean_42_axes_0 = const()[name = string("mean_42_axes_0"), val = tensor([-1])]; bool mean_42_keep_dims_0 = const()[name = string("mean_42_keep_dims_0"), val = bool(true)]; tensor mean_42_cast_fp16 = reduce_mean(axes = mean_42_axes_0, keep_dims = mean_42_keep_dims_0, x = pow_43_cast_fp16)[name = string("mean_42_cast_fp16")]; fp16 const_1158_to_fp16 = const()[name = string("const_1158_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_119_cast_fp16 = add(x = mean_42_cast_fp16, y = const_1158_to_fp16)[name = string("add_119_cast_fp16")]; fp32 rsqrt_42_epsilon_0 = const()[name = string("rsqrt_42_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_42_cast_fp16 = rsqrt(epsilon = rsqrt_42_epsilon_0, x = add_119_cast_fp16)[name = string("rsqrt_42_cast_fp16")]; tensor mul_177_cast_fp16 = mul(x = add_118_cast_fp16, y = rsqrt_42_cast_fp16)[name = string("mul_177_cast_fp16")]; tensor add_120_to_fp16 = const()[name = string("add_120_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232100480)))]; tensor mul_178_cast_fp16 = mul(x = mul_177_cast_fp16, y = add_120_to_fp16)[name = string("mul_178_cast_fp16")]; tensor p_st_0_model_layers_7_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232102080))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232691968))))[name = string("p_st_0_model_layers_7_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_49_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_7_self_attn_q_proj_weight_to_fp16_quantized, x = mul_178_cast_fp16)[name = string("linear_49_cast_fp16")]; tensor const_1162 = const()[name = string("const_1162"), val = tensor([1, 512, -1, 256])]; tensor view_42_cast_fp16 = reshape(shape = const_1162, x = linear_49_cast_fp16)[name = string("view_42_cast_fp16")]; tensor transpose_59_perm_0 = const()[name = string("transpose_59_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_7_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232693568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232890240))))[name = string("p_st_0_model_layers_7_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_50_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_7_self_attn_k_proj_weight_to_fp16_quantized, x = mul_178_cast_fp16)[name = string("linear_50_cast_fp16")]; tensor const_1165 = const()[name = string("const_1165"), val = tensor([1, 512, -1, 256])]; tensor view_43_cast_fp16 = reshape(shape = const_1165, x = linear_50_cast_fp16)[name = string("view_43_cast_fp16")]; tensor transpose_60_perm_0 = const()[name = string("transpose_60_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_7_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232890816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233087488))))[name = string("p_st_0_model_layers_7_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_51_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_7_self_attn_v_proj_weight_to_fp16_quantized, x = mul_178_cast_fp16)[name = string("linear_51_cast_fp16")]; tensor const_1168 = const()[name = string("const_1168"), val = tensor([1, 512, -1, 256])]; tensor view_44_cast_fp16 = reshape(shape = const_1168, x = linear_51_cast_fp16)[name = string("view_44_cast_fp16")]; tensor transpose_61_perm_0 = const()[name = string("transpose_61_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1172_promoted_to_fp16 = const()[name = string("const_1172_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_59_cast_fp16 = transpose(perm = transpose_59_perm_0, x = view_42_cast_fp16)[name = string("transpose_67")]; tensor pow_44_cast_fp16 = pow(x = transpose_59_cast_fp16, y = const_1172_promoted_to_fp16)[name = string("pow_44_cast_fp16")]; tensor mean_43_axes_0 = const()[name = string("mean_43_axes_0"), val = tensor([-1])]; bool mean_43_keep_dims_0 = const()[name = string("mean_43_keep_dims_0"), val = bool(true)]; tensor mean_43_cast_fp16 = reduce_mean(axes = mean_43_axes_0, keep_dims = mean_43_keep_dims_0, x = pow_44_cast_fp16)[name = string("mean_43_cast_fp16")]; fp16 const_1175_to_fp16 = const()[name = string("const_1175_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_121_cast_fp16 = add(x = mean_43_cast_fp16, y = const_1175_to_fp16)[name = string("add_121_cast_fp16")]; fp32 rsqrt_43_epsilon_0 = const()[name = string("rsqrt_43_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_43_cast_fp16 = rsqrt(epsilon = rsqrt_43_epsilon_0, x = add_121_cast_fp16)[name = string("rsqrt_43_cast_fp16")]; tensor mul_179_cast_fp16 = mul(x = transpose_59_cast_fp16, y = rsqrt_43_cast_fp16)[name = string("mul_179_cast_fp16")]; tensor add_122_to_fp16 = const()[name = string("add_122_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233088064)))]; tensor mul_180_cast_fp16 = mul(x = mul_179_cast_fp16, y = add_122_to_fp16)[name = string("mul_180_cast_fp16")]; fp16 const_1180_promoted_to_fp16 = const()[name = string("const_1180_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_60_cast_fp16 = transpose(perm = transpose_60_perm_0, x = view_43_cast_fp16)[name = string("transpose_66")]; tensor pow_45_cast_fp16 = pow(x = transpose_60_cast_fp16, y = const_1180_promoted_to_fp16)[name = string("pow_45_cast_fp16")]; tensor mean_44_axes_0 = const()[name = string("mean_44_axes_0"), val = tensor([-1])]; bool mean_44_keep_dims_0 = const()[name = string("mean_44_keep_dims_0"), val = bool(true)]; tensor mean_44_cast_fp16 = reduce_mean(axes = mean_44_axes_0, keep_dims = mean_44_keep_dims_0, x = pow_45_cast_fp16)[name = string("mean_44_cast_fp16")]; fp16 const_1183_to_fp16 = const()[name = string("const_1183_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_123_cast_fp16 = add(x = mean_44_cast_fp16, y = const_1183_to_fp16)[name = string("add_123_cast_fp16")]; fp32 rsqrt_44_epsilon_0 = const()[name = string("rsqrt_44_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_44_cast_fp16 = rsqrt(epsilon = rsqrt_44_epsilon_0, x = add_123_cast_fp16)[name = string("rsqrt_44_cast_fp16")]; tensor mul_181_cast_fp16 = mul(x = transpose_60_cast_fp16, y = rsqrt_44_cast_fp16)[name = string("mul_181_cast_fp16")]; tensor add_124_to_fp16 = const()[name = string("add_124_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233088640)))]; tensor mul_182_cast_fp16 = mul(x = mul_181_cast_fp16, y = add_124_to_fp16)[name = string("mul_182_cast_fp16")]; tensor mul_183_cast_fp16 = mul(x = mul_180_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_183_cast_fp16")]; tensor slice_230_begin_0 = const()[name = string("slice_230_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_230_end_0 = const()[name = string("slice_230_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_230_end_mask_0 = const()[name = string("slice_230_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_230_cast_fp16 = slice_by_index(begin = slice_230_begin_0, end = slice_230_end_0, end_mask = slice_230_end_mask_0, x = mul_180_cast_fp16)[name = string("slice_230_cast_fp16")]; tensor slice_231_begin_0 = const()[name = string("slice_231_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_231_end_0 = const()[name = string("slice_231_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_231_end_mask_0 = const()[name = string("slice_231_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_231_cast_fp16 = slice_by_index(begin = slice_231_begin_0, end = slice_231_end_0, end_mask = slice_231_end_mask_0, x = mul_180_cast_fp16)[name = string("slice_231_cast_fp16")]; fp16 const_1195_promoted_to_fp16 = const()[name = string("const_1195_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_14_cast_fp16 = mul(x = slice_231_cast_fp16, y = const_1195_promoted_to_fp16)[name = string("neg_14_cast_fp16")]; int32 const_1196 = const()[name = string("const_1196"), val = int32(-1)]; bool cat_52_interleave_0 = const()[name = string("cat_52_interleave_0"), val = bool(false)]; tensor cat_52_cast_fp16 = concat(axis = const_1196, interleave = cat_52_interleave_0, values = (neg_14_cast_fp16, slice_230_cast_fp16))[name = string("cat_52_cast_fp16")]; tensor mul_184_cast_fp16 = mul(x = cat_52_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_184_cast_fp16")]; tensor add_125_cast_fp16 = add(x = mul_183_cast_fp16, y = mul_184_cast_fp16)[name = string("add_125_cast_fp16")]; tensor mul_185_cast_fp16 = mul(x = mul_182_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_185_cast_fp16")]; tensor slice_232_begin_0 = const()[name = string("slice_232_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_232_end_0 = const()[name = string("slice_232_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_232_end_mask_0 = const()[name = string("slice_232_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_232_cast_fp16 = slice_by_index(begin = slice_232_begin_0, end = slice_232_end_0, end_mask = slice_232_end_mask_0, x = mul_182_cast_fp16)[name = string("slice_232_cast_fp16")]; tensor slice_233_begin_0 = const()[name = string("slice_233_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_233_end_0 = const()[name = string("slice_233_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_233_end_mask_0 = const()[name = string("slice_233_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_233_cast_fp16 = slice_by_index(begin = slice_233_begin_0, end = slice_233_end_0, end_mask = slice_233_end_mask_0, x = mul_182_cast_fp16)[name = string("slice_233_cast_fp16")]; fp16 const_1203_promoted_to_fp16 = const()[name = string("const_1203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_15_cast_fp16 = mul(x = slice_233_cast_fp16, y = const_1203_promoted_to_fp16)[name = string("neg_15_cast_fp16")]; int32 const_1204 = const()[name = string("const_1204"), val = int32(-1)]; bool cat_53_interleave_0 = const()[name = string("cat_53_interleave_0"), val = bool(false)]; tensor cat_53_cast_fp16 = concat(axis = const_1204, interleave = cat_53_interleave_0, values = (neg_15_cast_fp16, slice_232_cast_fp16))[name = string("cat_53_cast_fp16")]; tensor mul_186_cast_fp16 = mul(x = cat_53_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_186_cast_fp16")]; tensor add_126_cast_fp16 = add(x = mul_185_cast_fp16, y = mul_186_cast_fp16)[name = string("add_126_cast_fp16")]; int32 const_1205 = const()[name = string("const_1205"), val = int32(-2)]; bool cat_54_interleave_0 = const()[name = string("cat_54_interleave_0"), val = bool(false)]; tensor cat_54_cast_fp16 = concat(axis = const_1205, interleave = cat_54_interleave_0, values = add_126_cast_fp16)[name = string("cat_54_cast_fp16")]; int32 const_1206 = const()[name = string("const_1206"), val = int32(-2)]; bool cat_55_interleave_0 = const()[name = string("cat_55_interleave_0"), val = bool(false)]; tensor transpose_61_cast_fp16 = transpose(perm = transpose_61_perm_0, x = view_44_cast_fp16)[name = string("transpose_65")]; tensor cat_55_cast_fp16 = concat(axis = const_1206, interleave = cat_55_interleave_0, values = transpose_61_cast_fp16)[name = string("cat_55_cast_fp16")]; tensor unsqueeze_107_axes_0 = const()[name = string("unsqueeze_107_axes_0"), val = tensor([2])]; tensor unsqueeze_107_cast_fp16 = expand_dims(axes = unsqueeze_107_axes_0, x = cat_54_cast_fp16)[name = string("unsqueeze_107_cast_fp16")]; tensor expand_40_reps_0 = const()[name = string("expand_40_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_40_cast_fp16 = tile(reps = expand_40_reps_0, x = unsqueeze_107_cast_fp16)[name = string("expand_40_cast_fp16")]; tensor const_1221 = const()[name = string("const_1221"), val = tensor([1, 3, 512, 256])]; tensor view_45_cast_fp16 = reshape(shape = const_1221, x = expand_40_cast_fp16)[name = string("view_45_cast_fp16")]; tensor unsqueeze_108_axes_0 = const()[name = string("unsqueeze_108_axes_0"), val = tensor([2])]; tensor unsqueeze_108_cast_fp16 = expand_dims(axes = unsqueeze_108_axes_0, x = cat_55_cast_fp16)[name = string("unsqueeze_108_cast_fp16")]; tensor expand_41_reps_0 = const()[name = string("expand_41_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_41_cast_fp16 = tile(reps = expand_41_reps_0, x = unsqueeze_108_cast_fp16)[name = string("expand_41_cast_fp16")]; tensor const_1236 = const()[name = string("const_1236"), val = tensor([1, 3, 512, 256])]; tensor view_46_cast_fp16 = reshape(shape = const_1236, x = expand_41_cast_fp16)[name = string("view_46_cast_fp16")]; bool matmul_38_transpose_x_1 = const()[name = string("matmul_38_transpose_x_1"), val = bool(false)]; bool matmul_38_transpose_y_1 = const()[name = string("matmul_38_transpose_y_1"), val = bool(true)]; tensor matmul_38_cast_fp16 = matmul(transpose_x = matmul_38_transpose_x_1, transpose_y = matmul_38_transpose_y_1, x = add_125_cast_fp16, y = view_45_cast_fp16)[name = string("matmul_38_cast_fp16")]; fp16 const_1239_to_fp16 = const()[name = string("const_1239_to_fp16"), val = fp16(0x1p-4)]; tensor mul_187_cast_fp16 = mul(x = matmul_38_cast_fp16, y = const_1239_to_fp16)[name = string("mul_187_cast_fp16")]; tensor add_127_cast_fp16 = add(x = mul_187_cast_fp16, y = expand_cast_fp16)[name = string("add_127_cast_fp16")]; int32 const_1249 = const()[name = string("const_1249"), val = int32(-1)]; tensor softmax_7_cast_fp16 = softmax(axis = const_1249, x = add_127_cast_fp16)[name = string("softmax_7_cast_fp16")]; bool matmul_39_transpose_x_0 = const()[name = string("matmul_39_transpose_x_0"), val = bool(false)]; bool matmul_39_transpose_y_0 = const()[name = string("matmul_39_transpose_y_0"), val = bool(false)]; tensor matmul_39_cast_fp16 = matmul(transpose_x = matmul_39_transpose_x_0, transpose_y = matmul_39_transpose_y_0, x = softmax_7_cast_fp16, y = view_46_cast_fp16)[name = string("matmul_39_cast_fp16")]; tensor transpose_63_perm_0 = const()[name = string("transpose_63_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1254 = const()[name = string("const_1254"), val = tensor([1, 512, -1])]; tensor transpose_63_cast_fp16 = transpose(perm = transpose_63_perm_0, x = matmul_39_cast_fp16)[name = string("transpose_64")]; tensor view_47_cast_fp16 = reshape(shape = const_1254, x = transpose_63_cast_fp16)[name = string("view_47_cast_fp16")]; tensor p_st_0_model_layers_7_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233089216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233679104))))[name = string("p_st_0_model_layers_7_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_52_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_7_self_attn_o_proj_weight_to_fp16_quantized, x = view_47_cast_fp16)[name = string("linear_52_cast_fp16")]; fp16 const_1256_promoted_to_fp16 = const()[name = string("const_1256_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_46_cast_fp16 = pow(x = linear_52_cast_fp16, y = const_1256_promoted_to_fp16)[name = string("pow_46_cast_fp16")]; tensor mean_45_axes_0 = const()[name = string("mean_45_axes_0"), val = tensor([-1])]; bool mean_45_keep_dims_0 = const()[name = string("mean_45_keep_dims_0"), val = bool(true)]; tensor mean_45_cast_fp16 = reduce_mean(axes = mean_45_axes_0, keep_dims = mean_45_keep_dims_0, x = pow_46_cast_fp16)[name = string("mean_45_cast_fp16")]; fp16 const_1259_to_fp16 = const()[name = string("const_1259_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_128_cast_fp16 = add(x = mean_45_cast_fp16, y = const_1259_to_fp16)[name = string("add_128_cast_fp16")]; fp32 rsqrt_45_epsilon_0 = const()[name = string("rsqrt_45_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_45_cast_fp16 = rsqrt(epsilon = rsqrt_45_epsilon_0, x = add_128_cast_fp16)[name = string("rsqrt_45_cast_fp16")]; tensor mul_188_cast_fp16 = mul(x = linear_52_cast_fp16, y = rsqrt_45_cast_fp16)[name = string("mul_188_cast_fp16")]; tensor add_129_to_fp16 = const()[name = string("add_129_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233680704)))]; tensor mul_189_cast_fp16 = mul(x = mul_188_cast_fp16, y = add_129_to_fp16)[name = string("mul_189_cast_fp16")]; tensor add_130_cast_fp16 = add(x = add_118_cast_fp16, y = mul_189_cast_fp16)[name = string("add_130_cast_fp16")]; fp16 const_1264_promoted_to_fp16 = const()[name = string("const_1264_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_47_cast_fp16 = pow(x = add_130_cast_fp16, y = const_1264_promoted_to_fp16)[name = string("pow_47_cast_fp16")]; tensor mean_46_axes_0 = const()[name = string("mean_46_axes_0"), val = tensor([-1])]; bool mean_46_keep_dims_0 = const()[name = string("mean_46_keep_dims_0"), val = bool(true)]; tensor mean_46_cast_fp16 = reduce_mean(axes = mean_46_axes_0, keep_dims = mean_46_keep_dims_0, x = pow_47_cast_fp16)[name = string("mean_46_cast_fp16")]; fp16 const_1267_to_fp16 = const()[name = string("const_1267_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_131_cast_fp16 = add(x = mean_46_cast_fp16, y = const_1267_to_fp16)[name = string("add_131_cast_fp16")]; fp32 rsqrt_46_epsilon_0 = const()[name = string("rsqrt_46_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_46_cast_fp16 = rsqrt(epsilon = rsqrt_46_epsilon_0, x = add_131_cast_fp16)[name = string("rsqrt_46_cast_fp16")]; tensor mul_190_cast_fp16 = mul(x = add_130_cast_fp16, y = rsqrt_46_cast_fp16)[name = string("mul_190_cast_fp16")]; tensor add_132_to_fp16 = const()[name = string("add_132_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233682304)))]; tensor mul_191_cast_fp16 = mul(x = mul_190_cast_fp16, y = add_132_to_fp16)[name = string("mul_191_cast_fp16")]; tensor p_st_0_model_layers_7_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233683904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234568704))))[name = string("p_st_0_model_layers_7_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_53_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_7_mlp_gate_proj_weight_to_fp16_quantized, x = mul_191_cast_fp16)[name = string("linear_53_cast_fp16")]; string gelu_7_mode_0 = const()[name = string("gelu_7_mode_0"), val = string("EXACT")]; tensor gelu_7_cast_fp16 = gelu(mode = gelu_7_mode_0, x = linear_53_cast_fp16)[name = string("gelu_7_cast_fp16")]; tensor p_st_0_model_layers_7_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234571072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235455872))))[name = string("p_st_0_model_layers_7_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_54_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_7_mlp_up_proj_weight_to_fp16_quantized, x = mul_191_cast_fp16)[name = string("linear_54_cast_fp16")]; tensor mul_192_cast_fp16 = mul(x = gelu_7_cast_fp16, y = linear_54_cast_fp16)[name = string("mul_192_cast_fp16")]; tensor p_st_0_model_layers_7_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235458240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236343040))))[name = string("p_st_0_model_layers_7_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_55_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_7_mlp_down_proj_weight_to_fp16_quantized, x = mul_192_cast_fp16)[name = string("linear_55_cast_fp16")]; fp16 const_1272_promoted_to_fp16 = const()[name = string("const_1272_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_48_cast_fp16 = pow(x = linear_55_cast_fp16, y = const_1272_promoted_to_fp16)[name = string("pow_48_cast_fp16")]; tensor mean_47_axes_0 = const()[name = string("mean_47_axes_0"), val = tensor([-1])]; bool mean_47_keep_dims_0 = const()[name = string("mean_47_keep_dims_0"), val = bool(true)]; tensor mean_47_cast_fp16 = reduce_mean(axes = mean_47_axes_0, keep_dims = mean_47_keep_dims_0, x = pow_48_cast_fp16)[name = string("mean_47_cast_fp16")]; fp16 const_1275_to_fp16 = const()[name = string("const_1275_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_133_cast_fp16 = add(x = mean_47_cast_fp16, y = const_1275_to_fp16)[name = string("add_133_cast_fp16")]; fp32 rsqrt_47_epsilon_0 = const()[name = string("rsqrt_47_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_47_cast_fp16 = rsqrt(epsilon = rsqrt_47_epsilon_0, x = add_133_cast_fp16)[name = string("rsqrt_47_cast_fp16")]; tensor mul_193_cast_fp16 = mul(x = linear_55_cast_fp16, y = rsqrt_47_cast_fp16)[name = string("mul_193_cast_fp16")]; tensor add_134_to_fp16 = const()[name = string("add_134_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236344640)))]; tensor mul_194_cast_fp16 = mul(x = mul_193_cast_fp16, y = add_134_to_fp16)[name = string("mul_194_cast_fp16")]; tensor add_135_cast_fp16 = add(x = add_130_cast_fp16, y = mul_194_cast_fp16)[name = string("add_135_cast_fp16")]; fp16 const_1280_promoted_to_fp16 = const()[name = string("const_1280_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_49_cast_fp16 = pow(x = add_135_cast_fp16, y = const_1280_promoted_to_fp16)[name = string("pow_49_cast_fp16")]; tensor mean_48_axes_0 = const()[name = string("mean_48_axes_0"), val = tensor([-1])]; bool mean_48_keep_dims_0 = const()[name = string("mean_48_keep_dims_0"), val = bool(true)]; tensor mean_48_cast_fp16 = reduce_mean(axes = mean_48_axes_0, keep_dims = mean_48_keep_dims_0, x = pow_49_cast_fp16)[name = string("mean_48_cast_fp16")]; fp16 const_1283_to_fp16 = const()[name = string("const_1283_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_136_cast_fp16 = add(x = mean_48_cast_fp16, y = const_1283_to_fp16)[name = string("add_136_cast_fp16")]; fp32 rsqrt_48_epsilon_0 = const()[name = string("rsqrt_48_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_48_cast_fp16 = rsqrt(epsilon = rsqrt_48_epsilon_0, x = add_136_cast_fp16)[name = string("rsqrt_48_cast_fp16")]; tensor mul_195_cast_fp16 = mul(x = add_135_cast_fp16, y = rsqrt_48_cast_fp16)[name = string("mul_195_cast_fp16")]; tensor add_137_to_fp16 = const()[name = string("add_137_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236346240)))]; tensor mul_196_cast_fp16 = mul(x = mul_195_cast_fp16, y = add_137_to_fp16)[name = string("mul_196_cast_fp16")]; tensor p_st_0_model_layers_8_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236347840))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236937728))))[name = string("p_st_0_model_layers_8_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_56_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_8_self_attn_q_proj_weight_to_fp16_quantized, x = mul_196_cast_fp16)[name = string("linear_56_cast_fp16")]; tensor const_1287 = const()[name = string("const_1287"), val = tensor([1, 512, -1, 256])]; tensor view_48_cast_fp16 = reshape(shape = const_1287, x = linear_56_cast_fp16)[name = string("view_48_cast_fp16")]; tensor transpose_64_perm_0 = const()[name = string("transpose_64_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_8_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236939328))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237136000))))[name = string("p_st_0_model_layers_8_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_57_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_8_self_attn_k_proj_weight_to_fp16_quantized, x = mul_196_cast_fp16)[name = string("linear_57_cast_fp16")]; tensor const_1290 = const()[name = string("const_1290"), val = tensor([1, 512, -1, 256])]; tensor view_49_cast_fp16 = reshape(shape = const_1290, x = linear_57_cast_fp16)[name = string("view_49_cast_fp16")]; tensor transpose_65_perm_0 = const()[name = string("transpose_65_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_8_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237136576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237333248))))[name = string("p_st_0_model_layers_8_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_58_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_8_self_attn_v_proj_weight_to_fp16_quantized, x = mul_196_cast_fp16)[name = string("linear_58_cast_fp16")]; tensor const_1293 = const()[name = string("const_1293"), val = tensor([1, 512, -1, 256])]; tensor view_50_cast_fp16 = reshape(shape = const_1293, x = linear_58_cast_fp16)[name = string("view_50_cast_fp16")]; tensor transpose_66_perm_0 = const()[name = string("transpose_66_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1297_promoted_to_fp16 = const()[name = string("const_1297_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_64_cast_fp16 = transpose(perm = transpose_64_perm_0, x = view_48_cast_fp16)[name = string("transpose_63")]; tensor pow_50_cast_fp16 = pow(x = transpose_64_cast_fp16, y = const_1297_promoted_to_fp16)[name = string("pow_50_cast_fp16")]; tensor mean_49_axes_0 = const()[name = string("mean_49_axes_0"), val = tensor([-1])]; bool mean_49_keep_dims_0 = const()[name = string("mean_49_keep_dims_0"), val = bool(true)]; tensor mean_49_cast_fp16 = reduce_mean(axes = mean_49_axes_0, keep_dims = mean_49_keep_dims_0, x = pow_50_cast_fp16)[name = string("mean_49_cast_fp16")]; fp16 const_1300_to_fp16 = const()[name = string("const_1300_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_138_cast_fp16 = add(x = mean_49_cast_fp16, y = const_1300_to_fp16)[name = string("add_138_cast_fp16")]; fp32 rsqrt_49_epsilon_0 = const()[name = string("rsqrt_49_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_49_cast_fp16 = rsqrt(epsilon = rsqrt_49_epsilon_0, x = add_138_cast_fp16)[name = string("rsqrt_49_cast_fp16")]; tensor mul_197_cast_fp16 = mul(x = transpose_64_cast_fp16, y = rsqrt_49_cast_fp16)[name = string("mul_197_cast_fp16")]; tensor add_139_to_fp16 = const()[name = string("add_139_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237333824)))]; tensor mul_198_cast_fp16 = mul(x = mul_197_cast_fp16, y = add_139_to_fp16)[name = string("mul_198_cast_fp16")]; fp16 const_1305_promoted_to_fp16 = const()[name = string("const_1305_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_65_cast_fp16 = transpose(perm = transpose_65_perm_0, x = view_49_cast_fp16)[name = string("transpose_62")]; tensor pow_51_cast_fp16 = pow(x = transpose_65_cast_fp16, y = const_1305_promoted_to_fp16)[name = string("pow_51_cast_fp16")]; tensor mean_50_axes_0 = const()[name = string("mean_50_axes_0"), val = tensor([-1])]; bool mean_50_keep_dims_0 = const()[name = string("mean_50_keep_dims_0"), val = bool(true)]; tensor mean_50_cast_fp16 = reduce_mean(axes = mean_50_axes_0, keep_dims = mean_50_keep_dims_0, x = pow_51_cast_fp16)[name = string("mean_50_cast_fp16")]; fp16 const_1308_to_fp16 = const()[name = string("const_1308_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_140_cast_fp16 = add(x = mean_50_cast_fp16, y = const_1308_to_fp16)[name = string("add_140_cast_fp16")]; fp32 rsqrt_50_epsilon_0 = const()[name = string("rsqrt_50_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_50_cast_fp16 = rsqrt(epsilon = rsqrt_50_epsilon_0, x = add_140_cast_fp16)[name = string("rsqrt_50_cast_fp16")]; tensor mul_199_cast_fp16 = mul(x = transpose_65_cast_fp16, y = rsqrt_50_cast_fp16)[name = string("mul_199_cast_fp16")]; tensor add_141_to_fp16 = const()[name = string("add_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237334400)))]; tensor mul_200_cast_fp16 = mul(x = mul_199_cast_fp16, y = add_141_to_fp16)[name = string("mul_200_cast_fp16")]; tensor mul_201_cast_fp16 = mul(x = mul_198_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_201_cast_fp16")]; tensor slice_253_begin_0 = const()[name = string("slice_253_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_253_end_0 = const()[name = string("slice_253_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_253_end_mask_0 = const()[name = string("slice_253_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_253_cast_fp16 = slice_by_index(begin = slice_253_begin_0, end = slice_253_end_0, end_mask = slice_253_end_mask_0, x = mul_198_cast_fp16)[name = string("slice_253_cast_fp16")]; tensor slice_254_begin_0 = const()[name = string("slice_254_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_254_end_0 = const()[name = string("slice_254_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_254_end_mask_0 = const()[name = string("slice_254_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_254_cast_fp16 = slice_by_index(begin = slice_254_begin_0, end = slice_254_end_0, end_mask = slice_254_end_mask_0, x = mul_198_cast_fp16)[name = string("slice_254_cast_fp16")]; fp16 const_1320_promoted_to_fp16 = const()[name = string("const_1320_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_16_cast_fp16 = mul(x = slice_254_cast_fp16, y = const_1320_promoted_to_fp16)[name = string("neg_16_cast_fp16")]; int32 const_1321 = const()[name = string("const_1321"), val = int32(-1)]; bool cat_56_interleave_0 = const()[name = string("cat_56_interleave_0"), val = bool(false)]; tensor cat_56_cast_fp16 = concat(axis = const_1321, interleave = cat_56_interleave_0, values = (neg_16_cast_fp16, slice_253_cast_fp16))[name = string("cat_56_cast_fp16")]; tensor mul_202_cast_fp16 = mul(x = cat_56_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_202_cast_fp16")]; tensor add_142_cast_fp16 = add(x = mul_201_cast_fp16, y = mul_202_cast_fp16)[name = string("add_142_cast_fp16")]; tensor mul_203_cast_fp16 = mul(x = mul_200_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_203_cast_fp16")]; tensor slice_255_begin_0 = const()[name = string("slice_255_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_255_end_0 = const()[name = string("slice_255_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_255_end_mask_0 = const()[name = string("slice_255_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_255_cast_fp16 = slice_by_index(begin = slice_255_begin_0, end = slice_255_end_0, end_mask = slice_255_end_mask_0, x = mul_200_cast_fp16)[name = string("slice_255_cast_fp16")]; tensor slice_256_begin_0 = const()[name = string("slice_256_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_256_end_0 = const()[name = string("slice_256_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_256_end_mask_0 = const()[name = string("slice_256_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_256_cast_fp16 = slice_by_index(begin = slice_256_begin_0, end = slice_256_end_0, end_mask = slice_256_end_mask_0, x = mul_200_cast_fp16)[name = string("slice_256_cast_fp16")]; fp16 const_1328_promoted_to_fp16 = const()[name = string("const_1328_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_17_cast_fp16 = mul(x = slice_256_cast_fp16, y = const_1328_promoted_to_fp16)[name = string("neg_17_cast_fp16")]; int32 const_1329 = const()[name = string("const_1329"), val = int32(-1)]; bool cat_57_interleave_0 = const()[name = string("cat_57_interleave_0"), val = bool(false)]; tensor cat_57_cast_fp16 = concat(axis = const_1329, interleave = cat_57_interleave_0, values = (neg_17_cast_fp16, slice_255_cast_fp16))[name = string("cat_57_cast_fp16")]; tensor mul_204_cast_fp16 = mul(x = cat_57_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_204_cast_fp16")]; tensor add_143_cast_fp16 = add(x = mul_203_cast_fp16, y = mul_204_cast_fp16)[name = string("add_143_cast_fp16")]; int32 const_1330 = const()[name = string("const_1330"), val = int32(-2)]; bool cat_58_interleave_0 = const()[name = string("cat_58_interleave_0"), val = bool(false)]; tensor cat_58_cast_fp16 = concat(axis = const_1330, interleave = cat_58_interleave_0, values = add_143_cast_fp16)[name = string("cat_58_cast_fp16")]; int32 const_1331 = const()[name = string("const_1331"), val = int32(-2)]; bool cat_59_interleave_0 = const()[name = string("cat_59_interleave_0"), val = bool(false)]; tensor transpose_66_cast_fp16 = transpose(perm = transpose_66_perm_0, x = view_50_cast_fp16)[name = string("transpose_61")]; tensor cat_59_cast_fp16 = concat(axis = const_1331, interleave = cat_59_interleave_0, values = transpose_66_cast_fp16)[name = string("cat_59_cast_fp16")]; tensor unsqueeze_111_axes_0 = const()[name = string("unsqueeze_111_axes_0"), val = tensor([2])]; tensor unsqueeze_111_cast_fp16 = expand_dims(axes = unsqueeze_111_axes_0, x = cat_58_cast_fp16)[name = string("unsqueeze_111_cast_fp16")]; tensor expand_42_reps_0 = const()[name = string("expand_42_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_42_cast_fp16 = tile(reps = expand_42_reps_0, x = unsqueeze_111_cast_fp16)[name = string("expand_42_cast_fp16")]; tensor const_1346 = const()[name = string("const_1346"), val = tensor([1, 3, 512, 256])]; tensor view_51_cast_fp16 = reshape(shape = const_1346, x = expand_42_cast_fp16)[name = string("view_51_cast_fp16")]; tensor unsqueeze_112_axes_0 = const()[name = string("unsqueeze_112_axes_0"), val = tensor([2])]; tensor unsqueeze_112_cast_fp16 = expand_dims(axes = unsqueeze_112_axes_0, x = cat_59_cast_fp16)[name = string("unsqueeze_112_cast_fp16")]; tensor expand_43_reps_0 = const()[name = string("expand_43_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_43_cast_fp16 = tile(reps = expand_43_reps_0, x = unsqueeze_112_cast_fp16)[name = string("expand_43_cast_fp16")]; tensor const_1361 = const()[name = string("const_1361"), val = tensor([1, 3, 512, 256])]; tensor view_52_cast_fp16 = reshape(shape = const_1361, x = expand_43_cast_fp16)[name = string("view_52_cast_fp16")]; bool matmul_40_transpose_x_1 = const()[name = string("matmul_40_transpose_x_1"), val = bool(false)]; bool matmul_40_transpose_y_1 = const()[name = string("matmul_40_transpose_y_1"), val = bool(true)]; tensor matmul_40_cast_fp16 = matmul(transpose_x = matmul_40_transpose_x_1, transpose_y = matmul_40_transpose_y_1, x = add_142_cast_fp16, y = view_51_cast_fp16)[name = string("matmul_40_cast_fp16")]; fp16 const_1364_to_fp16 = const()[name = string("const_1364_to_fp16"), val = fp16(0x1p-4)]; tensor mul_205_cast_fp16 = mul(x = matmul_40_cast_fp16, y = const_1364_to_fp16)[name = string("mul_205_cast_fp16")]; tensor add_144_cast_fp16 = add(x = mul_205_cast_fp16, y = expand_cast_fp16)[name = string("add_144_cast_fp16")]; int32 const_1374 = const()[name = string("const_1374"), val = int32(-1)]; tensor softmax_8_cast_fp16 = softmax(axis = const_1374, x = add_144_cast_fp16)[name = string("softmax_8_cast_fp16")]; bool matmul_41_transpose_x_0 = const()[name = string("matmul_41_transpose_x_0"), val = bool(false)]; bool matmul_41_transpose_y_0 = const()[name = string("matmul_41_transpose_y_0"), val = bool(false)]; tensor matmul_41_cast_fp16 = matmul(transpose_x = matmul_41_transpose_x_0, transpose_y = matmul_41_transpose_y_0, x = softmax_8_cast_fp16, y = view_52_cast_fp16)[name = string("matmul_41_cast_fp16")]; tensor transpose_68_perm_0 = const()[name = string("transpose_68_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1379 = const()[name = string("const_1379"), val = tensor([1, 512, -1])]; tensor transpose_68_cast_fp16 = transpose(perm = transpose_68_perm_0, x = matmul_41_cast_fp16)[name = string("transpose_60")]; tensor view_53_cast_fp16 = reshape(shape = const_1379, x = transpose_68_cast_fp16)[name = string("view_53_cast_fp16")]; tensor p_st_0_model_layers_8_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237924864))))[name = string("p_st_0_model_layers_8_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_59_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_8_self_attn_o_proj_weight_to_fp16_quantized, x = view_53_cast_fp16)[name = string("linear_59_cast_fp16")]; fp16 const_1381_promoted_to_fp16 = const()[name = string("const_1381_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_52_cast_fp16 = pow(x = linear_59_cast_fp16, y = const_1381_promoted_to_fp16)[name = string("pow_52_cast_fp16")]; tensor mean_51_axes_0 = const()[name = string("mean_51_axes_0"), val = tensor([-1])]; bool mean_51_keep_dims_0 = const()[name = string("mean_51_keep_dims_0"), val = bool(true)]; tensor mean_51_cast_fp16 = reduce_mean(axes = mean_51_axes_0, keep_dims = mean_51_keep_dims_0, x = pow_52_cast_fp16)[name = string("mean_51_cast_fp16")]; fp16 const_1384_to_fp16 = const()[name = string("const_1384_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_145_cast_fp16 = add(x = mean_51_cast_fp16, y = const_1384_to_fp16)[name = string("add_145_cast_fp16")]; fp32 rsqrt_51_epsilon_0 = const()[name = string("rsqrt_51_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_51_cast_fp16 = rsqrt(epsilon = rsqrt_51_epsilon_0, x = add_145_cast_fp16)[name = string("rsqrt_51_cast_fp16")]; tensor mul_206_cast_fp16 = mul(x = linear_59_cast_fp16, y = rsqrt_51_cast_fp16)[name = string("mul_206_cast_fp16")]; tensor add_146_to_fp16 = const()[name = string("add_146_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237926464)))]; tensor mul_207_cast_fp16 = mul(x = mul_206_cast_fp16, y = add_146_to_fp16)[name = string("mul_207_cast_fp16")]; tensor add_147_cast_fp16 = add(x = add_135_cast_fp16, y = mul_207_cast_fp16)[name = string("add_147_cast_fp16")]; fp16 const_1389_promoted_to_fp16 = const()[name = string("const_1389_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_53_cast_fp16 = pow(x = add_147_cast_fp16, y = const_1389_promoted_to_fp16)[name = string("pow_53_cast_fp16")]; tensor mean_52_axes_0 = const()[name = string("mean_52_axes_0"), val = tensor([-1])]; bool mean_52_keep_dims_0 = const()[name = string("mean_52_keep_dims_0"), val = bool(true)]; tensor mean_52_cast_fp16 = reduce_mean(axes = mean_52_axes_0, keep_dims = mean_52_keep_dims_0, x = pow_53_cast_fp16)[name = string("mean_52_cast_fp16")]; fp16 const_1392_to_fp16 = const()[name = string("const_1392_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_148_cast_fp16 = add(x = mean_52_cast_fp16, y = const_1392_to_fp16)[name = string("add_148_cast_fp16")]; fp32 rsqrt_52_epsilon_0 = const()[name = string("rsqrt_52_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_52_cast_fp16 = rsqrt(epsilon = rsqrt_52_epsilon_0, x = add_148_cast_fp16)[name = string("rsqrt_52_cast_fp16")]; tensor mul_208_cast_fp16 = mul(x = add_147_cast_fp16, y = rsqrt_52_cast_fp16)[name = string("mul_208_cast_fp16")]; tensor add_149_to_fp16 = const()[name = string("add_149_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237928064)))]; tensor mul_209_cast_fp16 = mul(x = mul_208_cast_fp16, y = add_149_to_fp16)[name = string("mul_209_cast_fp16")]; tensor p_st_0_model_layers_8_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237929664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238814464))))[name = string("p_st_0_model_layers_8_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_60_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_8_mlp_gate_proj_weight_to_fp16_quantized, x = mul_209_cast_fp16)[name = string("linear_60_cast_fp16")]; string gelu_8_mode_0 = const()[name = string("gelu_8_mode_0"), val = string("EXACT")]; tensor gelu_8_cast_fp16 = gelu(mode = gelu_8_mode_0, x = linear_60_cast_fp16)[name = string("gelu_8_cast_fp16")]; tensor p_st_0_model_layers_8_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238816832))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239701632))))[name = string("p_st_0_model_layers_8_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_61_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_8_mlp_up_proj_weight_to_fp16_quantized, x = mul_209_cast_fp16)[name = string("linear_61_cast_fp16")]; tensor mul_210_cast_fp16 = mul(x = gelu_8_cast_fp16, y = linear_61_cast_fp16)[name = string("mul_210_cast_fp16")]; tensor p_st_0_model_layers_8_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239704000))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240588800))))[name = string("p_st_0_model_layers_8_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_62_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_8_mlp_down_proj_weight_to_fp16_quantized, x = mul_210_cast_fp16)[name = string("linear_62_cast_fp16")]; fp16 const_1397_promoted_to_fp16 = const()[name = string("const_1397_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_54_cast_fp16 = pow(x = linear_62_cast_fp16, y = const_1397_promoted_to_fp16)[name = string("pow_54_cast_fp16")]; tensor mean_53_axes_0 = const()[name = string("mean_53_axes_0"), val = tensor([-1])]; bool mean_53_keep_dims_0 = const()[name = string("mean_53_keep_dims_0"), val = bool(true)]; tensor mean_53_cast_fp16 = reduce_mean(axes = mean_53_axes_0, keep_dims = mean_53_keep_dims_0, x = pow_54_cast_fp16)[name = string("mean_53_cast_fp16")]; fp16 const_1400_to_fp16 = const()[name = string("const_1400_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_150_cast_fp16 = add(x = mean_53_cast_fp16, y = const_1400_to_fp16)[name = string("add_150_cast_fp16")]; fp32 rsqrt_53_epsilon_0 = const()[name = string("rsqrt_53_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_53_cast_fp16 = rsqrt(epsilon = rsqrt_53_epsilon_0, x = add_150_cast_fp16)[name = string("rsqrt_53_cast_fp16")]; tensor mul_211_cast_fp16 = mul(x = linear_62_cast_fp16, y = rsqrt_53_cast_fp16)[name = string("mul_211_cast_fp16")]; tensor add_151_to_fp16 = const()[name = string("add_151_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240590400)))]; tensor mul_212_cast_fp16 = mul(x = mul_211_cast_fp16, y = add_151_to_fp16)[name = string("mul_212_cast_fp16")]; tensor add_152_cast_fp16 = add(x = add_147_cast_fp16, y = mul_212_cast_fp16)[name = string("add_152_cast_fp16")]; fp16 const_1405_promoted_to_fp16 = const()[name = string("const_1405_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_55_cast_fp16 = pow(x = add_152_cast_fp16, y = const_1405_promoted_to_fp16)[name = string("pow_55_cast_fp16")]; tensor mean_54_axes_0 = const()[name = string("mean_54_axes_0"), val = tensor([-1])]; bool mean_54_keep_dims_0 = const()[name = string("mean_54_keep_dims_0"), val = bool(true)]; tensor mean_54_cast_fp16 = reduce_mean(axes = mean_54_axes_0, keep_dims = mean_54_keep_dims_0, x = pow_55_cast_fp16)[name = string("mean_54_cast_fp16")]; fp16 const_1408_to_fp16 = const()[name = string("const_1408_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_153_cast_fp16 = add(x = mean_54_cast_fp16, y = const_1408_to_fp16)[name = string("add_153_cast_fp16")]; fp32 rsqrt_54_epsilon_0 = const()[name = string("rsqrt_54_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_54_cast_fp16 = rsqrt(epsilon = rsqrt_54_epsilon_0, x = add_153_cast_fp16)[name = string("rsqrt_54_cast_fp16")]; tensor mul_213_cast_fp16 = mul(x = add_152_cast_fp16, y = rsqrt_54_cast_fp16)[name = string("mul_213_cast_fp16")]; tensor add_154_to_fp16 = const()[name = string("add_154_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240592000)))]; tensor mul_214_cast_fp16 = mul(x = mul_213_cast_fp16, y = add_154_to_fp16)[name = string("mul_214_cast_fp16")]; tensor p_st_0_model_layers_9_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240593600))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241183488))))[name = string("p_st_0_model_layers_9_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_63_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_9_self_attn_q_proj_weight_to_fp16_quantized, x = mul_214_cast_fp16)[name = string("linear_63_cast_fp16")]; tensor const_1412 = const()[name = string("const_1412"), val = tensor([1, 512, -1, 256])]; tensor view_54_cast_fp16 = reshape(shape = const_1412, x = linear_63_cast_fp16)[name = string("view_54_cast_fp16")]; tensor transpose_69_perm_0 = const()[name = string("transpose_69_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_9_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241185088))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241381760))))[name = string("p_st_0_model_layers_9_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_64_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_9_self_attn_k_proj_weight_to_fp16_quantized, x = mul_214_cast_fp16)[name = string("linear_64_cast_fp16")]; tensor const_1415 = const()[name = string("const_1415"), val = tensor([1, 512, -1, 256])]; tensor view_55_cast_fp16 = reshape(shape = const_1415, x = linear_64_cast_fp16)[name = string("view_55_cast_fp16")]; tensor transpose_70_perm_0 = const()[name = string("transpose_70_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_9_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241382336))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241579008))))[name = string("p_st_0_model_layers_9_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_65_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_9_self_attn_v_proj_weight_to_fp16_quantized, x = mul_214_cast_fp16)[name = string("linear_65_cast_fp16")]; tensor const_1418 = const()[name = string("const_1418"), val = tensor([1, 512, -1, 256])]; tensor view_56_cast_fp16 = reshape(shape = const_1418, x = linear_65_cast_fp16)[name = string("view_56_cast_fp16")]; tensor transpose_71_perm_0 = const()[name = string("transpose_71_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1422_promoted_to_fp16 = const()[name = string("const_1422_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_69_cast_fp16 = transpose(perm = transpose_69_perm_0, x = view_54_cast_fp16)[name = string("transpose_59")]; tensor pow_56_cast_fp16 = pow(x = transpose_69_cast_fp16, y = const_1422_promoted_to_fp16)[name = string("pow_56_cast_fp16")]; tensor mean_55_axes_0 = const()[name = string("mean_55_axes_0"), val = tensor([-1])]; bool mean_55_keep_dims_0 = const()[name = string("mean_55_keep_dims_0"), val = bool(true)]; tensor mean_55_cast_fp16 = reduce_mean(axes = mean_55_axes_0, keep_dims = mean_55_keep_dims_0, x = pow_56_cast_fp16)[name = string("mean_55_cast_fp16")]; fp16 const_1425_to_fp16 = const()[name = string("const_1425_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_155_cast_fp16 = add(x = mean_55_cast_fp16, y = const_1425_to_fp16)[name = string("add_155_cast_fp16")]; fp32 rsqrt_55_epsilon_0 = const()[name = string("rsqrt_55_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_55_cast_fp16 = rsqrt(epsilon = rsqrt_55_epsilon_0, x = add_155_cast_fp16)[name = string("rsqrt_55_cast_fp16")]; tensor mul_215_cast_fp16 = mul(x = transpose_69_cast_fp16, y = rsqrt_55_cast_fp16)[name = string("mul_215_cast_fp16")]; tensor add_156_to_fp16 = const()[name = string("add_156_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241579584)))]; tensor mul_216_cast_fp16 = mul(x = mul_215_cast_fp16, y = add_156_to_fp16)[name = string("mul_216_cast_fp16")]; fp16 const_1430_promoted_to_fp16 = const()[name = string("const_1430_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_70_cast_fp16 = transpose(perm = transpose_70_perm_0, x = view_55_cast_fp16)[name = string("transpose_58")]; tensor pow_57_cast_fp16 = pow(x = transpose_70_cast_fp16, y = const_1430_promoted_to_fp16)[name = string("pow_57_cast_fp16")]; tensor mean_56_axes_0 = const()[name = string("mean_56_axes_0"), val = tensor([-1])]; bool mean_56_keep_dims_0 = const()[name = string("mean_56_keep_dims_0"), val = bool(true)]; tensor mean_56_cast_fp16 = reduce_mean(axes = mean_56_axes_0, keep_dims = mean_56_keep_dims_0, x = pow_57_cast_fp16)[name = string("mean_56_cast_fp16")]; fp16 const_1433_to_fp16 = const()[name = string("const_1433_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_157_cast_fp16 = add(x = mean_56_cast_fp16, y = const_1433_to_fp16)[name = string("add_157_cast_fp16")]; fp32 rsqrt_56_epsilon_0 = const()[name = string("rsqrt_56_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_56_cast_fp16 = rsqrt(epsilon = rsqrt_56_epsilon_0, x = add_157_cast_fp16)[name = string("rsqrt_56_cast_fp16")]; tensor mul_217_cast_fp16 = mul(x = transpose_70_cast_fp16, y = rsqrt_56_cast_fp16)[name = string("mul_217_cast_fp16")]; tensor add_158_to_fp16 = const()[name = string("add_158_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241580160)))]; tensor mul_218_cast_fp16 = mul(x = mul_217_cast_fp16, y = add_158_to_fp16)[name = string("mul_218_cast_fp16")]; tensor mul_219_cast_fp16 = mul(x = mul_216_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_219_cast_fp16")]; tensor slice_276_begin_0 = const()[name = string("slice_276_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_276_end_0 = const()[name = string("slice_276_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_276_end_mask_0 = const()[name = string("slice_276_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_276_cast_fp16 = slice_by_index(begin = slice_276_begin_0, end = slice_276_end_0, end_mask = slice_276_end_mask_0, x = mul_216_cast_fp16)[name = string("slice_276_cast_fp16")]; tensor slice_277_begin_0 = const()[name = string("slice_277_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_277_end_0 = const()[name = string("slice_277_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_277_end_mask_0 = const()[name = string("slice_277_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_277_cast_fp16 = slice_by_index(begin = slice_277_begin_0, end = slice_277_end_0, end_mask = slice_277_end_mask_0, x = mul_216_cast_fp16)[name = string("slice_277_cast_fp16")]; fp16 const_1445_promoted_to_fp16 = const()[name = string("const_1445_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_18_cast_fp16 = mul(x = slice_277_cast_fp16, y = const_1445_promoted_to_fp16)[name = string("neg_18_cast_fp16")]; int32 const_1446 = const()[name = string("const_1446"), val = int32(-1)]; bool cat_60_interleave_0 = const()[name = string("cat_60_interleave_0"), val = bool(false)]; tensor cat_60_cast_fp16 = concat(axis = const_1446, interleave = cat_60_interleave_0, values = (neg_18_cast_fp16, slice_276_cast_fp16))[name = string("cat_60_cast_fp16")]; tensor mul_220_cast_fp16 = mul(x = cat_60_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_220_cast_fp16")]; tensor add_159_cast_fp16 = add(x = mul_219_cast_fp16, y = mul_220_cast_fp16)[name = string("add_159_cast_fp16")]; tensor mul_221_cast_fp16 = mul(x = mul_218_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_221_cast_fp16")]; tensor slice_278_begin_0 = const()[name = string("slice_278_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_278_end_0 = const()[name = string("slice_278_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_278_end_mask_0 = const()[name = string("slice_278_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_278_cast_fp16 = slice_by_index(begin = slice_278_begin_0, end = slice_278_end_0, end_mask = slice_278_end_mask_0, x = mul_218_cast_fp16)[name = string("slice_278_cast_fp16")]; tensor slice_279_begin_0 = const()[name = string("slice_279_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_279_end_0 = const()[name = string("slice_279_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_279_end_mask_0 = const()[name = string("slice_279_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_279_cast_fp16 = slice_by_index(begin = slice_279_begin_0, end = slice_279_end_0, end_mask = slice_279_end_mask_0, x = mul_218_cast_fp16)[name = string("slice_279_cast_fp16")]; fp16 const_1453_promoted_to_fp16 = const()[name = string("const_1453_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_19_cast_fp16 = mul(x = slice_279_cast_fp16, y = const_1453_promoted_to_fp16)[name = string("neg_19_cast_fp16")]; int32 const_1454 = const()[name = string("const_1454"), val = int32(-1)]; bool cat_61_interleave_0 = const()[name = string("cat_61_interleave_0"), val = bool(false)]; tensor cat_61_cast_fp16 = concat(axis = const_1454, interleave = cat_61_interleave_0, values = (neg_19_cast_fp16, slice_278_cast_fp16))[name = string("cat_61_cast_fp16")]; tensor mul_222_cast_fp16 = mul(x = cat_61_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_222_cast_fp16")]; tensor add_160_cast_fp16 = add(x = mul_221_cast_fp16, y = mul_222_cast_fp16)[name = string("add_160_cast_fp16")]; int32 const_1455 = const()[name = string("const_1455"), val = int32(-2)]; bool cat_62_interleave_0 = const()[name = string("cat_62_interleave_0"), val = bool(false)]; tensor cat_62_cast_fp16 = concat(axis = const_1455, interleave = cat_62_interleave_0, values = add_160_cast_fp16)[name = string("cat_62_cast_fp16")]; int32 const_1456 = const()[name = string("const_1456"), val = int32(-2)]; bool cat_63_interleave_0 = const()[name = string("cat_63_interleave_0"), val = bool(false)]; tensor transpose_71_cast_fp16 = transpose(perm = transpose_71_perm_0, x = view_56_cast_fp16)[name = string("transpose_57")]; tensor cat_63_cast_fp16 = concat(axis = const_1456, interleave = cat_63_interleave_0, values = transpose_71_cast_fp16)[name = string("cat_63_cast_fp16")]; tensor unsqueeze_115_axes_0 = const()[name = string("unsqueeze_115_axes_0"), val = tensor([2])]; tensor unsqueeze_115_cast_fp16 = expand_dims(axes = unsqueeze_115_axes_0, x = cat_62_cast_fp16)[name = string("unsqueeze_115_cast_fp16")]; tensor expand_44_reps_0 = const()[name = string("expand_44_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_44_cast_fp16 = tile(reps = expand_44_reps_0, x = unsqueeze_115_cast_fp16)[name = string("expand_44_cast_fp16")]; tensor const_1471 = const()[name = string("const_1471"), val = tensor([1, 3, 512, 256])]; tensor view_57_cast_fp16 = reshape(shape = const_1471, x = expand_44_cast_fp16)[name = string("view_57_cast_fp16")]; tensor unsqueeze_116_axes_0 = const()[name = string("unsqueeze_116_axes_0"), val = tensor([2])]; tensor unsqueeze_116_cast_fp16 = expand_dims(axes = unsqueeze_116_axes_0, x = cat_63_cast_fp16)[name = string("unsqueeze_116_cast_fp16")]; tensor expand_45_reps_0 = const()[name = string("expand_45_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_45_cast_fp16 = tile(reps = expand_45_reps_0, x = unsqueeze_116_cast_fp16)[name = string("expand_45_cast_fp16")]; tensor const_1486 = const()[name = string("const_1486"), val = tensor([1, 3, 512, 256])]; tensor view_58_cast_fp16 = reshape(shape = const_1486, x = expand_45_cast_fp16)[name = string("view_58_cast_fp16")]; bool matmul_42_transpose_x_1 = const()[name = string("matmul_42_transpose_x_1"), val = bool(false)]; bool matmul_42_transpose_y_1 = const()[name = string("matmul_42_transpose_y_1"), val = bool(true)]; tensor matmul_42_cast_fp16 = matmul(transpose_x = matmul_42_transpose_x_1, transpose_y = matmul_42_transpose_y_1, x = add_159_cast_fp16, y = view_57_cast_fp16)[name = string("matmul_42_cast_fp16")]; fp16 const_1489_to_fp16 = const()[name = string("const_1489_to_fp16"), val = fp16(0x1p-4)]; tensor mul_223_cast_fp16 = mul(x = matmul_42_cast_fp16, y = const_1489_to_fp16)[name = string("mul_223_cast_fp16")]; tensor add_161_cast_fp16 = add(x = mul_223_cast_fp16, y = expand_cast_fp16)[name = string("add_161_cast_fp16")]; int32 const_1499 = const()[name = string("const_1499"), val = int32(-1)]; tensor softmax_9_cast_fp16 = softmax(axis = const_1499, x = add_161_cast_fp16)[name = string("softmax_9_cast_fp16")]; bool matmul_43_transpose_x_0 = const()[name = string("matmul_43_transpose_x_0"), val = bool(false)]; bool matmul_43_transpose_y_0 = const()[name = string("matmul_43_transpose_y_0"), val = bool(false)]; tensor matmul_43_cast_fp16 = matmul(transpose_x = matmul_43_transpose_x_0, transpose_y = matmul_43_transpose_y_0, x = softmax_9_cast_fp16, y = view_58_cast_fp16)[name = string("matmul_43_cast_fp16")]; tensor transpose_73_perm_0 = const()[name = string("transpose_73_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1504 = const()[name = string("const_1504"), val = tensor([1, 512, -1])]; tensor transpose_73_cast_fp16 = transpose(perm = transpose_73_perm_0, x = matmul_43_cast_fp16)[name = string("transpose_56")]; tensor view_59_cast_fp16 = reshape(shape = const_1504, x = transpose_73_cast_fp16)[name = string("view_59_cast_fp16")]; tensor p_st_0_model_layers_9_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241580736))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242170624))))[name = string("p_st_0_model_layers_9_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_66_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_9_self_attn_o_proj_weight_to_fp16_quantized, x = view_59_cast_fp16)[name = string("linear_66_cast_fp16")]; fp16 const_1506_promoted_to_fp16 = const()[name = string("const_1506_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_58_cast_fp16 = pow(x = linear_66_cast_fp16, y = const_1506_promoted_to_fp16)[name = string("pow_58_cast_fp16")]; tensor mean_57_axes_0 = const()[name = string("mean_57_axes_0"), val = tensor([-1])]; bool mean_57_keep_dims_0 = const()[name = string("mean_57_keep_dims_0"), val = bool(true)]; tensor mean_57_cast_fp16 = reduce_mean(axes = mean_57_axes_0, keep_dims = mean_57_keep_dims_0, x = pow_58_cast_fp16)[name = string("mean_57_cast_fp16")]; fp16 const_1509_to_fp16 = const()[name = string("const_1509_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_162_cast_fp16 = add(x = mean_57_cast_fp16, y = const_1509_to_fp16)[name = string("add_162_cast_fp16")]; fp32 rsqrt_57_epsilon_0 = const()[name = string("rsqrt_57_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_57_cast_fp16 = rsqrt(epsilon = rsqrt_57_epsilon_0, x = add_162_cast_fp16)[name = string("rsqrt_57_cast_fp16")]; tensor mul_224_cast_fp16 = mul(x = linear_66_cast_fp16, y = rsqrt_57_cast_fp16)[name = string("mul_224_cast_fp16")]; tensor add_163_to_fp16 = const()[name = string("add_163_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242172224)))]; tensor mul_225_cast_fp16 = mul(x = mul_224_cast_fp16, y = add_163_to_fp16)[name = string("mul_225_cast_fp16")]; tensor add_164_cast_fp16 = add(x = add_152_cast_fp16, y = mul_225_cast_fp16)[name = string("add_164_cast_fp16")]; fp16 const_1514_promoted_to_fp16 = const()[name = string("const_1514_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_59_cast_fp16 = pow(x = add_164_cast_fp16, y = const_1514_promoted_to_fp16)[name = string("pow_59_cast_fp16")]; tensor mean_58_axes_0 = const()[name = string("mean_58_axes_0"), val = tensor([-1])]; bool mean_58_keep_dims_0 = const()[name = string("mean_58_keep_dims_0"), val = bool(true)]; tensor mean_58_cast_fp16 = reduce_mean(axes = mean_58_axes_0, keep_dims = mean_58_keep_dims_0, x = pow_59_cast_fp16)[name = string("mean_58_cast_fp16")]; fp16 const_1517_to_fp16 = const()[name = string("const_1517_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_165_cast_fp16 = add(x = mean_58_cast_fp16, y = const_1517_to_fp16)[name = string("add_165_cast_fp16")]; fp32 rsqrt_58_epsilon_0 = const()[name = string("rsqrt_58_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_58_cast_fp16 = rsqrt(epsilon = rsqrt_58_epsilon_0, x = add_165_cast_fp16)[name = string("rsqrt_58_cast_fp16")]; tensor mul_226_cast_fp16 = mul(x = add_164_cast_fp16, y = rsqrt_58_cast_fp16)[name = string("mul_226_cast_fp16")]; tensor add_166_to_fp16 = const()[name = string("add_166_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242173824)))]; tensor mul_227_cast_fp16 = mul(x = mul_226_cast_fp16, y = add_166_to_fp16)[name = string("mul_227_cast_fp16")]; tensor p_st_0_model_layers_9_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242175424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243060224))))[name = string("p_st_0_model_layers_9_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_67_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_9_mlp_gate_proj_weight_to_fp16_quantized, x = mul_227_cast_fp16)[name = string("linear_67_cast_fp16")]; string gelu_9_mode_0 = const()[name = string("gelu_9_mode_0"), val = string("EXACT")]; tensor gelu_9_cast_fp16 = gelu(mode = gelu_9_mode_0, x = linear_67_cast_fp16)[name = string("gelu_9_cast_fp16")]; tensor p_st_0_model_layers_9_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243062592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243947392))))[name = string("p_st_0_model_layers_9_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_68_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_9_mlp_up_proj_weight_to_fp16_quantized, x = mul_227_cast_fp16)[name = string("linear_68_cast_fp16")]; tensor mul_228_cast_fp16 = mul(x = gelu_9_cast_fp16, y = linear_68_cast_fp16)[name = string("mul_228_cast_fp16")]; tensor p_st_0_model_layers_9_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243949760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244834560))))[name = string("p_st_0_model_layers_9_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_69_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_9_mlp_down_proj_weight_to_fp16_quantized, x = mul_228_cast_fp16)[name = string("linear_69_cast_fp16")]; fp16 const_1522_promoted_to_fp16 = const()[name = string("const_1522_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_60_cast_fp16 = pow(x = linear_69_cast_fp16, y = const_1522_promoted_to_fp16)[name = string("pow_60_cast_fp16")]; tensor mean_59_axes_0 = const()[name = string("mean_59_axes_0"), val = tensor([-1])]; bool mean_59_keep_dims_0 = const()[name = string("mean_59_keep_dims_0"), val = bool(true)]; tensor mean_59_cast_fp16 = reduce_mean(axes = mean_59_axes_0, keep_dims = mean_59_keep_dims_0, x = pow_60_cast_fp16)[name = string("mean_59_cast_fp16")]; fp16 const_1525_to_fp16 = const()[name = string("const_1525_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_167_cast_fp16 = add(x = mean_59_cast_fp16, y = const_1525_to_fp16)[name = string("add_167_cast_fp16")]; fp32 rsqrt_59_epsilon_0 = const()[name = string("rsqrt_59_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_59_cast_fp16 = rsqrt(epsilon = rsqrt_59_epsilon_0, x = add_167_cast_fp16)[name = string("rsqrt_59_cast_fp16")]; tensor mul_229_cast_fp16 = mul(x = linear_69_cast_fp16, y = rsqrt_59_cast_fp16)[name = string("mul_229_cast_fp16")]; tensor add_168_to_fp16 = const()[name = string("add_168_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244836160)))]; tensor mul_230_cast_fp16 = mul(x = mul_229_cast_fp16, y = add_168_to_fp16)[name = string("mul_230_cast_fp16")]; tensor add_169_cast_fp16 = add(x = add_164_cast_fp16, y = mul_230_cast_fp16)[name = string("add_169_cast_fp16")]; fp16 const_1530_promoted_to_fp16 = const()[name = string("const_1530_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_61_cast_fp16 = pow(x = add_169_cast_fp16, y = const_1530_promoted_to_fp16)[name = string("pow_61_cast_fp16")]; tensor mean_60_axes_0 = const()[name = string("mean_60_axes_0"), val = tensor([-1])]; bool mean_60_keep_dims_0 = const()[name = string("mean_60_keep_dims_0"), val = bool(true)]; tensor mean_60_cast_fp16 = reduce_mean(axes = mean_60_axes_0, keep_dims = mean_60_keep_dims_0, x = pow_61_cast_fp16)[name = string("mean_60_cast_fp16")]; fp16 const_1533_to_fp16 = const()[name = string("const_1533_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_170_cast_fp16 = add(x = mean_60_cast_fp16, y = const_1533_to_fp16)[name = string("add_170_cast_fp16")]; fp32 rsqrt_60_epsilon_0 = const()[name = string("rsqrt_60_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_60_cast_fp16 = rsqrt(epsilon = rsqrt_60_epsilon_0, x = add_170_cast_fp16)[name = string("rsqrt_60_cast_fp16")]; tensor mul_231_cast_fp16 = mul(x = add_169_cast_fp16, y = rsqrt_60_cast_fp16)[name = string("mul_231_cast_fp16")]; tensor add_171_to_fp16 = const()[name = string("add_171_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244837760)))]; tensor mul_232_cast_fp16 = mul(x = mul_231_cast_fp16, y = add_171_to_fp16)[name = string("mul_232_cast_fp16")]; tensor p_st_0_model_layers_10_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244839360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245429248))))[name = string("p_st_0_model_layers_10_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_70_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_10_self_attn_q_proj_weight_to_fp16_quantized, x = mul_232_cast_fp16)[name = string("linear_70_cast_fp16")]; tensor const_1537 = const()[name = string("const_1537"), val = tensor([1, 512, -1, 256])]; tensor view_60_cast_fp16 = reshape(shape = const_1537, x = linear_70_cast_fp16)[name = string("view_60_cast_fp16")]; tensor transpose_74_perm_0 = const()[name = string("transpose_74_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_10_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245430848))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245627520))))[name = string("p_st_0_model_layers_10_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_71_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_10_self_attn_k_proj_weight_to_fp16_quantized, x = mul_232_cast_fp16)[name = string("linear_71_cast_fp16")]; tensor const_1540 = const()[name = string("const_1540"), val = tensor([1, 512, -1, 256])]; tensor view_61_cast_fp16 = reshape(shape = const_1540, x = linear_71_cast_fp16)[name = string("view_61_cast_fp16")]; tensor transpose_75_perm_0 = const()[name = string("transpose_75_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_10_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245628096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245824768))))[name = string("p_st_0_model_layers_10_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_72_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_10_self_attn_v_proj_weight_to_fp16_quantized, x = mul_232_cast_fp16)[name = string("linear_72_cast_fp16")]; tensor const_1543 = const()[name = string("const_1543"), val = tensor([1, 512, -1, 256])]; tensor view_62_cast_fp16 = reshape(shape = const_1543, x = linear_72_cast_fp16)[name = string("view_62_cast_fp16")]; tensor transpose_76_perm_0 = const()[name = string("transpose_76_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1547_promoted_to_fp16 = const()[name = string("const_1547_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_74_cast_fp16 = transpose(perm = transpose_74_perm_0, x = view_60_cast_fp16)[name = string("transpose_55")]; tensor pow_62_cast_fp16 = pow(x = transpose_74_cast_fp16, y = const_1547_promoted_to_fp16)[name = string("pow_62_cast_fp16")]; tensor mean_61_axes_0 = const()[name = string("mean_61_axes_0"), val = tensor([-1])]; bool mean_61_keep_dims_0 = const()[name = string("mean_61_keep_dims_0"), val = bool(true)]; tensor mean_61_cast_fp16 = reduce_mean(axes = mean_61_axes_0, keep_dims = mean_61_keep_dims_0, x = pow_62_cast_fp16)[name = string("mean_61_cast_fp16")]; fp16 const_1550_to_fp16 = const()[name = string("const_1550_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_172_cast_fp16 = add(x = mean_61_cast_fp16, y = const_1550_to_fp16)[name = string("add_172_cast_fp16")]; fp32 rsqrt_61_epsilon_0 = const()[name = string("rsqrt_61_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_61_cast_fp16 = rsqrt(epsilon = rsqrt_61_epsilon_0, x = add_172_cast_fp16)[name = string("rsqrt_61_cast_fp16")]; tensor mul_233_cast_fp16 = mul(x = transpose_74_cast_fp16, y = rsqrt_61_cast_fp16)[name = string("mul_233_cast_fp16")]; tensor add_173_to_fp16 = const()[name = string("add_173_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245825344)))]; tensor mul_234_cast_fp16 = mul(x = mul_233_cast_fp16, y = add_173_to_fp16)[name = string("mul_234_cast_fp16")]; fp16 const_1555_promoted_to_fp16 = const()[name = string("const_1555_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_75_cast_fp16 = transpose(perm = transpose_75_perm_0, x = view_61_cast_fp16)[name = string("transpose_54")]; tensor pow_63_cast_fp16 = pow(x = transpose_75_cast_fp16, y = const_1555_promoted_to_fp16)[name = string("pow_63_cast_fp16")]; tensor mean_62_axes_0 = const()[name = string("mean_62_axes_0"), val = tensor([-1])]; bool mean_62_keep_dims_0 = const()[name = string("mean_62_keep_dims_0"), val = bool(true)]; tensor mean_62_cast_fp16 = reduce_mean(axes = mean_62_axes_0, keep_dims = mean_62_keep_dims_0, x = pow_63_cast_fp16)[name = string("mean_62_cast_fp16")]; fp16 const_1558_to_fp16 = const()[name = string("const_1558_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_174_cast_fp16 = add(x = mean_62_cast_fp16, y = const_1558_to_fp16)[name = string("add_174_cast_fp16")]; fp32 rsqrt_62_epsilon_0 = const()[name = string("rsqrt_62_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_62_cast_fp16 = rsqrt(epsilon = rsqrt_62_epsilon_0, x = add_174_cast_fp16)[name = string("rsqrt_62_cast_fp16")]; tensor mul_235_cast_fp16 = mul(x = transpose_75_cast_fp16, y = rsqrt_62_cast_fp16)[name = string("mul_235_cast_fp16")]; tensor add_175_to_fp16 = const()[name = string("add_175_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245825920)))]; tensor mul_236_cast_fp16 = mul(x = mul_235_cast_fp16, y = add_175_to_fp16)[name = string("mul_236_cast_fp16")]; tensor mul_237_cast_fp16 = mul(x = mul_234_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_237_cast_fp16")]; tensor slice_299_begin_0 = const()[name = string("slice_299_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_299_end_0 = const()[name = string("slice_299_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_299_end_mask_0 = const()[name = string("slice_299_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_299_cast_fp16 = slice_by_index(begin = slice_299_begin_0, end = slice_299_end_0, end_mask = slice_299_end_mask_0, x = mul_234_cast_fp16)[name = string("slice_299_cast_fp16")]; tensor slice_300_begin_0 = const()[name = string("slice_300_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_300_end_0 = const()[name = string("slice_300_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_300_end_mask_0 = const()[name = string("slice_300_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_300_cast_fp16 = slice_by_index(begin = slice_300_begin_0, end = slice_300_end_0, end_mask = slice_300_end_mask_0, x = mul_234_cast_fp16)[name = string("slice_300_cast_fp16")]; fp16 const_1570_promoted_to_fp16 = const()[name = string("const_1570_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_20_cast_fp16 = mul(x = slice_300_cast_fp16, y = const_1570_promoted_to_fp16)[name = string("neg_20_cast_fp16")]; int32 const_1571 = const()[name = string("const_1571"), val = int32(-1)]; bool cat_64_interleave_0 = const()[name = string("cat_64_interleave_0"), val = bool(false)]; tensor cat_64_cast_fp16 = concat(axis = const_1571, interleave = cat_64_interleave_0, values = (neg_20_cast_fp16, slice_299_cast_fp16))[name = string("cat_64_cast_fp16")]; tensor mul_238_cast_fp16 = mul(x = cat_64_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_238_cast_fp16")]; tensor add_176_cast_fp16 = add(x = mul_237_cast_fp16, y = mul_238_cast_fp16)[name = string("add_176_cast_fp16")]; tensor mul_239_cast_fp16 = mul(x = mul_236_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_239_cast_fp16")]; tensor slice_301_begin_0 = const()[name = string("slice_301_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_301_end_0 = const()[name = string("slice_301_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_301_end_mask_0 = const()[name = string("slice_301_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_301_cast_fp16 = slice_by_index(begin = slice_301_begin_0, end = slice_301_end_0, end_mask = slice_301_end_mask_0, x = mul_236_cast_fp16)[name = string("slice_301_cast_fp16")]; tensor slice_302_begin_0 = const()[name = string("slice_302_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_302_end_0 = const()[name = string("slice_302_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_302_end_mask_0 = const()[name = string("slice_302_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_302_cast_fp16 = slice_by_index(begin = slice_302_begin_0, end = slice_302_end_0, end_mask = slice_302_end_mask_0, x = mul_236_cast_fp16)[name = string("slice_302_cast_fp16")]; fp16 const_1578_promoted_to_fp16 = const()[name = string("const_1578_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_21_cast_fp16 = mul(x = slice_302_cast_fp16, y = const_1578_promoted_to_fp16)[name = string("neg_21_cast_fp16")]; int32 const_1579 = const()[name = string("const_1579"), val = int32(-1)]; bool cat_65_interleave_0 = const()[name = string("cat_65_interleave_0"), val = bool(false)]; tensor cat_65_cast_fp16 = concat(axis = const_1579, interleave = cat_65_interleave_0, values = (neg_21_cast_fp16, slice_301_cast_fp16))[name = string("cat_65_cast_fp16")]; tensor mul_240_cast_fp16 = mul(x = cat_65_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_240_cast_fp16")]; tensor add_177_cast_fp16 = add(x = mul_239_cast_fp16, y = mul_240_cast_fp16)[name = string("add_177_cast_fp16")]; int32 const_1580 = const()[name = string("const_1580"), val = int32(-2)]; bool cat_66_interleave_0 = const()[name = string("cat_66_interleave_0"), val = bool(false)]; tensor cat_66_cast_fp16 = concat(axis = const_1580, interleave = cat_66_interleave_0, values = add_177_cast_fp16)[name = string("cat_66_cast_fp16")]; int32 const_1581 = const()[name = string("const_1581"), val = int32(-2)]; bool cat_67_interleave_0 = const()[name = string("cat_67_interleave_0"), val = bool(false)]; tensor transpose_76_cast_fp16 = transpose(perm = transpose_76_perm_0, x = view_62_cast_fp16)[name = string("transpose_53")]; tensor cat_67_cast_fp16 = concat(axis = const_1581, interleave = cat_67_interleave_0, values = transpose_76_cast_fp16)[name = string("cat_67_cast_fp16")]; tensor unsqueeze_119_axes_0 = const()[name = string("unsqueeze_119_axes_0"), val = tensor([2])]; tensor unsqueeze_119_cast_fp16 = expand_dims(axes = unsqueeze_119_axes_0, x = cat_66_cast_fp16)[name = string("unsqueeze_119_cast_fp16")]; tensor expand_46_reps_0 = const()[name = string("expand_46_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_46_cast_fp16 = tile(reps = expand_46_reps_0, x = unsqueeze_119_cast_fp16)[name = string("expand_46_cast_fp16")]; tensor const_1596 = const()[name = string("const_1596"), val = tensor([1, 3, 512, 256])]; tensor view_63_cast_fp16 = reshape(shape = const_1596, x = expand_46_cast_fp16)[name = string("view_63_cast_fp16")]; tensor unsqueeze_120_axes_0 = const()[name = string("unsqueeze_120_axes_0"), val = tensor([2])]; tensor unsqueeze_120_cast_fp16 = expand_dims(axes = unsqueeze_120_axes_0, x = cat_67_cast_fp16)[name = string("unsqueeze_120_cast_fp16")]; tensor expand_47_reps_0 = const()[name = string("expand_47_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_47_cast_fp16 = tile(reps = expand_47_reps_0, x = unsqueeze_120_cast_fp16)[name = string("expand_47_cast_fp16")]; tensor const_1611 = const()[name = string("const_1611"), val = tensor([1, 3, 512, 256])]; tensor view_64_cast_fp16 = reshape(shape = const_1611, x = expand_47_cast_fp16)[name = string("view_64_cast_fp16")]; bool matmul_44_transpose_x_1 = const()[name = string("matmul_44_transpose_x_1"), val = bool(false)]; bool matmul_44_transpose_y_1 = const()[name = string("matmul_44_transpose_y_1"), val = bool(true)]; tensor matmul_44_cast_fp16 = matmul(transpose_x = matmul_44_transpose_x_1, transpose_y = matmul_44_transpose_y_1, x = add_176_cast_fp16, y = view_63_cast_fp16)[name = string("matmul_44_cast_fp16")]; fp16 const_1614_to_fp16 = const()[name = string("const_1614_to_fp16"), val = fp16(0x1p-4)]; tensor mul_241_cast_fp16 = mul(x = matmul_44_cast_fp16, y = const_1614_to_fp16)[name = string("mul_241_cast_fp16")]; tensor add_178_cast_fp16 = add(x = mul_241_cast_fp16, y = expand_cast_fp16)[name = string("add_178_cast_fp16")]; int32 const_1624 = const()[name = string("const_1624"), val = int32(-1)]; tensor softmax_10_cast_fp16 = softmax(axis = const_1624, x = add_178_cast_fp16)[name = string("softmax_10_cast_fp16")]; bool matmul_45_transpose_x_0 = const()[name = string("matmul_45_transpose_x_0"), val = bool(false)]; bool matmul_45_transpose_y_0 = const()[name = string("matmul_45_transpose_y_0"), val = bool(false)]; tensor matmul_45_cast_fp16 = matmul(transpose_x = matmul_45_transpose_x_0, transpose_y = matmul_45_transpose_y_0, x = softmax_10_cast_fp16, y = view_64_cast_fp16)[name = string("matmul_45_cast_fp16")]; tensor transpose_78_perm_0 = const()[name = string("transpose_78_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1629 = const()[name = string("const_1629"), val = tensor([1, 512, -1])]; tensor transpose_78_cast_fp16 = transpose(perm = transpose_78_perm_0, x = matmul_45_cast_fp16)[name = string("transpose_52")]; tensor view_65_cast_fp16 = reshape(shape = const_1629, x = transpose_78_cast_fp16)[name = string("view_65_cast_fp16")]; tensor p_st_0_model_layers_10_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245826496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(246416384))))[name = string("p_st_0_model_layers_10_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_73_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_10_self_attn_o_proj_weight_to_fp16_quantized, x = view_65_cast_fp16)[name = string("linear_73_cast_fp16")]; fp16 const_1631_promoted_to_fp16 = const()[name = string("const_1631_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_64_cast_fp16 = pow(x = linear_73_cast_fp16, y = const_1631_promoted_to_fp16)[name = string("pow_64_cast_fp16")]; tensor mean_63_axes_0 = const()[name = string("mean_63_axes_0"), val = tensor([-1])]; bool mean_63_keep_dims_0 = const()[name = string("mean_63_keep_dims_0"), val = bool(true)]; tensor mean_63_cast_fp16 = reduce_mean(axes = mean_63_axes_0, keep_dims = mean_63_keep_dims_0, x = pow_64_cast_fp16)[name = string("mean_63_cast_fp16")]; fp16 const_1634_to_fp16 = const()[name = string("const_1634_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_179_cast_fp16 = add(x = mean_63_cast_fp16, y = const_1634_to_fp16)[name = string("add_179_cast_fp16")]; fp32 rsqrt_63_epsilon_0 = const()[name = string("rsqrt_63_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_63_cast_fp16 = rsqrt(epsilon = rsqrt_63_epsilon_0, x = add_179_cast_fp16)[name = string("rsqrt_63_cast_fp16")]; tensor mul_242_cast_fp16 = mul(x = linear_73_cast_fp16, y = rsqrt_63_cast_fp16)[name = string("mul_242_cast_fp16")]; tensor add_180_to_fp16 = const()[name = string("add_180_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(246417984)))]; tensor mul_243_cast_fp16 = mul(x = mul_242_cast_fp16, y = add_180_to_fp16)[name = string("mul_243_cast_fp16")]; tensor add_181_cast_fp16 = add(x = add_169_cast_fp16, y = mul_243_cast_fp16)[name = string("add_181_cast_fp16")]; fp16 const_1639_promoted_to_fp16 = const()[name = string("const_1639_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_65_cast_fp16 = pow(x = add_181_cast_fp16, y = const_1639_promoted_to_fp16)[name = string("pow_65_cast_fp16")]; tensor mean_64_axes_0 = const()[name = string("mean_64_axes_0"), val = tensor([-1])]; bool mean_64_keep_dims_0 = const()[name = string("mean_64_keep_dims_0"), val = bool(true)]; tensor mean_64_cast_fp16 = reduce_mean(axes = mean_64_axes_0, keep_dims = mean_64_keep_dims_0, x = pow_65_cast_fp16)[name = string("mean_64_cast_fp16")]; fp16 const_1642_to_fp16 = const()[name = string("const_1642_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_182_cast_fp16 = add(x = mean_64_cast_fp16, y = const_1642_to_fp16)[name = string("add_182_cast_fp16")]; fp32 rsqrt_64_epsilon_0 = const()[name = string("rsqrt_64_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_64_cast_fp16 = rsqrt(epsilon = rsqrt_64_epsilon_0, x = add_182_cast_fp16)[name = string("rsqrt_64_cast_fp16")]; tensor mul_244_cast_fp16 = mul(x = add_181_cast_fp16, y = rsqrt_64_cast_fp16)[name = string("mul_244_cast_fp16")]; tensor add_183_to_fp16 = const()[name = string("add_183_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(246419584)))]; tensor mul_245_cast_fp16 = mul(x = mul_244_cast_fp16, y = add_183_to_fp16)[name = string("mul_245_cast_fp16")]; tensor p_st_0_model_layers_10_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(246421184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247305984))))[name = string("p_st_0_model_layers_10_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_74_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_10_mlp_gate_proj_weight_to_fp16_quantized, x = mul_245_cast_fp16)[name = string("linear_74_cast_fp16")]; string gelu_10_mode_0 = const()[name = string("gelu_10_mode_0"), val = string("EXACT")]; tensor gelu_10_cast_fp16 = gelu(mode = gelu_10_mode_0, x = linear_74_cast_fp16)[name = string("gelu_10_cast_fp16")]; tensor p_st_0_model_layers_10_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247308352))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248193152))))[name = string("p_st_0_model_layers_10_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_75_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_10_mlp_up_proj_weight_to_fp16_quantized, x = mul_245_cast_fp16)[name = string("linear_75_cast_fp16")]; tensor mul_246_cast_fp16 = mul(x = gelu_10_cast_fp16, y = linear_75_cast_fp16)[name = string("mul_246_cast_fp16")]; tensor p_st_0_model_layers_10_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248195520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249080320))))[name = string("p_st_0_model_layers_10_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_76_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_10_mlp_down_proj_weight_to_fp16_quantized, x = mul_246_cast_fp16)[name = string("linear_76_cast_fp16")]; fp16 const_1647_promoted_to_fp16 = const()[name = string("const_1647_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_66_cast_fp16 = pow(x = linear_76_cast_fp16, y = const_1647_promoted_to_fp16)[name = string("pow_66_cast_fp16")]; tensor mean_65_axes_0 = const()[name = string("mean_65_axes_0"), val = tensor([-1])]; bool mean_65_keep_dims_0 = const()[name = string("mean_65_keep_dims_0"), val = bool(true)]; tensor mean_65_cast_fp16 = reduce_mean(axes = mean_65_axes_0, keep_dims = mean_65_keep_dims_0, x = pow_66_cast_fp16)[name = string("mean_65_cast_fp16")]; fp16 const_1650_to_fp16 = const()[name = string("const_1650_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_184_cast_fp16 = add(x = mean_65_cast_fp16, y = const_1650_to_fp16)[name = string("add_184_cast_fp16")]; fp32 rsqrt_65_epsilon_0 = const()[name = string("rsqrt_65_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_65_cast_fp16 = rsqrt(epsilon = rsqrt_65_epsilon_0, x = add_184_cast_fp16)[name = string("rsqrt_65_cast_fp16")]; tensor mul_247_cast_fp16 = mul(x = linear_76_cast_fp16, y = rsqrt_65_cast_fp16)[name = string("mul_247_cast_fp16")]; tensor add_185_to_fp16 = const()[name = string("add_185_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249081920)))]; tensor mul_248_cast_fp16 = mul(x = mul_247_cast_fp16, y = add_185_to_fp16)[name = string("mul_248_cast_fp16")]; tensor add_186_cast_fp16 = add(x = add_181_cast_fp16, y = mul_248_cast_fp16)[name = string("add_186_cast_fp16")]; fp16 const_1655_promoted_to_fp16 = const()[name = string("const_1655_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_67_cast_fp16 = pow(x = add_186_cast_fp16, y = const_1655_promoted_to_fp16)[name = string("pow_67_cast_fp16")]; tensor mean_66_axes_0 = const()[name = string("mean_66_axes_0"), val = tensor([-1])]; bool mean_66_keep_dims_0 = const()[name = string("mean_66_keep_dims_0"), val = bool(true)]; tensor mean_66_cast_fp16 = reduce_mean(axes = mean_66_axes_0, keep_dims = mean_66_keep_dims_0, x = pow_67_cast_fp16)[name = string("mean_66_cast_fp16")]; fp16 const_1658_to_fp16 = const()[name = string("const_1658_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_187_cast_fp16 = add(x = mean_66_cast_fp16, y = const_1658_to_fp16)[name = string("add_187_cast_fp16")]; fp32 rsqrt_66_epsilon_0 = const()[name = string("rsqrt_66_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_66_cast_fp16 = rsqrt(epsilon = rsqrt_66_epsilon_0, x = add_187_cast_fp16)[name = string("rsqrt_66_cast_fp16")]; tensor mul_249_cast_fp16 = mul(x = add_186_cast_fp16, y = rsqrt_66_cast_fp16)[name = string("mul_249_cast_fp16")]; tensor add_188_to_fp16 = const()[name = string("add_188_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249083520)))]; tensor mul_250_cast_fp16 = mul(x = mul_249_cast_fp16, y = add_188_to_fp16)[name = string("mul_250_cast_fp16")]; tensor p_st_0_model_layers_11_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249085120))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249675008))))[name = string("p_st_0_model_layers_11_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_77_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_11_self_attn_q_proj_weight_to_fp16_quantized, x = mul_250_cast_fp16)[name = string("linear_77_cast_fp16")]; tensor const_1662 = const()[name = string("const_1662"), val = tensor([1, 512, -1, 256])]; tensor view_66_cast_fp16 = reshape(shape = const_1662, x = linear_77_cast_fp16)[name = string("view_66_cast_fp16")]; tensor transpose_79_perm_0 = const()[name = string("transpose_79_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_11_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249676608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249873280))))[name = string("p_st_0_model_layers_11_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_78_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_11_self_attn_k_proj_weight_to_fp16_quantized, x = mul_250_cast_fp16)[name = string("linear_78_cast_fp16")]; tensor const_1665 = const()[name = string("const_1665"), val = tensor([1, 512, -1, 256])]; tensor view_67_cast_fp16 = reshape(shape = const_1665, x = linear_78_cast_fp16)[name = string("view_67_cast_fp16")]; tensor transpose_80_perm_0 = const()[name = string("transpose_80_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_11_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249873856))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250070528))))[name = string("p_st_0_model_layers_11_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_79_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_11_self_attn_v_proj_weight_to_fp16_quantized, x = mul_250_cast_fp16)[name = string("linear_79_cast_fp16")]; tensor const_1668 = const()[name = string("const_1668"), val = tensor([1, 512, -1, 256])]; tensor view_68_cast_fp16 = reshape(shape = const_1668, x = linear_79_cast_fp16)[name = string("view_68_cast_fp16")]; tensor transpose_81_perm_0 = const()[name = string("transpose_81_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1672_promoted_to_fp16 = const()[name = string("const_1672_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_79_cast_fp16 = transpose(perm = transpose_79_perm_0, x = view_66_cast_fp16)[name = string("transpose_51")]; tensor pow_68_cast_fp16 = pow(x = transpose_79_cast_fp16, y = const_1672_promoted_to_fp16)[name = string("pow_68_cast_fp16")]; tensor mean_67_axes_0 = const()[name = string("mean_67_axes_0"), val = tensor([-1])]; bool mean_67_keep_dims_0 = const()[name = string("mean_67_keep_dims_0"), val = bool(true)]; tensor mean_67_cast_fp16 = reduce_mean(axes = mean_67_axes_0, keep_dims = mean_67_keep_dims_0, x = pow_68_cast_fp16)[name = string("mean_67_cast_fp16")]; fp16 const_1675_to_fp16 = const()[name = string("const_1675_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_189_cast_fp16 = add(x = mean_67_cast_fp16, y = const_1675_to_fp16)[name = string("add_189_cast_fp16")]; fp32 rsqrt_67_epsilon_0 = const()[name = string("rsqrt_67_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_67_cast_fp16 = rsqrt(epsilon = rsqrt_67_epsilon_0, x = add_189_cast_fp16)[name = string("rsqrt_67_cast_fp16")]; tensor mul_251_cast_fp16 = mul(x = transpose_79_cast_fp16, y = rsqrt_67_cast_fp16)[name = string("mul_251_cast_fp16")]; tensor add_190_to_fp16 = const()[name = string("add_190_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250071104)))]; tensor mul_252_cast_fp16 = mul(x = mul_251_cast_fp16, y = add_190_to_fp16)[name = string("mul_252_cast_fp16")]; fp16 const_1680_promoted_to_fp16 = const()[name = string("const_1680_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_80_cast_fp16 = transpose(perm = transpose_80_perm_0, x = view_67_cast_fp16)[name = string("transpose_50")]; tensor pow_69_cast_fp16 = pow(x = transpose_80_cast_fp16, y = const_1680_promoted_to_fp16)[name = string("pow_69_cast_fp16")]; tensor mean_68_axes_0 = const()[name = string("mean_68_axes_0"), val = tensor([-1])]; bool mean_68_keep_dims_0 = const()[name = string("mean_68_keep_dims_0"), val = bool(true)]; tensor mean_68_cast_fp16 = reduce_mean(axes = mean_68_axes_0, keep_dims = mean_68_keep_dims_0, x = pow_69_cast_fp16)[name = string("mean_68_cast_fp16")]; fp16 const_1683_to_fp16 = const()[name = string("const_1683_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_191_cast_fp16 = add(x = mean_68_cast_fp16, y = const_1683_to_fp16)[name = string("add_191_cast_fp16")]; fp32 rsqrt_68_epsilon_0 = const()[name = string("rsqrt_68_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_68_cast_fp16 = rsqrt(epsilon = rsqrt_68_epsilon_0, x = add_191_cast_fp16)[name = string("rsqrt_68_cast_fp16")]; tensor mul_253_cast_fp16 = mul(x = transpose_80_cast_fp16, y = rsqrt_68_cast_fp16)[name = string("mul_253_cast_fp16")]; tensor add_192_to_fp16 = const()[name = string("add_192_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250071680)))]; tensor mul_254_cast_fp16 = mul(x = mul_253_cast_fp16, y = add_192_to_fp16)[name = string("mul_254_cast_fp16")]; tensor mul_255_cast_fp16 = mul(x = mul_252_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_255_cast_fp16")]; tensor slice_322_begin_0 = const()[name = string("slice_322_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_322_end_0 = const()[name = string("slice_322_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_322_end_mask_0 = const()[name = string("slice_322_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_322_cast_fp16 = slice_by_index(begin = slice_322_begin_0, end = slice_322_end_0, end_mask = slice_322_end_mask_0, x = mul_252_cast_fp16)[name = string("slice_322_cast_fp16")]; tensor slice_323_begin_0 = const()[name = string("slice_323_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_323_end_0 = const()[name = string("slice_323_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_323_end_mask_0 = const()[name = string("slice_323_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_323_cast_fp16 = slice_by_index(begin = slice_323_begin_0, end = slice_323_end_0, end_mask = slice_323_end_mask_0, x = mul_252_cast_fp16)[name = string("slice_323_cast_fp16")]; fp16 const_1695_promoted_to_fp16 = const()[name = string("const_1695_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_22_cast_fp16 = mul(x = slice_323_cast_fp16, y = const_1695_promoted_to_fp16)[name = string("neg_22_cast_fp16")]; int32 const_1696 = const()[name = string("const_1696"), val = int32(-1)]; bool cat_68_interleave_0 = const()[name = string("cat_68_interleave_0"), val = bool(false)]; tensor cat_68_cast_fp16 = concat(axis = const_1696, interleave = cat_68_interleave_0, values = (neg_22_cast_fp16, slice_322_cast_fp16))[name = string("cat_68_cast_fp16")]; tensor mul_256_cast_fp16 = mul(x = cat_68_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_256_cast_fp16")]; tensor add_193_cast_fp16 = add(x = mul_255_cast_fp16, y = mul_256_cast_fp16)[name = string("add_193_cast_fp16")]; tensor mul_257_cast_fp16 = mul(x = mul_254_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_257_cast_fp16")]; tensor slice_324_begin_0 = const()[name = string("slice_324_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_324_end_0 = const()[name = string("slice_324_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_324_end_mask_0 = const()[name = string("slice_324_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_324_cast_fp16 = slice_by_index(begin = slice_324_begin_0, end = slice_324_end_0, end_mask = slice_324_end_mask_0, x = mul_254_cast_fp16)[name = string("slice_324_cast_fp16")]; tensor slice_325_begin_0 = const()[name = string("slice_325_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_325_end_0 = const()[name = string("slice_325_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_325_end_mask_0 = const()[name = string("slice_325_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_325_cast_fp16 = slice_by_index(begin = slice_325_begin_0, end = slice_325_end_0, end_mask = slice_325_end_mask_0, x = mul_254_cast_fp16)[name = string("slice_325_cast_fp16")]; fp16 const_1703_promoted_to_fp16 = const()[name = string("const_1703_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_23_cast_fp16 = mul(x = slice_325_cast_fp16, y = const_1703_promoted_to_fp16)[name = string("neg_23_cast_fp16")]; int32 const_1704 = const()[name = string("const_1704"), val = int32(-1)]; bool cat_69_interleave_0 = const()[name = string("cat_69_interleave_0"), val = bool(false)]; tensor cat_69_cast_fp16 = concat(axis = const_1704, interleave = cat_69_interleave_0, values = (neg_23_cast_fp16, slice_324_cast_fp16))[name = string("cat_69_cast_fp16")]; tensor mul_258_cast_fp16 = mul(x = cat_69_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_258_cast_fp16")]; tensor add_194_cast_fp16 = add(x = mul_257_cast_fp16, y = mul_258_cast_fp16)[name = string("add_194_cast_fp16")]; int32 const_1705 = const()[name = string("const_1705"), val = int32(-2)]; bool cat_70_interleave_0 = const()[name = string("cat_70_interleave_0"), val = bool(false)]; tensor cat_70_cast_fp16 = concat(axis = const_1705, interleave = cat_70_interleave_0, values = add_194_cast_fp16)[name = string("cat_70_cast_fp16")]; int32 const_1706 = const()[name = string("const_1706"), val = int32(-2)]; bool cat_71_interleave_0 = const()[name = string("cat_71_interleave_0"), val = bool(false)]; tensor transpose_81_cast_fp16 = transpose(perm = transpose_81_perm_0, x = view_68_cast_fp16)[name = string("transpose_49")]; tensor cat_71_cast_fp16 = concat(axis = const_1706, interleave = cat_71_interleave_0, values = transpose_81_cast_fp16)[name = string("cat_71_cast_fp16")]; tensor unsqueeze_123_axes_0 = const()[name = string("unsqueeze_123_axes_0"), val = tensor([2])]; tensor unsqueeze_123_cast_fp16 = expand_dims(axes = unsqueeze_123_axes_0, x = cat_70_cast_fp16)[name = string("unsqueeze_123_cast_fp16")]; tensor expand_48_reps_0 = const()[name = string("expand_48_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_48_cast_fp16 = tile(reps = expand_48_reps_0, x = unsqueeze_123_cast_fp16)[name = string("expand_48_cast_fp16")]; tensor const_1721 = const()[name = string("const_1721"), val = tensor([1, 3, 512, 256])]; tensor view_69_cast_fp16 = reshape(shape = const_1721, x = expand_48_cast_fp16)[name = string("view_69_cast_fp16")]; tensor unsqueeze_124_axes_0 = const()[name = string("unsqueeze_124_axes_0"), val = tensor([2])]; tensor unsqueeze_124_cast_fp16 = expand_dims(axes = unsqueeze_124_axes_0, x = cat_71_cast_fp16)[name = string("unsqueeze_124_cast_fp16")]; tensor expand_49_reps_0 = const()[name = string("expand_49_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_49_cast_fp16 = tile(reps = expand_49_reps_0, x = unsqueeze_124_cast_fp16)[name = string("expand_49_cast_fp16")]; tensor const_1736 = const()[name = string("const_1736"), val = tensor([1, 3, 512, 256])]; tensor view_70_cast_fp16 = reshape(shape = const_1736, x = expand_49_cast_fp16)[name = string("view_70_cast_fp16")]; bool matmul_46_transpose_x_1 = const()[name = string("matmul_46_transpose_x_1"), val = bool(false)]; bool matmul_46_transpose_y_1 = const()[name = string("matmul_46_transpose_y_1"), val = bool(true)]; tensor matmul_46_cast_fp16 = matmul(transpose_x = matmul_46_transpose_x_1, transpose_y = matmul_46_transpose_y_1, x = add_193_cast_fp16, y = view_69_cast_fp16)[name = string("matmul_46_cast_fp16")]; fp16 const_1739_to_fp16 = const()[name = string("const_1739_to_fp16"), val = fp16(0x1p-4)]; tensor mul_259_cast_fp16 = mul(x = matmul_46_cast_fp16, y = const_1739_to_fp16)[name = string("mul_259_cast_fp16")]; tensor add_195_cast_fp16 = add(x = mul_259_cast_fp16, y = expand_cast_fp16)[name = string("add_195_cast_fp16")]; int32 const_1749 = const()[name = string("const_1749"), val = int32(-1)]; tensor softmax_11_cast_fp16 = softmax(axis = const_1749, x = add_195_cast_fp16)[name = string("softmax_11_cast_fp16")]; bool matmul_47_transpose_x_0 = const()[name = string("matmul_47_transpose_x_0"), val = bool(false)]; bool matmul_47_transpose_y_0 = const()[name = string("matmul_47_transpose_y_0"), val = bool(false)]; tensor matmul_47_cast_fp16 = matmul(transpose_x = matmul_47_transpose_x_0, transpose_y = matmul_47_transpose_y_0, x = softmax_11_cast_fp16, y = view_70_cast_fp16)[name = string("matmul_47_cast_fp16")]; tensor transpose_83_perm_0 = const()[name = string("transpose_83_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1754 = const()[name = string("const_1754"), val = tensor([1, 512, -1])]; tensor transpose_83_cast_fp16 = transpose(perm = transpose_83_perm_0, x = matmul_47_cast_fp16)[name = string("transpose_48")]; tensor view_71_cast_fp16 = reshape(shape = const_1754, x = transpose_83_cast_fp16)[name = string("view_71_cast_fp16")]; tensor p_st_0_model_layers_11_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250072256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250662144))))[name = string("p_st_0_model_layers_11_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_80_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_11_self_attn_o_proj_weight_to_fp16_quantized, x = view_71_cast_fp16)[name = string("linear_80_cast_fp16")]; fp16 const_1756_promoted_to_fp16 = const()[name = string("const_1756_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_70_cast_fp16 = pow(x = linear_80_cast_fp16, y = const_1756_promoted_to_fp16)[name = string("pow_70_cast_fp16")]; tensor mean_69_axes_0 = const()[name = string("mean_69_axes_0"), val = tensor([-1])]; bool mean_69_keep_dims_0 = const()[name = string("mean_69_keep_dims_0"), val = bool(true)]; tensor mean_69_cast_fp16 = reduce_mean(axes = mean_69_axes_0, keep_dims = mean_69_keep_dims_0, x = pow_70_cast_fp16)[name = string("mean_69_cast_fp16")]; fp16 const_1759_to_fp16 = const()[name = string("const_1759_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_196_cast_fp16 = add(x = mean_69_cast_fp16, y = const_1759_to_fp16)[name = string("add_196_cast_fp16")]; fp32 rsqrt_69_epsilon_0 = const()[name = string("rsqrt_69_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_69_cast_fp16 = rsqrt(epsilon = rsqrt_69_epsilon_0, x = add_196_cast_fp16)[name = string("rsqrt_69_cast_fp16")]; tensor mul_260_cast_fp16 = mul(x = linear_80_cast_fp16, y = rsqrt_69_cast_fp16)[name = string("mul_260_cast_fp16")]; tensor add_197_to_fp16 = const()[name = string("add_197_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250663744)))]; tensor mul_261_cast_fp16 = mul(x = mul_260_cast_fp16, y = add_197_to_fp16)[name = string("mul_261_cast_fp16")]; tensor add_198_cast_fp16 = add(x = add_186_cast_fp16, y = mul_261_cast_fp16)[name = string("add_198_cast_fp16")]; fp16 const_1764_promoted_to_fp16 = const()[name = string("const_1764_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_71_cast_fp16 = pow(x = add_198_cast_fp16, y = const_1764_promoted_to_fp16)[name = string("pow_71_cast_fp16")]; tensor mean_70_axes_0 = const()[name = string("mean_70_axes_0"), val = tensor([-1])]; bool mean_70_keep_dims_0 = const()[name = string("mean_70_keep_dims_0"), val = bool(true)]; tensor mean_70_cast_fp16 = reduce_mean(axes = mean_70_axes_0, keep_dims = mean_70_keep_dims_0, x = pow_71_cast_fp16)[name = string("mean_70_cast_fp16")]; fp16 const_1767_to_fp16 = const()[name = string("const_1767_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_199_cast_fp16 = add(x = mean_70_cast_fp16, y = const_1767_to_fp16)[name = string("add_199_cast_fp16")]; fp32 rsqrt_70_epsilon_0 = const()[name = string("rsqrt_70_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_70_cast_fp16 = rsqrt(epsilon = rsqrt_70_epsilon_0, x = add_199_cast_fp16)[name = string("rsqrt_70_cast_fp16")]; tensor mul_262_cast_fp16 = mul(x = add_198_cast_fp16, y = rsqrt_70_cast_fp16)[name = string("mul_262_cast_fp16")]; tensor add_200_to_fp16 = const()[name = string("add_200_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250665344)))]; tensor mul_263_cast_fp16 = mul(x = mul_262_cast_fp16, y = add_200_to_fp16)[name = string("mul_263_cast_fp16")]; tensor p_st_0_model_layers_11_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250666944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251551744))))[name = string("p_st_0_model_layers_11_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_81_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_11_mlp_gate_proj_weight_to_fp16_quantized, x = mul_263_cast_fp16)[name = string("linear_81_cast_fp16")]; string gelu_11_mode_0 = const()[name = string("gelu_11_mode_0"), val = string("EXACT")]; tensor gelu_11_cast_fp16 = gelu(mode = gelu_11_mode_0, x = linear_81_cast_fp16)[name = string("gelu_11_cast_fp16")]; tensor p_st_0_model_layers_11_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251554112))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252438912))))[name = string("p_st_0_model_layers_11_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_82_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_11_mlp_up_proj_weight_to_fp16_quantized, x = mul_263_cast_fp16)[name = string("linear_82_cast_fp16")]; tensor mul_264_cast_fp16 = mul(x = gelu_11_cast_fp16, y = linear_82_cast_fp16)[name = string("mul_264_cast_fp16")]; tensor p_st_0_model_layers_11_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252441280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253326080))))[name = string("p_st_0_model_layers_11_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_83_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_11_mlp_down_proj_weight_to_fp16_quantized, x = mul_264_cast_fp16)[name = string("linear_83_cast_fp16")]; fp16 const_1772_promoted_to_fp16 = const()[name = string("const_1772_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_72_cast_fp16 = pow(x = linear_83_cast_fp16, y = const_1772_promoted_to_fp16)[name = string("pow_72_cast_fp16")]; tensor mean_71_axes_0 = const()[name = string("mean_71_axes_0"), val = tensor([-1])]; bool mean_71_keep_dims_0 = const()[name = string("mean_71_keep_dims_0"), val = bool(true)]; tensor mean_71_cast_fp16 = reduce_mean(axes = mean_71_axes_0, keep_dims = mean_71_keep_dims_0, x = pow_72_cast_fp16)[name = string("mean_71_cast_fp16")]; fp16 const_1775_to_fp16 = const()[name = string("const_1775_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_201_cast_fp16 = add(x = mean_71_cast_fp16, y = const_1775_to_fp16)[name = string("add_201_cast_fp16")]; fp32 rsqrt_71_epsilon_0 = const()[name = string("rsqrt_71_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_71_cast_fp16 = rsqrt(epsilon = rsqrt_71_epsilon_0, x = add_201_cast_fp16)[name = string("rsqrt_71_cast_fp16")]; tensor mul_265_cast_fp16 = mul(x = linear_83_cast_fp16, y = rsqrt_71_cast_fp16)[name = string("mul_265_cast_fp16")]; tensor add_202_to_fp16 = const()[name = string("add_202_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253327680)))]; tensor mul_266_cast_fp16 = mul(x = mul_265_cast_fp16, y = add_202_to_fp16)[name = string("mul_266_cast_fp16")]; tensor add_203_cast_fp16 = add(x = add_198_cast_fp16, y = mul_266_cast_fp16)[name = string("add_203_cast_fp16")]; fp16 const_1780_promoted_to_fp16 = const()[name = string("const_1780_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_73_cast_fp16 = pow(x = add_203_cast_fp16, y = const_1780_promoted_to_fp16)[name = string("pow_73_cast_fp16")]; tensor mean_72_axes_0 = const()[name = string("mean_72_axes_0"), val = tensor([-1])]; bool mean_72_keep_dims_0 = const()[name = string("mean_72_keep_dims_0"), val = bool(true)]; tensor mean_72_cast_fp16 = reduce_mean(axes = mean_72_axes_0, keep_dims = mean_72_keep_dims_0, x = pow_73_cast_fp16)[name = string("mean_72_cast_fp16")]; fp16 const_1783_to_fp16 = const()[name = string("const_1783_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_204_cast_fp16 = add(x = mean_72_cast_fp16, y = const_1783_to_fp16)[name = string("add_204_cast_fp16")]; fp32 rsqrt_72_epsilon_0 = const()[name = string("rsqrt_72_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_72_cast_fp16 = rsqrt(epsilon = rsqrt_72_epsilon_0, x = add_204_cast_fp16)[name = string("rsqrt_72_cast_fp16")]; tensor mul_267_cast_fp16 = mul(x = add_203_cast_fp16, y = rsqrt_72_cast_fp16)[name = string("mul_267_cast_fp16")]; tensor add_205_to_fp16 = const()[name = string("add_205_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253329280)))]; tensor mul_268_cast_fp16 = mul(x = mul_267_cast_fp16, y = add_205_to_fp16)[name = string("mul_268_cast_fp16")]; tensor p_st_0_model_layers_12_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253330880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253920768))))[name = string("p_st_0_model_layers_12_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_84_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_12_self_attn_q_proj_weight_to_fp16_quantized, x = mul_268_cast_fp16)[name = string("linear_84_cast_fp16")]; tensor const_1787 = const()[name = string("const_1787"), val = tensor([1, 512, -1, 256])]; tensor view_72_cast_fp16 = reshape(shape = const_1787, x = linear_84_cast_fp16)[name = string("view_72_cast_fp16")]; tensor transpose_84_perm_0 = const()[name = string("transpose_84_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_12_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253922368))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254119040))))[name = string("p_st_0_model_layers_12_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_85_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_12_self_attn_k_proj_weight_to_fp16_quantized, x = mul_268_cast_fp16)[name = string("linear_85_cast_fp16")]; tensor const_1790 = const()[name = string("const_1790"), val = tensor([1, 512, -1, 256])]; tensor view_73_cast_fp16 = reshape(shape = const_1790, x = linear_85_cast_fp16)[name = string("view_73_cast_fp16")]; tensor transpose_85_perm_0 = const()[name = string("transpose_85_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_12_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254119616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254316288))))[name = string("p_st_0_model_layers_12_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_86_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_12_self_attn_v_proj_weight_to_fp16_quantized, x = mul_268_cast_fp16)[name = string("linear_86_cast_fp16")]; tensor const_1793 = const()[name = string("const_1793"), val = tensor([1, 512, -1, 256])]; tensor view_74_cast_fp16 = reshape(shape = const_1793, x = linear_86_cast_fp16)[name = string("view_74_cast_fp16")]; tensor transpose_86_perm_0 = const()[name = string("transpose_86_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1797_promoted_to_fp16 = const()[name = string("const_1797_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_84_cast_fp16 = transpose(perm = transpose_84_perm_0, x = view_72_cast_fp16)[name = string("transpose_47")]; tensor pow_74_cast_fp16 = pow(x = transpose_84_cast_fp16, y = const_1797_promoted_to_fp16)[name = string("pow_74_cast_fp16")]; tensor mean_73_axes_0 = const()[name = string("mean_73_axes_0"), val = tensor([-1])]; bool mean_73_keep_dims_0 = const()[name = string("mean_73_keep_dims_0"), val = bool(true)]; tensor mean_73_cast_fp16 = reduce_mean(axes = mean_73_axes_0, keep_dims = mean_73_keep_dims_0, x = pow_74_cast_fp16)[name = string("mean_73_cast_fp16")]; fp16 const_1800_to_fp16 = const()[name = string("const_1800_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_206_cast_fp16 = add(x = mean_73_cast_fp16, y = const_1800_to_fp16)[name = string("add_206_cast_fp16")]; fp32 rsqrt_73_epsilon_0 = const()[name = string("rsqrt_73_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_73_cast_fp16 = rsqrt(epsilon = rsqrt_73_epsilon_0, x = add_206_cast_fp16)[name = string("rsqrt_73_cast_fp16")]; tensor mul_269_cast_fp16 = mul(x = transpose_84_cast_fp16, y = rsqrt_73_cast_fp16)[name = string("mul_269_cast_fp16")]; tensor add_207_to_fp16 = const()[name = string("add_207_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254316864)))]; tensor mul_270_cast_fp16 = mul(x = mul_269_cast_fp16, y = add_207_to_fp16)[name = string("mul_270_cast_fp16")]; fp16 const_1805_promoted_to_fp16 = const()[name = string("const_1805_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_85_cast_fp16 = transpose(perm = transpose_85_perm_0, x = view_73_cast_fp16)[name = string("transpose_46")]; tensor pow_75_cast_fp16 = pow(x = transpose_85_cast_fp16, y = const_1805_promoted_to_fp16)[name = string("pow_75_cast_fp16")]; tensor mean_74_axes_0 = const()[name = string("mean_74_axes_0"), val = tensor([-1])]; bool mean_74_keep_dims_0 = const()[name = string("mean_74_keep_dims_0"), val = bool(true)]; tensor mean_74_cast_fp16 = reduce_mean(axes = mean_74_axes_0, keep_dims = mean_74_keep_dims_0, x = pow_75_cast_fp16)[name = string("mean_74_cast_fp16")]; fp16 const_1808_to_fp16 = const()[name = string("const_1808_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_208_cast_fp16 = add(x = mean_74_cast_fp16, y = const_1808_to_fp16)[name = string("add_208_cast_fp16")]; fp32 rsqrt_74_epsilon_0 = const()[name = string("rsqrt_74_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_74_cast_fp16 = rsqrt(epsilon = rsqrt_74_epsilon_0, x = add_208_cast_fp16)[name = string("rsqrt_74_cast_fp16")]; tensor mul_271_cast_fp16 = mul(x = transpose_85_cast_fp16, y = rsqrt_74_cast_fp16)[name = string("mul_271_cast_fp16")]; tensor add_209_to_fp16 = const()[name = string("add_209_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254317440)))]; tensor mul_272_cast_fp16 = mul(x = mul_271_cast_fp16, y = add_209_to_fp16)[name = string("mul_272_cast_fp16")]; tensor mul_273_cast_fp16 = mul(x = mul_270_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_273_cast_fp16")]; tensor slice_337_begin_0 = const()[name = string("slice_337_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_337_end_0 = const()[name = string("slice_337_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_337_end_mask_0 = const()[name = string("slice_337_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_337_cast_fp16 = slice_by_index(begin = slice_337_begin_0, end = slice_337_end_0, end_mask = slice_337_end_mask_0, x = mul_270_cast_fp16)[name = string("slice_337_cast_fp16")]; tensor slice_338_begin_0 = const()[name = string("slice_338_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_338_end_0 = const()[name = string("slice_338_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_338_end_mask_0 = const()[name = string("slice_338_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_338_cast_fp16 = slice_by_index(begin = slice_338_begin_0, end = slice_338_end_0, end_mask = slice_338_end_mask_0, x = mul_270_cast_fp16)[name = string("slice_338_cast_fp16")]; fp16 const_1820_promoted_to_fp16 = const()[name = string("const_1820_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_24_cast_fp16 = mul(x = slice_338_cast_fp16, y = const_1820_promoted_to_fp16)[name = string("neg_24_cast_fp16")]; int32 const_1821 = const()[name = string("const_1821"), val = int32(-1)]; bool cat_72_interleave_0 = const()[name = string("cat_72_interleave_0"), val = bool(false)]; tensor cat_72_cast_fp16 = concat(axis = const_1821, interleave = cat_72_interleave_0, values = (neg_24_cast_fp16, slice_337_cast_fp16))[name = string("cat_72_cast_fp16")]; tensor mul_274_cast_fp16 = mul(x = cat_72_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_274_cast_fp16")]; tensor add_210_cast_fp16 = add(x = mul_273_cast_fp16, y = mul_274_cast_fp16)[name = string("add_210_cast_fp16")]; tensor mul_275_cast_fp16 = mul(x = mul_272_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_275_cast_fp16")]; tensor slice_339_begin_0 = const()[name = string("slice_339_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_339_end_0 = const()[name = string("slice_339_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_339_end_mask_0 = const()[name = string("slice_339_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_339_cast_fp16 = slice_by_index(begin = slice_339_begin_0, end = slice_339_end_0, end_mask = slice_339_end_mask_0, x = mul_272_cast_fp16)[name = string("slice_339_cast_fp16")]; tensor slice_340_begin_0 = const()[name = string("slice_340_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_340_end_0 = const()[name = string("slice_340_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_340_end_mask_0 = const()[name = string("slice_340_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_340_cast_fp16 = slice_by_index(begin = slice_340_begin_0, end = slice_340_end_0, end_mask = slice_340_end_mask_0, x = mul_272_cast_fp16)[name = string("slice_340_cast_fp16")]; fp16 const_1828_promoted_to_fp16 = const()[name = string("const_1828_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_25_cast_fp16 = mul(x = slice_340_cast_fp16, y = const_1828_promoted_to_fp16)[name = string("neg_25_cast_fp16")]; int32 const_1829 = const()[name = string("const_1829"), val = int32(-1)]; bool cat_73_interleave_0 = const()[name = string("cat_73_interleave_0"), val = bool(false)]; tensor cat_73_cast_fp16 = concat(axis = const_1829, interleave = cat_73_interleave_0, values = (neg_25_cast_fp16, slice_339_cast_fp16))[name = string("cat_73_cast_fp16")]; tensor mul_276_cast_fp16 = mul(x = cat_73_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_276_cast_fp16")]; tensor add_211_cast_fp16 = add(x = mul_275_cast_fp16, y = mul_276_cast_fp16)[name = string("add_211_cast_fp16")]; int32 const_1830 = const()[name = string("const_1830"), val = int32(-2)]; bool cat_74_interleave_0 = const()[name = string("cat_74_interleave_0"), val = bool(false)]; tensor cat_74_cast_fp16 = concat(axis = const_1830, interleave = cat_74_interleave_0, values = add_211_cast_fp16)[name = string("cat_74_cast_fp16")]; int32 const_1831 = const()[name = string("const_1831"), val = int32(-2)]; bool cat_75_interleave_0 = const()[name = string("cat_75_interleave_0"), val = bool(false)]; tensor transpose_86_cast_fp16 = transpose(perm = transpose_86_perm_0, x = view_74_cast_fp16)[name = string("transpose_45")]; tensor cat_75_cast_fp16 = concat(axis = const_1831, interleave = cat_75_interleave_0, values = transpose_86_cast_fp16)[name = string("cat_75_cast_fp16")]; tensor unsqueeze_127_axes_0 = const()[name = string("unsqueeze_127_axes_0"), val = tensor([2])]; tensor unsqueeze_127_cast_fp16 = expand_dims(axes = unsqueeze_127_axes_0, x = cat_74_cast_fp16)[name = string("unsqueeze_127_cast_fp16")]; tensor expand_50_reps_0 = const()[name = string("expand_50_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_50_cast_fp16 = tile(reps = expand_50_reps_0, x = unsqueeze_127_cast_fp16)[name = string("expand_50_cast_fp16")]; tensor const_1846 = const()[name = string("const_1846"), val = tensor([1, 3, 512, 256])]; tensor view_75_cast_fp16 = reshape(shape = const_1846, x = expand_50_cast_fp16)[name = string("view_75_cast_fp16")]; tensor unsqueeze_128_axes_0 = const()[name = string("unsqueeze_128_axes_0"), val = tensor([2])]; tensor unsqueeze_128_cast_fp16 = expand_dims(axes = unsqueeze_128_axes_0, x = cat_75_cast_fp16)[name = string("unsqueeze_128_cast_fp16")]; tensor expand_51_reps_0 = const()[name = string("expand_51_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_51_cast_fp16 = tile(reps = expand_51_reps_0, x = unsqueeze_128_cast_fp16)[name = string("expand_51_cast_fp16")]; tensor const_1861 = const()[name = string("const_1861"), val = tensor([1, 3, 512, 256])]; tensor view_76_cast_fp16 = reshape(shape = const_1861, x = expand_51_cast_fp16)[name = string("view_76_cast_fp16")]; bool matmul_48_transpose_x_1 = const()[name = string("matmul_48_transpose_x_1"), val = bool(false)]; bool matmul_48_transpose_y_1 = const()[name = string("matmul_48_transpose_y_1"), val = bool(true)]; tensor matmul_48_cast_fp16 = matmul(transpose_x = matmul_48_transpose_x_1, transpose_y = matmul_48_transpose_y_1, x = add_210_cast_fp16, y = view_75_cast_fp16)[name = string("matmul_48_cast_fp16")]; fp16 const_1864_to_fp16 = const()[name = string("const_1864_to_fp16"), val = fp16(0x1p-4)]; tensor mul_277_cast_fp16 = mul(x = matmul_48_cast_fp16, y = const_1864_to_fp16)[name = string("mul_277_cast_fp16")]; tensor add_212_cast_fp16 = add(x = mul_277_cast_fp16, y = expand_cast_fp16)[name = string("add_212_cast_fp16")]; int32 const_1874 = const()[name = string("const_1874"), val = int32(-1)]; tensor softmax_12_cast_fp16 = softmax(axis = const_1874, x = add_212_cast_fp16)[name = string("softmax_12_cast_fp16")]; bool matmul_49_transpose_x_0 = const()[name = string("matmul_49_transpose_x_0"), val = bool(false)]; bool matmul_49_transpose_y_0 = const()[name = string("matmul_49_transpose_y_0"), val = bool(false)]; tensor matmul_49_cast_fp16 = matmul(transpose_x = matmul_49_transpose_x_0, transpose_y = matmul_49_transpose_y_0, x = softmax_12_cast_fp16, y = view_76_cast_fp16)[name = string("matmul_49_cast_fp16")]; tensor transpose_88_perm_0 = const()[name = string("transpose_88_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_1879 = const()[name = string("const_1879"), val = tensor([1, 512, -1])]; tensor transpose_88_cast_fp16 = transpose(perm = transpose_88_perm_0, x = matmul_49_cast_fp16)[name = string("transpose_44")]; tensor view_77_cast_fp16 = reshape(shape = const_1879, x = transpose_88_cast_fp16)[name = string("view_77_cast_fp16")]; tensor p_st_0_model_layers_12_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254318016))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254907904))))[name = string("p_st_0_model_layers_12_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_87_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_12_self_attn_o_proj_weight_to_fp16_quantized, x = view_77_cast_fp16)[name = string("linear_87_cast_fp16")]; fp16 const_1881_promoted_to_fp16 = const()[name = string("const_1881_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_76_cast_fp16 = pow(x = linear_87_cast_fp16, y = const_1881_promoted_to_fp16)[name = string("pow_76_cast_fp16")]; tensor mean_75_axes_0 = const()[name = string("mean_75_axes_0"), val = tensor([-1])]; bool mean_75_keep_dims_0 = const()[name = string("mean_75_keep_dims_0"), val = bool(true)]; tensor mean_75_cast_fp16 = reduce_mean(axes = mean_75_axes_0, keep_dims = mean_75_keep_dims_0, x = pow_76_cast_fp16)[name = string("mean_75_cast_fp16")]; fp16 const_1884_to_fp16 = const()[name = string("const_1884_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_213_cast_fp16 = add(x = mean_75_cast_fp16, y = const_1884_to_fp16)[name = string("add_213_cast_fp16")]; fp32 rsqrt_75_epsilon_0 = const()[name = string("rsqrt_75_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_75_cast_fp16 = rsqrt(epsilon = rsqrt_75_epsilon_0, x = add_213_cast_fp16)[name = string("rsqrt_75_cast_fp16")]; tensor mul_278_cast_fp16 = mul(x = linear_87_cast_fp16, y = rsqrt_75_cast_fp16)[name = string("mul_278_cast_fp16")]; tensor add_214_to_fp16 = const()[name = string("add_214_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254909504)))]; tensor mul_279_cast_fp16 = mul(x = mul_278_cast_fp16, y = add_214_to_fp16)[name = string("mul_279_cast_fp16")]; tensor add_215_cast_fp16 = add(x = add_203_cast_fp16, y = mul_279_cast_fp16)[name = string("add_215_cast_fp16")]; fp16 const_1889_promoted_to_fp16 = const()[name = string("const_1889_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_77_cast_fp16 = pow(x = add_215_cast_fp16, y = const_1889_promoted_to_fp16)[name = string("pow_77_cast_fp16")]; tensor mean_76_axes_0 = const()[name = string("mean_76_axes_0"), val = tensor([-1])]; bool mean_76_keep_dims_0 = const()[name = string("mean_76_keep_dims_0"), val = bool(true)]; tensor mean_76_cast_fp16 = reduce_mean(axes = mean_76_axes_0, keep_dims = mean_76_keep_dims_0, x = pow_77_cast_fp16)[name = string("mean_76_cast_fp16")]; fp16 const_1892_to_fp16 = const()[name = string("const_1892_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_216_cast_fp16 = add(x = mean_76_cast_fp16, y = const_1892_to_fp16)[name = string("add_216_cast_fp16")]; fp32 rsqrt_76_epsilon_0 = const()[name = string("rsqrt_76_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_76_cast_fp16 = rsqrt(epsilon = rsqrt_76_epsilon_0, x = add_216_cast_fp16)[name = string("rsqrt_76_cast_fp16")]; tensor mul_280_cast_fp16 = mul(x = add_215_cast_fp16, y = rsqrt_76_cast_fp16)[name = string("mul_280_cast_fp16")]; tensor add_217_to_fp16 = const()[name = string("add_217_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254911104)))]; tensor mul_281_cast_fp16 = mul(x = mul_280_cast_fp16, y = add_217_to_fp16)[name = string("mul_281_cast_fp16")]; tensor p_st_0_model_layers_12_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(254912704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255797504))))[name = string("p_st_0_model_layers_12_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_88_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_12_mlp_gate_proj_weight_to_fp16_quantized, x = mul_281_cast_fp16)[name = string("linear_88_cast_fp16")]; string gelu_12_mode_0 = const()[name = string("gelu_12_mode_0"), val = string("EXACT")]; tensor gelu_12_cast_fp16 = gelu(mode = gelu_12_mode_0, x = linear_88_cast_fp16)[name = string("gelu_12_cast_fp16")]; tensor p_st_0_model_layers_12_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255799872))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(256684672))))[name = string("p_st_0_model_layers_12_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_89_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_12_mlp_up_proj_weight_to_fp16_quantized, x = mul_281_cast_fp16)[name = string("linear_89_cast_fp16")]; tensor mul_282_cast_fp16 = mul(x = gelu_12_cast_fp16, y = linear_89_cast_fp16)[name = string("mul_282_cast_fp16")]; tensor p_st_0_model_layers_12_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(256687040))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257571840))))[name = string("p_st_0_model_layers_12_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_90_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_12_mlp_down_proj_weight_to_fp16_quantized, x = mul_282_cast_fp16)[name = string("linear_90_cast_fp16")]; fp16 const_1897_promoted_to_fp16 = const()[name = string("const_1897_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_78_cast_fp16 = pow(x = linear_90_cast_fp16, y = const_1897_promoted_to_fp16)[name = string("pow_78_cast_fp16")]; tensor mean_77_axes_0 = const()[name = string("mean_77_axes_0"), val = tensor([-1])]; bool mean_77_keep_dims_0 = const()[name = string("mean_77_keep_dims_0"), val = bool(true)]; tensor mean_77_cast_fp16 = reduce_mean(axes = mean_77_axes_0, keep_dims = mean_77_keep_dims_0, x = pow_78_cast_fp16)[name = string("mean_77_cast_fp16")]; fp16 const_1900_to_fp16 = const()[name = string("const_1900_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_218_cast_fp16 = add(x = mean_77_cast_fp16, y = const_1900_to_fp16)[name = string("add_218_cast_fp16")]; fp32 rsqrt_77_epsilon_0 = const()[name = string("rsqrt_77_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_77_cast_fp16 = rsqrt(epsilon = rsqrt_77_epsilon_0, x = add_218_cast_fp16)[name = string("rsqrt_77_cast_fp16")]; tensor mul_283_cast_fp16 = mul(x = linear_90_cast_fp16, y = rsqrt_77_cast_fp16)[name = string("mul_283_cast_fp16")]; tensor add_219_to_fp16 = const()[name = string("add_219_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257573440)))]; tensor mul_284_cast_fp16 = mul(x = mul_283_cast_fp16, y = add_219_to_fp16)[name = string("mul_284_cast_fp16")]; tensor add_220_cast_fp16 = add(x = add_215_cast_fp16, y = mul_284_cast_fp16)[name = string("add_220_cast_fp16")]; fp16 const_1905_promoted_to_fp16 = const()[name = string("const_1905_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_79_cast_fp16 = pow(x = add_220_cast_fp16, y = const_1905_promoted_to_fp16)[name = string("pow_79_cast_fp16")]; tensor mean_78_axes_0 = const()[name = string("mean_78_axes_0"), val = tensor([-1])]; bool mean_78_keep_dims_0 = const()[name = string("mean_78_keep_dims_0"), val = bool(true)]; tensor mean_78_cast_fp16 = reduce_mean(axes = mean_78_axes_0, keep_dims = mean_78_keep_dims_0, x = pow_79_cast_fp16)[name = string("mean_78_cast_fp16")]; fp16 const_1908_to_fp16 = const()[name = string("const_1908_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_221_cast_fp16 = add(x = mean_78_cast_fp16, y = const_1908_to_fp16)[name = string("add_221_cast_fp16")]; fp32 rsqrt_78_epsilon_0 = const()[name = string("rsqrt_78_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_78_cast_fp16 = rsqrt(epsilon = rsqrt_78_epsilon_0, x = add_221_cast_fp16)[name = string("rsqrt_78_cast_fp16")]; tensor mul_285_cast_fp16 = mul(x = add_220_cast_fp16, y = rsqrt_78_cast_fp16)[name = string("mul_285_cast_fp16")]; tensor add_222_to_fp16 = const()[name = string("add_222_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257575040)))]; tensor mul_286_cast_fp16 = mul(x = mul_285_cast_fp16, y = add_222_to_fp16)[name = string("mul_286_cast_fp16")]; tensor p_st_0_model_layers_13_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257576640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258166528))))[name = string("p_st_0_model_layers_13_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_91_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_13_self_attn_q_proj_weight_to_fp16_quantized, x = mul_286_cast_fp16)[name = string("linear_91_cast_fp16")]; tensor const_1912 = const()[name = string("const_1912"), val = tensor([1, 512, -1, 256])]; tensor view_78_cast_fp16 = reshape(shape = const_1912, x = linear_91_cast_fp16)[name = string("view_78_cast_fp16")]; tensor transpose_89_perm_0 = const()[name = string("transpose_89_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_13_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258168128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258364800))))[name = string("p_st_0_model_layers_13_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_92_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_13_self_attn_k_proj_weight_to_fp16_quantized, x = mul_286_cast_fp16)[name = string("linear_92_cast_fp16")]; tensor const_1915 = const()[name = string("const_1915"), val = tensor([1, 512, -1, 256])]; tensor view_79_cast_fp16 = reshape(shape = const_1915, x = linear_92_cast_fp16)[name = string("view_79_cast_fp16")]; tensor transpose_90_perm_0 = const()[name = string("transpose_90_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_13_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258562048))))[name = string("p_st_0_model_layers_13_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_93_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_13_self_attn_v_proj_weight_to_fp16_quantized, x = mul_286_cast_fp16)[name = string("linear_93_cast_fp16")]; tensor const_1918 = const()[name = string("const_1918"), val = tensor([1, 512, -1, 256])]; tensor view_80_cast_fp16 = reshape(shape = const_1918, x = linear_93_cast_fp16)[name = string("view_80_cast_fp16")]; tensor transpose_91_perm_0 = const()[name = string("transpose_91_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_1922_promoted_to_fp16 = const()[name = string("const_1922_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_89_cast_fp16 = transpose(perm = transpose_89_perm_0, x = view_78_cast_fp16)[name = string("transpose_43")]; tensor pow_80_cast_fp16 = pow(x = transpose_89_cast_fp16, y = const_1922_promoted_to_fp16)[name = string("pow_80_cast_fp16")]; tensor mean_79_axes_0 = const()[name = string("mean_79_axes_0"), val = tensor([-1])]; bool mean_79_keep_dims_0 = const()[name = string("mean_79_keep_dims_0"), val = bool(true)]; tensor mean_79_cast_fp16 = reduce_mean(axes = mean_79_axes_0, keep_dims = mean_79_keep_dims_0, x = pow_80_cast_fp16)[name = string("mean_79_cast_fp16")]; fp16 const_1925_to_fp16 = const()[name = string("const_1925_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_223_cast_fp16 = add(x = mean_79_cast_fp16, y = const_1925_to_fp16)[name = string("add_223_cast_fp16")]; fp32 rsqrt_79_epsilon_0 = const()[name = string("rsqrt_79_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_79_cast_fp16 = rsqrt(epsilon = rsqrt_79_epsilon_0, x = add_223_cast_fp16)[name = string("rsqrt_79_cast_fp16")]; tensor mul_287_cast_fp16 = mul(x = transpose_89_cast_fp16, y = rsqrt_79_cast_fp16)[name = string("mul_287_cast_fp16")]; tensor add_224_to_fp16 = const()[name = string("add_224_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258562624)))]; tensor mul_288_cast_fp16 = mul(x = mul_287_cast_fp16, y = add_224_to_fp16)[name = string("mul_288_cast_fp16")]; fp16 const_1930_promoted_to_fp16 = const()[name = string("const_1930_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_90_cast_fp16 = transpose(perm = transpose_90_perm_0, x = view_79_cast_fp16)[name = string("transpose_42")]; tensor pow_81_cast_fp16 = pow(x = transpose_90_cast_fp16, y = const_1930_promoted_to_fp16)[name = string("pow_81_cast_fp16")]; tensor mean_80_axes_0 = const()[name = string("mean_80_axes_0"), val = tensor([-1])]; bool mean_80_keep_dims_0 = const()[name = string("mean_80_keep_dims_0"), val = bool(true)]; tensor mean_80_cast_fp16 = reduce_mean(axes = mean_80_axes_0, keep_dims = mean_80_keep_dims_0, x = pow_81_cast_fp16)[name = string("mean_80_cast_fp16")]; fp16 const_1933_to_fp16 = const()[name = string("const_1933_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_225_cast_fp16 = add(x = mean_80_cast_fp16, y = const_1933_to_fp16)[name = string("add_225_cast_fp16")]; fp32 rsqrt_80_epsilon_0 = const()[name = string("rsqrt_80_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_80_cast_fp16 = rsqrt(epsilon = rsqrt_80_epsilon_0, x = add_225_cast_fp16)[name = string("rsqrt_80_cast_fp16")]; tensor mul_289_cast_fp16 = mul(x = transpose_90_cast_fp16, y = rsqrt_80_cast_fp16)[name = string("mul_289_cast_fp16")]; tensor add_226_to_fp16 = const()[name = string("add_226_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258563200)))]; tensor mul_290_cast_fp16 = mul(x = mul_289_cast_fp16, y = add_226_to_fp16)[name = string("mul_290_cast_fp16")]; tensor mul_291_cast_fp16 = mul(x = mul_288_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_291_cast_fp16")]; tensor slice_360_begin_0 = const()[name = string("slice_360_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_360_end_0 = const()[name = string("slice_360_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_360_end_mask_0 = const()[name = string("slice_360_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_360_cast_fp16 = slice_by_index(begin = slice_360_begin_0, end = slice_360_end_0, end_mask = slice_360_end_mask_0, x = mul_288_cast_fp16)[name = string("slice_360_cast_fp16")]; tensor slice_361_begin_0 = const()[name = string("slice_361_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_361_end_0 = const()[name = string("slice_361_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_361_end_mask_0 = const()[name = string("slice_361_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_361_cast_fp16 = slice_by_index(begin = slice_361_begin_0, end = slice_361_end_0, end_mask = slice_361_end_mask_0, x = mul_288_cast_fp16)[name = string("slice_361_cast_fp16")]; fp16 const_1945_promoted_to_fp16 = const()[name = string("const_1945_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_26_cast_fp16 = mul(x = slice_361_cast_fp16, y = const_1945_promoted_to_fp16)[name = string("neg_26_cast_fp16")]; int32 const_1946 = const()[name = string("const_1946"), val = int32(-1)]; bool cat_76_interleave_0 = const()[name = string("cat_76_interleave_0"), val = bool(false)]; tensor cat_76_cast_fp16 = concat(axis = const_1946, interleave = cat_76_interleave_0, values = (neg_26_cast_fp16, slice_360_cast_fp16))[name = string("cat_76_cast_fp16")]; tensor mul_292_cast_fp16 = mul(x = cat_76_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_292_cast_fp16")]; tensor add_227_cast_fp16 = add(x = mul_291_cast_fp16, y = mul_292_cast_fp16)[name = string("add_227_cast_fp16")]; tensor mul_293_cast_fp16 = mul(x = mul_290_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_293_cast_fp16")]; tensor slice_362_begin_0 = const()[name = string("slice_362_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_362_end_0 = const()[name = string("slice_362_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_362_end_mask_0 = const()[name = string("slice_362_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_362_cast_fp16 = slice_by_index(begin = slice_362_begin_0, end = slice_362_end_0, end_mask = slice_362_end_mask_0, x = mul_290_cast_fp16)[name = string("slice_362_cast_fp16")]; tensor slice_363_begin_0 = const()[name = string("slice_363_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_363_end_0 = const()[name = string("slice_363_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_363_end_mask_0 = const()[name = string("slice_363_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_363_cast_fp16 = slice_by_index(begin = slice_363_begin_0, end = slice_363_end_0, end_mask = slice_363_end_mask_0, x = mul_290_cast_fp16)[name = string("slice_363_cast_fp16")]; fp16 const_1953_promoted_to_fp16 = const()[name = string("const_1953_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_27_cast_fp16 = mul(x = slice_363_cast_fp16, y = const_1953_promoted_to_fp16)[name = string("neg_27_cast_fp16")]; int32 const_1954 = const()[name = string("const_1954"), val = int32(-1)]; bool cat_77_interleave_0 = const()[name = string("cat_77_interleave_0"), val = bool(false)]; tensor cat_77_cast_fp16 = concat(axis = const_1954, interleave = cat_77_interleave_0, values = (neg_27_cast_fp16, slice_362_cast_fp16))[name = string("cat_77_cast_fp16")]; tensor mul_294_cast_fp16 = mul(x = cat_77_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_294_cast_fp16")]; tensor add_228_cast_fp16 = add(x = mul_293_cast_fp16, y = mul_294_cast_fp16)[name = string("add_228_cast_fp16")]; int32 const_1955 = const()[name = string("const_1955"), val = int32(-2)]; bool cat_78_interleave_0 = const()[name = string("cat_78_interleave_0"), val = bool(false)]; tensor cat_78_cast_fp16 = concat(axis = const_1955, interleave = cat_78_interleave_0, values = add_228_cast_fp16)[name = string("cat_78_cast_fp16")]; int32 const_1956 = const()[name = string("const_1956"), val = int32(-2)]; bool cat_79_interleave_0 = const()[name = string("cat_79_interleave_0"), val = bool(false)]; tensor transpose_91_cast_fp16 = transpose(perm = transpose_91_perm_0, x = view_80_cast_fp16)[name = string("transpose_41")]; tensor cat_79_cast_fp16 = concat(axis = const_1956, interleave = cat_79_interleave_0, values = transpose_91_cast_fp16)[name = string("cat_79_cast_fp16")]; tensor unsqueeze_131_axes_0 = const()[name = string("unsqueeze_131_axes_0"), val = tensor([2])]; tensor unsqueeze_131_cast_fp16 = expand_dims(axes = unsqueeze_131_axes_0, x = cat_78_cast_fp16)[name = string("unsqueeze_131_cast_fp16")]; tensor expand_52_reps_0 = const()[name = string("expand_52_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_52_cast_fp16 = tile(reps = expand_52_reps_0, x = unsqueeze_131_cast_fp16)[name = string("expand_52_cast_fp16")]; tensor const_1971 = const()[name = string("const_1971"), val = tensor([1, 3, 512, 256])]; tensor view_81_cast_fp16 = reshape(shape = const_1971, x = expand_52_cast_fp16)[name = string("view_81_cast_fp16")]; tensor unsqueeze_132_axes_0 = const()[name = string("unsqueeze_132_axes_0"), val = tensor([2])]; tensor unsqueeze_132_cast_fp16 = expand_dims(axes = unsqueeze_132_axes_0, x = cat_79_cast_fp16)[name = string("unsqueeze_132_cast_fp16")]; tensor expand_53_reps_0 = const()[name = string("expand_53_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_53_cast_fp16 = tile(reps = expand_53_reps_0, x = unsqueeze_132_cast_fp16)[name = string("expand_53_cast_fp16")]; tensor const_1986 = const()[name = string("const_1986"), val = tensor([1, 3, 512, 256])]; tensor view_82_cast_fp16 = reshape(shape = const_1986, x = expand_53_cast_fp16)[name = string("view_82_cast_fp16")]; bool matmul_50_transpose_x_1 = const()[name = string("matmul_50_transpose_x_1"), val = bool(false)]; bool matmul_50_transpose_y_1 = const()[name = string("matmul_50_transpose_y_1"), val = bool(true)]; tensor matmul_50_cast_fp16 = matmul(transpose_x = matmul_50_transpose_x_1, transpose_y = matmul_50_transpose_y_1, x = add_227_cast_fp16, y = view_81_cast_fp16)[name = string("matmul_50_cast_fp16")]; fp16 const_1989_to_fp16 = const()[name = string("const_1989_to_fp16"), val = fp16(0x1p-4)]; tensor mul_295_cast_fp16 = mul(x = matmul_50_cast_fp16, y = const_1989_to_fp16)[name = string("mul_295_cast_fp16")]; tensor add_229_cast_fp16 = add(x = mul_295_cast_fp16, y = expand_cast_fp16)[name = string("add_229_cast_fp16")]; int32 const_1999 = const()[name = string("const_1999"), val = int32(-1)]; tensor softmax_13_cast_fp16 = softmax(axis = const_1999, x = add_229_cast_fp16)[name = string("softmax_13_cast_fp16")]; bool matmul_51_transpose_x_0 = const()[name = string("matmul_51_transpose_x_0"), val = bool(false)]; bool matmul_51_transpose_y_0 = const()[name = string("matmul_51_transpose_y_0"), val = bool(false)]; tensor matmul_51_cast_fp16 = matmul(transpose_x = matmul_51_transpose_x_0, transpose_y = matmul_51_transpose_y_0, x = softmax_13_cast_fp16, y = view_82_cast_fp16)[name = string("matmul_51_cast_fp16")]; tensor transpose_93_perm_0 = const()[name = string("transpose_93_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2004 = const()[name = string("const_2004"), val = tensor([1, 512, -1])]; tensor transpose_93_cast_fp16 = transpose(perm = transpose_93_perm_0, x = matmul_51_cast_fp16)[name = string("transpose_40")]; tensor view_83_cast_fp16 = reshape(shape = const_2004, x = transpose_93_cast_fp16)[name = string("view_83_cast_fp16")]; tensor p_st_0_model_layers_13_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258563776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(259153664))))[name = string("p_st_0_model_layers_13_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_94_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_13_self_attn_o_proj_weight_to_fp16_quantized, x = view_83_cast_fp16)[name = string("linear_94_cast_fp16")]; fp16 const_2006_promoted_to_fp16 = const()[name = string("const_2006_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_82_cast_fp16 = pow(x = linear_94_cast_fp16, y = const_2006_promoted_to_fp16)[name = string("pow_82_cast_fp16")]; tensor mean_81_axes_0 = const()[name = string("mean_81_axes_0"), val = tensor([-1])]; bool mean_81_keep_dims_0 = const()[name = string("mean_81_keep_dims_0"), val = bool(true)]; tensor mean_81_cast_fp16 = reduce_mean(axes = mean_81_axes_0, keep_dims = mean_81_keep_dims_0, x = pow_82_cast_fp16)[name = string("mean_81_cast_fp16")]; fp16 const_2009_to_fp16 = const()[name = string("const_2009_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_230_cast_fp16 = add(x = mean_81_cast_fp16, y = const_2009_to_fp16)[name = string("add_230_cast_fp16")]; fp32 rsqrt_81_epsilon_0 = const()[name = string("rsqrt_81_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_81_cast_fp16 = rsqrt(epsilon = rsqrt_81_epsilon_0, x = add_230_cast_fp16)[name = string("rsqrt_81_cast_fp16")]; tensor mul_296_cast_fp16 = mul(x = linear_94_cast_fp16, y = rsqrt_81_cast_fp16)[name = string("mul_296_cast_fp16")]; tensor add_231_to_fp16 = const()[name = string("add_231_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(259155264)))]; tensor mul_297_cast_fp16 = mul(x = mul_296_cast_fp16, y = add_231_to_fp16)[name = string("mul_297_cast_fp16")]; tensor add_232_cast_fp16 = add(x = add_220_cast_fp16, y = mul_297_cast_fp16)[name = string("add_232_cast_fp16")]; fp16 const_2014_promoted_to_fp16 = const()[name = string("const_2014_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_83_cast_fp16 = pow(x = add_232_cast_fp16, y = const_2014_promoted_to_fp16)[name = string("pow_83_cast_fp16")]; tensor mean_82_axes_0 = const()[name = string("mean_82_axes_0"), val = tensor([-1])]; bool mean_82_keep_dims_0 = const()[name = string("mean_82_keep_dims_0"), val = bool(true)]; tensor mean_82_cast_fp16 = reduce_mean(axes = mean_82_axes_0, keep_dims = mean_82_keep_dims_0, x = pow_83_cast_fp16)[name = string("mean_82_cast_fp16")]; fp16 const_2017_to_fp16 = const()[name = string("const_2017_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_233_cast_fp16 = add(x = mean_82_cast_fp16, y = const_2017_to_fp16)[name = string("add_233_cast_fp16")]; fp32 rsqrt_82_epsilon_0 = const()[name = string("rsqrt_82_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_82_cast_fp16 = rsqrt(epsilon = rsqrt_82_epsilon_0, x = add_233_cast_fp16)[name = string("rsqrt_82_cast_fp16")]; tensor mul_298_cast_fp16 = mul(x = add_232_cast_fp16, y = rsqrt_82_cast_fp16)[name = string("mul_298_cast_fp16")]; tensor add_234_to_fp16 = const()[name = string("add_234_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(259156864)))]; tensor mul_299_cast_fp16 = mul(x = mul_298_cast_fp16, y = add_234_to_fp16)[name = string("mul_299_cast_fp16")]; tensor p_st_0_model_layers_13_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(259158464))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260043264))))[name = string("p_st_0_model_layers_13_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_95_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_13_mlp_gate_proj_weight_to_fp16_quantized, x = mul_299_cast_fp16)[name = string("linear_95_cast_fp16")]; string gelu_13_mode_0 = const()[name = string("gelu_13_mode_0"), val = string("EXACT")]; tensor gelu_13_cast_fp16 = gelu(mode = gelu_13_mode_0, x = linear_95_cast_fp16)[name = string("gelu_13_cast_fp16")]; tensor p_st_0_model_layers_13_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260045632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260930432))))[name = string("p_st_0_model_layers_13_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_96_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_13_mlp_up_proj_weight_to_fp16_quantized, x = mul_299_cast_fp16)[name = string("linear_96_cast_fp16")]; tensor mul_300_cast_fp16 = mul(x = gelu_13_cast_fp16, y = linear_96_cast_fp16)[name = string("mul_300_cast_fp16")]; tensor p_st_0_model_layers_13_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260932800))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261817600))))[name = string("p_st_0_model_layers_13_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_97_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_13_mlp_down_proj_weight_to_fp16_quantized, x = mul_300_cast_fp16)[name = string("linear_97_cast_fp16")]; fp16 const_2022_promoted_to_fp16 = const()[name = string("const_2022_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_84_cast_fp16 = pow(x = linear_97_cast_fp16, y = const_2022_promoted_to_fp16)[name = string("pow_84_cast_fp16")]; tensor mean_83_axes_0 = const()[name = string("mean_83_axes_0"), val = tensor([-1])]; bool mean_83_keep_dims_0 = const()[name = string("mean_83_keep_dims_0"), val = bool(true)]; tensor mean_83_cast_fp16 = reduce_mean(axes = mean_83_axes_0, keep_dims = mean_83_keep_dims_0, x = pow_84_cast_fp16)[name = string("mean_83_cast_fp16")]; fp16 const_2025_to_fp16 = const()[name = string("const_2025_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_235_cast_fp16 = add(x = mean_83_cast_fp16, y = const_2025_to_fp16)[name = string("add_235_cast_fp16")]; fp32 rsqrt_83_epsilon_0 = const()[name = string("rsqrt_83_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_83_cast_fp16 = rsqrt(epsilon = rsqrt_83_epsilon_0, x = add_235_cast_fp16)[name = string("rsqrt_83_cast_fp16")]; tensor mul_301_cast_fp16 = mul(x = linear_97_cast_fp16, y = rsqrt_83_cast_fp16)[name = string("mul_301_cast_fp16")]; tensor add_236_to_fp16 = const()[name = string("add_236_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261819200)))]; tensor mul_302_cast_fp16 = mul(x = mul_301_cast_fp16, y = add_236_to_fp16)[name = string("mul_302_cast_fp16")]; tensor add_237_cast_fp16 = add(x = add_232_cast_fp16, y = mul_302_cast_fp16)[name = string("add_237_cast_fp16")]; fp16 const_2030_promoted_to_fp16 = const()[name = string("const_2030_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_85_cast_fp16 = pow(x = add_237_cast_fp16, y = const_2030_promoted_to_fp16)[name = string("pow_85_cast_fp16")]; tensor mean_84_axes_0 = const()[name = string("mean_84_axes_0"), val = tensor([-1])]; bool mean_84_keep_dims_0 = const()[name = string("mean_84_keep_dims_0"), val = bool(true)]; tensor mean_84_cast_fp16 = reduce_mean(axes = mean_84_axes_0, keep_dims = mean_84_keep_dims_0, x = pow_85_cast_fp16)[name = string("mean_84_cast_fp16")]; fp16 const_2033_to_fp16 = const()[name = string("const_2033_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_238_cast_fp16 = add(x = mean_84_cast_fp16, y = const_2033_to_fp16)[name = string("add_238_cast_fp16")]; fp32 rsqrt_84_epsilon_0 = const()[name = string("rsqrt_84_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_84_cast_fp16 = rsqrt(epsilon = rsqrt_84_epsilon_0, x = add_238_cast_fp16)[name = string("rsqrt_84_cast_fp16")]; tensor mul_303_cast_fp16 = mul(x = add_237_cast_fp16, y = rsqrt_84_cast_fp16)[name = string("mul_303_cast_fp16")]; tensor add_239_to_fp16 = const()[name = string("add_239_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261820800)))]; tensor mul_304_cast_fp16 = mul(x = mul_303_cast_fp16, y = add_239_to_fp16)[name = string("mul_304_cast_fp16")]; tensor p_st_0_model_layers_14_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261822400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262412288))))[name = string("p_st_0_model_layers_14_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_98_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_14_self_attn_q_proj_weight_to_fp16_quantized, x = mul_304_cast_fp16)[name = string("linear_98_cast_fp16")]; tensor const_2037 = const()[name = string("const_2037"), val = tensor([1, 512, -1, 256])]; tensor view_84_cast_fp16 = reshape(shape = const_2037, x = linear_98_cast_fp16)[name = string("view_84_cast_fp16")]; tensor transpose_94_perm_0 = const()[name = string("transpose_94_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_14_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262413888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262610560))))[name = string("p_st_0_model_layers_14_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_99_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_14_self_attn_k_proj_weight_to_fp16_quantized, x = mul_304_cast_fp16)[name = string("linear_99_cast_fp16")]; tensor const_2040 = const()[name = string("const_2040"), val = tensor([1, 512, -1, 256])]; tensor view_85_cast_fp16 = reshape(shape = const_2040, x = linear_99_cast_fp16)[name = string("view_85_cast_fp16")]; tensor transpose_95_perm_0 = const()[name = string("transpose_95_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_14_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262611136))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262807808))))[name = string("p_st_0_model_layers_14_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_100_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_14_self_attn_v_proj_weight_to_fp16_quantized, x = mul_304_cast_fp16)[name = string("linear_100_cast_fp16")]; tensor const_2043 = const()[name = string("const_2043"), val = tensor([1, 512, -1, 256])]; tensor view_86_cast_fp16 = reshape(shape = const_2043, x = linear_100_cast_fp16)[name = string("view_86_cast_fp16")]; tensor transpose_96_perm_0 = const()[name = string("transpose_96_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2047_promoted_to_fp16 = const()[name = string("const_2047_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_94_cast_fp16 = transpose(perm = transpose_94_perm_0, x = view_84_cast_fp16)[name = string("transpose_39")]; tensor pow_86_cast_fp16 = pow(x = transpose_94_cast_fp16, y = const_2047_promoted_to_fp16)[name = string("pow_86_cast_fp16")]; tensor mean_85_axes_0 = const()[name = string("mean_85_axes_0"), val = tensor([-1])]; bool mean_85_keep_dims_0 = const()[name = string("mean_85_keep_dims_0"), val = bool(true)]; tensor mean_85_cast_fp16 = reduce_mean(axes = mean_85_axes_0, keep_dims = mean_85_keep_dims_0, x = pow_86_cast_fp16)[name = string("mean_85_cast_fp16")]; fp16 const_2050_to_fp16 = const()[name = string("const_2050_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_240_cast_fp16 = add(x = mean_85_cast_fp16, y = const_2050_to_fp16)[name = string("add_240_cast_fp16")]; fp32 rsqrt_85_epsilon_0 = const()[name = string("rsqrt_85_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_85_cast_fp16 = rsqrt(epsilon = rsqrt_85_epsilon_0, x = add_240_cast_fp16)[name = string("rsqrt_85_cast_fp16")]; tensor mul_305_cast_fp16 = mul(x = transpose_94_cast_fp16, y = rsqrt_85_cast_fp16)[name = string("mul_305_cast_fp16")]; tensor add_241_to_fp16 = const()[name = string("add_241_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262808384)))]; tensor mul_306_cast_fp16 = mul(x = mul_305_cast_fp16, y = add_241_to_fp16)[name = string("mul_306_cast_fp16")]; fp16 const_2055_promoted_to_fp16 = const()[name = string("const_2055_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_95_cast_fp16 = transpose(perm = transpose_95_perm_0, x = view_85_cast_fp16)[name = string("transpose_38")]; tensor pow_87_cast_fp16 = pow(x = transpose_95_cast_fp16, y = const_2055_promoted_to_fp16)[name = string("pow_87_cast_fp16")]; tensor mean_86_axes_0 = const()[name = string("mean_86_axes_0"), val = tensor([-1])]; bool mean_86_keep_dims_0 = const()[name = string("mean_86_keep_dims_0"), val = bool(true)]; tensor mean_86_cast_fp16 = reduce_mean(axes = mean_86_axes_0, keep_dims = mean_86_keep_dims_0, x = pow_87_cast_fp16)[name = string("mean_86_cast_fp16")]; fp16 const_2058_to_fp16 = const()[name = string("const_2058_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_242_cast_fp16 = add(x = mean_86_cast_fp16, y = const_2058_to_fp16)[name = string("add_242_cast_fp16")]; fp32 rsqrt_86_epsilon_0 = const()[name = string("rsqrt_86_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_86_cast_fp16 = rsqrt(epsilon = rsqrt_86_epsilon_0, x = add_242_cast_fp16)[name = string("rsqrt_86_cast_fp16")]; tensor mul_307_cast_fp16 = mul(x = transpose_95_cast_fp16, y = rsqrt_86_cast_fp16)[name = string("mul_307_cast_fp16")]; tensor add_243_to_fp16 = const()[name = string("add_243_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262808960)))]; tensor mul_308_cast_fp16 = mul(x = mul_307_cast_fp16, y = add_243_to_fp16)[name = string("mul_308_cast_fp16")]; tensor mul_309_cast_fp16 = mul(x = mul_306_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_309_cast_fp16")]; tensor slice_383_begin_0 = const()[name = string("slice_383_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_383_end_0 = const()[name = string("slice_383_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_383_end_mask_0 = const()[name = string("slice_383_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_383_cast_fp16 = slice_by_index(begin = slice_383_begin_0, end = slice_383_end_0, end_mask = slice_383_end_mask_0, x = mul_306_cast_fp16)[name = string("slice_383_cast_fp16")]; tensor slice_384_begin_0 = const()[name = string("slice_384_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_384_end_0 = const()[name = string("slice_384_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_384_end_mask_0 = const()[name = string("slice_384_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_384_cast_fp16 = slice_by_index(begin = slice_384_begin_0, end = slice_384_end_0, end_mask = slice_384_end_mask_0, x = mul_306_cast_fp16)[name = string("slice_384_cast_fp16")]; fp16 const_2070_promoted_to_fp16 = const()[name = string("const_2070_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_28_cast_fp16 = mul(x = slice_384_cast_fp16, y = const_2070_promoted_to_fp16)[name = string("neg_28_cast_fp16")]; int32 const_2071 = const()[name = string("const_2071"), val = int32(-1)]; bool cat_80_interleave_0 = const()[name = string("cat_80_interleave_0"), val = bool(false)]; tensor cat_80_cast_fp16 = concat(axis = const_2071, interleave = cat_80_interleave_0, values = (neg_28_cast_fp16, slice_383_cast_fp16))[name = string("cat_80_cast_fp16")]; tensor mul_310_cast_fp16 = mul(x = cat_80_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_310_cast_fp16")]; tensor add_244_cast_fp16 = add(x = mul_309_cast_fp16, y = mul_310_cast_fp16)[name = string("add_244_cast_fp16")]; tensor mul_311_cast_fp16 = mul(x = mul_308_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_311_cast_fp16")]; tensor slice_385_begin_0 = const()[name = string("slice_385_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_385_end_0 = const()[name = string("slice_385_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_385_end_mask_0 = const()[name = string("slice_385_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_385_cast_fp16 = slice_by_index(begin = slice_385_begin_0, end = slice_385_end_0, end_mask = slice_385_end_mask_0, x = mul_308_cast_fp16)[name = string("slice_385_cast_fp16")]; tensor slice_386_begin_0 = const()[name = string("slice_386_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_386_end_0 = const()[name = string("slice_386_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_386_end_mask_0 = const()[name = string("slice_386_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_386_cast_fp16 = slice_by_index(begin = slice_386_begin_0, end = slice_386_end_0, end_mask = slice_386_end_mask_0, x = mul_308_cast_fp16)[name = string("slice_386_cast_fp16")]; fp16 const_2078_promoted_to_fp16 = const()[name = string("const_2078_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_29_cast_fp16 = mul(x = slice_386_cast_fp16, y = const_2078_promoted_to_fp16)[name = string("neg_29_cast_fp16")]; int32 const_2079 = const()[name = string("const_2079"), val = int32(-1)]; bool cat_81_interleave_0 = const()[name = string("cat_81_interleave_0"), val = bool(false)]; tensor cat_81_cast_fp16 = concat(axis = const_2079, interleave = cat_81_interleave_0, values = (neg_29_cast_fp16, slice_385_cast_fp16))[name = string("cat_81_cast_fp16")]; tensor mul_312_cast_fp16 = mul(x = cat_81_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_312_cast_fp16")]; tensor add_245_cast_fp16 = add(x = mul_311_cast_fp16, y = mul_312_cast_fp16)[name = string("add_245_cast_fp16")]; int32 const_2080 = const()[name = string("const_2080"), val = int32(-2)]; bool cat_82_interleave_0 = const()[name = string("cat_82_interleave_0"), val = bool(false)]; tensor cat_82_cast_fp16 = concat(axis = const_2080, interleave = cat_82_interleave_0, values = add_245_cast_fp16)[name = string("cat_82_cast_fp16")]; int32 const_2081 = const()[name = string("const_2081"), val = int32(-2)]; bool cat_83_interleave_0 = const()[name = string("cat_83_interleave_0"), val = bool(false)]; tensor transpose_96_cast_fp16 = transpose(perm = transpose_96_perm_0, x = view_86_cast_fp16)[name = string("transpose_37")]; tensor cat_83_cast_fp16 = concat(axis = const_2081, interleave = cat_83_interleave_0, values = transpose_96_cast_fp16)[name = string("cat_83_cast_fp16")]; tensor unsqueeze_135_axes_0 = const()[name = string("unsqueeze_135_axes_0"), val = tensor([2])]; tensor unsqueeze_135_cast_fp16 = expand_dims(axes = unsqueeze_135_axes_0, x = cat_82_cast_fp16)[name = string("unsqueeze_135_cast_fp16")]; tensor expand_54_reps_0 = const()[name = string("expand_54_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_54_cast_fp16 = tile(reps = expand_54_reps_0, x = unsqueeze_135_cast_fp16)[name = string("expand_54_cast_fp16")]; tensor const_2096 = const()[name = string("const_2096"), val = tensor([1, 3, 512, 256])]; tensor view_87_cast_fp16 = reshape(shape = const_2096, x = expand_54_cast_fp16)[name = string("view_87_cast_fp16")]; tensor unsqueeze_136_axes_0 = const()[name = string("unsqueeze_136_axes_0"), val = tensor([2])]; tensor unsqueeze_136_cast_fp16 = expand_dims(axes = unsqueeze_136_axes_0, x = cat_83_cast_fp16)[name = string("unsqueeze_136_cast_fp16")]; tensor expand_55_reps_0 = const()[name = string("expand_55_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_55_cast_fp16 = tile(reps = expand_55_reps_0, x = unsqueeze_136_cast_fp16)[name = string("expand_55_cast_fp16")]; tensor const_2111 = const()[name = string("const_2111"), val = tensor([1, 3, 512, 256])]; tensor view_88_cast_fp16 = reshape(shape = const_2111, x = expand_55_cast_fp16)[name = string("view_88_cast_fp16")]; bool matmul_52_transpose_x_1 = const()[name = string("matmul_52_transpose_x_1"), val = bool(false)]; bool matmul_52_transpose_y_1 = const()[name = string("matmul_52_transpose_y_1"), val = bool(true)]; tensor matmul_52_cast_fp16 = matmul(transpose_x = matmul_52_transpose_x_1, transpose_y = matmul_52_transpose_y_1, x = add_244_cast_fp16, y = view_87_cast_fp16)[name = string("matmul_52_cast_fp16")]; fp16 const_2114_to_fp16 = const()[name = string("const_2114_to_fp16"), val = fp16(0x1p-4)]; tensor mul_313_cast_fp16 = mul(x = matmul_52_cast_fp16, y = const_2114_to_fp16)[name = string("mul_313_cast_fp16")]; tensor add_246_cast_fp16 = add(x = mul_313_cast_fp16, y = expand_cast_fp16)[name = string("add_246_cast_fp16")]; int32 const_2124 = const()[name = string("const_2124"), val = int32(-1)]; tensor softmax_14_cast_fp16 = softmax(axis = const_2124, x = add_246_cast_fp16)[name = string("softmax_14_cast_fp16")]; bool matmul_53_transpose_x_0 = const()[name = string("matmul_53_transpose_x_0"), val = bool(false)]; bool matmul_53_transpose_y_0 = const()[name = string("matmul_53_transpose_y_0"), val = bool(false)]; tensor matmul_53_cast_fp16 = matmul(transpose_x = matmul_53_transpose_x_0, transpose_y = matmul_53_transpose_y_0, x = softmax_14_cast_fp16, y = view_88_cast_fp16)[name = string("matmul_53_cast_fp16")]; tensor transpose_98_perm_0 = const()[name = string("transpose_98_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2129 = const()[name = string("const_2129"), val = tensor([1, 512, -1])]; tensor transpose_98_cast_fp16 = transpose(perm = transpose_98_perm_0, x = matmul_53_cast_fp16)[name = string("transpose_36")]; tensor view_89_cast_fp16 = reshape(shape = const_2129, x = transpose_98_cast_fp16)[name = string("view_89_cast_fp16")]; tensor p_st_0_model_layers_14_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262809536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263399424))))[name = string("p_st_0_model_layers_14_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_101_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_14_self_attn_o_proj_weight_to_fp16_quantized, x = view_89_cast_fp16)[name = string("linear_101_cast_fp16")]; fp16 const_2131_promoted_to_fp16 = const()[name = string("const_2131_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_88_cast_fp16 = pow(x = linear_101_cast_fp16, y = const_2131_promoted_to_fp16)[name = string("pow_88_cast_fp16")]; tensor mean_87_axes_0 = const()[name = string("mean_87_axes_0"), val = tensor([-1])]; bool mean_87_keep_dims_0 = const()[name = string("mean_87_keep_dims_0"), val = bool(true)]; tensor mean_87_cast_fp16 = reduce_mean(axes = mean_87_axes_0, keep_dims = mean_87_keep_dims_0, x = pow_88_cast_fp16)[name = string("mean_87_cast_fp16")]; fp16 const_2134_to_fp16 = const()[name = string("const_2134_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_247_cast_fp16 = add(x = mean_87_cast_fp16, y = const_2134_to_fp16)[name = string("add_247_cast_fp16")]; fp32 rsqrt_87_epsilon_0 = const()[name = string("rsqrt_87_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_87_cast_fp16 = rsqrt(epsilon = rsqrt_87_epsilon_0, x = add_247_cast_fp16)[name = string("rsqrt_87_cast_fp16")]; tensor mul_314_cast_fp16 = mul(x = linear_101_cast_fp16, y = rsqrt_87_cast_fp16)[name = string("mul_314_cast_fp16")]; tensor add_248_to_fp16 = const()[name = string("add_248_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263401024)))]; tensor mul_315_cast_fp16 = mul(x = mul_314_cast_fp16, y = add_248_to_fp16)[name = string("mul_315_cast_fp16")]; tensor add_249_cast_fp16 = add(x = add_237_cast_fp16, y = mul_315_cast_fp16)[name = string("add_249_cast_fp16")]; fp16 const_2139_promoted_to_fp16 = const()[name = string("const_2139_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_89_cast_fp16 = pow(x = add_249_cast_fp16, y = const_2139_promoted_to_fp16)[name = string("pow_89_cast_fp16")]; tensor mean_88_axes_0 = const()[name = string("mean_88_axes_0"), val = tensor([-1])]; bool mean_88_keep_dims_0 = const()[name = string("mean_88_keep_dims_0"), val = bool(true)]; tensor mean_88_cast_fp16 = reduce_mean(axes = mean_88_axes_0, keep_dims = mean_88_keep_dims_0, x = pow_89_cast_fp16)[name = string("mean_88_cast_fp16")]; fp16 const_2142_to_fp16 = const()[name = string("const_2142_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_250_cast_fp16 = add(x = mean_88_cast_fp16, y = const_2142_to_fp16)[name = string("add_250_cast_fp16")]; fp32 rsqrt_88_epsilon_0 = const()[name = string("rsqrt_88_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_88_cast_fp16 = rsqrt(epsilon = rsqrt_88_epsilon_0, x = add_250_cast_fp16)[name = string("rsqrt_88_cast_fp16")]; tensor mul_316_cast_fp16 = mul(x = add_249_cast_fp16, y = rsqrt_88_cast_fp16)[name = string("mul_316_cast_fp16")]; tensor add_251_to_fp16 = const()[name = string("add_251_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263402624)))]; tensor mul_317_cast_fp16 = mul(x = mul_316_cast_fp16, y = add_251_to_fp16)[name = string("mul_317_cast_fp16")]; tensor p_st_0_model_layers_14_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263404224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(264289024))))[name = string("p_st_0_model_layers_14_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_102_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_14_mlp_gate_proj_weight_to_fp16_quantized, x = mul_317_cast_fp16)[name = string("linear_102_cast_fp16")]; string gelu_14_mode_0 = const()[name = string("gelu_14_mode_0"), val = string("EXACT")]; tensor gelu_14_cast_fp16 = gelu(mode = gelu_14_mode_0, x = linear_102_cast_fp16)[name = string("gelu_14_cast_fp16")]; tensor p_st_0_model_layers_14_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(264291392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265176192))))[name = string("p_st_0_model_layers_14_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_103_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_14_mlp_up_proj_weight_to_fp16_quantized, x = mul_317_cast_fp16)[name = string("linear_103_cast_fp16")]; tensor mul_318_cast_fp16 = mul(x = gelu_14_cast_fp16, y = linear_103_cast_fp16)[name = string("mul_318_cast_fp16")]; tensor p_st_0_model_layers_14_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265178560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266063360))))[name = string("p_st_0_model_layers_14_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_104_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_14_mlp_down_proj_weight_to_fp16_quantized, x = mul_318_cast_fp16)[name = string("linear_104_cast_fp16")]; fp16 const_2147_promoted_to_fp16 = const()[name = string("const_2147_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_90_cast_fp16 = pow(x = linear_104_cast_fp16, y = const_2147_promoted_to_fp16)[name = string("pow_90_cast_fp16")]; tensor mean_89_axes_0 = const()[name = string("mean_89_axes_0"), val = tensor([-1])]; bool mean_89_keep_dims_0 = const()[name = string("mean_89_keep_dims_0"), val = bool(true)]; tensor mean_89_cast_fp16 = reduce_mean(axes = mean_89_axes_0, keep_dims = mean_89_keep_dims_0, x = pow_90_cast_fp16)[name = string("mean_89_cast_fp16")]; fp16 const_2150_to_fp16 = const()[name = string("const_2150_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_252_cast_fp16 = add(x = mean_89_cast_fp16, y = const_2150_to_fp16)[name = string("add_252_cast_fp16")]; fp32 rsqrt_89_epsilon_0 = const()[name = string("rsqrt_89_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_89_cast_fp16 = rsqrt(epsilon = rsqrt_89_epsilon_0, x = add_252_cast_fp16)[name = string("rsqrt_89_cast_fp16")]; tensor mul_319_cast_fp16 = mul(x = linear_104_cast_fp16, y = rsqrt_89_cast_fp16)[name = string("mul_319_cast_fp16")]; tensor add_253_to_fp16 = const()[name = string("add_253_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266064960)))]; tensor mul_320_cast_fp16 = mul(x = mul_319_cast_fp16, y = add_253_to_fp16)[name = string("mul_320_cast_fp16")]; tensor add_254_cast_fp16 = add(x = add_249_cast_fp16, y = mul_320_cast_fp16)[name = string("add_254_cast_fp16")]; fp16 const_2155_promoted_to_fp16 = const()[name = string("const_2155_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_91_cast_fp16 = pow(x = add_254_cast_fp16, y = const_2155_promoted_to_fp16)[name = string("pow_91_cast_fp16")]; tensor mean_90_axes_0 = const()[name = string("mean_90_axes_0"), val = tensor([-1])]; bool mean_90_keep_dims_0 = const()[name = string("mean_90_keep_dims_0"), val = bool(true)]; tensor mean_90_cast_fp16 = reduce_mean(axes = mean_90_axes_0, keep_dims = mean_90_keep_dims_0, x = pow_91_cast_fp16)[name = string("mean_90_cast_fp16")]; fp16 const_2158_to_fp16 = const()[name = string("const_2158_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_255_cast_fp16 = add(x = mean_90_cast_fp16, y = const_2158_to_fp16)[name = string("add_255_cast_fp16")]; fp32 rsqrt_90_epsilon_0 = const()[name = string("rsqrt_90_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_90_cast_fp16 = rsqrt(epsilon = rsqrt_90_epsilon_0, x = add_255_cast_fp16)[name = string("rsqrt_90_cast_fp16")]; tensor mul_321_cast_fp16 = mul(x = add_254_cast_fp16, y = rsqrt_90_cast_fp16)[name = string("mul_321_cast_fp16")]; tensor add_256_to_fp16 = const()[name = string("add_256_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266066560)))]; tensor mul_322_cast_fp16 = mul(x = mul_321_cast_fp16, y = add_256_to_fp16)[name = string("mul_322_cast_fp16")]; tensor p_st_0_model_layers_15_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266068160))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266658048))))[name = string("p_st_0_model_layers_15_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_105_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_15_self_attn_q_proj_weight_to_fp16_quantized, x = mul_322_cast_fp16)[name = string("linear_105_cast_fp16")]; tensor const_2162 = const()[name = string("const_2162"), val = tensor([1, 512, -1, 256])]; tensor view_90_cast_fp16 = reshape(shape = const_2162, x = linear_105_cast_fp16)[name = string("view_90_cast_fp16")]; tensor transpose_99_perm_0 = const()[name = string("transpose_99_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_15_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266659648))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266856320))))[name = string("p_st_0_model_layers_15_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_106_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_15_self_attn_k_proj_weight_to_fp16_quantized, x = mul_322_cast_fp16)[name = string("linear_106_cast_fp16")]; tensor const_2165 = const()[name = string("const_2165"), val = tensor([1, 512, -1, 256])]; tensor view_91_cast_fp16 = reshape(shape = const_2165, x = linear_106_cast_fp16)[name = string("view_91_cast_fp16")]; tensor transpose_100_perm_0 = const()[name = string("transpose_100_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_15_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(266856896))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267053568))))[name = string("p_st_0_model_layers_15_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_107_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_15_self_attn_v_proj_weight_to_fp16_quantized, x = mul_322_cast_fp16)[name = string("linear_107_cast_fp16")]; tensor const_2168 = const()[name = string("const_2168"), val = tensor([1, 512, -1, 256])]; tensor view_92_cast_fp16 = reshape(shape = const_2168, x = linear_107_cast_fp16)[name = string("view_92_cast_fp16")]; tensor transpose_101_perm_0 = const()[name = string("transpose_101_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2172_promoted_to_fp16 = const()[name = string("const_2172_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_99_cast_fp16 = transpose(perm = transpose_99_perm_0, x = view_90_cast_fp16)[name = string("transpose_35")]; tensor pow_92_cast_fp16 = pow(x = transpose_99_cast_fp16, y = const_2172_promoted_to_fp16)[name = string("pow_92_cast_fp16")]; tensor mean_91_axes_0 = const()[name = string("mean_91_axes_0"), val = tensor([-1])]; bool mean_91_keep_dims_0 = const()[name = string("mean_91_keep_dims_0"), val = bool(true)]; tensor mean_91_cast_fp16 = reduce_mean(axes = mean_91_axes_0, keep_dims = mean_91_keep_dims_0, x = pow_92_cast_fp16)[name = string("mean_91_cast_fp16")]; fp16 const_2175_to_fp16 = const()[name = string("const_2175_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_257_cast_fp16 = add(x = mean_91_cast_fp16, y = const_2175_to_fp16)[name = string("add_257_cast_fp16")]; fp32 rsqrt_91_epsilon_0 = const()[name = string("rsqrt_91_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_91_cast_fp16 = rsqrt(epsilon = rsqrt_91_epsilon_0, x = add_257_cast_fp16)[name = string("rsqrt_91_cast_fp16")]; tensor mul_323_cast_fp16 = mul(x = transpose_99_cast_fp16, y = rsqrt_91_cast_fp16)[name = string("mul_323_cast_fp16")]; tensor add_258_to_fp16 = const()[name = string("add_258_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267054144)))]; tensor mul_324_cast_fp16 = mul(x = mul_323_cast_fp16, y = add_258_to_fp16)[name = string("mul_324_cast_fp16")]; fp16 const_2180_promoted_to_fp16 = const()[name = string("const_2180_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_100_cast_fp16 = transpose(perm = transpose_100_perm_0, x = view_91_cast_fp16)[name = string("transpose_34")]; tensor pow_93_cast_fp16 = pow(x = transpose_100_cast_fp16, y = const_2180_promoted_to_fp16)[name = string("pow_93_cast_fp16")]; tensor mean_92_axes_0 = const()[name = string("mean_92_axes_0"), val = tensor([-1])]; bool mean_92_keep_dims_0 = const()[name = string("mean_92_keep_dims_0"), val = bool(true)]; tensor mean_92_cast_fp16 = reduce_mean(axes = mean_92_axes_0, keep_dims = mean_92_keep_dims_0, x = pow_93_cast_fp16)[name = string("mean_92_cast_fp16")]; fp16 const_2183_to_fp16 = const()[name = string("const_2183_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_259_cast_fp16 = add(x = mean_92_cast_fp16, y = const_2183_to_fp16)[name = string("add_259_cast_fp16")]; fp32 rsqrt_92_epsilon_0 = const()[name = string("rsqrt_92_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_92_cast_fp16 = rsqrt(epsilon = rsqrt_92_epsilon_0, x = add_259_cast_fp16)[name = string("rsqrt_92_cast_fp16")]; tensor mul_325_cast_fp16 = mul(x = transpose_100_cast_fp16, y = rsqrt_92_cast_fp16)[name = string("mul_325_cast_fp16")]; tensor add_260_to_fp16 = const()[name = string("add_260_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267054720)))]; tensor mul_326_cast_fp16 = mul(x = mul_325_cast_fp16, y = add_260_to_fp16)[name = string("mul_326_cast_fp16")]; tensor mul_327_cast_fp16 = mul(x = mul_324_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_327_cast_fp16")]; tensor slice_406_begin_0 = const()[name = string("slice_406_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_406_end_0 = const()[name = string("slice_406_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_406_end_mask_0 = const()[name = string("slice_406_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_406_cast_fp16 = slice_by_index(begin = slice_406_begin_0, end = slice_406_end_0, end_mask = slice_406_end_mask_0, x = mul_324_cast_fp16)[name = string("slice_406_cast_fp16")]; tensor slice_407_begin_0 = const()[name = string("slice_407_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_407_end_0 = const()[name = string("slice_407_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_407_end_mask_0 = const()[name = string("slice_407_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_407_cast_fp16 = slice_by_index(begin = slice_407_begin_0, end = slice_407_end_0, end_mask = slice_407_end_mask_0, x = mul_324_cast_fp16)[name = string("slice_407_cast_fp16")]; fp16 const_2195_promoted_to_fp16 = const()[name = string("const_2195_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_30_cast_fp16 = mul(x = slice_407_cast_fp16, y = const_2195_promoted_to_fp16)[name = string("neg_30_cast_fp16")]; int32 const_2196 = const()[name = string("const_2196"), val = int32(-1)]; bool cat_84_interleave_0 = const()[name = string("cat_84_interleave_0"), val = bool(false)]; tensor cat_84_cast_fp16 = concat(axis = const_2196, interleave = cat_84_interleave_0, values = (neg_30_cast_fp16, slice_406_cast_fp16))[name = string("cat_84_cast_fp16")]; tensor mul_328_cast_fp16 = mul(x = cat_84_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_328_cast_fp16")]; tensor add_261_cast_fp16 = add(x = mul_327_cast_fp16, y = mul_328_cast_fp16)[name = string("add_261_cast_fp16")]; tensor mul_329_cast_fp16 = mul(x = mul_326_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_329_cast_fp16")]; tensor slice_408_begin_0 = const()[name = string("slice_408_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_408_end_0 = const()[name = string("slice_408_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_408_end_mask_0 = const()[name = string("slice_408_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_408_cast_fp16 = slice_by_index(begin = slice_408_begin_0, end = slice_408_end_0, end_mask = slice_408_end_mask_0, x = mul_326_cast_fp16)[name = string("slice_408_cast_fp16")]; tensor slice_409_begin_0 = const()[name = string("slice_409_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_409_end_0 = const()[name = string("slice_409_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_409_end_mask_0 = const()[name = string("slice_409_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_409_cast_fp16 = slice_by_index(begin = slice_409_begin_0, end = slice_409_end_0, end_mask = slice_409_end_mask_0, x = mul_326_cast_fp16)[name = string("slice_409_cast_fp16")]; fp16 const_2203_promoted_to_fp16 = const()[name = string("const_2203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_31_cast_fp16 = mul(x = slice_409_cast_fp16, y = const_2203_promoted_to_fp16)[name = string("neg_31_cast_fp16")]; int32 const_2204 = const()[name = string("const_2204"), val = int32(-1)]; bool cat_85_interleave_0 = const()[name = string("cat_85_interleave_0"), val = bool(false)]; tensor cat_85_cast_fp16 = concat(axis = const_2204, interleave = cat_85_interleave_0, values = (neg_31_cast_fp16, slice_408_cast_fp16))[name = string("cat_85_cast_fp16")]; tensor mul_330_cast_fp16 = mul(x = cat_85_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_330_cast_fp16")]; tensor add_262_cast_fp16 = add(x = mul_329_cast_fp16, y = mul_330_cast_fp16)[name = string("add_262_cast_fp16")]; int32 const_2205 = const()[name = string("const_2205"), val = int32(-2)]; bool cat_86_interleave_0 = const()[name = string("cat_86_interleave_0"), val = bool(false)]; tensor cat_86_cast_fp16 = concat(axis = const_2205, interleave = cat_86_interleave_0, values = add_262_cast_fp16)[name = string("cat_86_cast_fp16")]; int32 const_2206 = const()[name = string("const_2206"), val = int32(-2)]; bool cat_87_interleave_0 = const()[name = string("cat_87_interleave_0"), val = bool(false)]; tensor transpose_101_cast_fp16 = transpose(perm = transpose_101_perm_0, x = view_92_cast_fp16)[name = string("transpose_33")]; tensor cat_87_cast_fp16 = concat(axis = const_2206, interleave = cat_87_interleave_0, values = transpose_101_cast_fp16)[name = string("cat_87_cast_fp16")]; tensor unsqueeze_139_axes_0 = const()[name = string("unsqueeze_139_axes_0"), val = tensor([2])]; tensor unsqueeze_139_cast_fp16 = expand_dims(axes = unsqueeze_139_axes_0, x = cat_86_cast_fp16)[name = string("unsqueeze_139_cast_fp16")]; tensor expand_56_reps_0 = const()[name = string("expand_56_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_56_cast_fp16 = tile(reps = expand_56_reps_0, x = unsqueeze_139_cast_fp16)[name = string("expand_56_cast_fp16")]; tensor const_2221 = const()[name = string("const_2221"), val = tensor([1, 3, 512, 256])]; tensor view_93_cast_fp16 = reshape(shape = const_2221, x = expand_56_cast_fp16)[name = string("view_93_cast_fp16")]; tensor unsqueeze_140_axes_0 = const()[name = string("unsqueeze_140_axes_0"), val = tensor([2])]; tensor unsqueeze_140_cast_fp16 = expand_dims(axes = unsqueeze_140_axes_0, x = cat_87_cast_fp16)[name = string("unsqueeze_140_cast_fp16")]; tensor expand_57_reps_0 = const()[name = string("expand_57_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_57_cast_fp16 = tile(reps = expand_57_reps_0, x = unsqueeze_140_cast_fp16)[name = string("expand_57_cast_fp16")]; tensor const_2236 = const()[name = string("const_2236"), val = tensor([1, 3, 512, 256])]; tensor view_94_cast_fp16 = reshape(shape = const_2236, x = expand_57_cast_fp16)[name = string("view_94_cast_fp16")]; bool matmul_54_transpose_x_1 = const()[name = string("matmul_54_transpose_x_1"), val = bool(false)]; bool matmul_54_transpose_y_1 = const()[name = string("matmul_54_transpose_y_1"), val = bool(true)]; tensor matmul_54_cast_fp16 = matmul(transpose_x = matmul_54_transpose_x_1, transpose_y = matmul_54_transpose_y_1, x = add_261_cast_fp16, y = view_93_cast_fp16)[name = string("matmul_54_cast_fp16")]; fp16 const_2239_to_fp16 = const()[name = string("const_2239_to_fp16"), val = fp16(0x1p-4)]; tensor mul_331_cast_fp16 = mul(x = matmul_54_cast_fp16, y = const_2239_to_fp16)[name = string("mul_331_cast_fp16")]; tensor add_263_cast_fp16 = add(x = mul_331_cast_fp16, y = expand_cast_fp16)[name = string("add_263_cast_fp16")]; int32 const_2249 = const()[name = string("const_2249"), val = int32(-1)]; tensor softmax_15_cast_fp16 = softmax(axis = const_2249, x = add_263_cast_fp16)[name = string("softmax_15_cast_fp16")]; bool matmul_55_transpose_x_0 = const()[name = string("matmul_55_transpose_x_0"), val = bool(false)]; bool matmul_55_transpose_y_0 = const()[name = string("matmul_55_transpose_y_0"), val = bool(false)]; tensor matmul_55_cast_fp16 = matmul(transpose_x = matmul_55_transpose_x_0, transpose_y = matmul_55_transpose_y_0, x = softmax_15_cast_fp16, y = view_94_cast_fp16)[name = string("matmul_55_cast_fp16")]; tensor transpose_103_perm_0 = const()[name = string("transpose_103_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2254 = const()[name = string("const_2254"), val = tensor([1, 512, -1])]; tensor transpose_103_cast_fp16 = transpose(perm = transpose_103_perm_0, x = matmul_55_cast_fp16)[name = string("transpose_32")]; tensor view_95_cast_fp16 = reshape(shape = const_2254, x = transpose_103_cast_fp16)[name = string("view_95_cast_fp16")]; tensor p_st_0_model_layers_15_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267055296))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267645184))))[name = string("p_st_0_model_layers_15_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_108_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_15_self_attn_o_proj_weight_to_fp16_quantized, x = view_95_cast_fp16)[name = string("linear_108_cast_fp16")]; fp16 const_2256_promoted_to_fp16 = const()[name = string("const_2256_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_94_cast_fp16 = pow(x = linear_108_cast_fp16, y = const_2256_promoted_to_fp16)[name = string("pow_94_cast_fp16")]; tensor mean_93_axes_0 = const()[name = string("mean_93_axes_0"), val = tensor([-1])]; bool mean_93_keep_dims_0 = const()[name = string("mean_93_keep_dims_0"), val = bool(true)]; tensor mean_93_cast_fp16 = reduce_mean(axes = mean_93_axes_0, keep_dims = mean_93_keep_dims_0, x = pow_94_cast_fp16)[name = string("mean_93_cast_fp16")]; fp16 const_2259_to_fp16 = const()[name = string("const_2259_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_264_cast_fp16 = add(x = mean_93_cast_fp16, y = const_2259_to_fp16)[name = string("add_264_cast_fp16")]; fp32 rsqrt_93_epsilon_0 = const()[name = string("rsqrt_93_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_93_cast_fp16 = rsqrt(epsilon = rsqrt_93_epsilon_0, x = add_264_cast_fp16)[name = string("rsqrt_93_cast_fp16")]; tensor mul_332_cast_fp16 = mul(x = linear_108_cast_fp16, y = rsqrt_93_cast_fp16)[name = string("mul_332_cast_fp16")]; tensor add_265_to_fp16 = const()[name = string("add_265_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267646784)))]; tensor mul_333_cast_fp16 = mul(x = mul_332_cast_fp16, y = add_265_to_fp16)[name = string("mul_333_cast_fp16")]; tensor add_266_cast_fp16 = add(x = add_254_cast_fp16, y = mul_333_cast_fp16)[name = string("add_266_cast_fp16")]; fp16 const_2264_promoted_to_fp16 = const()[name = string("const_2264_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_95_cast_fp16 = pow(x = add_266_cast_fp16, y = const_2264_promoted_to_fp16)[name = string("pow_95_cast_fp16")]; tensor mean_94_axes_0 = const()[name = string("mean_94_axes_0"), val = tensor([-1])]; bool mean_94_keep_dims_0 = const()[name = string("mean_94_keep_dims_0"), val = bool(true)]; tensor mean_94_cast_fp16 = reduce_mean(axes = mean_94_axes_0, keep_dims = mean_94_keep_dims_0, x = pow_95_cast_fp16)[name = string("mean_94_cast_fp16")]; fp16 const_2267_to_fp16 = const()[name = string("const_2267_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_267_cast_fp16 = add(x = mean_94_cast_fp16, y = const_2267_to_fp16)[name = string("add_267_cast_fp16")]; fp32 rsqrt_94_epsilon_0 = const()[name = string("rsqrt_94_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_94_cast_fp16 = rsqrt(epsilon = rsqrt_94_epsilon_0, x = add_267_cast_fp16)[name = string("rsqrt_94_cast_fp16")]; tensor mul_334_cast_fp16 = mul(x = add_266_cast_fp16, y = rsqrt_94_cast_fp16)[name = string("mul_334_cast_fp16")]; tensor add_268_to_fp16 = const()[name = string("add_268_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267648384)))]; tensor mul_335_cast_fp16 = mul(x = mul_334_cast_fp16, y = add_268_to_fp16)[name = string("mul_335_cast_fp16")]; tensor p_st_0_model_layers_15_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267649984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268534784))))[name = string("p_st_0_model_layers_15_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_109_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_15_mlp_gate_proj_weight_to_fp16_quantized, x = mul_335_cast_fp16)[name = string("linear_109_cast_fp16")]; string gelu_15_mode_0 = const()[name = string("gelu_15_mode_0"), val = string("EXACT")]; tensor gelu_15_cast_fp16 = gelu(mode = gelu_15_mode_0, x = linear_109_cast_fp16)[name = string("gelu_15_cast_fp16")]; tensor p_st_0_model_layers_15_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268537152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269421952))))[name = string("p_st_0_model_layers_15_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_110_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_15_mlp_up_proj_weight_to_fp16_quantized, x = mul_335_cast_fp16)[name = string("linear_110_cast_fp16")]; tensor mul_336_cast_fp16 = mul(x = gelu_15_cast_fp16, y = linear_110_cast_fp16)[name = string("mul_336_cast_fp16")]; tensor p_st_0_model_layers_15_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269424320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270309120))))[name = string("p_st_0_model_layers_15_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_111_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_15_mlp_down_proj_weight_to_fp16_quantized, x = mul_336_cast_fp16)[name = string("linear_111_cast_fp16")]; fp16 const_2272_promoted_to_fp16 = const()[name = string("const_2272_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_96_cast_fp16 = pow(x = linear_111_cast_fp16, y = const_2272_promoted_to_fp16)[name = string("pow_96_cast_fp16")]; tensor mean_95_axes_0 = const()[name = string("mean_95_axes_0"), val = tensor([-1])]; bool mean_95_keep_dims_0 = const()[name = string("mean_95_keep_dims_0"), val = bool(true)]; tensor mean_95_cast_fp16 = reduce_mean(axes = mean_95_axes_0, keep_dims = mean_95_keep_dims_0, x = pow_96_cast_fp16)[name = string("mean_95_cast_fp16")]; fp16 const_2275_to_fp16 = const()[name = string("const_2275_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_269_cast_fp16 = add(x = mean_95_cast_fp16, y = const_2275_to_fp16)[name = string("add_269_cast_fp16")]; fp32 rsqrt_95_epsilon_0 = const()[name = string("rsqrt_95_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_95_cast_fp16 = rsqrt(epsilon = rsqrt_95_epsilon_0, x = add_269_cast_fp16)[name = string("rsqrt_95_cast_fp16")]; tensor mul_337_cast_fp16 = mul(x = linear_111_cast_fp16, y = rsqrt_95_cast_fp16)[name = string("mul_337_cast_fp16")]; tensor add_270_to_fp16 = const()[name = string("add_270_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270310720)))]; tensor mul_338_cast_fp16 = mul(x = mul_337_cast_fp16, y = add_270_to_fp16)[name = string("mul_338_cast_fp16")]; tensor add_271_cast_fp16 = add(x = add_266_cast_fp16, y = mul_338_cast_fp16)[name = string("add_271_cast_fp16")]; fp16 const_2280_promoted_to_fp16 = const()[name = string("const_2280_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_97_cast_fp16 = pow(x = add_271_cast_fp16, y = const_2280_promoted_to_fp16)[name = string("pow_97_cast_fp16")]; tensor mean_96_axes_0 = const()[name = string("mean_96_axes_0"), val = tensor([-1])]; bool mean_96_keep_dims_0 = const()[name = string("mean_96_keep_dims_0"), val = bool(true)]; tensor mean_96_cast_fp16 = reduce_mean(axes = mean_96_axes_0, keep_dims = mean_96_keep_dims_0, x = pow_97_cast_fp16)[name = string("mean_96_cast_fp16")]; fp16 const_2283_to_fp16 = const()[name = string("const_2283_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_272_cast_fp16 = add(x = mean_96_cast_fp16, y = const_2283_to_fp16)[name = string("add_272_cast_fp16")]; fp32 rsqrt_96_epsilon_0 = const()[name = string("rsqrt_96_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_96_cast_fp16 = rsqrt(epsilon = rsqrt_96_epsilon_0, x = add_272_cast_fp16)[name = string("rsqrt_96_cast_fp16")]; tensor mul_339_cast_fp16 = mul(x = add_271_cast_fp16, y = rsqrt_96_cast_fp16)[name = string("mul_339_cast_fp16")]; tensor add_273_to_fp16 = const()[name = string("add_273_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270312320)))]; tensor mul_340_cast_fp16 = mul(x = mul_339_cast_fp16, y = add_273_to_fp16)[name = string("mul_340_cast_fp16")]; tensor p_st_0_model_layers_16_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270313920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270903808))))[name = string("p_st_0_model_layers_16_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_112_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_16_self_attn_q_proj_weight_to_fp16_quantized, x = mul_340_cast_fp16)[name = string("linear_112_cast_fp16")]; tensor const_2287 = const()[name = string("const_2287"), val = tensor([1, 512, -1, 256])]; tensor view_96_cast_fp16 = reshape(shape = const_2287, x = linear_112_cast_fp16)[name = string("view_96_cast_fp16")]; tensor transpose_104_perm_0 = const()[name = string("transpose_104_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_16_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270905408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271102080))))[name = string("p_st_0_model_layers_16_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_113_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_16_self_attn_k_proj_weight_to_fp16_quantized, x = mul_340_cast_fp16)[name = string("linear_113_cast_fp16")]; tensor const_2290 = const()[name = string("const_2290"), val = tensor([1, 512, -1, 256])]; tensor view_97_cast_fp16 = reshape(shape = const_2290, x = linear_113_cast_fp16)[name = string("view_97_cast_fp16")]; tensor transpose_105_perm_0 = const()[name = string("transpose_105_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_16_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271102656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271299328))))[name = string("p_st_0_model_layers_16_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_114_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_16_self_attn_v_proj_weight_to_fp16_quantized, x = mul_340_cast_fp16)[name = string("linear_114_cast_fp16")]; tensor const_2293 = const()[name = string("const_2293"), val = tensor([1, 512, -1, 256])]; tensor view_98_cast_fp16 = reshape(shape = const_2293, x = linear_114_cast_fp16)[name = string("view_98_cast_fp16")]; tensor transpose_106_perm_0 = const()[name = string("transpose_106_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2297_promoted_to_fp16 = const()[name = string("const_2297_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_104_cast_fp16 = transpose(perm = transpose_104_perm_0, x = view_96_cast_fp16)[name = string("transpose_31")]; tensor pow_98_cast_fp16 = pow(x = transpose_104_cast_fp16, y = const_2297_promoted_to_fp16)[name = string("pow_98_cast_fp16")]; tensor mean_97_axes_0 = const()[name = string("mean_97_axes_0"), val = tensor([-1])]; bool mean_97_keep_dims_0 = const()[name = string("mean_97_keep_dims_0"), val = bool(true)]; tensor mean_97_cast_fp16 = reduce_mean(axes = mean_97_axes_0, keep_dims = mean_97_keep_dims_0, x = pow_98_cast_fp16)[name = string("mean_97_cast_fp16")]; fp16 const_2300_to_fp16 = const()[name = string("const_2300_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_274_cast_fp16 = add(x = mean_97_cast_fp16, y = const_2300_to_fp16)[name = string("add_274_cast_fp16")]; fp32 rsqrt_97_epsilon_0 = const()[name = string("rsqrt_97_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_97_cast_fp16 = rsqrt(epsilon = rsqrt_97_epsilon_0, x = add_274_cast_fp16)[name = string("rsqrt_97_cast_fp16")]; tensor mul_341_cast_fp16 = mul(x = transpose_104_cast_fp16, y = rsqrt_97_cast_fp16)[name = string("mul_341_cast_fp16")]; tensor add_275_to_fp16 = const()[name = string("add_275_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271299904)))]; tensor mul_342_cast_fp16 = mul(x = mul_341_cast_fp16, y = add_275_to_fp16)[name = string("mul_342_cast_fp16")]; fp16 const_2305_promoted_to_fp16 = const()[name = string("const_2305_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_105_cast_fp16 = transpose(perm = transpose_105_perm_0, x = view_97_cast_fp16)[name = string("transpose_30")]; tensor pow_99_cast_fp16 = pow(x = transpose_105_cast_fp16, y = const_2305_promoted_to_fp16)[name = string("pow_99_cast_fp16")]; tensor mean_98_axes_0 = const()[name = string("mean_98_axes_0"), val = tensor([-1])]; bool mean_98_keep_dims_0 = const()[name = string("mean_98_keep_dims_0"), val = bool(true)]; tensor mean_98_cast_fp16 = reduce_mean(axes = mean_98_axes_0, keep_dims = mean_98_keep_dims_0, x = pow_99_cast_fp16)[name = string("mean_98_cast_fp16")]; fp16 const_2308_to_fp16 = const()[name = string("const_2308_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_276_cast_fp16 = add(x = mean_98_cast_fp16, y = const_2308_to_fp16)[name = string("add_276_cast_fp16")]; fp32 rsqrt_98_epsilon_0 = const()[name = string("rsqrt_98_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_98_cast_fp16 = rsqrt(epsilon = rsqrt_98_epsilon_0, x = add_276_cast_fp16)[name = string("rsqrt_98_cast_fp16")]; tensor mul_343_cast_fp16 = mul(x = transpose_105_cast_fp16, y = rsqrt_98_cast_fp16)[name = string("mul_343_cast_fp16")]; tensor add_277_to_fp16 = const()[name = string("add_277_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271300480)))]; tensor mul_344_cast_fp16 = mul(x = mul_343_cast_fp16, y = add_277_to_fp16)[name = string("mul_344_cast_fp16")]; tensor mul_345_cast_fp16 = mul(x = mul_342_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_345_cast_fp16")]; tensor slice_429_begin_0 = const()[name = string("slice_429_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_429_end_0 = const()[name = string("slice_429_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_429_end_mask_0 = const()[name = string("slice_429_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_429_cast_fp16 = slice_by_index(begin = slice_429_begin_0, end = slice_429_end_0, end_mask = slice_429_end_mask_0, x = mul_342_cast_fp16)[name = string("slice_429_cast_fp16")]; tensor slice_430_begin_0 = const()[name = string("slice_430_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_430_end_0 = const()[name = string("slice_430_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_430_end_mask_0 = const()[name = string("slice_430_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_430_cast_fp16 = slice_by_index(begin = slice_430_begin_0, end = slice_430_end_0, end_mask = slice_430_end_mask_0, x = mul_342_cast_fp16)[name = string("slice_430_cast_fp16")]; fp16 const_2320_promoted_to_fp16 = const()[name = string("const_2320_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_32_cast_fp16 = mul(x = slice_430_cast_fp16, y = const_2320_promoted_to_fp16)[name = string("neg_32_cast_fp16")]; int32 const_2321 = const()[name = string("const_2321"), val = int32(-1)]; bool cat_88_interleave_0 = const()[name = string("cat_88_interleave_0"), val = bool(false)]; tensor cat_88_cast_fp16 = concat(axis = const_2321, interleave = cat_88_interleave_0, values = (neg_32_cast_fp16, slice_429_cast_fp16))[name = string("cat_88_cast_fp16")]; tensor mul_346_cast_fp16 = mul(x = cat_88_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_346_cast_fp16")]; tensor add_278_cast_fp16 = add(x = mul_345_cast_fp16, y = mul_346_cast_fp16)[name = string("add_278_cast_fp16")]; tensor mul_347_cast_fp16 = mul(x = mul_344_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_347_cast_fp16")]; tensor slice_431_begin_0 = const()[name = string("slice_431_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_431_end_0 = const()[name = string("slice_431_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_431_end_mask_0 = const()[name = string("slice_431_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_431_cast_fp16 = slice_by_index(begin = slice_431_begin_0, end = slice_431_end_0, end_mask = slice_431_end_mask_0, x = mul_344_cast_fp16)[name = string("slice_431_cast_fp16")]; tensor slice_432_begin_0 = const()[name = string("slice_432_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_432_end_0 = const()[name = string("slice_432_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_432_end_mask_0 = const()[name = string("slice_432_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_432_cast_fp16 = slice_by_index(begin = slice_432_begin_0, end = slice_432_end_0, end_mask = slice_432_end_mask_0, x = mul_344_cast_fp16)[name = string("slice_432_cast_fp16")]; fp16 const_2328_promoted_to_fp16 = const()[name = string("const_2328_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_33_cast_fp16 = mul(x = slice_432_cast_fp16, y = const_2328_promoted_to_fp16)[name = string("neg_33_cast_fp16")]; int32 const_2329 = const()[name = string("const_2329"), val = int32(-1)]; bool cat_89_interleave_0 = const()[name = string("cat_89_interleave_0"), val = bool(false)]; tensor cat_89_cast_fp16 = concat(axis = const_2329, interleave = cat_89_interleave_0, values = (neg_33_cast_fp16, slice_431_cast_fp16))[name = string("cat_89_cast_fp16")]; tensor mul_348_cast_fp16 = mul(x = cat_89_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_348_cast_fp16")]; tensor add_279_cast_fp16 = add(x = mul_347_cast_fp16, y = mul_348_cast_fp16)[name = string("add_279_cast_fp16")]; int32 const_2330 = const()[name = string("const_2330"), val = int32(-2)]; bool cat_90_interleave_0 = const()[name = string("cat_90_interleave_0"), val = bool(false)]; tensor cat_90_cast_fp16 = concat(axis = const_2330, interleave = cat_90_interleave_0, values = add_279_cast_fp16)[name = string("cat_90_cast_fp16")]; int32 const_2331 = const()[name = string("const_2331"), val = int32(-2)]; bool cat_91_interleave_0 = const()[name = string("cat_91_interleave_0"), val = bool(false)]; tensor transpose_106_cast_fp16 = transpose(perm = transpose_106_perm_0, x = view_98_cast_fp16)[name = string("transpose_29")]; tensor cat_91_cast_fp16 = concat(axis = const_2331, interleave = cat_91_interleave_0, values = transpose_106_cast_fp16)[name = string("cat_91_cast_fp16")]; tensor unsqueeze_143_axes_0 = const()[name = string("unsqueeze_143_axes_0"), val = tensor([2])]; tensor unsqueeze_143_cast_fp16 = expand_dims(axes = unsqueeze_143_axes_0, x = cat_90_cast_fp16)[name = string("unsqueeze_143_cast_fp16")]; tensor expand_58_reps_0 = const()[name = string("expand_58_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_58_cast_fp16 = tile(reps = expand_58_reps_0, x = unsqueeze_143_cast_fp16)[name = string("expand_58_cast_fp16")]; tensor const_2346 = const()[name = string("const_2346"), val = tensor([1, 3, 512, 256])]; tensor view_99_cast_fp16 = reshape(shape = const_2346, x = expand_58_cast_fp16)[name = string("view_99_cast_fp16")]; tensor unsqueeze_144_axes_0 = const()[name = string("unsqueeze_144_axes_0"), val = tensor([2])]; tensor unsqueeze_144_cast_fp16 = expand_dims(axes = unsqueeze_144_axes_0, x = cat_91_cast_fp16)[name = string("unsqueeze_144_cast_fp16")]; tensor expand_59_reps_0 = const()[name = string("expand_59_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_59_cast_fp16 = tile(reps = expand_59_reps_0, x = unsqueeze_144_cast_fp16)[name = string("expand_59_cast_fp16")]; tensor const_2361 = const()[name = string("const_2361"), val = tensor([1, 3, 512, 256])]; tensor view_100_cast_fp16 = reshape(shape = const_2361, x = expand_59_cast_fp16)[name = string("view_100_cast_fp16")]; bool matmul_56_transpose_x_1 = const()[name = string("matmul_56_transpose_x_1"), val = bool(false)]; bool matmul_56_transpose_y_1 = const()[name = string("matmul_56_transpose_y_1"), val = bool(true)]; tensor matmul_56_cast_fp16 = matmul(transpose_x = matmul_56_transpose_x_1, transpose_y = matmul_56_transpose_y_1, x = add_278_cast_fp16, y = view_99_cast_fp16)[name = string("matmul_56_cast_fp16")]; fp16 const_2364_to_fp16 = const()[name = string("const_2364_to_fp16"), val = fp16(0x1p-4)]; tensor mul_349_cast_fp16 = mul(x = matmul_56_cast_fp16, y = const_2364_to_fp16)[name = string("mul_349_cast_fp16")]; tensor add_280_cast_fp16 = add(x = mul_349_cast_fp16, y = expand_cast_fp16)[name = string("add_280_cast_fp16")]; int32 const_2374 = const()[name = string("const_2374"), val = int32(-1)]; tensor softmax_16_cast_fp16 = softmax(axis = const_2374, x = add_280_cast_fp16)[name = string("softmax_16_cast_fp16")]; bool matmul_57_transpose_x_0 = const()[name = string("matmul_57_transpose_x_0"), val = bool(false)]; bool matmul_57_transpose_y_0 = const()[name = string("matmul_57_transpose_y_0"), val = bool(false)]; tensor matmul_57_cast_fp16 = matmul(transpose_x = matmul_57_transpose_x_0, transpose_y = matmul_57_transpose_y_0, x = softmax_16_cast_fp16, y = view_100_cast_fp16)[name = string("matmul_57_cast_fp16")]; tensor transpose_108_perm_0 = const()[name = string("transpose_108_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2379 = const()[name = string("const_2379"), val = tensor([1, 512, -1])]; tensor transpose_108_cast_fp16 = transpose(perm = transpose_108_perm_0, x = matmul_57_cast_fp16)[name = string("transpose_28")]; tensor view_101_cast_fp16 = reshape(shape = const_2379, x = transpose_108_cast_fp16)[name = string("view_101_cast_fp16")]; tensor p_st_0_model_layers_16_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271301056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271890944))))[name = string("p_st_0_model_layers_16_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_115_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_16_self_attn_o_proj_weight_to_fp16_quantized, x = view_101_cast_fp16)[name = string("linear_115_cast_fp16")]; fp16 const_2381_promoted_to_fp16 = const()[name = string("const_2381_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_100_cast_fp16 = pow(x = linear_115_cast_fp16, y = const_2381_promoted_to_fp16)[name = string("pow_100_cast_fp16")]; tensor mean_99_axes_0 = const()[name = string("mean_99_axes_0"), val = tensor([-1])]; bool mean_99_keep_dims_0 = const()[name = string("mean_99_keep_dims_0"), val = bool(true)]; tensor mean_99_cast_fp16 = reduce_mean(axes = mean_99_axes_0, keep_dims = mean_99_keep_dims_0, x = pow_100_cast_fp16)[name = string("mean_99_cast_fp16")]; fp16 const_2384_to_fp16 = const()[name = string("const_2384_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_281_cast_fp16 = add(x = mean_99_cast_fp16, y = const_2384_to_fp16)[name = string("add_281_cast_fp16")]; fp32 rsqrt_99_epsilon_0 = const()[name = string("rsqrt_99_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_99_cast_fp16 = rsqrt(epsilon = rsqrt_99_epsilon_0, x = add_281_cast_fp16)[name = string("rsqrt_99_cast_fp16")]; tensor mul_350_cast_fp16 = mul(x = linear_115_cast_fp16, y = rsqrt_99_cast_fp16)[name = string("mul_350_cast_fp16")]; tensor add_282_to_fp16 = const()[name = string("add_282_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271892544)))]; tensor mul_351_cast_fp16 = mul(x = mul_350_cast_fp16, y = add_282_to_fp16)[name = string("mul_351_cast_fp16")]; tensor add_283_cast_fp16 = add(x = add_271_cast_fp16, y = mul_351_cast_fp16)[name = string("add_283_cast_fp16")]; fp16 const_2389_promoted_to_fp16 = const()[name = string("const_2389_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_101_cast_fp16 = pow(x = add_283_cast_fp16, y = const_2389_promoted_to_fp16)[name = string("pow_101_cast_fp16")]; tensor mean_100_axes_0 = const()[name = string("mean_100_axes_0"), val = tensor([-1])]; bool mean_100_keep_dims_0 = const()[name = string("mean_100_keep_dims_0"), val = bool(true)]; tensor mean_100_cast_fp16 = reduce_mean(axes = mean_100_axes_0, keep_dims = mean_100_keep_dims_0, x = pow_101_cast_fp16)[name = string("mean_100_cast_fp16")]; fp16 const_2392_to_fp16 = const()[name = string("const_2392_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_284_cast_fp16 = add(x = mean_100_cast_fp16, y = const_2392_to_fp16)[name = string("add_284_cast_fp16")]; fp32 rsqrt_100_epsilon_0 = const()[name = string("rsqrt_100_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_100_cast_fp16 = rsqrt(epsilon = rsqrt_100_epsilon_0, x = add_284_cast_fp16)[name = string("rsqrt_100_cast_fp16")]; tensor mul_352_cast_fp16 = mul(x = add_283_cast_fp16, y = rsqrt_100_cast_fp16)[name = string("mul_352_cast_fp16")]; tensor add_285_to_fp16 = const()[name = string("add_285_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271894144)))]; tensor mul_353_cast_fp16 = mul(x = mul_352_cast_fp16, y = add_285_to_fp16)[name = string("mul_353_cast_fp16")]; tensor p_st_0_model_layers_16_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271895744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272780544))))[name = string("p_st_0_model_layers_16_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_116_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_16_mlp_gate_proj_weight_to_fp16_quantized, x = mul_353_cast_fp16)[name = string("linear_116_cast_fp16")]; string gelu_16_mode_0 = const()[name = string("gelu_16_mode_0"), val = string("EXACT")]; tensor gelu_16_cast_fp16 = gelu(mode = gelu_16_mode_0, x = linear_116_cast_fp16)[name = string("gelu_16_cast_fp16")]; tensor p_st_0_model_layers_16_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272782912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273667712))))[name = string("p_st_0_model_layers_16_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_117_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_16_mlp_up_proj_weight_to_fp16_quantized, x = mul_353_cast_fp16)[name = string("linear_117_cast_fp16")]; tensor mul_354_cast_fp16 = mul(x = gelu_16_cast_fp16, y = linear_117_cast_fp16)[name = string("mul_354_cast_fp16")]; tensor p_st_0_model_layers_16_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273670080))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274554880))))[name = string("p_st_0_model_layers_16_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_118_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_16_mlp_down_proj_weight_to_fp16_quantized, x = mul_354_cast_fp16)[name = string("linear_118_cast_fp16")]; fp16 const_2397_promoted_to_fp16 = const()[name = string("const_2397_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_102_cast_fp16 = pow(x = linear_118_cast_fp16, y = const_2397_promoted_to_fp16)[name = string("pow_102_cast_fp16")]; tensor mean_101_axes_0 = const()[name = string("mean_101_axes_0"), val = tensor([-1])]; bool mean_101_keep_dims_0 = const()[name = string("mean_101_keep_dims_0"), val = bool(true)]; tensor mean_101_cast_fp16 = reduce_mean(axes = mean_101_axes_0, keep_dims = mean_101_keep_dims_0, x = pow_102_cast_fp16)[name = string("mean_101_cast_fp16")]; fp16 const_2400_to_fp16 = const()[name = string("const_2400_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_286_cast_fp16 = add(x = mean_101_cast_fp16, y = const_2400_to_fp16)[name = string("add_286_cast_fp16")]; fp32 rsqrt_101_epsilon_0 = const()[name = string("rsqrt_101_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_101_cast_fp16 = rsqrt(epsilon = rsqrt_101_epsilon_0, x = add_286_cast_fp16)[name = string("rsqrt_101_cast_fp16")]; tensor mul_355_cast_fp16 = mul(x = linear_118_cast_fp16, y = rsqrt_101_cast_fp16)[name = string("mul_355_cast_fp16")]; tensor add_287_to_fp16 = const()[name = string("add_287_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274556480)))]; tensor mul_356_cast_fp16 = mul(x = mul_355_cast_fp16, y = add_287_to_fp16)[name = string("mul_356_cast_fp16")]; tensor add_288_cast_fp16 = add(x = add_283_cast_fp16, y = mul_356_cast_fp16)[name = string("add_288_cast_fp16")]; fp16 const_2405_promoted_to_fp16 = const()[name = string("const_2405_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_103_cast_fp16 = pow(x = add_288_cast_fp16, y = const_2405_promoted_to_fp16)[name = string("pow_103_cast_fp16")]; tensor mean_102_axes_0 = const()[name = string("mean_102_axes_0"), val = tensor([-1])]; bool mean_102_keep_dims_0 = const()[name = string("mean_102_keep_dims_0"), val = bool(true)]; tensor mean_102_cast_fp16 = reduce_mean(axes = mean_102_axes_0, keep_dims = mean_102_keep_dims_0, x = pow_103_cast_fp16)[name = string("mean_102_cast_fp16")]; fp16 const_2408_to_fp16 = const()[name = string("const_2408_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_289_cast_fp16 = add(x = mean_102_cast_fp16, y = const_2408_to_fp16)[name = string("add_289_cast_fp16")]; fp32 rsqrt_102_epsilon_0 = const()[name = string("rsqrt_102_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_102_cast_fp16 = rsqrt(epsilon = rsqrt_102_epsilon_0, x = add_289_cast_fp16)[name = string("rsqrt_102_cast_fp16")]; tensor mul_357_cast_fp16 = mul(x = add_288_cast_fp16, y = rsqrt_102_cast_fp16)[name = string("mul_357_cast_fp16")]; tensor add_290_to_fp16 = const()[name = string("add_290_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274558080)))]; tensor mul_358_cast_fp16 = mul(x = mul_357_cast_fp16, y = add_290_to_fp16)[name = string("mul_358_cast_fp16")]; tensor p_st_0_model_layers_17_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274559680))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275149568))))[name = string("p_st_0_model_layers_17_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_119_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_17_self_attn_q_proj_weight_to_fp16_quantized, x = mul_358_cast_fp16)[name = string("linear_119_cast_fp16")]; tensor const_2412 = const()[name = string("const_2412"), val = tensor([1, 512, -1, 256])]; tensor view_102_cast_fp16 = reshape(shape = const_2412, x = linear_119_cast_fp16)[name = string("view_102_cast_fp16")]; tensor transpose_109_perm_0 = const()[name = string("transpose_109_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_17_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275151168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275347840))))[name = string("p_st_0_model_layers_17_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_120_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_17_self_attn_k_proj_weight_to_fp16_quantized, x = mul_358_cast_fp16)[name = string("linear_120_cast_fp16")]; tensor const_2415 = const()[name = string("const_2415"), val = tensor([1, 512, -1, 256])]; tensor view_103_cast_fp16 = reshape(shape = const_2415, x = linear_120_cast_fp16)[name = string("view_103_cast_fp16")]; tensor transpose_110_perm_0 = const()[name = string("transpose_110_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_17_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275348416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275545088))))[name = string("p_st_0_model_layers_17_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_121_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_17_self_attn_v_proj_weight_to_fp16_quantized, x = mul_358_cast_fp16)[name = string("linear_121_cast_fp16")]; tensor const_2418 = const()[name = string("const_2418"), val = tensor([1, 512, -1, 256])]; tensor view_104_cast_fp16 = reshape(shape = const_2418, x = linear_121_cast_fp16)[name = string("view_104_cast_fp16")]; tensor transpose_111_perm_0 = const()[name = string("transpose_111_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2422_promoted_to_fp16 = const()[name = string("const_2422_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_109_cast_fp16 = transpose(perm = transpose_109_perm_0, x = view_102_cast_fp16)[name = string("transpose_27")]; tensor pow_104_cast_fp16 = pow(x = transpose_109_cast_fp16, y = const_2422_promoted_to_fp16)[name = string("pow_104_cast_fp16")]; tensor mean_103_axes_0 = const()[name = string("mean_103_axes_0"), val = tensor([-1])]; bool mean_103_keep_dims_0 = const()[name = string("mean_103_keep_dims_0"), val = bool(true)]; tensor mean_103_cast_fp16 = reduce_mean(axes = mean_103_axes_0, keep_dims = mean_103_keep_dims_0, x = pow_104_cast_fp16)[name = string("mean_103_cast_fp16")]; fp16 const_2425_to_fp16 = const()[name = string("const_2425_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_291_cast_fp16 = add(x = mean_103_cast_fp16, y = const_2425_to_fp16)[name = string("add_291_cast_fp16")]; fp32 rsqrt_103_epsilon_0 = const()[name = string("rsqrt_103_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_103_cast_fp16 = rsqrt(epsilon = rsqrt_103_epsilon_0, x = add_291_cast_fp16)[name = string("rsqrt_103_cast_fp16")]; tensor mul_359_cast_fp16 = mul(x = transpose_109_cast_fp16, y = rsqrt_103_cast_fp16)[name = string("mul_359_cast_fp16")]; tensor add_292_to_fp16 = const()[name = string("add_292_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275545664)))]; tensor mul_360_cast_fp16 = mul(x = mul_359_cast_fp16, y = add_292_to_fp16)[name = string("mul_360_cast_fp16")]; fp16 const_2430_promoted_to_fp16 = const()[name = string("const_2430_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_110_cast_fp16 = transpose(perm = transpose_110_perm_0, x = view_103_cast_fp16)[name = string("transpose_26")]; tensor pow_105_cast_fp16 = pow(x = transpose_110_cast_fp16, y = const_2430_promoted_to_fp16)[name = string("pow_105_cast_fp16")]; tensor mean_104_axes_0 = const()[name = string("mean_104_axes_0"), val = tensor([-1])]; bool mean_104_keep_dims_0 = const()[name = string("mean_104_keep_dims_0"), val = bool(true)]; tensor mean_104_cast_fp16 = reduce_mean(axes = mean_104_axes_0, keep_dims = mean_104_keep_dims_0, x = pow_105_cast_fp16)[name = string("mean_104_cast_fp16")]; fp16 const_2433_to_fp16 = const()[name = string("const_2433_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_293_cast_fp16 = add(x = mean_104_cast_fp16, y = const_2433_to_fp16)[name = string("add_293_cast_fp16")]; fp32 rsqrt_104_epsilon_0 = const()[name = string("rsqrt_104_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_104_cast_fp16 = rsqrt(epsilon = rsqrt_104_epsilon_0, x = add_293_cast_fp16)[name = string("rsqrt_104_cast_fp16")]; tensor mul_361_cast_fp16 = mul(x = transpose_110_cast_fp16, y = rsqrt_104_cast_fp16)[name = string("mul_361_cast_fp16")]; tensor add_294_to_fp16 = const()[name = string("add_294_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275546240)))]; tensor mul_362_cast_fp16 = mul(x = mul_361_cast_fp16, y = add_294_to_fp16)[name = string("mul_362_cast_fp16")]; tensor mul_363_cast_fp16 = mul(x = mul_360_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_363_cast_fp16")]; tensor slice_452_begin_0 = const()[name = string("slice_452_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_452_end_0 = const()[name = string("slice_452_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_452_end_mask_0 = const()[name = string("slice_452_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_452_cast_fp16 = slice_by_index(begin = slice_452_begin_0, end = slice_452_end_0, end_mask = slice_452_end_mask_0, x = mul_360_cast_fp16)[name = string("slice_452_cast_fp16")]; tensor slice_453_begin_0 = const()[name = string("slice_453_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_453_end_0 = const()[name = string("slice_453_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_453_end_mask_0 = const()[name = string("slice_453_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_453_cast_fp16 = slice_by_index(begin = slice_453_begin_0, end = slice_453_end_0, end_mask = slice_453_end_mask_0, x = mul_360_cast_fp16)[name = string("slice_453_cast_fp16")]; fp16 const_2445_promoted_to_fp16 = const()[name = string("const_2445_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_34_cast_fp16 = mul(x = slice_453_cast_fp16, y = const_2445_promoted_to_fp16)[name = string("neg_34_cast_fp16")]; int32 const_2446 = const()[name = string("const_2446"), val = int32(-1)]; bool cat_92_interleave_0 = const()[name = string("cat_92_interleave_0"), val = bool(false)]; tensor cat_92_cast_fp16 = concat(axis = const_2446, interleave = cat_92_interleave_0, values = (neg_34_cast_fp16, slice_452_cast_fp16))[name = string("cat_92_cast_fp16")]; tensor mul_364_cast_fp16 = mul(x = cat_92_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_364_cast_fp16")]; tensor add_295_cast_fp16 = add(x = mul_363_cast_fp16, y = mul_364_cast_fp16)[name = string("add_295_cast_fp16")]; tensor mul_365_cast_fp16 = mul(x = mul_362_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_365_cast_fp16")]; tensor slice_454_begin_0 = const()[name = string("slice_454_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_454_end_0 = const()[name = string("slice_454_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_454_end_mask_0 = const()[name = string("slice_454_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_454_cast_fp16 = slice_by_index(begin = slice_454_begin_0, end = slice_454_end_0, end_mask = slice_454_end_mask_0, x = mul_362_cast_fp16)[name = string("slice_454_cast_fp16")]; tensor slice_455_begin_0 = const()[name = string("slice_455_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_455_end_0 = const()[name = string("slice_455_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_455_end_mask_0 = const()[name = string("slice_455_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_455_cast_fp16 = slice_by_index(begin = slice_455_begin_0, end = slice_455_end_0, end_mask = slice_455_end_mask_0, x = mul_362_cast_fp16)[name = string("slice_455_cast_fp16")]; fp16 const_2453_promoted_to_fp16 = const()[name = string("const_2453_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_35_cast_fp16 = mul(x = slice_455_cast_fp16, y = const_2453_promoted_to_fp16)[name = string("neg_35_cast_fp16")]; int32 const_2454 = const()[name = string("const_2454"), val = int32(-1)]; bool cat_93_interleave_0 = const()[name = string("cat_93_interleave_0"), val = bool(false)]; tensor cat_93_cast_fp16 = concat(axis = const_2454, interleave = cat_93_interleave_0, values = (neg_35_cast_fp16, slice_454_cast_fp16))[name = string("cat_93_cast_fp16")]; tensor mul_366_cast_fp16 = mul(x = cat_93_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_366_cast_fp16")]; tensor add_296_cast_fp16 = add(x = mul_365_cast_fp16, y = mul_366_cast_fp16)[name = string("add_296_cast_fp16")]; int32 const_2455 = const()[name = string("const_2455"), val = int32(-2)]; bool cat_94_interleave_0 = const()[name = string("cat_94_interleave_0"), val = bool(false)]; tensor cat_94_cast_fp16 = concat(axis = const_2455, interleave = cat_94_interleave_0, values = add_296_cast_fp16)[name = string("cat_94_cast_fp16")]; int32 const_2456 = const()[name = string("const_2456"), val = int32(-2)]; bool cat_95_interleave_0 = const()[name = string("cat_95_interleave_0"), val = bool(false)]; tensor transpose_111_cast_fp16 = transpose(perm = transpose_111_perm_0, x = view_104_cast_fp16)[name = string("transpose_25")]; tensor cat_95_cast_fp16 = concat(axis = const_2456, interleave = cat_95_interleave_0, values = transpose_111_cast_fp16)[name = string("cat_95_cast_fp16")]; tensor unsqueeze_147_axes_0 = const()[name = string("unsqueeze_147_axes_0"), val = tensor([2])]; tensor unsqueeze_147_cast_fp16 = expand_dims(axes = unsqueeze_147_axes_0, x = cat_94_cast_fp16)[name = string("unsqueeze_147_cast_fp16")]; tensor expand_60_reps_0 = const()[name = string("expand_60_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_60_cast_fp16 = tile(reps = expand_60_reps_0, x = unsqueeze_147_cast_fp16)[name = string("expand_60_cast_fp16")]; tensor const_2471 = const()[name = string("const_2471"), val = tensor([1, 3, 512, 256])]; tensor view_105_cast_fp16 = reshape(shape = const_2471, x = expand_60_cast_fp16)[name = string("view_105_cast_fp16")]; tensor unsqueeze_148_axes_0 = const()[name = string("unsqueeze_148_axes_0"), val = tensor([2])]; tensor unsqueeze_148_cast_fp16 = expand_dims(axes = unsqueeze_148_axes_0, x = cat_95_cast_fp16)[name = string("unsqueeze_148_cast_fp16")]; tensor expand_61_reps_0 = const()[name = string("expand_61_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_61_cast_fp16 = tile(reps = expand_61_reps_0, x = unsqueeze_148_cast_fp16)[name = string("expand_61_cast_fp16")]; tensor const_2486 = const()[name = string("const_2486"), val = tensor([1, 3, 512, 256])]; tensor view_106_cast_fp16 = reshape(shape = const_2486, x = expand_61_cast_fp16)[name = string("view_106_cast_fp16")]; bool matmul_58_transpose_x_1 = const()[name = string("matmul_58_transpose_x_1"), val = bool(false)]; bool matmul_58_transpose_y_1 = const()[name = string("matmul_58_transpose_y_1"), val = bool(true)]; tensor matmul_58_cast_fp16 = matmul(transpose_x = matmul_58_transpose_x_1, transpose_y = matmul_58_transpose_y_1, x = add_295_cast_fp16, y = view_105_cast_fp16)[name = string("matmul_58_cast_fp16")]; fp16 const_2489_to_fp16 = const()[name = string("const_2489_to_fp16"), val = fp16(0x1p-4)]; tensor mul_367_cast_fp16 = mul(x = matmul_58_cast_fp16, y = const_2489_to_fp16)[name = string("mul_367_cast_fp16")]; tensor add_297_cast_fp16 = add(x = mul_367_cast_fp16, y = expand_cast_fp16)[name = string("add_297_cast_fp16")]; int32 const_2499 = const()[name = string("const_2499"), val = int32(-1)]; tensor softmax_17_cast_fp16 = softmax(axis = const_2499, x = add_297_cast_fp16)[name = string("softmax_17_cast_fp16")]; bool matmul_59_transpose_x_0 = const()[name = string("matmul_59_transpose_x_0"), val = bool(false)]; bool matmul_59_transpose_y_0 = const()[name = string("matmul_59_transpose_y_0"), val = bool(false)]; tensor matmul_59_cast_fp16 = matmul(transpose_x = matmul_59_transpose_x_0, transpose_y = matmul_59_transpose_y_0, x = softmax_17_cast_fp16, y = view_106_cast_fp16)[name = string("matmul_59_cast_fp16")]; tensor transpose_113_perm_0 = const()[name = string("transpose_113_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2504 = const()[name = string("const_2504"), val = tensor([1, 512, -1])]; tensor transpose_113_cast_fp16 = transpose(perm = transpose_113_perm_0, x = matmul_59_cast_fp16)[name = string("transpose_24")]; tensor view_107_cast_fp16 = reshape(shape = const_2504, x = transpose_113_cast_fp16)[name = string("view_107_cast_fp16")]; tensor p_st_0_model_layers_17_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275546816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276136704))))[name = string("p_st_0_model_layers_17_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_122_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_17_self_attn_o_proj_weight_to_fp16_quantized, x = view_107_cast_fp16)[name = string("linear_122_cast_fp16")]; fp16 const_2506_promoted_to_fp16 = const()[name = string("const_2506_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_106_cast_fp16 = pow(x = linear_122_cast_fp16, y = const_2506_promoted_to_fp16)[name = string("pow_106_cast_fp16")]; tensor mean_105_axes_0 = const()[name = string("mean_105_axes_0"), val = tensor([-1])]; bool mean_105_keep_dims_0 = const()[name = string("mean_105_keep_dims_0"), val = bool(true)]; tensor mean_105_cast_fp16 = reduce_mean(axes = mean_105_axes_0, keep_dims = mean_105_keep_dims_0, x = pow_106_cast_fp16)[name = string("mean_105_cast_fp16")]; fp16 const_2509_to_fp16 = const()[name = string("const_2509_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_298_cast_fp16 = add(x = mean_105_cast_fp16, y = const_2509_to_fp16)[name = string("add_298_cast_fp16")]; fp32 rsqrt_105_epsilon_0 = const()[name = string("rsqrt_105_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_105_cast_fp16 = rsqrt(epsilon = rsqrt_105_epsilon_0, x = add_298_cast_fp16)[name = string("rsqrt_105_cast_fp16")]; tensor mul_368_cast_fp16 = mul(x = linear_122_cast_fp16, y = rsqrt_105_cast_fp16)[name = string("mul_368_cast_fp16")]; tensor add_299_to_fp16 = const()[name = string("add_299_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276138304)))]; tensor mul_369_cast_fp16 = mul(x = mul_368_cast_fp16, y = add_299_to_fp16)[name = string("mul_369_cast_fp16")]; tensor add_300_cast_fp16 = add(x = add_288_cast_fp16, y = mul_369_cast_fp16)[name = string("add_300_cast_fp16")]; fp16 const_2514_promoted_to_fp16 = const()[name = string("const_2514_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_107_cast_fp16 = pow(x = add_300_cast_fp16, y = const_2514_promoted_to_fp16)[name = string("pow_107_cast_fp16")]; tensor mean_106_axes_0 = const()[name = string("mean_106_axes_0"), val = tensor([-1])]; bool mean_106_keep_dims_0 = const()[name = string("mean_106_keep_dims_0"), val = bool(true)]; tensor mean_106_cast_fp16 = reduce_mean(axes = mean_106_axes_0, keep_dims = mean_106_keep_dims_0, x = pow_107_cast_fp16)[name = string("mean_106_cast_fp16")]; fp16 const_2517_to_fp16 = const()[name = string("const_2517_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_301_cast_fp16 = add(x = mean_106_cast_fp16, y = const_2517_to_fp16)[name = string("add_301_cast_fp16")]; fp32 rsqrt_106_epsilon_0 = const()[name = string("rsqrt_106_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_106_cast_fp16 = rsqrt(epsilon = rsqrt_106_epsilon_0, x = add_301_cast_fp16)[name = string("rsqrt_106_cast_fp16")]; tensor mul_370_cast_fp16 = mul(x = add_300_cast_fp16, y = rsqrt_106_cast_fp16)[name = string("mul_370_cast_fp16")]; tensor add_302_to_fp16 = const()[name = string("add_302_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276139904)))]; tensor mul_371_cast_fp16 = mul(x = mul_370_cast_fp16, y = add_302_to_fp16)[name = string("mul_371_cast_fp16")]; tensor p_st_0_model_layers_17_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276141504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(277026304))))[name = string("p_st_0_model_layers_17_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_123_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_17_mlp_gate_proj_weight_to_fp16_quantized, x = mul_371_cast_fp16)[name = string("linear_123_cast_fp16")]; string gelu_17_mode_0 = const()[name = string("gelu_17_mode_0"), val = string("EXACT")]; tensor gelu_17_cast_fp16 = gelu(mode = gelu_17_mode_0, x = linear_123_cast_fp16)[name = string("gelu_17_cast_fp16")]; tensor p_st_0_model_layers_17_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(277028672))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(277913472))))[name = string("p_st_0_model_layers_17_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_124_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_17_mlp_up_proj_weight_to_fp16_quantized, x = mul_371_cast_fp16)[name = string("linear_124_cast_fp16")]; tensor mul_372_cast_fp16 = mul(x = gelu_17_cast_fp16, y = linear_124_cast_fp16)[name = string("mul_372_cast_fp16")]; tensor p_st_0_model_layers_17_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(277915840))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278800640))))[name = string("p_st_0_model_layers_17_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_125_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_17_mlp_down_proj_weight_to_fp16_quantized, x = mul_372_cast_fp16)[name = string("linear_125_cast_fp16")]; fp16 const_2522_promoted_to_fp16 = const()[name = string("const_2522_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_108_cast_fp16 = pow(x = linear_125_cast_fp16, y = const_2522_promoted_to_fp16)[name = string("pow_108_cast_fp16")]; tensor mean_107_axes_0 = const()[name = string("mean_107_axes_0"), val = tensor([-1])]; bool mean_107_keep_dims_0 = const()[name = string("mean_107_keep_dims_0"), val = bool(true)]; tensor mean_107_cast_fp16 = reduce_mean(axes = mean_107_axes_0, keep_dims = mean_107_keep_dims_0, x = pow_108_cast_fp16)[name = string("mean_107_cast_fp16")]; fp16 const_2525_to_fp16 = const()[name = string("const_2525_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_303_cast_fp16 = add(x = mean_107_cast_fp16, y = const_2525_to_fp16)[name = string("add_303_cast_fp16")]; fp32 rsqrt_107_epsilon_0 = const()[name = string("rsqrt_107_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_107_cast_fp16 = rsqrt(epsilon = rsqrt_107_epsilon_0, x = add_303_cast_fp16)[name = string("rsqrt_107_cast_fp16")]; tensor mul_373_cast_fp16 = mul(x = linear_125_cast_fp16, y = rsqrt_107_cast_fp16)[name = string("mul_373_cast_fp16")]; tensor add_304_to_fp16 = const()[name = string("add_304_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278802240)))]; tensor mul_374_cast_fp16 = mul(x = mul_373_cast_fp16, y = add_304_to_fp16)[name = string("mul_374_cast_fp16")]; tensor add_305_cast_fp16 = add(x = add_300_cast_fp16, y = mul_374_cast_fp16)[name = string("add_305_cast_fp16")]; fp16 const_2530_promoted_to_fp16 = const()[name = string("const_2530_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_109_cast_fp16 = pow(x = add_305_cast_fp16, y = const_2530_promoted_to_fp16)[name = string("pow_109_cast_fp16")]; tensor mean_108_axes_0 = const()[name = string("mean_108_axes_0"), val = tensor([-1])]; bool mean_108_keep_dims_0 = const()[name = string("mean_108_keep_dims_0"), val = bool(true)]; tensor mean_108_cast_fp16 = reduce_mean(axes = mean_108_axes_0, keep_dims = mean_108_keep_dims_0, x = pow_109_cast_fp16)[name = string("mean_108_cast_fp16")]; fp16 const_2533_to_fp16 = const()[name = string("const_2533_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_306_cast_fp16 = add(x = mean_108_cast_fp16, y = const_2533_to_fp16)[name = string("add_306_cast_fp16")]; fp32 rsqrt_108_epsilon_0 = const()[name = string("rsqrt_108_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_108_cast_fp16 = rsqrt(epsilon = rsqrt_108_epsilon_0, x = add_306_cast_fp16)[name = string("rsqrt_108_cast_fp16")]; tensor mul_375_cast_fp16 = mul(x = add_305_cast_fp16, y = rsqrt_108_cast_fp16)[name = string("mul_375_cast_fp16")]; tensor add_307_to_fp16 = const()[name = string("add_307_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278803840)))]; tensor mul_376_cast_fp16 = mul(x = mul_375_cast_fp16, y = add_307_to_fp16)[name = string("mul_376_cast_fp16")]; tensor p_st_0_model_layers_18_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278805440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279395328))))[name = string("p_st_0_model_layers_18_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_126_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_18_self_attn_q_proj_weight_to_fp16_quantized, x = mul_376_cast_fp16)[name = string("linear_126_cast_fp16")]; tensor const_2537 = const()[name = string("const_2537"), val = tensor([1, 512, -1, 256])]; tensor view_108_cast_fp16 = reshape(shape = const_2537, x = linear_126_cast_fp16)[name = string("view_108_cast_fp16")]; tensor transpose_114_perm_0 = const()[name = string("transpose_114_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_18_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279396928))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279593600))))[name = string("p_st_0_model_layers_18_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_127_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_18_self_attn_k_proj_weight_to_fp16_quantized, x = mul_376_cast_fp16)[name = string("linear_127_cast_fp16")]; tensor const_2540 = const()[name = string("const_2540"), val = tensor([1, 512, -1, 256])]; tensor view_109_cast_fp16 = reshape(shape = const_2540, x = linear_127_cast_fp16)[name = string("view_109_cast_fp16")]; tensor transpose_115_perm_0 = const()[name = string("transpose_115_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_18_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279594176))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279790848))))[name = string("p_st_0_model_layers_18_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_128_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_18_self_attn_v_proj_weight_to_fp16_quantized, x = mul_376_cast_fp16)[name = string("linear_128_cast_fp16")]; tensor const_2543 = const()[name = string("const_2543"), val = tensor([1, 512, -1, 256])]; tensor view_110_cast_fp16 = reshape(shape = const_2543, x = linear_128_cast_fp16)[name = string("view_110_cast_fp16")]; tensor transpose_116_perm_0 = const()[name = string("transpose_116_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2547_promoted_to_fp16 = const()[name = string("const_2547_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_114_cast_fp16 = transpose(perm = transpose_114_perm_0, x = view_108_cast_fp16)[name = string("transpose_23")]; tensor pow_110_cast_fp16 = pow(x = transpose_114_cast_fp16, y = const_2547_promoted_to_fp16)[name = string("pow_110_cast_fp16")]; tensor mean_109_axes_0 = const()[name = string("mean_109_axes_0"), val = tensor([-1])]; bool mean_109_keep_dims_0 = const()[name = string("mean_109_keep_dims_0"), val = bool(true)]; tensor mean_109_cast_fp16 = reduce_mean(axes = mean_109_axes_0, keep_dims = mean_109_keep_dims_0, x = pow_110_cast_fp16)[name = string("mean_109_cast_fp16")]; fp16 const_2550_to_fp16 = const()[name = string("const_2550_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_308_cast_fp16 = add(x = mean_109_cast_fp16, y = const_2550_to_fp16)[name = string("add_308_cast_fp16")]; fp32 rsqrt_109_epsilon_0 = const()[name = string("rsqrt_109_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_109_cast_fp16 = rsqrt(epsilon = rsqrt_109_epsilon_0, x = add_308_cast_fp16)[name = string("rsqrt_109_cast_fp16")]; tensor mul_377_cast_fp16 = mul(x = transpose_114_cast_fp16, y = rsqrt_109_cast_fp16)[name = string("mul_377_cast_fp16")]; tensor add_309_to_fp16 = const()[name = string("add_309_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279791424)))]; tensor mul_378_cast_fp16 = mul(x = mul_377_cast_fp16, y = add_309_to_fp16)[name = string("mul_378_cast_fp16")]; fp16 const_2555_promoted_to_fp16 = const()[name = string("const_2555_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_115_cast_fp16 = transpose(perm = transpose_115_perm_0, x = view_109_cast_fp16)[name = string("transpose_22")]; tensor pow_111_cast_fp16 = pow(x = transpose_115_cast_fp16, y = const_2555_promoted_to_fp16)[name = string("pow_111_cast_fp16")]; tensor mean_110_axes_0 = const()[name = string("mean_110_axes_0"), val = tensor([-1])]; bool mean_110_keep_dims_0 = const()[name = string("mean_110_keep_dims_0"), val = bool(true)]; tensor mean_110_cast_fp16 = reduce_mean(axes = mean_110_axes_0, keep_dims = mean_110_keep_dims_0, x = pow_111_cast_fp16)[name = string("mean_110_cast_fp16")]; fp16 const_2558_to_fp16 = const()[name = string("const_2558_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_310_cast_fp16 = add(x = mean_110_cast_fp16, y = const_2558_to_fp16)[name = string("add_310_cast_fp16")]; fp32 rsqrt_110_epsilon_0 = const()[name = string("rsqrt_110_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_110_cast_fp16 = rsqrt(epsilon = rsqrt_110_epsilon_0, x = add_310_cast_fp16)[name = string("rsqrt_110_cast_fp16")]; tensor mul_379_cast_fp16 = mul(x = transpose_115_cast_fp16, y = rsqrt_110_cast_fp16)[name = string("mul_379_cast_fp16")]; tensor add_311_to_fp16 = const()[name = string("add_311_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279792000)))]; tensor mul_380_cast_fp16 = mul(x = mul_379_cast_fp16, y = add_311_to_fp16)[name = string("mul_380_cast_fp16")]; tensor mul_381_cast_fp16 = mul(x = mul_378_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_381_cast_fp16")]; tensor slice_467_begin_0 = const()[name = string("slice_467_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_467_end_0 = const()[name = string("slice_467_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_467_end_mask_0 = const()[name = string("slice_467_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_467_cast_fp16 = slice_by_index(begin = slice_467_begin_0, end = slice_467_end_0, end_mask = slice_467_end_mask_0, x = mul_378_cast_fp16)[name = string("slice_467_cast_fp16")]; tensor slice_468_begin_0 = const()[name = string("slice_468_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_468_end_0 = const()[name = string("slice_468_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_468_end_mask_0 = const()[name = string("slice_468_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_468_cast_fp16 = slice_by_index(begin = slice_468_begin_0, end = slice_468_end_0, end_mask = slice_468_end_mask_0, x = mul_378_cast_fp16)[name = string("slice_468_cast_fp16")]; fp16 const_2570_promoted_to_fp16 = const()[name = string("const_2570_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_36_cast_fp16 = mul(x = slice_468_cast_fp16, y = const_2570_promoted_to_fp16)[name = string("neg_36_cast_fp16")]; int32 const_2571 = const()[name = string("const_2571"), val = int32(-1)]; bool cat_96_interleave_0 = const()[name = string("cat_96_interleave_0"), val = bool(false)]; tensor cat_96_cast_fp16 = concat(axis = const_2571, interleave = cat_96_interleave_0, values = (neg_36_cast_fp16, slice_467_cast_fp16))[name = string("cat_96_cast_fp16")]; tensor mul_382_cast_fp16 = mul(x = cat_96_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_382_cast_fp16")]; tensor add_312_cast_fp16 = add(x = mul_381_cast_fp16, y = mul_382_cast_fp16)[name = string("add_312_cast_fp16")]; tensor mul_383_cast_fp16 = mul(x = mul_380_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_383_cast_fp16")]; tensor slice_469_begin_0 = const()[name = string("slice_469_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_469_end_0 = const()[name = string("slice_469_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_469_end_mask_0 = const()[name = string("slice_469_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_469_cast_fp16 = slice_by_index(begin = slice_469_begin_0, end = slice_469_end_0, end_mask = slice_469_end_mask_0, x = mul_380_cast_fp16)[name = string("slice_469_cast_fp16")]; tensor slice_470_begin_0 = const()[name = string("slice_470_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_470_end_0 = const()[name = string("slice_470_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_470_end_mask_0 = const()[name = string("slice_470_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_470_cast_fp16 = slice_by_index(begin = slice_470_begin_0, end = slice_470_end_0, end_mask = slice_470_end_mask_0, x = mul_380_cast_fp16)[name = string("slice_470_cast_fp16")]; fp16 const_2578_promoted_to_fp16 = const()[name = string("const_2578_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_37_cast_fp16 = mul(x = slice_470_cast_fp16, y = const_2578_promoted_to_fp16)[name = string("neg_37_cast_fp16")]; int32 const_2579 = const()[name = string("const_2579"), val = int32(-1)]; bool cat_97_interleave_0 = const()[name = string("cat_97_interleave_0"), val = bool(false)]; tensor cat_97_cast_fp16 = concat(axis = const_2579, interleave = cat_97_interleave_0, values = (neg_37_cast_fp16, slice_469_cast_fp16))[name = string("cat_97_cast_fp16")]; tensor mul_384_cast_fp16 = mul(x = cat_97_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_384_cast_fp16")]; tensor add_313_cast_fp16 = add(x = mul_383_cast_fp16, y = mul_384_cast_fp16)[name = string("add_313_cast_fp16")]; int32 const_2580 = const()[name = string("const_2580"), val = int32(-2)]; bool cat_98_interleave_0 = const()[name = string("cat_98_interleave_0"), val = bool(false)]; tensor cat_98_cast_fp16 = concat(axis = const_2580, interleave = cat_98_interleave_0, values = add_313_cast_fp16)[name = string("cat_98_cast_fp16")]; int32 const_2581 = const()[name = string("const_2581"), val = int32(-2)]; bool cat_99_interleave_0 = const()[name = string("cat_99_interleave_0"), val = bool(false)]; tensor transpose_116_cast_fp16 = transpose(perm = transpose_116_perm_0, x = view_110_cast_fp16)[name = string("transpose_21")]; tensor cat_99_cast_fp16 = concat(axis = const_2581, interleave = cat_99_interleave_0, values = transpose_116_cast_fp16)[name = string("cat_99_cast_fp16")]; tensor unsqueeze_151_axes_0 = const()[name = string("unsqueeze_151_axes_0"), val = tensor([2])]; tensor unsqueeze_151_cast_fp16 = expand_dims(axes = unsqueeze_151_axes_0, x = cat_98_cast_fp16)[name = string("unsqueeze_151_cast_fp16")]; tensor expand_62_reps_0 = const()[name = string("expand_62_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_62_cast_fp16 = tile(reps = expand_62_reps_0, x = unsqueeze_151_cast_fp16)[name = string("expand_62_cast_fp16")]; tensor const_2596 = const()[name = string("const_2596"), val = tensor([1, 3, 512, 256])]; tensor view_111_cast_fp16 = reshape(shape = const_2596, x = expand_62_cast_fp16)[name = string("view_111_cast_fp16")]; tensor unsqueeze_152_axes_0 = const()[name = string("unsqueeze_152_axes_0"), val = tensor([2])]; tensor unsqueeze_152_cast_fp16 = expand_dims(axes = unsqueeze_152_axes_0, x = cat_99_cast_fp16)[name = string("unsqueeze_152_cast_fp16")]; tensor expand_63_reps_0 = const()[name = string("expand_63_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_63_cast_fp16 = tile(reps = expand_63_reps_0, x = unsqueeze_152_cast_fp16)[name = string("expand_63_cast_fp16")]; tensor const_2611 = const()[name = string("const_2611"), val = tensor([1, 3, 512, 256])]; tensor view_112_cast_fp16 = reshape(shape = const_2611, x = expand_63_cast_fp16)[name = string("view_112_cast_fp16")]; bool matmul_60_transpose_x_1 = const()[name = string("matmul_60_transpose_x_1"), val = bool(false)]; bool matmul_60_transpose_y_1 = const()[name = string("matmul_60_transpose_y_1"), val = bool(true)]; tensor matmul_60_cast_fp16 = matmul(transpose_x = matmul_60_transpose_x_1, transpose_y = matmul_60_transpose_y_1, x = add_312_cast_fp16, y = view_111_cast_fp16)[name = string("matmul_60_cast_fp16")]; fp16 const_2614_to_fp16 = const()[name = string("const_2614_to_fp16"), val = fp16(0x1p-4)]; tensor mul_385_cast_fp16 = mul(x = matmul_60_cast_fp16, y = const_2614_to_fp16)[name = string("mul_385_cast_fp16")]; tensor add_314_cast_fp16 = add(x = mul_385_cast_fp16, y = expand_cast_fp16)[name = string("add_314_cast_fp16")]; int32 const_2624 = const()[name = string("const_2624"), val = int32(-1)]; tensor softmax_18_cast_fp16 = softmax(axis = const_2624, x = add_314_cast_fp16)[name = string("softmax_18_cast_fp16")]; bool matmul_61_transpose_x_0 = const()[name = string("matmul_61_transpose_x_0"), val = bool(false)]; bool matmul_61_transpose_y_0 = const()[name = string("matmul_61_transpose_y_0"), val = bool(false)]; tensor matmul_61_cast_fp16 = matmul(transpose_x = matmul_61_transpose_x_0, transpose_y = matmul_61_transpose_y_0, x = softmax_18_cast_fp16, y = view_112_cast_fp16)[name = string("matmul_61_cast_fp16")]; tensor transpose_118_perm_0 = const()[name = string("transpose_118_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2629 = const()[name = string("const_2629"), val = tensor([1, 512, -1])]; tensor transpose_118_cast_fp16 = transpose(perm = transpose_118_perm_0, x = matmul_61_cast_fp16)[name = string("transpose_20")]; tensor view_113_cast_fp16 = reshape(shape = const_2629, x = transpose_118_cast_fp16)[name = string("view_113_cast_fp16")]; tensor p_st_0_model_layers_18_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(279792576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280382464))))[name = string("p_st_0_model_layers_18_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_129_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_18_self_attn_o_proj_weight_to_fp16_quantized, x = view_113_cast_fp16)[name = string("linear_129_cast_fp16")]; fp16 const_2631_promoted_to_fp16 = const()[name = string("const_2631_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_112_cast_fp16 = pow(x = linear_129_cast_fp16, y = const_2631_promoted_to_fp16)[name = string("pow_112_cast_fp16")]; tensor mean_111_axes_0 = const()[name = string("mean_111_axes_0"), val = tensor([-1])]; bool mean_111_keep_dims_0 = const()[name = string("mean_111_keep_dims_0"), val = bool(true)]; tensor mean_111_cast_fp16 = reduce_mean(axes = mean_111_axes_0, keep_dims = mean_111_keep_dims_0, x = pow_112_cast_fp16)[name = string("mean_111_cast_fp16")]; fp16 const_2634_to_fp16 = const()[name = string("const_2634_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_315_cast_fp16 = add(x = mean_111_cast_fp16, y = const_2634_to_fp16)[name = string("add_315_cast_fp16")]; fp32 rsqrt_111_epsilon_0 = const()[name = string("rsqrt_111_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_111_cast_fp16 = rsqrt(epsilon = rsqrt_111_epsilon_0, x = add_315_cast_fp16)[name = string("rsqrt_111_cast_fp16")]; tensor mul_386_cast_fp16 = mul(x = linear_129_cast_fp16, y = rsqrt_111_cast_fp16)[name = string("mul_386_cast_fp16")]; tensor add_316_to_fp16 = const()[name = string("add_316_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280384064)))]; tensor mul_387_cast_fp16 = mul(x = mul_386_cast_fp16, y = add_316_to_fp16)[name = string("mul_387_cast_fp16")]; tensor add_317_cast_fp16 = add(x = add_305_cast_fp16, y = mul_387_cast_fp16)[name = string("add_317_cast_fp16")]; fp16 const_2639_promoted_to_fp16 = const()[name = string("const_2639_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_113_cast_fp16 = pow(x = add_317_cast_fp16, y = const_2639_promoted_to_fp16)[name = string("pow_113_cast_fp16")]; tensor mean_112_axes_0 = const()[name = string("mean_112_axes_0"), val = tensor([-1])]; bool mean_112_keep_dims_0 = const()[name = string("mean_112_keep_dims_0"), val = bool(true)]; tensor mean_112_cast_fp16 = reduce_mean(axes = mean_112_axes_0, keep_dims = mean_112_keep_dims_0, x = pow_113_cast_fp16)[name = string("mean_112_cast_fp16")]; fp16 const_2642_to_fp16 = const()[name = string("const_2642_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_318_cast_fp16 = add(x = mean_112_cast_fp16, y = const_2642_to_fp16)[name = string("add_318_cast_fp16")]; fp32 rsqrt_112_epsilon_0 = const()[name = string("rsqrt_112_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_112_cast_fp16 = rsqrt(epsilon = rsqrt_112_epsilon_0, x = add_318_cast_fp16)[name = string("rsqrt_112_cast_fp16")]; tensor mul_388_cast_fp16 = mul(x = add_317_cast_fp16, y = rsqrt_112_cast_fp16)[name = string("mul_388_cast_fp16")]; tensor add_319_to_fp16 = const()[name = string("add_319_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280385664)))]; tensor mul_389_cast_fp16 = mul(x = mul_388_cast_fp16, y = add_319_to_fp16)[name = string("mul_389_cast_fp16")]; tensor p_st_0_model_layers_18_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280387264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(281272064))))[name = string("p_st_0_model_layers_18_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_130_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_18_mlp_gate_proj_weight_to_fp16_quantized, x = mul_389_cast_fp16)[name = string("linear_130_cast_fp16")]; string gelu_18_mode_0 = const()[name = string("gelu_18_mode_0"), val = string("EXACT")]; tensor gelu_18_cast_fp16 = gelu(mode = gelu_18_mode_0, x = linear_130_cast_fp16)[name = string("gelu_18_cast_fp16")]; tensor p_st_0_model_layers_18_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(281274432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282159232))))[name = string("p_st_0_model_layers_18_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_131_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_18_mlp_up_proj_weight_to_fp16_quantized, x = mul_389_cast_fp16)[name = string("linear_131_cast_fp16")]; tensor mul_390_cast_fp16 = mul(x = gelu_18_cast_fp16, y = linear_131_cast_fp16)[name = string("mul_390_cast_fp16")]; tensor p_st_0_model_layers_18_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282161600))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283046400))))[name = string("p_st_0_model_layers_18_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_132_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_18_mlp_down_proj_weight_to_fp16_quantized, x = mul_390_cast_fp16)[name = string("linear_132_cast_fp16")]; fp16 const_2647_promoted_to_fp16 = const()[name = string("const_2647_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_114_cast_fp16 = pow(x = linear_132_cast_fp16, y = const_2647_promoted_to_fp16)[name = string("pow_114_cast_fp16")]; tensor mean_113_axes_0 = const()[name = string("mean_113_axes_0"), val = tensor([-1])]; bool mean_113_keep_dims_0 = const()[name = string("mean_113_keep_dims_0"), val = bool(true)]; tensor mean_113_cast_fp16 = reduce_mean(axes = mean_113_axes_0, keep_dims = mean_113_keep_dims_0, x = pow_114_cast_fp16)[name = string("mean_113_cast_fp16")]; fp16 const_2650_to_fp16 = const()[name = string("const_2650_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_320_cast_fp16 = add(x = mean_113_cast_fp16, y = const_2650_to_fp16)[name = string("add_320_cast_fp16")]; fp32 rsqrt_113_epsilon_0 = const()[name = string("rsqrt_113_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_113_cast_fp16 = rsqrt(epsilon = rsqrt_113_epsilon_0, x = add_320_cast_fp16)[name = string("rsqrt_113_cast_fp16")]; tensor mul_391_cast_fp16 = mul(x = linear_132_cast_fp16, y = rsqrt_113_cast_fp16)[name = string("mul_391_cast_fp16")]; tensor add_321_to_fp16 = const()[name = string("add_321_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283048000)))]; tensor mul_392_cast_fp16 = mul(x = mul_391_cast_fp16, y = add_321_to_fp16)[name = string("mul_392_cast_fp16")]; tensor add_322_cast_fp16 = add(x = add_317_cast_fp16, y = mul_392_cast_fp16)[name = string("add_322_cast_fp16")]; fp16 const_2655_promoted_to_fp16 = const()[name = string("const_2655_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_115_cast_fp16 = pow(x = add_322_cast_fp16, y = const_2655_promoted_to_fp16)[name = string("pow_115_cast_fp16")]; tensor mean_114_axes_0 = const()[name = string("mean_114_axes_0"), val = tensor([-1])]; bool mean_114_keep_dims_0 = const()[name = string("mean_114_keep_dims_0"), val = bool(true)]; tensor mean_114_cast_fp16 = reduce_mean(axes = mean_114_axes_0, keep_dims = mean_114_keep_dims_0, x = pow_115_cast_fp16)[name = string("mean_114_cast_fp16")]; fp16 const_2658_to_fp16 = const()[name = string("const_2658_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_323_cast_fp16 = add(x = mean_114_cast_fp16, y = const_2658_to_fp16)[name = string("add_323_cast_fp16")]; fp32 rsqrt_114_epsilon_0 = const()[name = string("rsqrt_114_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_114_cast_fp16 = rsqrt(epsilon = rsqrt_114_epsilon_0, x = add_323_cast_fp16)[name = string("rsqrt_114_cast_fp16")]; tensor mul_393_cast_fp16 = mul(x = add_322_cast_fp16, y = rsqrt_114_cast_fp16)[name = string("mul_393_cast_fp16")]; tensor add_324_to_fp16 = const()[name = string("add_324_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283049600)))]; tensor mul_394_cast_fp16 = mul(x = mul_393_cast_fp16, y = add_324_to_fp16)[name = string("mul_394_cast_fp16")]; tensor p_st_0_model_layers_19_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283051200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283641088))))[name = string("p_st_0_model_layers_19_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_133_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_19_self_attn_q_proj_weight_to_fp16_quantized, x = mul_394_cast_fp16)[name = string("linear_133_cast_fp16")]; tensor const_2662 = const()[name = string("const_2662"), val = tensor([1, 512, -1, 256])]; tensor view_114_cast_fp16 = reshape(shape = const_2662, x = linear_133_cast_fp16)[name = string("view_114_cast_fp16")]; tensor transpose_119_perm_0 = const()[name = string("transpose_119_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_19_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283642688))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283839360))))[name = string("p_st_0_model_layers_19_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_134_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_19_self_attn_k_proj_weight_to_fp16_quantized, x = mul_394_cast_fp16)[name = string("linear_134_cast_fp16")]; tensor const_2665 = const()[name = string("const_2665"), val = tensor([1, 512, -1, 256])]; tensor view_115_cast_fp16 = reshape(shape = const_2665, x = linear_134_cast_fp16)[name = string("view_115_cast_fp16")]; tensor transpose_120_perm_0 = const()[name = string("transpose_120_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_19_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283839936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284036608))))[name = string("p_st_0_model_layers_19_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_135_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_19_self_attn_v_proj_weight_to_fp16_quantized, x = mul_394_cast_fp16)[name = string("linear_135_cast_fp16")]; tensor const_2668 = const()[name = string("const_2668"), val = tensor([1, 512, -1, 256])]; tensor view_116_cast_fp16 = reshape(shape = const_2668, x = linear_135_cast_fp16)[name = string("view_116_cast_fp16")]; tensor transpose_121_perm_0 = const()[name = string("transpose_121_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2672_promoted_to_fp16 = const()[name = string("const_2672_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_119_cast_fp16 = transpose(perm = transpose_119_perm_0, x = view_114_cast_fp16)[name = string("transpose_19")]; tensor pow_116_cast_fp16 = pow(x = transpose_119_cast_fp16, y = const_2672_promoted_to_fp16)[name = string("pow_116_cast_fp16")]; tensor mean_115_axes_0 = const()[name = string("mean_115_axes_0"), val = tensor([-1])]; bool mean_115_keep_dims_0 = const()[name = string("mean_115_keep_dims_0"), val = bool(true)]; tensor mean_115_cast_fp16 = reduce_mean(axes = mean_115_axes_0, keep_dims = mean_115_keep_dims_0, x = pow_116_cast_fp16)[name = string("mean_115_cast_fp16")]; fp16 const_2675_to_fp16 = const()[name = string("const_2675_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_325_cast_fp16 = add(x = mean_115_cast_fp16, y = const_2675_to_fp16)[name = string("add_325_cast_fp16")]; fp32 rsqrt_115_epsilon_0 = const()[name = string("rsqrt_115_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_115_cast_fp16 = rsqrt(epsilon = rsqrt_115_epsilon_0, x = add_325_cast_fp16)[name = string("rsqrt_115_cast_fp16")]; tensor mul_395_cast_fp16 = mul(x = transpose_119_cast_fp16, y = rsqrt_115_cast_fp16)[name = string("mul_395_cast_fp16")]; tensor add_326_to_fp16 = const()[name = string("add_326_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284037184)))]; tensor mul_396_cast_fp16 = mul(x = mul_395_cast_fp16, y = add_326_to_fp16)[name = string("mul_396_cast_fp16")]; fp16 const_2680_promoted_to_fp16 = const()[name = string("const_2680_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_120_cast_fp16 = transpose(perm = transpose_120_perm_0, x = view_115_cast_fp16)[name = string("transpose_18")]; tensor pow_117_cast_fp16 = pow(x = transpose_120_cast_fp16, y = const_2680_promoted_to_fp16)[name = string("pow_117_cast_fp16")]; tensor mean_116_axes_0 = const()[name = string("mean_116_axes_0"), val = tensor([-1])]; bool mean_116_keep_dims_0 = const()[name = string("mean_116_keep_dims_0"), val = bool(true)]; tensor mean_116_cast_fp16 = reduce_mean(axes = mean_116_axes_0, keep_dims = mean_116_keep_dims_0, x = pow_117_cast_fp16)[name = string("mean_116_cast_fp16")]; fp16 const_2683_to_fp16 = const()[name = string("const_2683_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_327_cast_fp16 = add(x = mean_116_cast_fp16, y = const_2683_to_fp16)[name = string("add_327_cast_fp16")]; fp32 rsqrt_116_epsilon_0 = const()[name = string("rsqrt_116_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_116_cast_fp16 = rsqrt(epsilon = rsqrt_116_epsilon_0, x = add_327_cast_fp16)[name = string("rsqrt_116_cast_fp16")]; tensor mul_397_cast_fp16 = mul(x = transpose_120_cast_fp16, y = rsqrt_116_cast_fp16)[name = string("mul_397_cast_fp16")]; tensor add_328_to_fp16 = const()[name = string("add_328_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284037760)))]; tensor mul_398_cast_fp16 = mul(x = mul_397_cast_fp16, y = add_328_to_fp16)[name = string("mul_398_cast_fp16")]; tensor mul_399_cast_fp16 = mul(x = mul_396_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_399_cast_fp16")]; tensor slice_490_begin_0 = const()[name = string("slice_490_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_490_end_0 = const()[name = string("slice_490_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_490_end_mask_0 = const()[name = string("slice_490_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_490_cast_fp16 = slice_by_index(begin = slice_490_begin_0, end = slice_490_end_0, end_mask = slice_490_end_mask_0, x = mul_396_cast_fp16)[name = string("slice_490_cast_fp16")]; tensor slice_491_begin_0 = const()[name = string("slice_491_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_491_end_0 = const()[name = string("slice_491_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_491_end_mask_0 = const()[name = string("slice_491_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_491_cast_fp16 = slice_by_index(begin = slice_491_begin_0, end = slice_491_end_0, end_mask = slice_491_end_mask_0, x = mul_396_cast_fp16)[name = string("slice_491_cast_fp16")]; fp16 const_2695_promoted_to_fp16 = const()[name = string("const_2695_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_38_cast_fp16 = mul(x = slice_491_cast_fp16, y = const_2695_promoted_to_fp16)[name = string("neg_38_cast_fp16")]; int32 const_2696 = const()[name = string("const_2696"), val = int32(-1)]; bool cat_100_interleave_0 = const()[name = string("cat_100_interleave_0"), val = bool(false)]; tensor cat_100_cast_fp16 = concat(axis = const_2696, interleave = cat_100_interleave_0, values = (neg_38_cast_fp16, slice_490_cast_fp16))[name = string("cat_100_cast_fp16")]; tensor mul_400_cast_fp16 = mul(x = cat_100_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_400_cast_fp16")]; tensor add_329_cast_fp16 = add(x = mul_399_cast_fp16, y = mul_400_cast_fp16)[name = string("add_329_cast_fp16")]; tensor mul_401_cast_fp16 = mul(x = mul_398_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_401_cast_fp16")]; tensor slice_492_begin_0 = const()[name = string("slice_492_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_492_end_0 = const()[name = string("slice_492_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_492_end_mask_0 = const()[name = string("slice_492_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_492_cast_fp16 = slice_by_index(begin = slice_492_begin_0, end = slice_492_end_0, end_mask = slice_492_end_mask_0, x = mul_398_cast_fp16)[name = string("slice_492_cast_fp16")]; tensor slice_493_begin_0 = const()[name = string("slice_493_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_493_end_0 = const()[name = string("slice_493_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_493_end_mask_0 = const()[name = string("slice_493_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_493_cast_fp16 = slice_by_index(begin = slice_493_begin_0, end = slice_493_end_0, end_mask = slice_493_end_mask_0, x = mul_398_cast_fp16)[name = string("slice_493_cast_fp16")]; fp16 const_2703_promoted_to_fp16 = const()[name = string("const_2703_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_39_cast_fp16 = mul(x = slice_493_cast_fp16, y = const_2703_promoted_to_fp16)[name = string("neg_39_cast_fp16")]; int32 const_2704 = const()[name = string("const_2704"), val = int32(-1)]; bool cat_101_interleave_0 = const()[name = string("cat_101_interleave_0"), val = bool(false)]; tensor cat_101_cast_fp16 = concat(axis = const_2704, interleave = cat_101_interleave_0, values = (neg_39_cast_fp16, slice_492_cast_fp16))[name = string("cat_101_cast_fp16")]; tensor mul_402_cast_fp16 = mul(x = cat_101_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_402_cast_fp16")]; tensor add_330_cast_fp16 = add(x = mul_401_cast_fp16, y = mul_402_cast_fp16)[name = string("add_330_cast_fp16")]; int32 const_2705 = const()[name = string("const_2705"), val = int32(-2)]; bool cat_102_interleave_0 = const()[name = string("cat_102_interleave_0"), val = bool(false)]; tensor cat_102_cast_fp16 = concat(axis = const_2705, interleave = cat_102_interleave_0, values = add_330_cast_fp16)[name = string("cat_102_cast_fp16")]; int32 const_2706 = const()[name = string("const_2706"), val = int32(-2)]; bool cat_103_interleave_0 = const()[name = string("cat_103_interleave_0"), val = bool(false)]; tensor transpose_121_cast_fp16 = transpose(perm = transpose_121_perm_0, x = view_116_cast_fp16)[name = string("transpose_17")]; tensor cat_103_cast_fp16 = concat(axis = const_2706, interleave = cat_103_interleave_0, values = transpose_121_cast_fp16)[name = string("cat_103_cast_fp16")]; tensor unsqueeze_155_axes_0 = const()[name = string("unsqueeze_155_axes_0"), val = tensor([2])]; tensor unsqueeze_155_cast_fp16 = expand_dims(axes = unsqueeze_155_axes_0, x = cat_102_cast_fp16)[name = string("unsqueeze_155_cast_fp16")]; tensor expand_64_reps_0 = const()[name = string("expand_64_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_64_cast_fp16 = tile(reps = expand_64_reps_0, x = unsqueeze_155_cast_fp16)[name = string("expand_64_cast_fp16")]; tensor const_2721 = const()[name = string("const_2721"), val = tensor([1, 3, 512, 256])]; tensor view_117_cast_fp16 = reshape(shape = const_2721, x = expand_64_cast_fp16)[name = string("view_117_cast_fp16")]; tensor unsqueeze_156_axes_0 = const()[name = string("unsqueeze_156_axes_0"), val = tensor([2])]; tensor unsqueeze_156_cast_fp16 = expand_dims(axes = unsqueeze_156_axes_0, x = cat_103_cast_fp16)[name = string("unsqueeze_156_cast_fp16")]; tensor expand_65_reps_0 = const()[name = string("expand_65_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_65_cast_fp16 = tile(reps = expand_65_reps_0, x = unsqueeze_156_cast_fp16)[name = string("expand_65_cast_fp16")]; tensor const_2736 = const()[name = string("const_2736"), val = tensor([1, 3, 512, 256])]; tensor view_118_cast_fp16 = reshape(shape = const_2736, x = expand_65_cast_fp16)[name = string("view_118_cast_fp16")]; bool matmul_62_transpose_x_1 = const()[name = string("matmul_62_transpose_x_1"), val = bool(false)]; bool matmul_62_transpose_y_1 = const()[name = string("matmul_62_transpose_y_1"), val = bool(true)]; tensor matmul_62_cast_fp16 = matmul(transpose_x = matmul_62_transpose_x_1, transpose_y = matmul_62_transpose_y_1, x = add_329_cast_fp16, y = view_117_cast_fp16)[name = string("matmul_62_cast_fp16")]; fp16 const_2739_to_fp16 = const()[name = string("const_2739_to_fp16"), val = fp16(0x1p-4)]; tensor mul_403_cast_fp16 = mul(x = matmul_62_cast_fp16, y = const_2739_to_fp16)[name = string("mul_403_cast_fp16")]; tensor add_331_cast_fp16 = add(x = mul_403_cast_fp16, y = expand_cast_fp16)[name = string("add_331_cast_fp16")]; int32 const_2749 = const()[name = string("const_2749"), val = int32(-1)]; tensor softmax_19_cast_fp16 = softmax(axis = const_2749, x = add_331_cast_fp16)[name = string("softmax_19_cast_fp16")]; bool matmul_63_transpose_x_0 = const()[name = string("matmul_63_transpose_x_0"), val = bool(false)]; bool matmul_63_transpose_y_0 = const()[name = string("matmul_63_transpose_y_0"), val = bool(false)]; tensor matmul_63_cast_fp16 = matmul(transpose_x = matmul_63_transpose_x_0, transpose_y = matmul_63_transpose_y_0, x = softmax_19_cast_fp16, y = view_118_cast_fp16)[name = string("matmul_63_cast_fp16")]; tensor transpose_123_perm_0 = const()[name = string("transpose_123_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2754 = const()[name = string("const_2754"), val = tensor([1, 512, -1])]; tensor transpose_123_cast_fp16 = transpose(perm = transpose_123_perm_0, x = matmul_63_cast_fp16)[name = string("transpose_16")]; tensor view_119_cast_fp16 = reshape(shape = const_2754, x = transpose_123_cast_fp16)[name = string("view_119_cast_fp16")]; tensor p_st_0_model_layers_19_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284038336))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284628224))))[name = string("p_st_0_model_layers_19_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_136_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_19_self_attn_o_proj_weight_to_fp16_quantized, x = view_119_cast_fp16)[name = string("linear_136_cast_fp16")]; fp16 const_2756_promoted_to_fp16 = const()[name = string("const_2756_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_118_cast_fp16 = pow(x = linear_136_cast_fp16, y = const_2756_promoted_to_fp16)[name = string("pow_118_cast_fp16")]; tensor mean_117_axes_0 = const()[name = string("mean_117_axes_0"), val = tensor([-1])]; bool mean_117_keep_dims_0 = const()[name = string("mean_117_keep_dims_0"), val = bool(true)]; tensor mean_117_cast_fp16 = reduce_mean(axes = mean_117_axes_0, keep_dims = mean_117_keep_dims_0, x = pow_118_cast_fp16)[name = string("mean_117_cast_fp16")]; fp16 const_2759_to_fp16 = const()[name = string("const_2759_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_332_cast_fp16 = add(x = mean_117_cast_fp16, y = const_2759_to_fp16)[name = string("add_332_cast_fp16")]; fp32 rsqrt_117_epsilon_0 = const()[name = string("rsqrt_117_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_117_cast_fp16 = rsqrt(epsilon = rsqrt_117_epsilon_0, x = add_332_cast_fp16)[name = string("rsqrt_117_cast_fp16")]; tensor mul_404_cast_fp16 = mul(x = linear_136_cast_fp16, y = rsqrt_117_cast_fp16)[name = string("mul_404_cast_fp16")]; tensor add_333_to_fp16 = const()[name = string("add_333_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284629824)))]; tensor mul_405_cast_fp16 = mul(x = mul_404_cast_fp16, y = add_333_to_fp16)[name = string("mul_405_cast_fp16")]; tensor add_334_cast_fp16 = add(x = add_322_cast_fp16, y = mul_405_cast_fp16)[name = string("add_334_cast_fp16")]; fp16 const_2764_promoted_to_fp16 = const()[name = string("const_2764_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_119_cast_fp16 = pow(x = add_334_cast_fp16, y = const_2764_promoted_to_fp16)[name = string("pow_119_cast_fp16")]; tensor mean_118_axes_0 = const()[name = string("mean_118_axes_0"), val = tensor([-1])]; bool mean_118_keep_dims_0 = const()[name = string("mean_118_keep_dims_0"), val = bool(true)]; tensor mean_118_cast_fp16 = reduce_mean(axes = mean_118_axes_0, keep_dims = mean_118_keep_dims_0, x = pow_119_cast_fp16)[name = string("mean_118_cast_fp16")]; fp16 const_2767_to_fp16 = const()[name = string("const_2767_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_335_cast_fp16 = add(x = mean_118_cast_fp16, y = const_2767_to_fp16)[name = string("add_335_cast_fp16")]; fp32 rsqrt_118_epsilon_0 = const()[name = string("rsqrt_118_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_118_cast_fp16 = rsqrt(epsilon = rsqrt_118_epsilon_0, x = add_335_cast_fp16)[name = string("rsqrt_118_cast_fp16")]; tensor mul_406_cast_fp16 = mul(x = add_334_cast_fp16, y = rsqrt_118_cast_fp16)[name = string("mul_406_cast_fp16")]; tensor add_336_to_fp16 = const()[name = string("add_336_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284631424)))]; tensor mul_407_cast_fp16 = mul(x = mul_406_cast_fp16, y = add_336_to_fp16)[name = string("mul_407_cast_fp16")]; tensor p_st_0_model_layers_19_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284633024))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(285517824))))[name = string("p_st_0_model_layers_19_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_137_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_19_mlp_gate_proj_weight_to_fp16_quantized, x = mul_407_cast_fp16)[name = string("linear_137_cast_fp16")]; string gelu_19_mode_0 = const()[name = string("gelu_19_mode_0"), val = string("EXACT")]; tensor gelu_19_cast_fp16 = gelu(mode = gelu_19_mode_0, x = linear_137_cast_fp16)[name = string("gelu_19_cast_fp16")]; tensor p_st_0_model_layers_19_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(285520192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286404992))))[name = string("p_st_0_model_layers_19_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_138_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_19_mlp_up_proj_weight_to_fp16_quantized, x = mul_407_cast_fp16)[name = string("linear_138_cast_fp16")]; tensor mul_408_cast_fp16 = mul(x = gelu_19_cast_fp16, y = linear_138_cast_fp16)[name = string("mul_408_cast_fp16")]; tensor p_st_0_model_layers_19_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286407360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287292160))))[name = string("p_st_0_model_layers_19_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_139_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_19_mlp_down_proj_weight_to_fp16_quantized, x = mul_408_cast_fp16)[name = string("linear_139_cast_fp16")]; fp16 const_2772_promoted_to_fp16 = const()[name = string("const_2772_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_120_cast_fp16 = pow(x = linear_139_cast_fp16, y = const_2772_promoted_to_fp16)[name = string("pow_120_cast_fp16")]; tensor mean_119_axes_0 = const()[name = string("mean_119_axes_0"), val = tensor([-1])]; bool mean_119_keep_dims_0 = const()[name = string("mean_119_keep_dims_0"), val = bool(true)]; tensor mean_119_cast_fp16 = reduce_mean(axes = mean_119_axes_0, keep_dims = mean_119_keep_dims_0, x = pow_120_cast_fp16)[name = string("mean_119_cast_fp16")]; fp16 const_2775_to_fp16 = const()[name = string("const_2775_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_337_cast_fp16 = add(x = mean_119_cast_fp16, y = const_2775_to_fp16)[name = string("add_337_cast_fp16")]; fp32 rsqrt_119_epsilon_0 = const()[name = string("rsqrt_119_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_119_cast_fp16 = rsqrt(epsilon = rsqrt_119_epsilon_0, x = add_337_cast_fp16)[name = string("rsqrt_119_cast_fp16")]; tensor mul_409_cast_fp16 = mul(x = linear_139_cast_fp16, y = rsqrt_119_cast_fp16)[name = string("mul_409_cast_fp16")]; tensor add_338_to_fp16 = const()[name = string("add_338_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287293760)))]; tensor mul_410_cast_fp16 = mul(x = mul_409_cast_fp16, y = add_338_to_fp16)[name = string("mul_410_cast_fp16")]; tensor add_339_cast_fp16 = add(x = add_334_cast_fp16, y = mul_410_cast_fp16)[name = string("add_339_cast_fp16")]; fp16 const_2780_promoted_to_fp16 = const()[name = string("const_2780_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_121_cast_fp16 = pow(x = add_339_cast_fp16, y = const_2780_promoted_to_fp16)[name = string("pow_121_cast_fp16")]; tensor mean_120_axes_0 = const()[name = string("mean_120_axes_0"), val = tensor([-1])]; bool mean_120_keep_dims_0 = const()[name = string("mean_120_keep_dims_0"), val = bool(true)]; tensor mean_120_cast_fp16 = reduce_mean(axes = mean_120_axes_0, keep_dims = mean_120_keep_dims_0, x = pow_121_cast_fp16)[name = string("mean_120_cast_fp16")]; fp16 const_2783_to_fp16 = const()[name = string("const_2783_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_340_cast_fp16 = add(x = mean_120_cast_fp16, y = const_2783_to_fp16)[name = string("add_340_cast_fp16")]; fp32 rsqrt_120_epsilon_0 = const()[name = string("rsqrt_120_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_120_cast_fp16 = rsqrt(epsilon = rsqrt_120_epsilon_0, x = add_340_cast_fp16)[name = string("rsqrt_120_cast_fp16")]; tensor mul_411_cast_fp16 = mul(x = add_339_cast_fp16, y = rsqrt_120_cast_fp16)[name = string("mul_411_cast_fp16")]; tensor add_341_to_fp16 = const()[name = string("add_341_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287295360)))]; tensor mul_412_cast_fp16 = mul(x = mul_411_cast_fp16, y = add_341_to_fp16)[name = string("mul_412_cast_fp16")]; tensor p_st_0_model_layers_20_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287296960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287886848))))[name = string("p_st_0_model_layers_20_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_140_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_20_self_attn_q_proj_weight_to_fp16_quantized, x = mul_412_cast_fp16)[name = string("linear_140_cast_fp16")]; tensor const_2787 = const()[name = string("const_2787"), val = tensor([1, 512, -1, 256])]; tensor view_120_cast_fp16 = reshape(shape = const_2787, x = linear_140_cast_fp16)[name = string("view_120_cast_fp16")]; tensor transpose_124_perm_0 = const()[name = string("transpose_124_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_20_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287888448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288085120))))[name = string("p_st_0_model_layers_20_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_141_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_20_self_attn_k_proj_weight_to_fp16_quantized, x = mul_412_cast_fp16)[name = string("linear_141_cast_fp16")]; tensor const_2790 = const()[name = string("const_2790"), val = tensor([1, 512, -1, 256])]; tensor view_121_cast_fp16 = reshape(shape = const_2790, x = linear_141_cast_fp16)[name = string("view_121_cast_fp16")]; tensor transpose_125_perm_0 = const()[name = string("transpose_125_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_20_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288085696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288282368))))[name = string("p_st_0_model_layers_20_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_142_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_20_self_attn_v_proj_weight_to_fp16_quantized, x = mul_412_cast_fp16)[name = string("linear_142_cast_fp16")]; tensor const_2793 = const()[name = string("const_2793"), val = tensor([1, 512, -1, 256])]; tensor view_122_cast_fp16 = reshape(shape = const_2793, x = linear_142_cast_fp16)[name = string("view_122_cast_fp16")]; tensor transpose_126_perm_0 = const()[name = string("transpose_126_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2797_promoted_to_fp16 = const()[name = string("const_2797_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_124_cast_fp16 = transpose(perm = transpose_124_perm_0, x = view_120_cast_fp16)[name = string("transpose_15")]; tensor pow_122_cast_fp16 = pow(x = transpose_124_cast_fp16, y = const_2797_promoted_to_fp16)[name = string("pow_122_cast_fp16")]; tensor mean_121_axes_0 = const()[name = string("mean_121_axes_0"), val = tensor([-1])]; bool mean_121_keep_dims_0 = const()[name = string("mean_121_keep_dims_0"), val = bool(true)]; tensor mean_121_cast_fp16 = reduce_mean(axes = mean_121_axes_0, keep_dims = mean_121_keep_dims_0, x = pow_122_cast_fp16)[name = string("mean_121_cast_fp16")]; fp16 const_2800_to_fp16 = const()[name = string("const_2800_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_342_cast_fp16 = add(x = mean_121_cast_fp16, y = const_2800_to_fp16)[name = string("add_342_cast_fp16")]; fp32 rsqrt_121_epsilon_0 = const()[name = string("rsqrt_121_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_121_cast_fp16 = rsqrt(epsilon = rsqrt_121_epsilon_0, x = add_342_cast_fp16)[name = string("rsqrt_121_cast_fp16")]; tensor mul_413_cast_fp16 = mul(x = transpose_124_cast_fp16, y = rsqrt_121_cast_fp16)[name = string("mul_413_cast_fp16")]; tensor add_343_to_fp16 = const()[name = string("add_343_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288282944)))]; tensor mul_414_cast_fp16 = mul(x = mul_413_cast_fp16, y = add_343_to_fp16)[name = string("mul_414_cast_fp16")]; fp16 const_2805_promoted_to_fp16 = const()[name = string("const_2805_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_125_cast_fp16 = transpose(perm = transpose_125_perm_0, x = view_121_cast_fp16)[name = string("transpose_14")]; tensor pow_123_cast_fp16 = pow(x = transpose_125_cast_fp16, y = const_2805_promoted_to_fp16)[name = string("pow_123_cast_fp16")]; tensor mean_122_axes_0 = const()[name = string("mean_122_axes_0"), val = tensor([-1])]; bool mean_122_keep_dims_0 = const()[name = string("mean_122_keep_dims_0"), val = bool(true)]; tensor mean_122_cast_fp16 = reduce_mean(axes = mean_122_axes_0, keep_dims = mean_122_keep_dims_0, x = pow_123_cast_fp16)[name = string("mean_122_cast_fp16")]; fp16 const_2808_to_fp16 = const()[name = string("const_2808_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_344_cast_fp16 = add(x = mean_122_cast_fp16, y = const_2808_to_fp16)[name = string("add_344_cast_fp16")]; fp32 rsqrt_122_epsilon_0 = const()[name = string("rsqrt_122_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_122_cast_fp16 = rsqrt(epsilon = rsqrt_122_epsilon_0, x = add_344_cast_fp16)[name = string("rsqrt_122_cast_fp16")]; tensor mul_415_cast_fp16 = mul(x = transpose_125_cast_fp16, y = rsqrt_122_cast_fp16)[name = string("mul_415_cast_fp16")]; tensor add_345_to_fp16 = const()[name = string("add_345_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288283520)))]; tensor mul_416_cast_fp16 = mul(x = mul_415_cast_fp16, y = add_345_to_fp16)[name = string("mul_416_cast_fp16")]; tensor mul_417_cast_fp16 = mul(x = mul_414_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_417_cast_fp16")]; tensor slice_513_begin_0 = const()[name = string("slice_513_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_513_end_0 = const()[name = string("slice_513_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_513_end_mask_0 = const()[name = string("slice_513_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_513_cast_fp16 = slice_by_index(begin = slice_513_begin_0, end = slice_513_end_0, end_mask = slice_513_end_mask_0, x = mul_414_cast_fp16)[name = string("slice_513_cast_fp16")]; tensor slice_514_begin_0 = const()[name = string("slice_514_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_514_end_0 = const()[name = string("slice_514_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_514_end_mask_0 = const()[name = string("slice_514_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_514_cast_fp16 = slice_by_index(begin = slice_514_begin_0, end = slice_514_end_0, end_mask = slice_514_end_mask_0, x = mul_414_cast_fp16)[name = string("slice_514_cast_fp16")]; fp16 const_2820_promoted_to_fp16 = const()[name = string("const_2820_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_40_cast_fp16 = mul(x = slice_514_cast_fp16, y = const_2820_promoted_to_fp16)[name = string("neg_40_cast_fp16")]; int32 const_2821 = const()[name = string("const_2821"), val = int32(-1)]; bool cat_104_interleave_0 = const()[name = string("cat_104_interleave_0"), val = bool(false)]; tensor cat_104_cast_fp16 = concat(axis = const_2821, interleave = cat_104_interleave_0, values = (neg_40_cast_fp16, slice_513_cast_fp16))[name = string("cat_104_cast_fp16")]; tensor mul_418_cast_fp16 = mul(x = cat_104_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_418_cast_fp16")]; tensor add_346_cast_fp16 = add(x = mul_417_cast_fp16, y = mul_418_cast_fp16)[name = string("add_346_cast_fp16")]; tensor mul_419_cast_fp16 = mul(x = mul_416_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_419_cast_fp16")]; tensor slice_515_begin_0 = const()[name = string("slice_515_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_515_end_0 = const()[name = string("slice_515_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_515_end_mask_0 = const()[name = string("slice_515_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_515_cast_fp16 = slice_by_index(begin = slice_515_begin_0, end = slice_515_end_0, end_mask = slice_515_end_mask_0, x = mul_416_cast_fp16)[name = string("slice_515_cast_fp16")]; tensor slice_516_begin_0 = const()[name = string("slice_516_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_516_end_0 = const()[name = string("slice_516_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_516_end_mask_0 = const()[name = string("slice_516_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_516_cast_fp16 = slice_by_index(begin = slice_516_begin_0, end = slice_516_end_0, end_mask = slice_516_end_mask_0, x = mul_416_cast_fp16)[name = string("slice_516_cast_fp16")]; fp16 const_2828_promoted_to_fp16 = const()[name = string("const_2828_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_41_cast_fp16 = mul(x = slice_516_cast_fp16, y = const_2828_promoted_to_fp16)[name = string("neg_41_cast_fp16")]; int32 const_2829 = const()[name = string("const_2829"), val = int32(-1)]; bool cat_105_interleave_0 = const()[name = string("cat_105_interleave_0"), val = bool(false)]; tensor cat_105_cast_fp16 = concat(axis = const_2829, interleave = cat_105_interleave_0, values = (neg_41_cast_fp16, slice_515_cast_fp16))[name = string("cat_105_cast_fp16")]; tensor mul_420_cast_fp16 = mul(x = cat_105_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_420_cast_fp16")]; tensor add_347_cast_fp16 = add(x = mul_419_cast_fp16, y = mul_420_cast_fp16)[name = string("add_347_cast_fp16")]; int32 const_2830 = const()[name = string("const_2830"), val = int32(-2)]; bool cat_106_interleave_0 = const()[name = string("cat_106_interleave_0"), val = bool(false)]; tensor cat_106_cast_fp16 = concat(axis = const_2830, interleave = cat_106_interleave_0, values = add_347_cast_fp16)[name = string("cat_106_cast_fp16")]; int32 const_2831 = const()[name = string("const_2831"), val = int32(-2)]; bool cat_107_interleave_0 = const()[name = string("cat_107_interleave_0"), val = bool(false)]; tensor transpose_126_cast_fp16 = transpose(perm = transpose_126_perm_0, x = view_122_cast_fp16)[name = string("transpose_13")]; tensor cat_107_cast_fp16 = concat(axis = const_2831, interleave = cat_107_interleave_0, values = transpose_126_cast_fp16)[name = string("cat_107_cast_fp16")]; tensor unsqueeze_159_axes_0 = const()[name = string("unsqueeze_159_axes_0"), val = tensor([2])]; tensor unsqueeze_159_cast_fp16 = expand_dims(axes = unsqueeze_159_axes_0, x = cat_106_cast_fp16)[name = string("unsqueeze_159_cast_fp16")]; tensor expand_66_reps_0 = const()[name = string("expand_66_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_66_cast_fp16 = tile(reps = expand_66_reps_0, x = unsqueeze_159_cast_fp16)[name = string("expand_66_cast_fp16")]; tensor const_2846 = const()[name = string("const_2846"), val = tensor([1, 3, 512, 256])]; tensor view_123_cast_fp16 = reshape(shape = const_2846, x = expand_66_cast_fp16)[name = string("view_123_cast_fp16")]; tensor unsqueeze_160_axes_0 = const()[name = string("unsqueeze_160_axes_0"), val = tensor([2])]; tensor unsqueeze_160_cast_fp16 = expand_dims(axes = unsqueeze_160_axes_0, x = cat_107_cast_fp16)[name = string("unsqueeze_160_cast_fp16")]; tensor expand_67_reps_0 = const()[name = string("expand_67_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_67_cast_fp16 = tile(reps = expand_67_reps_0, x = unsqueeze_160_cast_fp16)[name = string("expand_67_cast_fp16")]; tensor const_2861 = const()[name = string("const_2861"), val = tensor([1, 3, 512, 256])]; tensor view_124_cast_fp16 = reshape(shape = const_2861, x = expand_67_cast_fp16)[name = string("view_124_cast_fp16")]; bool matmul_64_transpose_x_1 = const()[name = string("matmul_64_transpose_x_1"), val = bool(false)]; bool matmul_64_transpose_y_1 = const()[name = string("matmul_64_transpose_y_1"), val = bool(true)]; tensor matmul_64_cast_fp16 = matmul(transpose_x = matmul_64_transpose_x_1, transpose_y = matmul_64_transpose_y_1, x = add_346_cast_fp16, y = view_123_cast_fp16)[name = string("matmul_64_cast_fp16")]; fp16 const_2864_to_fp16 = const()[name = string("const_2864_to_fp16"), val = fp16(0x1p-4)]; tensor mul_421_cast_fp16 = mul(x = matmul_64_cast_fp16, y = const_2864_to_fp16)[name = string("mul_421_cast_fp16")]; tensor add_348_cast_fp16 = add(x = mul_421_cast_fp16, y = expand_cast_fp16)[name = string("add_348_cast_fp16")]; int32 const_2874 = const()[name = string("const_2874"), val = int32(-1)]; tensor softmax_20_cast_fp16 = softmax(axis = const_2874, x = add_348_cast_fp16)[name = string("softmax_20_cast_fp16")]; bool matmul_65_transpose_x_0 = const()[name = string("matmul_65_transpose_x_0"), val = bool(false)]; bool matmul_65_transpose_y_0 = const()[name = string("matmul_65_transpose_y_0"), val = bool(false)]; tensor matmul_65_cast_fp16 = matmul(transpose_x = matmul_65_transpose_x_0, transpose_y = matmul_65_transpose_y_0, x = softmax_20_cast_fp16, y = view_124_cast_fp16)[name = string("matmul_65_cast_fp16")]; tensor transpose_128_perm_0 = const()[name = string("transpose_128_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_2879 = const()[name = string("const_2879"), val = tensor([1, 512, -1])]; tensor transpose_128_cast_fp16 = transpose(perm = transpose_128_perm_0, x = matmul_65_cast_fp16)[name = string("transpose_12")]; tensor view_125_cast_fp16 = reshape(shape = const_2879, x = transpose_128_cast_fp16)[name = string("view_125_cast_fp16")]; tensor p_st_0_model_layers_20_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288284096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288873984))))[name = string("p_st_0_model_layers_20_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_143_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_20_self_attn_o_proj_weight_to_fp16_quantized, x = view_125_cast_fp16)[name = string("linear_143_cast_fp16")]; fp16 const_2881_promoted_to_fp16 = const()[name = string("const_2881_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_124_cast_fp16 = pow(x = linear_143_cast_fp16, y = const_2881_promoted_to_fp16)[name = string("pow_124_cast_fp16")]; tensor mean_123_axes_0 = const()[name = string("mean_123_axes_0"), val = tensor([-1])]; bool mean_123_keep_dims_0 = const()[name = string("mean_123_keep_dims_0"), val = bool(true)]; tensor mean_123_cast_fp16 = reduce_mean(axes = mean_123_axes_0, keep_dims = mean_123_keep_dims_0, x = pow_124_cast_fp16)[name = string("mean_123_cast_fp16")]; fp16 const_2884_to_fp16 = const()[name = string("const_2884_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_349_cast_fp16 = add(x = mean_123_cast_fp16, y = const_2884_to_fp16)[name = string("add_349_cast_fp16")]; fp32 rsqrt_123_epsilon_0 = const()[name = string("rsqrt_123_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_123_cast_fp16 = rsqrt(epsilon = rsqrt_123_epsilon_0, x = add_349_cast_fp16)[name = string("rsqrt_123_cast_fp16")]; tensor mul_422_cast_fp16 = mul(x = linear_143_cast_fp16, y = rsqrt_123_cast_fp16)[name = string("mul_422_cast_fp16")]; tensor add_350_to_fp16 = const()[name = string("add_350_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288875584)))]; tensor mul_423_cast_fp16 = mul(x = mul_422_cast_fp16, y = add_350_to_fp16)[name = string("mul_423_cast_fp16")]; tensor add_351_cast_fp16 = add(x = add_339_cast_fp16, y = mul_423_cast_fp16)[name = string("add_351_cast_fp16")]; fp16 const_2889_promoted_to_fp16 = const()[name = string("const_2889_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_125_cast_fp16 = pow(x = add_351_cast_fp16, y = const_2889_promoted_to_fp16)[name = string("pow_125_cast_fp16")]; tensor mean_124_axes_0 = const()[name = string("mean_124_axes_0"), val = tensor([-1])]; bool mean_124_keep_dims_0 = const()[name = string("mean_124_keep_dims_0"), val = bool(true)]; tensor mean_124_cast_fp16 = reduce_mean(axes = mean_124_axes_0, keep_dims = mean_124_keep_dims_0, x = pow_125_cast_fp16)[name = string("mean_124_cast_fp16")]; fp16 const_2892_to_fp16 = const()[name = string("const_2892_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_352_cast_fp16 = add(x = mean_124_cast_fp16, y = const_2892_to_fp16)[name = string("add_352_cast_fp16")]; fp32 rsqrt_124_epsilon_0 = const()[name = string("rsqrt_124_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_124_cast_fp16 = rsqrt(epsilon = rsqrt_124_epsilon_0, x = add_352_cast_fp16)[name = string("rsqrt_124_cast_fp16")]; tensor mul_424_cast_fp16 = mul(x = add_351_cast_fp16, y = rsqrt_124_cast_fp16)[name = string("mul_424_cast_fp16")]; tensor add_353_to_fp16 = const()[name = string("add_353_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288877184)))]; tensor mul_425_cast_fp16 = mul(x = mul_424_cast_fp16, y = add_353_to_fp16)[name = string("mul_425_cast_fp16")]; tensor p_st_0_model_layers_20_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288878784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(289763584))))[name = string("p_st_0_model_layers_20_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_144_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_20_mlp_gate_proj_weight_to_fp16_quantized, x = mul_425_cast_fp16)[name = string("linear_144_cast_fp16")]; string gelu_20_mode_0 = const()[name = string("gelu_20_mode_0"), val = string("EXACT")]; tensor gelu_20_cast_fp16 = gelu(mode = gelu_20_mode_0, x = linear_144_cast_fp16)[name = string("gelu_20_cast_fp16")]; tensor p_st_0_model_layers_20_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(289765952))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290650752))))[name = string("p_st_0_model_layers_20_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_145_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_20_mlp_up_proj_weight_to_fp16_quantized, x = mul_425_cast_fp16)[name = string("linear_145_cast_fp16")]; tensor mul_426_cast_fp16 = mul(x = gelu_20_cast_fp16, y = linear_145_cast_fp16)[name = string("mul_426_cast_fp16")]; tensor p_st_0_model_layers_20_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290653120))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291537920))))[name = string("p_st_0_model_layers_20_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_146_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_20_mlp_down_proj_weight_to_fp16_quantized, x = mul_426_cast_fp16)[name = string("linear_146_cast_fp16")]; fp16 const_2897_promoted_to_fp16 = const()[name = string("const_2897_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_126_cast_fp16 = pow(x = linear_146_cast_fp16, y = const_2897_promoted_to_fp16)[name = string("pow_126_cast_fp16")]; tensor mean_125_axes_0 = const()[name = string("mean_125_axes_0"), val = tensor([-1])]; bool mean_125_keep_dims_0 = const()[name = string("mean_125_keep_dims_0"), val = bool(true)]; tensor mean_125_cast_fp16 = reduce_mean(axes = mean_125_axes_0, keep_dims = mean_125_keep_dims_0, x = pow_126_cast_fp16)[name = string("mean_125_cast_fp16")]; fp16 const_2900_to_fp16 = const()[name = string("const_2900_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_354_cast_fp16 = add(x = mean_125_cast_fp16, y = const_2900_to_fp16)[name = string("add_354_cast_fp16")]; fp32 rsqrt_125_epsilon_0 = const()[name = string("rsqrt_125_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_125_cast_fp16 = rsqrt(epsilon = rsqrt_125_epsilon_0, x = add_354_cast_fp16)[name = string("rsqrt_125_cast_fp16")]; tensor mul_427_cast_fp16 = mul(x = linear_146_cast_fp16, y = rsqrt_125_cast_fp16)[name = string("mul_427_cast_fp16")]; tensor add_355_to_fp16 = const()[name = string("add_355_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291539520)))]; tensor mul_428_cast_fp16 = mul(x = mul_427_cast_fp16, y = add_355_to_fp16)[name = string("mul_428_cast_fp16")]; tensor add_356_cast_fp16 = add(x = add_351_cast_fp16, y = mul_428_cast_fp16)[name = string("add_356_cast_fp16")]; fp16 const_2905_promoted_to_fp16 = const()[name = string("const_2905_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_127_cast_fp16 = pow(x = add_356_cast_fp16, y = const_2905_promoted_to_fp16)[name = string("pow_127_cast_fp16")]; tensor mean_126_axes_0 = const()[name = string("mean_126_axes_0"), val = tensor([-1])]; bool mean_126_keep_dims_0 = const()[name = string("mean_126_keep_dims_0"), val = bool(true)]; tensor mean_126_cast_fp16 = reduce_mean(axes = mean_126_axes_0, keep_dims = mean_126_keep_dims_0, x = pow_127_cast_fp16)[name = string("mean_126_cast_fp16")]; fp16 const_2908_to_fp16 = const()[name = string("const_2908_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_357_cast_fp16 = add(x = mean_126_cast_fp16, y = const_2908_to_fp16)[name = string("add_357_cast_fp16")]; fp32 rsqrt_126_epsilon_0 = const()[name = string("rsqrt_126_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_126_cast_fp16 = rsqrt(epsilon = rsqrt_126_epsilon_0, x = add_357_cast_fp16)[name = string("rsqrt_126_cast_fp16")]; tensor mul_429_cast_fp16 = mul(x = add_356_cast_fp16, y = rsqrt_126_cast_fp16)[name = string("mul_429_cast_fp16")]; tensor add_358_to_fp16 = const()[name = string("add_358_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291541120)))]; tensor mul_430_cast_fp16 = mul(x = mul_429_cast_fp16, y = add_358_to_fp16)[name = string("mul_430_cast_fp16")]; tensor p_st_0_model_layers_21_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291542720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292132608))))[name = string("p_st_0_model_layers_21_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_147_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_21_self_attn_q_proj_weight_to_fp16_quantized, x = mul_430_cast_fp16)[name = string("linear_147_cast_fp16")]; tensor const_2912 = const()[name = string("const_2912"), val = tensor([1, 512, -1, 256])]; tensor view_126_cast_fp16 = reshape(shape = const_2912, x = linear_147_cast_fp16)[name = string("view_126_cast_fp16")]; tensor transpose_129_perm_0 = const()[name = string("transpose_129_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_21_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292134208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292330880))))[name = string("p_st_0_model_layers_21_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_148_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_21_self_attn_k_proj_weight_to_fp16_quantized, x = mul_430_cast_fp16)[name = string("linear_148_cast_fp16")]; tensor const_2915 = const()[name = string("const_2915"), val = tensor([1, 512, -1, 256])]; tensor view_127_cast_fp16 = reshape(shape = const_2915, x = linear_148_cast_fp16)[name = string("view_127_cast_fp16")]; tensor transpose_130_perm_0 = const()[name = string("transpose_130_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_21_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292331456))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292528128))))[name = string("p_st_0_model_layers_21_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_149_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_21_self_attn_v_proj_weight_to_fp16_quantized, x = mul_430_cast_fp16)[name = string("linear_149_cast_fp16")]; tensor const_2918 = const()[name = string("const_2918"), val = tensor([1, 512, -1, 256])]; tensor view_128_cast_fp16 = reshape(shape = const_2918, x = linear_149_cast_fp16)[name = string("view_128_cast_fp16")]; tensor transpose_131_perm_0 = const()[name = string("transpose_131_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_2922_promoted_to_fp16 = const()[name = string("const_2922_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_129_cast_fp16 = transpose(perm = transpose_129_perm_0, x = view_126_cast_fp16)[name = string("transpose_11")]; tensor pow_128_cast_fp16 = pow(x = transpose_129_cast_fp16, y = const_2922_promoted_to_fp16)[name = string("pow_128_cast_fp16")]; tensor mean_127_axes_0 = const()[name = string("mean_127_axes_0"), val = tensor([-1])]; bool mean_127_keep_dims_0 = const()[name = string("mean_127_keep_dims_0"), val = bool(true)]; tensor mean_127_cast_fp16 = reduce_mean(axes = mean_127_axes_0, keep_dims = mean_127_keep_dims_0, x = pow_128_cast_fp16)[name = string("mean_127_cast_fp16")]; fp16 const_2925_to_fp16 = const()[name = string("const_2925_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_359_cast_fp16 = add(x = mean_127_cast_fp16, y = const_2925_to_fp16)[name = string("add_359_cast_fp16")]; fp32 rsqrt_127_epsilon_0 = const()[name = string("rsqrt_127_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_127_cast_fp16 = rsqrt(epsilon = rsqrt_127_epsilon_0, x = add_359_cast_fp16)[name = string("rsqrt_127_cast_fp16")]; tensor mul_431_cast_fp16 = mul(x = transpose_129_cast_fp16, y = rsqrt_127_cast_fp16)[name = string("mul_431_cast_fp16")]; tensor add_360_to_fp16 = const()[name = string("add_360_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292528704)))]; tensor mul_432_cast_fp16 = mul(x = mul_431_cast_fp16, y = add_360_to_fp16)[name = string("mul_432_cast_fp16")]; fp16 const_2930_promoted_to_fp16 = const()[name = string("const_2930_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_130_cast_fp16 = transpose(perm = transpose_130_perm_0, x = view_127_cast_fp16)[name = string("transpose_10")]; tensor pow_129_cast_fp16 = pow(x = transpose_130_cast_fp16, y = const_2930_promoted_to_fp16)[name = string("pow_129_cast_fp16")]; tensor mean_128_axes_0 = const()[name = string("mean_128_axes_0"), val = tensor([-1])]; bool mean_128_keep_dims_0 = const()[name = string("mean_128_keep_dims_0"), val = bool(true)]; tensor mean_128_cast_fp16 = reduce_mean(axes = mean_128_axes_0, keep_dims = mean_128_keep_dims_0, x = pow_129_cast_fp16)[name = string("mean_128_cast_fp16")]; fp16 const_2933_to_fp16 = const()[name = string("const_2933_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_361_cast_fp16 = add(x = mean_128_cast_fp16, y = const_2933_to_fp16)[name = string("add_361_cast_fp16")]; fp32 rsqrt_128_epsilon_0 = const()[name = string("rsqrt_128_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_128_cast_fp16 = rsqrt(epsilon = rsqrt_128_epsilon_0, x = add_361_cast_fp16)[name = string("rsqrt_128_cast_fp16")]; tensor mul_433_cast_fp16 = mul(x = transpose_130_cast_fp16, y = rsqrt_128_cast_fp16)[name = string("mul_433_cast_fp16")]; tensor add_362_to_fp16 = const()[name = string("add_362_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292529280)))]; tensor mul_434_cast_fp16 = mul(x = mul_433_cast_fp16, y = add_362_to_fp16)[name = string("mul_434_cast_fp16")]; tensor mul_435_cast_fp16 = mul(x = mul_432_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_435_cast_fp16")]; tensor slice_536_begin_0 = const()[name = string("slice_536_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_536_end_0 = const()[name = string("slice_536_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_536_end_mask_0 = const()[name = string("slice_536_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_536_cast_fp16 = slice_by_index(begin = slice_536_begin_0, end = slice_536_end_0, end_mask = slice_536_end_mask_0, x = mul_432_cast_fp16)[name = string("slice_536_cast_fp16")]; tensor slice_537_begin_0 = const()[name = string("slice_537_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_537_end_0 = const()[name = string("slice_537_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_537_end_mask_0 = const()[name = string("slice_537_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_537_cast_fp16 = slice_by_index(begin = slice_537_begin_0, end = slice_537_end_0, end_mask = slice_537_end_mask_0, x = mul_432_cast_fp16)[name = string("slice_537_cast_fp16")]; fp16 const_2945_promoted_to_fp16 = const()[name = string("const_2945_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_42_cast_fp16 = mul(x = slice_537_cast_fp16, y = const_2945_promoted_to_fp16)[name = string("neg_42_cast_fp16")]; int32 const_2946 = const()[name = string("const_2946"), val = int32(-1)]; bool cat_108_interleave_0 = const()[name = string("cat_108_interleave_0"), val = bool(false)]; tensor cat_108_cast_fp16 = concat(axis = const_2946, interleave = cat_108_interleave_0, values = (neg_42_cast_fp16, slice_536_cast_fp16))[name = string("cat_108_cast_fp16")]; tensor mul_436_cast_fp16 = mul(x = cat_108_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_436_cast_fp16")]; tensor add_363_cast_fp16 = add(x = mul_435_cast_fp16, y = mul_436_cast_fp16)[name = string("add_363_cast_fp16")]; tensor mul_437_cast_fp16 = mul(x = mul_434_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_437_cast_fp16")]; tensor slice_538_begin_0 = const()[name = string("slice_538_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_538_end_0 = const()[name = string("slice_538_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_538_end_mask_0 = const()[name = string("slice_538_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_538_cast_fp16 = slice_by_index(begin = slice_538_begin_0, end = slice_538_end_0, end_mask = slice_538_end_mask_0, x = mul_434_cast_fp16)[name = string("slice_538_cast_fp16")]; tensor slice_539_begin_0 = const()[name = string("slice_539_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_539_end_0 = const()[name = string("slice_539_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_539_end_mask_0 = const()[name = string("slice_539_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_539_cast_fp16 = slice_by_index(begin = slice_539_begin_0, end = slice_539_end_0, end_mask = slice_539_end_mask_0, x = mul_434_cast_fp16)[name = string("slice_539_cast_fp16")]; fp16 const_2953_promoted_to_fp16 = const()[name = string("const_2953_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_43_cast_fp16 = mul(x = slice_539_cast_fp16, y = const_2953_promoted_to_fp16)[name = string("neg_43_cast_fp16")]; int32 const_2954 = const()[name = string("const_2954"), val = int32(-1)]; bool cat_109_interleave_0 = const()[name = string("cat_109_interleave_0"), val = bool(false)]; tensor cat_109_cast_fp16 = concat(axis = const_2954, interleave = cat_109_interleave_0, values = (neg_43_cast_fp16, slice_538_cast_fp16))[name = string("cat_109_cast_fp16")]; tensor mul_438_cast_fp16 = mul(x = cat_109_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_438_cast_fp16")]; tensor add_364_cast_fp16 = add(x = mul_437_cast_fp16, y = mul_438_cast_fp16)[name = string("add_364_cast_fp16")]; int32 const_2955 = const()[name = string("const_2955"), val = int32(-2)]; bool cat_110_interleave_0 = const()[name = string("cat_110_interleave_0"), val = bool(false)]; tensor cat_110_cast_fp16 = concat(axis = const_2955, interleave = cat_110_interleave_0, values = add_364_cast_fp16)[name = string("cat_110_cast_fp16")]; int32 const_2956 = const()[name = string("const_2956"), val = int32(-2)]; bool cat_111_interleave_0 = const()[name = string("cat_111_interleave_0"), val = bool(false)]; tensor transpose_131_cast_fp16 = transpose(perm = transpose_131_perm_0, x = view_128_cast_fp16)[name = string("transpose_9")]; tensor cat_111_cast_fp16 = concat(axis = const_2956, interleave = cat_111_interleave_0, values = transpose_131_cast_fp16)[name = string("cat_111_cast_fp16")]; tensor unsqueeze_163_axes_0 = const()[name = string("unsqueeze_163_axes_0"), val = tensor([2])]; tensor unsqueeze_163_cast_fp16 = expand_dims(axes = unsqueeze_163_axes_0, x = cat_110_cast_fp16)[name = string("unsqueeze_163_cast_fp16")]; tensor expand_68_reps_0 = const()[name = string("expand_68_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_68_cast_fp16 = tile(reps = expand_68_reps_0, x = unsqueeze_163_cast_fp16)[name = string("expand_68_cast_fp16")]; tensor const_2971 = const()[name = string("const_2971"), val = tensor([1, 3, 512, 256])]; tensor view_129_cast_fp16 = reshape(shape = const_2971, x = expand_68_cast_fp16)[name = string("view_129_cast_fp16")]; tensor unsqueeze_164_axes_0 = const()[name = string("unsqueeze_164_axes_0"), val = tensor([2])]; tensor unsqueeze_164_cast_fp16 = expand_dims(axes = unsqueeze_164_axes_0, x = cat_111_cast_fp16)[name = string("unsqueeze_164_cast_fp16")]; tensor expand_69_reps_0 = const()[name = string("expand_69_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_69_cast_fp16 = tile(reps = expand_69_reps_0, x = unsqueeze_164_cast_fp16)[name = string("expand_69_cast_fp16")]; tensor const_2986 = const()[name = string("const_2986"), val = tensor([1, 3, 512, 256])]; tensor view_130_cast_fp16 = reshape(shape = const_2986, x = expand_69_cast_fp16)[name = string("view_130_cast_fp16")]; bool matmul_66_transpose_x_1 = const()[name = string("matmul_66_transpose_x_1"), val = bool(false)]; bool matmul_66_transpose_y_1 = const()[name = string("matmul_66_transpose_y_1"), val = bool(true)]; tensor matmul_66_cast_fp16 = matmul(transpose_x = matmul_66_transpose_x_1, transpose_y = matmul_66_transpose_y_1, x = add_363_cast_fp16, y = view_129_cast_fp16)[name = string("matmul_66_cast_fp16")]; fp16 const_2989_to_fp16 = const()[name = string("const_2989_to_fp16"), val = fp16(0x1p-4)]; tensor mul_439_cast_fp16 = mul(x = matmul_66_cast_fp16, y = const_2989_to_fp16)[name = string("mul_439_cast_fp16")]; tensor add_365_cast_fp16 = add(x = mul_439_cast_fp16, y = expand_cast_fp16)[name = string("add_365_cast_fp16")]; int32 const_2999 = const()[name = string("const_2999"), val = int32(-1)]; tensor softmax_21_cast_fp16 = softmax(axis = const_2999, x = add_365_cast_fp16)[name = string("softmax_21_cast_fp16")]; bool matmul_67_transpose_x_0 = const()[name = string("matmul_67_transpose_x_0"), val = bool(false)]; bool matmul_67_transpose_y_0 = const()[name = string("matmul_67_transpose_y_0"), val = bool(false)]; tensor matmul_67_cast_fp16 = matmul(transpose_x = matmul_67_transpose_x_0, transpose_y = matmul_67_transpose_y_0, x = softmax_21_cast_fp16, y = view_130_cast_fp16)[name = string("matmul_67_cast_fp16")]; tensor transpose_133_perm_0 = const()[name = string("transpose_133_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_3004 = const()[name = string("const_3004"), val = tensor([1, 512, -1])]; tensor transpose_133_cast_fp16 = transpose(perm = transpose_133_perm_0, x = matmul_67_cast_fp16)[name = string("transpose_8")]; tensor view_131_cast_fp16 = reshape(shape = const_3004, x = transpose_133_cast_fp16)[name = string("view_131_cast_fp16")]; tensor p_st_0_model_layers_21_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292529856))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(293119744))))[name = string("p_st_0_model_layers_21_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_150_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_21_self_attn_o_proj_weight_to_fp16_quantized, x = view_131_cast_fp16)[name = string("linear_150_cast_fp16")]; fp16 const_3006_promoted_to_fp16 = const()[name = string("const_3006_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_130_cast_fp16 = pow(x = linear_150_cast_fp16, y = const_3006_promoted_to_fp16)[name = string("pow_130_cast_fp16")]; tensor mean_129_axes_0 = const()[name = string("mean_129_axes_0"), val = tensor([-1])]; bool mean_129_keep_dims_0 = const()[name = string("mean_129_keep_dims_0"), val = bool(true)]; tensor mean_129_cast_fp16 = reduce_mean(axes = mean_129_axes_0, keep_dims = mean_129_keep_dims_0, x = pow_130_cast_fp16)[name = string("mean_129_cast_fp16")]; fp16 const_3009_to_fp16 = const()[name = string("const_3009_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_366_cast_fp16 = add(x = mean_129_cast_fp16, y = const_3009_to_fp16)[name = string("add_366_cast_fp16")]; fp32 rsqrt_129_epsilon_0 = const()[name = string("rsqrt_129_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_129_cast_fp16 = rsqrt(epsilon = rsqrt_129_epsilon_0, x = add_366_cast_fp16)[name = string("rsqrt_129_cast_fp16")]; tensor mul_440_cast_fp16 = mul(x = linear_150_cast_fp16, y = rsqrt_129_cast_fp16)[name = string("mul_440_cast_fp16")]; tensor add_367_to_fp16 = const()[name = string("add_367_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(293121344)))]; tensor mul_441_cast_fp16 = mul(x = mul_440_cast_fp16, y = add_367_to_fp16)[name = string("mul_441_cast_fp16")]; tensor add_368_cast_fp16 = add(x = add_356_cast_fp16, y = mul_441_cast_fp16)[name = string("add_368_cast_fp16")]; fp16 const_3014_promoted_to_fp16 = const()[name = string("const_3014_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_131_cast_fp16 = pow(x = add_368_cast_fp16, y = const_3014_promoted_to_fp16)[name = string("pow_131_cast_fp16")]; tensor mean_130_axes_0 = const()[name = string("mean_130_axes_0"), val = tensor([-1])]; bool mean_130_keep_dims_0 = const()[name = string("mean_130_keep_dims_0"), val = bool(true)]; tensor mean_130_cast_fp16 = reduce_mean(axes = mean_130_axes_0, keep_dims = mean_130_keep_dims_0, x = pow_131_cast_fp16)[name = string("mean_130_cast_fp16")]; fp16 const_3017_to_fp16 = const()[name = string("const_3017_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_369_cast_fp16 = add(x = mean_130_cast_fp16, y = const_3017_to_fp16)[name = string("add_369_cast_fp16")]; fp32 rsqrt_130_epsilon_0 = const()[name = string("rsqrt_130_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_130_cast_fp16 = rsqrt(epsilon = rsqrt_130_epsilon_0, x = add_369_cast_fp16)[name = string("rsqrt_130_cast_fp16")]; tensor mul_442_cast_fp16 = mul(x = add_368_cast_fp16, y = rsqrt_130_cast_fp16)[name = string("mul_442_cast_fp16")]; tensor add_370_to_fp16 = const()[name = string("add_370_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(293122944)))]; tensor mul_443_cast_fp16 = mul(x = mul_442_cast_fp16, y = add_370_to_fp16)[name = string("mul_443_cast_fp16")]; tensor p_st_0_model_layers_21_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(293124544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294009344))))[name = string("p_st_0_model_layers_21_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_151_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_21_mlp_gate_proj_weight_to_fp16_quantized, x = mul_443_cast_fp16)[name = string("linear_151_cast_fp16")]; string gelu_21_mode_0 = const()[name = string("gelu_21_mode_0"), val = string("EXACT")]; tensor gelu_21_cast_fp16 = gelu(mode = gelu_21_mode_0, x = linear_151_cast_fp16)[name = string("gelu_21_cast_fp16")]; tensor p_st_0_model_layers_21_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294011712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294896512))))[name = string("p_st_0_model_layers_21_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_152_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_21_mlp_up_proj_weight_to_fp16_quantized, x = mul_443_cast_fp16)[name = string("linear_152_cast_fp16")]; tensor mul_444_cast_fp16 = mul(x = gelu_21_cast_fp16, y = linear_152_cast_fp16)[name = string("mul_444_cast_fp16")]; tensor p_st_0_model_layers_21_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294898880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295783680))))[name = string("p_st_0_model_layers_21_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_153_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_21_mlp_down_proj_weight_to_fp16_quantized, x = mul_444_cast_fp16)[name = string("linear_153_cast_fp16")]; fp16 const_3022_promoted_to_fp16 = const()[name = string("const_3022_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_132_cast_fp16 = pow(x = linear_153_cast_fp16, y = const_3022_promoted_to_fp16)[name = string("pow_132_cast_fp16")]; tensor mean_131_axes_0 = const()[name = string("mean_131_axes_0"), val = tensor([-1])]; bool mean_131_keep_dims_0 = const()[name = string("mean_131_keep_dims_0"), val = bool(true)]; tensor mean_131_cast_fp16 = reduce_mean(axes = mean_131_axes_0, keep_dims = mean_131_keep_dims_0, x = pow_132_cast_fp16)[name = string("mean_131_cast_fp16")]; fp16 const_3025_to_fp16 = const()[name = string("const_3025_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_371_cast_fp16 = add(x = mean_131_cast_fp16, y = const_3025_to_fp16)[name = string("add_371_cast_fp16")]; fp32 rsqrt_131_epsilon_0 = const()[name = string("rsqrt_131_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_131_cast_fp16 = rsqrt(epsilon = rsqrt_131_epsilon_0, x = add_371_cast_fp16)[name = string("rsqrt_131_cast_fp16")]; tensor mul_445_cast_fp16 = mul(x = linear_153_cast_fp16, y = rsqrt_131_cast_fp16)[name = string("mul_445_cast_fp16")]; tensor add_372_to_fp16 = const()[name = string("add_372_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295785280)))]; tensor mul_446_cast_fp16 = mul(x = mul_445_cast_fp16, y = add_372_to_fp16)[name = string("mul_446_cast_fp16")]; tensor add_373_cast_fp16 = add(x = add_368_cast_fp16, y = mul_446_cast_fp16)[name = string("add_373_cast_fp16")]; fp16 const_3030_promoted_to_fp16 = const()[name = string("const_3030_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_133_cast_fp16 = pow(x = add_373_cast_fp16, y = const_3030_promoted_to_fp16)[name = string("pow_133_cast_fp16")]; tensor mean_132_axes_0 = const()[name = string("mean_132_axes_0"), val = tensor([-1])]; bool mean_132_keep_dims_0 = const()[name = string("mean_132_keep_dims_0"), val = bool(true)]; tensor mean_132_cast_fp16 = reduce_mean(axes = mean_132_axes_0, keep_dims = mean_132_keep_dims_0, x = pow_133_cast_fp16)[name = string("mean_132_cast_fp16")]; fp16 const_3033_to_fp16 = const()[name = string("const_3033_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_374_cast_fp16 = add(x = mean_132_cast_fp16, y = const_3033_to_fp16)[name = string("add_374_cast_fp16")]; fp32 rsqrt_132_epsilon_0 = const()[name = string("rsqrt_132_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_132_cast_fp16 = rsqrt(epsilon = rsqrt_132_epsilon_0, x = add_374_cast_fp16)[name = string("rsqrt_132_cast_fp16")]; tensor mul_447_cast_fp16 = mul(x = add_373_cast_fp16, y = rsqrt_132_cast_fp16)[name = string("mul_447_cast_fp16")]; tensor add_375_to_fp16 = const()[name = string("add_375_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295786880)))]; tensor mul_448_cast_fp16 = mul(x = mul_447_cast_fp16, y = add_375_to_fp16)[name = string("mul_448_cast_fp16")]; tensor p_st_0_model_layers_22_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295788480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296378368))))[name = string("p_st_0_model_layers_22_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_154_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_22_self_attn_q_proj_weight_to_fp16_quantized, x = mul_448_cast_fp16)[name = string("linear_154_cast_fp16")]; tensor const_3037 = const()[name = string("const_3037"), val = tensor([1, 512, -1, 256])]; tensor view_132_cast_fp16 = reshape(shape = const_3037, x = linear_154_cast_fp16)[name = string("view_132_cast_fp16")]; tensor transpose_134_perm_0 = const()[name = string("transpose_134_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_22_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296379968))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296576640))))[name = string("p_st_0_model_layers_22_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_155_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_22_self_attn_k_proj_weight_to_fp16_quantized, x = mul_448_cast_fp16)[name = string("linear_155_cast_fp16")]; tensor const_3040 = const()[name = string("const_3040"), val = tensor([1, 512, -1, 256])]; tensor view_133_cast_fp16 = reshape(shape = const_3040, x = linear_155_cast_fp16)[name = string("view_133_cast_fp16")]; tensor transpose_135_perm_0 = const()[name = string("transpose_135_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_22_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296577216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296773888))))[name = string("p_st_0_model_layers_22_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_156_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_22_self_attn_v_proj_weight_to_fp16_quantized, x = mul_448_cast_fp16)[name = string("linear_156_cast_fp16")]; tensor const_3043 = const()[name = string("const_3043"), val = tensor([1, 512, -1, 256])]; tensor view_134_cast_fp16 = reshape(shape = const_3043, x = linear_156_cast_fp16)[name = string("view_134_cast_fp16")]; tensor transpose_136_perm_0 = const()[name = string("transpose_136_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_3047_promoted_to_fp16 = const()[name = string("const_3047_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_134_cast_fp16 = transpose(perm = transpose_134_perm_0, x = view_132_cast_fp16)[name = string("transpose_7")]; tensor pow_134_cast_fp16 = pow(x = transpose_134_cast_fp16, y = const_3047_promoted_to_fp16)[name = string("pow_134_cast_fp16")]; tensor mean_133_axes_0 = const()[name = string("mean_133_axes_0"), val = tensor([-1])]; bool mean_133_keep_dims_0 = const()[name = string("mean_133_keep_dims_0"), val = bool(true)]; tensor mean_133_cast_fp16 = reduce_mean(axes = mean_133_axes_0, keep_dims = mean_133_keep_dims_0, x = pow_134_cast_fp16)[name = string("mean_133_cast_fp16")]; fp16 const_3050_to_fp16 = const()[name = string("const_3050_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_376_cast_fp16 = add(x = mean_133_cast_fp16, y = const_3050_to_fp16)[name = string("add_376_cast_fp16")]; fp32 rsqrt_133_epsilon_0 = const()[name = string("rsqrt_133_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_133_cast_fp16 = rsqrt(epsilon = rsqrt_133_epsilon_0, x = add_376_cast_fp16)[name = string("rsqrt_133_cast_fp16")]; tensor mul_449_cast_fp16 = mul(x = transpose_134_cast_fp16, y = rsqrt_133_cast_fp16)[name = string("mul_449_cast_fp16")]; tensor add_377_to_fp16 = const()[name = string("add_377_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296774464)))]; tensor mul_450_cast_fp16 = mul(x = mul_449_cast_fp16, y = add_377_to_fp16)[name = string("mul_450_cast_fp16")]; fp16 const_3055_promoted_to_fp16 = const()[name = string("const_3055_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_135_cast_fp16 = transpose(perm = transpose_135_perm_0, x = view_133_cast_fp16)[name = string("transpose_6")]; tensor pow_135_cast_fp16 = pow(x = transpose_135_cast_fp16, y = const_3055_promoted_to_fp16)[name = string("pow_135_cast_fp16")]; tensor mean_134_axes_0 = const()[name = string("mean_134_axes_0"), val = tensor([-1])]; bool mean_134_keep_dims_0 = const()[name = string("mean_134_keep_dims_0"), val = bool(true)]; tensor mean_134_cast_fp16 = reduce_mean(axes = mean_134_axes_0, keep_dims = mean_134_keep_dims_0, x = pow_135_cast_fp16)[name = string("mean_134_cast_fp16")]; fp16 const_3058_to_fp16 = const()[name = string("const_3058_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_378_cast_fp16 = add(x = mean_134_cast_fp16, y = const_3058_to_fp16)[name = string("add_378_cast_fp16")]; fp32 rsqrt_134_epsilon_0 = const()[name = string("rsqrt_134_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_134_cast_fp16 = rsqrt(epsilon = rsqrt_134_epsilon_0, x = add_378_cast_fp16)[name = string("rsqrt_134_cast_fp16")]; tensor mul_451_cast_fp16 = mul(x = transpose_135_cast_fp16, y = rsqrt_134_cast_fp16)[name = string("mul_451_cast_fp16")]; tensor add_379_to_fp16 = const()[name = string("add_379_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296775040)))]; tensor mul_452_cast_fp16 = mul(x = mul_451_cast_fp16, y = add_379_to_fp16)[name = string("mul_452_cast_fp16")]; tensor mul_453_cast_fp16 = mul(x = mul_450_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_453_cast_fp16")]; tensor slice_559_begin_0 = const()[name = string("slice_559_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_559_end_0 = const()[name = string("slice_559_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_559_end_mask_0 = const()[name = string("slice_559_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_559_cast_fp16 = slice_by_index(begin = slice_559_begin_0, end = slice_559_end_0, end_mask = slice_559_end_mask_0, x = mul_450_cast_fp16)[name = string("slice_559_cast_fp16")]; tensor slice_560_begin_0 = const()[name = string("slice_560_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_560_end_0 = const()[name = string("slice_560_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_560_end_mask_0 = const()[name = string("slice_560_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_560_cast_fp16 = slice_by_index(begin = slice_560_begin_0, end = slice_560_end_0, end_mask = slice_560_end_mask_0, x = mul_450_cast_fp16)[name = string("slice_560_cast_fp16")]; fp16 const_3070_promoted_to_fp16 = const()[name = string("const_3070_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_44_cast_fp16 = mul(x = slice_560_cast_fp16, y = const_3070_promoted_to_fp16)[name = string("neg_44_cast_fp16")]; int32 const_3071 = const()[name = string("const_3071"), val = int32(-1)]; bool cat_112_interleave_0 = const()[name = string("cat_112_interleave_0"), val = bool(false)]; tensor cat_112_cast_fp16 = concat(axis = const_3071, interleave = cat_112_interleave_0, values = (neg_44_cast_fp16, slice_559_cast_fp16))[name = string("cat_112_cast_fp16")]; tensor mul_454_cast_fp16 = mul(x = cat_112_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_454_cast_fp16")]; tensor add_380_cast_fp16 = add(x = mul_453_cast_fp16, y = mul_454_cast_fp16)[name = string("add_380_cast_fp16")]; tensor mul_455_cast_fp16 = mul(x = mul_452_cast_fp16, y = unsqueeze_77_to_fp16_quantized)[name = string("mul_455_cast_fp16")]; tensor slice_561_begin_0 = const()[name = string("slice_561_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_561_end_0 = const()[name = string("slice_561_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_561_end_mask_0 = const()[name = string("slice_561_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_561_cast_fp16 = slice_by_index(begin = slice_561_begin_0, end = slice_561_end_0, end_mask = slice_561_end_mask_0, x = mul_452_cast_fp16)[name = string("slice_561_cast_fp16")]; tensor slice_562_begin_0 = const()[name = string("slice_562_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_562_end_0 = const()[name = string("slice_562_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_562_end_mask_0 = const()[name = string("slice_562_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_562_cast_fp16 = slice_by_index(begin = slice_562_begin_0, end = slice_562_end_0, end_mask = slice_562_end_mask_0, x = mul_452_cast_fp16)[name = string("slice_562_cast_fp16")]; fp16 const_3078_promoted_to_fp16 = const()[name = string("const_3078_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_45_cast_fp16 = mul(x = slice_562_cast_fp16, y = const_3078_promoted_to_fp16)[name = string("neg_45_cast_fp16")]; int32 const_3079 = const()[name = string("const_3079"), val = int32(-1)]; bool cat_113_interleave_0 = const()[name = string("cat_113_interleave_0"), val = bool(false)]; tensor cat_113_cast_fp16 = concat(axis = const_3079, interleave = cat_113_interleave_0, values = (neg_45_cast_fp16, slice_561_cast_fp16))[name = string("cat_113_cast_fp16")]; tensor mul_456_cast_fp16 = mul(x = cat_113_cast_fp16, y = unsqueeze_78_to_fp16_quantized)[name = string("mul_456_cast_fp16")]; tensor add_381_cast_fp16 = add(x = mul_455_cast_fp16, y = mul_456_cast_fp16)[name = string("add_381_cast_fp16")]; int32 const_3080 = const()[name = string("const_3080"), val = int32(-2)]; bool cat_114_interleave_0 = const()[name = string("cat_114_interleave_0"), val = bool(false)]; tensor cat_114_cast_fp16 = concat(axis = const_3080, interleave = cat_114_interleave_0, values = add_381_cast_fp16)[name = string("cat_114_cast_fp16")]; int32 const_3081 = const()[name = string("const_3081"), val = int32(-2)]; bool cat_115_interleave_0 = const()[name = string("cat_115_interleave_0"), val = bool(false)]; tensor transpose_136_cast_fp16 = transpose(perm = transpose_136_perm_0, x = view_134_cast_fp16)[name = string("transpose_5")]; tensor cat_115_cast_fp16 = concat(axis = const_3081, interleave = cat_115_interleave_0, values = transpose_136_cast_fp16)[name = string("cat_115_cast_fp16")]; tensor unsqueeze_167_axes_0 = const()[name = string("unsqueeze_167_axes_0"), val = tensor([2])]; tensor unsqueeze_167_cast_fp16 = expand_dims(axes = unsqueeze_167_axes_0, x = cat_114_cast_fp16)[name = string("unsqueeze_167_cast_fp16")]; tensor expand_70_reps_0 = const()[name = string("expand_70_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_70_cast_fp16 = tile(reps = expand_70_reps_0, x = unsqueeze_167_cast_fp16)[name = string("expand_70_cast_fp16")]; tensor const_3096 = const()[name = string("const_3096"), val = tensor([1, 3, 512, 256])]; tensor view_135_cast_fp16 = reshape(shape = const_3096, x = expand_70_cast_fp16)[name = string("view_135_cast_fp16")]; tensor unsqueeze_168_axes_0 = const()[name = string("unsqueeze_168_axes_0"), val = tensor([2])]; tensor unsqueeze_168_cast_fp16 = expand_dims(axes = unsqueeze_168_axes_0, x = cat_115_cast_fp16)[name = string("unsqueeze_168_cast_fp16")]; tensor expand_71_reps_0 = const()[name = string("expand_71_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_71_cast_fp16 = tile(reps = expand_71_reps_0, x = unsqueeze_168_cast_fp16)[name = string("expand_71_cast_fp16")]; tensor const_3111 = const()[name = string("const_3111"), val = tensor([1, 3, 512, 256])]; tensor view_136_cast_fp16 = reshape(shape = const_3111, x = expand_71_cast_fp16)[name = string("view_136_cast_fp16")]; bool matmul_68_transpose_x_1 = const()[name = string("matmul_68_transpose_x_1"), val = bool(false)]; bool matmul_68_transpose_y_1 = const()[name = string("matmul_68_transpose_y_1"), val = bool(true)]; tensor matmul_68_cast_fp16 = matmul(transpose_x = matmul_68_transpose_x_1, transpose_y = matmul_68_transpose_y_1, x = add_380_cast_fp16, y = view_135_cast_fp16)[name = string("matmul_68_cast_fp16")]; fp16 const_3114_to_fp16 = const()[name = string("const_3114_to_fp16"), val = fp16(0x1p-4)]; tensor mul_457_cast_fp16 = mul(x = matmul_68_cast_fp16, y = const_3114_to_fp16)[name = string("mul_457_cast_fp16")]; tensor add_382_cast_fp16 = add(x = mul_457_cast_fp16, y = expand_cast_fp16)[name = string("add_382_cast_fp16")]; int32 const_3124 = const()[name = string("const_3124"), val = int32(-1)]; tensor softmax_22_cast_fp16 = softmax(axis = const_3124, x = add_382_cast_fp16)[name = string("softmax_22_cast_fp16")]; bool matmul_69_transpose_x_0 = const()[name = string("matmul_69_transpose_x_0"), val = bool(false)]; bool matmul_69_transpose_y_0 = const()[name = string("matmul_69_transpose_y_0"), val = bool(false)]; tensor matmul_69_cast_fp16 = matmul(transpose_x = matmul_69_transpose_x_0, transpose_y = matmul_69_transpose_y_0, x = softmax_22_cast_fp16, y = view_136_cast_fp16)[name = string("matmul_69_cast_fp16")]; tensor transpose_138_perm_0 = const()[name = string("transpose_138_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_3129 = const()[name = string("const_3129"), val = tensor([1, 512, -1])]; tensor transpose_138_cast_fp16 = transpose(perm = transpose_138_perm_0, x = matmul_69_cast_fp16)[name = string("transpose_4")]; tensor view_137_cast_fp16 = reshape(shape = const_3129, x = transpose_138_cast_fp16)[name = string("view_137_cast_fp16")]; tensor p_st_0_model_layers_22_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296775616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297365504))))[name = string("p_st_0_model_layers_22_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_157_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_22_self_attn_o_proj_weight_to_fp16_quantized, x = view_137_cast_fp16)[name = string("linear_157_cast_fp16")]; fp16 const_3131_promoted_to_fp16 = const()[name = string("const_3131_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_136_cast_fp16 = pow(x = linear_157_cast_fp16, y = const_3131_promoted_to_fp16)[name = string("pow_136_cast_fp16")]; tensor mean_135_axes_0 = const()[name = string("mean_135_axes_0"), val = tensor([-1])]; bool mean_135_keep_dims_0 = const()[name = string("mean_135_keep_dims_0"), val = bool(true)]; tensor mean_135_cast_fp16 = reduce_mean(axes = mean_135_axes_0, keep_dims = mean_135_keep_dims_0, x = pow_136_cast_fp16)[name = string("mean_135_cast_fp16")]; fp16 const_3134_to_fp16 = const()[name = string("const_3134_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_383_cast_fp16 = add(x = mean_135_cast_fp16, y = const_3134_to_fp16)[name = string("add_383_cast_fp16")]; fp32 rsqrt_135_epsilon_0 = const()[name = string("rsqrt_135_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_135_cast_fp16 = rsqrt(epsilon = rsqrt_135_epsilon_0, x = add_383_cast_fp16)[name = string("rsqrt_135_cast_fp16")]; tensor mul_458_cast_fp16 = mul(x = linear_157_cast_fp16, y = rsqrt_135_cast_fp16)[name = string("mul_458_cast_fp16")]; tensor add_384_to_fp16 = const()[name = string("add_384_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297367104)))]; tensor mul_459_cast_fp16 = mul(x = mul_458_cast_fp16, y = add_384_to_fp16)[name = string("mul_459_cast_fp16")]; tensor add_385_cast_fp16 = add(x = add_373_cast_fp16, y = mul_459_cast_fp16)[name = string("add_385_cast_fp16")]; fp16 const_3139_promoted_to_fp16 = const()[name = string("const_3139_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_137_cast_fp16 = pow(x = add_385_cast_fp16, y = const_3139_promoted_to_fp16)[name = string("pow_137_cast_fp16")]; tensor mean_136_axes_0 = const()[name = string("mean_136_axes_0"), val = tensor([-1])]; bool mean_136_keep_dims_0 = const()[name = string("mean_136_keep_dims_0"), val = bool(true)]; tensor mean_136_cast_fp16 = reduce_mean(axes = mean_136_axes_0, keep_dims = mean_136_keep_dims_0, x = pow_137_cast_fp16)[name = string("mean_136_cast_fp16")]; fp16 const_3142_to_fp16 = const()[name = string("const_3142_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_386_cast_fp16 = add(x = mean_136_cast_fp16, y = const_3142_to_fp16)[name = string("add_386_cast_fp16")]; fp32 rsqrt_136_epsilon_0 = const()[name = string("rsqrt_136_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_136_cast_fp16 = rsqrt(epsilon = rsqrt_136_epsilon_0, x = add_386_cast_fp16)[name = string("rsqrt_136_cast_fp16")]; tensor mul_460_cast_fp16 = mul(x = add_385_cast_fp16, y = rsqrt_136_cast_fp16)[name = string("mul_460_cast_fp16")]; tensor add_387_to_fp16 = const()[name = string("add_387_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297368704)))]; tensor mul_461_cast_fp16 = mul(x = mul_460_cast_fp16, y = add_387_to_fp16)[name = string("mul_461_cast_fp16")]; tensor p_st_0_model_layers_22_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297370304))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298255104))))[name = string("p_st_0_model_layers_22_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_158_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_22_mlp_gate_proj_weight_to_fp16_quantized, x = mul_461_cast_fp16)[name = string("linear_158_cast_fp16")]; string gelu_22_mode_0 = const()[name = string("gelu_22_mode_0"), val = string("EXACT")]; tensor gelu_22_cast_fp16 = gelu(mode = gelu_22_mode_0, x = linear_158_cast_fp16)[name = string("gelu_22_cast_fp16")]; tensor p_st_0_model_layers_22_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298257472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299142272))))[name = string("p_st_0_model_layers_22_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_159_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_22_mlp_up_proj_weight_to_fp16_quantized, x = mul_461_cast_fp16)[name = string("linear_159_cast_fp16")]; tensor mul_462_cast_fp16 = mul(x = gelu_22_cast_fp16, y = linear_159_cast_fp16)[name = string("mul_462_cast_fp16")]; tensor p_st_0_model_layers_22_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299144640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300029440))))[name = string("p_st_0_model_layers_22_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_160_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_22_mlp_down_proj_weight_to_fp16_quantized, x = mul_462_cast_fp16)[name = string("linear_160_cast_fp16")]; fp16 const_3147_promoted_to_fp16 = const()[name = string("const_3147_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_138_cast_fp16 = pow(x = linear_160_cast_fp16, y = const_3147_promoted_to_fp16)[name = string("pow_138_cast_fp16")]; tensor mean_137_axes_0 = const()[name = string("mean_137_axes_0"), val = tensor([-1])]; bool mean_137_keep_dims_0 = const()[name = string("mean_137_keep_dims_0"), val = bool(true)]; tensor mean_137_cast_fp16 = reduce_mean(axes = mean_137_axes_0, keep_dims = mean_137_keep_dims_0, x = pow_138_cast_fp16)[name = string("mean_137_cast_fp16")]; fp16 const_3150_to_fp16 = const()[name = string("const_3150_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_388_cast_fp16 = add(x = mean_137_cast_fp16, y = const_3150_to_fp16)[name = string("add_388_cast_fp16")]; fp32 rsqrt_137_epsilon_0 = const()[name = string("rsqrt_137_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_137_cast_fp16 = rsqrt(epsilon = rsqrt_137_epsilon_0, x = add_388_cast_fp16)[name = string("rsqrt_137_cast_fp16")]; tensor mul_463_cast_fp16 = mul(x = linear_160_cast_fp16, y = rsqrt_137_cast_fp16)[name = string("mul_463_cast_fp16")]; tensor add_389_to_fp16 = const()[name = string("add_389_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300031040)))]; tensor mul_464_cast_fp16 = mul(x = mul_463_cast_fp16, y = add_389_to_fp16)[name = string("mul_464_cast_fp16")]; tensor add_390_cast_fp16 = add(x = add_385_cast_fp16, y = mul_464_cast_fp16)[name = string("add_390_cast_fp16")]; fp16 const_3155_promoted_to_fp16 = const()[name = string("const_3155_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_139_cast_fp16 = pow(x = add_390_cast_fp16, y = const_3155_promoted_to_fp16)[name = string("pow_139_cast_fp16")]; tensor mean_138_axes_0 = const()[name = string("mean_138_axes_0"), val = tensor([-1])]; bool mean_138_keep_dims_0 = const()[name = string("mean_138_keep_dims_0"), val = bool(true)]; tensor mean_138_cast_fp16 = reduce_mean(axes = mean_138_axes_0, keep_dims = mean_138_keep_dims_0, x = pow_139_cast_fp16)[name = string("mean_138_cast_fp16")]; fp16 const_3158_to_fp16 = const()[name = string("const_3158_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_391_cast_fp16 = add(x = mean_138_cast_fp16, y = const_3158_to_fp16)[name = string("add_391_cast_fp16")]; fp32 rsqrt_138_epsilon_0 = const()[name = string("rsqrt_138_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_138_cast_fp16 = rsqrt(epsilon = rsqrt_138_epsilon_0, x = add_391_cast_fp16)[name = string("rsqrt_138_cast_fp16")]; tensor mul_465_cast_fp16 = mul(x = add_390_cast_fp16, y = rsqrt_138_cast_fp16)[name = string("mul_465_cast_fp16")]; tensor add_392_to_fp16 = const()[name = string("add_392_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300032640)))]; tensor mul_466_cast_fp16 = mul(x = mul_465_cast_fp16, y = add_392_to_fp16)[name = string("mul_466_cast_fp16")]; tensor p_st_0_model_layers_23_self_attn_q_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300034240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300624128))))[name = string("p_st_0_model_layers_23_self_attn_q_proj_weight_to_fp16_quantized")]; tensor linear_161_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_23_self_attn_q_proj_weight_to_fp16_quantized, x = mul_466_cast_fp16)[name = string("linear_161_cast_fp16")]; tensor const_3162 = const()[name = string("const_3162"), val = tensor([1, 512, -1, 256])]; tensor view_138_cast_fp16 = reshape(shape = const_3162, x = linear_161_cast_fp16)[name = string("view_138_cast_fp16")]; tensor transpose_139_perm_0 = const()[name = string("transpose_139_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_23_self_attn_k_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300625728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300822400))))[name = string("p_st_0_model_layers_23_self_attn_k_proj_weight_to_fp16_quantized")]; tensor linear_162_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_23_self_attn_k_proj_weight_to_fp16_quantized, x = mul_466_cast_fp16)[name = string("linear_162_cast_fp16")]; tensor const_3165 = const()[name = string("const_3165"), val = tensor([1, 512, -1, 256])]; tensor view_139_cast_fp16 = reshape(shape = const_3165, x = linear_162_cast_fp16)[name = string("view_139_cast_fp16")]; tensor transpose_140_perm_0 = const()[name = string("transpose_140_perm_0"), val = tensor([0, 2, 1, 3])]; tensor p_st_0_model_layers_23_self_attn_v_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300822976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301019648))))[name = string("p_st_0_model_layers_23_self_attn_v_proj_weight_to_fp16_quantized")]; tensor linear_163_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = p_st_0_model_layers_23_self_attn_v_proj_weight_to_fp16_quantized, x = mul_466_cast_fp16)[name = string("linear_163_cast_fp16")]; tensor const_3168 = const()[name = string("const_3168"), val = tensor([1, 512, -1, 256])]; tensor view_140_cast_fp16 = reshape(shape = const_3168, x = linear_163_cast_fp16)[name = string("view_140_cast_fp16")]; tensor transpose_141_perm_0 = const()[name = string("transpose_141_perm_0"), val = tensor([0, 2, 1, 3])]; fp16 const_3172_promoted_to_fp16 = const()[name = string("const_3172_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_139_cast_fp16 = transpose(perm = transpose_139_perm_0, x = view_138_cast_fp16)[name = string("transpose_3")]; tensor pow_140_cast_fp16 = pow(x = transpose_139_cast_fp16, y = const_3172_promoted_to_fp16)[name = string("pow_140_cast_fp16")]; tensor mean_139_axes_0 = const()[name = string("mean_139_axes_0"), val = tensor([-1])]; bool mean_139_keep_dims_0 = const()[name = string("mean_139_keep_dims_0"), val = bool(true)]; tensor mean_139_cast_fp16 = reduce_mean(axes = mean_139_axes_0, keep_dims = mean_139_keep_dims_0, x = pow_140_cast_fp16)[name = string("mean_139_cast_fp16")]; fp16 const_3175_to_fp16 = const()[name = string("const_3175_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_393_cast_fp16 = add(x = mean_139_cast_fp16, y = const_3175_to_fp16)[name = string("add_393_cast_fp16")]; fp32 rsqrt_139_epsilon_0 = const()[name = string("rsqrt_139_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_139_cast_fp16 = rsqrt(epsilon = rsqrt_139_epsilon_0, x = add_393_cast_fp16)[name = string("rsqrt_139_cast_fp16")]; tensor mul_467_cast_fp16 = mul(x = transpose_139_cast_fp16, y = rsqrt_139_cast_fp16)[name = string("mul_467_cast_fp16")]; tensor add_394_to_fp16 = const()[name = string("add_394_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301020224)))]; tensor mul_468_cast_fp16 = mul(x = mul_467_cast_fp16, y = add_394_to_fp16)[name = string("mul_468_cast_fp16")]; fp16 const_3180_promoted_to_fp16 = const()[name = string("const_3180_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor transpose_140_cast_fp16 = transpose(perm = transpose_140_perm_0, x = view_139_cast_fp16)[name = string("transpose_2")]; tensor pow_141_cast_fp16 = pow(x = transpose_140_cast_fp16, y = const_3180_promoted_to_fp16)[name = string("pow_141_cast_fp16")]; tensor mean_140_axes_0 = const()[name = string("mean_140_axes_0"), val = tensor([-1])]; bool mean_140_keep_dims_0 = const()[name = string("mean_140_keep_dims_0"), val = bool(true)]; tensor mean_140_cast_fp16 = reduce_mean(axes = mean_140_axes_0, keep_dims = mean_140_keep_dims_0, x = pow_141_cast_fp16)[name = string("mean_140_cast_fp16")]; fp16 const_3183_to_fp16 = const()[name = string("const_3183_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_395_cast_fp16 = add(x = mean_140_cast_fp16, y = const_3183_to_fp16)[name = string("add_395_cast_fp16")]; fp32 rsqrt_140_epsilon_0 = const()[name = string("rsqrt_140_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_140_cast_fp16 = rsqrt(epsilon = rsqrt_140_epsilon_0, x = add_395_cast_fp16)[name = string("rsqrt_140_cast_fp16")]; tensor mul_469_cast_fp16 = mul(x = transpose_140_cast_fp16, y = rsqrt_140_cast_fp16)[name = string("mul_469_cast_fp16")]; tensor add_396_to_fp16 = const()[name = string("add_396_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301020800)))]; tensor mul_470_cast_fp16 = mul(x = mul_469_cast_fp16, y = add_396_to_fp16)[name = string("mul_470_cast_fp16")]; tensor mul_471_cast_fp16 = mul(x = mul_468_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_471_cast_fp16")]; tensor slice_582_begin_0 = const()[name = string("slice_582_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_582_end_0 = const()[name = string("slice_582_end_0"), val = tensor([1, 3, 512, 128])]; tensor slice_582_end_mask_0 = const()[name = string("slice_582_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_582_cast_fp16 = slice_by_index(begin = slice_582_begin_0, end = slice_582_end_0, end_mask = slice_582_end_mask_0, x = mul_468_cast_fp16)[name = string("slice_582_cast_fp16")]; tensor slice_583_begin_0 = const()[name = string("slice_583_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_583_end_0 = const()[name = string("slice_583_end_0"), val = tensor([1, 3, 512, 1])]; tensor slice_583_end_mask_0 = const()[name = string("slice_583_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_583_cast_fp16 = slice_by_index(begin = slice_583_begin_0, end = slice_583_end_0, end_mask = slice_583_end_mask_0, x = mul_468_cast_fp16)[name = string("slice_583_cast_fp16")]; fp16 const_3195_promoted_to_fp16 = const()[name = string("const_3195_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_46_cast_fp16 = mul(x = slice_583_cast_fp16, y = const_3195_promoted_to_fp16)[name = string("neg_46_cast_fp16")]; int32 const_3196 = const()[name = string("const_3196"), val = int32(-1)]; bool cat_116_interleave_0 = const()[name = string("cat_116_interleave_0"), val = bool(false)]; tensor cat_116_cast_fp16 = concat(axis = const_3196, interleave = cat_116_interleave_0, values = (neg_46_cast_fp16, slice_582_cast_fp16))[name = string("cat_116_cast_fp16")]; tensor mul_472_cast_fp16 = mul(x = cat_116_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_472_cast_fp16")]; tensor add_397_cast_fp16 = add(x = mul_471_cast_fp16, y = mul_472_cast_fp16)[name = string("add_397_cast_fp16")]; tensor mul_473_cast_fp16 = mul(x = mul_470_cast_fp16, y = unsqueeze_97_to_fp16_quantized)[name = string("mul_473_cast_fp16")]; tensor slice_584_begin_0 = const()[name = string("slice_584_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slice_584_end_0 = const()[name = string("slice_584_end_0"), val = tensor([1, 1, 512, 128])]; tensor slice_584_end_mask_0 = const()[name = string("slice_584_end_mask_0"), val = tensor([true, true, true, false])]; tensor slice_584_cast_fp16 = slice_by_index(begin = slice_584_begin_0, end = slice_584_end_0, end_mask = slice_584_end_mask_0, x = mul_470_cast_fp16)[name = string("slice_584_cast_fp16")]; tensor slice_585_begin_0 = const()[name = string("slice_585_begin_0"), val = tensor([0, 0, 0, 128])]; tensor slice_585_end_0 = const()[name = string("slice_585_end_0"), val = tensor([1, 1, 512, 1])]; tensor slice_585_end_mask_0 = const()[name = string("slice_585_end_mask_0"), val = tensor([true, true, true, true])]; tensor slice_585_cast_fp16 = slice_by_index(begin = slice_585_begin_0, end = slice_585_end_0, end_mask = slice_585_end_mask_0, x = mul_470_cast_fp16)[name = string("slice_585_cast_fp16")]; fp16 const_3203_promoted_to_fp16 = const()[name = string("const_3203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor neg_47_cast_fp16 = mul(x = slice_585_cast_fp16, y = const_3203_promoted_to_fp16)[name = string("neg_47_cast_fp16")]; int32 const_3204 = const()[name = string("const_3204"), val = int32(-1)]; bool cat_117_interleave_0 = const()[name = string("cat_117_interleave_0"), val = bool(false)]; tensor cat_117_cast_fp16 = concat(axis = const_3204, interleave = cat_117_interleave_0, values = (neg_47_cast_fp16, slice_584_cast_fp16))[name = string("cat_117_cast_fp16")]; tensor mul_474_cast_fp16 = mul(x = cat_117_cast_fp16, y = unsqueeze_98_to_fp16_quantized)[name = string("mul_474_cast_fp16")]; tensor add_398_cast_fp16 = add(x = mul_473_cast_fp16, y = mul_474_cast_fp16)[name = string("add_398_cast_fp16")]; int32 const_3205 = const()[name = string("const_3205"), val = int32(-2)]; bool cat_118_interleave_0 = const()[name = string("cat_118_interleave_0"), val = bool(false)]; tensor cat_118_cast_fp16 = concat(axis = const_3205, interleave = cat_118_interleave_0, values = add_398_cast_fp16)[name = string("cat_118_cast_fp16")]; int32 const_3206 = const()[name = string("const_3206"), val = int32(-2)]; bool cat_119_interleave_0 = const()[name = string("cat_119_interleave_0"), val = bool(false)]; tensor transpose_141_cast_fp16 = transpose(perm = transpose_141_perm_0, x = view_140_cast_fp16)[name = string("transpose_1")]; tensor cat_119_cast_fp16 = concat(axis = const_3206, interleave = cat_119_interleave_0, values = transpose_141_cast_fp16)[name = string("cat_119_cast_fp16")]; tensor unsqueeze_171_axes_0 = const()[name = string("unsqueeze_171_axes_0"), val = tensor([2])]; tensor unsqueeze_171_cast_fp16 = expand_dims(axes = unsqueeze_171_axes_0, x = cat_118_cast_fp16)[name = string("unsqueeze_171_cast_fp16")]; tensor expand_72_reps_0 = const()[name = string("expand_72_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_72_cast_fp16 = tile(reps = expand_72_reps_0, x = unsqueeze_171_cast_fp16)[name = string("expand_72_cast_fp16")]; tensor const_3221 = const()[name = string("const_3221"), val = tensor([1, 3, 512, 256])]; tensor view_141_cast_fp16 = reshape(shape = const_3221, x = expand_72_cast_fp16)[name = string("view_141_cast_fp16")]; tensor unsqueeze_172_axes_0 = const()[name = string("unsqueeze_172_axes_0"), val = tensor([2])]; tensor unsqueeze_172_cast_fp16 = expand_dims(axes = unsqueeze_172_axes_0, x = cat_119_cast_fp16)[name = string("unsqueeze_172_cast_fp16")]; tensor expand_73_reps_0 = const()[name = string("expand_73_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor expand_73_cast_fp16 = tile(reps = expand_73_reps_0, x = unsqueeze_172_cast_fp16)[name = string("expand_73_cast_fp16")]; tensor const_3236 = const()[name = string("const_3236"), val = tensor([1, 3, 512, 256])]; tensor view_142_cast_fp16 = reshape(shape = const_3236, x = expand_73_cast_fp16)[name = string("view_142_cast_fp16")]; bool matmul_70_transpose_x_1 = const()[name = string("matmul_70_transpose_x_1"), val = bool(false)]; bool matmul_70_transpose_y_1 = const()[name = string("matmul_70_transpose_y_1"), val = bool(true)]; tensor matmul_70_cast_fp16 = matmul(transpose_x = matmul_70_transpose_x_1, transpose_y = matmul_70_transpose_y_1, x = add_397_cast_fp16, y = view_141_cast_fp16)[name = string("matmul_70_cast_fp16")]; fp16 const_3239_to_fp16 = const()[name = string("const_3239_to_fp16"), val = fp16(0x1p-4)]; tensor mul_475_cast_fp16 = mul(x = matmul_70_cast_fp16, y = const_3239_to_fp16)[name = string("mul_475_cast_fp16")]; tensor add_399_cast_fp16 = add(x = mul_475_cast_fp16, y = expand_cast_fp16)[name = string("add_399_cast_fp16")]; int32 const_3249 = const()[name = string("const_3249"), val = int32(-1)]; tensor softmax_23_cast_fp16 = softmax(axis = const_3249, x = add_399_cast_fp16)[name = string("softmax_23_cast_fp16")]; bool matmul_71_transpose_x_0 = const()[name = string("matmul_71_transpose_x_0"), val = bool(false)]; bool matmul_71_transpose_y_0 = const()[name = string("matmul_71_transpose_y_0"), val = bool(false)]; tensor matmul_71_cast_fp16 = matmul(transpose_x = matmul_71_transpose_x_0, transpose_y = matmul_71_transpose_y_0, x = softmax_23_cast_fp16, y = view_142_cast_fp16)[name = string("matmul_71_cast_fp16")]; tensor transpose_143_perm_0 = const()[name = string("transpose_143_perm_0"), val = tensor([0, 2, 1, 3])]; tensor const_3254 = const()[name = string("const_3254"), val = tensor([1, 512, -1])]; tensor transpose_143_cast_fp16 = transpose(perm = transpose_143_perm_0, x = matmul_71_cast_fp16)[name = string("transpose_0")]; tensor view_143_cast_fp16 = reshape(shape = const_3254, x = transpose_143_cast_fp16)[name = string("view_143_cast_fp16")]; tensor p_st_0_model_layers_23_self_attn_o_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301021376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301611264))))[name = string("p_st_0_model_layers_23_self_attn_o_proj_weight_to_fp16_quantized")]; tensor linear_164_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_23_self_attn_o_proj_weight_to_fp16_quantized, x = view_143_cast_fp16)[name = string("linear_164_cast_fp16")]; fp16 const_3256_promoted_to_fp16 = const()[name = string("const_3256_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_142_cast_fp16 = pow(x = linear_164_cast_fp16, y = const_3256_promoted_to_fp16)[name = string("pow_142_cast_fp16")]; tensor mean_141_axes_0 = const()[name = string("mean_141_axes_0"), val = tensor([-1])]; bool mean_141_keep_dims_0 = const()[name = string("mean_141_keep_dims_0"), val = bool(true)]; tensor mean_141_cast_fp16 = reduce_mean(axes = mean_141_axes_0, keep_dims = mean_141_keep_dims_0, x = pow_142_cast_fp16)[name = string("mean_141_cast_fp16")]; fp16 const_3259_to_fp16 = const()[name = string("const_3259_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_400_cast_fp16 = add(x = mean_141_cast_fp16, y = const_3259_to_fp16)[name = string("add_400_cast_fp16")]; fp32 rsqrt_141_epsilon_0 = const()[name = string("rsqrt_141_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_141_cast_fp16 = rsqrt(epsilon = rsqrt_141_epsilon_0, x = add_400_cast_fp16)[name = string("rsqrt_141_cast_fp16")]; tensor mul_476_cast_fp16 = mul(x = linear_164_cast_fp16, y = rsqrt_141_cast_fp16)[name = string("mul_476_cast_fp16")]; tensor add_401_to_fp16 = const()[name = string("add_401_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301612864)))]; tensor mul_477_cast_fp16 = mul(x = mul_476_cast_fp16, y = add_401_to_fp16)[name = string("mul_477_cast_fp16")]; tensor add_402_cast_fp16 = add(x = add_390_cast_fp16, y = mul_477_cast_fp16)[name = string("add_402_cast_fp16")]; fp16 const_3264_promoted_to_fp16 = const()[name = string("const_3264_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_143_cast_fp16 = pow(x = add_402_cast_fp16, y = const_3264_promoted_to_fp16)[name = string("pow_143_cast_fp16")]; tensor mean_142_axes_0 = const()[name = string("mean_142_axes_0"), val = tensor([-1])]; bool mean_142_keep_dims_0 = const()[name = string("mean_142_keep_dims_0"), val = bool(true)]; tensor mean_142_cast_fp16 = reduce_mean(axes = mean_142_axes_0, keep_dims = mean_142_keep_dims_0, x = pow_143_cast_fp16)[name = string("mean_142_cast_fp16")]; fp16 const_3267_to_fp16 = const()[name = string("const_3267_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_403_cast_fp16 = add(x = mean_142_cast_fp16, y = const_3267_to_fp16)[name = string("add_403_cast_fp16")]; fp32 rsqrt_142_epsilon_0 = const()[name = string("rsqrt_142_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_142_cast_fp16 = rsqrt(epsilon = rsqrt_142_epsilon_0, x = add_403_cast_fp16)[name = string("rsqrt_142_cast_fp16")]; tensor mul_478_cast_fp16 = mul(x = add_402_cast_fp16, y = rsqrt_142_cast_fp16)[name = string("mul_478_cast_fp16")]; tensor add_404_to_fp16 = const()[name = string("add_404_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301614464)))]; tensor mul_479_cast_fp16 = mul(x = mul_478_cast_fp16, y = add_404_to_fp16)[name = string("mul_479_cast_fp16")]; tensor p_st_0_model_layers_23_mlp_gate_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301616064))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302500864))))[name = string("p_st_0_model_layers_23_mlp_gate_proj_weight_to_fp16_quantized")]; tensor linear_165_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_23_mlp_gate_proj_weight_to_fp16_quantized, x = mul_479_cast_fp16)[name = string("linear_165_cast_fp16")]; string gelu_23_mode_0 = const()[name = string("gelu_23_mode_0"), val = string("EXACT")]; tensor gelu_23_cast_fp16 = gelu(mode = gelu_23_mode_0, x = linear_165_cast_fp16)[name = string("gelu_23_cast_fp16")]; tensor p_st_0_model_layers_23_mlp_up_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302503232))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303388032))))[name = string("p_st_0_model_layers_23_mlp_up_proj_weight_to_fp16_quantized")]; tensor linear_166_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = p_st_0_model_layers_23_mlp_up_proj_weight_to_fp16_quantized, x = mul_479_cast_fp16)[name = string("linear_166_cast_fp16")]; tensor mul_480_cast_fp16 = mul(x = gelu_23_cast_fp16, y = linear_166_cast_fp16)[name = string("mul_480_cast_fp16")]; tensor p_st_0_model_layers_23_mlp_down_proj_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303390400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304275200))))[name = string("p_st_0_model_layers_23_mlp_down_proj_weight_to_fp16_quantized")]; tensor linear_167_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_0_model_layers_23_mlp_down_proj_weight_to_fp16_quantized, x = mul_480_cast_fp16)[name = string("linear_167_cast_fp16")]; fp16 const_3272_promoted_to_fp16 = const()[name = string("const_3272_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_144_cast_fp16 = pow(x = linear_167_cast_fp16, y = const_3272_promoted_to_fp16)[name = string("pow_144_cast_fp16")]; tensor mean_143_axes_0 = const()[name = string("mean_143_axes_0"), val = tensor([-1])]; bool mean_143_keep_dims_0 = const()[name = string("mean_143_keep_dims_0"), val = bool(true)]; tensor mean_143_cast_fp16 = reduce_mean(axes = mean_143_axes_0, keep_dims = mean_143_keep_dims_0, x = pow_144_cast_fp16)[name = string("mean_143_cast_fp16")]; fp16 const_3275_to_fp16 = const()[name = string("const_3275_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_405_cast_fp16 = add(x = mean_143_cast_fp16, y = const_3275_to_fp16)[name = string("add_405_cast_fp16")]; fp32 rsqrt_143_epsilon_0 = const()[name = string("rsqrt_143_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_143_cast_fp16 = rsqrt(epsilon = rsqrt_143_epsilon_0, x = add_405_cast_fp16)[name = string("rsqrt_143_cast_fp16")]; tensor mul_481_cast_fp16 = mul(x = linear_167_cast_fp16, y = rsqrt_143_cast_fp16)[name = string("mul_481_cast_fp16")]; tensor add_406_to_fp16 = const()[name = string("add_406_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304276800)))]; tensor mul_482_cast_fp16 = mul(x = mul_481_cast_fp16, y = add_406_to_fp16)[name = string("mul_482_cast_fp16")]; tensor add_407_cast_fp16 = add(x = add_402_cast_fp16, y = mul_482_cast_fp16)[name = string("add_407_cast_fp16")]; fp16 const_3280_promoted_to_fp16 = const()[name = string("const_3280_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor pow_145_cast_fp16 = pow(x = add_407_cast_fp16, y = const_3280_promoted_to_fp16)[name = string("pow_145_cast_fp16")]; tensor mean_144_axes_0 = const()[name = string("mean_144_axes_0"), val = tensor([-1])]; bool mean_144_keep_dims_0 = const()[name = string("mean_144_keep_dims_0"), val = bool(true)]; tensor mean_144_cast_fp16 = reduce_mean(axes = mean_144_axes_0, keep_dims = mean_144_keep_dims_0, x = pow_145_cast_fp16)[name = string("mean_144_cast_fp16")]; fp16 const_3283_to_fp16 = const()[name = string("const_3283_to_fp16"), val = fp16(0x1.1p-20)]; tensor add_408_cast_fp16 = add(x = mean_144_cast_fp16, y = const_3283_to_fp16)[name = string("add_408_cast_fp16")]; fp32 rsqrt_144_epsilon_0 = const()[name = string("rsqrt_144_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor rsqrt_144_cast_fp16 = rsqrt(epsilon = rsqrt_144_epsilon_0, x = add_408_cast_fp16)[name = string("rsqrt_144_cast_fp16")]; tensor mul_483_cast_fp16 = mul(x = add_407_cast_fp16, y = rsqrt_144_cast_fp16)[name = string("mul_483_cast_fp16")]; tensor add_409_to_fp16 = const()[name = string("add_409_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304278400)))]; tensor mul_484_cast_fp16 = mul(x = mul_483_cast_fp16, y = add_409_to_fp16)[name = string("mul_484_cast_fp16")]; tensor unsqueeze_173_axes_0 = const()[name = string("unsqueeze_173_axes_0"), val = tensor([-1])]; tensor unsqueeze_173 = expand_dims(axes = unsqueeze_173_axes_0, x = attention_mask)[name = string("unsqueeze_173")]; tensor expand_74_reps_0 = const()[name = string("expand_74_reps_0"), val = tensor([1, 1, 768])]; tensor expand_74 = tile(reps = expand_74_reps_0, x = unsqueeze_173)[name = string("expand_74")]; string _to_copy_507_to_fp16_dtype_0 = const()[name = string("_to_copy_507_to_fp16_dtype_0"), val = string("fp16")]; tensor expand_74_to_fp16 = cast(dtype = _to_copy_507_to_fp16_dtype_0, x = expand_74)[name = string("cast_0")]; tensor mul_485_cast_fp16 = mul(x = mul_484_cast_fp16, y = expand_74_to_fp16)[name = string("mul_485_cast_fp16")]; tensor sum_1_axes_0 = const()[name = string("sum_1_axes_0"), val = tensor([1])]; bool sum_1_keep_dims_0 = const()[name = string("sum_1_keep_dims_0"), val = bool(false)]; tensor sum_1_cast_fp16 = reduce_sum(axes = sum_1_axes_0, keep_dims = sum_1_keep_dims_0, x = mul_485_cast_fp16)[name = string("sum_1_cast_fp16")]; tensor sum_2_axes_0 = const()[name = string("sum_2_axes_0"), val = tensor([1])]; bool sum_2_keep_dims_0 = const()[name = string("sum_2_keep_dims_0"), val = bool(false)]; tensor sum_2_cast_fp16 = reduce_sum(axes = sum_2_axes_0, keep_dims = sum_2_keep_dims_0, x = expand_74_to_fp16)[name = string("sum_2_cast_fp16")]; fp16 const_3293_to_fp16 = const()[name = string("const_3293_to_fp16"), val = fp16(0x1p-24)]; fp16 const_3294_to_fp16 = const()[name = string("const_3294_to_fp16"), val = fp16(inf)]; tensor clip_0_cast_fp16 = clip(alpha = const_3293_to_fp16, beta = const_3294_to_fp16, x = sum_2_cast_fp16)[name = string("clip_0_cast_fp16")]; tensor div_cast_fp16 = real_div(x = sum_1_cast_fp16, y = clip_0_cast_fp16)[name = string("div_cast_fp16")]; int32 const_3295 = const()[name = string("const_3295"), val = int32(-1)]; bool cat_120_interleave_0 = const()[name = string("cat_120_interleave_0"), val = bool(false)]; tensor cat_120_cast_fp16 = concat(axis = const_3295, interleave = cat_120_interleave_0, values = div_cast_fp16)[name = string("cat_120_cast_fp16")]; tensor p_st_2_linear_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304280000))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306639360))))[name = string("p_st_2_linear_weight_to_fp16_quantized")]; tensor linear_168_bias_0_to_fp16 = const()[name = string("linear_168_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306645568)))]; tensor linear_168_cast_fp16 = linear(bias = linear_168_bias_0_to_fp16, weight = p_st_2_linear_weight_to_fp16_quantized, x = cat_120_cast_fp16)[name = string("linear_168_cast_fp16")]; tensor p_st_3_linear_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(309011136))))[name = string("p_st_3_linear_weight_to_fp16_quantized")]; tensor linear_169_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = p_st_3_linear_weight_to_fp16_quantized, x = linear_168_cast_fp16)[name = string("linear_169_cast_fp16")]; tensor const_3297 = const()[name = string("const_3297"), val = tensor([1])]; bool const_3298 = const()[name = string("const_3298"), val = bool(true)]; tensor linalg_vector_norm_cast_fp16 = reduce_l2_norm(axes = const_3297, keep_dims = const_3298, x = linear_169_cast_fp16)[name = string("linalg_vector_norm_cast_fp16")]; fp16 const_3299_to_fp16 = const()[name = string("const_3299_to_fp16"), val = fp16(0x1p-24)]; tensor clamp_min_cast_fp16 = maximum(x = linalg_vector_norm_cast_fp16, y = const_3299_to_fp16)[name = string("clamp_min_cast_fp16")]; tensor expand_75_reps_0 = const()[name = string("expand_75_reps_0"), val = tensor([1, 768])]; tensor expand_75_cast_fp16 = tile(reps = expand_75_reps_0, x = clamp_min_cast_fp16)[name = string("expand_75_cast_fp16")]; tensor embedding = real_div(x = linear_169_cast_fp16, y = expand_75_cast_fp16)[name = string("div_1_cast_fp16")]; } -> (embedding); }