program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.22.1"}})] { func main(tensor causal_mask, tensor conv_state_in, tensor hidden_in, tensor kv_cache_in, tensor position_ids, tensor update_mask) { tensor layers_2_self_attn_k_layernorm_weight = const()[name = string("layers_2_self_attn_k_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))]; tensor layers_2_self_attn_q_layernorm_weight = const()[name = string("layers_2_self_attn_q_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(256)))]; fp16 attn_scale = const()[name = string("attn_scale"), val = fp16(0x1p-3)]; tensor layers_0_self_attn_k_layernorm_weight = const()[name = string("layers_0_self_attn_k_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(448)))]; tensor layers_0_self_attn_q_layernorm_weight = const()[name = string("layers_0_self_attn_q_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(640)))]; tensor layers_0_operator_norm_weight = const()[name = string("layers_0_operator_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(832)))]; tensor sin_cached_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101312))))[name = string("sin_cached_palettized")]; tensor cos_cached_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207936))))[name = string("cos_cached_palettized")]; tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002688))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1006848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1400128))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1402240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1795520))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1797632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5336640))))[name = string("layers_0_feed_forward_w1_weight_palettized")]; tensor layers_0_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5355136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8894144))))[name = string("layers_0_feed_forward_w3_weight_palettized")]; tensor layers_0_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8912640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12451648))))[name = string("layers_0_feed_forward_w2_weight_palettized")]; tensor layers_1_conv_in_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12455808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14815168))))[name = string("layers_1_conv_in_proj_weight_palettized")]; tensor layers_1_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14827520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18366528))))[name = string("layers_1_feed_forward_w1_weight_palettized")]; tensor layers_1_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18385024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21924032))))[name = string("layers_1_feed_forward_w3_weight_palettized")]; tensor layers_1_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21942528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25481536))))[name = string("layers_1_feed_forward_w2_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25485696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26272192))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26276352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26669632))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26671744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27065024))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27067136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30606144))))[name = string("layers_2_feed_forward_w1_weight_palettized")]; tensor layers_2_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30624640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34163648))))[name = string("layers_2_feed_forward_w3_weight_palettized")]; tensor layers_2_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34182144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37721152))))[name = string("layers_2_feed_forward_w2_weight_palettized")]; tensor layers_3_conv_in_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37725312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40084672))))[name = string("layers_3_conv_in_proj_weight_palettized")]; tensor layers_3_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40097024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43636032))))[name = string("layers_3_feed_forward_w1_weight_palettized")]; tensor layers_3_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43654528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47193536))))[name = string("layers_3_feed_forward_w3_weight_palettized")]; tensor layers_3_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47212032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50751040))))[name = string("layers_3_feed_forward_w2_weight_palettized")]; int32 var_175_batch_dims_0 = const()[name = string("op_175_batch_dims_0"), val = int32(0)]; bool var_175_validate_indices_0 = const()[name = string("op_175_validate_indices_0"), val = bool(false)]; string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; string cast_20_dtype_0 = const()[name = string("cast_20_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; tensor cast_20 = cast(dtype = cast_20_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; tensor greater_equal_0 = greater_equal(x = cast_20, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; tensor add_0 = add(x = cast_20, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = cast_20, b = add_0, cond = greater_equal_0)[name = string("select_0")]; string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(2048)]; tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_175_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_175_cast_uint16_cast_uint16_axis_0"), val = int32(0)]; tensor op_175_cast_uint16_cast_uint16 = gather(axis = op_175_cast_uint16_cast_uint16_axis_0, batch_dims = var_175_batch_dims_0, indices = select_0_1, validate_indices = var_175_validate_indices_0, x = cos_cached_palettized)[name = string("op_175_cast_uint16_cast_uint16")]; tensor var_180 = const()[name = string("op_180"), val = tensor([1, 1, 1, 64])]; tensor cos = reshape(shape = var_180, x = op_175_cast_uint16_cast_uint16)[name = string("cos")]; int32 var_182 = const()[name = string("op_182"), val = int32(0)]; int32 var_183_batch_dims_0 = const()[name = string("op_183_batch_dims_0"), val = int32(0)]; bool var_183_validate_indices_0 = const()[name = string("op_183_validate_indices_0"), val = bool(false)]; string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; tensor var_183_cast_uint16 = gather(axis = var_182, batch_dims = var_183_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_183_validate_indices_0, x = sin_cached_palettized)[name = string("op_183_cast_uint16")]; tensor var_188 = const()[name = string("op_188"), val = tensor([1, 1, 1, 64])]; tensor sin = reshape(shape = var_188, x = var_183_cast_uint16)[name = string("sin")]; fp16 const_0_promoted = const()[name = string("const_0_promoted"), val = fp16(-0x1p+0)]; tensor var_190 = mul(x = hidden_in, y = const_0_promoted)[name = string("op_190")]; int32 var_192 = const()[name = string("op_192"), val = int32(-1)]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1 = concat(axis = var_192, interleave = input_1_interleave_0, values = (hidden_in, var_190))[name = string("input_1")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_198_to_fp16 = const()[name = string("op_198_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_198_to_fp16, x = input_1)[name = string("normed_1_cast_fp16")]; tensor var_201_split_sizes_0 = const()[name = string("op_201_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_201_axis_0 = const()[name = string("op_201_axis_0"), val = int32(-1)]; tensor var_201_0, tensor var_201_1 = split(axis = var_201_axis_0, split_sizes = var_201_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_201")]; tensor hidden_states_1 = mul(x = var_201_0, y = layers_0_operator_norm_weight)[name = string("hidden_states_1")]; tensor var_207 = const()[name = string("op_207"), val = tensor([0, 2, 1])]; tensor var_210_axes_0 = const()[name = string("op_210_axes_0"), val = tensor([2])]; tensor var_208 = transpose(perm = var_207, x = hidden_states_1)[name = string("transpose_27")]; tensor var_210 = expand_dims(axes = var_210_axes_0, x = var_208)[name = string("op_210")]; string var_226_pad_type_0 = const()[name = string("op_226_pad_type_0"), val = string("valid")]; tensor var_226_strides_0 = const()[name = string("op_226_strides_0"), val = tensor([1, 1])]; tensor var_226_pad_0 = const()[name = string("op_226_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_226_dilations_0 = const()[name = string("op_226_dilations_0"), val = tensor([1, 1])]; int32 var_226_groups_0 = const()[name = string("op_226_groups_0"), val = int32(1)]; tensor var_226 = conv(dilations = var_226_dilations_0, groups = var_226_groups_0, pad = var_226_pad_0, pad_type = var_226_pad_type_0, strides = var_226_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_210)[name = string("op_226")]; tensor var_231 = const()[name = string("op_231"), val = tensor([1, 16, 64, 1])]; tensor var_232 = reshape(shape = var_231, x = var_226)[name = string("op_232")]; tensor var_237 = const()[name = string("op_237"), val = tensor([0, 1, 3, 2])]; string var_254_pad_type_0 = const()[name = string("op_254_pad_type_0"), val = string("valid")]; tensor var_254_strides_0 = const()[name = string("op_254_strides_0"), val = tensor([1, 1])]; tensor var_254_pad_0 = const()[name = string("op_254_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_254_dilations_0 = const()[name = string("op_254_dilations_0"), val = tensor([1, 1])]; int32 var_254_groups_0 = const()[name = string("op_254_groups_0"), val = int32(1)]; tensor var_254 = conv(dilations = var_254_dilations_0, groups = var_254_groups_0, pad = var_254_pad_0, pad_type = var_254_pad_type_0, strides = var_254_strides_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_210)[name = string("op_254")]; tensor var_259 = const()[name = string("op_259"), val = tensor([1, 8, 64, 1])]; tensor var_260 = reshape(shape = var_259, x = var_254)[name = string("op_260")]; tensor var_265 = const()[name = string("op_265"), val = tensor([0, 1, 3, 2])]; string var_282_pad_type_0 = const()[name = string("op_282_pad_type_0"), val = string("valid")]; tensor var_282_strides_0 = const()[name = string("op_282_strides_0"), val = tensor([1, 1])]; tensor var_282_pad_0 = const()[name = string("op_282_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_282_dilations_0 = const()[name = string("op_282_dilations_0"), val = tensor([1, 1])]; int32 var_282_groups_0 = const()[name = string("op_282_groups_0"), val = int32(1)]; tensor var_282 = conv(dilations = var_282_dilations_0, groups = var_282_groups_0, pad = var_282_pad_0, pad_type = var_282_pad_type_0, strides = var_282_strides_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_210)[name = string("op_282")]; fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; tensor var_238 = transpose(perm = var_237, x = var_232)[name = string("transpose_26")]; tensor var_300 = mul(x = var_238, y = const_1_promoted)[name = string("op_300")]; int32 var_302 = const()[name = string("op_302"), val = int32(-1)]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_302, interleave = input_5_interleave_0, values = (var_238, var_300))[name = string("input_5")]; tensor normed_3_axes_0 = const()[name = string("normed_3_axes_0"), val = tensor([-1])]; fp16 var_308_to_fp16 = const()[name = string("op_308_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_3_cast_fp16 = layer_norm(axes = normed_3_axes_0, epsilon = var_308_to_fp16, x = input_5)[name = string("normed_3_cast_fp16")]; tensor var_311_split_sizes_0 = const()[name = string("op_311_split_sizes_0"), val = tensor([64, 64])]; int32 var_311_axis_0 = const()[name = string("op_311_axis_0"), val = int32(-1)]; tensor var_311_0, tensor var_311_1 = split(axis = var_311_axis_0, split_sizes = var_311_split_sizes_0, x = normed_3_cast_fp16)[name = string("op_311")]; tensor q_1 = mul(x = var_311_0, y = layers_0_self_attn_q_layernorm_weight)[name = string("q_1")]; fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; tensor var_266 = transpose(perm = var_265, x = var_260)[name = string("transpose_25")]; tensor var_314 = mul(x = var_266, y = const_2_promoted)[name = string("op_314")]; int32 var_316 = const()[name = string("op_316"), val = int32(-1)]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7 = concat(axis = var_316, interleave = input_7_interleave_0, values = (var_266, var_314))[name = string("input_7")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_322_to_fp16 = const()[name = string("op_322_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_322_to_fp16, x = input_7)[name = string("normed_5_cast_fp16")]; tensor var_325_split_sizes_0 = const()[name = string("op_325_split_sizes_0"), val = tensor([64, 64])]; int32 var_325_axis_0 = const()[name = string("op_325_axis_0"), val = int32(-1)]; tensor var_325_0, tensor var_325_1 = split(axis = var_325_axis_0, split_sizes = var_325_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_325")]; tensor k_1 = mul(x = var_325_0, y = layers_0_self_attn_k_layernorm_weight)[name = string("k_1")]; tensor var_328 = mul(x = q_1, y = cos)[name = string("op_328")]; tensor var_329_split_sizes_0 = const()[name = string("op_329_split_sizes_0"), val = tensor([32, 32])]; int32 var_329_axis_0 = const()[name = string("op_329_axis_0"), val = int32(-1)]; tensor var_329_0, tensor var_329_1 = split(axis = var_329_axis_0, split_sizes = var_329_split_sizes_0, x = q_1)[name = string("op_329")]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_331 = mul(x = var_329_1, y = const_3_promoted)[name = string("op_331")]; int32 var_333 = const()[name = string("op_333"), val = int32(-1)]; bool var_334_interleave_0 = const()[name = string("op_334_interleave_0"), val = bool(false)]; tensor var_334 = concat(axis = var_333, interleave = var_334_interleave_0, values = (var_331, var_329_0))[name = string("op_334")]; tensor var_335 = mul(x = var_334, y = sin)[name = string("op_335")]; tensor q_3 = add(x = var_328, y = var_335)[name = string("q_3")]; tensor var_338 = mul(x = k_1, y = cos)[name = string("op_338")]; tensor var_339_split_sizes_0 = const()[name = string("op_339_split_sizes_0"), val = tensor([32, 32])]; int32 var_339_axis_0 = const()[name = string("op_339_axis_0"), val = int32(-1)]; tensor var_339_0, tensor var_339_1 = split(axis = var_339_axis_0, split_sizes = var_339_split_sizes_0, x = k_1)[name = string("op_339")]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_341 = mul(x = var_339_1, y = const_4_promoted)[name = string("op_341")]; int32 var_343 = const()[name = string("op_343"), val = int32(-1)]; bool var_344_interleave_0 = const()[name = string("op_344_interleave_0"), val = bool(false)]; tensor var_344 = concat(axis = var_343, interleave = var_344_interleave_0, values = (var_341, var_339_0))[name = string("op_344")]; tensor var_345 = mul(x = var_344, y = sin)[name = string("op_345")]; tensor k_3 = add(x = var_338, y = var_345)[name = string("k_3")]; tensor K_cache_1_begin_0 = const()[name = string("K_cache_1_begin_0"), val = tensor([0, 0, 0, 0, 0])]; tensor K_cache_1_end_0 = const()[name = string("K_cache_1_end_0"), val = tensor([1, 1, 512, 1, 1024])]; tensor K_cache_1_end_mask_0 = const()[name = string("K_cache_1_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor K_cache_1_squeeze_mask_0 = const()[name = string("K_cache_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor K_cache_1_cast_fp16 = slice_by_index(begin = K_cache_1_begin_0, end = K_cache_1_end_0, end_mask = K_cache_1_end_mask_0, squeeze_mask = K_cache_1_squeeze_mask_0, x = kv_cache_in)[name = string("K_cache_1_cast_fp16")]; tensor V_cache_1_begin_0 = const()[name = string("V_cache_1_begin_0"), val = tensor([2, 0, 0, 0, 0])]; tensor V_cache_1_end_0 = const()[name = string("V_cache_1_end_0"), val = tensor([3, 1, 512, 1, 1024])]; tensor V_cache_1_end_mask_0 = const()[name = string("V_cache_1_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor V_cache_1_squeeze_mask_0 = const()[name = string("V_cache_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor V_cache_1_cast_fp16 = slice_by_index(begin = V_cache_1_begin_0, end = V_cache_1_end_0, end_mask = V_cache_1_end_mask_0, squeeze_mask = V_cache_1_squeeze_mask_0, x = kv_cache_in)[name = string("V_cache_1_cast_fp16")]; tensor var_358 = const()[name = string("op_358"), val = tensor([0, 1, 3, 2])]; tensor var_364 = const()[name = string("op_364"), val = tensor([1, 1024, 1, 1])]; tensor var_359 = transpose(perm = var_358, x = q_3)[name = string("transpose_24")]; tensor query_1 = reshape(shape = var_364, x = var_359)[name = string("query_1")]; tensor var_370 = const()[name = string("op_370"), val = tensor([0, 1, 3, 2])]; tensor var_376 = const()[name = string("op_376"), val = tensor([1, 512, 1, 1])]; tensor var_371 = transpose(perm = var_370, x = k_3)[name = string("transpose_23")]; tensor k_slice_1 = reshape(shape = var_376, x = var_371)[name = string("k_slice_1")]; int32 var_391 = const()[name = string("op_391"), val = int32(-1)]; bool key_1_interleave_0 = const()[name = string("key_1_interleave_0"), val = bool(false)]; tensor key_1_cast_fp16 = concat(axis = var_391, interleave = key_1_interleave_0, values = (K_cache_1_cast_fp16, k_slice_1))[name = string("key_1_cast_fp16")]; int32 var_394 = const()[name = string("op_394"), val = int32(-1)]; bool var_395_interleave_0 = const()[name = string("op_395_interleave_0"), val = bool(false)]; tensor var_395_cast_fp16 = concat(axis = var_394, interleave = var_395_interleave_0, values = (V_cache_1_cast_fp16, var_282))[name = string("op_395_cast_fp16")]; tensor var_396 = mul(x = query_1, y = attn_scale)[name = string("op_396")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50755200)))]; int32 var_399_axis_0 = const()[name = string("op_399_axis_0"), val = int32(1)]; tensor var_399_0, tensor var_399_1, tensor var_399_2, tensor var_399_3, tensor var_399_4, tensor var_399_5, tensor var_399_6, tensor var_399_7, tensor var_399_8, tensor var_399_9, tensor var_399_10, tensor var_399_11, tensor var_399_12, tensor var_399_13, tensor var_399_14, tensor var_399_15 = split(axis = var_399_axis_0, split_sizes = tile_0, x = var_396)[name = string("op_399")]; tensor var_418_perm_0 = const()[name = string("op_418_perm_0"), val = tensor([0, 3, 2, 1])]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_421_axis_0 = const()[name = string("op_421_axis_0"), val = int32(3)]; tensor var_418_cast_fp16 = transpose(perm = var_418_perm_0, x = key_1_cast_fp16)[name = string("transpose_22")]; tensor var_421_cast_fp16_0, tensor var_421_cast_fp16_1, tensor var_421_cast_fp16_2, tensor var_421_cast_fp16_3, tensor var_421_cast_fp16_4, tensor var_421_cast_fp16_5, tensor var_421_cast_fp16_6, tensor var_421_cast_fp16_7 = split(axis = var_421_axis_0, split_sizes = tile_1, x = var_418_cast_fp16)[name = string("op_421_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_432_axis_0 = const()[name = string("op_432_axis_0"), val = int32(1)]; tensor var_432_cast_fp16_0, tensor var_432_cast_fp16_1, tensor var_432_cast_fp16_2, tensor var_432_cast_fp16_3, tensor var_432_cast_fp16_4, tensor var_432_cast_fp16_5, tensor var_432_cast_fp16_6, tensor var_432_cast_fp16_7 = split(axis = var_432_axis_0, split_sizes = tile_2, x = var_395_cast_fp16)[name = string("op_432_cast_fp16")]; string scores_1_equation_0 = const()[name = string("scores_1_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_1_cast_fp16 = einsum(equation = scores_1_equation_0, values = (var_421_cast_fp16_0, var_399_0))[name = string("scores_1_cast_fp16")]; tensor var_446_cast_fp16 = add(x = scores_1_cast_fp16, y = causal_mask)[name = string("op_446_cast_fp16")]; int32 var_447 = const()[name = string("op_447"), val = int32(1)]; tensor var_449_cast_fp16 = softmax(axis = var_447, x = var_446_cast_fp16)[name = string("op_449_cast_fp16")]; string var_453_equation_0 = const()[name = string("op_453_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_453_cast_fp16 = einsum(equation = var_453_equation_0, values = (var_432_cast_fp16_0, var_449_cast_fp16))[name = string("op_453_cast_fp16")]; string scores_3_equation_0 = const()[name = string("scores_3_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_3_cast_fp16 = einsum(equation = scores_3_equation_0, values = (var_421_cast_fp16_0, var_399_1))[name = string("scores_3_cast_fp16")]; tensor var_459_cast_fp16 = add(x = scores_3_cast_fp16, y = causal_mask)[name = string("op_459_cast_fp16")]; int32 var_460 = const()[name = string("op_460"), val = int32(1)]; tensor var_462_cast_fp16 = softmax(axis = var_460, x = var_459_cast_fp16)[name = string("op_462_cast_fp16")]; string var_466_equation_0 = const()[name = string("op_466_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_466_cast_fp16 = einsum(equation = var_466_equation_0, values = (var_432_cast_fp16_0, var_462_cast_fp16))[name = string("op_466_cast_fp16")]; string scores_5_equation_0 = const()[name = string("scores_5_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_5_cast_fp16 = einsum(equation = scores_5_equation_0, values = (var_421_cast_fp16_1, var_399_2))[name = string("scores_5_cast_fp16")]; tensor var_472_cast_fp16 = add(x = scores_5_cast_fp16, y = causal_mask)[name = string("op_472_cast_fp16")]; int32 var_473 = const()[name = string("op_473"), val = int32(1)]; tensor var_475_cast_fp16 = softmax(axis = var_473, x = var_472_cast_fp16)[name = string("op_475_cast_fp16")]; string var_479_equation_0 = const()[name = string("op_479_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_479_cast_fp16 = einsum(equation = var_479_equation_0, values = (var_432_cast_fp16_1, var_475_cast_fp16))[name = string("op_479_cast_fp16")]; string scores_7_equation_0 = const()[name = string("scores_7_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_7_cast_fp16 = einsum(equation = scores_7_equation_0, values = (var_421_cast_fp16_1, var_399_3))[name = string("scores_7_cast_fp16")]; tensor var_485_cast_fp16 = add(x = scores_7_cast_fp16, y = causal_mask)[name = string("op_485_cast_fp16")]; int32 var_486 = const()[name = string("op_486"), val = int32(1)]; tensor var_488_cast_fp16 = softmax(axis = var_486, x = var_485_cast_fp16)[name = string("op_488_cast_fp16")]; string var_492_equation_0 = const()[name = string("op_492_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_492_cast_fp16 = einsum(equation = var_492_equation_0, values = (var_432_cast_fp16_1, var_488_cast_fp16))[name = string("op_492_cast_fp16")]; string scores_9_equation_0 = const()[name = string("scores_9_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_9_cast_fp16 = einsum(equation = scores_9_equation_0, values = (var_421_cast_fp16_2, var_399_4))[name = string("scores_9_cast_fp16")]; tensor var_498_cast_fp16 = add(x = scores_9_cast_fp16, y = causal_mask)[name = string("op_498_cast_fp16")]; int32 var_499 = const()[name = string("op_499"), val = int32(1)]; tensor var_501_cast_fp16 = softmax(axis = var_499, x = var_498_cast_fp16)[name = string("op_501_cast_fp16")]; string var_505_equation_0 = const()[name = string("op_505_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_505_cast_fp16 = einsum(equation = var_505_equation_0, values = (var_432_cast_fp16_2, var_501_cast_fp16))[name = string("op_505_cast_fp16")]; string scores_11_equation_0 = const()[name = string("scores_11_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_11_cast_fp16 = einsum(equation = scores_11_equation_0, values = (var_421_cast_fp16_2, var_399_5))[name = string("scores_11_cast_fp16")]; tensor var_511_cast_fp16 = add(x = scores_11_cast_fp16, y = causal_mask)[name = string("op_511_cast_fp16")]; int32 var_512 = const()[name = string("op_512"), val = int32(1)]; tensor var_514_cast_fp16 = softmax(axis = var_512, x = var_511_cast_fp16)[name = string("op_514_cast_fp16")]; string var_518_equation_0 = const()[name = string("op_518_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_518_cast_fp16 = einsum(equation = var_518_equation_0, values = (var_432_cast_fp16_2, var_514_cast_fp16))[name = string("op_518_cast_fp16")]; string scores_13_equation_0 = const()[name = string("scores_13_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_13_cast_fp16 = einsum(equation = scores_13_equation_0, values = (var_421_cast_fp16_3, var_399_6))[name = string("scores_13_cast_fp16")]; tensor var_524_cast_fp16 = add(x = scores_13_cast_fp16, y = causal_mask)[name = string("op_524_cast_fp16")]; int32 var_525 = const()[name = string("op_525"), val = int32(1)]; tensor var_527_cast_fp16 = softmax(axis = var_525, x = var_524_cast_fp16)[name = string("op_527_cast_fp16")]; string var_531_equation_0 = const()[name = string("op_531_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_531_cast_fp16 = einsum(equation = var_531_equation_0, values = (var_432_cast_fp16_3, var_527_cast_fp16))[name = string("op_531_cast_fp16")]; string scores_15_equation_0 = const()[name = string("scores_15_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_15_cast_fp16 = einsum(equation = scores_15_equation_0, values = (var_421_cast_fp16_3, var_399_7))[name = string("scores_15_cast_fp16")]; tensor var_537_cast_fp16 = add(x = scores_15_cast_fp16, y = causal_mask)[name = string("op_537_cast_fp16")]; int32 var_538 = const()[name = string("op_538"), val = int32(1)]; tensor var_540_cast_fp16 = softmax(axis = var_538, x = var_537_cast_fp16)[name = string("op_540_cast_fp16")]; string var_544_equation_0 = const()[name = string("op_544_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_544_cast_fp16 = einsum(equation = var_544_equation_0, values = (var_432_cast_fp16_3, var_540_cast_fp16))[name = string("op_544_cast_fp16")]; string scores_17_equation_0 = const()[name = string("scores_17_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_17_cast_fp16 = einsum(equation = scores_17_equation_0, values = (var_421_cast_fp16_4, var_399_8))[name = string("scores_17_cast_fp16")]; tensor var_550_cast_fp16 = add(x = scores_17_cast_fp16, y = causal_mask)[name = string("op_550_cast_fp16")]; int32 var_551 = const()[name = string("op_551"), val = int32(1)]; tensor var_553_cast_fp16 = softmax(axis = var_551, x = var_550_cast_fp16)[name = string("op_553_cast_fp16")]; string var_557_equation_0 = const()[name = string("op_557_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_557_cast_fp16 = einsum(equation = var_557_equation_0, values = (var_432_cast_fp16_4, var_553_cast_fp16))[name = string("op_557_cast_fp16")]; string scores_19_equation_0 = const()[name = string("scores_19_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_19_cast_fp16 = einsum(equation = scores_19_equation_0, values = (var_421_cast_fp16_4, var_399_9))[name = string("scores_19_cast_fp16")]; tensor var_563_cast_fp16 = add(x = scores_19_cast_fp16, y = causal_mask)[name = string("op_563_cast_fp16")]; int32 var_564 = const()[name = string("op_564"), val = int32(1)]; tensor var_566_cast_fp16 = softmax(axis = var_564, x = var_563_cast_fp16)[name = string("op_566_cast_fp16")]; string var_570_equation_0 = const()[name = string("op_570_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_570_cast_fp16 = einsum(equation = var_570_equation_0, values = (var_432_cast_fp16_4, var_566_cast_fp16))[name = string("op_570_cast_fp16")]; string scores_21_equation_0 = const()[name = string("scores_21_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_21_cast_fp16 = einsum(equation = scores_21_equation_0, values = (var_421_cast_fp16_5, var_399_10))[name = string("scores_21_cast_fp16")]; tensor var_576_cast_fp16 = add(x = scores_21_cast_fp16, y = causal_mask)[name = string("op_576_cast_fp16")]; int32 var_577 = const()[name = string("op_577"), val = int32(1)]; tensor var_579_cast_fp16 = softmax(axis = var_577, x = var_576_cast_fp16)[name = string("op_579_cast_fp16")]; string var_583_equation_0 = const()[name = string("op_583_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_583_cast_fp16 = einsum(equation = var_583_equation_0, values = (var_432_cast_fp16_5, var_579_cast_fp16))[name = string("op_583_cast_fp16")]; string scores_23_equation_0 = const()[name = string("scores_23_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_23_cast_fp16 = einsum(equation = scores_23_equation_0, values = (var_421_cast_fp16_5, var_399_11))[name = string("scores_23_cast_fp16")]; tensor var_589_cast_fp16 = add(x = scores_23_cast_fp16, y = causal_mask)[name = string("op_589_cast_fp16")]; int32 var_590 = const()[name = string("op_590"), val = int32(1)]; tensor var_592_cast_fp16 = softmax(axis = var_590, x = var_589_cast_fp16)[name = string("op_592_cast_fp16")]; string var_596_equation_0 = const()[name = string("op_596_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_596_cast_fp16 = einsum(equation = var_596_equation_0, values = (var_432_cast_fp16_5, var_592_cast_fp16))[name = string("op_596_cast_fp16")]; string scores_25_equation_0 = const()[name = string("scores_25_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_25_cast_fp16 = einsum(equation = scores_25_equation_0, values = (var_421_cast_fp16_6, var_399_12))[name = string("scores_25_cast_fp16")]; tensor var_602_cast_fp16 = add(x = scores_25_cast_fp16, y = causal_mask)[name = string("op_602_cast_fp16")]; int32 var_603 = const()[name = string("op_603"), val = int32(1)]; tensor var_605_cast_fp16 = softmax(axis = var_603, x = var_602_cast_fp16)[name = string("op_605_cast_fp16")]; string var_609_equation_0 = const()[name = string("op_609_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_609_cast_fp16 = einsum(equation = var_609_equation_0, values = (var_432_cast_fp16_6, var_605_cast_fp16))[name = string("op_609_cast_fp16")]; string scores_27_equation_0 = const()[name = string("scores_27_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_27_cast_fp16 = einsum(equation = scores_27_equation_0, values = (var_421_cast_fp16_6, var_399_13))[name = string("scores_27_cast_fp16")]; tensor var_615_cast_fp16 = add(x = scores_27_cast_fp16, y = causal_mask)[name = string("op_615_cast_fp16")]; int32 var_616 = const()[name = string("op_616"), val = int32(1)]; tensor var_618_cast_fp16 = softmax(axis = var_616, x = var_615_cast_fp16)[name = string("op_618_cast_fp16")]; string var_622_equation_0 = const()[name = string("op_622_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_622_cast_fp16 = einsum(equation = var_622_equation_0, values = (var_432_cast_fp16_6, var_618_cast_fp16))[name = string("op_622_cast_fp16")]; string scores_29_equation_0 = const()[name = string("scores_29_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_29_cast_fp16 = einsum(equation = scores_29_equation_0, values = (var_421_cast_fp16_7, var_399_14))[name = string("scores_29_cast_fp16")]; tensor var_628_cast_fp16 = add(x = scores_29_cast_fp16, y = causal_mask)[name = string("op_628_cast_fp16")]; int32 var_629 = const()[name = string("op_629"), val = int32(1)]; tensor var_631_cast_fp16 = softmax(axis = var_629, x = var_628_cast_fp16)[name = string("op_631_cast_fp16")]; string var_635_equation_0 = const()[name = string("op_635_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_635_cast_fp16 = einsum(equation = var_635_equation_0, values = (var_432_cast_fp16_7, var_631_cast_fp16))[name = string("op_635_cast_fp16")]; string scores_31_equation_0 = const()[name = string("scores_31_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_31_cast_fp16 = einsum(equation = scores_31_equation_0, values = (var_421_cast_fp16_7, var_399_15))[name = string("scores_31_cast_fp16")]; tensor var_641_cast_fp16 = add(x = scores_31_cast_fp16, y = causal_mask)[name = string("op_641_cast_fp16")]; int32 var_642 = const()[name = string("op_642"), val = int32(1)]; tensor var_644_cast_fp16 = softmax(axis = var_642, x = var_641_cast_fp16)[name = string("op_644_cast_fp16")]; string var_648_equation_0 = const()[name = string("op_648_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_648_cast_fp16 = einsum(equation = var_648_equation_0, values = (var_432_cast_fp16_7, var_644_cast_fp16))[name = string("op_648_cast_fp16")]; int32 var_650 = const()[name = string("op_650"), val = int32(1)]; bool input_9_interleave_0 = const()[name = string("input_9_interleave_0"), val = bool(false)]; tensor input_9_cast_fp16 = concat(axis = var_650, interleave = input_9_interleave_0, values = (var_453_cast_fp16, var_466_cast_fp16, var_479_cast_fp16, var_492_cast_fp16, var_505_cast_fp16, var_518_cast_fp16, var_531_cast_fp16, var_544_cast_fp16, var_557_cast_fp16, var_570_cast_fp16, var_583_cast_fp16, var_596_cast_fp16, var_609_cast_fp16, var_622_cast_fp16, var_635_cast_fp16, var_648_cast_fp16))[name = string("input_9_cast_fp16")]; string out_1_pad_type_0 = const()[name = string("out_1_pad_type_0"), val = string("valid")]; tensor out_1_strides_0 = const()[name = string("out_1_strides_0"), val = tensor([1, 1])]; tensor out_1_pad_0 = const()[name = string("out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor out_1_dilations_0 = const()[name = string("out_1_dilations_0"), val = tensor([1, 1])]; int32 out_1_groups_0 = const()[name = string("out_1_groups_0"), val = int32(1)]; tensor layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50755328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51541824))))[name = string("layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized")]; tensor out_1_cast_fp16 = conv(dilations = out_1_dilations_0, groups = out_1_groups_0, pad = out_1_pad_0, pad_type = out_1_pad_type_0, strides = out_1_strides_0, weight = layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized, x = input_9_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_664_axes_0 = const()[name = string("op_664_axes_0"), val = tensor([2])]; tensor var_664_cast_fp16 = squeeze(axes = var_664_axes_0, x = out_1_cast_fp16)[name = string("op_664_cast_fp16")]; tensor var_668 = const()[name = string("op_668"), val = tensor([0, 2, 1])]; tensor op_out_1_cast_fp16 = transpose(perm = var_668, x = var_664_cast_fp16)[name = string("transpose_21")]; tensor x_7_cast_fp16 = add(x = hidden_in, y = op_out_1_cast_fp16)[name = string("x_7_cast_fp16")]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_672_cast_fp16 = mul(x = x_7_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_672_cast_fp16")]; int32 var_674 = const()[name = string("op_674"), val = int32(-1)]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11_cast_fp16 = concat(axis = var_674, interleave = input_11_interleave_0, values = (x_7_cast_fp16, var_672_cast_fp16))[name = string("input_11_cast_fp16")]; tensor normed_7_axes_0 = const()[name = string("normed_7_axes_0"), val = tensor([-1])]; fp16 var_680_to_fp16 = const()[name = string("op_680_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_7_cast_fp16 = layer_norm(axes = normed_7_axes_0, epsilon = var_680_to_fp16, x = input_11_cast_fp16)[name = string("normed_7_cast_fp16")]; tensor var_683_split_sizes_0 = const()[name = string("op_683_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_683_axis_0 = const()[name = string("op_683_axis_0"), val = int32(-1)]; tensor var_683_cast_fp16_0, tensor var_683_cast_fp16_1 = split(axis = var_683_axis_0, split_sizes = var_683_split_sizes_0, x = normed_7_cast_fp16)[name = string("op_683_cast_fp16")]; tensor layers_0_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51545984)))]; tensor normed_9_cast_fp16 = mul(x = var_683_cast_fp16_0, y = layers_0_ffn_norm_weight_promoted_to_fp16)[name = string("normed_9_cast_fp16")]; tensor var_689 = const()[name = string("op_689"), val = tensor([0, 2, 1])]; tensor var_692_axes_0 = const()[name = string("op_692_axes_0"), val = tensor([2])]; tensor var_690_cast_fp16 = transpose(perm = var_689, x = normed_9_cast_fp16)[name = string("transpose_20")]; tensor var_692_cast_fp16 = expand_dims(axes = var_692_axes_0, x = var_690_cast_fp16)[name = string("op_692_cast_fp16")]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_0_feed_forward_w1_weight_palettized, x = var_692_cast_fp16)[name = string("input_15")]; string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = layers_0_feed_forward_w3_weight_palettized, x = var_692_cast_fp16)[name = string("b_1")]; tensor var_720 = silu(x = input_15)[name = string("op_720")]; tensor input_17 = mul(x = var_720, y = b_1)[name = string("input_17")]; string mlp_1_pad_type_0 = const()[name = string("mlp_1_pad_type_0"), val = string("valid")]; tensor mlp_1_strides_0 = const()[name = string("mlp_1_strides_0"), val = tensor([1, 1])]; tensor mlp_1_pad_0 = const()[name = string("mlp_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_1_dilations_0 = const()[name = string("mlp_1_dilations_0"), val = tensor([1, 1])]; int32 mlp_1_groups_0 = const()[name = string("mlp_1_groups_0"), val = int32(1)]; tensor mlp_1 = conv(dilations = mlp_1_dilations_0, groups = mlp_1_groups_0, pad = mlp_1_pad_0, pad_type = mlp_1_pad_type_0, strides = mlp_1_strides_0, weight = layers_0_feed_forward_w2_weight_palettized, x = input_17)[name = string("mlp_1")]; tensor var_734_axes_0 = const()[name = string("op_734_axes_0"), val = tensor([2])]; tensor var_734 = squeeze(axes = var_734_axes_0, x = mlp_1)[name = string("op_734")]; tensor var_738 = const()[name = string("op_738"), val = tensor([0, 2, 1])]; tensor mlp_3 = transpose(perm = var_738, x = var_734)[name = string("transpose_19")]; tensor x_9_cast_fp16 = add(x = x_7_cast_fp16, y = mlp_3)[name = string("x_9_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_742_cast_fp16 = mul(x = x_9_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_742_cast_fp16")]; int32 var_744 = const()[name = string("op_744"), val = int32(-1)]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_744, interleave = input_19_interleave_0, values = (x_9_cast_fp16, var_742_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_11_axes_0 = const()[name = string("normed_11_axes_0"), val = tensor([-1])]; fp16 var_750_to_fp16 = const()[name = string("op_750_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_11_cast_fp16 = layer_norm(axes = normed_11_axes_0, epsilon = var_750_to_fp16, x = input_19_cast_fp16)[name = string("normed_11_cast_fp16")]; tensor var_753_split_sizes_0 = const()[name = string("op_753_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_753_axis_0 = const()[name = string("op_753_axis_0"), val = int32(-1)]; tensor var_753_cast_fp16_0, tensor var_753_cast_fp16_1 = split(axis = var_753_axis_0, split_sizes = var_753_split_sizes_0, x = normed_11_cast_fp16)[name = string("op_753_cast_fp16")]; tensor layers_1_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_operator_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51548096)))]; tensor hidden_states_3_cast_fp16 = mul(x = var_753_cast_fp16_0, y = layers_1_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor var_759 = const()[name = string("op_759"), val = tensor([0, 2, 1])]; tensor var_762_axes_0 = const()[name = string("op_762_axes_0"), val = tensor([2])]; tensor var_760_cast_fp16 = transpose(perm = var_759, x = hidden_states_3_cast_fp16)[name = string("transpose_18")]; tensor var_762_cast_fp16 = expand_dims(axes = var_762_axes_0, x = var_760_cast_fp16)[name = string("op_762_cast_fp16")]; string BCx_1_pad_type_0 = const()[name = string("BCx_1_pad_type_0"), val = string("valid")]; tensor BCx_1_strides_0 = const()[name = string("BCx_1_strides_0"), val = tensor([1, 1])]; tensor BCx_1_pad_0 = const()[name = string("BCx_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor BCx_1_dilations_0 = const()[name = string("BCx_1_dilations_0"), val = tensor([1, 1])]; int32 BCx_1_groups_0 = const()[name = string("BCx_1_groups_0"), val = int32(1)]; tensor BCx_1 = conv(dilations = BCx_1_dilations_0, groups = BCx_1_groups_0, pad = BCx_1_pad_0, pad_type = BCx_1_pad_type_0, strides = BCx_1_strides_0, weight = layers_1_conv_in_proj_weight_palettized, x = var_762_cast_fp16)[name = string("BCx_1")]; tensor var_779_split_sizes_0 = const()[name = string("op_779_split_sizes_0"), val = tensor([1024, 1024, 1024])]; int32 var_779_axis_0 = const()[name = string("op_779_axis_0"), val = int32(1)]; tensor var_779_0, tensor var_779_1, tensor var_779_2 = split(axis = var_779_axis_0, split_sizes = var_779_split_sizes_0, x = BCx_1)[name = string("op_779")]; tensor Bx_1 = mul(x = var_779_0, y = var_779_2)[name = string("Bx_1")]; tensor var_785_begin_0 = const()[name = string("op_785_begin_0"), val = tensor([0, 0, 0])]; tensor var_785_end_0 = const()[name = string("op_785_end_0"), val = tensor([1, 1024, 3])]; tensor var_785_end_mask_0 = const()[name = string("op_785_end_mask_0"), val = tensor([false, true, true])]; tensor var_785_squeeze_mask_0 = const()[name = string("op_785_squeeze_mask_0"), val = tensor([true, false, false])]; tensor var_785_cast_fp16 = slice_by_index(begin = var_785_begin_0, end = var_785_end_0, end_mask = var_785_end_mask_0, squeeze_mask = var_785_squeeze_mask_0, x = conv_state_in)[name = string("op_785_cast_fp16")]; tensor var_787_axes_0 = const()[name = string("op_787_axes_0"), val = tensor([0])]; tensor var_787_cast_fp16 = expand_dims(axes = var_787_axes_0, x = var_785_cast_fp16)[name = string("op_787_cast_fp16")]; tensor slot_1_axes_0 = const()[name = string("slot_1_axes_0"), val = tensor([2])]; tensor slot_1_cast_fp16 = expand_dims(axes = slot_1_axes_0, x = var_787_cast_fp16)[name = string("slot_1_cast_fp16")]; tensor live_tail_1_begin_0 = const()[name = string("live_tail_1_begin_0"), val = tensor([0, 0, 0, 1])]; tensor live_tail_1_end_0 = const()[name = string("live_tail_1_end_0"), val = tensor([1, 1024, 1, 1])]; tensor live_tail_1_end_mask_0 = const()[name = string("live_tail_1_end_mask_0"), val = tensor([true, true, true, true])]; tensor live_tail_1_cast_fp16 = slice_by_index(begin = live_tail_1_begin_0, end = live_tail_1_end_0, end_mask = live_tail_1_end_mask_0, x = slot_1_cast_fp16)[name = string("live_tail_1_cast_fp16")]; int32 var_796 = const()[name = string("op_796"), val = int32(-1)]; bool new_state_1_interleave_0 = const()[name = string("new_state_1_interleave_0"), val = bool(false)]; tensor new_state_1_cast_fp16 = concat(axis = var_796, interleave = new_state_1_interleave_0, values = (live_tail_1_cast_fp16, Bx_1))[name = string("new_state_1_cast_fp16")]; tensor var_799_axes_0 = const()[name = string("op_799_axes_0"), val = tensor([0])]; tensor var_799_cast_fp16 = squeeze(axes = var_799_axes_0, x = new_state_1_cast_fp16)[name = string("op_799_cast_fp16")]; tensor var_801_axes_0 = const()[name = string("op_801_axes_0"), val = tensor([1])]; tensor var_801_cast_fp16 = squeeze(axes = var_801_axes_0, x = var_799_cast_fp16)[name = string("op_801_cast_fp16")]; string conv_out_1_pad_type_0 = const()[name = string("conv_out_1_pad_type_0"), val = string("valid")]; int32 conv_out_1_groups_0 = const()[name = string("conv_out_1_groups_0"), val = int32(1024)]; tensor conv_out_1_strides_0 = const()[name = string("conv_out_1_strides_0"), val = tensor([1, 1])]; tensor conv_out_1_pad_0 = const()[name = string("conv_out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor conv_out_1_dilations_0 = const()[name = string("conv_out_1_dilations_0"), val = tensor([1, 1])]; tensor layers_1_conv_conv_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51550208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51552576))))[name = string("layers_1_conv_conv_weight_promoted_to_fp16_palettized")]; tensor conv_out_1_cast_fp16 = conv(dilations = conv_out_1_dilations_0, groups = conv_out_1_groups_0, pad = conv_out_1_pad_0, pad_type = conv_out_1_pad_type_0, strides = conv_out_1_strides_0, weight = layers_1_conv_conv_weight_promoted_to_fp16_palettized, x = new_state_1_cast_fp16)[name = string("conv_out_1_cast_fp16")]; tensor input_23_cast_fp16 = mul(x = var_779_1, y = conv_out_1_cast_fp16)[name = string("input_23_cast_fp16")]; string y_1_pad_type_0 = const()[name = string("y_1_pad_type_0"), val = string("valid")]; tensor y_1_strides_0 = const()[name = string("y_1_strides_0"), val = tensor([1, 1])]; tensor y_1_pad_0 = const()[name = string("y_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor y_1_dilations_0 = const()[name = string("y_1_dilations_0"), val = tensor([1, 1])]; int32 y_1_groups_0 = const()[name = string("y_1_groups_0"), val = int32(1)]; tensor layers_1_conv_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51556736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52343232))))[name = string("layers_1_conv_out_proj_weight_promoted_to_fp16_palettized")]; tensor y_1_cast_fp16 = conv(dilations = y_1_dilations_0, groups = y_1_groups_0, pad = y_1_pad_0, pad_type = y_1_pad_type_0, strides = y_1_strides_0, weight = layers_1_conv_out_proj_weight_promoted_to_fp16_palettized, x = input_23_cast_fp16)[name = string("y_1_cast_fp16")]; tensor var_827_axes_0 = const()[name = string("op_827_axes_0"), val = tensor([2])]; tensor var_827_cast_fp16 = squeeze(axes = var_827_axes_0, x = y_1_cast_fp16)[name = string("op_827_cast_fp16")]; tensor var_831 = const()[name = string("op_831"), val = tensor([0, 2, 1])]; tensor op_out_3_cast_fp16 = transpose(perm = var_831, x = var_827_cast_fp16)[name = string("transpose_17")]; tensor x_11_cast_fp16 = add(x = x_9_cast_fp16, y = op_out_3_cast_fp16)[name = string("x_11_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_835_cast_fp16 = mul(x = x_11_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_835_cast_fp16")]; int32 var_837 = const()[name = string("op_837"), val = int32(-1)]; bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; tensor input_25_cast_fp16 = concat(axis = var_837, interleave = input_25_interleave_0, values = (x_11_cast_fp16, var_835_cast_fp16))[name = string("input_25_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_843_to_fp16 = const()[name = string("op_843_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_843_to_fp16, x = input_25_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor var_846_split_sizes_0 = const()[name = string("op_846_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_846_axis_0 = const()[name = string("op_846_axis_0"), val = int32(-1)]; tensor var_846_cast_fp16_0, tensor var_846_cast_fp16_1 = split(axis = var_846_axis_0, split_sizes = var_846_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_846_cast_fp16")]; tensor layers_1_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52347392)))]; tensor normed_15_cast_fp16 = mul(x = var_846_cast_fp16_0, y = layers_1_ffn_norm_weight_promoted_to_fp16)[name = string("normed_15_cast_fp16")]; tensor var_852 = const()[name = string("op_852"), val = tensor([0, 2, 1])]; tensor var_855_axes_0 = const()[name = string("op_855_axes_0"), val = tensor([2])]; tensor var_853_cast_fp16 = transpose(perm = var_852, x = normed_15_cast_fp16)[name = string("transpose_16")]; tensor var_855_cast_fp16 = expand_dims(axes = var_855_axes_0, x = var_853_cast_fp16)[name = string("op_855_cast_fp16")]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_1_feed_forward_w1_weight_palettized, x = var_855_cast_fp16)[name = string("input_29")]; string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = layers_1_feed_forward_w3_weight_palettized, x = var_855_cast_fp16)[name = string("b_3")]; tensor var_883 = silu(x = input_29)[name = string("op_883")]; tensor input_31 = mul(x = var_883, y = b_3)[name = string("input_31")]; string mlp_5_pad_type_0 = const()[name = string("mlp_5_pad_type_0"), val = string("valid")]; tensor mlp_5_strides_0 = const()[name = string("mlp_5_strides_0"), val = tensor([1, 1])]; tensor mlp_5_pad_0 = const()[name = string("mlp_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_5_dilations_0 = const()[name = string("mlp_5_dilations_0"), val = tensor([1, 1])]; int32 mlp_5_groups_0 = const()[name = string("mlp_5_groups_0"), val = int32(1)]; tensor mlp_5 = conv(dilations = mlp_5_dilations_0, groups = mlp_5_groups_0, pad = mlp_5_pad_0, pad_type = mlp_5_pad_type_0, strides = mlp_5_strides_0, weight = layers_1_feed_forward_w2_weight_palettized, x = input_31)[name = string("mlp_5")]; tensor var_897_axes_0 = const()[name = string("op_897_axes_0"), val = tensor([2])]; tensor var_897 = squeeze(axes = var_897_axes_0, x = mlp_5)[name = string("op_897")]; tensor var_901 = const()[name = string("op_901"), val = tensor([0, 2, 1])]; tensor mlp_7 = transpose(perm = var_901, x = var_897)[name = string("transpose_15")]; tensor x_13_cast_fp16 = add(x = x_11_cast_fp16, y = mlp_7)[name = string("x_13_cast_fp16")]; fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_905_cast_fp16 = mul(x = x_13_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_905_cast_fp16")]; int32 var_907 = const()[name = string("op_907"), val = int32(-1)]; bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; tensor input_33_cast_fp16 = concat(axis = var_907, interleave = input_33_interleave_0, values = (x_13_cast_fp16, var_905_cast_fp16))[name = string("input_33_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_913_to_fp16 = const()[name = string("op_913_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_913_to_fp16, x = input_33_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_916_split_sizes_0 = const()[name = string("op_916_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_916_axis_0 = const()[name = string("op_916_axis_0"), val = int32(-1)]; tensor var_916_cast_fp16_0, tensor var_916_cast_fp16_1 = split(axis = var_916_axis_0, split_sizes = var_916_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_916_cast_fp16")]; tensor layers_2_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_operator_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52349504)))]; tensor hidden_states_5_cast_fp16 = mul(x = var_916_cast_fp16_0, y = layers_2_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_5_cast_fp16")]; tensor var_922 = const()[name = string("op_922"), val = tensor([0, 2, 1])]; tensor var_925_axes_0 = const()[name = string("op_925_axes_0"), val = tensor([2])]; tensor var_923_cast_fp16 = transpose(perm = var_922, x = hidden_states_5_cast_fp16)[name = string("transpose_14")]; tensor var_925_cast_fp16 = expand_dims(axes = var_925_axes_0, x = var_923_cast_fp16)[name = string("op_925_cast_fp16")]; string var_941_pad_type_0 = const()[name = string("op_941_pad_type_0"), val = string("valid")]; tensor var_941_strides_0 = const()[name = string("op_941_strides_0"), val = tensor([1, 1])]; tensor var_941_pad_0 = const()[name = string("op_941_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_941_dilations_0 = const()[name = string("op_941_dilations_0"), val = tensor([1, 1])]; int32 var_941_groups_0 = const()[name = string("op_941_groups_0"), val = int32(1)]; tensor var_941 = conv(dilations = var_941_dilations_0, groups = var_941_groups_0, pad = var_941_pad_0, pad_type = var_941_pad_type_0, strides = var_941_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_925_cast_fp16)[name = string("op_941")]; tensor var_946 = const()[name = string("op_946"), val = tensor([1, 16, 64, 1])]; tensor var_947 = reshape(shape = var_946, x = var_941)[name = string("op_947")]; tensor var_952 = const()[name = string("op_952"), val = tensor([0, 1, 3, 2])]; string var_969_pad_type_0 = const()[name = string("op_969_pad_type_0"), val = string("valid")]; tensor var_969_strides_0 = const()[name = string("op_969_strides_0"), val = tensor([1, 1])]; tensor var_969_pad_0 = const()[name = string("op_969_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_969_dilations_0 = const()[name = string("op_969_dilations_0"), val = tensor([1, 1])]; int32 var_969_groups_0 = const()[name = string("op_969_groups_0"), val = int32(1)]; tensor var_969 = conv(dilations = var_969_dilations_0, groups = var_969_groups_0, pad = var_969_pad_0, pad_type = var_969_pad_type_0, strides = var_969_strides_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_925_cast_fp16)[name = string("op_969")]; tensor var_974 = const()[name = string("op_974"), val = tensor([1, 8, 64, 1])]; tensor var_975 = reshape(shape = var_974, x = var_969)[name = string("op_975")]; tensor var_980 = const()[name = string("op_980"), val = tensor([0, 1, 3, 2])]; string var_997_pad_type_0 = const()[name = string("op_997_pad_type_0"), val = string("valid")]; tensor var_997_strides_0 = const()[name = string("op_997_strides_0"), val = tensor([1, 1])]; tensor var_997_pad_0 = const()[name = string("op_997_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_997_dilations_0 = const()[name = string("op_997_dilations_0"), val = tensor([1, 1])]; int32 var_997_groups_0 = const()[name = string("op_997_groups_0"), val = int32(1)]; tensor var_997 = conv(dilations = var_997_dilations_0, groups = var_997_groups_0, pad = var_997_pad_0, pad_type = var_997_pad_type_0, strides = var_997_strides_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_925_cast_fp16)[name = string("op_997")]; fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; tensor var_953 = transpose(perm = var_952, x = var_947)[name = string("transpose_13")]; tensor var_1015 = mul(x = var_953, y = const_15_promoted)[name = string("op_1015")]; int32 var_1017 = const()[name = string("op_1017"), val = int32(-1)]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37 = concat(axis = var_1017, interleave = input_37_interleave_0, values = (var_953, var_1015))[name = string("input_37")]; tensor normed_19_axes_0 = const()[name = string("normed_19_axes_0"), val = tensor([-1])]; fp16 var_1023_to_fp16 = const()[name = string("op_1023_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_19_cast_fp16 = layer_norm(axes = normed_19_axes_0, epsilon = var_1023_to_fp16, x = input_37)[name = string("normed_19_cast_fp16")]; tensor var_1026_split_sizes_0 = const()[name = string("op_1026_split_sizes_0"), val = tensor([64, 64])]; int32 var_1026_axis_0 = const()[name = string("op_1026_axis_0"), val = int32(-1)]; tensor var_1026_0, tensor var_1026_1 = split(axis = var_1026_axis_0, split_sizes = var_1026_split_sizes_0, x = normed_19_cast_fp16)[name = string("op_1026")]; tensor q_5 = mul(x = var_1026_0, y = layers_2_self_attn_q_layernorm_weight)[name = string("q_5")]; fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; tensor var_981 = transpose(perm = var_980, x = var_975)[name = string("transpose_12")]; tensor var_1029 = mul(x = var_981, y = const_16_promoted)[name = string("op_1029")]; int32 var_1031 = const()[name = string("op_1031"), val = int32(-1)]; bool input_39_interleave_0 = const()[name = string("input_39_interleave_0"), val = bool(false)]; tensor input_39 = concat(axis = var_1031, interleave = input_39_interleave_0, values = (var_981, var_1029))[name = string("input_39")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_1037_to_fp16 = const()[name = string("op_1037_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1037_to_fp16, x = input_39)[name = string("normed_21_cast_fp16")]; tensor var_1040_split_sizes_0 = const()[name = string("op_1040_split_sizes_0"), val = tensor([64, 64])]; int32 var_1040_axis_0 = const()[name = string("op_1040_axis_0"), val = int32(-1)]; tensor var_1040_0, tensor var_1040_1 = split(axis = var_1040_axis_0, split_sizes = var_1040_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_1040")]; tensor k_5 = mul(x = var_1040_0, y = layers_2_self_attn_k_layernorm_weight)[name = string("k_5")]; tensor var_1043 = mul(x = q_5, y = cos)[name = string("op_1043")]; tensor var_1044_split_sizes_0 = const()[name = string("op_1044_split_sizes_0"), val = tensor([32, 32])]; int32 var_1044_axis_0 = const()[name = string("op_1044_axis_0"), val = int32(-1)]; tensor var_1044_0, tensor var_1044_1 = split(axis = var_1044_axis_0, split_sizes = var_1044_split_sizes_0, x = q_5)[name = string("op_1044")]; fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; tensor var_1046 = mul(x = var_1044_1, y = const_17_promoted)[name = string("op_1046")]; int32 var_1048 = const()[name = string("op_1048"), val = int32(-1)]; bool var_1049_interleave_0 = const()[name = string("op_1049_interleave_0"), val = bool(false)]; tensor var_1049 = concat(axis = var_1048, interleave = var_1049_interleave_0, values = (var_1046, var_1044_0))[name = string("op_1049")]; tensor var_1050 = mul(x = var_1049, y = sin)[name = string("op_1050")]; tensor q = add(x = var_1043, y = var_1050)[name = string("q")]; tensor var_1053 = mul(x = k_5, y = cos)[name = string("op_1053")]; tensor var_1054_split_sizes_0 = const()[name = string("op_1054_split_sizes_0"), val = tensor([32, 32])]; int32 var_1054_axis_0 = const()[name = string("op_1054_axis_0"), val = int32(-1)]; tensor var_1054_0, tensor var_1054_1 = split(axis = var_1054_axis_0, split_sizes = var_1054_split_sizes_0, x = k_5)[name = string("op_1054")]; fp16 const_18_promoted = const()[name = string("const_18_promoted"), val = fp16(-0x1p+0)]; tensor var_1056 = mul(x = var_1054_1, y = const_18_promoted)[name = string("op_1056")]; int32 var_1058 = const()[name = string("op_1058"), val = int32(-1)]; bool var_1059_interleave_0 = const()[name = string("op_1059_interleave_0"), val = bool(false)]; tensor var_1059 = concat(axis = var_1058, interleave = var_1059_interleave_0, values = (var_1056, var_1054_0))[name = string("op_1059")]; tensor var_1060 = mul(x = var_1059, y = sin)[name = string("op_1060")]; tensor k = add(x = var_1053, y = var_1060)[name = string("k")]; tensor K_cache_begin_0 = const()[name = string("K_cache_begin_0"), val = tensor([1, 0, 0, 0, 0])]; tensor K_cache_end_0 = const()[name = string("K_cache_end_0"), val = tensor([2, 1, 512, 1, 1024])]; tensor K_cache_end_mask_0 = const()[name = string("K_cache_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor K_cache_squeeze_mask_0 = const()[name = string("K_cache_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor K_cache_cast_fp16 = slice_by_index(begin = K_cache_begin_0, end = K_cache_end_0, end_mask = K_cache_end_mask_0, squeeze_mask = K_cache_squeeze_mask_0, x = kv_cache_in)[name = string("K_cache_cast_fp16")]; tensor V_cache_begin_0 = const()[name = string("V_cache_begin_0"), val = tensor([3, 0, 0, 0, 0])]; tensor V_cache_end_0 = const()[name = string("V_cache_end_0"), val = tensor([4, 1, 512, 1, 1024])]; tensor V_cache_end_mask_0 = const()[name = string("V_cache_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor V_cache_squeeze_mask_0 = const()[name = string("V_cache_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor V_cache_cast_fp16 = slice_by_index(begin = V_cache_begin_0, end = V_cache_end_0, end_mask = V_cache_end_mask_0, squeeze_mask = V_cache_squeeze_mask_0, x = kv_cache_in)[name = string("V_cache_cast_fp16")]; tensor var_1073 = const()[name = string("op_1073"), val = tensor([0, 1, 3, 2])]; tensor var_1079 = const()[name = string("op_1079"), val = tensor([1, 1024, 1, 1])]; tensor var_1074 = transpose(perm = var_1073, x = q)[name = string("transpose_11")]; tensor query = reshape(shape = var_1079, x = var_1074)[name = string("query")]; tensor var_1085 = const()[name = string("op_1085"), val = tensor([0, 1, 3, 2])]; tensor var_1091 = const()[name = string("op_1091"), val = tensor([1, 512, 1, 1])]; tensor var_1086 = transpose(perm = var_1085, x = k)[name = string("transpose_10")]; tensor k_slice = reshape(shape = var_1091, x = var_1086)[name = string("k_slice")]; int32 var_1106 = const()[name = string("op_1106"), val = int32(-1)]; bool key_interleave_0 = const()[name = string("key_interleave_0"), val = bool(false)]; tensor key_cast_fp16 = concat(axis = var_1106, interleave = key_interleave_0, values = (K_cache_cast_fp16, k_slice))[name = string("key_cast_fp16")]; int32 var_1109 = const()[name = string("op_1109"), val = int32(-1)]; bool var_1110_interleave_0 = const()[name = string("op_1110_interleave_0"), val = bool(false)]; tensor var_1110_cast_fp16 = concat(axis = var_1109, interleave = var_1110_interleave_0, values = (V_cache_cast_fp16, var_997))[name = string("op_1110_cast_fp16")]; tensor var_1111 = mul(x = query, y = attn_scale)[name = string("op_1111")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52351616)))]; int32 var_1114_axis_0 = const()[name = string("op_1114_axis_0"), val = int32(1)]; tensor var_1114_0, tensor var_1114_1, tensor var_1114_2, tensor var_1114_3, tensor var_1114_4, tensor var_1114_5, tensor var_1114_6, tensor var_1114_7, tensor var_1114_8, tensor var_1114_9, tensor var_1114_10, tensor var_1114_11, tensor var_1114_12, tensor var_1114_13, tensor var_1114_14, tensor var_1114_15 = split(axis = var_1114_axis_0, split_sizes = tile_3, x = var_1111)[name = string("op_1114")]; tensor var_1133_perm_0 = const()[name = string("op_1133_perm_0"), val = tensor([0, 3, 2, 1])]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_1136_axis_0 = const()[name = string("op_1136_axis_0"), val = int32(3)]; tensor var_1133_cast_fp16 = transpose(perm = var_1133_perm_0, x = key_cast_fp16)[name = string("transpose_9")]; tensor var_1136_cast_fp16_0, tensor var_1136_cast_fp16_1, tensor var_1136_cast_fp16_2, tensor var_1136_cast_fp16_3, tensor var_1136_cast_fp16_4, tensor var_1136_cast_fp16_5, tensor var_1136_cast_fp16_6, tensor var_1136_cast_fp16_7 = split(axis = var_1136_axis_0, split_sizes = tile_4, x = var_1133_cast_fp16)[name = string("op_1136_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_1147_axis_0 = const()[name = string("op_1147_axis_0"), val = int32(1)]; tensor var_1147_cast_fp16_0, tensor var_1147_cast_fp16_1, tensor var_1147_cast_fp16_2, tensor var_1147_cast_fp16_3, tensor var_1147_cast_fp16_4, tensor var_1147_cast_fp16_5, tensor var_1147_cast_fp16_6, tensor var_1147_cast_fp16_7 = split(axis = var_1147_axis_0, split_sizes = tile_5, x = var_1110_cast_fp16)[name = string("op_1147_cast_fp16")]; string scores_33_equation_0 = const()[name = string("scores_33_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_33_cast_fp16 = einsum(equation = scores_33_equation_0, values = (var_1136_cast_fp16_0, var_1114_0))[name = string("scores_33_cast_fp16")]; tensor var_1161_cast_fp16 = add(x = scores_33_cast_fp16, y = causal_mask)[name = string("op_1161_cast_fp16")]; int32 var_1162 = const()[name = string("op_1162"), val = int32(1)]; tensor var_1164_cast_fp16 = softmax(axis = var_1162, x = var_1161_cast_fp16)[name = string("op_1164_cast_fp16")]; string var_1168_equation_0 = const()[name = string("op_1168_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1168_cast_fp16 = einsum(equation = var_1168_equation_0, values = (var_1147_cast_fp16_0, var_1164_cast_fp16))[name = string("op_1168_cast_fp16")]; string scores_35_equation_0 = const()[name = string("scores_35_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_35_cast_fp16 = einsum(equation = scores_35_equation_0, values = (var_1136_cast_fp16_0, var_1114_1))[name = string("scores_35_cast_fp16")]; tensor var_1174_cast_fp16 = add(x = scores_35_cast_fp16, y = causal_mask)[name = string("op_1174_cast_fp16")]; int32 var_1175 = const()[name = string("op_1175"), val = int32(1)]; tensor var_1177_cast_fp16 = softmax(axis = var_1175, x = var_1174_cast_fp16)[name = string("op_1177_cast_fp16")]; string var_1181_equation_0 = const()[name = string("op_1181_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1181_cast_fp16 = einsum(equation = var_1181_equation_0, values = (var_1147_cast_fp16_0, var_1177_cast_fp16))[name = string("op_1181_cast_fp16")]; string scores_37_equation_0 = const()[name = string("scores_37_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_37_cast_fp16 = einsum(equation = scores_37_equation_0, values = (var_1136_cast_fp16_1, var_1114_2))[name = string("scores_37_cast_fp16")]; tensor var_1187_cast_fp16 = add(x = scores_37_cast_fp16, y = causal_mask)[name = string("op_1187_cast_fp16")]; int32 var_1188 = const()[name = string("op_1188"), val = int32(1)]; tensor var_1190_cast_fp16 = softmax(axis = var_1188, x = var_1187_cast_fp16)[name = string("op_1190_cast_fp16")]; string var_1194_equation_0 = const()[name = string("op_1194_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1194_cast_fp16 = einsum(equation = var_1194_equation_0, values = (var_1147_cast_fp16_1, var_1190_cast_fp16))[name = string("op_1194_cast_fp16")]; string scores_39_equation_0 = const()[name = string("scores_39_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_39_cast_fp16 = einsum(equation = scores_39_equation_0, values = (var_1136_cast_fp16_1, var_1114_3))[name = string("scores_39_cast_fp16")]; tensor var_1200_cast_fp16 = add(x = scores_39_cast_fp16, y = causal_mask)[name = string("op_1200_cast_fp16")]; int32 var_1201 = const()[name = string("op_1201"), val = int32(1)]; tensor var_1203_cast_fp16 = softmax(axis = var_1201, x = var_1200_cast_fp16)[name = string("op_1203_cast_fp16")]; string var_1207_equation_0 = const()[name = string("op_1207_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1207_cast_fp16 = einsum(equation = var_1207_equation_0, values = (var_1147_cast_fp16_1, var_1203_cast_fp16))[name = string("op_1207_cast_fp16")]; string scores_41_equation_0 = const()[name = string("scores_41_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_41_cast_fp16 = einsum(equation = scores_41_equation_0, values = (var_1136_cast_fp16_2, var_1114_4))[name = string("scores_41_cast_fp16")]; tensor var_1213_cast_fp16 = add(x = scores_41_cast_fp16, y = causal_mask)[name = string("op_1213_cast_fp16")]; int32 var_1214 = const()[name = string("op_1214"), val = int32(1)]; tensor var_1216_cast_fp16 = softmax(axis = var_1214, x = var_1213_cast_fp16)[name = string("op_1216_cast_fp16")]; string var_1220_equation_0 = const()[name = string("op_1220_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1220_cast_fp16 = einsum(equation = var_1220_equation_0, values = (var_1147_cast_fp16_2, var_1216_cast_fp16))[name = string("op_1220_cast_fp16")]; string scores_43_equation_0 = const()[name = string("scores_43_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_43_cast_fp16 = einsum(equation = scores_43_equation_0, values = (var_1136_cast_fp16_2, var_1114_5))[name = string("scores_43_cast_fp16")]; tensor var_1226_cast_fp16 = add(x = scores_43_cast_fp16, y = causal_mask)[name = string("op_1226_cast_fp16")]; int32 var_1227 = const()[name = string("op_1227"), val = int32(1)]; tensor var_1229_cast_fp16 = softmax(axis = var_1227, x = var_1226_cast_fp16)[name = string("op_1229_cast_fp16")]; string var_1233_equation_0 = const()[name = string("op_1233_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1233_cast_fp16 = einsum(equation = var_1233_equation_0, values = (var_1147_cast_fp16_2, var_1229_cast_fp16))[name = string("op_1233_cast_fp16")]; string scores_45_equation_0 = const()[name = string("scores_45_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_45_cast_fp16 = einsum(equation = scores_45_equation_0, values = (var_1136_cast_fp16_3, var_1114_6))[name = string("scores_45_cast_fp16")]; tensor var_1239_cast_fp16 = add(x = scores_45_cast_fp16, y = causal_mask)[name = string("op_1239_cast_fp16")]; int32 var_1240 = const()[name = string("op_1240"), val = int32(1)]; tensor var_1242_cast_fp16 = softmax(axis = var_1240, x = var_1239_cast_fp16)[name = string("op_1242_cast_fp16")]; string var_1246_equation_0 = const()[name = string("op_1246_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1246_cast_fp16 = einsum(equation = var_1246_equation_0, values = (var_1147_cast_fp16_3, var_1242_cast_fp16))[name = string("op_1246_cast_fp16")]; string scores_47_equation_0 = const()[name = string("scores_47_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_47_cast_fp16 = einsum(equation = scores_47_equation_0, values = (var_1136_cast_fp16_3, var_1114_7))[name = string("scores_47_cast_fp16")]; tensor var_1252_cast_fp16 = add(x = scores_47_cast_fp16, y = causal_mask)[name = string("op_1252_cast_fp16")]; int32 var_1253 = const()[name = string("op_1253"), val = int32(1)]; tensor var_1255_cast_fp16 = softmax(axis = var_1253, x = var_1252_cast_fp16)[name = string("op_1255_cast_fp16")]; string var_1259_equation_0 = const()[name = string("op_1259_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1259_cast_fp16 = einsum(equation = var_1259_equation_0, values = (var_1147_cast_fp16_3, var_1255_cast_fp16))[name = string("op_1259_cast_fp16")]; string scores_49_equation_0 = const()[name = string("scores_49_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_49_cast_fp16 = einsum(equation = scores_49_equation_0, values = (var_1136_cast_fp16_4, var_1114_8))[name = string("scores_49_cast_fp16")]; tensor var_1265_cast_fp16 = add(x = scores_49_cast_fp16, y = causal_mask)[name = string("op_1265_cast_fp16")]; int32 var_1266 = const()[name = string("op_1266"), val = int32(1)]; tensor var_1268_cast_fp16 = softmax(axis = var_1266, x = var_1265_cast_fp16)[name = string("op_1268_cast_fp16")]; string var_1272_equation_0 = const()[name = string("op_1272_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1272_cast_fp16 = einsum(equation = var_1272_equation_0, values = (var_1147_cast_fp16_4, var_1268_cast_fp16))[name = string("op_1272_cast_fp16")]; string scores_51_equation_0 = const()[name = string("scores_51_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_51_cast_fp16 = einsum(equation = scores_51_equation_0, values = (var_1136_cast_fp16_4, var_1114_9))[name = string("scores_51_cast_fp16")]; tensor var_1278_cast_fp16 = add(x = scores_51_cast_fp16, y = causal_mask)[name = string("op_1278_cast_fp16")]; int32 var_1279 = const()[name = string("op_1279"), val = int32(1)]; tensor var_1281_cast_fp16 = softmax(axis = var_1279, x = var_1278_cast_fp16)[name = string("op_1281_cast_fp16")]; string var_1285_equation_0 = const()[name = string("op_1285_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1285_cast_fp16 = einsum(equation = var_1285_equation_0, values = (var_1147_cast_fp16_4, var_1281_cast_fp16))[name = string("op_1285_cast_fp16")]; string scores_53_equation_0 = const()[name = string("scores_53_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_53_cast_fp16 = einsum(equation = scores_53_equation_0, values = (var_1136_cast_fp16_5, var_1114_10))[name = string("scores_53_cast_fp16")]; tensor var_1291_cast_fp16 = add(x = scores_53_cast_fp16, y = causal_mask)[name = string("op_1291_cast_fp16")]; int32 var_1292 = const()[name = string("op_1292"), val = int32(1)]; tensor var_1294_cast_fp16 = softmax(axis = var_1292, x = var_1291_cast_fp16)[name = string("op_1294_cast_fp16")]; string var_1298_equation_0 = const()[name = string("op_1298_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1298_cast_fp16 = einsum(equation = var_1298_equation_0, values = (var_1147_cast_fp16_5, var_1294_cast_fp16))[name = string("op_1298_cast_fp16")]; string scores_55_equation_0 = const()[name = string("scores_55_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_55_cast_fp16 = einsum(equation = scores_55_equation_0, values = (var_1136_cast_fp16_5, var_1114_11))[name = string("scores_55_cast_fp16")]; tensor var_1304_cast_fp16 = add(x = scores_55_cast_fp16, y = causal_mask)[name = string("op_1304_cast_fp16")]; int32 var_1305 = const()[name = string("op_1305"), val = int32(1)]; tensor var_1307_cast_fp16 = softmax(axis = var_1305, x = var_1304_cast_fp16)[name = string("op_1307_cast_fp16")]; string var_1311_equation_0 = const()[name = string("op_1311_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1311_cast_fp16 = einsum(equation = var_1311_equation_0, values = (var_1147_cast_fp16_5, var_1307_cast_fp16))[name = string("op_1311_cast_fp16")]; string scores_57_equation_0 = const()[name = string("scores_57_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_57_cast_fp16 = einsum(equation = scores_57_equation_0, values = (var_1136_cast_fp16_6, var_1114_12))[name = string("scores_57_cast_fp16")]; tensor var_1317_cast_fp16 = add(x = scores_57_cast_fp16, y = causal_mask)[name = string("op_1317_cast_fp16")]; int32 var_1318 = const()[name = string("op_1318"), val = int32(1)]; tensor var_1320_cast_fp16 = softmax(axis = var_1318, x = var_1317_cast_fp16)[name = string("op_1320_cast_fp16")]; string var_1324_equation_0 = const()[name = string("op_1324_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1324_cast_fp16 = einsum(equation = var_1324_equation_0, values = (var_1147_cast_fp16_6, var_1320_cast_fp16))[name = string("op_1324_cast_fp16")]; string scores_59_equation_0 = const()[name = string("scores_59_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_59_cast_fp16 = einsum(equation = scores_59_equation_0, values = (var_1136_cast_fp16_6, var_1114_13))[name = string("scores_59_cast_fp16")]; tensor var_1330_cast_fp16 = add(x = scores_59_cast_fp16, y = causal_mask)[name = string("op_1330_cast_fp16")]; int32 var_1331 = const()[name = string("op_1331"), val = int32(1)]; tensor var_1333_cast_fp16 = softmax(axis = var_1331, x = var_1330_cast_fp16)[name = string("op_1333_cast_fp16")]; string var_1337_equation_0 = const()[name = string("op_1337_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1337_cast_fp16 = einsum(equation = var_1337_equation_0, values = (var_1147_cast_fp16_6, var_1333_cast_fp16))[name = string("op_1337_cast_fp16")]; string scores_61_equation_0 = const()[name = string("scores_61_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_61_cast_fp16 = einsum(equation = scores_61_equation_0, values = (var_1136_cast_fp16_7, var_1114_14))[name = string("scores_61_cast_fp16")]; tensor var_1343_cast_fp16 = add(x = scores_61_cast_fp16, y = causal_mask)[name = string("op_1343_cast_fp16")]; int32 var_1344 = const()[name = string("op_1344"), val = int32(1)]; tensor var_1346_cast_fp16 = softmax(axis = var_1344, x = var_1343_cast_fp16)[name = string("op_1346_cast_fp16")]; string var_1350_equation_0 = const()[name = string("op_1350_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1350_cast_fp16 = einsum(equation = var_1350_equation_0, values = (var_1147_cast_fp16_7, var_1346_cast_fp16))[name = string("op_1350_cast_fp16")]; string scores_equation_0 = const()[name = string("scores_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_cast_fp16 = einsum(equation = scores_equation_0, values = (var_1136_cast_fp16_7, var_1114_15))[name = string("scores_cast_fp16")]; tensor var_1356_cast_fp16 = add(x = scores_cast_fp16, y = causal_mask)[name = string("op_1356_cast_fp16")]; int32 var_1357 = const()[name = string("op_1357"), val = int32(1)]; tensor var_1359_cast_fp16 = softmax(axis = var_1357, x = var_1356_cast_fp16)[name = string("op_1359_cast_fp16")]; string var_1363_equation_0 = const()[name = string("op_1363_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1363_cast_fp16 = einsum(equation = var_1363_equation_0, values = (var_1147_cast_fp16_7, var_1359_cast_fp16))[name = string("op_1363_cast_fp16")]; int32 var_1365 = const()[name = string("op_1365"), val = int32(1)]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41_cast_fp16 = concat(axis = var_1365, interleave = input_41_interleave_0, values = (var_1168_cast_fp16, var_1181_cast_fp16, var_1194_cast_fp16, var_1207_cast_fp16, var_1220_cast_fp16, var_1233_cast_fp16, var_1246_cast_fp16, var_1259_cast_fp16, var_1272_cast_fp16, var_1285_cast_fp16, var_1298_cast_fp16, var_1311_cast_fp16, var_1324_cast_fp16, var_1337_cast_fp16, var_1350_cast_fp16, var_1363_cast_fp16))[name = string("input_41_cast_fp16")]; string out_pad_type_0 = const()[name = string("out_pad_type_0"), val = string("valid")]; tensor out_strides_0 = const()[name = string("out_strides_0"), val = tensor([1, 1])]; tensor out_pad_0 = const()[name = string("out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor out_dilations_0 = const()[name = string("out_dilations_0"), val = tensor([1, 1])]; int32 out_groups_0 = const()[name = string("out_groups_0"), val = int32(1)]; tensor layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52351744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53138240))))[name = string("layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized")]; tensor out_cast_fp16 = conv(dilations = out_dilations_0, groups = out_groups_0, pad = out_pad_0, pad_type = out_pad_type_0, strides = out_strides_0, weight = layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized, x = input_41_cast_fp16)[name = string("out_cast_fp16")]; tensor var_1379_axes_0 = const()[name = string("op_1379_axes_0"), val = tensor([2])]; tensor var_1379_cast_fp16 = squeeze(axes = var_1379_axes_0, x = out_cast_fp16)[name = string("op_1379_cast_fp16")]; tensor var_1383 = const()[name = string("op_1383"), val = tensor([0, 2, 1])]; tensor op_out_5_cast_fp16 = transpose(perm = var_1383, x = var_1379_cast_fp16)[name = string("transpose_8")]; tensor x_19_cast_fp16 = add(x = x_13_cast_fp16, y = op_out_5_cast_fp16)[name = string("x_19_cast_fp16")]; fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1387_cast_fp16 = mul(x = x_19_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_1387_cast_fp16")]; int32 var_1389 = const()[name = string("op_1389"), val = int32(-1)]; bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; tensor input_43_cast_fp16 = concat(axis = var_1389, interleave = input_43_interleave_0, values = (x_19_cast_fp16, var_1387_cast_fp16))[name = string("input_43_cast_fp16")]; tensor normed_23_axes_0 = const()[name = string("normed_23_axes_0"), val = tensor([-1])]; fp16 var_1395_to_fp16 = const()[name = string("op_1395_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_23_cast_fp16 = layer_norm(axes = normed_23_axes_0, epsilon = var_1395_to_fp16, x = input_43_cast_fp16)[name = string("normed_23_cast_fp16")]; tensor var_1398_split_sizes_0 = const()[name = string("op_1398_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1398_axis_0 = const()[name = string("op_1398_axis_0"), val = int32(-1)]; tensor var_1398_cast_fp16_0, tensor var_1398_cast_fp16_1 = split(axis = var_1398_axis_0, split_sizes = var_1398_split_sizes_0, x = normed_23_cast_fp16)[name = string("op_1398_cast_fp16")]; tensor layers_2_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53142400)))]; tensor normed_25_cast_fp16 = mul(x = var_1398_cast_fp16_0, y = layers_2_ffn_norm_weight_promoted_to_fp16)[name = string("normed_25_cast_fp16")]; tensor var_1404 = const()[name = string("op_1404"), val = tensor([0, 2, 1])]; tensor var_1407_axes_0 = const()[name = string("op_1407_axes_0"), val = tensor([2])]; tensor var_1405_cast_fp16 = transpose(perm = var_1404, x = normed_25_cast_fp16)[name = string("transpose_7")]; tensor var_1407_cast_fp16 = expand_dims(axes = var_1407_axes_0, x = var_1405_cast_fp16)[name = string("op_1407_cast_fp16")]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_2_feed_forward_w1_weight_palettized, x = var_1407_cast_fp16)[name = string("input_47")]; string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = layers_2_feed_forward_w3_weight_palettized, x = var_1407_cast_fp16)[name = string("b_5")]; tensor var_1435 = silu(x = input_47)[name = string("op_1435")]; tensor input_49 = mul(x = var_1435, y = b_5)[name = string("input_49")]; string mlp_9_pad_type_0 = const()[name = string("mlp_9_pad_type_0"), val = string("valid")]; tensor mlp_9_strides_0 = const()[name = string("mlp_9_strides_0"), val = tensor([1, 1])]; tensor mlp_9_pad_0 = const()[name = string("mlp_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_9_dilations_0 = const()[name = string("mlp_9_dilations_0"), val = tensor([1, 1])]; int32 mlp_9_groups_0 = const()[name = string("mlp_9_groups_0"), val = int32(1)]; tensor mlp_9 = conv(dilations = mlp_9_dilations_0, groups = mlp_9_groups_0, pad = mlp_9_pad_0, pad_type = mlp_9_pad_type_0, strides = mlp_9_strides_0, weight = layers_2_feed_forward_w2_weight_palettized, x = input_49)[name = string("mlp_9")]; tensor var_1449_axes_0 = const()[name = string("op_1449_axes_0"), val = tensor([2])]; tensor var_1449 = squeeze(axes = var_1449_axes_0, x = mlp_9)[name = string("op_1449")]; tensor var_1453 = const()[name = string("op_1453"), val = tensor([0, 2, 1])]; tensor mlp_11 = transpose(perm = var_1453, x = var_1449)[name = string("transpose_6")]; tensor x_21_cast_fp16 = add(x = x_19_cast_fp16, y = mlp_11)[name = string("x_21_cast_fp16")]; fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1457_cast_fp16 = mul(x = x_21_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_1457_cast_fp16")]; int32 var_1459 = const()[name = string("op_1459"), val = int32(-1)]; bool input_51_interleave_0 = const()[name = string("input_51_interleave_0"), val = bool(false)]; tensor input_51_cast_fp16 = concat(axis = var_1459, interleave = input_51_interleave_0, values = (x_21_cast_fp16, var_1457_cast_fp16))[name = string("input_51_cast_fp16")]; tensor normed_27_axes_0 = const()[name = string("normed_27_axes_0"), val = tensor([-1])]; fp16 var_1465_to_fp16 = const()[name = string("op_1465_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_27_cast_fp16 = layer_norm(axes = normed_27_axes_0, epsilon = var_1465_to_fp16, x = input_51_cast_fp16)[name = string("normed_27_cast_fp16")]; tensor var_1468_split_sizes_0 = const()[name = string("op_1468_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1468_axis_0 = const()[name = string("op_1468_axis_0"), val = int32(-1)]; tensor var_1468_cast_fp16_0, tensor var_1468_cast_fp16_1 = split(axis = var_1468_axis_0, split_sizes = var_1468_split_sizes_0, x = normed_27_cast_fp16)[name = string("op_1468_cast_fp16")]; tensor layers_3_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_operator_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53144512)))]; tensor hidden_states_7_cast_fp16 = mul(x = var_1468_cast_fp16_0, y = layers_3_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor var_1474 = const()[name = string("op_1474"), val = tensor([0, 2, 1])]; tensor var_1477_axes_0 = const()[name = string("op_1477_axes_0"), val = tensor([2])]; tensor var_1475_cast_fp16 = transpose(perm = var_1474, x = hidden_states_7_cast_fp16)[name = string("transpose_5")]; tensor var_1477_cast_fp16 = expand_dims(axes = var_1477_axes_0, x = var_1475_cast_fp16)[name = string("op_1477_cast_fp16")]; string BCx_pad_type_0 = const()[name = string("BCx_pad_type_0"), val = string("valid")]; tensor BCx_strides_0 = const()[name = string("BCx_strides_0"), val = tensor([1, 1])]; tensor BCx_pad_0 = const()[name = string("BCx_pad_0"), val = tensor([0, 0, 0, 0])]; tensor BCx_dilations_0 = const()[name = string("BCx_dilations_0"), val = tensor([1, 1])]; int32 BCx_groups_0 = const()[name = string("BCx_groups_0"), val = int32(1)]; tensor BCx = conv(dilations = BCx_dilations_0, groups = BCx_groups_0, pad = BCx_pad_0, pad_type = BCx_pad_type_0, strides = BCx_strides_0, weight = layers_3_conv_in_proj_weight_palettized, x = var_1477_cast_fp16)[name = string("BCx")]; tensor var_1494_split_sizes_0 = const()[name = string("op_1494_split_sizes_0"), val = tensor([1024, 1024, 1024])]; int32 var_1494_axis_0 = const()[name = string("op_1494_axis_0"), val = int32(1)]; tensor var_1494_0, tensor var_1494_1, tensor var_1494_2 = split(axis = var_1494_axis_0, split_sizes = var_1494_split_sizes_0, x = BCx)[name = string("op_1494")]; tensor Bx = mul(x = var_1494_0, y = var_1494_2)[name = string("Bx")]; tensor var_1500_begin_0 = const()[name = string("op_1500_begin_0"), val = tensor([1, 0, 0])]; tensor var_1500_end_0 = const()[name = string("op_1500_end_0"), val = tensor([2, 1024, 3])]; tensor var_1500_end_mask_0 = const()[name = string("op_1500_end_mask_0"), val = tensor([false, true, true])]; tensor var_1500_squeeze_mask_0 = const()[name = string("op_1500_squeeze_mask_0"), val = tensor([true, false, false])]; tensor var_1500_cast_fp16 = slice_by_index(begin = var_1500_begin_0, end = var_1500_end_0, end_mask = var_1500_end_mask_0, squeeze_mask = var_1500_squeeze_mask_0, x = conv_state_in)[name = string("op_1500_cast_fp16")]; tensor var_1502_axes_0 = const()[name = string("op_1502_axes_0"), val = tensor([0])]; tensor var_1502_cast_fp16 = expand_dims(axes = var_1502_axes_0, x = var_1500_cast_fp16)[name = string("op_1502_cast_fp16")]; tensor slot_axes_0 = const()[name = string("slot_axes_0"), val = tensor([2])]; tensor slot_cast_fp16 = expand_dims(axes = slot_axes_0, x = var_1502_cast_fp16)[name = string("slot_cast_fp16")]; tensor live_tail_begin_0 = const()[name = string("live_tail_begin_0"), val = tensor([0, 0, 0, 1])]; tensor live_tail_end_0 = const()[name = string("live_tail_end_0"), val = tensor([1, 1024, 1, 1])]; tensor live_tail_end_mask_0 = const()[name = string("live_tail_end_mask_0"), val = tensor([true, true, true, true])]; tensor live_tail_cast_fp16 = slice_by_index(begin = live_tail_begin_0, end = live_tail_end_0, end_mask = live_tail_end_mask_0, x = slot_cast_fp16)[name = string("live_tail_cast_fp16")]; int32 var_1511 = const()[name = string("op_1511"), val = int32(-1)]; bool new_state_interleave_0 = const()[name = string("new_state_interleave_0"), val = bool(false)]; tensor new_state_cast_fp16 = concat(axis = var_1511, interleave = new_state_interleave_0, values = (live_tail_cast_fp16, Bx))[name = string("new_state_cast_fp16")]; tensor var_1514_axes_0 = const()[name = string("op_1514_axes_0"), val = tensor([0])]; tensor var_1514_cast_fp16 = squeeze(axes = var_1514_axes_0, x = new_state_cast_fp16)[name = string("op_1514_cast_fp16")]; tensor new_slot_axes_0 = const()[name = string("new_slot_axes_0"), val = tensor([1])]; tensor new_slot_cast_fp16 = squeeze(axes = new_slot_axes_0, x = var_1514_cast_fp16)[name = string("new_slot_cast_fp16")]; string conv_out_pad_type_0 = const()[name = string("conv_out_pad_type_0"), val = string("valid")]; int32 conv_out_groups_0 = const()[name = string("conv_out_groups_0"), val = int32(1024)]; tensor conv_out_strides_0 = const()[name = string("conv_out_strides_0"), val = tensor([1, 1])]; tensor conv_out_pad_0 = const()[name = string("conv_out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor conv_out_dilations_0 = const()[name = string("conv_out_dilations_0"), val = tensor([1, 1])]; tensor layers_3_conv_conv_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53146624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53148992))))[name = string("layers_3_conv_conv_weight_promoted_to_fp16_palettized")]; tensor conv_out_cast_fp16 = conv(dilations = conv_out_dilations_0, groups = conv_out_groups_0, pad = conv_out_pad_0, pad_type = conv_out_pad_type_0, strides = conv_out_strides_0, weight = layers_3_conv_conv_weight_promoted_to_fp16_palettized, x = new_state_cast_fp16)[name = string("conv_out_cast_fp16")]; tensor input_55_cast_fp16 = mul(x = var_1494_1, y = conv_out_cast_fp16)[name = string("input_55_cast_fp16")]; string y_pad_type_0 = const()[name = string("y_pad_type_0"), val = string("valid")]; tensor y_strides_0 = const()[name = string("y_strides_0"), val = tensor([1, 1])]; tensor y_pad_0 = const()[name = string("y_pad_0"), val = tensor([0, 0, 0, 0])]; tensor y_dilations_0 = const()[name = string("y_dilations_0"), val = tensor([1, 1])]; int32 y_groups_0 = const()[name = string("y_groups_0"), val = int32(1)]; tensor layers_3_conv_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53153152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53939648))))[name = string("layers_3_conv_out_proj_weight_promoted_to_fp16_palettized")]; tensor y_cast_fp16 = conv(dilations = y_dilations_0, groups = y_groups_0, pad = y_pad_0, pad_type = y_pad_type_0, strides = y_strides_0, weight = layers_3_conv_out_proj_weight_promoted_to_fp16_palettized, x = input_55_cast_fp16)[name = string("y_cast_fp16")]; tensor var_1542_axes_0 = const()[name = string("op_1542_axes_0"), val = tensor([2])]; tensor var_1542_cast_fp16 = squeeze(axes = var_1542_axes_0, x = y_cast_fp16)[name = string("op_1542_cast_fp16")]; tensor var_1546 = const()[name = string("op_1546"), val = tensor([0, 2, 1])]; tensor op_out_cast_fp16 = transpose(perm = var_1546, x = var_1542_cast_fp16)[name = string("transpose_4")]; tensor x_23_cast_fp16 = add(x = x_21_cast_fp16, y = op_out_cast_fp16)[name = string("x_23_cast_fp16")]; fp16 const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1550_cast_fp16 = mul(x = x_23_cast_fp16, y = const_27_promoted_to_fp16)[name = string("op_1550_cast_fp16")]; int32 var_1552 = const()[name = string("op_1552"), val = int32(-1)]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57_cast_fp16 = concat(axis = var_1552, interleave = input_57_interleave_0, values = (x_23_cast_fp16, var_1550_cast_fp16))[name = string("input_57_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1558_to_fp16, x = input_57_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_1561_split_sizes_0 = const()[name = string("op_1561_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1561_axis_0 = const()[name = string("op_1561_axis_0"), val = int32(-1)]; tensor var_1561_cast_fp16_0, tensor var_1561_cast_fp16_1 = split(axis = var_1561_axis_0, split_sizes = var_1561_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_1561_cast_fp16")]; tensor layers_3_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53943808)))]; tensor normed_31_cast_fp16 = mul(x = var_1561_cast_fp16_0, y = layers_3_ffn_norm_weight_promoted_to_fp16)[name = string("normed_31_cast_fp16")]; tensor var_1567 = const()[name = string("op_1567"), val = tensor([0, 2, 1])]; tensor var_1570_axes_0 = const()[name = string("op_1570_axes_0"), val = tensor([2])]; tensor var_1568_cast_fp16 = transpose(perm = var_1567, x = normed_31_cast_fp16)[name = string("transpose_3")]; tensor var_1570_cast_fp16 = expand_dims(axes = var_1570_axes_0, x = var_1568_cast_fp16)[name = string("op_1570_cast_fp16")]; string input_61_pad_type_0 = const()[name = string("input_61_pad_type_0"), val = string("valid")]; tensor input_61_strides_0 = const()[name = string("input_61_strides_0"), val = tensor([1, 1])]; tensor input_61_pad_0 = const()[name = string("input_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_61_dilations_0 = const()[name = string("input_61_dilations_0"), val = tensor([1, 1])]; int32 input_61_groups_0 = const()[name = string("input_61_groups_0"), val = int32(1)]; tensor input_61 = conv(dilations = input_61_dilations_0, groups = input_61_groups_0, pad = input_61_pad_0, pad_type = input_61_pad_type_0, strides = input_61_strides_0, weight = layers_3_feed_forward_w1_weight_palettized, x = var_1570_cast_fp16)[name = string("input_61")]; string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = layers_3_feed_forward_w3_weight_palettized, x = var_1570_cast_fp16)[name = string("b")]; tensor var_1598 = silu(x = input_61)[name = string("op_1598")]; tensor input_63 = mul(x = var_1598, y = b)[name = string("input_63")]; string mlp_13_pad_type_0 = const()[name = string("mlp_13_pad_type_0"), val = string("valid")]; tensor mlp_13_strides_0 = const()[name = string("mlp_13_strides_0"), val = tensor([1, 1])]; tensor mlp_13_pad_0 = const()[name = string("mlp_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_13_dilations_0 = const()[name = string("mlp_13_dilations_0"), val = tensor([1, 1])]; int32 mlp_13_groups_0 = const()[name = string("mlp_13_groups_0"), val = int32(1)]; tensor mlp_13 = conv(dilations = mlp_13_dilations_0, groups = mlp_13_groups_0, pad = mlp_13_pad_0, pad_type = mlp_13_pad_type_0, strides = mlp_13_strides_0, weight = layers_3_feed_forward_w2_weight_palettized, x = input_63)[name = string("mlp_13")]; tensor var_1612_axes_0 = const()[name = string("op_1612_axes_0"), val = tensor([2])]; tensor var_1612 = squeeze(axes = var_1612_axes_0, x = mlp_13)[name = string("op_1612")]; tensor var_1616 = const()[name = string("op_1616"), val = tensor([0, 2, 1])]; tensor mlp = transpose(perm = var_1616, x = var_1612)[name = string("transpose_2")]; tensor x_cast_fp16 = add(x = x_23_cast_fp16, y = mlp)[name = string("x_cast_fp16")]; int32 var_1622_axis_0 = const()[name = string("op_1622_axis_0"), val = int32(0)]; tensor conv_state_out = stack(axis = var_1622_axis_0, values = (var_801_cast_fp16, new_slot_cast_fp16))[name = string("op_1622_cast_fp16")]; int32 var_1625_axis_0 = const()[name = string("op_1625_axis_0"), val = int32(0)]; tensor var_1625 = stack(axis = var_1625_axis_0, values = (k_slice_1, k_slice))[name = string("op_1625")]; int32 var_1628_axis_0 = const()[name = string("op_1628_axis_0"), val = int32(0)]; tensor var_1628 = stack(axis = var_1628_axis_0, values = (var_282, var_997))[name = string("op_1628")]; int32 var_1630 = const()[name = string("op_1630"), val = int32(0)]; bool var_1631_interleave_0 = const()[name = string("op_1631_interleave_0"), val = bool(false)]; tensor kv_slice_out = concat(axis = var_1630, interleave = var_1631_interleave_0, values = (var_1625, var_1628))[name = string("op_1631")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1632_cast_fp16 = mul(x = x_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1632_cast_fp16")]; int32 var_1634 = const()[name = string("op_1634"), val = int32(-1)]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_1634, interleave = input_65_interleave_0, values = (x_cast_fp16, var_1632_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_axes_0 = const()[name = string("normed_axes_0"), val = tensor([-1])]; fp16 var_1640_to_fp16 = const()[name = string("op_1640_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_cast_fp16 = layer_norm(axes = normed_axes_0, epsilon = var_1640_to_fp16, x = input_65_cast_fp16)[name = string("normed_cast_fp16")]; tensor var_1643_split_sizes_0 = const()[name = string("op_1643_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1643_axis_0 = const()[name = string("op_1643_axis_0"), val = int32(-1)]; tensor var_1643_cast_fp16_0, tensor var_1643_cast_fp16_1 = split(axis = var_1643_axis_0, split_sizes = var_1643_split_sizes_0, x = normed_cast_fp16)[name = string("op_1643_cast_fp16")]; tensor embedding_norm_weight_promoted_to_fp16 = const()[name = string("embedding_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53945920)))]; tensor hidden_states_cast_fp16 = mul(x = var_1643_cast_fp16_0, y = embedding_norm_weight_promoted_to_fp16)[name = string("hidden_states_cast_fp16")]; tensor var_1649 = const()[name = string("op_1649"), val = tensor([0, 2, 1])]; tensor squeeze_0 = const()[name = string("squeeze_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53948032)))]; string var_1670_pad_type_0 = const()[name = string("var_1670_pad_type_0"), val = string("valid")]; int32 var_1670_groups_0 = const()[name = string("var_1670_groups_0"), val = int32(1)]; tensor var_1670_strides_0 = const()[name = string("var_1670_strides_0"), val = tensor([1])]; tensor var_1670_pad_0 = const()[name = string("var_1670_pad_0"), val = tensor([0, 0])]; tensor var_1670_dilations_0 = const()[name = string("var_1670_dilations_0"), val = tensor([1])]; tensor var_1650_cast_fp16 = transpose(perm = var_1649, x = hidden_states_cast_fp16)[name = string("transpose_1")]; tensor var_1670 = conv(dilations = var_1670_dilations_0, groups = var_1670_groups_0, pad = var_1670_pad_0, pad_type = var_1670_pad_type_0, strides = var_1670_strides_0, weight = squeeze_0, x = var_1650_cast_fp16)[name = string("var_1670")]; tensor var_1674 = const()[name = string("op_1674"), val = tensor([0, 2, 1])]; tensor logits_2d_axes_0 = const()[name = string("logits_2d_axes_0"), val = tensor([0])]; tensor logits = transpose(perm = var_1674, x = var_1670)[name = string("transpose_0")]; tensor logits_2d = squeeze(axes = logits_2d_axes_0, x = logits)[name = string("logits_2d")]; int32 token_id_axis_0 = const()[name = string("token_id_axis_0"), val = int32(-1)]; bool token_id_keep_dims_0 = const()[name = string("token_id_keep_dims_0"), val = bool(false)]; string token_id_output_dtype_0 = const()[name = string("token_id_output_dtype_0"), val = string("int32")]; tensor token_id = reduce_argmax(axis = token_id_axis_0, keep_dims = token_id_keep_dims_0, output_dtype = token_id_output_dtype_0, x = logits_2d)[name = string("token_id")]; tensor var_1682_axes_0 = const()[name = string("op_1682_axes_0"), val = tensor([-1])]; tensor var_1682 = expand_dims(axes = var_1682_axes_0, x = token_id)[name = string("op_1682")]; int32 var_1683 = const()[name = string("op_1683"), val = int32(-1)]; bool var_1685_validate_indices_0 = const()[name = string("op_1685_validate_indices_0"), val = bool(false)]; tensor var_1685 = gather_along_axis(axis = var_1683, indices = var_1682, validate_indices = var_1685_validate_indices_0, x = logits_2d)[name = string("op_1685")]; tensor var_1687_axes_0 = const()[name = string("op_1687_axes_0"), val = tensor([-1])]; tensor token_logit = squeeze(axes = var_1687_axes_0, x = var_1685)[name = string("op_1687")]; tensor update_mask_tmp = identity(x = update_mask)[name = string("update_mask_tmp")]; } -> (token_id, token_logit, kv_slice_out, conv_state_out); }