program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.22.1"}})] { func main(tensor causal_mask, tensor conv_state_in, tensor hidden_in, tensor kv_cache_in, tensor position_ids, tensor update_mask) { tensor layers_2_self_attn_k_layernorm_weight = const()[name = string("layers_2_self_attn_k_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))]; tensor layers_2_self_attn_q_layernorm_weight = const()[name = string("layers_2_self_attn_q_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(256)))]; fp16 attn_scale = const()[name = string("attn_scale"), val = fp16(0x1p-3)]; tensor layers_0_self_attn_k_layernorm_weight = const()[name = string("layers_0_self_attn_k_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(448)))]; tensor layers_0_self_attn_q_layernorm_weight = const()[name = string("layers_0_self_attn_q_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(640)))]; tensor layers_0_operator_norm_weight = const()[name = string("layers_0_operator_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(832)))]; tensor sin_cached_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101312))))[name = string("sin_cached_palettized")]; tensor cos_cached_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207936))))[name = string("cos_cached_palettized")]; tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002688))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1006848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1400128))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1402240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1795520))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1797632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5336640))))[name = string("layers_0_feed_forward_w1_weight_palettized")]; tensor layers_0_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5355136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8894144))))[name = string("layers_0_feed_forward_w3_weight_palettized")]; tensor layers_0_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8912640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12451648))))[name = string("layers_0_feed_forward_w2_weight_palettized")]; tensor layers_1_conv_in_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12455808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14815168))))[name = string("layers_1_conv_in_proj_weight_palettized")]; tensor layers_1_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14827520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18366528))))[name = string("layers_1_feed_forward_w1_weight_palettized")]; tensor layers_1_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18385024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21924032))))[name = string("layers_1_feed_forward_w3_weight_palettized")]; tensor layers_1_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21942528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25481536))))[name = string("layers_1_feed_forward_w2_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25485696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26272192))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26276352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26669632))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26671744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27065024))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27067136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30606144))))[name = string("layers_2_feed_forward_w1_weight_palettized")]; tensor layers_2_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30624640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34163648))))[name = string("layers_2_feed_forward_w3_weight_palettized")]; tensor layers_2_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34182144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37721152))))[name = string("layers_2_feed_forward_w2_weight_palettized")]; tensor layers_3_conv_in_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37725312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40084672))))[name = string("layers_3_conv_in_proj_weight_palettized")]; tensor layers_3_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40097024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43636032))))[name = string("layers_3_feed_forward_w1_weight_palettized")]; tensor layers_3_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43654528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47193536))))[name = string("layers_3_feed_forward_w3_weight_palettized")]; tensor layers_3_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47212032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50751040))))[name = string("layers_3_feed_forward_w2_weight_palettized")]; int32 var_172_batch_dims_0 = const()[name = string("op_172_batch_dims_0"), val = int32(0)]; bool var_172_validate_indices_0 = const()[name = string("op_172_validate_indices_0"), val = bool(false)]; string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; string cast_21_dtype_0 = const()[name = string("cast_21_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; tensor cast_21 = cast(dtype = cast_21_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; tensor greater_equal_0 = greater_equal(x = cast_21, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; tensor add_0 = add(x = cast_21, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = cast_21, b = add_0, cond = greater_equal_0)[name = string("select_0")]; string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(2048)]; tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_172_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_172_cast_uint16_cast_uint16_axis_0"), val = int32(0)]; tensor op_172_cast_uint16_cast_uint16 = gather(axis = op_172_cast_uint16_cast_uint16_axis_0, batch_dims = var_172_batch_dims_0, indices = select_0_1, validate_indices = var_172_validate_indices_0, x = cos_cached_palettized)[name = string("op_172_cast_uint16_cast_uint16")]; tensor var_177 = const()[name = string("op_177"), val = tensor([1, 1, 1, 64])]; tensor cos = reshape(shape = var_177, x = op_172_cast_uint16_cast_uint16)[name = string("cos")]; int32 var_179 = const()[name = string("op_179"), val = int32(0)]; int32 var_180_batch_dims_0 = const()[name = string("op_180_batch_dims_0"), val = int32(0)]; bool var_180_validate_indices_0 = const()[name = string("op_180_validate_indices_0"), val = bool(false)]; string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; tensor var_180_cast_uint16 = gather(axis = var_179, batch_dims = var_180_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_180_validate_indices_0, x = sin_cached_palettized)[name = string("op_180_cast_uint16")]; tensor var_185 = const()[name = string("op_185"), val = tensor([1, 1, 1, 64])]; tensor sin = reshape(shape = var_185, x = var_180_cast_uint16)[name = string("sin")]; fp16 const_0_promoted = const()[name = string("const_0_promoted"), val = fp16(-0x1p+0)]; tensor var_187 = mul(x = hidden_in, y = const_0_promoted)[name = string("op_187")]; int32 var_189 = const()[name = string("op_189"), val = int32(-1)]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1 = concat(axis = var_189, interleave = input_1_interleave_0, values = (hidden_in, var_187))[name = string("input_1")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_195_to_fp16 = const()[name = string("op_195_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_195_to_fp16, x = input_1)[name = string("normed_1_cast_fp16")]; tensor var_198_split_sizes_0 = const()[name = string("op_198_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_198_axis_0 = const()[name = string("op_198_axis_0"), val = int32(-1)]; tensor var_198_0, tensor var_198_1 = split(axis = var_198_axis_0, split_sizes = var_198_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_198")]; tensor hidden_states_1 = mul(x = var_198_0, y = layers_0_operator_norm_weight)[name = string("hidden_states_1")]; tensor var_204 = const()[name = string("op_204"), val = tensor([0, 2, 1])]; tensor var_207_axes_0 = const()[name = string("op_207_axes_0"), val = tensor([2])]; tensor var_205 = transpose(perm = var_204, x = hidden_states_1)[name = string("transpose_25")]; tensor var_207 = expand_dims(axes = var_207_axes_0, x = var_205)[name = string("op_207")]; string var_223_pad_type_0 = const()[name = string("op_223_pad_type_0"), val = string("valid")]; tensor var_223_strides_0 = const()[name = string("op_223_strides_0"), val = tensor([1, 1])]; tensor var_223_pad_0 = const()[name = string("op_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_223_dilations_0 = const()[name = string("op_223_dilations_0"), val = tensor([1, 1])]; int32 var_223_groups_0 = const()[name = string("op_223_groups_0"), val = int32(1)]; tensor var_223 = conv(dilations = var_223_dilations_0, groups = var_223_groups_0, pad = var_223_pad_0, pad_type = var_223_pad_type_0, strides = var_223_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_207)[name = string("op_223")]; tensor var_228 = const()[name = string("op_228"), val = tensor([1, 16, 64, 1])]; tensor var_229 = reshape(shape = var_228, x = var_223)[name = string("op_229")]; tensor var_234 = const()[name = string("op_234"), val = tensor([0, 1, 3, 2])]; string var_251_pad_type_0 = const()[name = string("op_251_pad_type_0"), val = string("valid")]; tensor var_251_strides_0 = const()[name = string("op_251_strides_0"), val = tensor([1, 1])]; tensor var_251_pad_0 = const()[name = string("op_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_251_dilations_0 = const()[name = string("op_251_dilations_0"), val = tensor([1, 1])]; int32 var_251_groups_0 = const()[name = string("op_251_groups_0"), val = int32(1)]; tensor var_251 = conv(dilations = var_251_dilations_0, groups = var_251_groups_0, pad = var_251_pad_0, pad_type = var_251_pad_type_0, strides = var_251_strides_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_207)[name = string("op_251")]; tensor var_256 = const()[name = string("op_256"), val = tensor([1, 8, 64, 1])]; tensor var_257 = reshape(shape = var_256, x = var_251)[name = string("op_257")]; tensor var_262 = const()[name = string("op_262"), val = tensor([0, 1, 3, 2])]; string var_279_pad_type_0 = const()[name = string("op_279_pad_type_0"), val = string("valid")]; tensor var_279_strides_0 = const()[name = string("op_279_strides_0"), val = tensor([1, 1])]; tensor var_279_pad_0 = const()[name = string("op_279_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_279_dilations_0 = const()[name = string("op_279_dilations_0"), val = tensor([1, 1])]; int32 var_279_groups_0 = const()[name = string("op_279_groups_0"), val = int32(1)]; tensor var_279 = conv(dilations = var_279_dilations_0, groups = var_279_groups_0, pad = var_279_pad_0, pad_type = var_279_pad_type_0, strides = var_279_strides_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_207)[name = string("op_279")]; fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; tensor var_235 = transpose(perm = var_234, x = var_229)[name = string("transpose_24")]; tensor var_297 = mul(x = var_235, y = const_1_promoted)[name = string("op_297")]; int32 var_299 = const()[name = string("op_299"), val = int32(-1)]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_299, interleave = input_5_interleave_0, values = (var_235, var_297))[name = string("input_5")]; tensor normed_3_axes_0 = const()[name = string("normed_3_axes_0"), val = tensor([-1])]; fp16 var_305_to_fp16 = const()[name = string("op_305_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_3_cast_fp16 = layer_norm(axes = normed_3_axes_0, epsilon = var_305_to_fp16, x = input_5)[name = string("normed_3_cast_fp16")]; tensor var_308_split_sizes_0 = const()[name = string("op_308_split_sizes_0"), val = tensor([64, 64])]; int32 var_308_axis_0 = const()[name = string("op_308_axis_0"), val = int32(-1)]; tensor var_308_0, tensor var_308_1 = split(axis = var_308_axis_0, split_sizes = var_308_split_sizes_0, x = normed_3_cast_fp16)[name = string("op_308")]; tensor q_1 = mul(x = var_308_0, y = layers_0_self_attn_q_layernorm_weight)[name = string("q_1")]; fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; tensor var_263 = transpose(perm = var_262, x = var_257)[name = string("transpose_23")]; tensor var_311 = mul(x = var_263, y = const_2_promoted)[name = string("op_311")]; int32 var_313 = const()[name = string("op_313"), val = int32(-1)]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7 = concat(axis = var_313, interleave = input_7_interleave_0, values = (var_263, var_311))[name = string("input_7")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_319_to_fp16 = const()[name = string("op_319_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_319_to_fp16, x = input_7)[name = string("normed_5_cast_fp16")]; tensor var_322_split_sizes_0 = const()[name = string("op_322_split_sizes_0"), val = tensor([64, 64])]; int32 var_322_axis_0 = const()[name = string("op_322_axis_0"), val = int32(-1)]; tensor var_322_0, tensor var_322_1 = split(axis = var_322_axis_0, split_sizes = var_322_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_322")]; tensor k_1 = mul(x = var_322_0, y = layers_0_self_attn_k_layernorm_weight)[name = string("k_1")]; tensor var_325 = mul(x = q_1, y = cos)[name = string("op_325")]; tensor var_326_split_sizes_0 = const()[name = string("op_326_split_sizes_0"), val = tensor([32, 32])]; int32 var_326_axis_0 = const()[name = string("op_326_axis_0"), val = int32(-1)]; tensor var_326_0, tensor var_326_1 = split(axis = var_326_axis_0, split_sizes = var_326_split_sizes_0, x = q_1)[name = string("op_326")]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_328 = mul(x = var_326_1, y = const_3_promoted)[name = string("op_328")]; int32 var_330 = const()[name = string("op_330"), val = int32(-1)]; bool var_331_interleave_0 = const()[name = string("op_331_interleave_0"), val = bool(false)]; tensor var_331 = concat(axis = var_330, interleave = var_331_interleave_0, values = (var_328, var_326_0))[name = string("op_331")]; tensor var_332 = mul(x = var_331, y = sin)[name = string("op_332")]; tensor q_3 = add(x = var_325, y = var_332)[name = string("q_3")]; tensor var_335 = mul(x = k_1, y = cos)[name = string("op_335")]; tensor var_336_split_sizes_0 = const()[name = string("op_336_split_sizes_0"), val = tensor([32, 32])]; int32 var_336_axis_0 = const()[name = string("op_336_axis_0"), val = int32(-1)]; tensor var_336_0, tensor var_336_1 = split(axis = var_336_axis_0, split_sizes = var_336_split_sizes_0, x = k_1)[name = string("op_336")]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_338 = mul(x = var_336_1, y = const_4_promoted)[name = string("op_338")]; int32 var_340 = const()[name = string("op_340"), val = int32(-1)]; bool var_341_interleave_0 = const()[name = string("op_341_interleave_0"), val = bool(false)]; tensor var_341 = concat(axis = var_340, interleave = var_341_interleave_0, values = (var_338, var_336_0))[name = string("op_341")]; tensor var_342 = mul(x = var_341, y = sin)[name = string("op_342")]; tensor k_3 = add(x = var_335, y = var_342)[name = string("k_3")]; tensor K_cache_1_begin_0 = const()[name = string("K_cache_1_begin_0"), val = tensor([0, 0, 0, 0, 0])]; tensor K_cache_1_end_0 = const()[name = string("K_cache_1_end_0"), val = tensor([1, 1, 512, 1, 1024])]; tensor K_cache_1_end_mask_0 = const()[name = string("K_cache_1_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor K_cache_1_squeeze_mask_0 = const()[name = string("K_cache_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor K_cache_1_cast_fp16 = slice_by_index(begin = K_cache_1_begin_0, end = K_cache_1_end_0, end_mask = K_cache_1_end_mask_0, squeeze_mask = K_cache_1_squeeze_mask_0, x = kv_cache_in)[name = string("K_cache_1_cast_fp16")]; tensor V_cache_1_begin_0 = const()[name = string("V_cache_1_begin_0"), val = tensor([2, 0, 0, 0, 0])]; tensor V_cache_1_end_0 = const()[name = string("V_cache_1_end_0"), val = tensor([3, 1, 512, 1, 1024])]; tensor V_cache_1_end_mask_0 = const()[name = string("V_cache_1_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor V_cache_1_squeeze_mask_0 = const()[name = string("V_cache_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor V_cache_1_cast_fp16 = slice_by_index(begin = V_cache_1_begin_0, end = V_cache_1_end_0, end_mask = V_cache_1_end_mask_0, squeeze_mask = V_cache_1_squeeze_mask_0, x = kv_cache_in)[name = string("V_cache_1_cast_fp16")]; tensor var_355 = const()[name = string("op_355"), val = tensor([0, 1, 3, 2])]; tensor var_361 = const()[name = string("op_361"), val = tensor([1, 1024, 1, 1])]; tensor var_356 = transpose(perm = var_355, x = q_3)[name = string("transpose_22")]; tensor query_1 = reshape(shape = var_361, x = var_356)[name = string("query_1")]; tensor var_367 = const()[name = string("op_367"), val = tensor([0, 1, 3, 2])]; tensor var_373 = const()[name = string("op_373"), val = tensor([1, 512, 1, 1])]; tensor var_368 = transpose(perm = var_367, x = k_3)[name = string("transpose_21")]; tensor k_slice_1 = reshape(shape = var_373, x = var_368)[name = string("k_slice_1")]; int32 var_388 = const()[name = string("op_388"), val = int32(-1)]; bool key_1_interleave_0 = const()[name = string("key_1_interleave_0"), val = bool(false)]; tensor key_1_cast_fp16 = concat(axis = var_388, interleave = key_1_interleave_0, values = (K_cache_1_cast_fp16, k_slice_1))[name = string("key_1_cast_fp16")]; int32 var_391 = const()[name = string("op_391"), val = int32(-1)]; bool var_392_interleave_0 = const()[name = string("op_392_interleave_0"), val = bool(false)]; tensor var_392_cast_fp16 = concat(axis = var_391, interleave = var_392_interleave_0, values = (V_cache_1_cast_fp16, var_279))[name = string("op_392_cast_fp16")]; tensor var_393 = mul(x = query_1, y = attn_scale)[name = string("op_393")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50755200)))]; int32 var_396_axis_0 = const()[name = string("op_396_axis_0"), val = int32(1)]; tensor var_396_0, tensor var_396_1, tensor var_396_2, tensor var_396_3, tensor var_396_4, tensor var_396_5, tensor var_396_6, tensor var_396_7, tensor var_396_8, tensor var_396_9, tensor var_396_10, tensor var_396_11, tensor var_396_12, tensor var_396_13, tensor var_396_14, tensor var_396_15 = split(axis = var_396_axis_0, split_sizes = tile_0, x = var_393)[name = string("op_396")]; tensor var_415_perm_0 = const()[name = string("op_415_perm_0"), val = tensor([0, 3, 2, 1])]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_418_axis_0 = const()[name = string("op_418_axis_0"), val = int32(3)]; tensor var_415_cast_fp16 = transpose(perm = var_415_perm_0, x = key_1_cast_fp16)[name = string("transpose_20")]; tensor var_418_cast_fp16_0, tensor var_418_cast_fp16_1, tensor var_418_cast_fp16_2, tensor var_418_cast_fp16_3, tensor var_418_cast_fp16_4, tensor var_418_cast_fp16_5, tensor var_418_cast_fp16_6, tensor var_418_cast_fp16_7 = split(axis = var_418_axis_0, split_sizes = tile_1, x = var_415_cast_fp16)[name = string("op_418_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_429_axis_0 = const()[name = string("op_429_axis_0"), val = int32(1)]; tensor var_429_cast_fp16_0, tensor var_429_cast_fp16_1, tensor var_429_cast_fp16_2, tensor var_429_cast_fp16_3, tensor var_429_cast_fp16_4, tensor var_429_cast_fp16_5, tensor var_429_cast_fp16_6, tensor var_429_cast_fp16_7 = split(axis = var_429_axis_0, split_sizes = tile_2, x = var_392_cast_fp16)[name = string("op_429_cast_fp16")]; string scores_1_equation_0 = const()[name = string("scores_1_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_1_cast_fp16 = einsum(equation = scores_1_equation_0, values = (var_418_cast_fp16_0, var_396_0))[name = string("scores_1_cast_fp16")]; tensor var_443_cast_fp16 = add(x = scores_1_cast_fp16, y = causal_mask)[name = string("op_443_cast_fp16")]; int32 var_444 = const()[name = string("op_444"), val = int32(1)]; tensor var_446_cast_fp16 = softmax(axis = var_444, x = var_443_cast_fp16)[name = string("op_446_cast_fp16")]; string var_450_equation_0 = const()[name = string("op_450_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_450_cast_fp16 = einsum(equation = var_450_equation_0, values = (var_429_cast_fp16_0, var_446_cast_fp16))[name = string("op_450_cast_fp16")]; string scores_3_equation_0 = const()[name = string("scores_3_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_3_cast_fp16 = einsum(equation = scores_3_equation_0, values = (var_418_cast_fp16_0, var_396_1))[name = string("scores_3_cast_fp16")]; tensor var_456_cast_fp16 = add(x = scores_3_cast_fp16, y = causal_mask)[name = string("op_456_cast_fp16")]; int32 var_457 = const()[name = string("op_457"), val = int32(1)]; tensor var_459_cast_fp16 = softmax(axis = var_457, x = var_456_cast_fp16)[name = string("op_459_cast_fp16")]; string var_463_equation_0 = const()[name = string("op_463_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_463_cast_fp16 = einsum(equation = var_463_equation_0, values = (var_429_cast_fp16_0, var_459_cast_fp16))[name = string("op_463_cast_fp16")]; string scores_5_equation_0 = const()[name = string("scores_5_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_5_cast_fp16 = einsum(equation = scores_5_equation_0, values = (var_418_cast_fp16_1, var_396_2))[name = string("scores_5_cast_fp16")]; tensor var_469_cast_fp16 = add(x = scores_5_cast_fp16, y = causal_mask)[name = string("op_469_cast_fp16")]; int32 var_470 = const()[name = string("op_470"), val = int32(1)]; tensor var_472_cast_fp16 = softmax(axis = var_470, x = var_469_cast_fp16)[name = string("op_472_cast_fp16")]; string var_476_equation_0 = const()[name = string("op_476_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_476_cast_fp16 = einsum(equation = var_476_equation_0, values = (var_429_cast_fp16_1, var_472_cast_fp16))[name = string("op_476_cast_fp16")]; string scores_7_equation_0 = const()[name = string("scores_7_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_7_cast_fp16 = einsum(equation = scores_7_equation_0, values = (var_418_cast_fp16_1, var_396_3))[name = string("scores_7_cast_fp16")]; tensor var_482_cast_fp16 = add(x = scores_7_cast_fp16, y = causal_mask)[name = string("op_482_cast_fp16")]; int32 var_483 = const()[name = string("op_483"), val = int32(1)]; tensor var_485_cast_fp16 = softmax(axis = var_483, x = var_482_cast_fp16)[name = string("op_485_cast_fp16")]; string var_489_equation_0 = const()[name = string("op_489_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_489_cast_fp16 = einsum(equation = var_489_equation_0, values = (var_429_cast_fp16_1, var_485_cast_fp16))[name = string("op_489_cast_fp16")]; string scores_9_equation_0 = const()[name = string("scores_9_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_9_cast_fp16 = einsum(equation = scores_9_equation_0, values = (var_418_cast_fp16_2, var_396_4))[name = string("scores_9_cast_fp16")]; tensor var_495_cast_fp16 = add(x = scores_9_cast_fp16, y = causal_mask)[name = string("op_495_cast_fp16")]; int32 var_496 = const()[name = string("op_496"), val = int32(1)]; tensor var_498_cast_fp16 = softmax(axis = var_496, x = var_495_cast_fp16)[name = string("op_498_cast_fp16")]; string var_502_equation_0 = const()[name = string("op_502_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_502_cast_fp16 = einsum(equation = var_502_equation_0, values = (var_429_cast_fp16_2, var_498_cast_fp16))[name = string("op_502_cast_fp16")]; string scores_11_equation_0 = const()[name = string("scores_11_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_11_cast_fp16 = einsum(equation = scores_11_equation_0, values = (var_418_cast_fp16_2, var_396_5))[name = string("scores_11_cast_fp16")]; tensor var_508_cast_fp16 = add(x = scores_11_cast_fp16, y = causal_mask)[name = string("op_508_cast_fp16")]; int32 var_509 = const()[name = string("op_509"), val = int32(1)]; tensor var_511_cast_fp16 = softmax(axis = var_509, x = var_508_cast_fp16)[name = string("op_511_cast_fp16")]; string var_515_equation_0 = const()[name = string("op_515_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_515_cast_fp16 = einsum(equation = var_515_equation_0, values = (var_429_cast_fp16_2, var_511_cast_fp16))[name = string("op_515_cast_fp16")]; string scores_13_equation_0 = const()[name = string("scores_13_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_13_cast_fp16 = einsum(equation = scores_13_equation_0, values = (var_418_cast_fp16_3, var_396_6))[name = string("scores_13_cast_fp16")]; tensor var_521_cast_fp16 = add(x = scores_13_cast_fp16, y = causal_mask)[name = string("op_521_cast_fp16")]; int32 var_522 = const()[name = string("op_522"), val = int32(1)]; tensor var_524_cast_fp16 = softmax(axis = var_522, x = var_521_cast_fp16)[name = string("op_524_cast_fp16")]; string var_528_equation_0 = const()[name = string("op_528_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_528_cast_fp16 = einsum(equation = var_528_equation_0, values = (var_429_cast_fp16_3, var_524_cast_fp16))[name = string("op_528_cast_fp16")]; string scores_15_equation_0 = const()[name = string("scores_15_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_15_cast_fp16 = einsum(equation = scores_15_equation_0, values = (var_418_cast_fp16_3, var_396_7))[name = string("scores_15_cast_fp16")]; tensor var_534_cast_fp16 = add(x = scores_15_cast_fp16, y = causal_mask)[name = string("op_534_cast_fp16")]; int32 var_535 = const()[name = string("op_535"), val = int32(1)]; tensor var_537_cast_fp16 = softmax(axis = var_535, x = var_534_cast_fp16)[name = string("op_537_cast_fp16")]; string var_541_equation_0 = const()[name = string("op_541_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_541_cast_fp16 = einsum(equation = var_541_equation_0, values = (var_429_cast_fp16_3, var_537_cast_fp16))[name = string("op_541_cast_fp16")]; string scores_17_equation_0 = const()[name = string("scores_17_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_17_cast_fp16 = einsum(equation = scores_17_equation_0, values = (var_418_cast_fp16_4, var_396_8))[name = string("scores_17_cast_fp16")]; tensor var_547_cast_fp16 = add(x = scores_17_cast_fp16, y = causal_mask)[name = string("op_547_cast_fp16")]; int32 var_548 = const()[name = string("op_548"), val = int32(1)]; tensor var_550_cast_fp16 = softmax(axis = var_548, x = var_547_cast_fp16)[name = string("op_550_cast_fp16")]; string var_554_equation_0 = const()[name = string("op_554_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_554_cast_fp16 = einsum(equation = var_554_equation_0, values = (var_429_cast_fp16_4, var_550_cast_fp16))[name = string("op_554_cast_fp16")]; string scores_19_equation_0 = const()[name = string("scores_19_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_19_cast_fp16 = einsum(equation = scores_19_equation_0, values = (var_418_cast_fp16_4, var_396_9))[name = string("scores_19_cast_fp16")]; tensor var_560_cast_fp16 = add(x = scores_19_cast_fp16, y = causal_mask)[name = string("op_560_cast_fp16")]; int32 var_561 = const()[name = string("op_561"), val = int32(1)]; tensor var_563_cast_fp16 = softmax(axis = var_561, x = var_560_cast_fp16)[name = string("op_563_cast_fp16")]; string var_567_equation_0 = const()[name = string("op_567_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_567_cast_fp16 = einsum(equation = var_567_equation_0, values = (var_429_cast_fp16_4, var_563_cast_fp16))[name = string("op_567_cast_fp16")]; string scores_21_equation_0 = const()[name = string("scores_21_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_21_cast_fp16 = einsum(equation = scores_21_equation_0, values = (var_418_cast_fp16_5, var_396_10))[name = string("scores_21_cast_fp16")]; tensor var_573_cast_fp16 = add(x = scores_21_cast_fp16, y = causal_mask)[name = string("op_573_cast_fp16")]; int32 var_574 = const()[name = string("op_574"), val = int32(1)]; tensor var_576_cast_fp16 = softmax(axis = var_574, x = var_573_cast_fp16)[name = string("op_576_cast_fp16")]; string var_580_equation_0 = const()[name = string("op_580_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_580_cast_fp16 = einsum(equation = var_580_equation_0, values = (var_429_cast_fp16_5, var_576_cast_fp16))[name = string("op_580_cast_fp16")]; string scores_23_equation_0 = const()[name = string("scores_23_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_23_cast_fp16 = einsum(equation = scores_23_equation_0, values = (var_418_cast_fp16_5, var_396_11))[name = string("scores_23_cast_fp16")]; tensor var_586_cast_fp16 = add(x = scores_23_cast_fp16, y = causal_mask)[name = string("op_586_cast_fp16")]; int32 var_587 = const()[name = string("op_587"), val = int32(1)]; tensor var_589_cast_fp16 = softmax(axis = var_587, x = var_586_cast_fp16)[name = string("op_589_cast_fp16")]; string var_593_equation_0 = const()[name = string("op_593_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_593_cast_fp16 = einsum(equation = var_593_equation_0, values = (var_429_cast_fp16_5, var_589_cast_fp16))[name = string("op_593_cast_fp16")]; string scores_25_equation_0 = const()[name = string("scores_25_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_25_cast_fp16 = einsum(equation = scores_25_equation_0, values = (var_418_cast_fp16_6, var_396_12))[name = string("scores_25_cast_fp16")]; tensor var_599_cast_fp16 = add(x = scores_25_cast_fp16, y = causal_mask)[name = string("op_599_cast_fp16")]; int32 var_600 = const()[name = string("op_600"), val = int32(1)]; tensor var_602_cast_fp16 = softmax(axis = var_600, x = var_599_cast_fp16)[name = string("op_602_cast_fp16")]; string var_606_equation_0 = const()[name = string("op_606_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_606_cast_fp16 = einsum(equation = var_606_equation_0, values = (var_429_cast_fp16_6, var_602_cast_fp16))[name = string("op_606_cast_fp16")]; string scores_27_equation_0 = const()[name = string("scores_27_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_27_cast_fp16 = einsum(equation = scores_27_equation_0, values = (var_418_cast_fp16_6, var_396_13))[name = string("scores_27_cast_fp16")]; tensor var_612_cast_fp16 = add(x = scores_27_cast_fp16, y = causal_mask)[name = string("op_612_cast_fp16")]; int32 var_613 = const()[name = string("op_613"), val = int32(1)]; tensor var_615_cast_fp16 = softmax(axis = var_613, x = var_612_cast_fp16)[name = string("op_615_cast_fp16")]; string var_619_equation_0 = const()[name = string("op_619_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_619_cast_fp16 = einsum(equation = var_619_equation_0, values = (var_429_cast_fp16_6, var_615_cast_fp16))[name = string("op_619_cast_fp16")]; string scores_29_equation_0 = const()[name = string("scores_29_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_29_cast_fp16 = einsum(equation = scores_29_equation_0, values = (var_418_cast_fp16_7, var_396_14))[name = string("scores_29_cast_fp16")]; tensor var_625_cast_fp16 = add(x = scores_29_cast_fp16, y = causal_mask)[name = string("op_625_cast_fp16")]; int32 var_626 = const()[name = string("op_626"), val = int32(1)]; tensor var_628_cast_fp16 = softmax(axis = var_626, x = var_625_cast_fp16)[name = string("op_628_cast_fp16")]; string var_632_equation_0 = const()[name = string("op_632_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_632_cast_fp16 = einsum(equation = var_632_equation_0, values = (var_429_cast_fp16_7, var_628_cast_fp16))[name = string("op_632_cast_fp16")]; string scores_31_equation_0 = const()[name = string("scores_31_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_31_cast_fp16 = einsum(equation = scores_31_equation_0, values = (var_418_cast_fp16_7, var_396_15))[name = string("scores_31_cast_fp16")]; tensor var_638_cast_fp16 = add(x = scores_31_cast_fp16, y = causal_mask)[name = string("op_638_cast_fp16")]; int32 var_639 = const()[name = string("op_639"), val = int32(1)]; tensor var_641_cast_fp16 = softmax(axis = var_639, x = var_638_cast_fp16)[name = string("op_641_cast_fp16")]; string var_645_equation_0 = const()[name = string("op_645_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_645_cast_fp16 = einsum(equation = var_645_equation_0, values = (var_429_cast_fp16_7, var_641_cast_fp16))[name = string("op_645_cast_fp16")]; int32 var_647 = const()[name = string("op_647"), val = int32(1)]; bool input_9_interleave_0 = const()[name = string("input_9_interleave_0"), val = bool(false)]; tensor input_9_cast_fp16 = concat(axis = var_647, interleave = input_9_interleave_0, values = (var_450_cast_fp16, var_463_cast_fp16, var_476_cast_fp16, var_489_cast_fp16, var_502_cast_fp16, var_515_cast_fp16, var_528_cast_fp16, var_541_cast_fp16, var_554_cast_fp16, var_567_cast_fp16, var_580_cast_fp16, var_593_cast_fp16, var_606_cast_fp16, var_619_cast_fp16, var_632_cast_fp16, var_645_cast_fp16))[name = string("input_9_cast_fp16")]; string out_1_pad_type_0 = const()[name = string("out_1_pad_type_0"), val = string("valid")]; tensor out_1_strides_0 = const()[name = string("out_1_strides_0"), val = tensor([1, 1])]; tensor out_1_pad_0 = const()[name = string("out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor out_1_dilations_0 = const()[name = string("out_1_dilations_0"), val = tensor([1, 1])]; int32 out_1_groups_0 = const()[name = string("out_1_groups_0"), val = int32(1)]; tensor layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50755328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51541824))))[name = string("layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized")]; tensor out_1_cast_fp16 = conv(dilations = out_1_dilations_0, groups = out_1_groups_0, pad = out_1_pad_0, pad_type = out_1_pad_type_0, strides = out_1_strides_0, weight = layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized, x = input_9_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_661_axes_0 = const()[name = string("op_661_axes_0"), val = tensor([2])]; tensor var_661_cast_fp16 = squeeze(axes = var_661_axes_0, x = out_1_cast_fp16)[name = string("op_661_cast_fp16")]; tensor var_665 = const()[name = string("op_665"), val = tensor([0, 2, 1])]; tensor op_out_1_cast_fp16 = transpose(perm = var_665, x = var_661_cast_fp16)[name = string("transpose_19")]; tensor x_7_cast_fp16 = add(x = hidden_in, y = op_out_1_cast_fp16)[name = string("x_7_cast_fp16")]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_669_cast_fp16 = mul(x = x_7_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_669_cast_fp16")]; int32 var_671 = const()[name = string("op_671"), val = int32(-1)]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11_cast_fp16 = concat(axis = var_671, interleave = input_11_interleave_0, values = (x_7_cast_fp16, var_669_cast_fp16))[name = string("input_11_cast_fp16")]; tensor normed_7_axes_0 = const()[name = string("normed_7_axes_0"), val = tensor([-1])]; fp16 var_677_to_fp16 = const()[name = string("op_677_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_7_cast_fp16 = layer_norm(axes = normed_7_axes_0, epsilon = var_677_to_fp16, x = input_11_cast_fp16)[name = string("normed_7_cast_fp16")]; tensor var_680_split_sizes_0 = const()[name = string("op_680_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_680_axis_0 = const()[name = string("op_680_axis_0"), val = int32(-1)]; tensor var_680_cast_fp16_0, tensor var_680_cast_fp16_1 = split(axis = var_680_axis_0, split_sizes = var_680_split_sizes_0, x = normed_7_cast_fp16)[name = string("op_680_cast_fp16")]; tensor layers_0_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51545984)))]; tensor normed_9_cast_fp16 = mul(x = var_680_cast_fp16_0, y = layers_0_ffn_norm_weight_promoted_to_fp16)[name = string("normed_9_cast_fp16")]; tensor var_686 = const()[name = string("op_686"), val = tensor([0, 2, 1])]; tensor var_689_axes_0 = const()[name = string("op_689_axes_0"), val = tensor([2])]; tensor var_687_cast_fp16 = transpose(perm = var_686, x = normed_9_cast_fp16)[name = string("transpose_18")]; tensor var_689_cast_fp16 = expand_dims(axes = var_689_axes_0, x = var_687_cast_fp16)[name = string("op_689_cast_fp16")]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_0_feed_forward_w1_weight_palettized, x = var_689_cast_fp16)[name = string("input_15")]; string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = layers_0_feed_forward_w3_weight_palettized, x = var_689_cast_fp16)[name = string("b_1")]; tensor var_717 = silu(x = input_15)[name = string("op_717")]; tensor input_17 = mul(x = var_717, y = b_1)[name = string("input_17")]; string mlp_1_pad_type_0 = const()[name = string("mlp_1_pad_type_0"), val = string("valid")]; tensor mlp_1_strides_0 = const()[name = string("mlp_1_strides_0"), val = tensor([1, 1])]; tensor mlp_1_pad_0 = const()[name = string("mlp_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_1_dilations_0 = const()[name = string("mlp_1_dilations_0"), val = tensor([1, 1])]; int32 mlp_1_groups_0 = const()[name = string("mlp_1_groups_0"), val = int32(1)]; tensor mlp_1 = conv(dilations = mlp_1_dilations_0, groups = mlp_1_groups_0, pad = mlp_1_pad_0, pad_type = mlp_1_pad_type_0, strides = mlp_1_strides_0, weight = layers_0_feed_forward_w2_weight_palettized, x = input_17)[name = string("mlp_1")]; tensor var_731_axes_0 = const()[name = string("op_731_axes_0"), val = tensor([2])]; tensor var_731 = squeeze(axes = var_731_axes_0, x = mlp_1)[name = string("op_731")]; tensor var_735 = const()[name = string("op_735"), val = tensor([0, 2, 1])]; tensor mlp_3 = transpose(perm = var_735, x = var_731)[name = string("transpose_17")]; tensor x_9_cast_fp16 = add(x = x_7_cast_fp16, y = mlp_3)[name = string("x_9_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_739_cast_fp16 = mul(x = x_9_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_739_cast_fp16")]; int32 var_741 = const()[name = string("op_741"), val = int32(-1)]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_741, interleave = input_19_interleave_0, values = (x_9_cast_fp16, var_739_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_11_axes_0 = const()[name = string("normed_11_axes_0"), val = tensor([-1])]; fp16 var_747_to_fp16 = const()[name = string("op_747_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_11_cast_fp16 = layer_norm(axes = normed_11_axes_0, epsilon = var_747_to_fp16, x = input_19_cast_fp16)[name = string("normed_11_cast_fp16")]; tensor var_750_split_sizes_0 = const()[name = string("op_750_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_750_axis_0 = const()[name = string("op_750_axis_0"), val = int32(-1)]; tensor var_750_cast_fp16_0, tensor var_750_cast_fp16_1 = split(axis = var_750_axis_0, split_sizes = var_750_split_sizes_0, x = normed_11_cast_fp16)[name = string("op_750_cast_fp16")]; tensor layers_1_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_operator_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51548096)))]; tensor hidden_states_3_cast_fp16 = mul(x = var_750_cast_fp16_0, y = layers_1_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor var_756 = const()[name = string("op_756"), val = tensor([0, 2, 1])]; tensor var_759_axes_0 = const()[name = string("op_759_axes_0"), val = tensor([2])]; tensor var_757_cast_fp16 = transpose(perm = var_756, x = hidden_states_3_cast_fp16)[name = string("transpose_16")]; tensor var_759_cast_fp16 = expand_dims(axes = var_759_axes_0, x = var_757_cast_fp16)[name = string("op_759_cast_fp16")]; string BCx_1_pad_type_0 = const()[name = string("BCx_1_pad_type_0"), val = string("valid")]; tensor BCx_1_strides_0 = const()[name = string("BCx_1_strides_0"), val = tensor([1, 1])]; tensor BCx_1_pad_0 = const()[name = string("BCx_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor BCx_1_dilations_0 = const()[name = string("BCx_1_dilations_0"), val = tensor([1, 1])]; int32 BCx_1_groups_0 = const()[name = string("BCx_1_groups_0"), val = int32(1)]; tensor BCx_1 = conv(dilations = BCx_1_dilations_0, groups = BCx_1_groups_0, pad = BCx_1_pad_0, pad_type = BCx_1_pad_type_0, strides = BCx_1_strides_0, weight = layers_1_conv_in_proj_weight_palettized, x = var_759_cast_fp16)[name = string("BCx_1")]; tensor var_776_split_sizes_0 = const()[name = string("op_776_split_sizes_0"), val = tensor([1024, 1024, 1024])]; int32 var_776_axis_0 = const()[name = string("op_776_axis_0"), val = int32(1)]; tensor var_776_0, tensor var_776_1, tensor var_776_2 = split(axis = var_776_axis_0, split_sizes = var_776_split_sizes_0, x = BCx_1)[name = string("op_776")]; tensor Bx_1 = mul(x = var_776_0, y = var_776_2)[name = string("Bx_1")]; tensor var_782_begin_0 = const()[name = string("op_782_begin_0"), val = tensor([0, 0, 0])]; tensor var_782_end_0 = const()[name = string("op_782_end_0"), val = tensor([1, 1024, 3])]; tensor var_782_end_mask_0 = const()[name = string("op_782_end_mask_0"), val = tensor([false, true, true])]; tensor var_782_squeeze_mask_0 = const()[name = string("op_782_squeeze_mask_0"), val = tensor([true, false, false])]; tensor var_782_cast_fp16 = slice_by_index(begin = var_782_begin_0, end = var_782_end_0, end_mask = var_782_end_mask_0, squeeze_mask = var_782_squeeze_mask_0, x = conv_state_in)[name = string("op_782_cast_fp16")]; tensor var_784_axes_0 = const()[name = string("op_784_axes_0"), val = tensor([0])]; tensor var_784_cast_fp16 = expand_dims(axes = var_784_axes_0, x = var_782_cast_fp16)[name = string("op_784_cast_fp16")]; tensor slot_1_axes_0 = const()[name = string("slot_1_axes_0"), val = tensor([2])]; tensor slot_1_cast_fp16 = expand_dims(axes = slot_1_axes_0, x = var_784_cast_fp16)[name = string("slot_1_cast_fp16")]; tensor live_tail_1_begin_0 = const()[name = string("live_tail_1_begin_0"), val = tensor([0, 0, 0, 1])]; tensor live_tail_1_end_0 = const()[name = string("live_tail_1_end_0"), val = tensor([1, 1024, 1, 1])]; tensor live_tail_1_end_mask_0 = const()[name = string("live_tail_1_end_mask_0"), val = tensor([true, true, true, true])]; tensor live_tail_1_cast_fp16 = slice_by_index(begin = live_tail_1_begin_0, end = live_tail_1_end_0, end_mask = live_tail_1_end_mask_0, x = slot_1_cast_fp16)[name = string("live_tail_1_cast_fp16")]; int32 var_793 = const()[name = string("op_793"), val = int32(-1)]; bool new_state_1_interleave_0 = const()[name = string("new_state_1_interleave_0"), val = bool(false)]; tensor new_state_1_cast_fp16 = concat(axis = var_793, interleave = new_state_1_interleave_0, values = (live_tail_1_cast_fp16, Bx_1))[name = string("new_state_1_cast_fp16")]; tensor var_796_axes_0 = const()[name = string("op_796_axes_0"), val = tensor([0])]; tensor var_796_cast_fp16 = squeeze(axes = var_796_axes_0, x = new_state_1_cast_fp16)[name = string("op_796_cast_fp16")]; tensor var_798_axes_0 = const()[name = string("op_798_axes_0"), val = tensor([1])]; tensor var_798_cast_fp16 = squeeze(axes = var_798_axes_0, x = var_796_cast_fp16)[name = string("op_798_cast_fp16")]; string conv_out_1_pad_type_0 = const()[name = string("conv_out_1_pad_type_0"), val = string("valid")]; int32 conv_out_1_groups_0 = const()[name = string("conv_out_1_groups_0"), val = int32(1024)]; tensor conv_out_1_strides_0 = const()[name = string("conv_out_1_strides_0"), val = tensor([1, 1])]; tensor conv_out_1_pad_0 = const()[name = string("conv_out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor conv_out_1_dilations_0 = const()[name = string("conv_out_1_dilations_0"), val = tensor([1, 1])]; tensor layers_1_conv_conv_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51550208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51552576))))[name = string("layers_1_conv_conv_weight_promoted_to_fp16_palettized")]; tensor conv_out_1_cast_fp16 = conv(dilations = conv_out_1_dilations_0, groups = conv_out_1_groups_0, pad = conv_out_1_pad_0, pad_type = conv_out_1_pad_type_0, strides = conv_out_1_strides_0, weight = layers_1_conv_conv_weight_promoted_to_fp16_palettized, x = new_state_1_cast_fp16)[name = string("conv_out_1_cast_fp16")]; tensor input_23_cast_fp16 = mul(x = var_776_1, y = conv_out_1_cast_fp16)[name = string("input_23_cast_fp16")]; string y_1_pad_type_0 = const()[name = string("y_1_pad_type_0"), val = string("valid")]; tensor y_1_strides_0 = const()[name = string("y_1_strides_0"), val = tensor([1, 1])]; tensor y_1_pad_0 = const()[name = string("y_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor y_1_dilations_0 = const()[name = string("y_1_dilations_0"), val = tensor([1, 1])]; int32 y_1_groups_0 = const()[name = string("y_1_groups_0"), val = int32(1)]; tensor layers_1_conv_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51556736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52343232))))[name = string("layers_1_conv_out_proj_weight_promoted_to_fp16_palettized")]; tensor y_1_cast_fp16 = conv(dilations = y_1_dilations_0, groups = y_1_groups_0, pad = y_1_pad_0, pad_type = y_1_pad_type_0, strides = y_1_strides_0, weight = layers_1_conv_out_proj_weight_promoted_to_fp16_palettized, x = input_23_cast_fp16)[name = string("y_1_cast_fp16")]; tensor var_824_axes_0 = const()[name = string("op_824_axes_0"), val = tensor([2])]; tensor var_824_cast_fp16 = squeeze(axes = var_824_axes_0, x = y_1_cast_fp16)[name = string("op_824_cast_fp16")]; tensor var_828 = const()[name = string("op_828"), val = tensor([0, 2, 1])]; tensor op_out_3_cast_fp16 = transpose(perm = var_828, x = var_824_cast_fp16)[name = string("transpose_15")]; tensor x_11_cast_fp16 = add(x = x_9_cast_fp16, y = op_out_3_cast_fp16)[name = string("x_11_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_832_cast_fp16 = mul(x = x_11_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_832_cast_fp16")]; int32 var_834 = const()[name = string("op_834"), val = int32(-1)]; bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; tensor input_25_cast_fp16 = concat(axis = var_834, interleave = input_25_interleave_0, values = (x_11_cast_fp16, var_832_cast_fp16))[name = string("input_25_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_840_to_fp16 = const()[name = string("op_840_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_840_to_fp16, x = input_25_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor var_843_split_sizes_0 = const()[name = string("op_843_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_843_axis_0 = const()[name = string("op_843_axis_0"), val = int32(-1)]; tensor var_843_cast_fp16_0, tensor var_843_cast_fp16_1 = split(axis = var_843_axis_0, split_sizes = var_843_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_843_cast_fp16")]; tensor layers_1_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52347392)))]; tensor normed_15_cast_fp16 = mul(x = var_843_cast_fp16_0, y = layers_1_ffn_norm_weight_promoted_to_fp16)[name = string("normed_15_cast_fp16")]; tensor var_849 = const()[name = string("op_849"), val = tensor([0, 2, 1])]; tensor var_852_axes_0 = const()[name = string("op_852_axes_0"), val = tensor([2])]; tensor var_850_cast_fp16 = transpose(perm = var_849, x = normed_15_cast_fp16)[name = string("transpose_14")]; tensor var_852_cast_fp16 = expand_dims(axes = var_852_axes_0, x = var_850_cast_fp16)[name = string("op_852_cast_fp16")]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_1_feed_forward_w1_weight_palettized, x = var_852_cast_fp16)[name = string("input_29")]; string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = layers_1_feed_forward_w3_weight_palettized, x = var_852_cast_fp16)[name = string("b_3")]; tensor var_880 = silu(x = input_29)[name = string("op_880")]; tensor input_31 = mul(x = var_880, y = b_3)[name = string("input_31")]; string mlp_5_pad_type_0 = const()[name = string("mlp_5_pad_type_0"), val = string("valid")]; tensor mlp_5_strides_0 = const()[name = string("mlp_5_strides_0"), val = tensor([1, 1])]; tensor mlp_5_pad_0 = const()[name = string("mlp_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_5_dilations_0 = const()[name = string("mlp_5_dilations_0"), val = tensor([1, 1])]; int32 mlp_5_groups_0 = const()[name = string("mlp_5_groups_0"), val = int32(1)]; tensor mlp_5 = conv(dilations = mlp_5_dilations_0, groups = mlp_5_groups_0, pad = mlp_5_pad_0, pad_type = mlp_5_pad_type_0, strides = mlp_5_strides_0, weight = layers_1_feed_forward_w2_weight_palettized, x = input_31)[name = string("mlp_5")]; tensor var_894_axes_0 = const()[name = string("op_894_axes_0"), val = tensor([2])]; tensor var_894 = squeeze(axes = var_894_axes_0, x = mlp_5)[name = string("op_894")]; tensor var_898 = const()[name = string("op_898"), val = tensor([0, 2, 1])]; tensor mlp_7 = transpose(perm = var_898, x = var_894)[name = string("transpose_13")]; tensor x_13_cast_fp16 = add(x = x_11_cast_fp16, y = mlp_7)[name = string("x_13_cast_fp16")]; fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_902_cast_fp16 = mul(x = x_13_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_902_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(-1)]; bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; tensor input_33_cast_fp16 = concat(axis = var_904, interleave = input_33_interleave_0, values = (x_13_cast_fp16, var_902_cast_fp16))[name = string("input_33_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_910_to_fp16 = const()[name = string("op_910_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_910_to_fp16, x = input_33_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_913_split_sizes_0 = const()[name = string("op_913_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_913_axis_0 = const()[name = string("op_913_axis_0"), val = int32(-1)]; tensor var_913_cast_fp16_0, tensor var_913_cast_fp16_1 = split(axis = var_913_axis_0, split_sizes = var_913_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_913_cast_fp16")]; tensor layers_2_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_operator_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52349504)))]; tensor hidden_states_5_cast_fp16 = mul(x = var_913_cast_fp16_0, y = layers_2_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_5_cast_fp16")]; tensor var_919 = const()[name = string("op_919"), val = tensor([0, 2, 1])]; tensor var_922_axes_0 = const()[name = string("op_922_axes_0"), val = tensor([2])]; tensor var_920_cast_fp16 = transpose(perm = var_919, x = hidden_states_5_cast_fp16)[name = string("transpose_12")]; tensor var_922_cast_fp16 = expand_dims(axes = var_922_axes_0, x = var_920_cast_fp16)[name = string("op_922_cast_fp16")]; string var_938_pad_type_0 = const()[name = string("op_938_pad_type_0"), val = string("valid")]; tensor var_938_strides_0 = const()[name = string("op_938_strides_0"), val = tensor([1, 1])]; tensor var_938_pad_0 = const()[name = string("op_938_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_938_dilations_0 = const()[name = string("op_938_dilations_0"), val = tensor([1, 1])]; int32 var_938_groups_0 = const()[name = string("op_938_groups_0"), val = int32(1)]; tensor var_938 = conv(dilations = var_938_dilations_0, groups = var_938_groups_0, pad = var_938_pad_0, pad_type = var_938_pad_type_0, strides = var_938_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_922_cast_fp16)[name = string("op_938")]; tensor var_943 = const()[name = string("op_943"), val = tensor([1, 16, 64, 1])]; tensor var_944 = reshape(shape = var_943, x = var_938)[name = string("op_944")]; tensor var_949 = const()[name = string("op_949"), val = tensor([0, 1, 3, 2])]; string var_966_pad_type_0 = const()[name = string("op_966_pad_type_0"), val = string("valid")]; tensor var_966_strides_0 = const()[name = string("op_966_strides_0"), val = tensor([1, 1])]; tensor var_966_pad_0 = const()[name = string("op_966_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_966_dilations_0 = const()[name = string("op_966_dilations_0"), val = tensor([1, 1])]; int32 var_966_groups_0 = const()[name = string("op_966_groups_0"), val = int32(1)]; tensor var_966 = conv(dilations = var_966_dilations_0, groups = var_966_groups_0, pad = var_966_pad_0, pad_type = var_966_pad_type_0, strides = var_966_strides_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_922_cast_fp16)[name = string("op_966")]; tensor var_971 = const()[name = string("op_971"), val = tensor([1, 8, 64, 1])]; tensor var_972 = reshape(shape = var_971, x = var_966)[name = string("op_972")]; tensor var_977 = const()[name = string("op_977"), val = tensor([0, 1, 3, 2])]; string var_994_pad_type_0 = const()[name = string("op_994_pad_type_0"), val = string("valid")]; tensor var_994_strides_0 = const()[name = string("op_994_strides_0"), val = tensor([1, 1])]; tensor var_994_pad_0 = const()[name = string("op_994_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_994_dilations_0 = const()[name = string("op_994_dilations_0"), val = tensor([1, 1])]; int32 var_994_groups_0 = const()[name = string("op_994_groups_0"), val = int32(1)]; tensor var_994 = conv(dilations = var_994_dilations_0, groups = var_994_groups_0, pad = var_994_pad_0, pad_type = var_994_pad_type_0, strides = var_994_strides_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_922_cast_fp16)[name = string("op_994")]; fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; tensor var_950 = transpose(perm = var_949, x = var_944)[name = string("transpose_11")]; tensor var_1012 = mul(x = var_950, y = const_15_promoted)[name = string("op_1012")]; int32 var_1014 = const()[name = string("op_1014"), val = int32(-1)]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37 = concat(axis = var_1014, interleave = input_37_interleave_0, values = (var_950, var_1012))[name = string("input_37")]; tensor normed_19_axes_0 = const()[name = string("normed_19_axes_0"), val = tensor([-1])]; fp16 var_1020_to_fp16 = const()[name = string("op_1020_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_19_cast_fp16 = layer_norm(axes = normed_19_axes_0, epsilon = var_1020_to_fp16, x = input_37)[name = string("normed_19_cast_fp16")]; tensor var_1023_split_sizes_0 = const()[name = string("op_1023_split_sizes_0"), val = tensor([64, 64])]; int32 var_1023_axis_0 = const()[name = string("op_1023_axis_0"), val = int32(-1)]; tensor var_1023_0, tensor var_1023_1 = split(axis = var_1023_axis_0, split_sizes = var_1023_split_sizes_0, x = normed_19_cast_fp16)[name = string("op_1023")]; tensor q_5 = mul(x = var_1023_0, y = layers_2_self_attn_q_layernorm_weight)[name = string("q_5")]; fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; tensor var_978 = transpose(perm = var_977, x = var_972)[name = string("transpose_10")]; tensor var_1026 = mul(x = var_978, y = const_16_promoted)[name = string("op_1026")]; int32 var_1028 = const()[name = string("op_1028"), val = int32(-1)]; bool input_39_interleave_0 = const()[name = string("input_39_interleave_0"), val = bool(false)]; tensor input_39 = concat(axis = var_1028, interleave = input_39_interleave_0, values = (var_978, var_1026))[name = string("input_39")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_1034_to_fp16 = const()[name = string("op_1034_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1034_to_fp16, x = input_39)[name = string("normed_21_cast_fp16")]; tensor var_1037_split_sizes_0 = const()[name = string("op_1037_split_sizes_0"), val = tensor([64, 64])]; int32 var_1037_axis_0 = const()[name = string("op_1037_axis_0"), val = int32(-1)]; tensor var_1037_0, tensor var_1037_1 = split(axis = var_1037_axis_0, split_sizes = var_1037_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_1037")]; tensor k_5 = mul(x = var_1037_0, y = layers_2_self_attn_k_layernorm_weight)[name = string("k_5")]; tensor var_1040 = mul(x = q_5, y = cos)[name = string("op_1040")]; tensor var_1041_split_sizes_0 = const()[name = string("op_1041_split_sizes_0"), val = tensor([32, 32])]; int32 var_1041_axis_0 = const()[name = string("op_1041_axis_0"), val = int32(-1)]; tensor var_1041_0, tensor var_1041_1 = split(axis = var_1041_axis_0, split_sizes = var_1041_split_sizes_0, x = q_5)[name = string("op_1041")]; fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; tensor var_1043 = mul(x = var_1041_1, y = const_17_promoted)[name = string("op_1043")]; int32 var_1045 = const()[name = string("op_1045"), val = int32(-1)]; bool var_1046_interleave_0 = const()[name = string("op_1046_interleave_0"), val = bool(false)]; tensor var_1046 = concat(axis = var_1045, interleave = var_1046_interleave_0, values = (var_1043, var_1041_0))[name = string("op_1046")]; tensor var_1047 = mul(x = var_1046, y = sin)[name = string("op_1047")]; tensor q = add(x = var_1040, y = var_1047)[name = string("q")]; tensor var_1050 = mul(x = k_5, y = cos)[name = string("op_1050")]; tensor var_1051_split_sizes_0 = const()[name = string("op_1051_split_sizes_0"), val = tensor([32, 32])]; int32 var_1051_axis_0 = const()[name = string("op_1051_axis_0"), val = int32(-1)]; tensor var_1051_0, tensor var_1051_1 = split(axis = var_1051_axis_0, split_sizes = var_1051_split_sizes_0, x = k_5)[name = string("op_1051")]; fp16 const_18_promoted = const()[name = string("const_18_promoted"), val = fp16(-0x1p+0)]; tensor var_1053 = mul(x = var_1051_1, y = const_18_promoted)[name = string("op_1053")]; int32 var_1055 = const()[name = string("op_1055"), val = int32(-1)]; bool var_1056_interleave_0 = const()[name = string("op_1056_interleave_0"), val = bool(false)]; tensor var_1056 = concat(axis = var_1055, interleave = var_1056_interleave_0, values = (var_1053, var_1051_0))[name = string("op_1056")]; tensor var_1057 = mul(x = var_1056, y = sin)[name = string("op_1057")]; tensor k = add(x = var_1050, y = var_1057)[name = string("k")]; tensor K_cache_begin_0 = const()[name = string("K_cache_begin_0"), val = tensor([1, 0, 0, 0, 0])]; tensor K_cache_end_0 = const()[name = string("K_cache_end_0"), val = tensor([2, 1, 512, 1, 1024])]; tensor K_cache_end_mask_0 = const()[name = string("K_cache_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor K_cache_squeeze_mask_0 = const()[name = string("K_cache_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor K_cache_cast_fp16 = slice_by_index(begin = K_cache_begin_0, end = K_cache_end_0, end_mask = K_cache_end_mask_0, squeeze_mask = K_cache_squeeze_mask_0, x = kv_cache_in)[name = string("K_cache_cast_fp16")]; tensor V_cache_begin_0 = const()[name = string("V_cache_begin_0"), val = tensor([3, 0, 0, 0, 0])]; tensor V_cache_end_0 = const()[name = string("V_cache_end_0"), val = tensor([4, 1, 512, 1, 1024])]; tensor V_cache_end_mask_0 = const()[name = string("V_cache_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor V_cache_squeeze_mask_0 = const()[name = string("V_cache_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor V_cache_cast_fp16 = slice_by_index(begin = V_cache_begin_0, end = V_cache_end_0, end_mask = V_cache_end_mask_0, squeeze_mask = V_cache_squeeze_mask_0, x = kv_cache_in)[name = string("V_cache_cast_fp16")]; tensor var_1070 = const()[name = string("op_1070"), val = tensor([0, 1, 3, 2])]; tensor var_1076 = const()[name = string("op_1076"), val = tensor([1, 1024, 1, 1])]; tensor var_1071 = transpose(perm = var_1070, x = q)[name = string("transpose_9")]; tensor query = reshape(shape = var_1076, x = var_1071)[name = string("query")]; tensor var_1082 = const()[name = string("op_1082"), val = tensor([0, 1, 3, 2])]; tensor var_1088 = const()[name = string("op_1088"), val = tensor([1, 512, 1, 1])]; tensor var_1083 = transpose(perm = var_1082, x = k)[name = string("transpose_8")]; tensor k_slice = reshape(shape = var_1088, x = var_1083)[name = string("k_slice")]; int32 var_1103 = const()[name = string("op_1103"), val = int32(-1)]; bool key_interleave_0 = const()[name = string("key_interleave_0"), val = bool(false)]; tensor key_cast_fp16 = concat(axis = var_1103, interleave = key_interleave_0, values = (K_cache_cast_fp16, k_slice))[name = string("key_cast_fp16")]; int32 var_1106 = const()[name = string("op_1106"), val = int32(-1)]; bool var_1107_interleave_0 = const()[name = string("op_1107_interleave_0"), val = bool(false)]; tensor var_1107_cast_fp16 = concat(axis = var_1106, interleave = var_1107_interleave_0, values = (V_cache_cast_fp16, var_994))[name = string("op_1107_cast_fp16")]; tensor var_1108 = mul(x = query, y = attn_scale)[name = string("op_1108")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52351616)))]; int32 var_1111_axis_0 = const()[name = string("op_1111_axis_0"), val = int32(1)]; tensor var_1111_0, tensor var_1111_1, tensor var_1111_2, tensor var_1111_3, tensor var_1111_4, tensor var_1111_5, tensor var_1111_6, tensor var_1111_7, tensor var_1111_8, tensor var_1111_9, tensor var_1111_10, tensor var_1111_11, tensor var_1111_12, tensor var_1111_13, tensor var_1111_14, tensor var_1111_15 = split(axis = var_1111_axis_0, split_sizes = tile_3, x = var_1108)[name = string("op_1111")]; tensor var_1130_perm_0 = const()[name = string("op_1130_perm_0"), val = tensor([0, 3, 2, 1])]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_1133_axis_0 = const()[name = string("op_1133_axis_0"), val = int32(3)]; tensor var_1130_cast_fp16 = transpose(perm = var_1130_perm_0, x = key_cast_fp16)[name = string("transpose_7")]; tensor var_1133_cast_fp16_0, tensor var_1133_cast_fp16_1, tensor var_1133_cast_fp16_2, tensor var_1133_cast_fp16_3, tensor var_1133_cast_fp16_4, tensor var_1133_cast_fp16_5, tensor var_1133_cast_fp16_6, tensor var_1133_cast_fp16_7 = split(axis = var_1133_axis_0, split_sizes = tile_4, x = var_1130_cast_fp16)[name = string("op_1133_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([64, 64, 64, 64, 64, 64, 64, 64])]; int32 var_1144_axis_0 = const()[name = string("op_1144_axis_0"), val = int32(1)]; tensor var_1144_cast_fp16_0, tensor var_1144_cast_fp16_1, tensor var_1144_cast_fp16_2, tensor var_1144_cast_fp16_3, tensor var_1144_cast_fp16_4, tensor var_1144_cast_fp16_5, tensor var_1144_cast_fp16_6, tensor var_1144_cast_fp16_7 = split(axis = var_1144_axis_0, split_sizes = tile_5, x = var_1107_cast_fp16)[name = string("op_1144_cast_fp16")]; string scores_33_equation_0 = const()[name = string("scores_33_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_33_cast_fp16 = einsum(equation = scores_33_equation_0, values = (var_1133_cast_fp16_0, var_1111_0))[name = string("scores_33_cast_fp16")]; tensor var_1158_cast_fp16 = add(x = scores_33_cast_fp16, y = causal_mask)[name = string("op_1158_cast_fp16")]; int32 var_1159 = const()[name = string("op_1159"), val = int32(1)]; tensor var_1161_cast_fp16 = softmax(axis = var_1159, x = var_1158_cast_fp16)[name = string("op_1161_cast_fp16")]; string var_1165_equation_0 = const()[name = string("op_1165_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1165_cast_fp16 = einsum(equation = var_1165_equation_0, values = (var_1144_cast_fp16_0, var_1161_cast_fp16))[name = string("op_1165_cast_fp16")]; string scores_35_equation_0 = const()[name = string("scores_35_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_35_cast_fp16 = einsum(equation = scores_35_equation_0, values = (var_1133_cast_fp16_0, var_1111_1))[name = string("scores_35_cast_fp16")]; tensor var_1171_cast_fp16 = add(x = scores_35_cast_fp16, y = causal_mask)[name = string("op_1171_cast_fp16")]; int32 var_1172 = const()[name = string("op_1172"), val = int32(1)]; tensor var_1174_cast_fp16 = softmax(axis = var_1172, x = var_1171_cast_fp16)[name = string("op_1174_cast_fp16")]; string var_1178_equation_0 = const()[name = string("op_1178_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1178_cast_fp16 = einsum(equation = var_1178_equation_0, values = (var_1144_cast_fp16_0, var_1174_cast_fp16))[name = string("op_1178_cast_fp16")]; string scores_37_equation_0 = const()[name = string("scores_37_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_37_cast_fp16 = einsum(equation = scores_37_equation_0, values = (var_1133_cast_fp16_1, var_1111_2))[name = string("scores_37_cast_fp16")]; tensor var_1184_cast_fp16 = add(x = scores_37_cast_fp16, y = causal_mask)[name = string("op_1184_cast_fp16")]; int32 var_1185 = const()[name = string("op_1185"), val = int32(1)]; tensor var_1187_cast_fp16 = softmax(axis = var_1185, x = var_1184_cast_fp16)[name = string("op_1187_cast_fp16")]; string var_1191_equation_0 = const()[name = string("op_1191_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1191_cast_fp16 = einsum(equation = var_1191_equation_0, values = (var_1144_cast_fp16_1, var_1187_cast_fp16))[name = string("op_1191_cast_fp16")]; string scores_39_equation_0 = const()[name = string("scores_39_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_39_cast_fp16 = einsum(equation = scores_39_equation_0, values = (var_1133_cast_fp16_1, var_1111_3))[name = string("scores_39_cast_fp16")]; tensor var_1197_cast_fp16 = add(x = scores_39_cast_fp16, y = causal_mask)[name = string("op_1197_cast_fp16")]; int32 var_1198 = const()[name = string("op_1198"), val = int32(1)]; tensor var_1200_cast_fp16 = softmax(axis = var_1198, x = var_1197_cast_fp16)[name = string("op_1200_cast_fp16")]; string var_1204_equation_0 = const()[name = string("op_1204_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1204_cast_fp16 = einsum(equation = var_1204_equation_0, values = (var_1144_cast_fp16_1, var_1200_cast_fp16))[name = string("op_1204_cast_fp16")]; string scores_41_equation_0 = const()[name = string("scores_41_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_41_cast_fp16 = einsum(equation = scores_41_equation_0, values = (var_1133_cast_fp16_2, var_1111_4))[name = string("scores_41_cast_fp16")]; tensor var_1210_cast_fp16 = add(x = scores_41_cast_fp16, y = causal_mask)[name = string("op_1210_cast_fp16")]; int32 var_1211 = const()[name = string("op_1211"), val = int32(1)]; tensor var_1213_cast_fp16 = softmax(axis = var_1211, x = var_1210_cast_fp16)[name = string("op_1213_cast_fp16")]; string var_1217_equation_0 = const()[name = string("op_1217_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1217_cast_fp16 = einsum(equation = var_1217_equation_0, values = (var_1144_cast_fp16_2, var_1213_cast_fp16))[name = string("op_1217_cast_fp16")]; string scores_43_equation_0 = const()[name = string("scores_43_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_43_cast_fp16 = einsum(equation = scores_43_equation_0, values = (var_1133_cast_fp16_2, var_1111_5))[name = string("scores_43_cast_fp16")]; tensor var_1223_cast_fp16 = add(x = scores_43_cast_fp16, y = causal_mask)[name = string("op_1223_cast_fp16")]; int32 var_1224 = const()[name = string("op_1224"), val = int32(1)]; tensor var_1226_cast_fp16 = softmax(axis = var_1224, x = var_1223_cast_fp16)[name = string("op_1226_cast_fp16")]; string var_1230_equation_0 = const()[name = string("op_1230_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1230_cast_fp16 = einsum(equation = var_1230_equation_0, values = (var_1144_cast_fp16_2, var_1226_cast_fp16))[name = string("op_1230_cast_fp16")]; string scores_45_equation_0 = const()[name = string("scores_45_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_45_cast_fp16 = einsum(equation = scores_45_equation_0, values = (var_1133_cast_fp16_3, var_1111_6))[name = string("scores_45_cast_fp16")]; tensor var_1236_cast_fp16 = add(x = scores_45_cast_fp16, y = causal_mask)[name = string("op_1236_cast_fp16")]; int32 var_1237 = const()[name = string("op_1237"), val = int32(1)]; tensor var_1239_cast_fp16 = softmax(axis = var_1237, x = var_1236_cast_fp16)[name = string("op_1239_cast_fp16")]; string var_1243_equation_0 = const()[name = string("op_1243_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1243_cast_fp16 = einsum(equation = var_1243_equation_0, values = (var_1144_cast_fp16_3, var_1239_cast_fp16))[name = string("op_1243_cast_fp16")]; string scores_47_equation_0 = const()[name = string("scores_47_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_47_cast_fp16 = einsum(equation = scores_47_equation_0, values = (var_1133_cast_fp16_3, var_1111_7))[name = string("scores_47_cast_fp16")]; tensor var_1249_cast_fp16 = add(x = scores_47_cast_fp16, y = causal_mask)[name = string("op_1249_cast_fp16")]; int32 var_1250 = const()[name = string("op_1250"), val = int32(1)]; tensor var_1252_cast_fp16 = softmax(axis = var_1250, x = var_1249_cast_fp16)[name = string("op_1252_cast_fp16")]; string var_1256_equation_0 = const()[name = string("op_1256_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1256_cast_fp16 = einsum(equation = var_1256_equation_0, values = (var_1144_cast_fp16_3, var_1252_cast_fp16))[name = string("op_1256_cast_fp16")]; string scores_49_equation_0 = const()[name = string("scores_49_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_49_cast_fp16 = einsum(equation = scores_49_equation_0, values = (var_1133_cast_fp16_4, var_1111_8))[name = string("scores_49_cast_fp16")]; tensor var_1262_cast_fp16 = add(x = scores_49_cast_fp16, y = causal_mask)[name = string("op_1262_cast_fp16")]; int32 var_1263 = const()[name = string("op_1263"), val = int32(1)]; tensor var_1265_cast_fp16 = softmax(axis = var_1263, x = var_1262_cast_fp16)[name = string("op_1265_cast_fp16")]; string var_1269_equation_0 = const()[name = string("op_1269_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1269_cast_fp16 = einsum(equation = var_1269_equation_0, values = (var_1144_cast_fp16_4, var_1265_cast_fp16))[name = string("op_1269_cast_fp16")]; string scores_51_equation_0 = const()[name = string("scores_51_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_51_cast_fp16 = einsum(equation = scores_51_equation_0, values = (var_1133_cast_fp16_4, var_1111_9))[name = string("scores_51_cast_fp16")]; tensor var_1275_cast_fp16 = add(x = scores_51_cast_fp16, y = causal_mask)[name = string("op_1275_cast_fp16")]; int32 var_1276 = const()[name = string("op_1276"), val = int32(1)]; tensor var_1278_cast_fp16 = softmax(axis = var_1276, x = var_1275_cast_fp16)[name = string("op_1278_cast_fp16")]; string var_1282_equation_0 = const()[name = string("op_1282_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1282_cast_fp16 = einsum(equation = var_1282_equation_0, values = (var_1144_cast_fp16_4, var_1278_cast_fp16))[name = string("op_1282_cast_fp16")]; string scores_53_equation_0 = const()[name = string("scores_53_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_53_cast_fp16 = einsum(equation = scores_53_equation_0, values = (var_1133_cast_fp16_5, var_1111_10))[name = string("scores_53_cast_fp16")]; tensor var_1288_cast_fp16 = add(x = scores_53_cast_fp16, y = causal_mask)[name = string("op_1288_cast_fp16")]; int32 var_1289 = const()[name = string("op_1289"), val = int32(1)]; tensor var_1291_cast_fp16 = softmax(axis = var_1289, x = var_1288_cast_fp16)[name = string("op_1291_cast_fp16")]; string var_1295_equation_0 = const()[name = string("op_1295_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1295_cast_fp16 = einsum(equation = var_1295_equation_0, values = (var_1144_cast_fp16_5, var_1291_cast_fp16))[name = string("op_1295_cast_fp16")]; string scores_55_equation_0 = const()[name = string("scores_55_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_55_cast_fp16 = einsum(equation = scores_55_equation_0, values = (var_1133_cast_fp16_5, var_1111_11))[name = string("scores_55_cast_fp16")]; tensor var_1301_cast_fp16 = add(x = scores_55_cast_fp16, y = causal_mask)[name = string("op_1301_cast_fp16")]; int32 var_1302 = const()[name = string("op_1302"), val = int32(1)]; tensor var_1304_cast_fp16 = softmax(axis = var_1302, x = var_1301_cast_fp16)[name = string("op_1304_cast_fp16")]; string var_1308_equation_0 = const()[name = string("op_1308_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1308_cast_fp16 = einsum(equation = var_1308_equation_0, values = (var_1144_cast_fp16_5, var_1304_cast_fp16))[name = string("op_1308_cast_fp16")]; string scores_57_equation_0 = const()[name = string("scores_57_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_57_cast_fp16 = einsum(equation = scores_57_equation_0, values = (var_1133_cast_fp16_6, var_1111_12))[name = string("scores_57_cast_fp16")]; tensor var_1314_cast_fp16 = add(x = scores_57_cast_fp16, y = causal_mask)[name = string("op_1314_cast_fp16")]; int32 var_1315 = const()[name = string("op_1315"), val = int32(1)]; tensor var_1317_cast_fp16 = softmax(axis = var_1315, x = var_1314_cast_fp16)[name = string("op_1317_cast_fp16")]; string var_1321_equation_0 = const()[name = string("op_1321_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1321_cast_fp16 = einsum(equation = var_1321_equation_0, values = (var_1144_cast_fp16_6, var_1317_cast_fp16))[name = string("op_1321_cast_fp16")]; string scores_59_equation_0 = const()[name = string("scores_59_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_59_cast_fp16 = einsum(equation = scores_59_equation_0, values = (var_1133_cast_fp16_6, var_1111_13))[name = string("scores_59_cast_fp16")]; tensor var_1327_cast_fp16 = add(x = scores_59_cast_fp16, y = causal_mask)[name = string("op_1327_cast_fp16")]; int32 var_1328 = const()[name = string("op_1328"), val = int32(1)]; tensor var_1330_cast_fp16 = softmax(axis = var_1328, x = var_1327_cast_fp16)[name = string("op_1330_cast_fp16")]; string var_1334_equation_0 = const()[name = string("op_1334_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1334_cast_fp16 = einsum(equation = var_1334_equation_0, values = (var_1144_cast_fp16_6, var_1330_cast_fp16))[name = string("op_1334_cast_fp16")]; string scores_61_equation_0 = const()[name = string("scores_61_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_61_cast_fp16 = einsum(equation = scores_61_equation_0, values = (var_1133_cast_fp16_7, var_1111_14))[name = string("scores_61_cast_fp16")]; tensor var_1340_cast_fp16 = add(x = scores_61_cast_fp16, y = causal_mask)[name = string("op_1340_cast_fp16")]; int32 var_1341 = const()[name = string("op_1341"), val = int32(1)]; tensor var_1343_cast_fp16 = softmax(axis = var_1341, x = var_1340_cast_fp16)[name = string("op_1343_cast_fp16")]; string var_1347_equation_0 = const()[name = string("op_1347_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1347_cast_fp16 = einsum(equation = var_1347_equation_0, values = (var_1144_cast_fp16_7, var_1343_cast_fp16))[name = string("op_1347_cast_fp16")]; string scores_equation_0 = const()[name = string("scores_equation_0"), val = string("bkhc,bchq->bkhq")]; tensor scores_cast_fp16 = einsum(equation = scores_equation_0, values = (var_1133_cast_fp16_7, var_1111_15))[name = string("scores_cast_fp16")]; tensor var_1353_cast_fp16 = add(x = scores_cast_fp16, y = causal_mask)[name = string("op_1353_cast_fp16")]; int32 var_1354 = const()[name = string("op_1354"), val = int32(1)]; tensor var_1356_cast_fp16 = softmax(axis = var_1354, x = var_1353_cast_fp16)[name = string("op_1356_cast_fp16")]; string var_1360_equation_0 = const()[name = string("op_1360_equation_0"), val = string("bchk,bkhq->bchq")]; tensor var_1360_cast_fp16 = einsum(equation = var_1360_equation_0, values = (var_1144_cast_fp16_7, var_1356_cast_fp16))[name = string("op_1360_cast_fp16")]; int32 var_1362 = const()[name = string("op_1362"), val = int32(1)]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41_cast_fp16 = concat(axis = var_1362, interleave = input_41_interleave_0, values = (var_1165_cast_fp16, var_1178_cast_fp16, var_1191_cast_fp16, var_1204_cast_fp16, var_1217_cast_fp16, var_1230_cast_fp16, var_1243_cast_fp16, var_1256_cast_fp16, var_1269_cast_fp16, var_1282_cast_fp16, var_1295_cast_fp16, var_1308_cast_fp16, var_1321_cast_fp16, var_1334_cast_fp16, var_1347_cast_fp16, var_1360_cast_fp16))[name = string("input_41_cast_fp16")]; string out_pad_type_0 = const()[name = string("out_pad_type_0"), val = string("valid")]; tensor out_strides_0 = const()[name = string("out_strides_0"), val = tensor([1, 1])]; tensor out_pad_0 = const()[name = string("out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor out_dilations_0 = const()[name = string("out_dilations_0"), val = tensor([1, 1])]; int32 out_groups_0 = const()[name = string("out_groups_0"), val = int32(1)]; tensor layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52351744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53138240))))[name = string("layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized")]; tensor out_cast_fp16 = conv(dilations = out_dilations_0, groups = out_groups_0, pad = out_pad_0, pad_type = out_pad_type_0, strides = out_strides_0, weight = layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized, x = input_41_cast_fp16)[name = string("out_cast_fp16")]; tensor var_1376_axes_0 = const()[name = string("op_1376_axes_0"), val = tensor([2])]; tensor var_1376_cast_fp16 = squeeze(axes = var_1376_axes_0, x = out_cast_fp16)[name = string("op_1376_cast_fp16")]; tensor var_1380 = const()[name = string("op_1380"), val = tensor([0, 2, 1])]; tensor op_out_5_cast_fp16 = transpose(perm = var_1380, x = var_1376_cast_fp16)[name = string("transpose_6")]; tensor x_19_cast_fp16 = add(x = x_13_cast_fp16, y = op_out_5_cast_fp16)[name = string("x_19_cast_fp16")]; fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1384_cast_fp16 = mul(x = x_19_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_1384_cast_fp16")]; int32 var_1386 = const()[name = string("op_1386"), val = int32(-1)]; bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; tensor input_43_cast_fp16 = concat(axis = var_1386, interleave = input_43_interleave_0, values = (x_19_cast_fp16, var_1384_cast_fp16))[name = string("input_43_cast_fp16")]; tensor normed_23_axes_0 = const()[name = string("normed_23_axes_0"), val = tensor([-1])]; fp16 var_1392_to_fp16 = const()[name = string("op_1392_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_23_cast_fp16 = layer_norm(axes = normed_23_axes_0, epsilon = var_1392_to_fp16, x = input_43_cast_fp16)[name = string("normed_23_cast_fp16")]; tensor var_1395_split_sizes_0 = const()[name = string("op_1395_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1395_axis_0 = const()[name = string("op_1395_axis_0"), val = int32(-1)]; tensor var_1395_cast_fp16_0, tensor var_1395_cast_fp16_1 = split(axis = var_1395_axis_0, split_sizes = var_1395_split_sizes_0, x = normed_23_cast_fp16)[name = string("op_1395_cast_fp16")]; tensor layers_2_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53142400)))]; tensor normed_25_cast_fp16 = mul(x = var_1395_cast_fp16_0, y = layers_2_ffn_norm_weight_promoted_to_fp16)[name = string("normed_25_cast_fp16")]; tensor var_1401 = const()[name = string("op_1401"), val = tensor([0, 2, 1])]; tensor var_1404_axes_0 = const()[name = string("op_1404_axes_0"), val = tensor([2])]; tensor var_1402_cast_fp16 = transpose(perm = var_1401, x = normed_25_cast_fp16)[name = string("transpose_5")]; tensor var_1404_cast_fp16 = expand_dims(axes = var_1404_axes_0, x = var_1402_cast_fp16)[name = string("op_1404_cast_fp16")]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_2_feed_forward_w1_weight_palettized, x = var_1404_cast_fp16)[name = string("input_47")]; string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = layers_2_feed_forward_w3_weight_palettized, x = var_1404_cast_fp16)[name = string("b_5")]; tensor var_1432 = silu(x = input_47)[name = string("op_1432")]; tensor input_49 = mul(x = var_1432, y = b_5)[name = string("input_49")]; string mlp_9_pad_type_0 = const()[name = string("mlp_9_pad_type_0"), val = string("valid")]; tensor mlp_9_strides_0 = const()[name = string("mlp_9_strides_0"), val = tensor([1, 1])]; tensor mlp_9_pad_0 = const()[name = string("mlp_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_9_dilations_0 = const()[name = string("mlp_9_dilations_0"), val = tensor([1, 1])]; int32 mlp_9_groups_0 = const()[name = string("mlp_9_groups_0"), val = int32(1)]; tensor mlp_9 = conv(dilations = mlp_9_dilations_0, groups = mlp_9_groups_0, pad = mlp_9_pad_0, pad_type = mlp_9_pad_type_0, strides = mlp_9_strides_0, weight = layers_2_feed_forward_w2_weight_palettized, x = input_49)[name = string("mlp_9")]; tensor var_1446_axes_0 = const()[name = string("op_1446_axes_0"), val = tensor([2])]; tensor var_1446 = squeeze(axes = var_1446_axes_0, x = mlp_9)[name = string("op_1446")]; tensor var_1450 = const()[name = string("op_1450"), val = tensor([0, 2, 1])]; tensor mlp_11 = transpose(perm = var_1450, x = var_1446)[name = string("transpose_4")]; tensor x_21_cast_fp16 = add(x = x_19_cast_fp16, y = mlp_11)[name = string("x_21_cast_fp16")]; fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1454_cast_fp16 = mul(x = x_21_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_1454_cast_fp16")]; int32 var_1456 = const()[name = string("op_1456"), val = int32(-1)]; bool input_51_interleave_0 = const()[name = string("input_51_interleave_0"), val = bool(false)]; tensor input_51_cast_fp16 = concat(axis = var_1456, interleave = input_51_interleave_0, values = (x_21_cast_fp16, var_1454_cast_fp16))[name = string("input_51_cast_fp16")]; tensor normed_27_axes_0 = const()[name = string("normed_27_axes_0"), val = tensor([-1])]; fp16 var_1462_to_fp16 = const()[name = string("op_1462_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_27_cast_fp16 = layer_norm(axes = normed_27_axes_0, epsilon = var_1462_to_fp16, x = input_51_cast_fp16)[name = string("normed_27_cast_fp16")]; tensor var_1465_split_sizes_0 = const()[name = string("op_1465_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1465_axis_0 = const()[name = string("op_1465_axis_0"), val = int32(-1)]; tensor var_1465_cast_fp16_0, tensor var_1465_cast_fp16_1 = split(axis = var_1465_axis_0, split_sizes = var_1465_split_sizes_0, x = normed_27_cast_fp16)[name = string("op_1465_cast_fp16")]; tensor layers_3_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_operator_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53144512)))]; tensor hidden_states_cast_fp16 = mul(x = var_1465_cast_fp16_0, y = layers_3_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_cast_fp16")]; tensor var_1471 = const()[name = string("op_1471"), val = tensor([0, 2, 1])]; tensor var_1474_axes_0 = const()[name = string("op_1474_axes_0"), val = tensor([2])]; tensor var_1472_cast_fp16 = transpose(perm = var_1471, x = hidden_states_cast_fp16)[name = string("transpose_3")]; tensor var_1474_cast_fp16 = expand_dims(axes = var_1474_axes_0, x = var_1472_cast_fp16)[name = string("op_1474_cast_fp16")]; string BCx_pad_type_0 = const()[name = string("BCx_pad_type_0"), val = string("valid")]; tensor BCx_strides_0 = const()[name = string("BCx_strides_0"), val = tensor([1, 1])]; tensor BCx_pad_0 = const()[name = string("BCx_pad_0"), val = tensor([0, 0, 0, 0])]; tensor BCx_dilations_0 = const()[name = string("BCx_dilations_0"), val = tensor([1, 1])]; int32 BCx_groups_0 = const()[name = string("BCx_groups_0"), val = int32(1)]; tensor BCx = conv(dilations = BCx_dilations_0, groups = BCx_groups_0, pad = BCx_pad_0, pad_type = BCx_pad_type_0, strides = BCx_strides_0, weight = layers_3_conv_in_proj_weight_palettized, x = var_1474_cast_fp16)[name = string("BCx")]; tensor var_1491_split_sizes_0 = const()[name = string("op_1491_split_sizes_0"), val = tensor([1024, 1024, 1024])]; int32 var_1491_axis_0 = const()[name = string("op_1491_axis_0"), val = int32(1)]; tensor var_1491_0, tensor var_1491_1, tensor var_1491_2 = split(axis = var_1491_axis_0, split_sizes = var_1491_split_sizes_0, x = BCx)[name = string("op_1491")]; tensor Bx = mul(x = var_1491_0, y = var_1491_2)[name = string("Bx")]; tensor var_1497_begin_0 = const()[name = string("op_1497_begin_0"), val = tensor([1, 0, 0])]; tensor var_1497_end_0 = const()[name = string("op_1497_end_0"), val = tensor([2, 1024, 3])]; tensor var_1497_end_mask_0 = const()[name = string("op_1497_end_mask_0"), val = tensor([false, true, true])]; tensor var_1497_squeeze_mask_0 = const()[name = string("op_1497_squeeze_mask_0"), val = tensor([true, false, false])]; tensor var_1497_cast_fp16 = slice_by_index(begin = var_1497_begin_0, end = var_1497_end_0, end_mask = var_1497_end_mask_0, squeeze_mask = var_1497_squeeze_mask_0, x = conv_state_in)[name = string("op_1497_cast_fp16")]; tensor var_1499_axes_0 = const()[name = string("op_1499_axes_0"), val = tensor([0])]; tensor var_1499_cast_fp16 = expand_dims(axes = var_1499_axes_0, x = var_1497_cast_fp16)[name = string("op_1499_cast_fp16")]; tensor slot_axes_0 = const()[name = string("slot_axes_0"), val = tensor([2])]; tensor slot_cast_fp16 = expand_dims(axes = slot_axes_0, x = var_1499_cast_fp16)[name = string("slot_cast_fp16")]; tensor live_tail_begin_0 = const()[name = string("live_tail_begin_0"), val = tensor([0, 0, 0, 1])]; tensor live_tail_end_0 = const()[name = string("live_tail_end_0"), val = tensor([1, 1024, 1, 1])]; tensor live_tail_end_mask_0 = const()[name = string("live_tail_end_mask_0"), val = tensor([true, true, true, true])]; tensor live_tail_cast_fp16 = slice_by_index(begin = live_tail_begin_0, end = live_tail_end_0, end_mask = live_tail_end_mask_0, x = slot_cast_fp16)[name = string("live_tail_cast_fp16")]; int32 var_1508 = const()[name = string("op_1508"), val = int32(-1)]; bool new_state_interleave_0 = const()[name = string("new_state_interleave_0"), val = bool(false)]; tensor new_state_cast_fp16 = concat(axis = var_1508, interleave = new_state_interleave_0, values = (live_tail_cast_fp16, Bx))[name = string("new_state_cast_fp16")]; tensor var_1511_axes_0 = const()[name = string("op_1511_axes_0"), val = tensor([0])]; tensor var_1511_cast_fp16 = squeeze(axes = var_1511_axes_0, x = new_state_cast_fp16)[name = string("op_1511_cast_fp16")]; tensor new_slot_axes_0 = const()[name = string("new_slot_axes_0"), val = tensor([1])]; tensor new_slot_cast_fp16 = squeeze(axes = new_slot_axes_0, x = var_1511_cast_fp16)[name = string("new_slot_cast_fp16")]; string conv_out_pad_type_0 = const()[name = string("conv_out_pad_type_0"), val = string("valid")]; int32 conv_out_groups_0 = const()[name = string("conv_out_groups_0"), val = int32(1024)]; tensor conv_out_strides_0 = const()[name = string("conv_out_strides_0"), val = tensor([1, 1])]; tensor conv_out_pad_0 = const()[name = string("conv_out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor conv_out_dilations_0 = const()[name = string("conv_out_dilations_0"), val = tensor([1, 1])]; tensor layers_3_conv_conv_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53146624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53148992))))[name = string("layers_3_conv_conv_weight_promoted_to_fp16_palettized")]; tensor conv_out_cast_fp16 = conv(dilations = conv_out_dilations_0, groups = conv_out_groups_0, pad = conv_out_pad_0, pad_type = conv_out_pad_type_0, strides = conv_out_strides_0, weight = layers_3_conv_conv_weight_promoted_to_fp16_palettized, x = new_state_cast_fp16)[name = string("conv_out_cast_fp16")]; tensor input_55_cast_fp16 = mul(x = var_1491_1, y = conv_out_cast_fp16)[name = string("input_55_cast_fp16")]; string y_pad_type_0 = const()[name = string("y_pad_type_0"), val = string("valid")]; tensor y_strides_0 = const()[name = string("y_strides_0"), val = tensor([1, 1])]; tensor y_pad_0 = const()[name = string("y_pad_0"), val = tensor([0, 0, 0, 0])]; tensor y_dilations_0 = const()[name = string("y_dilations_0"), val = tensor([1, 1])]; int32 y_groups_0 = const()[name = string("y_groups_0"), val = int32(1)]; tensor layers_3_conv_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53153152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53939648))))[name = string("layers_3_conv_out_proj_weight_promoted_to_fp16_palettized")]; tensor y_cast_fp16 = conv(dilations = y_dilations_0, groups = y_groups_0, pad = y_pad_0, pad_type = y_pad_type_0, strides = y_strides_0, weight = layers_3_conv_out_proj_weight_promoted_to_fp16_palettized, x = input_55_cast_fp16)[name = string("y_cast_fp16")]; tensor var_1539_axes_0 = const()[name = string("op_1539_axes_0"), val = tensor([2])]; tensor var_1539_cast_fp16 = squeeze(axes = var_1539_axes_0, x = y_cast_fp16)[name = string("op_1539_cast_fp16")]; tensor var_1543 = const()[name = string("op_1543"), val = tensor([0, 2, 1])]; tensor op_out_cast_fp16 = transpose(perm = var_1543, x = var_1539_cast_fp16)[name = string("transpose_2")]; tensor x_cast_fp16 = add(x = x_21_cast_fp16, y = op_out_cast_fp16)[name = string("x_cast_fp16")]; fp16 const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1547_cast_fp16 = mul(x = x_cast_fp16, y = const_27_promoted_to_fp16)[name = string("op_1547_cast_fp16")]; int32 var_1549 = const()[name = string("op_1549"), val = int32(-1)]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57_cast_fp16 = concat(axis = var_1549, interleave = input_57_interleave_0, values = (x_cast_fp16, var_1547_cast_fp16))[name = string("input_57_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_1555_to_fp16 = const()[name = string("op_1555_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1555_to_fp16, x = input_57_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_1558_split_sizes_0 = const()[name = string("op_1558_split_sizes_0"), val = tensor([1024, 1024])]; int32 var_1558_axis_0 = const()[name = string("op_1558_axis_0"), val = int32(-1)]; tensor var_1558_cast_fp16_0, tensor var_1558_cast_fp16_1 = split(axis = var_1558_axis_0, split_sizes = var_1558_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_1558_cast_fp16")]; tensor layers_3_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_ffn_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53943808)))]; tensor normed_cast_fp16 = mul(x = var_1558_cast_fp16_0, y = layers_3_ffn_norm_weight_promoted_to_fp16)[name = string("normed_cast_fp16")]; tensor var_1564 = const()[name = string("op_1564"), val = tensor([0, 2, 1])]; tensor var_1567_axes_0 = const()[name = string("op_1567_axes_0"), val = tensor([2])]; tensor var_1565_cast_fp16 = transpose(perm = var_1564, x = normed_cast_fp16)[name = string("transpose_1")]; tensor var_1567_cast_fp16 = expand_dims(axes = var_1567_axes_0, x = var_1565_cast_fp16)[name = string("op_1567_cast_fp16")]; string input_61_pad_type_0 = const()[name = string("input_61_pad_type_0"), val = string("valid")]; tensor input_61_strides_0 = const()[name = string("input_61_strides_0"), val = tensor([1, 1])]; tensor input_61_pad_0 = const()[name = string("input_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_61_dilations_0 = const()[name = string("input_61_dilations_0"), val = tensor([1, 1])]; int32 input_61_groups_0 = const()[name = string("input_61_groups_0"), val = int32(1)]; tensor input_61 = conv(dilations = input_61_dilations_0, groups = input_61_groups_0, pad = input_61_pad_0, pad_type = input_61_pad_type_0, strides = input_61_strides_0, weight = layers_3_feed_forward_w1_weight_palettized, x = var_1567_cast_fp16)[name = string("input_61")]; string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = layers_3_feed_forward_w3_weight_palettized, x = var_1567_cast_fp16)[name = string("b")]; tensor var_1595 = silu(x = input_61)[name = string("op_1595")]; tensor input = mul(x = var_1595, y = b)[name = string("input")]; string mlp_13_pad_type_0 = const()[name = string("mlp_13_pad_type_0"), val = string("valid")]; tensor mlp_13_strides_0 = const()[name = string("mlp_13_strides_0"), val = tensor([1, 1])]; tensor mlp_13_pad_0 = const()[name = string("mlp_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_13_dilations_0 = const()[name = string("mlp_13_dilations_0"), val = tensor([1, 1])]; int32 mlp_13_groups_0 = const()[name = string("mlp_13_groups_0"), val = int32(1)]; tensor mlp_13 = conv(dilations = mlp_13_dilations_0, groups = mlp_13_groups_0, pad = mlp_13_pad_0, pad_type = mlp_13_pad_type_0, strides = mlp_13_strides_0, weight = layers_3_feed_forward_w2_weight_palettized, x = input)[name = string("mlp_13")]; tensor var_1609_axes_0 = const()[name = string("op_1609_axes_0"), val = tensor([2])]; tensor var_1609 = squeeze(axes = var_1609_axes_0, x = mlp_13)[name = string("op_1609")]; tensor var_1613 = const()[name = string("op_1613"), val = tensor([0, 2, 1])]; tensor mlp = transpose(perm = var_1613, x = var_1609)[name = string("transpose_0")]; tensor hidden_out = add(x = x_cast_fp16, y = mlp)[name = string("op_1616_cast_fp16")]; int32 var_1619_axis_0 = const()[name = string("op_1619_axis_0"), val = int32(0)]; tensor conv_state_out = stack(axis = var_1619_axis_0, values = (var_798_cast_fp16, new_slot_cast_fp16))[name = string("op_1619_cast_fp16")]; int32 var_1622_axis_0 = const()[name = string("op_1622_axis_0"), val = int32(0)]; tensor var_1622 = stack(axis = var_1622_axis_0, values = (k_slice_1, k_slice))[name = string("op_1622")]; int32 var_1625_axis_0 = const()[name = string("op_1625_axis_0"), val = int32(0)]; tensor var_1625 = stack(axis = var_1625_axis_0, values = (var_279, var_994))[name = string("op_1625")]; int32 var_1627 = const()[name = string("op_1627"), val = int32(0)]; bool var_1628_interleave_0 = const()[name = string("op_1628_interleave_0"), val = bool(false)]; tensor kv_slice_out = concat(axis = var_1627, interleave = var_1628_interleave_0, values = (var_1622, var_1625))[name = string("op_1628")]; tensor update_mask_tmp = identity(x = update_mask)[name = string("update_mask_tmp")]; } -> (hidden_out, kv_slice_out, conv_state_out); }