| program(1.3) |
| [buildInfo = dict<string, string>({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.22.1"}})] |
| { |
| func main<ios18>(tensor<fp16, [1, 1025, 1, 1]> causal_mask, tensor<fp16, [2, 1024, 3]> conv_state_in, tensor<fp16, [1, 1, 1024]> hidden_in, tensor<fp16, [4, 1, 512, 1, 1024]> kv_cache_in, tensor<int32, [1]> position_ids, tensor<fp16, [1, 1, 1024, 1]> update_mask) { |
| tensor<fp16, [64]> layers_2_self_attn_k_layernorm_weight = const()[name = string("layers_2_self_attn_k_layernorm_weight"), val = tensor<fp16, [64]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))]; |
| tensor<fp16, [64]> layers_2_self_attn_q_layernorm_weight = const()[name = string("layers_2_self_attn_q_layernorm_weight"), val = tensor<fp16, [64]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(256)))]; |
| fp16 attn_scale = const()[name = string("attn_scale"), val = fp16(0x1p-3)]; |
| tensor<fp16, [64]> layers_0_self_attn_k_layernorm_weight = const()[name = string("layers_0_self_attn_k_layernorm_weight"), val = tensor<fp16, [64]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(448)))]; |
| tensor<fp16, [64]> layers_0_self_attn_q_layernorm_weight = const()[name = string("layers_0_self_attn_q_layernorm_weight"), val = tensor<fp16, [64]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(640)))]; |
| tensor<fp16, [1024]> layers_0_operator_norm_weight = const()[name = string("layers_0_operator_norm_weight"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(832)))]; |
| tensor<fp16, [2048, 64]> sin_cached_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [2048, 64]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2944))), lut = tensor<fp16, [64, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101312))))[name = string("sin_cached_palettized")]; |
| tensor<fp16, [2048, 64]> cos_cached_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [2048, 64]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109568))), lut = tensor<fp16, [64, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207936))))[name = string("cos_cached_palettized")]; |
| tensor<fp16, [1024, 1024, 1, 1]> layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216192))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002688))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; |
| tensor<fp16, [512, 1024, 1, 1]> layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [512, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1006848))), lut = tensor<fp16, [16, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1400128))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; |
| tensor<fp16, [512, 1024, 1, 1]> layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [512, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1402240))), lut = tensor<fp16, [16, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1795520))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_0_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1797632))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5336640))))[name = string("layers_0_feed_forward_w1_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_0_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5355136))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8894144))))[name = string("layers_0_feed_forward_w3_weight_palettized")]; |
| tensor<fp16, [1024, 4608, 1, 1]> layers_0_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 4608, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8912640))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12451648))))[name = string("layers_0_feed_forward_w2_weight_palettized")]; |
| tensor<fp16, [3072, 1024, 1, 1]> layers_1_conv_in_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [3072, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12455808))), lut = tensor<fp16, [96, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14815168))))[name = string("layers_1_conv_in_proj_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_1_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14827520))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18366528))))[name = string("layers_1_feed_forward_w1_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_1_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18385024))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21924032))))[name = string("layers_1_feed_forward_w3_weight_palettized")]; |
| tensor<fp16, [1024, 4608, 1, 1]> layers_1_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 4608, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21942528))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25481536))))[name = string("layers_1_feed_forward_w2_weight_palettized")]; |
| tensor<fp16, [1024, 1024, 1, 1]> layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25485696))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26272192))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; |
| tensor<fp16, [512, 1024, 1, 1]> layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [512, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26276352))), lut = tensor<fp16, [16, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26669632))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; |
| tensor<fp16, [512, 1024, 1, 1]> layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [512, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26671744))), lut = tensor<fp16, [16, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27065024))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_2_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27067136))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30606144))))[name = string("layers_2_feed_forward_w1_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_2_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30624640))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34163648))))[name = string("layers_2_feed_forward_w3_weight_palettized")]; |
| tensor<fp16, [1024, 4608, 1, 1]> layers_2_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 4608, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34182144))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37721152))))[name = string("layers_2_feed_forward_w2_weight_palettized")]; |
| tensor<fp16, [3072, 1024, 1, 1]> layers_3_conv_in_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [3072, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37725312))), lut = tensor<fp16, [96, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40084672))))[name = string("layers_3_conv_in_proj_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_3_feed_forward_w1_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40097024))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43636032))))[name = string("layers_3_feed_forward_w1_weight_palettized")]; |
| tensor<fp16, [4608, 1024, 1, 1]> layers_3_feed_forward_w3_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [4608, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43654528))), lut = tensor<fp16, [144, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47193536))))[name = string("layers_3_feed_forward_w3_weight_palettized")]; |
| tensor<fp16, [1024, 4608, 1, 1]> layers_3_feed_forward_w2_weight_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 4608, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47212032))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50751040))))[name = string("layers_3_feed_forward_w2_weight_palettized")]; |
| int32 var_175_batch_dims_0 = const()[name = string("op_175_batch_dims_0"), val = int32(0)]; |
| bool var_175_validate_indices_0 = const()[name = string("op_175_validate_indices_0"), val = bool(false)]; |
| string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; |
| string cast_20_dtype_0 = const()[name = string("cast_20_dtype_0"), val = string("int32")]; |
| int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; |
| tensor<int16, [1]> position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; |
| tensor<int32, [1]> cast_20 = cast(dtype = cast_20_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; |
| tensor<bool, [1]> greater_equal_0 = greater_equal(x = cast_20, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; |
| int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; |
| tensor<int32, [1]> add_0 = add(x = cast_20, y = slice_by_index_0)[name = string("add_0")]; |
| tensor<int32, [1]> select_0 = select(a = cast_20, b = add_0, cond = greater_equal_0)[name = string("select_0")]; |
| string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; |
| string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; |
| int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; |
| tensor<int16, [1]> select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; |
| tensor<int32, [1]> cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; |
| tensor<bool, [1]> greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; |
| int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(2048)]; |
| tensor<int32, [1]> add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; |
| tensor<int32, [1]> select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; |
| int32 op_175_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_175_cast_uint16_cast_uint16_axis_0"), val = int32(0)]; |
| tensor<fp16, [1, 64]> op_175_cast_uint16_cast_uint16 = gather(axis = op_175_cast_uint16_cast_uint16_axis_0, batch_dims = var_175_batch_dims_0, indices = select_0_1, validate_indices = var_175_validate_indices_0, x = cos_cached_palettized)[name = string("op_175_cast_uint16_cast_uint16")]; |
| tensor<int32, [4]> var_180 = const()[name = string("op_180"), val = tensor<int32, [4]>([1, 1, 1, 64])]; |
| tensor<fp16, [1, 1, 1, 64]> cos = reshape(shape = var_180, x = op_175_cast_uint16_cast_uint16)[name = string("cos")]; |
| int32 var_182 = const()[name = string("op_182"), val = int32(0)]; |
| int32 var_183_batch_dims_0 = const()[name = string("op_183_batch_dims_0"), val = int32(0)]; |
| bool var_183_validate_indices_0 = const()[name = string("op_183_validate_indices_0"), val = bool(false)]; |
| string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; |
| tensor<uint16, [1]> position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; |
| tensor<fp16, [1, 64]> var_183_cast_uint16 = gather(axis = var_182, batch_dims = var_183_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_183_validate_indices_0, x = sin_cached_palettized)[name = string("op_183_cast_uint16")]; |
| tensor<int32, [4]> var_188 = const()[name = string("op_188"), val = tensor<int32, [4]>([1, 1, 1, 64])]; |
| tensor<fp16, [1, 1, 1, 64]> sin = reshape(shape = var_188, x = var_183_cast_uint16)[name = string("sin")]; |
| fp16 const_0_promoted = const()[name = string("const_0_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_190 = mul(x = hidden_in, y = const_0_promoted)[name = string("op_190")]; |
| int32 var_192 = const()[name = string("op_192"), val = int32(-1)]; |
| bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_1 = concat(axis = var_192, interleave = input_1_interleave_0, values = (hidden_in, var_190))[name = string("input_1")]; |
| tensor<int32, [1]> normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_198_to_fp16 = const()[name = string("op_198_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_198_to_fp16, x = input_1)[name = string("normed_1_cast_fp16")]; |
| tensor<int32, [2]> var_201_split_sizes_0 = const()[name = string("op_201_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_201_axis_0 = const()[name = string("op_201_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_201_0, tensor<fp16, [1, 1, 1024]> var_201_1 = split(axis = var_201_axis_0, split_sizes = var_201_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_201")]; |
| tensor<fp16, [1, 1, 1024]> hidden_states_1 = mul(x = var_201_0, y = layers_0_operator_norm_weight)[name = string("hidden_states_1")]; |
| tensor<int32, [3]> var_207 = const()[name = string("op_207"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_210_axes_0 = const()[name = string("op_210_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_208 = transpose(perm = var_207, x = hidden_states_1)[name = string("transpose_27")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_210 = expand_dims(axes = var_210_axes_0, x = var_208)[name = string("op_210")]; |
| string var_226_pad_type_0 = const()[name = string("op_226_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> var_226_strides_0 = const()[name = string("op_226_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> var_226_pad_0 = const()[name = string("op_226_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> var_226_dilations_0 = const()[name = string("op_226_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 var_226_groups_0 = const()[name = string("op_226_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> var_226 = conv(dilations = var_226_dilations_0, groups = var_226_groups_0, pad = var_226_pad_0, pad_type = var_226_pad_type_0, strides = var_226_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_210)[name = string("op_226")]; |
| tensor<int32, [4]> var_231 = const()[name = string("op_231"), val = tensor<int32, [4]>([1, 16, 64, 1])]; |
| tensor<fp16, [1, 16, 64, 1]> var_232 = reshape(shape = var_231, x = var_226)[name = string("op_232")]; |
| tensor<int32, [4]> var_237 = const()[name = string("op_237"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| string var_254_pad_type_0 = const()[name = string("op_254_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> var_254_strides_0 = const()[name = string("op_254_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> var_254_pad_0 = const()[name = string("op_254_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> var_254_dilations_0 = const()[name = string("op_254_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 var_254_groups_0 = const()[name = string("op_254_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 512, 1, 1]> var_254 = conv(dilations = var_254_dilations_0, groups = var_254_groups_0, pad = var_254_pad_0, pad_type = var_254_pad_type_0, strides = var_254_strides_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_210)[name = string("op_254")]; |
| tensor<int32, [4]> var_259 = const()[name = string("op_259"), val = tensor<int32, [4]>([1, 8, 64, 1])]; |
| tensor<fp16, [1, 8, 64, 1]> var_260 = reshape(shape = var_259, x = var_254)[name = string("op_260")]; |
| tensor<int32, [4]> var_265 = const()[name = string("op_265"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| string var_282_pad_type_0 = const()[name = string("op_282_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> var_282_strides_0 = const()[name = string("op_282_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> var_282_pad_0 = const()[name = string("op_282_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> var_282_dilations_0 = const()[name = string("op_282_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 var_282_groups_0 = const()[name = string("op_282_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 512, 1, 1]> var_282 = conv(dilations = var_282_dilations_0, groups = var_282_groups_0, pad = var_282_pad_0, pad_type = var_282_pad_type_0, strides = var_282_strides_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_210)[name = string("op_282")]; |
| fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 16, 1, 64]> var_238 = transpose(perm = var_237, x = var_232)[name = string("transpose_26")]; |
| tensor<fp16, [1, 16, 1, 64]> var_300 = mul(x = var_238, y = const_1_promoted)[name = string("op_300")]; |
| int32 var_302 = const()[name = string("op_302"), val = int32(-1)]; |
| bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 16, 1, 128]> input_5 = concat(axis = var_302, interleave = input_5_interleave_0, values = (var_238, var_300))[name = string("input_5")]; |
| tensor<int32, [1]> normed_3_axes_0 = const()[name = string("normed_3_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_308_to_fp16 = const()[name = string("op_308_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 16, 1, 128]> normed_3_cast_fp16 = layer_norm(axes = normed_3_axes_0, epsilon = var_308_to_fp16, x = input_5)[name = string("normed_3_cast_fp16")]; |
| tensor<int32, [2]> var_311_split_sizes_0 = const()[name = string("op_311_split_sizes_0"), val = tensor<int32, [2]>([64, 64])]; |
| int32 var_311_axis_0 = const()[name = string("op_311_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 16, 1, 64]> var_311_0, tensor<fp16, [1, 16, 1, 64]> var_311_1 = split(axis = var_311_axis_0, split_sizes = var_311_split_sizes_0, x = normed_3_cast_fp16)[name = string("op_311")]; |
| tensor<fp16, [1, 16, 1, 64]> q_1 = mul(x = var_311_0, y = layers_0_self_attn_q_layernorm_weight)[name = string("q_1")]; |
| fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 8, 1, 64]> var_266 = transpose(perm = var_265, x = var_260)[name = string("transpose_25")]; |
| tensor<fp16, [1, 8, 1, 64]> var_314 = mul(x = var_266, y = const_2_promoted)[name = string("op_314")]; |
| int32 var_316 = const()[name = string("op_316"), val = int32(-1)]; |
| bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 8, 1, 128]> input_7 = concat(axis = var_316, interleave = input_7_interleave_0, values = (var_266, var_314))[name = string("input_7")]; |
| tensor<int32, [1]> normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_322_to_fp16 = const()[name = string("op_322_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 8, 1, 128]> normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_322_to_fp16, x = input_7)[name = string("normed_5_cast_fp16")]; |
| tensor<int32, [2]> var_325_split_sizes_0 = const()[name = string("op_325_split_sizes_0"), val = tensor<int32, [2]>([64, 64])]; |
| int32 var_325_axis_0 = const()[name = string("op_325_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 8, 1, 64]> var_325_0, tensor<fp16, [1, 8, 1, 64]> var_325_1 = split(axis = var_325_axis_0, split_sizes = var_325_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_325")]; |
| tensor<fp16, [1, 8, 1, 64]> k_1 = mul(x = var_325_0, y = layers_0_self_attn_k_layernorm_weight)[name = string("k_1")]; |
| tensor<fp16, [1, 16, 1, 64]> var_328 = mul(x = q_1, y = cos)[name = string("op_328")]; |
| tensor<int32, [2]> var_329_split_sizes_0 = const()[name = string("op_329_split_sizes_0"), val = tensor<int32, [2]>([32, 32])]; |
| int32 var_329_axis_0 = const()[name = string("op_329_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 16, 1, 32]> var_329_0, tensor<fp16, [1, 16, 1, 32]> var_329_1 = split(axis = var_329_axis_0, split_sizes = var_329_split_sizes_0, x = q_1)[name = string("op_329")]; |
| fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 16, 1, 32]> var_331 = mul(x = var_329_1, y = const_3_promoted)[name = string("op_331")]; |
| int32 var_333 = const()[name = string("op_333"), val = int32(-1)]; |
| bool var_334_interleave_0 = const()[name = string("op_334_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 16, 1, 64]> var_334 = concat(axis = var_333, interleave = var_334_interleave_0, values = (var_331, var_329_0))[name = string("op_334")]; |
| tensor<fp16, [1, 16, 1, 64]> var_335 = mul(x = var_334, y = sin)[name = string("op_335")]; |
| tensor<fp16, [1, 16, 1, 64]> q_3 = add(x = var_328, y = var_335)[name = string("q_3")]; |
| tensor<fp16, [1, 8, 1, 64]> var_338 = mul(x = k_1, y = cos)[name = string("op_338")]; |
| tensor<int32, [2]> var_339_split_sizes_0 = const()[name = string("op_339_split_sizes_0"), val = tensor<int32, [2]>([32, 32])]; |
| int32 var_339_axis_0 = const()[name = string("op_339_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 8, 1, 32]> var_339_0, tensor<fp16, [1, 8, 1, 32]> var_339_1 = split(axis = var_339_axis_0, split_sizes = var_339_split_sizes_0, x = k_1)[name = string("op_339")]; |
| fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 8, 1, 32]> var_341 = mul(x = var_339_1, y = const_4_promoted)[name = string("op_341")]; |
| int32 var_343 = const()[name = string("op_343"), val = int32(-1)]; |
| bool var_344_interleave_0 = const()[name = string("op_344_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 8, 1, 64]> var_344 = concat(axis = var_343, interleave = var_344_interleave_0, values = (var_341, var_339_0))[name = string("op_344")]; |
| tensor<fp16, [1, 8, 1, 64]> var_345 = mul(x = var_344, y = sin)[name = string("op_345")]; |
| tensor<fp16, [1, 8, 1, 64]> k_3 = add(x = var_338, y = var_345)[name = string("k_3")]; |
| tensor<int32, [5]> K_cache_1_begin_0 = const()[name = string("K_cache_1_begin_0"), val = tensor<int32, [5]>([0, 0, 0, 0, 0])]; |
| tensor<int32, [5]> K_cache_1_end_0 = const()[name = string("K_cache_1_end_0"), val = tensor<int32, [5]>([1, 1, 512, 1, 1024])]; |
| tensor<bool, [5]> K_cache_1_end_mask_0 = const()[name = string("K_cache_1_end_mask_0"), val = tensor<bool, [5]>([false, true, true, true, true])]; |
| tensor<bool, [5]> K_cache_1_squeeze_mask_0 = const()[name = string("K_cache_1_squeeze_mask_0"), val = tensor<bool, [5]>([true, false, false, false, false])]; |
| tensor<fp16, [1, 512, 1, 1024]> K_cache_1_cast_fp16 = slice_by_index(begin = K_cache_1_begin_0, end = K_cache_1_end_0, end_mask = K_cache_1_end_mask_0, squeeze_mask = K_cache_1_squeeze_mask_0, x = kv_cache_in)[name = string("K_cache_1_cast_fp16")]; |
| tensor<int32, [5]> V_cache_1_begin_0 = const()[name = string("V_cache_1_begin_0"), val = tensor<int32, [5]>([2, 0, 0, 0, 0])]; |
| tensor<int32, [5]> V_cache_1_end_0 = const()[name = string("V_cache_1_end_0"), val = tensor<int32, [5]>([3, 1, 512, 1, 1024])]; |
| tensor<bool, [5]> V_cache_1_end_mask_0 = const()[name = string("V_cache_1_end_mask_0"), val = tensor<bool, [5]>([false, true, true, true, true])]; |
| tensor<bool, [5]> V_cache_1_squeeze_mask_0 = const()[name = string("V_cache_1_squeeze_mask_0"), val = tensor<bool, [5]>([true, false, false, false, false])]; |
| tensor<fp16, [1, 512, 1, 1024]> V_cache_1_cast_fp16 = slice_by_index(begin = V_cache_1_begin_0, end = V_cache_1_end_0, end_mask = V_cache_1_end_mask_0, squeeze_mask = V_cache_1_squeeze_mask_0, x = kv_cache_in)[name = string("V_cache_1_cast_fp16")]; |
| tensor<int32, [4]> var_358 = const()[name = string("op_358"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| tensor<int32, [4]> var_364 = const()[name = string("op_364"), val = tensor<int32, [4]>([1, 1024, 1, 1])]; |
| tensor<fp16, [1, 16, 64, 1]> var_359 = transpose(perm = var_358, x = q_3)[name = string("transpose_24")]; |
| tensor<fp16, [1, 1024, 1, 1]> query_1 = reshape(shape = var_364, x = var_359)[name = string("query_1")]; |
| tensor<int32, [4]> var_370 = const()[name = string("op_370"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| tensor<int32, [4]> var_376 = const()[name = string("op_376"), val = tensor<int32, [4]>([1, 512, 1, 1])]; |
| tensor<fp16, [1, 8, 64, 1]> var_371 = transpose(perm = var_370, x = k_3)[name = string("transpose_23")]; |
| tensor<fp16, [1, 512, 1, 1]> k_slice_1 = reshape(shape = var_376, x = var_371)[name = string("k_slice_1")]; |
| int32 var_391 = const()[name = string("op_391"), val = int32(-1)]; |
| bool key_1_interleave_0 = const()[name = string("key_1_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 512, 1, 1025]> key_1_cast_fp16 = concat(axis = var_391, interleave = key_1_interleave_0, values = (K_cache_1_cast_fp16, k_slice_1))[name = string("key_1_cast_fp16")]; |
| int32 var_394 = const()[name = string("op_394"), val = int32(-1)]; |
| bool var_395_interleave_0 = const()[name = string("op_395_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 512, 1, 1025]> var_395_cast_fp16 = concat(axis = var_394, interleave = var_395_interleave_0, values = (V_cache_1_cast_fp16, var_282))[name = string("op_395_cast_fp16")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_396 = mul(x = query_1, y = attn_scale)[name = string("op_396")]; |
| tensor<int32, [16]> tile_0 = const()[name = string("tile_0"), val = tensor<int32, [16]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50755200)))]; |
| int32 var_399_axis_0 = const()[name = string("op_399_axis_0"), val = int32(1)]; |
| tensor<fp16, [1, 64, 1, 1]> var_399_0, tensor<fp16, [1, 64, 1, 1]> var_399_1, tensor<fp16, [1, 64, 1, 1]> var_399_2, tensor<fp16, [1, 64, 1, 1]> var_399_3, tensor<fp16, [1, 64, 1, 1]> var_399_4, tensor<fp16, [1, 64, 1, 1]> var_399_5, tensor<fp16, [1, 64, 1, 1]> var_399_6, tensor<fp16, [1, 64, 1, 1]> var_399_7, tensor<fp16, [1, 64, 1, 1]> var_399_8, tensor<fp16, [1, 64, 1, 1]> var_399_9, tensor<fp16, [1, 64, 1, 1]> var_399_10, tensor<fp16, [1, 64, 1, 1]> var_399_11, tensor<fp16, [1, 64, 1, 1]> var_399_12, tensor<fp16, [1, 64, 1, 1]> var_399_13, tensor<fp16, [1, 64, 1, 1]> var_399_14, tensor<fp16, [1, 64, 1, 1]> var_399_15 = split(axis = var_399_axis_0, split_sizes = tile_0, x = var_396)[name = string("op_399")]; |
| tensor<int32, [4]> var_418_perm_0 = const()[name = string("op_418_perm_0"), val = tensor<int32, [4]>([0, 3, 2, 1])]; |
| tensor<int32, [8]> tile_1 = const()[name = string("tile_1"), val = tensor<int32, [8]>([64, 64, 64, 64, 64, 64, 64, 64])]; |
| int32 var_421_axis_0 = const()[name = string("op_421_axis_0"), val = int32(3)]; |
| tensor<fp16, [1, 1025, 1, 512]> var_418_cast_fp16 = transpose(perm = var_418_perm_0, x = key_1_cast_fp16)[name = string("transpose_22")]; |
| tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_0, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_1, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_2, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_3, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_4, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_5, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_6, tensor<fp16, [1, 1025, 1, 64]> var_421_cast_fp16_7 = split(axis = var_421_axis_0, split_sizes = tile_1, x = var_418_cast_fp16)[name = string("op_421_cast_fp16")]; |
| tensor<int32, [8]> tile_2 = const()[name = string("tile_2"), val = tensor<int32, [8]>([64, 64, 64, 64, 64, 64, 64, 64])]; |
| int32 var_432_axis_0 = const()[name = string("op_432_axis_0"), val = int32(1)]; |
| tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_0, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_1, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_2, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_3, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_4, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_5, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_6, tensor<fp16, [1, 64, 1, 1025]> var_432_cast_fp16_7 = split(axis = var_432_axis_0, split_sizes = tile_2, x = var_395_cast_fp16)[name = string("op_432_cast_fp16")]; |
| string scores_1_equation_0 = const()[name = string("scores_1_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_1_cast_fp16 = einsum(equation = scores_1_equation_0, values = (var_421_cast_fp16_0, var_399_0))[name = string("scores_1_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_446_cast_fp16 = add(x = scores_1_cast_fp16, y = causal_mask)[name = string("op_446_cast_fp16")]; |
| int32 var_447 = const()[name = string("op_447"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_449_cast_fp16 = softmax(axis = var_447, x = var_446_cast_fp16)[name = string("op_449_cast_fp16")]; |
| string var_453_equation_0 = const()[name = string("op_453_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_453_cast_fp16 = einsum(equation = var_453_equation_0, values = (var_432_cast_fp16_0, var_449_cast_fp16))[name = string("op_453_cast_fp16")]; |
| string scores_3_equation_0 = const()[name = string("scores_3_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_3_cast_fp16 = einsum(equation = scores_3_equation_0, values = (var_421_cast_fp16_0, var_399_1))[name = string("scores_3_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_459_cast_fp16 = add(x = scores_3_cast_fp16, y = causal_mask)[name = string("op_459_cast_fp16")]; |
| int32 var_460 = const()[name = string("op_460"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_462_cast_fp16 = softmax(axis = var_460, x = var_459_cast_fp16)[name = string("op_462_cast_fp16")]; |
| string var_466_equation_0 = const()[name = string("op_466_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_466_cast_fp16 = einsum(equation = var_466_equation_0, values = (var_432_cast_fp16_0, var_462_cast_fp16))[name = string("op_466_cast_fp16")]; |
| string scores_5_equation_0 = const()[name = string("scores_5_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_5_cast_fp16 = einsum(equation = scores_5_equation_0, values = (var_421_cast_fp16_1, var_399_2))[name = string("scores_5_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_472_cast_fp16 = add(x = scores_5_cast_fp16, y = causal_mask)[name = string("op_472_cast_fp16")]; |
| int32 var_473 = const()[name = string("op_473"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_475_cast_fp16 = softmax(axis = var_473, x = var_472_cast_fp16)[name = string("op_475_cast_fp16")]; |
| string var_479_equation_0 = const()[name = string("op_479_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_479_cast_fp16 = einsum(equation = var_479_equation_0, values = (var_432_cast_fp16_1, var_475_cast_fp16))[name = string("op_479_cast_fp16")]; |
| string scores_7_equation_0 = const()[name = string("scores_7_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_7_cast_fp16 = einsum(equation = scores_7_equation_0, values = (var_421_cast_fp16_1, var_399_3))[name = string("scores_7_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_485_cast_fp16 = add(x = scores_7_cast_fp16, y = causal_mask)[name = string("op_485_cast_fp16")]; |
| int32 var_486 = const()[name = string("op_486"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_488_cast_fp16 = softmax(axis = var_486, x = var_485_cast_fp16)[name = string("op_488_cast_fp16")]; |
| string var_492_equation_0 = const()[name = string("op_492_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_492_cast_fp16 = einsum(equation = var_492_equation_0, values = (var_432_cast_fp16_1, var_488_cast_fp16))[name = string("op_492_cast_fp16")]; |
| string scores_9_equation_0 = const()[name = string("scores_9_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_9_cast_fp16 = einsum(equation = scores_9_equation_0, values = (var_421_cast_fp16_2, var_399_4))[name = string("scores_9_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_498_cast_fp16 = add(x = scores_9_cast_fp16, y = causal_mask)[name = string("op_498_cast_fp16")]; |
| int32 var_499 = const()[name = string("op_499"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_501_cast_fp16 = softmax(axis = var_499, x = var_498_cast_fp16)[name = string("op_501_cast_fp16")]; |
| string var_505_equation_0 = const()[name = string("op_505_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_505_cast_fp16 = einsum(equation = var_505_equation_0, values = (var_432_cast_fp16_2, var_501_cast_fp16))[name = string("op_505_cast_fp16")]; |
| string scores_11_equation_0 = const()[name = string("scores_11_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_11_cast_fp16 = einsum(equation = scores_11_equation_0, values = (var_421_cast_fp16_2, var_399_5))[name = string("scores_11_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_511_cast_fp16 = add(x = scores_11_cast_fp16, y = causal_mask)[name = string("op_511_cast_fp16")]; |
| int32 var_512 = const()[name = string("op_512"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_514_cast_fp16 = softmax(axis = var_512, x = var_511_cast_fp16)[name = string("op_514_cast_fp16")]; |
| string var_518_equation_0 = const()[name = string("op_518_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_518_cast_fp16 = einsum(equation = var_518_equation_0, values = (var_432_cast_fp16_2, var_514_cast_fp16))[name = string("op_518_cast_fp16")]; |
| string scores_13_equation_0 = const()[name = string("scores_13_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_13_cast_fp16 = einsum(equation = scores_13_equation_0, values = (var_421_cast_fp16_3, var_399_6))[name = string("scores_13_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_524_cast_fp16 = add(x = scores_13_cast_fp16, y = causal_mask)[name = string("op_524_cast_fp16")]; |
| int32 var_525 = const()[name = string("op_525"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_527_cast_fp16 = softmax(axis = var_525, x = var_524_cast_fp16)[name = string("op_527_cast_fp16")]; |
| string var_531_equation_0 = const()[name = string("op_531_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_531_cast_fp16 = einsum(equation = var_531_equation_0, values = (var_432_cast_fp16_3, var_527_cast_fp16))[name = string("op_531_cast_fp16")]; |
| string scores_15_equation_0 = const()[name = string("scores_15_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_15_cast_fp16 = einsum(equation = scores_15_equation_0, values = (var_421_cast_fp16_3, var_399_7))[name = string("scores_15_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_537_cast_fp16 = add(x = scores_15_cast_fp16, y = causal_mask)[name = string("op_537_cast_fp16")]; |
| int32 var_538 = const()[name = string("op_538"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_540_cast_fp16 = softmax(axis = var_538, x = var_537_cast_fp16)[name = string("op_540_cast_fp16")]; |
| string var_544_equation_0 = const()[name = string("op_544_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_544_cast_fp16 = einsum(equation = var_544_equation_0, values = (var_432_cast_fp16_3, var_540_cast_fp16))[name = string("op_544_cast_fp16")]; |
| string scores_17_equation_0 = const()[name = string("scores_17_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_17_cast_fp16 = einsum(equation = scores_17_equation_0, values = (var_421_cast_fp16_4, var_399_8))[name = string("scores_17_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_550_cast_fp16 = add(x = scores_17_cast_fp16, y = causal_mask)[name = string("op_550_cast_fp16")]; |
| int32 var_551 = const()[name = string("op_551"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_553_cast_fp16 = softmax(axis = var_551, x = var_550_cast_fp16)[name = string("op_553_cast_fp16")]; |
| string var_557_equation_0 = const()[name = string("op_557_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_557_cast_fp16 = einsum(equation = var_557_equation_0, values = (var_432_cast_fp16_4, var_553_cast_fp16))[name = string("op_557_cast_fp16")]; |
| string scores_19_equation_0 = const()[name = string("scores_19_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_19_cast_fp16 = einsum(equation = scores_19_equation_0, values = (var_421_cast_fp16_4, var_399_9))[name = string("scores_19_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_563_cast_fp16 = add(x = scores_19_cast_fp16, y = causal_mask)[name = string("op_563_cast_fp16")]; |
| int32 var_564 = const()[name = string("op_564"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_566_cast_fp16 = softmax(axis = var_564, x = var_563_cast_fp16)[name = string("op_566_cast_fp16")]; |
| string var_570_equation_0 = const()[name = string("op_570_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_570_cast_fp16 = einsum(equation = var_570_equation_0, values = (var_432_cast_fp16_4, var_566_cast_fp16))[name = string("op_570_cast_fp16")]; |
| string scores_21_equation_0 = const()[name = string("scores_21_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_21_cast_fp16 = einsum(equation = scores_21_equation_0, values = (var_421_cast_fp16_5, var_399_10))[name = string("scores_21_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_576_cast_fp16 = add(x = scores_21_cast_fp16, y = causal_mask)[name = string("op_576_cast_fp16")]; |
| int32 var_577 = const()[name = string("op_577"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_579_cast_fp16 = softmax(axis = var_577, x = var_576_cast_fp16)[name = string("op_579_cast_fp16")]; |
| string var_583_equation_0 = const()[name = string("op_583_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_583_cast_fp16 = einsum(equation = var_583_equation_0, values = (var_432_cast_fp16_5, var_579_cast_fp16))[name = string("op_583_cast_fp16")]; |
| string scores_23_equation_0 = const()[name = string("scores_23_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_23_cast_fp16 = einsum(equation = scores_23_equation_0, values = (var_421_cast_fp16_5, var_399_11))[name = string("scores_23_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_589_cast_fp16 = add(x = scores_23_cast_fp16, y = causal_mask)[name = string("op_589_cast_fp16")]; |
| int32 var_590 = const()[name = string("op_590"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_592_cast_fp16 = softmax(axis = var_590, x = var_589_cast_fp16)[name = string("op_592_cast_fp16")]; |
| string var_596_equation_0 = const()[name = string("op_596_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_596_cast_fp16 = einsum(equation = var_596_equation_0, values = (var_432_cast_fp16_5, var_592_cast_fp16))[name = string("op_596_cast_fp16")]; |
| string scores_25_equation_0 = const()[name = string("scores_25_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_25_cast_fp16 = einsum(equation = scores_25_equation_0, values = (var_421_cast_fp16_6, var_399_12))[name = string("scores_25_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_602_cast_fp16 = add(x = scores_25_cast_fp16, y = causal_mask)[name = string("op_602_cast_fp16")]; |
| int32 var_603 = const()[name = string("op_603"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_605_cast_fp16 = softmax(axis = var_603, x = var_602_cast_fp16)[name = string("op_605_cast_fp16")]; |
| string var_609_equation_0 = const()[name = string("op_609_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_609_cast_fp16 = einsum(equation = var_609_equation_0, values = (var_432_cast_fp16_6, var_605_cast_fp16))[name = string("op_609_cast_fp16")]; |
| string scores_27_equation_0 = const()[name = string("scores_27_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_27_cast_fp16 = einsum(equation = scores_27_equation_0, values = (var_421_cast_fp16_6, var_399_13))[name = string("scores_27_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_615_cast_fp16 = add(x = scores_27_cast_fp16, y = causal_mask)[name = string("op_615_cast_fp16")]; |
| int32 var_616 = const()[name = string("op_616"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_618_cast_fp16 = softmax(axis = var_616, x = var_615_cast_fp16)[name = string("op_618_cast_fp16")]; |
| string var_622_equation_0 = const()[name = string("op_622_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_622_cast_fp16 = einsum(equation = var_622_equation_0, values = (var_432_cast_fp16_6, var_618_cast_fp16))[name = string("op_622_cast_fp16")]; |
| string scores_29_equation_0 = const()[name = string("scores_29_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_29_cast_fp16 = einsum(equation = scores_29_equation_0, values = (var_421_cast_fp16_7, var_399_14))[name = string("scores_29_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_628_cast_fp16 = add(x = scores_29_cast_fp16, y = causal_mask)[name = string("op_628_cast_fp16")]; |
| int32 var_629 = const()[name = string("op_629"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_631_cast_fp16 = softmax(axis = var_629, x = var_628_cast_fp16)[name = string("op_631_cast_fp16")]; |
| string var_635_equation_0 = const()[name = string("op_635_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_635_cast_fp16 = einsum(equation = var_635_equation_0, values = (var_432_cast_fp16_7, var_631_cast_fp16))[name = string("op_635_cast_fp16")]; |
| string scores_31_equation_0 = const()[name = string("scores_31_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_31_cast_fp16 = einsum(equation = scores_31_equation_0, values = (var_421_cast_fp16_7, var_399_15))[name = string("scores_31_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_641_cast_fp16 = add(x = scores_31_cast_fp16, y = causal_mask)[name = string("op_641_cast_fp16")]; |
| int32 var_642 = const()[name = string("op_642"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_644_cast_fp16 = softmax(axis = var_642, x = var_641_cast_fp16)[name = string("op_644_cast_fp16")]; |
| string var_648_equation_0 = const()[name = string("op_648_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_648_cast_fp16 = einsum(equation = var_648_equation_0, values = (var_432_cast_fp16_7, var_644_cast_fp16))[name = string("op_648_cast_fp16")]; |
| int32 var_650 = const()[name = string("op_650"), val = int32(1)]; |
| bool input_9_interleave_0 = const()[name = string("input_9_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1024, 1, 1]> input_9_cast_fp16 = concat(axis = var_650, interleave = input_9_interleave_0, values = (var_453_cast_fp16, var_466_cast_fp16, var_479_cast_fp16, var_492_cast_fp16, var_505_cast_fp16, var_518_cast_fp16, var_531_cast_fp16, var_544_cast_fp16, var_557_cast_fp16, var_570_cast_fp16, var_583_cast_fp16, var_596_cast_fp16, var_609_cast_fp16, var_622_cast_fp16, var_635_cast_fp16, var_648_cast_fp16))[name = string("input_9_cast_fp16")]; |
| string out_1_pad_type_0 = const()[name = string("out_1_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> out_1_strides_0 = const()[name = string("out_1_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> out_1_pad_0 = const()[name = string("out_1_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> out_1_dilations_0 = const()[name = string("out_1_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 out_1_groups_0 = const()[name = string("out_1_groups_0"), val = int32(1)]; |
| tensor<fp16, [1024, 1024, 1, 1]> layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50755328))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51541824))))[name = string("layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized")]; |
| tensor<fp16, [1, 1024, 1, 1]> out_1_cast_fp16 = conv(dilations = out_1_dilations_0, groups = out_1_groups_0, pad = out_1_pad_0, pad_type = out_1_pad_type_0, strides = out_1_strides_0, weight = layers_0_self_attn_out_proj_weight_promoted_to_fp16_palettized, x = input_9_cast_fp16)[name = string("out_1_cast_fp16")]; |
| tensor<int32, [1]> var_664_axes_0 = const()[name = string("op_664_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_664_cast_fp16 = squeeze(axes = var_664_axes_0, x = out_1_cast_fp16)[name = string("op_664_cast_fp16")]; |
| tensor<int32, [3]> var_668 = const()[name = string("op_668"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> op_out_1_cast_fp16 = transpose(perm = var_668, x = var_664_cast_fp16)[name = string("transpose_21")]; |
| tensor<fp16, [1, 1, 1024]> x_7_cast_fp16 = add(x = hidden_in, y = op_out_1_cast_fp16)[name = string("x_7_cast_fp16")]; |
| fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_672_cast_fp16 = mul(x = x_7_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_672_cast_fp16")]; |
| int32 var_674 = const()[name = string("op_674"), val = int32(-1)]; |
| bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_11_cast_fp16 = concat(axis = var_674, interleave = input_11_interleave_0, values = (x_7_cast_fp16, var_672_cast_fp16))[name = string("input_11_cast_fp16")]; |
| tensor<int32, [1]> normed_7_axes_0 = const()[name = string("normed_7_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_680_to_fp16 = const()[name = string("op_680_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_7_cast_fp16 = layer_norm(axes = normed_7_axes_0, epsilon = var_680_to_fp16, x = input_11_cast_fp16)[name = string("normed_7_cast_fp16")]; |
| tensor<int32, [2]> var_683_split_sizes_0 = const()[name = string("op_683_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_683_axis_0 = const()[name = string("op_683_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_683_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_683_cast_fp16_1 = split(axis = var_683_axis_0, split_sizes = var_683_split_sizes_0, x = normed_7_cast_fp16)[name = string("op_683_cast_fp16")]; |
| tensor<fp16, [1024]> layers_0_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_ffn_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51545984)))]; |
| tensor<fp16, [1, 1, 1024]> normed_9_cast_fp16 = mul(x = var_683_cast_fp16_0, y = layers_0_ffn_norm_weight_promoted_to_fp16)[name = string("normed_9_cast_fp16")]; |
| tensor<int32, [3]> var_689 = const()[name = string("op_689"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_692_axes_0 = const()[name = string("op_692_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_690_cast_fp16 = transpose(perm = var_689, x = normed_9_cast_fp16)[name = string("transpose_20")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_692_cast_fp16 = expand_dims(axes = var_692_axes_0, x = var_690_cast_fp16)[name = string("op_692_cast_fp16")]; |
| string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_0_feed_forward_w1_weight_palettized, x = var_692_cast_fp16)[name = string("input_15")]; |
| string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = layers_0_feed_forward_w3_weight_palettized, x = var_692_cast_fp16)[name = string("b_1")]; |
| tensor<fp16, [1, 4608, 1, 1]> var_720 = silu(x = input_15)[name = string("op_720")]; |
| tensor<fp16, [1, 4608, 1, 1]> input_17 = mul(x = var_720, y = b_1)[name = string("input_17")]; |
| string mlp_1_pad_type_0 = const()[name = string("mlp_1_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> mlp_1_strides_0 = const()[name = string("mlp_1_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> mlp_1_pad_0 = const()[name = string("mlp_1_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> mlp_1_dilations_0 = const()[name = string("mlp_1_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 mlp_1_groups_0 = const()[name = string("mlp_1_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> mlp_1 = conv(dilations = mlp_1_dilations_0, groups = mlp_1_groups_0, pad = mlp_1_pad_0, pad_type = mlp_1_pad_type_0, strides = mlp_1_strides_0, weight = layers_0_feed_forward_w2_weight_palettized, x = input_17)[name = string("mlp_1")]; |
| tensor<int32, [1]> var_734_axes_0 = const()[name = string("op_734_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_734 = squeeze(axes = var_734_axes_0, x = mlp_1)[name = string("op_734")]; |
| tensor<int32, [3]> var_738 = const()[name = string("op_738"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> mlp_3 = transpose(perm = var_738, x = var_734)[name = string("transpose_19")]; |
| tensor<fp16, [1, 1, 1024]> x_9_cast_fp16 = add(x = x_7_cast_fp16, y = mlp_3)[name = string("x_9_cast_fp16")]; |
| fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_742_cast_fp16 = mul(x = x_9_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_742_cast_fp16")]; |
| int32 var_744 = const()[name = string("op_744"), val = int32(-1)]; |
| bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_19_cast_fp16 = concat(axis = var_744, interleave = input_19_interleave_0, values = (x_9_cast_fp16, var_742_cast_fp16))[name = string("input_19_cast_fp16")]; |
| tensor<int32, [1]> normed_11_axes_0 = const()[name = string("normed_11_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_750_to_fp16 = const()[name = string("op_750_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_11_cast_fp16 = layer_norm(axes = normed_11_axes_0, epsilon = var_750_to_fp16, x = input_19_cast_fp16)[name = string("normed_11_cast_fp16")]; |
| tensor<int32, [2]> var_753_split_sizes_0 = const()[name = string("op_753_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_753_axis_0 = const()[name = string("op_753_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_753_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_753_cast_fp16_1 = split(axis = var_753_axis_0, split_sizes = var_753_split_sizes_0, x = normed_11_cast_fp16)[name = string("op_753_cast_fp16")]; |
| tensor<fp16, [1024]> layers_1_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_operator_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51548096)))]; |
| tensor<fp16, [1, 1, 1024]> hidden_states_3_cast_fp16 = mul(x = var_753_cast_fp16_0, y = layers_1_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; |
| tensor<int32, [3]> var_759 = const()[name = string("op_759"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_762_axes_0 = const()[name = string("op_762_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_760_cast_fp16 = transpose(perm = var_759, x = hidden_states_3_cast_fp16)[name = string("transpose_18")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_762_cast_fp16 = expand_dims(axes = var_762_axes_0, x = var_760_cast_fp16)[name = string("op_762_cast_fp16")]; |
| string BCx_1_pad_type_0 = const()[name = string("BCx_1_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> BCx_1_strides_0 = const()[name = string("BCx_1_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> BCx_1_pad_0 = const()[name = string("BCx_1_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> BCx_1_dilations_0 = const()[name = string("BCx_1_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 BCx_1_groups_0 = const()[name = string("BCx_1_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 3072, 1, 1]> BCx_1 = conv(dilations = BCx_1_dilations_0, groups = BCx_1_groups_0, pad = BCx_1_pad_0, pad_type = BCx_1_pad_type_0, strides = BCx_1_strides_0, weight = layers_1_conv_in_proj_weight_palettized, x = var_762_cast_fp16)[name = string("BCx_1")]; |
| tensor<int32, [3]> var_779_split_sizes_0 = const()[name = string("op_779_split_sizes_0"), val = tensor<int32, [3]>([1024, 1024, 1024])]; |
| int32 var_779_axis_0 = const()[name = string("op_779_axis_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> var_779_0, tensor<fp16, [1, 1024, 1, 1]> var_779_1, tensor<fp16, [1, 1024, 1, 1]> var_779_2 = split(axis = var_779_axis_0, split_sizes = var_779_split_sizes_0, x = BCx_1)[name = string("op_779")]; |
| tensor<fp16, [1, 1024, 1, 1]> Bx_1 = mul(x = var_779_0, y = var_779_2)[name = string("Bx_1")]; |
| tensor<int32, [3]> var_785_begin_0 = const()[name = string("op_785_begin_0"), val = tensor<int32, [3]>([0, 0, 0])]; |
| tensor<int32, [3]> var_785_end_0 = const()[name = string("op_785_end_0"), val = tensor<int32, [3]>([1, 1024, 3])]; |
| tensor<bool, [3]> var_785_end_mask_0 = const()[name = string("op_785_end_mask_0"), val = tensor<bool, [3]>([false, true, true])]; |
| tensor<bool, [3]> var_785_squeeze_mask_0 = const()[name = string("op_785_squeeze_mask_0"), val = tensor<bool, [3]>([true, false, false])]; |
| tensor<fp16, [1024, 3]> var_785_cast_fp16 = slice_by_index(begin = var_785_begin_0, end = var_785_end_0, end_mask = var_785_end_mask_0, squeeze_mask = var_785_squeeze_mask_0, x = conv_state_in)[name = string("op_785_cast_fp16")]; |
| tensor<int32, [1]> var_787_axes_0 = const()[name = string("op_787_axes_0"), val = tensor<int32, [1]>([0])]; |
| tensor<fp16, [1, 1024, 3]> var_787_cast_fp16 = expand_dims(axes = var_787_axes_0, x = var_785_cast_fp16)[name = string("op_787_cast_fp16")]; |
| tensor<int32, [1]> slot_1_axes_0 = const()[name = string("slot_1_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1, 3]> slot_1_cast_fp16 = expand_dims(axes = slot_1_axes_0, x = var_787_cast_fp16)[name = string("slot_1_cast_fp16")]; |
| tensor<int32, [4]> live_tail_1_begin_0 = const()[name = string("live_tail_1_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 1])]; |
| tensor<int32, [4]> live_tail_1_end_0 = const()[name = string("live_tail_1_end_0"), val = tensor<int32, [4]>([1, 1024, 1, 1])]; |
| tensor<bool, [4]> live_tail_1_end_mask_0 = const()[name = string("live_tail_1_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])]; |
| tensor<fp16, [1, 1024, 1, 2]> live_tail_1_cast_fp16 = slice_by_index(begin = live_tail_1_begin_0, end = live_tail_1_end_0, end_mask = live_tail_1_end_mask_0, x = slot_1_cast_fp16)[name = string("live_tail_1_cast_fp16")]; |
| int32 var_796 = const()[name = string("op_796"), val = int32(-1)]; |
| bool new_state_1_interleave_0 = const()[name = string("new_state_1_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1024, 1, 3]> new_state_1_cast_fp16 = concat(axis = var_796, interleave = new_state_1_interleave_0, values = (live_tail_1_cast_fp16, Bx_1))[name = string("new_state_1_cast_fp16")]; |
| tensor<int32, [1]> var_799_axes_0 = const()[name = string("op_799_axes_0"), val = tensor<int32, [1]>([0])]; |
| tensor<fp16, [1024, 1, 3]> var_799_cast_fp16 = squeeze(axes = var_799_axes_0, x = new_state_1_cast_fp16)[name = string("op_799_cast_fp16")]; |
| tensor<int32, [1]> var_801_axes_0 = const()[name = string("op_801_axes_0"), val = tensor<int32, [1]>([1])]; |
| tensor<fp16, [1024, 3]> var_801_cast_fp16 = squeeze(axes = var_801_axes_0, x = var_799_cast_fp16)[name = string("op_801_cast_fp16")]; |
| string conv_out_1_pad_type_0 = const()[name = string("conv_out_1_pad_type_0"), val = string("valid")]; |
| int32 conv_out_1_groups_0 = const()[name = string("conv_out_1_groups_0"), val = int32(1024)]; |
| tensor<int32, [2]> conv_out_1_strides_0 = const()[name = string("conv_out_1_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> conv_out_1_pad_0 = const()[name = string("conv_out_1_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> conv_out_1_dilations_0 = const()[name = string("conv_out_1_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<fp16, [1024, 1, 1, 3]> layers_1_conv_conv_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1, 1, 3]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51550208))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51552576))))[name = string("layers_1_conv_conv_weight_promoted_to_fp16_palettized")]; |
| tensor<fp16, [1, 1024, 1, 1]> conv_out_1_cast_fp16 = conv(dilations = conv_out_1_dilations_0, groups = conv_out_1_groups_0, pad = conv_out_1_pad_0, pad_type = conv_out_1_pad_type_0, strides = conv_out_1_strides_0, weight = layers_1_conv_conv_weight_promoted_to_fp16_palettized, x = new_state_1_cast_fp16)[name = string("conv_out_1_cast_fp16")]; |
| tensor<fp16, [1, 1024, 1, 1]> input_23_cast_fp16 = mul(x = var_779_1, y = conv_out_1_cast_fp16)[name = string("input_23_cast_fp16")]; |
| string y_1_pad_type_0 = const()[name = string("y_1_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> y_1_strides_0 = const()[name = string("y_1_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> y_1_pad_0 = const()[name = string("y_1_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> y_1_dilations_0 = const()[name = string("y_1_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 y_1_groups_0 = const()[name = string("y_1_groups_0"), val = int32(1)]; |
| tensor<fp16, [1024, 1024, 1, 1]> layers_1_conv_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51556736))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52343232))))[name = string("layers_1_conv_out_proj_weight_promoted_to_fp16_palettized")]; |
| tensor<fp16, [1, 1024, 1, 1]> y_1_cast_fp16 = conv(dilations = y_1_dilations_0, groups = y_1_groups_0, pad = y_1_pad_0, pad_type = y_1_pad_type_0, strides = y_1_strides_0, weight = layers_1_conv_out_proj_weight_promoted_to_fp16_palettized, x = input_23_cast_fp16)[name = string("y_1_cast_fp16")]; |
| tensor<int32, [1]> var_827_axes_0 = const()[name = string("op_827_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_827_cast_fp16 = squeeze(axes = var_827_axes_0, x = y_1_cast_fp16)[name = string("op_827_cast_fp16")]; |
| tensor<int32, [3]> var_831 = const()[name = string("op_831"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> op_out_3_cast_fp16 = transpose(perm = var_831, x = var_827_cast_fp16)[name = string("transpose_17")]; |
| tensor<fp16, [1, 1, 1024]> x_11_cast_fp16 = add(x = x_9_cast_fp16, y = op_out_3_cast_fp16)[name = string("x_11_cast_fp16")]; |
| fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_835_cast_fp16 = mul(x = x_11_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_835_cast_fp16")]; |
| int32 var_837 = const()[name = string("op_837"), val = int32(-1)]; |
| bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_25_cast_fp16 = concat(axis = var_837, interleave = input_25_interleave_0, values = (x_11_cast_fp16, var_835_cast_fp16))[name = string("input_25_cast_fp16")]; |
| tensor<int32, [1]> normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_843_to_fp16 = const()[name = string("op_843_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_843_to_fp16, x = input_25_cast_fp16)[name = string("normed_13_cast_fp16")]; |
| tensor<int32, [2]> var_846_split_sizes_0 = const()[name = string("op_846_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_846_axis_0 = const()[name = string("op_846_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_846_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_846_cast_fp16_1 = split(axis = var_846_axis_0, split_sizes = var_846_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_846_cast_fp16")]; |
| tensor<fp16, [1024]> layers_1_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_ffn_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52347392)))]; |
| tensor<fp16, [1, 1, 1024]> normed_15_cast_fp16 = mul(x = var_846_cast_fp16_0, y = layers_1_ffn_norm_weight_promoted_to_fp16)[name = string("normed_15_cast_fp16")]; |
| tensor<int32, [3]> var_852 = const()[name = string("op_852"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_855_axes_0 = const()[name = string("op_855_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_853_cast_fp16 = transpose(perm = var_852, x = normed_15_cast_fp16)[name = string("transpose_16")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_855_cast_fp16 = expand_dims(axes = var_855_axes_0, x = var_853_cast_fp16)[name = string("op_855_cast_fp16")]; |
| string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> input_29 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_1_feed_forward_w1_weight_palettized, x = var_855_cast_fp16)[name = string("input_29")]; |
| string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = layers_1_feed_forward_w3_weight_palettized, x = var_855_cast_fp16)[name = string("b_3")]; |
| tensor<fp16, [1, 4608, 1, 1]> var_883 = silu(x = input_29)[name = string("op_883")]; |
| tensor<fp16, [1, 4608, 1, 1]> input_31 = mul(x = var_883, y = b_3)[name = string("input_31")]; |
| string mlp_5_pad_type_0 = const()[name = string("mlp_5_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> mlp_5_strides_0 = const()[name = string("mlp_5_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> mlp_5_pad_0 = const()[name = string("mlp_5_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> mlp_5_dilations_0 = const()[name = string("mlp_5_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 mlp_5_groups_0 = const()[name = string("mlp_5_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> mlp_5 = conv(dilations = mlp_5_dilations_0, groups = mlp_5_groups_0, pad = mlp_5_pad_0, pad_type = mlp_5_pad_type_0, strides = mlp_5_strides_0, weight = layers_1_feed_forward_w2_weight_palettized, x = input_31)[name = string("mlp_5")]; |
| tensor<int32, [1]> var_897_axes_0 = const()[name = string("op_897_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_897 = squeeze(axes = var_897_axes_0, x = mlp_5)[name = string("op_897")]; |
| tensor<int32, [3]> var_901 = const()[name = string("op_901"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> mlp_7 = transpose(perm = var_901, x = var_897)[name = string("transpose_15")]; |
| tensor<fp16, [1, 1, 1024]> x_13_cast_fp16 = add(x = x_11_cast_fp16, y = mlp_7)[name = string("x_13_cast_fp16")]; |
| fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_905_cast_fp16 = mul(x = x_13_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_905_cast_fp16")]; |
| int32 var_907 = const()[name = string("op_907"), val = int32(-1)]; |
| bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_33_cast_fp16 = concat(axis = var_907, interleave = input_33_interleave_0, values = (x_13_cast_fp16, var_905_cast_fp16))[name = string("input_33_cast_fp16")]; |
| tensor<int32, [1]> normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_913_to_fp16 = const()[name = string("op_913_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_913_to_fp16, x = input_33_cast_fp16)[name = string("normed_17_cast_fp16")]; |
| tensor<int32, [2]> var_916_split_sizes_0 = const()[name = string("op_916_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_916_axis_0 = const()[name = string("op_916_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_916_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_916_cast_fp16_1 = split(axis = var_916_axis_0, split_sizes = var_916_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_916_cast_fp16")]; |
| tensor<fp16, [1024]> layers_2_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_operator_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52349504)))]; |
| tensor<fp16, [1, 1, 1024]> hidden_states_5_cast_fp16 = mul(x = var_916_cast_fp16_0, y = layers_2_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_5_cast_fp16")]; |
| tensor<int32, [3]> var_922 = const()[name = string("op_922"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_925_axes_0 = const()[name = string("op_925_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_923_cast_fp16 = transpose(perm = var_922, x = hidden_states_5_cast_fp16)[name = string("transpose_14")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_925_cast_fp16 = expand_dims(axes = var_925_axes_0, x = var_923_cast_fp16)[name = string("op_925_cast_fp16")]; |
| string var_941_pad_type_0 = const()[name = string("op_941_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> var_941_strides_0 = const()[name = string("op_941_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> var_941_pad_0 = const()[name = string("op_941_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> var_941_dilations_0 = const()[name = string("op_941_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 var_941_groups_0 = const()[name = string("op_941_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> var_941 = conv(dilations = var_941_dilations_0, groups = var_941_groups_0, pad = var_941_pad_0, pad_type = var_941_pad_type_0, strides = var_941_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_925_cast_fp16)[name = string("op_941")]; |
| tensor<int32, [4]> var_946 = const()[name = string("op_946"), val = tensor<int32, [4]>([1, 16, 64, 1])]; |
| tensor<fp16, [1, 16, 64, 1]> var_947 = reshape(shape = var_946, x = var_941)[name = string("op_947")]; |
| tensor<int32, [4]> var_952 = const()[name = string("op_952"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| string var_969_pad_type_0 = const()[name = string("op_969_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> var_969_strides_0 = const()[name = string("op_969_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> var_969_pad_0 = const()[name = string("op_969_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> var_969_dilations_0 = const()[name = string("op_969_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 var_969_groups_0 = const()[name = string("op_969_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 512, 1, 1]> var_969 = conv(dilations = var_969_dilations_0, groups = var_969_groups_0, pad = var_969_pad_0, pad_type = var_969_pad_type_0, strides = var_969_strides_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_925_cast_fp16)[name = string("op_969")]; |
| tensor<int32, [4]> var_974 = const()[name = string("op_974"), val = tensor<int32, [4]>([1, 8, 64, 1])]; |
| tensor<fp16, [1, 8, 64, 1]> var_975 = reshape(shape = var_974, x = var_969)[name = string("op_975")]; |
| tensor<int32, [4]> var_980 = const()[name = string("op_980"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| string var_997_pad_type_0 = const()[name = string("op_997_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> var_997_strides_0 = const()[name = string("op_997_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> var_997_pad_0 = const()[name = string("op_997_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> var_997_dilations_0 = const()[name = string("op_997_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 var_997_groups_0 = const()[name = string("op_997_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 512, 1, 1]> var_997 = conv(dilations = var_997_dilations_0, groups = var_997_groups_0, pad = var_997_pad_0, pad_type = var_997_pad_type_0, strides = var_997_strides_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_925_cast_fp16)[name = string("op_997")]; |
| fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 16, 1, 64]> var_953 = transpose(perm = var_952, x = var_947)[name = string("transpose_13")]; |
| tensor<fp16, [1, 16, 1, 64]> var_1015 = mul(x = var_953, y = const_15_promoted)[name = string("op_1015")]; |
| int32 var_1017 = const()[name = string("op_1017"), val = int32(-1)]; |
| bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 16, 1, 128]> input_37 = concat(axis = var_1017, interleave = input_37_interleave_0, values = (var_953, var_1015))[name = string("input_37")]; |
| tensor<int32, [1]> normed_19_axes_0 = const()[name = string("normed_19_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_1023_to_fp16 = const()[name = string("op_1023_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 16, 1, 128]> normed_19_cast_fp16 = layer_norm(axes = normed_19_axes_0, epsilon = var_1023_to_fp16, x = input_37)[name = string("normed_19_cast_fp16")]; |
| tensor<int32, [2]> var_1026_split_sizes_0 = const()[name = string("op_1026_split_sizes_0"), val = tensor<int32, [2]>([64, 64])]; |
| int32 var_1026_axis_0 = const()[name = string("op_1026_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 16, 1, 64]> var_1026_0, tensor<fp16, [1, 16, 1, 64]> var_1026_1 = split(axis = var_1026_axis_0, split_sizes = var_1026_split_sizes_0, x = normed_19_cast_fp16)[name = string("op_1026")]; |
| tensor<fp16, [1, 16, 1, 64]> q_5 = mul(x = var_1026_0, y = layers_2_self_attn_q_layernorm_weight)[name = string("q_5")]; |
| fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 8, 1, 64]> var_981 = transpose(perm = var_980, x = var_975)[name = string("transpose_12")]; |
| tensor<fp16, [1, 8, 1, 64]> var_1029 = mul(x = var_981, y = const_16_promoted)[name = string("op_1029")]; |
| int32 var_1031 = const()[name = string("op_1031"), val = int32(-1)]; |
| bool input_39_interleave_0 = const()[name = string("input_39_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 8, 1, 128]> input_39 = concat(axis = var_1031, interleave = input_39_interleave_0, values = (var_981, var_1029))[name = string("input_39")]; |
| tensor<int32, [1]> normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_1037_to_fp16 = const()[name = string("op_1037_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 8, 1, 128]> normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1037_to_fp16, x = input_39)[name = string("normed_21_cast_fp16")]; |
| tensor<int32, [2]> var_1040_split_sizes_0 = const()[name = string("op_1040_split_sizes_0"), val = tensor<int32, [2]>([64, 64])]; |
| int32 var_1040_axis_0 = const()[name = string("op_1040_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 8, 1, 64]> var_1040_0, tensor<fp16, [1, 8, 1, 64]> var_1040_1 = split(axis = var_1040_axis_0, split_sizes = var_1040_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_1040")]; |
| tensor<fp16, [1, 8, 1, 64]> k_5 = mul(x = var_1040_0, y = layers_2_self_attn_k_layernorm_weight)[name = string("k_5")]; |
| tensor<fp16, [1, 16, 1, 64]> var_1043 = mul(x = q_5, y = cos)[name = string("op_1043")]; |
| tensor<int32, [2]> var_1044_split_sizes_0 = const()[name = string("op_1044_split_sizes_0"), val = tensor<int32, [2]>([32, 32])]; |
| int32 var_1044_axis_0 = const()[name = string("op_1044_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 16, 1, 32]> var_1044_0, tensor<fp16, [1, 16, 1, 32]> var_1044_1 = split(axis = var_1044_axis_0, split_sizes = var_1044_split_sizes_0, x = q_5)[name = string("op_1044")]; |
| fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 16, 1, 32]> var_1046 = mul(x = var_1044_1, y = const_17_promoted)[name = string("op_1046")]; |
| int32 var_1048 = const()[name = string("op_1048"), val = int32(-1)]; |
| bool var_1049_interleave_0 = const()[name = string("op_1049_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 16, 1, 64]> var_1049 = concat(axis = var_1048, interleave = var_1049_interleave_0, values = (var_1046, var_1044_0))[name = string("op_1049")]; |
| tensor<fp16, [1, 16, 1, 64]> var_1050 = mul(x = var_1049, y = sin)[name = string("op_1050")]; |
| tensor<fp16, [1, 16, 1, 64]> q = add(x = var_1043, y = var_1050)[name = string("q")]; |
| tensor<fp16, [1, 8, 1, 64]> var_1053 = mul(x = k_5, y = cos)[name = string("op_1053")]; |
| tensor<int32, [2]> var_1054_split_sizes_0 = const()[name = string("op_1054_split_sizes_0"), val = tensor<int32, [2]>([32, 32])]; |
| int32 var_1054_axis_0 = const()[name = string("op_1054_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 8, 1, 32]> var_1054_0, tensor<fp16, [1, 8, 1, 32]> var_1054_1 = split(axis = var_1054_axis_0, split_sizes = var_1054_split_sizes_0, x = k_5)[name = string("op_1054")]; |
| fp16 const_18_promoted = const()[name = string("const_18_promoted"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 8, 1, 32]> var_1056 = mul(x = var_1054_1, y = const_18_promoted)[name = string("op_1056")]; |
| int32 var_1058 = const()[name = string("op_1058"), val = int32(-1)]; |
| bool var_1059_interleave_0 = const()[name = string("op_1059_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 8, 1, 64]> var_1059 = concat(axis = var_1058, interleave = var_1059_interleave_0, values = (var_1056, var_1054_0))[name = string("op_1059")]; |
| tensor<fp16, [1, 8, 1, 64]> var_1060 = mul(x = var_1059, y = sin)[name = string("op_1060")]; |
| tensor<fp16, [1, 8, 1, 64]> k = add(x = var_1053, y = var_1060)[name = string("k")]; |
| tensor<int32, [5]> K_cache_begin_0 = const()[name = string("K_cache_begin_0"), val = tensor<int32, [5]>([1, 0, 0, 0, 0])]; |
| tensor<int32, [5]> K_cache_end_0 = const()[name = string("K_cache_end_0"), val = tensor<int32, [5]>([2, 1, 512, 1, 1024])]; |
| tensor<bool, [5]> K_cache_end_mask_0 = const()[name = string("K_cache_end_mask_0"), val = tensor<bool, [5]>([false, true, true, true, true])]; |
| tensor<bool, [5]> K_cache_squeeze_mask_0 = const()[name = string("K_cache_squeeze_mask_0"), val = tensor<bool, [5]>([true, false, false, false, false])]; |
| tensor<fp16, [1, 512, 1, 1024]> K_cache_cast_fp16 = slice_by_index(begin = K_cache_begin_0, end = K_cache_end_0, end_mask = K_cache_end_mask_0, squeeze_mask = K_cache_squeeze_mask_0, x = kv_cache_in)[name = string("K_cache_cast_fp16")]; |
| tensor<int32, [5]> V_cache_begin_0 = const()[name = string("V_cache_begin_0"), val = tensor<int32, [5]>([3, 0, 0, 0, 0])]; |
| tensor<int32, [5]> V_cache_end_0 = const()[name = string("V_cache_end_0"), val = tensor<int32, [5]>([4, 1, 512, 1, 1024])]; |
| tensor<bool, [5]> V_cache_end_mask_0 = const()[name = string("V_cache_end_mask_0"), val = tensor<bool, [5]>([false, true, true, true, true])]; |
| tensor<bool, [5]> V_cache_squeeze_mask_0 = const()[name = string("V_cache_squeeze_mask_0"), val = tensor<bool, [5]>([true, false, false, false, false])]; |
| tensor<fp16, [1, 512, 1, 1024]> V_cache_cast_fp16 = slice_by_index(begin = V_cache_begin_0, end = V_cache_end_0, end_mask = V_cache_end_mask_0, squeeze_mask = V_cache_squeeze_mask_0, x = kv_cache_in)[name = string("V_cache_cast_fp16")]; |
| tensor<int32, [4]> var_1073 = const()[name = string("op_1073"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| tensor<int32, [4]> var_1079 = const()[name = string("op_1079"), val = tensor<int32, [4]>([1, 1024, 1, 1])]; |
| tensor<fp16, [1, 16, 64, 1]> var_1074 = transpose(perm = var_1073, x = q)[name = string("transpose_11")]; |
| tensor<fp16, [1, 1024, 1, 1]> query = reshape(shape = var_1079, x = var_1074)[name = string("query")]; |
| tensor<int32, [4]> var_1085 = const()[name = string("op_1085"), val = tensor<int32, [4]>([0, 1, 3, 2])]; |
| tensor<int32, [4]> var_1091 = const()[name = string("op_1091"), val = tensor<int32, [4]>([1, 512, 1, 1])]; |
| tensor<fp16, [1, 8, 64, 1]> var_1086 = transpose(perm = var_1085, x = k)[name = string("transpose_10")]; |
| tensor<fp16, [1, 512, 1, 1]> k_slice = reshape(shape = var_1091, x = var_1086)[name = string("k_slice")]; |
| int32 var_1106 = const()[name = string("op_1106"), val = int32(-1)]; |
| bool key_interleave_0 = const()[name = string("key_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 512, 1, 1025]> key_cast_fp16 = concat(axis = var_1106, interleave = key_interleave_0, values = (K_cache_cast_fp16, k_slice))[name = string("key_cast_fp16")]; |
| int32 var_1109 = const()[name = string("op_1109"), val = int32(-1)]; |
| bool var_1110_interleave_0 = const()[name = string("op_1110_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 512, 1, 1025]> var_1110_cast_fp16 = concat(axis = var_1109, interleave = var_1110_interleave_0, values = (V_cache_cast_fp16, var_997))[name = string("op_1110_cast_fp16")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_1111 = mul(x = query, y = attn_scale)[name = string("op_1111")]; |
| tensor<int32, [16]> tile_3 = const()[name = string("tile_3"), val = tensor<int32, [16]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52351616)))]; |
| int32 var_1114_axis_0 = const()[name = string("op_1114_axis_0"), val = int32(1)]; |
| tensor<fp16, [1, 64, 1, 1]> var_1114_0, tensor<fp16, [1, 64, 1, 1]> var_1114_1, tensor<fp16, [1, 64, 1, 1]> var_1114_2, tensor<fp16, [1, 64, 1, 1]> var_1114_3, tensor<fp16, [1, 64, 1, 1]> var_1114_4, tensor<fp16, [1, 64, 1, 1]> var_1114_5, tensor<fp16, [1, 64, 1, 1]> var_1114_6, tensor<fp16, [1, 64, 1, 1]> var_1114_7, tensor<fp16, [1, 64, 1, 1]> var_1114_8, tensor<fp16, [1, 64, 1, 1]> var_1114_9, tensor<fp16, [1, 64, 1, 1]> var_1114_10, tensor<fp16, [1, 64, 1, 1]> var_1114_11, tensor<fp16, [1, 64, 1, 1]> var_1114_12, tensor<fp16, [1, 64, 1, 1]> var_1114_13, tensor<fp16, [1, 64, 1, 1]> var_1114_14, tensor<fp16, [1, 64, 1, 1]> var_1114_15 = split(axis = var_1114_axis_0, split_sizes = tile_3, x = var_1111)[name = string("op_1114")]; |
| tensor<int32, [4]> var_1133_perm_0 = const()[name = string("op_1133_perm_0"), val = tensor<int32, [4]>([0, 3, 2, 1])]; |
| tensor<int32, [8]> tile_4 = const()[name = string("tile_4"), val = tensor<int32, [8]>([64, 64, 64, 64, 64, 64, 64, 64])]; |
| int32 var_1136_axis_0 = const()[name = string("op_1136_axis_0"), val = int32(3)]; |
| tensor<fp16, [1, 1025, 1, 512]> var_1133_cast_fp16 = transpose(perm = var_1133_perm_0, x = key_cast_fp16)[name = string("transpose_9")]; |
| tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_0, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_1, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_2, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_3, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_4, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_5, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_6, tensor<fp16, [1, 1025, 1, 64]> var_1136_cast_fp16_7 = split(axis = var_1136_axis_0, split_sizes = tile_4, x = var_1133_cast_fp16)[name = string("op_1136_cast_fp16")]; |
| tensor<int32, [8]> tile_5 = const()[name = string("tile_5"), val = tensor<int32, [8]>([64, 64, 64, 64, 64, 64, 64, 64])]; |
| int32 var_1147_axis_0 = const()[name = string("op_1147_axis_0"), val = int32(1)]; |
| tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_0, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_1, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_2, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_3, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_4, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_5, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_6, tensor<fp16, [1, 64, 1, 1025]> var_1147_cast_fp16_7 = split(axis = var_1147_axis_0, split_sizes = tile_5, x = var_1110_cast_fp16)[name = string("op_1147_cast_fp16")]; |
| string scores_33_equation_0 = const()[name = string("scores_33_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_33_cast_fp16 = einsum(equation = scores_33_equation_0, values = (var_1136_cast_fp16_0, var_1114_0))[name = string("scores_33_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1161_cast_fp16 = add(x = scores_33_cast_fp16, y = causal_mask)[name = string("op_1161_cast_fp16")]; |
| int32 var_1162 = const()[name = string("op_1162"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1164_cast_fp16 = softmax(axis = var_1162, x = var_1161_cast_fp16)[name = string("op_1164_cast_fp16")]; |
| string var_1168_equation_0 = const()[name = string("op_1168_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1168_cast_fp16 = einsum(equation = var_1168_equation_0, values = (var_1147_cast_fp16_0, var_1164_cast_fp16))[name = string("op_1168_cast_fp16")]; |
| string scores_35_equation_0 = const()[name = string("scores_35_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_35_cast_fp16 = einsum(equation = scores_35_equation_0, values = (var_1136_cast_fp16_0, var_1114_1))[name = string("scores_35_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1174_cast_fp16 = add(x = scores_35_cast_fp16, y = causal_mask)[name = string("op_1174_cast_fp16")]; |
| int32 var_1175 = const()[name = string("op_1175"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1177_cast_fp16 = softmax(axis = var_1175, x = var_1174_cast_fp16)[name = string("op_1177_cast_fp16")]; |
| string var_1181_equation_0 = const()[name = string("op_1181_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1181_cast_fp16 = einsum(equation = var_1181_equation_0, values = (var_1147_cast_fp16_0, var_1177_cast_fp16))[name = string("op_1181_cast_fp16")]; |
| string scores_37_equation_0 = const()[name = string("scores_37_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_37_cast_fp16 = einsum(equation = scores_37_equation_0, values = (var_1136_cast_fp16_1, var_1114_2))[name = string("scores_37_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1187_cast_fp16 = add(x = scores_37_cast_fp16, y = causal_mask)[name = string("op_1187_cast_fp16")]; |
| int32 var_1188 = const()[name = string("op_1188"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1190_cast_fp16 = softmax(axis = var_1188, x = var_1187_cast_fp16)[name = string("op_1190_cast_fp16")]; |
| string var_1194_equation_0 = const()[name = string("op_1194_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1194_cast_fp16 = einsum(equation = var_1194_equation_0, values = (var_1147_cast_fp16_1, var_1190_cast_fp16))[name = string("op_1194_cast_fp16")]; |
| string scores_39_equation_0 = const()[name = string("scores_39_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_39_cast_fp16 = einsum(equation = scores_39_equation_0, values = (var_1136_cast_fp16_1, var_1114_3))[name = string("scores_39_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1200_cast_fp16 = add(x = scores_39_cast_fp16, y = causal_mask)[name = string("op_1200_cast_fp16")]; |
| int32 var_1201 = const()[name = string("op_1201"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1203_cast_fp16 = softmax(axis = var_1201, x = var_1200_cast_fp16)[name = string("op_1203_cast_fp16")]; |
| string var_1207_equation_0 = const()[name = string("op_1207_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1207_cast_fp16 = einsum(equation = var_1207_equation_0, values = (var_1147_cast_fp16_1, var_1203_cast_fp16))[name = string("op_1207_cast_fp16")]; |
| string scores_41_equation_0 = const()[name = string("scores_41_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_41_cast_fp16 = einsum(equation = scores_41_equation_0, values = (var_1136_cast_fp16_2, var_1114_4))[name = string("scores_41_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1213_cast_fp16 = add(x = scores_41_cast_fp16, y = causal_mask)[name = string("op_1213_cast_fp16")]; |
| int32 var_1214 = const()[name = string("op_1214"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1216_cast_fp16 = softmax(axis = var_1214, x = var_1213_cast_fp16)[name = string("op_1216_cast_fp16")]; |
| string var_1220_equation_0 = const()[name = string("op_1220_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1220_cast_fp16 = einsum(equation = var_1220_equation_0, values = (var_1147_cast_fp16_2, var_1216_cast_fp16))[name = string("op_1220_cast_fp16")]; |
| string scores_43_equation_0 = const()[name = string("scores_43_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_43_cast_fp16 = einsum(equation = scores_43_equation_0, values = (var_1136_cast_fp16_2, var_1114_5))[name = string("scores_43_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1226_cast_fp16 = add(x = scores_43_cast_fp16, y = causal_mask)[name = string("op_1226_cast_fp16")]; |
| int32 var_1227 = const()[name = string("op_1227"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1229_cast_fp16 = softmax(axis = var_1227, x = var_1226_cast_fp16)[name = string("op_1229_cast_fp16")]; |
| string var_1233_equation_0 = const()[name = string("op_1233_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1233_cast_fp16 = einsum(equation = var_1233_equation_0, values = (var_1147_cast_fp16_2, var_1229_cast_fp16))[name = string("op_1233_cast_fp16")]; |
| string scores_45_equation_0 = const()[name = string("scores_45_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_45_cast_fp16 = einsum(equation = scores_45_equation_0, values = (var_1136_cast_fp16_3, var_1114_6))[name = string("scores_45_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1239_cast_fp16 = add(x = scores_45_cast_fp16, y = causal_mask)[name = string("op_1239_cast_fp16")]; |
| int32 var_1240 = const()[name = string("op_1240"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1242_cast_fp16 = softmax(axis = var_1240, x = var_1239_cast_fp16)[name = string("op_1242_cast_fp16")]; |
| string var_1246_equation_0 = const()[name = string("op_1246_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1246_cast_fp16 = einsum(equation = var_1246_equation_0, values = (var_1147_cast_fp16_3, var_1242_cast_fp16))[name = string("op_1246_cast_fp16")]; |
| string scores_47_equation_0 = const()[name = string("scores_47_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_47_cast_fp16 = einsum(equation = scores_47_equation_0, values = (var_1136_cast_fp16_3, var_1114_7))[name = string("scores_47_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1252_cast_fp16 = add(x = scores_47_cast_fp16, y = causal_mask)[name = string("op_1252_cast_fp16")]; |
| int32 var_1253 = const()[name = string("op_1253"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1255_cast_fp16 = softmax(axis = var_1253, x = var_1252_cast_fp16)[name = string("op_1255_cast_fp16")]; |
| string var_1259_equation_0 = const()[name = string("op_1259_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1259_cast_fp16 = einsum(equation = var_1259_equation_0, values = (var_1147_cast_fp16_3, var_1255_cast_fp16))[name = string("op_1259_cast_fp16")]; |
| string scores_49_equation_0 = const()[name = string("scores_49_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_49_cast_fp16 = einsum(equation = scores_49_equation_0, values = (var_1136_cast_fp16_4, var_1114_8))[name = string("scores_49_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1265_cast_fp16 = add(x = scores_49_cast_fp16, y = causal_mask)[name = string("op_1265_cast_fp16")]; |
| int32 var_1266 = const()[name = string("op_1266"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1268_cast_fp16 = softmax(axis = var_1266, x = var_1265_cast_fp16)[name = string("op_1268_cast_fp16")]; |
| string var_1272_equation_0 = const()[name = string("op_1272_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1272_cast_fp16 = einsum(equation = var_1272_equation_0, values = (var_1147_cast_fp16_4, var_1268_cast_fp16))[name = string("op_1272_cast_fp16")]; |
| string scores_51_equation_0 = const()[name = string("scores_51_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_51_cast_fp16 = einsum(equation = scores_51_equation_0, values = (var_1136_cast_fp16_4, var_1114_9))[name = string("scores_51_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1278_cast_fp16 = add(x = scores_51_cast_fp16, y = causal_mask)[name = string("op_1278_cast_fp16")]; |
| int32 var_1279 = const()[name = string("op_1279"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1281_cast_fp16 = softmax(axis = var_1279, x = var_1278_cast_fp16)[name = string("op_1281_cast_fp16")]; |
| string var_1285_equation_0 = const()[name = string("op_1285_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1285_cast_fp16 = einsum(equation = var_1285_equation_0, values = (var_1147_cast_fp16_4, var_1281_cast_fp16))[name = string("op_1285_cast_fp16")]; |
| string scores_53_equation_0 = const()[name = string("scores_53_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_53_cast_fp16 = einsum(equation = scores_53_equation_0, values = (var_1136_cast_fp16_5, var_1114_10))[name = string("scores_53_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1291_cast_fp16 = add(x = scores_53_cast_fp16, y = causal_mask)[name = string("op_1291_cast_fp16")]; |
| int32 var_1292 = const()[name = string("op_1292"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1294_cast_fp16 = softmax(axis = var_1292, x = var_1291_cast_fp16)[name = string("op_1294_cast_fp16")]; |
| string var_1298_equation_0 = const()[name = string("op_1298_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1298_cast_fp16 = einsum(equation = var_1298_equation_0, values = (var_1147_cast_fp16_5, var_1294_cast_fp16))[name = string("op_1298_cast_fp16")]; |
| string scores_55_equation_0 = const()[name = string("scores_55_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_55_cast_fp16 = einsum(equation = scores_55_equation_0, values = (var_1136_cast_fp16_5, var_1114_11))[name = string("scores_55_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1304_cast_fp16 = add(x = scores_55_cast_fp16, y = causal_mask)[name = string("op_1304_cast_fp16")]; |
| int32 var_1305 = const()[name = string("op_1305"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1307_cast_fp16 = softmax(axis = var_1305, x = var_1304_cast_fp16)[name = string("op_1307_cast_fp16")]; |
| string var_1311_equation_0 = const()[name = string("op_1311_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1311_cast_fp16 = einsum(equation = var_1311_equation_0, values = (var_1147_cast_fp16_5, var_1307_cast_fp16))[name = string("op_1311_cast_fp16")]; |
| string scores_57_equation_0 = const()[name = string("scores_57_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_57_cast_fp16 = einsum(equation = scores_57_equation_0, values = (var_1136_cast_fp16_6, var_1114_12))[name = string("scores_57_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1317_cast_fp16 = add(x = scores_57_cast_fp16, y = causal_mask)[name = string("op_1317_cast_fp16")]; |
| int32 var_1318 = const()[name = string("op_1318"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1320_cast_fp16 = softmax(axis = var_1318, x = var_1317_cast_fp16)[name = string("op_1320_cast_fp16")]; |
| string var_1324_equation_0 = const()[name = string("op_1324_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1324_cast_fp16 = einsum(equation = var_1324_equation_0, values = (var_1147_cast_fp16_6, var_1320_cast_fp16))[name = string("op_1324_cast_fp16")]; |
| string scores_59_equation_0 = const()[name = string("scores_59_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_59_cast_fp16 = einsum(equation = scores_59_equation_0, values = (var_1136_cast_fp16_6, var_1114_13))[name = string("scores_59_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1330_cast_fp16 = add(x = scores_59_cast_fp16, y = causal_mask)[name = string("op_1330_cast_fp16")]; |
| int32 var_1331 = const()[name = string("op_1331"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1333_cast_fp16 = softmax(axis = var_1331, x = var_1330_cast_fp16)[name = string("op_1333_cast_fp16")]; |
| string var_1337_equation_0 = const()[name = string("op_1337_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1337_cast_fp16 = einsum(equation = var_1337_equation_0, values = (var_1147_cast_fp16_6, var_1333_cast_fp16))[name = string("op_1337_cast_fp16")]; |
| string scores_61_equation_0 = const()[name = string("scores_61_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_61_cast_fp16 = einsum(equation = scores_61_equation_0, values = (var_1136_cast_fp16_7, var_1114_14))[name = string("scores_61_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1343_cast_fp16 = add(x = scores_61_cast_fp16, y = causal_mask)[name = string("op_1343_cast_fp16")]; |
| int32 var_1344 = const()[name = string("op_1344"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1346_cast_fp16 = softmax(axis = var_1344, x = var_1343_cast_fp16)[name = string("op_1346_cast_fp16")]; |
| string var_1350_equation_0 = const()[name = string("op_1350_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1350_cast_fp16 = einsum(equation = var_1350_equation_0, values = (var_1147_cast_fp16_7, var_1346_cast_fp16))[name = string("op_1350_cast_fp16")]; |
| string scores_equation_0 = const()[name = string("scores_equation_0"), val = string("bkhc,bchq->bkhq")]; |
| tensor<fp16, [1, 1025, 1, 1]> scores_cast_fp16 = einsum(equation = scores_equation_0, values = (var_1136_cast_fp16_7, var_1114_15))[name = string("scores_cast_fp16")]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1356_cast_fp16 = add(x = scores_cast_fp16, y = causal_mask)[name = string("op_1356_cast_fp16")]; |
| int32 var_1357 = const()[name = string("op_1357"), val = int32(1)]; |
| tensor<fp16, [1, 1025, 1, 1]> var_1359_cast_fp16 = softmax(axis = var_1357, x = var_1356_cast_fp16)[name = string("op_1359_cast_fp16")]; |
| string var_1363_equation_0 = const()[name = string("op_1363_equation_0"), val = string("bchk,bkhq->bchq")]; |
| tensor<fp16, [1, 64, 1, 1]> var_1363_cast_fp16 = einsum(equation = var_1363_equation_0, values = (var_1147_cast_fp16_7, var_1359_cast_fp16))[name = string("op_1363_cast_fp16")]; |
| int32 var_1365 = const()[name = string("op_1365"), val = int32(1)]; |
| bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1024, 1, 1]> input_41_cast_fp16 = concat(axis = var_1365, interleave = input_41_interleave_0, values = (var_1168_cast_fp16, var_1181_cast_fp16, var_1194_cast_fp16, var_1207_cast_fp16, var_1220_cast_fp16, var_1233_cast_fp16, var_1246_cast_fp16, var_1259_cast_fp16, var_1272_cast_fp16, var_1285_cast_fp16, var_1298_cast_fp16, var_1311_cast_fp16, var_1324_cast_fp16, var_1337_cast_fp16, var_1350_cast_fp16, var_1363_cast_fp16))[name = string("input_41_cast_fp16")]; |
| string out_pad_type_0 = const()[name = string("out_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> out_strides_0 = const()[name = string("out_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> out_pad_0 = const()[name = string("out_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> out_dilations_0 = const()[name = string("out_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 out_groups_0 = const()[name = string("out_groups_0"), val = int32(1)]; |
| tensor<fp16, [1024, 1024, 1, 1]> layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52351744))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53138240))))[name = string("layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized")]; |
| tensor<fp16, [1, 1024, 1, 1]> out_cast_fp16 = conv(dilations = out_dilations_0, groups = out_groups_0, pad = out_pad_0, pad_type = out_pad_type_0, strides = out_strides_0, weight = layers_2_self_attn_out_proj_weight_promoted_to_fp16_palettized, x = input_41_cast_fp16)[name = string("out_cast_fp16")]; |
| tensor<int32, [1]> var_1379_axes_0 = const()[name = string("op_1379_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1379_cast_fp16 = squeeze(axes = var_1379_axes_0, x = out_cast_fp16)[name = string("op_1379_cast_fp16")]; |
| tensor<int32, [3]> var_1383 = const()[name = string("op_1383"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> op_out_5_cast_fp16 = transpose(perm = var_1383, x = var_1379_cast_fp16)[name = string("transpose_8")]; |
| tensor<fp16, [1, 1, 1024]> x_19_cast_fp16 = add(x = x_13_cast_fp16, y = op_out_5_cast_fp16)[name = string("x_19_cast_fp16")]; |
| fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_1387_cast_fp16 = mul(x = x_19_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_1387_cast_fp16")]; |
| int32 var_1389 = const()[name = string("op_1389"), val = int32(-1)]; |
| bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_43_cast_fp16 = concat(axis = var_1389, interleave = input_43_interleave_0, values = (x_19_cast_fp16, var_1387_cast_fp16))[name = string("input_43_cast_fp16")]; |
| tensor<int32, [1]> normed_23_axes_0 = const()[name = string("normed_23_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_1395_to_fp16 = const()[name = string("op_1395_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_23_cast_fp16 = layer_norm(axes = normed_23_axes_0, epsilon = var_1395_to_fp16, x = input_43_cast_fp16)[name = string("normed_23_cast_fp16")]; |
| tensor<int32, [2]> var_1398_split_sizes_0 = const()[name = string("op_1398_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_1398_axis_0 = const()[name = string("op_1398_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_1398_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_1398_cast_fp16_1 = split(axis = var_1398_axis_0, split_sizes = var_1398_split_sizes_0, x = normed_23_cast_fp16)[name = string("op_1398_cast_fp16")]; |
| tensor<fp16, [1024]> layers_2_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_ffn_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53142400)))]; |
| tensor<fp16, [1, 1, 1024]> normed_25_cast_fp16 = mul(x = var_1398_cast_fp16_0, y = layers_2_ffn_norm_weight_promoted_to_fp16)[name = string("normed_25_cast_fp16")]; |
| tensor<int32, [3]> var_1404 = const()[name = string("op_1404"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_1407_axes_0 = const()[name = string("op_1407_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1405_cast_fp16 = transpose(perm = var_1404, x = normed_25_cast_fp16)[name = string("transpose_7")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_1407_cast_fp16 = expand_dims(axes = var_1407_axes_0, x = var_1405_cast_fp16)[name = string("op_1407_cast_fp16")]; |
| string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> input_47 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_2_feed_forward_w1_weight_palettized, x = var_1407_cast_fp16)[name = string("input_47")]; |
| string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = layers_2_feed_forward_w3_weight_palettized, x = var_1407_cast_fp16)[name = string("b_5")]; |
| tensor<fp16, [1, 4608, 1, 1]> var_1435 = silu(x = input_47)[name = string("op_1435")]; |
| tensor<fp16, [1, 4608, 1, 1]> input_49 = mul(x = var_1435, y = b_5)[name = string("input_49")]; |
| string mlp_9_pad_type_0 = const()[name = string("mlp_9_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> mlp_9_strides_0 = const()[name = string("mlp_9_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> mlp_9_pad_0 = const()[name = string("mlp_9_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> mlp_9_dilations_0 = const()[name = string("mlp_9_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 mlp_9_groups_0 = const()[name = string("mlp_9_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> mlp_9 = conv(dilations = mlp_9_dilations_0, groups = mlp_9_groups_0, pad = mlp_9_pad_0, pad_type = mlp_9_pad_type_0, strides = mlp_9_strides_0, weight = layers_2_feed_forward_w2_weight_palettized, x = input_49)[name = string("mlp_9")]; |
| tensor<int32, [1]> var_1449_axes_0 = const()[name = string("op_1449_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1449 = squeeze(axes = var_1449_axes_0, x = mlp_9)[name = string("op_1449")]; |
| tensor<int32, [3]> var_1453 = const()[name = string("op_1453"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> mlp_11 = transpose(perm = var_1453, x = var_1449)[name = string("transpose_6")]; |
| tensor<fp16, [1, 1, 1024]> x_21_cast_fp16 = add(x = x_19_cast_fp16, y = mlp_11)[name = string("x_21_cast_fp16")]; |
| fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_1457_cast_fp16 = mul(x = x_21_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_1457_cast_fp16")]; |
| int32 var_1459 = const()[name = string("op_1459"), val = int32(-1)]; |
| bool input_51_interleave_0 = const()[name = string("input_51_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_51_cast_fp16 = concat(axis = var_1459, interleave = input_51_interleave_0, values = (x_21_cast_fp16, var_1457_cast_fp16))[name = string("input_51_cast_fp16")]; |
| tensor<int32, [1]> normed_27_axes_0 = const()[name = string("normed_27_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_1465_to_fp16 = const()[name = string("op_1465_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_27_cast_fp16 = layer_norm(axes = normed_27_axes_0, epsilon = var_1465_to_fp16, x = input_51_cast_fp16)[name = string("normed_27_cast_fp16")]; |
| tensor<int32, [2]> var_1468_split_sizes_0 = const()[name = string("op_1468_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_1468_axis_0 = const()[name = string("op_1468_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_1468_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_1468_cast_fp16_1 = split(axis = var_1468_axis_0, split_sizes = var_1468_split_sizes_0, x = normed_27_cast_fp16)[name = string("op_1468_cast_fp16")]; |
| tensor<fp16, [1024]> layers_3_operator_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_operator_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53144512)))]; |
| tensor<fp16, [1, 1, 1024]> hidden_states_7_cast_fp16 = mul(x = var_1468_cast_fp16_0, y = layers_3_operator_norm_weight_promoted_to_fp16)[name = string("hidden_states_7_cast_fp16")]; |
| tensor<int32, [3]> var_1474 = const()[name = string("op_1474"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_1477_axes_0 = const()[name = string("op_1477_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1475_cast_fp16 = transpose(perm = var_1474, x = hidden_states_7_cast_fp16)[name = string("transpose_5")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_1477_cast_fp16 = expand_dims(axes = var_1477_axes_0, x = var_1475_cast_fp16)[name = string("op_1477_cast_fp16")]; |
| string BCx_pad_type_0 = const()[name = string("BCx_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> BCx_strides_0 = const()[name = string("BCx_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> BCx_pad_0 = const()[name = string("BCx_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> BCx_dilations_0 = const()[name = string("BCx_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 BCx_groups_0 = const()[name = string("BCx_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 3072, 1, 1]> BCx = conv(dilations = BCx_dilations_0, groups = BCx_groups_0, pad = BCx_pad_0, pad_type = BCx_pad_type_0, strides = BCx_strides_0, weight = layers_3_conv_in_proj_weight_palettized, x = var_1477_cast_fp16)[name = string("BCx")]; |
| tensor<int32, [3]> var_1494_split_sizes_0 = const()[name = string("op_1494_split_sizes_0"), val = tensor<int32, [3]>([1024, 1024, 1024])]; |
| int32 var_1494_axis_0 = const()[name = string("op_1494_axis_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> var_1494_0, tensor<fp16, [1, 1024, 1, 1]> var_1494_1, tensor<fp16, [1, 1024, 1, 1]> var_1494_2 = split(axis = var_1494_axis_0, split_sizes = var_1494_split_sizes_0, x = BCx)[name = string("op_1494")]; |
| tensor<fp16, [1, 1024, 1, 1]> Bx = mul(x = var_1494_0, y = var_1494_2)[name = string("Bx")]; |
| tensor<int32, [3]> var_1500_begin_0 = const()[name = string("op_1500_begin_0"), val = tensor<int32, [3]>([1, 0, 0])]; |
| tensor<int32, [3]> var_1500_end_0 = const()[name = string("op_1500_end_0"), val = tensor<int32, [3]>([2, 1024, 3])]; |
| tensor<bool, [3]> var_1500_end_mask_0 = const()[name = string("op_1500_end_mask_0"), val = tensor<bool, [3]>([false, true, true])]; |
| tensor<bool, [3]> var_1500_squeeze_mask_0 = const()[name = string("op_1500_squeeze_mask_0"), val = tensor<bool, [3]>([true, false, false])]; |
| tensor<fp16, [1024, 3]> var_1500_cast_fp16 = slice_by_index(begin = var_1500_begin_0, end = var_1500_end_0, end_mask = var_1500_end_mask_0, squeeze_mask = var_1500_squeeze_mask_0, x = conv_state_in)[name = string("op_1500_cast_fp16")]; |
| tensor<int32, [1]> var_1502_axes_0 = const()[name = string("op_1502_axes_0"), val = tensor<int32, [1]>([0])]; |
| tensor<fp16, [1, 1024, 3]> var_1502_cast_fp16 = expand_dims(axes = var_1502_axes_0, x = var_1500_cast_fp16)[name = string("op_1502_cast_fp16")]; |
| tensor<int32, [1]> slot_axes_0 = const()[name = string("slot_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1, 3]> slot_cast_fp16 = expand_dims(axes = slot_axes_0, x = var_1502_cast_fp16)[name = string("slot_cast_fp16")]; |
| tensor<int32, [4]> live_tail_begin_0 = const()[name = string("live_tail_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 1])]; |
| tensor<int32, [4]> live_tail_end_0 = const()[name = string("live_tail_end_0"), val = tensor<int32, [4]>([1, 1024, 1, 1])]; |
| tensor<bool, [4]> live_tail_end_mask_0 = const()[name = string("live_tail_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])]; |
| tensor<fp16, [1, 1024, 1, 2]> live_tail_cast_fp16 = slice_by_index(begin = live_tail_begin_0, end = live_tail_end_0, end_mask = live_tail_end_mask_0, x = slot_cast_fp16)[name = string("live_tail_cast_fp16")]; |
| int32 var_1511 = const()[name = string("op_1511"), val = int32(-1)]; |
| bool new_state_interleave_0 = const()[name = string("new_state_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1024, 1, 3]> new_state_cast_fp16 = concat(axis = var_1511, interleave = new_state_interleave_0, values = (live_tail_cast_fp16, Bx))[name = string("new_state_cast_fp16")]; |
| tensor<int32, [1]> var_1514_axes_0 = const()[name = string("op_1514_axes_0"), val = tensor<int32, [1]>([0])]; |
| tensor<fp16, [1024, 1, 3]> var_1514_cast_fp16 = squeeze(axes = var_1514_axes_0, x = new_state_cast_fp16)[name = string("op_1514_cast_fp16")]; |
| tensor<int32, [1]> new_slot_axes_0 = const()[name = string("new_slot_axes_0"), val = tensor<int32, [1]>([1])]; |
| tensor<fp16, [1024, 3]> new_slot_cast_fp16 = squeeze(axes = new_slot_axes_0, x = var_1514_cast_fp16)[name = string("new_slot_cast_fp16")]; |
| string conv_out_pad_type_0 = const()[name = string("conv_out_pad_type_0"), val = string("valid")]; |
| int32 conv_out_groups_0 = const()[name = string("conv_out_groups_0"), val = int32(1024)]; |
| tensor<int32, [2]> conv_out_strides_0 = const()[name = string("conv_out_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> conv_out_pad_0 = const()[name = string("conv_out_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> conv_out_dilations_0 = const()[name = string("conv_out_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<fp16, [1024, 1, 1, 3]> layers_3_conv_conv_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1, 1, 3]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53146624))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53148992))))[name = string("layers_3_conv_conv_weight_promoted_to_fp16_palettized")]; |
| tensor<fp16, [1, 1024, 1, 1]> conv_out_cast_fp16 = conv(dilations = conv_out_dilations_0, groups = conv_out_groups_0, pad = conv_out_pad_0, pad_type = conv_out_pad_type_0, strides = conv_out_strides_0, weight = layers_3_conv_conv_weight_promoted_to_fp16_palettized, x = new_state_cast_fp16)[name = string("conv_out_cast_fp16")]; |
| tensor<fp16, [1, 1024, 1, 1]> input_55_cast_fp16 = mul(x = var_1494_1, y = conv_out_cast_fp16)[name = string("input_55_cast_fp16")]; |
| string y_pad_type_0 = const()[name = string("y_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> y_strides_0 = const()[name = string("y_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> y_pad_0 = const()[name = string("y_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> y_dilations_0 = const()[name = string("y_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 y_groups_0 = const()[name = string("y_groups_0"), val = int32(1)]; |
| tensor<fp16, [1024, 1024, 1, 1]> layers_3_conv_out_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor<uint6, [1024, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53153152))), lut = tensor<fp16, [32, 1, 1, 1, 64, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53939648))))[name = string("layers_3_conv_out_proj_weight_promoted_to_fp16_palettized")]; |
| tensor<fp16, [1, 1024, 1, 1]> y_cast_fp16 = conv(dilations = y_dilations_0, groups = y_groups_0, pad = y_pad_0, pad_type = y_pad_type_0, strides = y_strides_0, weight = layers_3_conv_out_proj_weight_promoted_to_fp16_palettized, x = input_55_cast_fp16)[name = string("y_cast_fp16")]; |
| tensor<int32, [1]> var_1542_axes_0 = const()[name = string("op_1542_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1542_cast_fp16 = squeeze(axes = var_1542_axes_0, x = y_cast_fp16)[name = string("op_1542_cast_fp16")]; |
| tensor<int32, [3]> var_1546 = const()[name = string("op_1546"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> op_out_cast_fp16 = transpose(perm = var_1546, x = var_1542_cast_fp16)[name = string("transpose_4")]; |
| tensor<fp16, [1, 1, 1024]> x_23_cast_fp16 = add(x = x_21_cast_fp16, y = op_out_cast_fp16)[name = string("x_23_cast_fp16")]; |
| fp16 const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_1550_cast_fp16 = mul(x = x_23_cast_fp16, y = const_27_promoted_to_fp16)[name = string("op_1550_cast_fp16")]; |
| int32 var_1552 = const()[name = string("op_1552"), val = int32(-1)]; |
| bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_57_cast_fp16 = concat(axis = var_1552, interleave = input_57_interleave_0, values = (x_23_cast_fp16, var_1550_cast_fp16))[name = string("input_57_cast_fp16")]; |
| tensor<int32, [1]> normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1558_to_fp16, x = input_57_cast_fp16)[name = string("normed_29_cast_fp16")]; |
| tensor<int32, [2]> var_1561_split_sizes_0 = const()[name = string("op_1561_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_1561_axis_0 = const()[name = string("op_1561_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_1561_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_1561_cast_fp16_1 = split(axis = var_1561_axis_0, split_sizes = var_1561_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_1561_cast_fp16")]; |
| tensor<fp16, [1024]> layers_3_ffn_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_ffn_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53943808)))]; |
| tensor<fp16, [1, 1, 1024]> normed_31_cast_fp16 = mul(x = var_1561_cast_fp16_0, y = layers_3_ffn_norm_weight_promoted_to_fp16)[name = string("normed_31_cast_fp16")]; |
| tensor<int32, [3]> var_1567 = const()[name = string("op_1567"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> var_1570_axes_0 = const()[name = string("op_1570_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1568_cast_fp16 = transpose(perm = var_1567, x = normed_31_cast_fp16)[name = string("transpose_3")]; |
| tensor<fp16, [1, 1024, 1, 1]> var_1570_cast_fp16 = expand_dims(axes = var_1570_axes_0, x = var_1568_cast_fp16)[name = string("op_1570_cast_fp16")]; |
| string input_61_pad_type_0 = const()[name = string("input_61_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> input_61_strides_0 = const()[name = string("input_61_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> input_61_pad_0 = const()[name = string("input_61_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> input_61_dilations_0 = const()[name = string("input_61_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 input_61_groups_0 = const()[name = string("input_61_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> input_61 = conv(dilations = input_61_dilations_0, groups = input_61_groups_0, pad = input_61_pad_0, pad_type = input_61_pad_type_0, strides = input_61_strides_0, weight = layers_3_feed_forward_w1_weight_palettized, x = var_1570_cast_fp16)[name = string("input_61")]; |
| string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> b_strides_0 = const()[name = string("b_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> b_pad_0 = const()[name = string("b_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 4608, 1, 1]> b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = layers_3_feed_forward_w3_weight_palettized, x = var_1570_cast_fp16)[name = string("b")]; |
| tensor<fp16, [1, 4608, 1, 1]> var_1598 = silu(x = input_61)[name = string("op_1598")]; |
| tensor<fp16, [1, 4608, 1, 1]> input_63 = mul(x = var_1598, y = b)[name = string("input_63")]; |
| string mlp_13_pad_type_0 = const()[name = string("mlp_13_pad_type_0"), val = string("valid")]; |
| tensor<int32, [2]> mlp_13_strides_0 = const()[name = string("mlp_13_strides_0"), val = tensor<int32, [2]>([1, 1])]; |
| tensor<int32, [4]> mlp_13_pad_0 = const()[name = string("mlp_13_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])]; |
| tensor<int32, [2]> mlp_13_dilations_0 = const()[name = string("mlp_13_dilations_0"), val = tensor<int32, [2]>([1, 1])]; |
| int32 mlp_13_groups_0 = const()[name = string("mlp_13_groups_0"), val = int32(1)]; |
| tensor<fp16, [1, 1024, 1, 1]> mlp_13 = conv(dilations = mlp_13_dilations_0, groups = mlp_13_groups_0, pad = mlp_13_pad_0, pad_type = mlp_13_pad_type_0, strides = mlp_13_strides_0, weight = layers_3_feed_forward_w2_weight_palettized, x = input_63)[name = string("mlp_13")]; |
| tensor<int32, [1]> var_1612_axes_0 = const()[name = string("op_1612_axes_0"), val = tensor<int32, [1]>([2])]; |
| tensor<fp16, [1, 1024, 1]> var_1612 = squeeze(axes = var_1612_axes_0, x = mlp_13)[name = string("op_1612")]; |
| tensor<int32, [3]> var_1616 = const()[name = string("op_1616"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [1, 1, 1024]> mlp = transpose(perm = var_1616, x = var_1612)[name = string("transpose_2")]; |
| tensor<fp16, [1, 1, 1024]> x_cast_fp16 = add(x = x_23_cast_fp16, y = mlp)[name = string("x_cast_fp16")]; |
| int32 var_1622_axis_0 = const()[name = string("op_1622_axis_0"), val = int32(0)]; |
| tensor<fp16, [2, 1024, 3]> conv_state_out = stack(axis = var_1622_axis_0, values = (var_801_cast_fp16, new_slot_cast_fp16))[name = string("op_1622_cast_fp16")]; |
| int32 var_1625_axis_0 = const()[name = string("op_1625_axis_0"), val = int32(0)]; |
| tensor<fp16, [2, 1, 512, 1, 1]> var_1625 = stack(axis = var_1625_axis_0, values = (k_slice_1, k_slice))[name = string("op_1625")]; |
| int32 var_1628_axis_0 = const()[name = string("op_1628_axis_0"), val = int32(0)]; |
| tensor<fp16, [2, 1, 512, 1, 1]> var_1628 = stack(axis = var_1628_axis_0, values = (var_282, var_997))[name = string("op_1628")]; |
| int32 var_1630 = const()[name = string("op_1630"), val = int32(0)]; |
| bool var_1631_interleave_0 = const()[name = string("op_1631_interleave_0"), val = bool(false)]; |
| tensor<fp16, [4, 1, 512, 1, 1]> kv_slice_out = concat(axis = var_1630, interleave = var_1631_interleave_0, values = (var_1625, var_1628))[name = string("op_1631")]; |
| fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; |
| tensor<fp16, [1, 1, 1024]> var_1632_cast_fp16 = mul(x = x_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1632_cast_fp16")]; |
| int32 var_1634 = const()[name = string("op_1634"), val = int32(-1)]; |
| bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; |
| tensor<fp16, [1, 1, 2048]> input_65_cast_fp16 = concat(axis = var_1634, interleave = input_65_interleave_0, values = (x_cast_fp16, var_1632_cast_fp16))[name = string("input_65_cast_fp16")]; |
| tensor<int32, [1]> normed_axes_0 = const()[name = string("normed_axes_0"), val = tensor<int32, [1]>([-1])]; |
| fp16 var_1640_to_fp16 = const()[name = string("op_1640_to_fp16"), val = fp16(0x1.5p-17)]; |
| tensor<fp16, [1, 1, 2048]> normed_cast_fp16 = layer_norm(axes = normed_axes_0, epsilon = var_1640_to_fp16, x = input_65_cast_fp16)[name = string("normed_cast_fp16")]; |
| tensor<int32, [2]> var_1643_split_sizes_0 = const()[name = string("op_1643_split_sizes_0"), val = tensor<int32, [2]>([1024, 1024])]; |
| int32 var_1643_axis_0 = const()[name = string("op_1643_axis_0"), val = int32(-1)]; |
| tensor<fp16, [1, 1, 1024]> var_1643_cast_fp16_0, tensor<fp16, [1, 1, 1024]> var_1643_cast_fp16_1 = split(axis = var_1643_axis_0, split_sizes = var_1643_split_sizes_0, x = normed_cast_fp16)[name = string("op_1643_cast_fp16")]; |
| tensor<fp16, [1024]> embedding_norm_weight_promoted_to_fp16 = const()[name = string("embedding_norm_weight_promoted_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53945920)))]; |
| tensor<fp16, [1, 1, 1024]> hidden_states_cast_fp16 = mul(x = var_1643_cast_fp16_0, y = embedding_norm_weight_promoted_to_fp16)[name = string("hidden_states_cast_fp16")]; |
| tensor<int32, [3]> var_1649 = const()[name = string("op_1649"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<fp16, [65536, 1024, 1]> squeeze_0 = const()[name = string("squeeze_0"), val = tensor<fp16, [65536, 1024, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53948032)))]; |
| string var_1670_pad_type_0 = const()[name = string("var_1670_pad_type_0"), val = string("valid")]; |
| int32 var_1670_groups_0 = const()[name = string("var_1670_groups_0"), val = int32(1)]; |
| tensor<int32, [1]> var_1670_strides_0 = const()[name = string("var_1670_strides_0"), val = tensor<int32, [1]>([1])]; |
| tensor<int32, [2]> var_1670_pad_0 = const()[name = string("var_1670_pad_0"), val = tensor<int32, [2]>([0, 0])]; |
| tensor<int32, [1]> var_1670_dilations_0 = const()[name = string("var_1670_dilations_0"), val = tensor<int32, [1]>([1])]; |
| tensor<fp16, [1, 1024, 1]> var_1650_cast_fp16 = transpose(perm = var_1649, x = hidden_states_cast_fp16)[name = string("transpose_1")]; |
| tensor<fp16, [1, 65536, 1]> var_1670 = conv(dilations = var_1670_dilations_0, groups = var_1670_groups_0, pad = var_1670_pad_0, pad_type = var_1670_pad_type_0, strides = var_1670_strides_0, weight = squeeze_0, x = var_1650_cast_fp16)[name = string("var_1670")]; |
| tensor<int32, [3]> var_1674 = const()[name = string("op_1674"), val = tensor<int32, [3]>([0, 2, 1])]; |
| tensor<int32, [1]> logits_2d_axes_0 = const()[name = string("logits_2d_axes_0"), val = tensor<int32, [1]>([0])]; |
| tensor<fp16, [1, 1, 65536]> logits = transpose(perm = var_1674, x = var_1670)[name = string("transpose_0")]; |
| tensor<fp16, [1, 65536]> logits_2d = squeeze(axes = logits_2d_axes_0, x = logits)[name = string("logits_2d")]; |
| int32 token_id_axis_0 = const()[name = string("token_id_axis_0"), val = int32(-1)]; |
| bool token_id_keep_dims_0 = const()[name = string("token_id_keep_dims_0"), val = bool(false)]; |
| string token_id_output_dtype_0 = const()[name = string("token_id_output_dtype_0"), val = string("int32")]; |
| tensor<int32, [1]> token_id = reduce_argmax(axis = token_id_axis_0, keep_dims = token_id_keep_dims_0, output_dtype = token_id_output_dtype_0, x = logits_2d)[name = string("token_id")]; |
| tensor<int32, [1]> var_1682_axes_0 = const()[name = string("op_1682_axes_0"), val = tensor<int32, [1]>([-1])]; |
| tensor<int32, [1, 1]> var_1682 = expand_dims(axes = var_1682_axes_0, x = token_id)[name = string("op_1682")]; |
| int32 var_1683 = const()[name = string("op_1683"), val = int32(-1)]; |
| bool var_1685_validate_indices_0 = const()[name = string("op_1685_validate_indices_0"), val = bool(false)]; |
| tensor<fp16, [1, 1]> var_1685 = gather_along_axis(axis = var_1683, indices = var_1682, validate_indices = var_1685_validate_indices_0, x = logits_2d)[name = string("op_1685")]; |
| tensor<int32, [1]> var_1687_axes_0 = const()[name = string("op_1687_axes_0"), val = tensor<int32, [1]>([-1])]; |
| tensor<fp16, [1]> token_logit = squeeze(axes = var_1687_axes_0, x = var_1685)[name = string("op_1687")]; |
| tensor<fp16, [1, 1, 1024, 1]> update_mask_tmp = identity(x = update_mask)[name = string("update_mask_tmp")]; |
| } -> (token_id, token_logit, kv_slice_out, conv_state_out); |
| } |