program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}})] { func main(tensor causal_mask, tensor image_embedding, tensor input_ids, state> kv_cache_0, tensor per_layer_combined, tensor position_ids, tensor update_mask) { tensor sin_full_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262272))))[name = string("sin_full_palettized")]; tensor cos_full_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525568))))[name = string("cos_full_palettized")]; tensor sin_sliding_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(657792))))[name = string("sin_sliding_palettized")]; tensor cos_sliding_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(658880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(790016))))[name = string("cos_sliding_palettized")]; tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(791104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2364032))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2366144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2562816))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2563136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2759808))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2760128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7478784))))[name = string("layers_0_mlp_gate_proj_weight_palettized")]; tensor layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7484992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12203648))))[name = string("layers_0_mlp_up_proj_weight_palettized")]; tensor layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12209856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16928512))))[name = string("layers_0_mlp_down_proj_weight_palettized")]; tensor layers_0_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16930112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17126784))))[name = string("layers_0_per_layer_input_gate_weight_palettized")]; tensor layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17127104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18700032))))[name = string("layers_1_self_attn_q_proj_weight_palettized")]; tensor layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18702144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18898816))))[name = string("layers_1_self_attn_k_proj_weight_palettized")]; tensor layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18899136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19095808))))[name = string("layers_1_self_attn_v_proj_weight_palettized")]; tensor layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19096128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23814784))))[name = string("layers_1_mlp_gate_proj_weight_palettized")]; tensor layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23820992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28539648))))[name = string("layers_1_mlp_up_proj_weight_palettized")]; tensor layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28545856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33264512))))[name = string("layers_1_mlp_down_proj_weight_palettized")]; tensor layers_1_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33266112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33462784))))[name = string("layers_1_per_layer_input_gate_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33463104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35036032))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35038144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35234816))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35235136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35431808))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35432128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40150784))))[name = string("layers_2_mlp_gate_proj_weight_palettized")]; tensor layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40156992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44875648))))[name = string("layers_2_mlp_up_proj_weight_palettized")]; tensor layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44881856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49600512))))[name = string("layers_2_mlp_down_proj_weight_palettized")]; tensor layers_2_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49602112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49798784))))[name = string("layers_2_per_layer_input_gate_weight_palettized")]; tensor layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49799104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51372032))))[name = string("layers_3_self_attn_q_proj_weight_palettized")]; tensor layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51374144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51570816))))[name = string("layers_3_self_attn_k_proj_weight_palettized")]; tensor layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51571136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51767808))))[name = string("layers_3_self_attn_v_proj_weight_palettized")]; tensor layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51768128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56486784))))[name = string("layers_3_mlp_gate_proj_weight_palettized")]; tensor layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56492992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61211648))))[name = string("layers_3_mlp_up_proj_weight_palettized")]; tensor layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61217856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65936512))))[name = string("layers_3_mlp_down_proj_weight_palettized")]; tensor layers_3_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65938112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66134784))))[name = string("layers_3_per_layer_input_gate_weight_palettized")]; tensor layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66135104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69280896))))[name = string("layers_4_self_attn_q_proj_weight_palettized")]; tensor layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69285056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69678336))))[name = string("layers_4_self_attn_k_proj_weight_palettized")]; tensor layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69678912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70072192))))[name = string("layers_4_self_attn_v_proj_weight_palettized")]; tensor layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70072768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74791424))))[name = string("layers_4_mlp_gate_proj_weight_palettized")]; tensor layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74797632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(79516288))))[name = string("layers_4_mlp_up_proj_weight_palettized")]; tensor layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(79522496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84241152))))[name = string("layers_4_mlp_down_proj_weight_palettized")]; tensor layers_4_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84242752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84439424))))[name = string("layers_4_per_layer_input_gate_weight_palettized")]; tensor layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84439744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86012672))))[name = string("layers_5_self_attn_q_proj_weight_palettized")]; tensor layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86014784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86211456))))[name = string("layers_5_self_attn_k_proj_weight_palettized")]; tensor layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86211776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86408448))))[name = string("layers_5_self_attn_v_proj_weight_palettized")]; tensor layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86408768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91127424))))[name = string("layers_5_mlp_gate_proj_weight_palettized")]; tensor layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91133632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95852288))))[name = string("layers_5_mlp_up_proj_weight_palettized")]; tensor layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95858496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(100577152))))[name = string("layers_5_mlp_down_proj_weight_palettized")]; tensor layers_5_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(100578752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(100775424))))[name = string("layers_5_per_layer_input_gate_weight_palettized")]; tensor layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(100775744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102348672))))[name = string("layers_6_self_attn_q_proj_weight_palettized")]; tensor layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102350784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102547456))))[name = string("layers_6_self_attn_k_proj_weight_palettized")]; tensor layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102547776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102744448))))[name = string("layers_6_self_attn_v_proj_weight_palettized")]; tensor layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102744768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(107463424))))[name = string("layers_6_mlp_gate_proj_weight_palettized")]; tensor layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(107469632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112188288))))[name = string("layers_6_mlp_up_proj_weight_palettized")]; tensor layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112194496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116913152))))[name = string("layers_6_mlp_down_proj_weight_palettized")]; tensor layers_6_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116914752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117111424))))[name = string("layers_6_per_layer_input_gate_weight_palettized")]; tensor layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117111744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118684672))))[name = string("layers_7_self_attn_q_proj_weight_palettized")]; tensor layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118686784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118883456))))[name = string("layers_7_self_attn_k_proj_weight_palettized")]; tensor layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118883776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(119080448))))[name = string("layers_7_self_attn_v_proj_weight_palettized")]; tensor layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(119080768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123799424))))[name = string("layers_7_mlp_gate_proj_weight_palettized")]; tensor layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123805632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128524288))))[name = string("layers_7_mlp_up_proj_weight_palettized")]; tensor layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128530496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133249152))))[name = string("layers_7_mlp_down_proj_weight_palettized")]; tensor layers_7_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133250752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133447424))))[name = string("layers_7_per_layer_input_gate_weight_palettized")]; tensor layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133447744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135020672))))[name = string("layers_8_self_attn_q_proj_weight_palettized")]; tensor layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135022784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135219456))))[name = string("layers_8_self_attn_k_proj_weight_palettized")]; tensor layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135219776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135416448))))[name = string("layers_8_self_attn_v_proj_weight_palettized")]; tensor layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135416768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(140135424))))[name = string("layers_8_mlp_gate_proj_weight_palettized")]; tensor layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(140141632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144860288))))[name = string("layers_8_mlp_up_proj_weight_palettized")]; tensor layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144866496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149585152))))[name = string("layers_8_mlp_down_proj_weight_palettized")]; tensor layers_8_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149586752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149783424))))[name = string("layers_8_per_layer_input_gate_weight_palettized")]; tensor layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149783744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(152929536))))[name = string("layers_9_self_attn_q_proj_weight_palettized")]; tensor layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(152933696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153326976))))[name = string("layers_9_self_attn_k_proj_weight_palettized")]; tensor layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153327552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153720832))))[name = string("layers_9_self_attn_v_proj_weight_palettized")]; tensor layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153721408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158440064))))[name = string("layers_9_mlp_gate_proj_weight_palettized")]; tensor layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158446272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163164928))))[name = string("layers_9_mlp_up_proj_weight_palettized")]; tensor layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163171136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(167889792))))[name = string("layers_9_mlp_down_proj_weight_palettized")]; tensor layers_9_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(167891392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168088064))))[name = string("layers_9_per_layer_input_gate_weight_palettized")]; tensor layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168088384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169661312))))[name = string("layers_10_self_attn_q_proj_weight_palettized")]; tensor layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169663424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169860096))))[name = string("layers_10_self_attn_k_proj_weight_palettized")]; tensor layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169860416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170057088))))[name = string("layers_10_self_attn_v_proj_weight_palettized")]; tensor layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170057408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174776064))))[name = string("layers_10_mlp_gate_proj_weight_palettized")]; tensor layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174782272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(179500928))))[name = string("layers_10_mlp_up_proj_weight_palettized")]; tensor layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(179507136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184225792))))[name = string("layers_10_mlp_down_proj_weight_palettized")]; tensor layers_10_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184227392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184424064))))[name = string("layers_10_per_layer_input_gate_weight_palettized")]; tensor layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184424384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(185997312))))[name = string("layers_11_self_attn_q_proj_weight_palettized")]; tensor layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(185999424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(186196096))))[name = string("layers_11_self_attn_k_proj_weight_palettized")]; tensor layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(186196416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(186393088))))[name = string("layers_11_self_attn_v_proj_weight_palettized")]; tensor layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(186393408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(191112064))))[name = string("layers_11_mlp_gate_proj_weight_palettized")]; tensor layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(191118272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195836928))))[name = string("layers_11_mlp_up_proj_weight_palettized")]; tensor layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195843136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200561792))))[name = string("layers_11_mlp_down_proj_weight_palettized")]; tensor layers_11_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200563392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200760064))))[name = string("layers_11_per_layer_input_gate_weight_palettized")]; tensor layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200760384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202333312))))[name = string("layers_12_self_attn_q_proj_weight_palettized")]; tensor layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202335424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202532096))))[name = string("layers_12_self_attn_k_proj_weight_palettized")]; tensor layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202532416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202729088))))[name = string("layers_12_self_attn_v_proj_weight_palettized")]; tensor layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(202729408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207448064))))[name = string("layers_12_mlp_gate_proj_weight_palettized")]; tensor layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(207454272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(212172928))))[name = string("layers_12_mlp_up_proj_weight_palettized")]; tensor layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(212179136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216897792))))[name = string("layers_12_mlp_down_proj_weight_palettized")]; tensor layers_12_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216899392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217096064))))[name = string("layers_12_per_layer_input_gate_weight_palettized")]; tensor layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217096384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218669312))))[name = string("layers_13_self_attn_q_proj_weight_palettized")]; tensor layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218671424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218868096))))[name = string("layers_13_self_attn_k_proj_weight_palettized")]; tensor layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218868416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219065088))))[name = string("layers_13_self_attn_v_proj_weight_palettized")]; tensor layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(219065408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223784064))))[name = string("layers_13_mlp_gate_proj_weight_palettized")]; tensor layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223790272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228508928))))[name = string("layers_13_mlp_up_proj_weight_palettized")]; tensor layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228515136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233233792))))[name = string("layers_13_mlp_down_proj_weight_palettized")]; tensor layers_13_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233235392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233432064))))[name = string("layers_13_per_layer_input_gate_weight_palettized")]; tensor layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(233432384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236578176))))[name = string("layers_14_self_attn_q_proj_weight_palettized")]; tensor layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236582336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236975616))))[name = string("layers_14_self_attn_k_proj_weight_palettized")]; tensor layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236976192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237369472))))[name = string("layers_14_self_attn_v_proj_weight_palettized")]; tensor layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(237370048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242088704))))[name = string("layers_14_mlp_gate_proj_weight_palettized")]; tensor layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242094912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(246813568))))[name = string("layers_14_mlp_up_proj_weight_palettized")]; tensor layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(246819776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251538432))))[name = string("layers_14_mlp_down_proj_weight_palettized")]; tensor layers_14_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251540032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251736704))))[name = string("layers_14_per_layer_input_gate_weight_palettized")]; tensor layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251737024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253309952))))[name = string("layers_15_self_attn_q_proj_weight_palettized")]; tensor layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(253312064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262749312))))[name = string("layers_15_mlp_gate_proj_weight_palettized")]; tensor layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262761664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272198912))))[name = string("layers_15_mlp_up_proj_weight_palettized")]; tensor layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272211264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(281648512))))[name = string("layers_15_mlp_down_proj_weight_palettized")]; tensor layers_15_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(281650112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(281846784))))[name = string("layers_15_per_layer_input_gate_weight_palettized")]; tensor layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(281847104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283420032))))[name = string("layers_16_self_attn_q_proj_weight_palettized")]; tensor layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283422144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292859392))))[name = string("layers_16_mlp_gate_proj_weight_palettized")]; tensor layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292871744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302308992))))[name = string("layers_16_mlp_up_proj_weight_palettized")]; tensor layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302321344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311758592))))[name = string("layers_16_mlp_down_proj_weight_palettized")]; tensor layers_16_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311760192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311956864))))[name = string("layers_16_per_layer_input_gate_weight_palettized")]; tensor layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311957184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313530112))))[name = string("layers_17_self_attn_q_proj_weight_palettized")]; tensor layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313532224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322969472))))[name = string("layers_17_mlp_gate_proj_weight_palettized")]; tensor layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322981824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332419072))))[name = string("layers_17_mlp_up_proj_weight_palettized")]; tensor layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332431424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(341868672))))[name = string("layers_17_mlp_down_proj_weight_palettized")]; tensor layers_17_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(341870272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342066944))))[name = string("layers_17_per_layer_input_gate_weight_palettized")]; tensor layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342067264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343640192))))[name = string("layers_18_self_attn_q_proj_weight_palettized")]; tensor layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343642304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353079552))))[name = string("layers_18_mlp_gate_proj_weight_palettized")]; tensor layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353091904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(362529152))))[name = string("layers_18_mlp_up_proj_weight_palettized")]; tensor layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(362541504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371978752))))[name = string("layers_18_mlp_down_proj_weight_palettized")]; tensor layers_18_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371980352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372177024))))[name = string("layers_18_per_layer_input_gate_weight_palettized")]; tensor layers_19_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372177344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(375323136))))[name = string("layers_19_self_attn_q_proj_weight_palettized")]; tensor layers_19_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(375327296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(384764544))))[name = string("layers_19_mlp_gate_proj_weight_palettized")]; tensor layers_19_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(384776896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(394214144))))[name = string("layers_19_mlp_up_proj_weight_palettized")]; tensor layers_19_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(394226496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403663744))))[name = string("layers_19_mlp_down_proj_weight_palettized")]; tensor layers_19_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403665344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403862016))))[name = string("layers_19_per_layer_input_gate_weight_palettized")]; tensor layers_20_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403862336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(405435264))))[name = string("layers_20_self_attn_q_proj_weight_palettized")]; tensor layers_20_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(405437376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(414874624))))[name = string("layers_20_mlp_gate_proj_weight_palettized")]; tensor layers_20_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(414886976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(424324224))))[name = string("layers_20_mlp_up_proj_weight_palettized")]; tensor layers_20_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(424336576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(433773824))))[name = string("layers_20_mlp_down_proj_weight_palettized")]; tensor layers_20_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(433775424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(433972096))))[name = string("layers_20_per_layer_input_gate_weight_palettized")]; tensor layers_21_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(433972416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(435545344))))[name = string("layers_21_self_attn_q_proj_weight_palettized")]; tensor layers_21_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(435547456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(444984704))))[name = string("layers_21_mlp_gate_proj_weight_palettized")]; tensor layers_21_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(444997056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454434304))))[name = string("layers_21_mlp_up_proj_weight_palettized")]; tensor layers_21_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454446656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(463883904))))[name = string("layers_21_mlp_down_proj_weight_palettized")]; tensor layers_21_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(463885504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464082176))))[name = string("layers_21_per_layer_input_gate_weight_palettized")]; tensor layers_22_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464082496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(465655424))))[name = string("layers_22_self_attn_q_proj_weight_palettized")]; tensor layers_22_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(465657536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475094784))))[name = string("layers_22_mlp_gate_proj_weight_palettized")]; tensor layers_22_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475107136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484544384))))[name = string("layers_22_mlp_up_proj_weight_palettized")]; tensor layers_22_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484556736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(493993984))))[name = string("layers_22_mlp_down_proj_weight_palettized")]; tensor layers_22_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(493995584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494192256))))[name = string("layers_22_per_layer_input_gate_weight_palettized")]; tensor layers_23_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494192576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(495765504))))[name = string("layers_23_self_attn_q_proj_weight_palettized")]; tensor layers_23_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(495767616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(505204864))))[name = string("layers_23_mlp_gate_proj_weight_palettized")]; tensor layers_23_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(505217216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(514654464))))[name = string("layers_23_mlp_up_proj_weight_palettized")]; tensor layers_23_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(514666816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524104064))))[name = string("layers_23_mlp_down_proj_weight_palettized")]; tensor layers_23_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524105664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524302336))))[name = string("layers_23_per_layer_input_gate_weight_palettized")]; tensor layers_24_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524302656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(527448448))))[name = string("layers_24_self_attn_q_proj_weight_palettized")]; tensor layers_24_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(527452608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536889856))))[name = string("layers_24_mlp_gate_proj_weight_palettized")]; tensor layers_24_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536902208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546339456))))[name = string("layers_24_mlp_up_proj_weight_palettized")]; tensor layers_24_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546351808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555789056))))[name = string("layers_24_mlp_down_proj_weight_palettized")]; tensor layers_24_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555790656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555987328))))[name = string("layers_24_per_layer_input_gate_weight_palettized")]; tensor layers_25_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555987648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557560576))))[name = string("layers_25_self_attn_q_proj_weight_palettized")]; tensor layers_25_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557562688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566999936))))[name = string("layers_25_mlp_gate_proj_weight_palettized")]; tensor layers_25_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567012288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576449536))))[name = string("layers_25_mlp_up_proj_weight_palettized")]; tensor layers_25_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576461888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(585899136))))[name = string("layers_25_mlp_down_proj_weight_palettized")]; tensor layers_25_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(585900736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(586097408))))[name = string("layers_25_per_layer_input_gate_weight_palettized")]; tensor layers_26_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(586097728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(587670656))))[name = string("layers_26_self_attn_q_proj_weight_palettized")]; tensor layers_26_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(587672768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(597110016))))[name = string("layers_26_mlp_gate_proj_weight_palettized")]; tensor layers_26_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(597122368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(606559616))))[name = string("layers_26_mlp_up_proj_weight_palettized")]; tensor layers_26_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(606571968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(616009216))))[name = string("layers_26_mlp_down_proj_weight_palettized")]; tensor layers_26_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(616010816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(616207488))))[name = string("layers_26_per_layer_input_gate_weight_palettized")]; tensor layers_27_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(616207808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617780736))))[name = string("layers_27_self_attn_q_proj_weight_palettized")]; tensor layers_27_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617782848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(627220096))))[name = string("layers_27_mlp_gate_proj_weight_palettized")]; tensor layers_27_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(627232448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636669696))))[name = string("layers_27_mlp_up_proj_weight_palettized")]; tensor layers_27_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636682048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(646119296))))[name = string("layers_27_mlp_down_proj_weight_palettized")]; tensor layers_27_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(646120896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(646317568))))[name = string("layers_27_per_layer_input_gate_weight_palettized")]; tensor layers_28_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(646317888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647890816))))[name = string("layers_28_self_attn_q_proj_weight_palettized")]; tensor layers_28_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647892928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(657330176))))[name = string("layers_28_mlp_gate_proj_weight_palettized")]; tensor layers_28_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(657342528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(666779776))))[name = string("layers_28_mlp_up_proj_weight_palettized")]; tensor layers_28_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(666792128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676229376))))[name = string("layers_28_mlp_down_proj_weight_palettized")]; tensor layers_28_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676230976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676427648))))[name = string("layers_28_per_layer_input_gate_weight_palettized")]; tensor layers_29_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676427968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(679573760))))[name = string("layers_29_self_attn_q_proj_weight_palettized")]; tensor layers_29_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(679577920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689015168))))[name = string("layers_29_mlp_gate_proj_weight_palettized")]; tensor layers_29_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689027520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698464768))))[name = string("layers_29_mlp_up_proj_weight_palettized")]; tensor layers_29_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698477120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707914368))))[name = string("layers_29_mlp_down_proj_weight_palettized")]; tensor layers_29_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707915968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(708112640))))[name = string("layers_29_per_layer_input_gate_weight_palettized")]; tensor layers_30_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(708112960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(709685888))))[name = string("layers_30_self_attn_q_proj_weight_palettized")]; tensor layers_30_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(709688000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719125248))))[name = string("layers_30_mlp_gate_proj_weight_palettized")]; tensor layers_30_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719137600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728574848))))[name = string("layers_30_mlp_up_proj_weight_palettized")]; tensor layers_30_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728587200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(738024448))))[name = string("layers_30_mlp_down_proj_weight_palettized")]; tensor layers_30_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(738026048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(738222720))))[name = string("layers_30_per_layer_input_gate_weight_palettized")]; tensor layers_31_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(738223040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(739795968))))[name = string("layers_31_self_attn_q_proj_weight_palettized")]; tensor layers_31_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(739798080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(749235328))))[name = string("layers_31_mlp_gate_proj_weight_palettized")]; tensor layers_31_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(749247680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(758684928))))[name = string("layers_31_mlp_up_proj_weight_palettized")]; tensor layers_31_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(758697280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(768134528))))[name = string("layers_31_mlp_down_proj_weight_palettized")]; tensor layers_31_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(768136128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(768332800))))[name = string("layers_31_per_layer_input_gate_weight_palettized")]; tensor layers_32_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(768333120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(769906048))))[name = string("layers_32_self_attn_q_proj_weight_palettized")]; tensor layers_32_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(769908160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(779345408))))[name = string("layers_32_mlp_gate_proj_weight_palettized")]; tensor layers_32_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(779357760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(788795008))))[name = string("layers_32_mlp_up_proj_weight_palettized")]; tensor layers_32_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(788807360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(798244608))))[name = string("layers_32_mlp_down_proj_weight_palettized")]; tensor layers_32_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(798246208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(798442880))))[name = string("layers_32_per_layer_input_gate_weight_palettized")]; tensor layers_33_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(798443200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(800016128))))[name = string("layers_33_self_attn_q_proj_weight_palettized")]; tensor layers_33_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(800018240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(809455488))))[name = string("layers_33_mlp_gate_proj_weight_palettized")]; tensor layers_33_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(809467840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(818905088))))[name = string("layers_33_mlp_up_proj_weight_palettized")]; tensor layers_33_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(818917440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(828354688))))[name = string("layers_33_mlp_down_proj_weight_palettized")]; tensor layers_33_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(828356288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(828552960))))[name = string("layers_33_per_layer_input_gate_weight_palettized")]; tensor layers_34_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(828553280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(831699072))))[name = string("layers_34_self_attn_q_proj_weight_palettized")]; tensor layers_34_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(831703232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(841140480))))[name = string("layers_34_mlp_gate_proj_weight_palettized")]; tensor layers_34_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(841152832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(850590080))))[name = string("layers_34_mlp_up_proj_weight_palettized")]; tensor layers_34_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(850602432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860039680))))[name = string("layers_34_mlp_down_proj_weight_palettized")]; tensor layers_34_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860041280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860237952))))[name = string("layers_34_per_layer_input_gate_weight_palettized")]; int32 var_1878_batch_dims_0 = const()[name = string("op_1878_batch_dims_0"), val = int32(0)]; bool var_1878_validate_indices_0 = const()[name = string("op_1878_validate_indices_0"), val = bool(false)]; tensor embed_tokens_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860238272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1061564928))))[name = string("embed_tokens_weight_to_fp16_palettized")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = input_ids, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_70 = const()[name = string("slice_by_index_70"), val = int32(262144)]; tensor add_0 = add(x = input_ids, y = slice_by_index_70)[name = string("add_0")]; tensor select_0 = select(a = input_ids, b = add_0, cond = greater_equal_0)[name = string("select_0")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor greater_equal_0_1 = greater_equal(x = select_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(262144)]; tensor add_0_1 = add(x = select_0, y = slice_by_index_0)[name = string("add_0_1")]; tensor select_0_1 = select(a = select_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_1878_cast_fp16_axis_0 = const()[name = string("op_1878_cast_fp16_axis_0"), val = int32(0)]; tensor op_1878_cast_fp16 = gather(axis = op_1878_cast_fp16_axis_0, batch_dims = var_1878_batch_dims_0, indices = select_0_1, validate_indices = var_1878_validate_indices_0, x = embed_tokens_weight_to_fp16_palettized)[name = string("op_1878_cast_fp16")]; fp16 const_0 = const()[name = string("const_0"), val = fp16(0x1.398p+5)]; tensor text_embedding = mul(x = op_1878_cast_fp16, y = const_0)[name = string("text_embedding")]; tensor var_1893_cast_fp16 = abs(x = image_embedding)[name = string("op_1893_cast_fp16")]; tensor var_1898_axes_0 = const()[name = string("op_1898_axes_0"), val = tensor([-1])]; bool var_1898_keep_dims_0 = const()[name = string("op_1898_keep_dims_0"), val = bool(true)]; tensor var_1898_cast_fp16 = reduce_sum(axes = var_1898_axes_0, keep_dims = var_1898_keep_dims_0, x = var_1893_cast_fp16)[name = string("op_1898_cast_fp16")]; fp16 var_1899_promoted_to_fp16 = const()[name = string("op_1899_promoted_to_fp16"), val = fp16(0x0p+0)]; tensor var_1900_cast_fp16 = greater(x = var_1898_cast_fp16, y = var_1899_promoted_to_fp16)[name = string("op_1900_cast_fp16")]; string is_image_dtype_0 = const()[name = string("is_image_dtype_0"), val = string("fp16")]; fp16 var_1906_promoted = const()[name = string("op_1906_promoted"), val = fp16(0x1p+0)]; tensor is_image = cast(dtype = is_image_dtype_0, x = var_1900_cast_fp16)[name = string("cast_1")]; tensor var_1908 = sub(x = var_1906_promoted, y = is_image)[name = string("op_1908")]; tensor var_1909 = mul(x = text_embedding, y = var_1908)[name = string("op_1909")]; tensor var_1910_cast_fp16 = mul(x = image_embedding, y = is_image)[name = string("op_1910_cast_fp16")]; tensor x_1_cast_fp16 = add(x = var_1909, y = var_1910_cast_fp16)[name = string("x_1_cast_fp16")]; int32 var_1913 = const()[name = string("op_1913"), val = int32(0)]; int32 var_1914_batch_dims_0 = const()[name = string("op_1914_batch_dims_0"), val = int32(0)]; bool var_1914_validate_indices_0 = const()[name = string("op_1914_validate_indices_0"), val = bool(false)]; string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_0")]; tensor var_1914_cast_uint16 = gather(axis = var_1913, batch_dims = var_1914_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_1914_validate_indices_0, x = cos_sliding_palettized)[name = string("op_1914_cast_uint16")]; tensor var_1916_axes_0 = const()[name = string("op_1916_axes_0"), val = tensor([0])]; tensor var_1916 = expand_dims(axes = var_1916_axes_0, x = var_1914_cast_uint16)[name = string("op_1916")]; tensor cos_1_axes_0 = const()[name = string("cos_1_axes_0"), val = tensor([0])]; tensor cos_1 = expand_dims(axes = cos_1_axes_0, x = var_1916)[name = string("cos_1")]; int32 var_1919 = const()[name = string("op_1919"), val = int32(0)]; int32 var_1920_batch_dims_0 = const()[name = string("op_1920_batch_dims_0"), val = int32(0)]; bool var_1920_validate_indices_0 = const()[name = string("op_1920_validate_indices_0"), val = bool(false)]; tensor var_1920_cast_uint16 = gather(axis = var_1919, batch_dims = var_1920_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_1920_validate_indices_0, x = sin_sliding_palettized)[name = string("op_1920_cast_uint16")]; tensor var_1922_axes_0 = const()[name = string("op_1922_axes_0"), val = tensor([0])]; tensor var_1922 = expand_dims(axes = var_1922_axes_0, x = var_1920_cast_uint16)[name = string("op_1922")]; tensor sin_1_axes_0 = const()[name = string("sin_1_axes_0"), val = tensor([0])]; tensor sin_1 = expand_dims(axes = sin_1_axes_0, x = var_1922)[name = string("sin_1")]; int32 var_1925 = const()[name = string("op_1925"), val = int32(0)]; int32 var_1926_batch_dims_0 = const()[name = string("op_1926_batch_dims_0"), val = int32(0)]; bool var_1926_validate_indices_0 = const()[name = string("op_1926_validate_indices_0"), val = bool(false)]; tensor var_1926_cast_uint16 = gather(axis = var_1925, batch_dims = var_1926_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_1926_validate_indices_0, x = cos_full_palettized)[name = string("op_1926_cast_uint16")]; tensor var_1928_axes_0 = const()[name = string("op_1928_axes_0"), val = tensor([0])]; tensor var_1928 = expand_dims(axes = var_1928_axes_0, x = var_1926_cast_uint16)[name = string("op_1928")]; tensor cos_axes_0 = const()[name = string("cos_axes_0"), val = tensor([0])]; tensor cos = expand_dims(axes = cos_axes_0, x = var_1928)[name = string("cos")]; int32 var_1931 = const()[name = string("op_1931"), val = int32(0)]; int32 var_1932_batch_dims_0 = const()[name = string("op_1932_batch_dims_0"), val = int32(0)]; bool var_1932_validate_indices_0 = const()[name = string("op_1932_validate_indices_0"), val = bool(false)]; tensor var_1932_cast_uint16 = gather(axis = var_1931, batch_dims = var_1932_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_1932_validate_indices_0, x = sin_full_palettized)[name = string("op_1932_cast_uint16")]; tensor var_1934_axes_0 = const()[name = string("op_1934_axes_0"), val = tensor([0])]; tensor var_1934 = expand_dims(axes = var_1934_axes_0, x = var_1932_cast_uint16)[name = string("op_1934")]; tensor sin_axes_0 = const()[name = string("sin_axes_0"), val = tensor([0])]; tensor sin = expand_dims(axes = sin_axes_0, x = var_1934)[name = string("sin")]; int32 var_1941 = const()[name = string("op_1941"), val = int32(-1)]; fp16 const_1_promoted_to_fp16 = const()[name = string("const_1_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1947_cast_fp16 = mul(x = x_1_cast_fp16, y = const_1_promoted_to_fp16)[name = string("op_1947_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_1941, interleave = input_1_interleave_0, values = (x_1_cast_fp16, var_1947_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_1939_to_fp16 = const()[name = string("op_1939_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_1939_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_1952_split_sizes_0 = const()[name = string("op_1952_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1952_axis_0 = const()[name = string("op_1952_axis_0"), val = int32(-1)]; tensor var_1952_cast_fp16_0, tensor var_1952_cast_fp16_1 = split(axis = var_1952_axis_0, split_sizes = var_1952_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_1952_cast_fp16")]; tensor const_2_to_fp16 = const()[name = string("const_2_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1061827136)))]; tensor var_1955_cast_fp16 = mul(x = var_1952_cast_fp16_0, y = const_2_to_fp16)[name = string("op_1955_cast_fp16")]; tensor var_1960 = const()[name = string("op_1960"), val = tensor([0, 2, 1])]; tensor var_1963_axes_0 = const()[name = string("op_1963_axes_0"), val = tensor([2])]; tensor var_1961 = transpose(perm = var_1960, x = var_1955_cast_fp16)[name = string("transpose_366")]; tensor var_1963 = expand_dims(axes = var_1963_axes_0, x = var_1961)[name = string("op_1963")]; string var_1979_pad_type_0 = const()[name = string("op_1979_pad_type_0"), val = string("valid")]; tensor var_1979_strides_0 = const()[name = string("op_1979_strides_0"), val = tensor([1, 1])]; tensor var_1979_pad_0 = const()[name = string("op_1979_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1979_dilations_0 = const()[name = string("op_1979_dilations_0"), val = tensor([1, 1])]; int32 var_1979_groups_0 = const()[name = string("op_1979_groups_0"), val = int32(1)]; tensor var_1979 = conv(dilations = var_1979_dilations_0, groups = var_1979_groups_0, pad = var_1979_pad_0, pad_type = var_1979_pad_type_0, strides = var_1979_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_1963)[name = string("op_1979")]; tensor var_1984 = const()[name = string("op_1984"), val = tensor([1, 8, 256, 1])]; tensor var_1985 = reshape(shape = var_1984, x = var_1979)[name = string("op_1985")]; tensor var_1990 = const()[name = string("op_1990"), val = tensor([0, 1, 3, 2])]; tensor var_2000 = const()[name = string("op_2000"), val = tensor([1, 8, 256])]; tensor var_1991 = transpose(perm = var_1990, x = var_1985)[name = string("transpose_365")]; tensor x_5 = reshape(shape = var_2000, x = var_1991)[name = string("x_5")]; int32 var_2006 = const()[name = string("op_2006"), val = int32(-1)]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2012_cast_fp16 = mul(x = x_5, y = const_3_promoted_to_fp16)[name = string("op_2012_cast_fp16")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5_cast_fp16 = concat(axis = var_2006, interleave = input_5_interleave_0, values = (x_5, var_2012_cast_fp16))[name = string("input_5_cast_fp16")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_2004_to_fp16 = const()[name = string("op_2004_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_2004_to_fp16, x = input_5_cast_fp16)[name = string("normed_5_cast_fp16")]; tensor var_2017_split_sizes_0 = const()[name = string("op_2017_split_sizes_0"), val = tensor([256, 256])]; int32 var_2017_axis_0 = const()[name = string("op_2017_axis_0"), val = int32(-1)]; tensor var_2017_cast_fp16_0, tensor var_2017_cast_fp16_1 = split(axis = var_2017_axis_0, split_sizes = var_2017_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_2017_cast_fp16")]; tensor const_4_to_fp16 = const()[name = string("const_4_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1061830272)))]; tensor var_2020_cast_fp16 = mul(x = var_2017_cast_fp16_0, y = const_4_to_fp16)[name = string("op_2020_cast_fp16")]; tensor var_2026 = const()[name = string("op_2026"), val = tensor([1, 8, 1, 256])]; tensor q_3 = reshape(shape = var_2026, x = var_2020_cast_fp16)[name = string("q_3")]; tensor var_2028 = mul(x = q_3, y = cos_1)[name = string("op_2028")]; tensor var_2029_split_sizes_0 = const()[name = string("op_2029_split_sizes_0"), val = tensor([128, 128])]; int32 var_2029_axis_0 = const()[name = string("op_2029_axis_0"), val = int32(-1)]; tensor var_2029_0, tensor var_2029_1 = split(axis = var_2029_axis_0, split_sizes = var_2029_split_sizes_0, x = q_3)[name = string("op_2029")]; fp16 const_5_promoted = const()[name = string("const_5_promoted"), val = fp16(-0x1p+0)]; tensor var_2031 = mul(x = var_2029_1, y = const_5_promoted)[name = string("op_2031")]; int32 var_2033 = const()[name = string("op_2033"), val = int32(-1)]; bool var_2034_interleave_0 = const()[name = string("op_2034_interleave_0"), val = bool(false)]; tensor var_2034 = concat(axis = var_2033, interleave = var_2034_interleave_0, values = (var_2031, var_2029_0))[name = string("op_2034")]; tensor var_2035 = mul(x = var_2034, y = sin_1)[name = string("op_2035")]; tensor q_7 = add(x = var_2028, y = var_2035)[name = string("q_7")]; string var_2048_pad_type_0 = const()[name = string("op_2048_pad_type_0"), val = string("valid")]; tensor var_2048_strides_0 = const()[name = string("op_2048_strides_0"), val = tensor([1, 1])]; tensor var_2048_pad_0 = const()[name = string("op_2048_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2048_dilations_0 = const()[name = string("op_2048_dilations_0"), val = tensor([1, 1])]; int32 var_2048_groups_0 = const()[name = string("op_2048_groups_0"), val = int32(1)]; tensor var_2048 = conv(dilations = var_2048_dilations_0, groups = var_2048_groups_0, pad = var_2048_pad_0, pad_type = var_2048_pad_type_0, strides = var_2048_strides_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_1963)[name = string("op_2048")]; tensor var_2053 = const()[name = string("op_2053"), val = tensor([1, 1, 256, 1])]; tensor var_2054 = reshape(shape = var_2053, x = var_2048)[name = string("op_2054")]; tensor var_2059 = const()[name = string("op_2059"), val = tensor([0, 1, 3, 2])]; string var_2076_pad_type_0 = const()[name = string("op_2076_pad_type_0"), val = string("valid")]; tensor var_2076_strides_0 = const()[name = string("op_2076_strides_0"), val = tensor([1, 1])]; tensor var_2076_pad_0 = const()[name = string("op_2076_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2076_dilations_0 = const()[name = string("op_2076_dilations_0"), val = tensor([1, 1])]; int32 var_2076_groups_0 = const()[name = string("op_2076_groups_0"), val = int32(1)]; tensor var_2076 = conv(dilations = var_2076_dilations_0, groups = var_2076_groups_0, pad = var_2076_pad_0, pad_type = var_2076_pad_type_0, strides = var_2076_strides_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_1963)[name = string("op_2076")]; tensor var_2081 = const()[name = string("op_2081"), val = tensor([1, 1, 256, 1])]; tensor var_2082 = reshape(shape = var_2081, x = var_2076)[name = string("op_2082")]; tensor var_2087 = const()[name = string("op_2087"), val = tensor([0, 1, 3, 2])]; tensor var_2097 = const()[name = string("op_2097"), val = tensor([1, 1, 256])]; tensor var_2060 = transpose(perm = var_2059, x = var_2054)[name = string("transpose_364")]; tensor x_9 = reshape(shape = var_2097, x = var_2060)[name = string("x_9")]; int32 var_2103 = const()[name = string("op_2103"), val = int32(-1)]; fp16 const_6_promoted_to_fp16 = const()[name = string("const_6_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2109_cast_fp16 = mul(x = x_9, y = const_6_promoted_to_fp16)[name = string("op_2109_cast_fp16")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7_cast_fp16 = concat(axis = var_2103, interleave = input_7_interleave_0, values = (x_9, var_2109_cast_fp16))[name = string("input_7_cast_fp16")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_2101_to_fp16 = const()[name = string("op_2101_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_2101_to_fp16, x = input_7_cast_fp16)[name = string("normed_9_cast_fp16")]; tensor var_2114_split_sizes_0 = const()[name = string("op_2114_split_sizes_0"), val = tensor([256, 256])]; int32 var_2114_axis_0 = const()[name = string("op_2114_axis_0"), val = int32(-1)]; tensor var_2114_cast_fp16_0, tensor var_2114_cast_fp16_1 = split(axis = var_2114_axis_0, split_sizes = var_2114_split_sizes_0, x = normed_9_cast_fp16)[name = string("op_2114_cast_fp16")]; tensor const_7_to_fp16 = const()[name = string("const_7_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1061830848)))]; tensor var_2117_cast_fp16 = mul(x = var_2114_cast_fp16_0, y = const_7_to_fp16)[name = string("op_2117_cast_fp16")]; tensor var_2123 = const()[name = string("op_2123"), val = tensor([1, 1, 1, 256])]; tensor q_5 = reshape(shape = var_2123, x = var_2117_cast_fp16)[name = string("q_5")]; fp16 var_2130_promoted_to_fp16 = const()[name = string("op_2130_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_2088 = transpose(perm = var_2087, x = var_2082)[name = string("transpose_363")]; tensor var_2131_cast_fp16 = pow(x = var_2088, y = var_2130_promoted_to_fp16)[name = string("op_2131_cast_fp16")]; tensor var_2136_axes_0 = const()[name = string("op_2136_axes_0"), val = tensor([-1])]; bool var_2136_keep_dims_0 = const()[name = string("op_2136_keep_dims_0"), val = bool(true)]; tensor var_2136_cast_fp16 = reduce_mean(axes = var_2136_axes_0, keep_dims = var_2136_keep_dims_0, x = var_2131_cast_fp16)[name = string("op_2136_cast_fp16")]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_1_cast_fp16 = add(x = var_2136_cast_fp16, y = var_2138_to_fp16)[name = string("mean_sq_1_cast_fp16")]; fp16 var_2145_to_fp16 = const()[name = string("op_2145_to_fp16"), val = fp16(-0x1p-1)]; tensor var_2146_cast_fp16 = pow(x = mean_sq_1_cast_fp16, y = var_2145_to_fp16)[name = string("op_2146_cast_fp16")]; tensor var_2147_cast_fp16 = mul(x = var_2088, y = var_2146_cast_fp16)[name = string("op_2147_cast_fp16")]; tensor var_2153 = mul(x = q_5, y = cos_1)[name = string("op_2153")]; tensor var_2154_split_sizes_0 = const()[name = string("op_2154_split_sizes_0"), val = tensor([128, 128])]; int32 var_2154_axis_0 = const()[name = string("op_2154_axis_0"), val = int32(-1)]; tensor var_2154_0, tensor var_2154_1 = split(axis = var_2154_axis_0, split_sizes = var_2154_split_sizes_0, x = q_5)[name = string("op_2154")]; fp16 const_8_promoted = const()[name = string("const_8_promoted"), val = fp16(-0x1p+0)]; tensor var_2156 = mul(x = var_2154_1, y = const_8_promoted)[name = string("op_2156")]; int32 var_2158 = const()[name = string("op_2158"), val = int32(-1)]; bool var_2159_interleave_0 = const()[name = string("op_2159_interleave_0"), val = bool(false)]; tensor var_2159 = concat(axis = var_2158, interleave = var_2159_interleave_0, values = (var_2156, var_2154_0))[name = string("op_2159")]; tensor var_2160 = mul(x = var_2159, y = sin_1)[name = string("op_2160")]; tensor input_9 = add(x = var_2153, y = var_2160)[name = string("input_9")]; tensor read_state_0 = read_state(input = kv_cache_0)[name = string("read_state_0")]; tensor var_2165_begin_0 = const()[name = string("op_2165_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2165_end_0 = const()[name = string("op_2165_end_0"), val = tensor([1, 1, 512, 512])]; tensor var_2165_end_mask_0 = const()[name = string("op_2165_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2165_squeeze_mask_0 = const()[name = string("op_2165_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2165_cast_fp16 = slice_by_index(begin = var_2165_begin_0, end = var_2165_end_0, end_mask = var_2165_end_mask_0, squeeze_mask = var_2165_squeeze_mask_0, x = read_state_0)[name = string("op_2165_cast_fp16")]; tensor K_cache_1_axes_0 = const()[name = string("K_cache_1_axes_0"), val = tensor([0])]; tensor K_cache_1_cast_fp16 = expand_dims(axes = K_cache_1_axes_0, x = var_2165_cast_fp16)[name = string("K_cache_1_cast_fp16")]; tensor var_2170_begin_0 = const()[name = string("op_2170_begin_0"), val = tensor([35, 0, 0, 0])]; tensor var_2170_end_0 = const()[name = string("op_2170_end_0"), val = tensor([36, 1, 512, 512])]; tensor var_2170_end_mask_0 = const()[name = string("op_2170_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2170_squeeze_mask_0 = const()[name = string("op_2170_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2170_cast_fp16 = slice_by_index(begin = var_2170_begin_0, end = var_2170_end_0, end_mask = var_2170_end_mask_0, squeeze_mask = var_2170_squeeze_mask_0, x = read_state_0)[name = string("op_2170_cast_fp16")]; tensor V_cache_1_axes_0 = const()[name = string("V_cache_1_axes_0"), val = tensor([0])]; tensor V_cache_1_cast_fp16 = expand_dims(axes = V_cache_1_axes_0, x = var_2170_cast_fp16)[name = string("V_cache_1_cast_fp16")]; tensor k_padded_1_pad_0 = const()[name = string("k_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_1_mode_0 = const()[name = string("k_padded_1_mode_0"), val = string("constant")]; fp16 const_9_to_fp16 = const()[name = string("const_9_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_1_cast_fp16 = pad(constant_val = const_9_to_fp16, mode = k_padded_1_mode_0, pad = k_padded_1_pad_0, x = input_9)[name = string("k_padded_1_cast_fp16")]; tensor v_padded_1_pad_0 = const()[name = string("v_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_1_mode_0 = const()[name = string("v_padded_1_mode_0"), val = string("constant")]; fp16 const_10_to_fp16 = const()[name = string("const_10_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_1_cast_fp16 = pad(constant_val = const_10_to_fp16, mode = v_padded_1_mode_0, pad = v_padded_1_pad_0, x = var_2147_cast_fp16)[name = string("v_padded_1_cast_fp16")]; fp16 var_2185_promoted_to_fp16 = const()[name = string("op_2185_promoted_to_fp16"), val = fp16(0x1p+0)]; tensor var_2187_cast_fp16 = sub(x = var_2185_promoted_to_fp16, y = update_mask)[name = string("op_2187_cast_fp16")]; tensor var_2188_cast_fp16 = mul(x = K_cache_1_cast_fp16, y = var_2187_cast_fp16)[name = string("op_2188_cast_fp16")]; tensor var_2189_reps_0 = const()[name = string("op_2189_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_2189_cast_fp16 = tile(reps = var_2189_reps_0, x = k_padded_1_cast_fp16)[name = string("op_2189_cast_fp16")]; tensor var_2190_cast_fp16 = mul(x = var_2189_cast_fp16, y = update_mask)[name = string("op_2190_cast_fp16")]; tensor K_new_1_cast_fp16 = add(x = var_2188_cast_fp16, y = var_2190_cast_fp16)[name = string("K_new_1_cast_fp16")]; tensor var_2196_cast_fp16 = mul(x = V_cache_1_cast_fp16, y = var_2187_cast_fp16)[name = string("op_2196_cast_fp16")]; tensor var_2197_reps_0 = const()[name = string("op_2197_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_2197_cast_fp16 = tile(reps = var_2197_reps_0, x = v_padded_1_cast_fp16)[name = string("op_2197_cast_fp16")]; tensor var_2198_cast_fp16 = mul(x = var_2197_cast_fp16, y = update_mask)[name = string("op_2198_cast_fp16")]; tensor V_new_1_cast_fp16 = add(x = var_2196_cast_fp16, y = var_2198_cast_fp16)[name = string("V_new_1_cast_fp16")]; tensor var_2202_axes_0 = const()[name = string("op_2202_axes_0"), val = tensor([0])]; tensor var_2202_cast_fp16 = squeeze(axes = var_2202_axes_0, x = K_new_1_cast_fp16)[name = string("op_2202_cast_fp16")]; tensor concat_0 = const()[name = string("concat_0"), val = tensor([0, 0, 0, 0])]; tensor concat_1 = const()[name = string("concat_1"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_0, begin_mask = kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_1, end_mask = kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_1_stride_0, update = var_2202_cast_fp16, x = read_state_0)[name = string("kv_cache_0_internal_tensor_assign_1_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_1_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_30_write_state")]; tensor coreml_update_state_30 = read_state(input = kv_cache_0)[name = string("coreml_update_state_30")]; tensor var_2209_axes_0 = const()[name = string("op_2209_axes_0"), val = tensor([0])]; tensor var_2209_cast_fp16 = squeeze(axes = var_2209_axes_0, x = V_new_1_cast_fp16)[name = string("op_2209_cast_fp16")]; tensor concat_2 = const()[name = string("concat_2"), val = tensor([35, 0, 0, 0])]; tensor concat_3 = const()[name = string("concat_3"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_2, begin_mask = kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_3, end_mask = kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_2_stride_0, update = var_2209_cast_fp16, x = coreml_update_state_30)[name = string("kv_cache_0_internal_tensor_assign_2_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_2_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_31_write_state")]; tensor coreml_update_state_31 = read_state(input = kv_cache_0)[name = string("coreml_update_state_31")]; tensor K_for_attn_1_begin_0 = const()[name = string("K_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_1_end_0 = const()[name = string("K_for_attn_1_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_1_end_mask_0 = const()[name = string("K_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_1_cast_fp16 = slice_by_index(begin = K_for_attn_1_begin_0, end = K_for_attn_1_end_0, end_mask = K_for_attn_1_end_mask_0, x = K_new_1_cast_fp16)[name = string("K_for_attn_1_cast_fp16")]; tensor V_for_attn_1_begin_0 = const()[name = string("V_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_1_end_0 = const()[name = string("V_for_attn_1_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_1_end_mask_0 = const()[name = string("V_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_1_cast_fp16 = slice_by_index(begin = V_for_attn_1_begin_0, end = V_for_attn_1_end_0, end_mask = V_for_attn_1_end_mask_0, x = V_new_1_cast_fp16)[name = string("V_for_attn_1_cast_fp16")]; tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = K_for_attn_1_cast_fp16)[name = string("transpose_362")]; tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; tensor concat_4 = const()[name = string("concat_4"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_4, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_5 = const()[name = string("concat_5"), val = tensor([-1, 1, 512, 256])]; tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_361")]; tensor reshape_1_cast_fp16 = reshape(shape = concat_5, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; tensor transpose_140_perm_0 = const()[name = string("transpose_140_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = V_for_attn_1_cast_fp16)[name = string("transpose_360")]; tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; tensor concat_6 = const()[name = string("concat_6"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_6, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_7 = const()[name = string("concat_7"), val = tensor([-1, 1, 512, 256])]; tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_359")]; tensor reshape_3_cast_fp16 = reshape(shape = concat_7, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor V_expanded_1_perm_0 = const()[name = string("V_expanded_1_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_2236_transpose_x_0 = const()[name = string("op_2236_transpose_x_0"), val = bool(false)]; bool var_2236_transpose_y_0 = const()[name = string("op_2236_transpose_y_0"), val = bool(false)]; tensor transpose_140_cast_fp16 = transpose(perm = transpose_140_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_358")]; tensor var_2236_cast_fp16 = matmul(transpose_x = var_2236_transpose_x_0, transpose_y = var_2236_transpose_y_0, x = q_7, y = transpose_140_cast_fp16)[name = string("op_2236_cast_fp16")]; tensor attn_weights_3_cast_fp16 = add(x = var_2236_cast_fp16, y = causal_mask)[name = string("attn_weights_3_cast_fp16")]; int32 var_2241 = const()[name = string("op_2241"), val = int32(-1)]; tensor attn_weights_5_cast_fp16 = softmax(axis = var_2241, x = attn_weights_3_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; tensor V_expanded_1_cast_fp16 = transpose(perm = V_expanded_1_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_357")]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = attn_weights_5_cast_fp16, y = V_expanded_1_cast_fp16)[name = string("attn_output_1_cast_fp16")]; tensor var_2249 = const()[name = string("op_2249"), val = tensor([0, 2, 1, 3])]; tensor var_2256 = const()[name = string("op_2256"), val = tensor([1, 1, -1])]; tensor var_2250_cast_fp16 = transpose(perm = var_2249, x = attn_output_1_cast_fp16)[name = string("transpose_356")]; tensor attn_output_3_cast_fp16 = reshape(shape = var_2256, x = var_2250_cast_fp16)[name = string("attn_output_3_cast_fp16")]; tensor var_2261 = const()[name = string("op_2261"), val = tensor([0, 2, 1])]; string var_2277_pad_type_0 = const()[name = string("op_2277_pad_type_0"), val = string("valid")]; int32 var_2277_groups_0 = const()[name = string("op_2277_groups_0"), val = int32(1)]; tensor var_2277_strides_0 = const()[name = string("op_2277_strides_0"), val = tensor([1])]; tensor var_2277_pad_0 = const()[name = string("op_2277_pad_0"), val = tensor([0, 0])]; tensor var_2277_dilations_0 = const()[name = string("op_2277_dilations_0"), val = tensor([1])]; tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1061831424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063404352))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2262_cast_fp16 = transpose(perm = var_2261, x = attn_output_3_cast_fp16)[name = string("transpose_355")]; tensor var_2277_cast_fp16 = conv(dilations = var_2277_dilations_0, groups = var_2277_groups_0, pad = var_2277_pad_0, pad_type = var_2277_pad_type_0, strides = var_2277_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_2262_cast_fp16)[name = string("op_2277_cast_fp16")]; tensor var_2281 = const()[name = string("op_2281"), val = tensor([0, 2, 1])]; int32 var_2287 = const()[name = string("op_2287"), val = int32(-1)]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_15_cast_fp16 = transpose(perm = var_2281, x = var_2277_cast_fp16)[name = string("transpose_354")]; tensor var_2293_cast_fp16 = mul(x = x_15_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_2293_cast_fp16")]; bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; tensor input_15_cast_fp16 = concat(axis = var_2287, interleave = input_15_interleave_0, values = (x_15_cast_fp16, var_2293_cast_fp16))[name = string("input_15_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_2285_to_fp16 = const()[name = string("op_2285_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_2285_to_fp16, x = input_15_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor var_2298_split_sizes_0 = const()[name = string("op_2298_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2298_axis_0 = const()[name = string("op_2298_axis_0"), val = int32(-1)]; tensor var_2298_cast_fp16_0, tensor var_2298_cast_fp16_1 = split(axis = var_2298_axis_0, split_sizes = var_2298_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_2298_cast_fp16")]; tensor const_12_to_fp16 = const()[name = string("const_12_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063405952)))]; tensor var_2301_cast_fp16 = mul(x = var_2298_cast_fp16_0, y = const_12_to_fp16)[name = string("op_2301_cast_fp16")]; tensor x_19_cast_fp16 = add(x = x_1_cast_fp16, y = var_2301_cast_fp16)[name = string("x_19_cast_fp16")]; int32 var_2309 = const()[name = string("op_2309"), val = int32(-1)]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2315_cast_fp16 = mul(x = x_19_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_2315_cast_fp16")]; bool input_17_interleave_0 = const()[name = string("input_17_interleave_0"), val = bool(false)]; tensor input_17_cast_fp16 = concat(axis = var_2309, interleave = input_17_interleave_0, values = (x_19_cast_fp16, var_2315_cast_fp16))[name = string("input_17_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_2307_to_fp16 = const()[name = string("op_2307_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_2307_to_fp16, x = input_17_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_2320_split_sizes_0 = const()[name = string("op_2320_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2320_axis_0 = const()[name = string("op_2320_axis_0"), val = int32(-1)]; tensor var_2320_cast_fp16_0, tensor var_2320_cast_fp16_1 = split(axis = var_2320_axis_0, split_sizes = var_2320_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_2320_cast_fp16")]; tensor const_14_to_fp16 = const()[name = string("const_14_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063409088)))]; tensor var_2323_cast_fp16 = mul(x = var_2320_cast_fp16_0, y = const_14_to_fp16)[name = string("op_2323_cast_fp16")]; tensor var_2333 = const()[name = string("op_2333"), val = tensor([0, 2, 1])]; tensor input_19_axes_0 = const()[name = string("input_19_axes_0"), val = tensor([2])]; tensor var_2334 = transpose(perm = var_2333, x = var_2323_cast_fp16)[name = string("transpose_353")]; tensor input_19 = expand_dims(axes = input_19_axes_0, x = var_2334)[name = string("input_19")]; string gate_1_pad_type_0 = const()[name = string("gate_1_pad_type_0"), val = string("valid")]; tensor gate_1_strides_0 = const()[name = string("gate_1_strides_0"), val = tensor([1, 1])]; tensor gate_1_pad_0 = const()[name = string("gate_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_1_dilations_0 = const()[name = string("gate_1_dilations_0"), val = tensor([1, 1])]; int32 gate_1_groups_0 = const()[name = string("gate_1_groups_0"), val = int32(1)]; tensor gate_1 = conv(dilations = gate_1_dilations_0, groups = gate_1_groups_0, pad = gate_1_pad_0, pad_type = gate_1_pad_type_0, strides = gate_1_strides_0, weight = layers_0_mlp_gate_proj_weight_palettized, x = input_19)[name = string("gate_1")]; string up_1_pad_type_0 = const()[name = string("up_1_pad_type_0"), val = string("valid")]; tensor up_1_strides_0 = const()[name = string("up_1_strides_0"), val = tensor([1, 1])]; tensor up_1_pad_0 = const()[name = string("up_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_1_dilations_0 = const()[name = string("up_1_dilations_0"), val = tensor([1, 1])]; int32 up_1_groups_0 = const()[name = string("up_1_groups_0"), val = int32(1)]; tensor up_1 = conv(dilations = up_1_dilations_0, groups = up_1_groups_0, pad = up_1_pad_0, pad_type = up_1_pad_type_0, strides = up_1_strides_0, weight = layers_0_mlp_up_proj_weight_palettized, x = input_19)[name = string("up_1")]; string gate_3_mode_0 = const()[name = string("gate_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_3 = gelu(mode = gate_3_mode_0, x = gate_1)[name = string("gate_3")]; tensor input_21 = mul(x = gate_3, y = up_1)[name = string("input_21")]; string mlp_out_1_pad_type_0 = const()[name = string("mlp_out_1_pad_type_0"), val = string("valid")]; tensor mlp_out_1_strides_0 = const()[name = string("mlp_out_1_strides_0"), val = tensor([1, 1])]; tensor mlp_out_1_pad_0 = const()[name = string("mlp_out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_1_dilations_0 = const()[name = string("mlp_out_1_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_1_groups_0 = const()[name = string("mlp_out_1_groups_0"), val = int32(1)]; tensor mlp_out_1 = conv(dilations = mlp_out_1_dilations_0, groups = mlp_out_1_groups_0, pad = mlp_out_1_pad_0, pad_type = mlp_out_1_pad_type_0, strides = mlp_out_1_strides_0, weight = layers_0_mlp_down_proj_weight_palettized, x = input_21)[name = string("mlp_out_1")]; tensor var_2374_axes_0 = const()[name = string("op_2374_axes_0"), val = tensor([2])]; tensor var_2374 = squeeze(axes = var_2374_axes_0, x = mlp_out_1)[name = string("op_2374")]; tensor var_2378 = const()[name = string("op_2378"), val = tensor([0, 2, 1])]; int32 var_2384 = const()[name = string("op_2384"), val = int32(-1)]; fp16 const_15_promoted_to_fp16 = const()[name = string("const_15_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_23 = transpose(perm = var_2378, x = var_2374)[name = string("transpose_352")]; tensor var_2390_cast_fp16 = mul(x = x_23, y = const_15_promoted_to_fp16)[name = string("op_2390_cast_fp16")]; bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; tensor input_23_cast_fp16 = concat(axis = var_2384, interleave = input_23_interleave_0, values = (x_23, var_2390_cast_fp16))[name = string("input_23_cast_fp16")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_2382_to_fp16 = const()[name = string("op_2382_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_2382_to_fp16, x = input_23_cast_fp16)[name = string("normed_21_cast_fp16")]; tensor var_2395_split_sizes_0 = const()[name = string("op_2395_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2395_axis_0 = const()[name = string("op_2395_axis_0"), val = int32(-1)]; tensor var_2395_cast_fp16_0, tensor var_2395_cast_fp16_1 = split(axis = var_2395_axis_0, split_sizes = var_2395_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_2395_cast_fp16")]; tensor const_16_to_fp16 = const()[name = string("const_16_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063412224)))]; tensor var_2398_cast_fp16 = mul(x = var_2395_cast_fp16_0, y = const_16_to_fp16)[name = string("op_2398_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = x_19_cast_fp16, y = var_2398_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; tensor per_layer_slice_1_begin_0 = const()[name = string("per_layer_slice_1_begin_0"), val = tensor([0, 0, 0])]; tensor per_layer_slice_1_end_0 = const()[name = string("per_layer_slice_1_end_0"), val = tensor([1, 1, 256])]; tensor per_layer_slice_1_end_mask_0 = const()[name = string("per_layer_slice_1_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_1_cast_fp16 = slice_by_index(begin = per_layer_slice_1_begin_0, end = per_layer_slice_1_end_0, end_mask = per_layer_slice_1_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_1_cast_fp16")]; tensor linear_0_bias_0 = const()[name = string("linear_0_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063415360)))]; tensor gated_1 = linear(bias = linear_0_bias_0, weight = layers_0_per_layer_input_gate_weight_palettized, x = hidden_states_9_cast_fp16)[name = string("linear_0")]; string gated_3_mode_0 = const()[name = string("gated_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_3 = gelu(mode = gated_3_mode_0, x = gated_1)[name = string("gated_3")]; tensor input_27_cast_fp16 = mul(x = gated_3, y = per_layer_slice_1_cast_fp16)[name = string("input_27_cast_fp16")]; tensor layers_0_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063415936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063612608))))[name = string("layers_0_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_1_bias_0_to_fp16 = const()[name = string("linear_1_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063614208)))]; tensor linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_0_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_27_cast_fp16)[name = string("linear_1_cast_fp16")]; int32 var_2435 = const()[name = string("op_2435"), val = int32(-1)]; fp16 const_17_promoted_to_fp16 = const()[name = string("const_17_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2441_cast_fp16 = mul(x = linear_1_cast_fp16, y = const_17_promoted_to_fp16)[name = string("op_2441_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_2435, interleave = input_29_interleave_0, values = (linear_1_cast_fp16, var_2441_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_2433_to_fp16 = const()[name = string("op_2433_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_2433_to_fp16, x = input_29_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor var_2446_split_sizes_0 = const()[name = string("op_2446_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2446_axis_0 = const()[name = string("op_2446_axis_0"), val = int32(-1)]; tensor var_2446_cast_fp16_0, tensor var_2446_cast_fp16_1 = split(axis = var_2446_axis_0, split_sizes = var_2446_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_2446_cast_fp16")]; tensor const_18_to_fp16 = const()[name = string("const_18_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063617344)))]; tensor var_2449_cast_fp16 = mul(x = var_2446_cast_fp16_0, y = const_18_to_fp16)[name = string("op_2449_cast_fp16")]; tensor hidden_states_13 = add(x = hidden_states_9_cast_fp16, y = var_2449_cast_fp16)[name = string("hidden_states_13")]; tensor layers_0_layer_scalar_to_fp16 = const()[name = string("layers_0_layer_scalar_to_fp16"), val = tensor([0x1.24p-6])]; tensor x_31_cast_fp16 = mul(x = hidden_states_13, y = layers_0_layer_scalar_to_fp16)[name = string("x_31_cast_fp16")]; int32 var_2457 = const()[name = string("op_2457"), val = int32(-1)]; fp16 const_19_promoted_to_fp16 = const()[name = string("const_19_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2463_cast_fp16 = mul(x = x_31_cast_fp16, y = const_19_promoted_to_fp16)[name = string("op_2463_cast_fp16")]; bool input_31_interleave_0 = const()[name = string("input_31_interleave_0"), val = bool(false)]; tensor input_31_cast_fp16 = concat(axis = var_2457, interleave = input_31_interleave_0, values = (x_31_cast_fp16, var_2463_cast_fp16))[name = string("input_31_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_2455_to_fp16 = const()[name = string("op_2455_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_2455_to_fp16, x = input_31_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_2468_split_sizes_0 = const()[name = string("op_2468_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2468_axis_0 = const()[name = string("op_2468_axis_0"), val = int32(-1)]; tensor var_2468_cast_fp16_0, tensor var_2468_cast_fp16_1 = split(axis = var_2468_axis_0, split_sizes = var_2468_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_2468_cast_fp16")]; tensor const_20_to_fp16 = const()[name = string("const_20_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063620480)))]; tensor var_2471_cast_fp16 = mul(x = var_2468_cast_fp16_0, y = const_20_to_fp16)[name = string("op_2471_cast_fp16")]; tensor var_2479 = const()[name = string("op_2479"), val = tensor([0, 2, 1])]; tensor var_2482_axes_0 = const()[name = string("op_2482_axes_0"), val = tensor([2])]; tensor var_2480_cast_fp16 = transpose(perm = var_2479, x = var_2471_cast_fp16)[name = string("transpose_351")]; tensor var_2482_cast_fp16 = expand_dims(axes = var_2482_axes_0, x = var_2480_cast_fp16)[name = string("op_2482_cast_fp16")]; string var_2498_pad_type_0 = const()[name = string("op_2498_pad_type_0"), val = string("valid")]; tensor var_2498_strides_0 = const()[name = string("op_2498_strides_0"), val = tensor([1, 1])]; tensor var_2498_pad_0 = const()[name = string("op_2498_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2498_dilations_0 = const()[name = string("op_2498_dilations_0"), val = tensor([1, 1])]; int32 var_2498_groups_0 = const()[name = string("op_2498_groups_0"), val = int32(1)]; tensor var_2498 = conv(dilations = var_2498_dilations_0, groups = var_2498_groups_0, pad = var_2498_pad_0, pad_type = var_2498_pad_type_0, strides = var_2498_strides_0, weight = layers_1_self_attn_q_proj_weight_palettized, x = var_2482_cast_fp16)[name = string("op_2498")]; tensor var_2503 = const()[name = string("op_2503"), val = tensor([1, 8, 256, 1])]; tensor var_2504 = reshape(shape = var_2503, x = var_2498)[name = string("op_2504")]; tensor var_2509 = const()[name = string("op_2509"), val = tensor([0, 1, 3, 2])]; tensor var_2519 = const()[name = string("op_2519"), val = tensor([1, 8, 256])]; tensor var_2510 = transpose(perm = var_2509, x = var_2504)[name = string("transpose_350")]; tensor x_35 = reshape(shape = var_2519, x = var_2510)[name = string("x_35")]; int32 var_2525 = const()[name = string("op_2525"), val = int32(-1)]; fp16 const_21_promoted_to_fp16 = const()[name = string("const_21_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2531_cast_fp16 = mul(x = x_35, y = const_21_promoted_to_fp16)[name = string("op_2531_cast_fp16")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35_cast_fp16 = concat(axis = var_2525, interleave = input_35_interleave_0, values = (x_35, var_2531_cast_fp16))[name = string("input_35_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_2523_to_fp16 = const()[name = string("op_2523_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_2523_to_fp16, x = input_35_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor var_2536_split_sizes_0 = const()[name = string("op_2536_split_sizes_0"), val = tensor([256, 256])]; int32 var_2536_axis_0 = const()[name = string("op_2536_axis_0"), val = int32(-1)]; tensor var_2536_cast_fp16_0, tensor var_2536_cast_fp16_1 = split(axis = var_2536_axis_0, split_sizes = var_2536_split_sizes_0, x = normed_33_cast_fp16)[name = string("op_2536_cast_fp16")]; tensor const_22_to_fp16 = const()[name = string("const_22_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063623616)))]; tensor var_2539_cast_fp16 = mul(x = var_2536_cast_fp16_0, y = const_22_to_fp16)[name = string("op_2539_cast_fp16")]; tensor var_2545 = const()[name = string("op_2545"), val = tensor([1, 8, 1, 256])]; tensor q_11 = reshape(shape = var_2545, x = var_2539_cast_fp16)[name = string("q_11")]; tensor var_2547 = mul(x = q_11, y = cos_1)[name = string("op_2547")]; tensor var_2548_split_sizes_0 = const()[name = string("op_2548_split_sizes_0"), val = tensor([128, 128])]; int32 var_2548_axis_0 = const()[name = string("op_2548_axis_0"), val = int32(-1)]; tensor var_2548_0, tensor var_2548_1 = split(axis = var_2548_axis_0, split_sizes = var_2548_split_sizes_0, x = q_11)[name = string("op_2548")]; fp16 const_23_promoted = const()[name = string("const_23_promoted"), val = fp16(-0x1p+0)]; tensor var_2550 = mul(x = var_2548_1, y = const_23_promoted)[name = string("op_2550")]; int32 var_2552 = const()[name = string("op_2552"), val = int32(-1)]; bool var_2553_interleave_0 = const()[name = string("op_2553_interleave_0"), val = bool(false)]; tensor var_2553 = concat(axis = var_2552, interleave = var_2553_interleave_0, values = (var_2550, var_2548_0))[name = string("op_2553")]; tensor var_2554 = mul(x = var_2553, y = sin_1)[name = string("op_2554")]; tensor q_15 = add(x = var_2547, y = var_2554)[name = string("q_15")]; string var_2567_pad_type_0 = const()[name = string("op_2567_pad_type_0"), val = string("valid")]; tensor var_2567_strides_0 = const()[name = string("op_2567_strides_0"), val = tensor([1, 1])]; tensor var_2567_pad_0 = const()[name = string("op_2567_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2567_dilations_0 = const()[name = string("op_2567_dilations_0"), val = tensor([1, 1])]; int32 var_2567_groups_0 = const()[name = string("op_2567_groups_0"), val = int32(1)]; tensor var_2567 = conv(dilations = var_2567_dilations_0, groups = var_2567_groups_0, pad = var_2567_pad_0, pad_type = var_2567_pad_type_0, strides = var_2567_strides_0, weight = layers_1_self_attn_k_proj_weight_palettized, x = var_2482_cast_fp16)[name = string("op_2567")]; tensor var_2572 = const()[name = string("op_2572"), val = tensor([1, 1, 256, 1])]; tensor var_2573 = reshape(shape = var_2572, x = var_2567)[name = string("op_2573")]; tensor var_2578 = const()[name = string("op_2578"), val = tensor([0, 1, 3, 2])]; string var_2595_pad_type_0 = const()[name = string("op_2595_pad_type_0"), val = string("valid")]; tensor var_2595_strides_0 = const()[name = string("op_2595_strides_0"), val = tensor([1, 1])]; tensor var_2595_pad_0 = const()[name = string("op_2595_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2595_dilations_0 = const()[name = string("op_2595_dilations_0"), val = tensor([1, 1])]; int32 var_2595_groups_0 = const()[name = string("op_2595_groups_0"), val = int32(1)]; tensor var_2595 = conv(dilations = var_2595_dilations_0, groups = var_2595_groups_0, pad = var_2595_pad_0, pad_type = var_2595_pad_type_0, strides = var_2595_strides_0, weight = layers_1_self_attn_v_proj_weight_palettized, x = var_2482_cast_fp16)[name = string("op_2595")]; tensor var_2600 = const()[name = string("op_2600"), val = tensor([1, 1, 256, 1])]; tensor var_2601 = reshape(shape = var_2600, x = var_2595)[name = string("op_2601")]; tensor var_2606 = const()[name = string("op_2606"), val = tensor([0, 1, 3, 2])]; tensor var_2616 = const()[name = string("op_2616"), val = tensor([1, 1, 256])]; tensor var_2579 = transpose(perm = var_2578, x = var_2573)[name = string("transpose_349")]; tensor x_39 = reshape(shape = var_2616, x = var_2579)[name = string("x_39")]; int32 var_2622 = const()[name = string("op_2622"), val = int32(-1)]; fp16 const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2628_cast_fp16 = mul(x = x_39, y = const_24_promoted_to_fp16)[name = string("op_2628_cast_fp16")]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37_cast_fp16 = concat(axis = var_2622, interleave = input_37_interleave_0, values = (x_39, var_2628_cast_fp16))[name = string("input_37_cast_fp16")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_2620_to_fp16 = const()[name = string("op_2620_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_2620_to_fp16, x = input_37_cast_fp16)[name = string("normed_37_cast_fp16")]; tensor var_2633_split_sizes_0 = const()[name = string("op_2633_split_sizes_0"), val = tensor([256, 256])]; int32 var_2633_axis_0 = const()[name = string("op_2633_axis_0"), val = int32(-1)]; tensor var_2633_cast_fp16_0, tensor var_2633_cast_fp16_1 = split(axis = var_2633_axis_0, split_sizes = var_2633_split_sizes_0, x = normed_37_cast_fp16)[name = string("op_2633_cast_fp16")]; tensor const_25_to_fp16 = const()[name = string("const_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063624192)))]; tensor var_2636_cast_fp16 = mul(x = var_2633_cast_fp16_0, y = const_25_to_fp16)[name = string("op_2636_cast_fp16")]; tensor var_2642 = const()[name = string("op_2642"), val = tensor([1, 1, 1, 256])]; tensor q_13 = reshape(shape = var_2642, x = var_2636_cast_fp16)[name = string("q_13")]; fp16 var_2649_promoted_to_fp16 = const()[name = string("op_2649_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_2607 = transpose(perm = var_2606, x = var_2601)[name = string("transpose_348")]; tensor var_2650_cast_fp16 = pow(x = var_2607, y = var_2649_promoted_to_fp16)[name = string("op_2650_cast_fp16")]; tensor var_2655_axes_0 = const()[name = string("op_2655_axes_0"), val = tensor([-1])]; bool var_2655_keep_dims_0 = const()[name = string("op_2655_keep_dims_0"), val = bool(true)]; tensor var_2655_cast_fp16 = reduce_mean(axes = var_2655_axes_0, keep_dims = var_2655_keep_dims_0, x = var_2650_cast_fp16)[name = string("op_2655_cast_fp16")]; fp16 var_2657_to_fp16 = const()[name = string("op_2657_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_3_cast_fp16 = add(x = var_2655_cast_fp16, y = var_2657_to_fp16)[name = string("mean_sq_3_cast_fp16")]; fp16 var_2664_to_fp16 = const()[name = string("op_2664_to_fp16"), val = fp16(-0x1p-1)]; tensor var_2665_cast_fp16 = pow(x = mean_sq_3_cast_fp16, y = var_2664_to_fp16)[name = string("op_2665_cast_fp16")]; tensor var_2666_cast_fp16 = mul(x = var_2607, y = var_2665_cast_fp16)[name = string("op_2666_cast_fp16")]; tensor var_2672 = mul(x = q_13, y = cos_1)[name = string("op_2672")]; tensor var_2673_split_sizes_0 = const()[name = string("op_2673_split_sizes_0"), val = tensor([128, 128])]; int32 var_2673_axis_0 = const()[name = string("op_2673_axis_0"), val = int32(-1)]; tensor var_2673_0, tensor var_2673_1 = split(axis = var_2673_axis_0, split_sizes = var_2673_split_sizes_0, x = q_13)[name = string("op_2673")]; fp16 const_26_promoted = const()[name = string("const_26_promoted"), val = fp16(-0x1p+0)]; tensor var_2675 = mul(x = var_2673_1, y = const_26_promoted)[name = string("op_2675")]; int32 var_2677 = const()[name = string("op_2677"), val = int32(-1)]; bool var_2678_interleave_0 = const()[name = string("op_2678_interleave_0"), val = bool(false)]; tensor var_2678 = concat(axis = var_2677, interleave = var_2678_interleave_0, values = (var_2675, var_2673_0))[name = string("op_2678")]; tensor var_2679 = mul(x = var_2678, y = sin_1)[name = string("op_2679")]; tensor input_39 = add(x = var_2672, y = var_2679)[name = string("input_39")]; tensor var_2684_begin_0 = const()[name = string("op_2684_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_2684_end_0 = const()[name = string("op_2684_end_0"), val = tensor([2, 1, 512, 512])]; tensor var_2684_end_mask_0 = const()[name = string("op_2684_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2684_squeeze_mask_0 = const()[name = string("op_2684_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2684_cast_fp16 = slice_by_index(begin = var_2684_begin_0, end = var_2684_end_0, end_mask = var_2684_end_mask_0, squeeze_mask = var_2684_squeeze_mask_0, x = coreml_update_state_31)[name = string("op_2684_cast_fp16")]; tensor K_cache_3_axes_0 = const()[name = string("K_cache_3_axes_0"), val = tensor([0])]; tensor K_cache_3_cast_fp16 = expand_dims(axes = K_cache_3_axes_0, x = var_2684_cast_fp16)[name = string("K_cache_3_cast_fp16")]; tensor var_2689_begin_0 = const()[name = string("op_2689_begin_0"), val = tensor([36, 0, 0, 0])]; tensor var_2689_end_0 = const()[name = string("op_2689_end_0"), val = tensor([37, 1, 512, 512])]; tensor var_2689_end_mask_0 = const()[name = string("op_2689_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2689_squeeze_mask_0 = const()[name = string("op_2689_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2689_cast_fp16 = slice_by_index(begin = var_2689_begin_0, end = var_2689_end_0, end_mask = var_2689_end_mask_0, squeeze_mask = var_2689_squeeze_mask_0, x = coreml_update_state_31)[name = string("op_2689_cast_fp16")]; tensor V_cache_3_axes_0 = const()[name = string("V_cache_3_axes_0"), val = tensor([0])]; tensor V_cache_3_cast_fp16 = expand_dims(axes = V_cache_3_axes_0, x = var_2689_cast_fp16)[name = string("V_cache_3_cast_fp16")]; tensor k_padded_3_pad_0 = const()[name = string("k_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_3_mode_0 = const()[name = string("k_padded_3_mode_0"), val = string("constant")]; fp16 const_27_to_fp16 = const()[name = string("const_27_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_3_cast_fp16 = pad(constant_val = const_27_to_fp16, mode = k_padded_3_mode_0, pad = k_padded_3_pad_0, x = input_39)[name = string("k_padded_3_cast_fp16")]; tensor v_padded_3_pad_0 = const()[name = string("v_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_3_mode_0 = const()[name = string("v_padded_3_mode_0"), val = string("constant")]; fp16 const_28_to_fp16 = const()[name = string("const_28_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_3_cast_fp16 = pad(constant_val = const_28_to_fp16, mode = v_padded_3_mode_0, pad = v_padded_3_pad_0, x = var_2666_cast_fp16)[name = string("v_padded_3_cast_fp16")]; tensor var_2707_cast_fp16 = mul(x = K_cache_3_cast_fp16, y = var_2187_cast_fp16)[name = string("op_2707_cast_fp16")]; tensor var_2708_reps_0 = const()[name = string("op_2708_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_2708_cast_fp16 = tile(reps = var_2708_reps_0, x = k_padded_3_cast_fp16)[name = string("op_2708_cast_fp16")]; tensor var_2709_cast_fp16 = mul(x = var_2708_cast_fp16, y = update_mask)[name = string("op_2709_cast_fp16")]; tensor K_new_3_cast_fp16 = add(x = var_2707_cast_fp16, y = var_2709_cast_fp16)[name = string("K_new_3_cast_fp16")]; tensor var_2715_cast_fp16 = mul(x = V_cache_3_cast_fp16, y = var_2187_cast_fp16)[name = string("op_2715_cast_fp16")]; tensor var_2716_reps_0 = const()[name = string("op_2716_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_2716_cast_fp16 = tile(reps = var_2716_reps_0, x = v_padded_3_cast_fp16)[name = string("op_2716_cast_fp16")]; tensor var_2717_cast_fp16 = mul(x = var_2716_cast_fp16, y = update_mask)[name = string("op_2717_cast_fp16")]; tensor V_new_3_cast_fp16 = add(x = var_2715_cast_fp16, y = var_2717_cast_fp16)[name = string("V_new_3_cast_fp16")]; tensor var_2721_axes_0 = const()[name = string("op_2721_axes_0"), val = tensor([0])]; tensor var_2721_cast_fp16 = squeeze(axes = var_2721_axes_0, x = K_new_3_cast_fp16)[name = string("op_2721_cast_fp16")]; tensor concat_8 = const()[name = string("concat_8"), val = tensor([1, 0, 0, 0])]; tensor concat_9 = const()[name = string("concat_9"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_8, begin_mask = kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_9, end_mask = kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_3_stride_0, update = var_2721_cast_fp16, x = coreml_update_state_31)[name = string("kv_cache_0_internal_tensor_assign_3_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_3_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_32_write_state")]; tensor coreml_update_state_32 = read_state(input = kv_cache_0)[name = string("coreml_update_state_32")]; tensor var_2728_axes_0 = const()[name = string("op_2728_axes_0"), val = tensor([0])]; tensor var_2728_cast_fp16 = squeeze(axes = var_2728_axes_0, x = V_new_3_cast_fp16)[name = string("op_2728_cast_fp16")]; tensor concat_10 = const()[name = string("concat_10"), val = tensor([36, 0, 0, 0])]; tensor concat_11 = const()[name = string("concat_11"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_10, begin_mask = kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_11, end_mask = kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_4_stride_0, update = var_2728_cast_fp16, x = coreml_update_state_32)[name = string("kv_cache_0_internal_tensor_assign_4_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_4_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_33_write_state")]; tensor coreml_update_state_33 = read_state(input = kv_cache_0)[name = string("coreml_update_state_33")]; tensor K_for_attn_3_begin_0 = const()[name = string("K_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_3_end_0 = const()[name = string("K_for_attn_3_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_3_end_mask_0 = const()[name = string("K_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_3_cast_fp16 = slice_by_index(begin = K_for_attn_3_begin_0, end = K_for_attn_3_end_0, end_mask = K_for_attn_3_end_mask_0, x = K_new_3_cast_fp16)[name = string("K_for_attn_3_cast_fp16")]; tensor V_for_attn_3_begin_0 = const()[name = string("V_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_3_end_0 = const()[name = string("V_for_attn_3_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_3_end_mask_0 = const()[name = string("V_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_3_cast_fp16 = slice_by_index(begin = V_for_attn_3_begin_0, end = V_for_attn_3_end_0, end_mask = V_for_attn_3_end_mask_0, x = V_new_3_cast_fp16)[name = string("V_for_attn_3_cast_fp16")]; tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = K_for_attn_3_cast_fp16)[name = string("transpose_347")]; tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; tensor concat_12 = const()[name = string("concat_12"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_12, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_13 = const()[name = string("concat_13"), val = tensor([-1, 1, 512, 256])]; tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_346")]; tensor reshape_5_cast_fp16 = reshape(shape = concat_13, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor transpose_141_perm_0 = const()[name = string("transpose_141_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = V_for_attn_3_cast_fp16)[name = string("transpose_345")]; tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; tensor concat_14 = const()[name = string("concat_14"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_14, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_15 = const()[name = string("concat_15"), val = tensor([-1, 1, 512, 256])]; tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_344")]; tensor reshape_7_cast_fp16 = reshape(shape = concat_15, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; tensor V_expanded_3_perm_0 = const()[name = string("V_expanded_3_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_2755_transpose_x_0 = const()[name = string("op_2755_transpose_x_0"), val = bool(false)]; bool var_2755_transpose_y_0 = const()[name = string("op_2755_transpose_y_0"), val = bool(false)]; tensor transpose_141_cast_fp16 = transpose(perm = transpose_141_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_343")]; tensor var_2755_cast_fp16 = matmul(transpose_x = var_2755_transpose_x_0, transpose_y = var_2755_transpose_y_0, x = q_15, y = transpose_141_cast_fp16)[name = string("op_2755_cast_fp16")]; tensor attn_weights_9_cast_fp16 = add(x = var_2755_cast_fp16, y = causal_mask)[name = string("attn_weights_9_cast_fp16")]; int32 var_2760 = const()[name = string("op_2760"), val = int32(-1)]; tensor attn_weights_11_cast_fp16 = softmax(axis = var_2760, x = attn_weights_9_cast_fp16)[name = string("attn_weights_11_cast_fp16")]; bool attn_output_7_transpose_x_0 = const()[name = string("attn_output_7_transpose_x_0"), val = bool(false)]; bool attn_output_7_transpose_y_0 = const()[name = string("attn_output_7_transpose_y_0"), val = bool(false)]; tensor V_expanded_3_cast_fp16 = transpose(perm = V_expanded_3_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_342")]; tensor attn_output_7_cast_fp16 = matmul(transpose_x = attn_output_7_transpose_x_0, transpose_y = attn_output_7_transpose_y_0, x = attn_weights_11_cast_fp16, y = V_expanded_3_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor var_2768 = const()[name = string("op_2768"), val = tensor([0, 2, 1, 3])]; tensor var_2775 = const()[name = string("op_2775"), val = tensor([1, 1, -1])]; tensor var_2769_cast_fp16 = transpose(perm = var_2768, x = attn_output_7_cast_fp16)[name = string("transpose_341")]; tensor attn_output_9_cast_fp16 = reshape(shape = var_2775, x = var_2769_cast_fp16)[name = string("attn_output_9_cast_fp16")]; tensor var_2780 = const()[name = string("op_2780"), val = tensor([0, 2, 1])]; string var_2796_pad_type_0 = const()[name = string("op_2796_pad_type_0"), val = string("valid")]; int32 var_2796_groups_0 = const()[name = string("op_2796_groups_0"), val = int32(1)]; tensor var_2796_strides_0 = const()[name = string("op_2796_strides_0"), val = tensor([1])]; tensor var_2796_pad_0 = const()[name = string("op_2796_pad_0"), val = tensor([0, 0])]; tensor var_2796_dilations_0 = const()[name = string("op_2796_dilations_0"), val = tensor([1])]; tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1063624768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065197696))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2781_cast_fp16 = transpose(perm = var_2780, x = attn_output_9_cast_fp16)[name = string("transpose_340")]; tensor var_2796_cast_fp16 = conv(dilations = var_2796_dilations_0, groups = var_2796_groups_0, pad = var_2796_pad_0, pad_type = var_2796_pad_type_0, strides = var_2796_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_2781_cast_fp16)[name = string("op_2796_cast_fp16")]; tensor var_2800 = const()[name = string("op_2800"), val = tensor([0, 2, 1])]; int32 var_2806 = const()[name = string("op_2806"), val = int32(-1)]; fp16 const_29_promoted_to_fp16 = const()[name = string("const_29_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_45_cast_fp16 = transpose(perm = var_2800, x = var_2796_cast_fp16)[name = string("transpose_339")]; tensor var_2812_cast_fp16 = mul(x = x_45_cast_fp16, y = const_29_promoted_to_fp16)[name = string("op_2812_cast_fp16")]; bool input_45_interleave_0 = const()[name = string("input_45_interleave_0"), val = bool(false)]; tensor input_45_cast_fp16 = concat(axis = var_2806, interleave = input_45_interleave_0, values = (x_45_cast_fp16, var_2812_cast_fp16))[name = string("input_45_cast_fp16")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_2804_to_fp16 = const()[name = string("op_2804_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2804_to_fp16, x = input_45_cast_fp16)[name = string("normed_41_cast_fp16")]; tensor var_2817_split_sizes_0 = const()[name = string("op_2817_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2817_axis_0 = const()[name = string("op_2817_axis_0"), val = int32(-1)]; tensor var_2817_cast_fp16_0, tensor var_2817_cast_fp16_1 = split(axis = var_2817_axis_0, split_sizes = var_2817_split_sizes_0, x = normed_41_cast_fp16)[name = string("op_2817_cast_fp16")]; tensor const_30_to_fp16 = const()[name = string("const_30_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065199296)))]; tensor var_2820_cast_fp16 = mul(x = var_2817_cast_fp16_0, y = const_30_to_fp16)[name = string("op_2820_cast_fp16")]; tensor x_49_cast_fp16 = add(x = x_31_cast_fp16, y = var_2820_cast_fp16)[name = string("x_49_cast_fp16")]; int32 var_2827 = const()[name = string("op_2827"), val = int32(-1)]; fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2833_cast_fp16 = mul(x = x_49_cast_fp16, y = const_31_promoted_to_fp16)[name = string("op_2833_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_2827, interleave = input_47_interleave_0, values = (x_49_cast_fp16, var_2833_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_2825_to_fp16 = const()[name = string("op_2825_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2825_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor var_2838_split_sizes_0 = const()[name = string("op_2838_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2838_axis_0 = const()[name = string("op_2838_axis_0"), val = int32(-1)]; tensor var_2838_cast_fp16_0, tensor var_2838_cast_fp16_1 = split(axis = var_2838_axis_0, split_sizes = var_2838_split_sizes_0, x = normed_45_cast_fp16)[name = string("op_2838_cast_fp16")]; tensor const_32_to_fp16 = const()[name = string("const_32_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065202432)))]; tensor var_2841_cast_fp16 = mul(x = var_2838_cast_fp16_0, y = const_32_to_fp16)[name = string("op_2841_cast_fp16")]; tensor var_2854 = const()[name = string("op_2854"), val = tensor([0, 2, 1])]; tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; tensor var_2855 = transpose(perm = var_2854, x = var_2841_cast_fp16)[name = string("transpose_338")]; tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2855)[name = string("input_49")]; string gate_5_pad_type_0 = const()[name = string("gate_5_pad_type_0"), val = string("valid")]; tensor gate_5_strides_0 = const()[name = string("gate_5_strides_0"), val = tensor([1, 1])]; tensor gate_5_pad_0 = const()[name = string("gate_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_5_dilations_0 = const()[name = string("gate_5_dilations_0"), val = tensor([1, 1])]; int32 gate_5_groups_0 = const()[name = string("gate_5_groups_0"), val = int32(1)]; tensor gate_5 = conv(dilations = gate_5_dilations_0, groups = gate_5_groups_0, pad = gate_5_pad_0, pad_type = gate_5_pad_type_0, strides = gate_5_strides_0, weight = layers_1_mlp_gate_proj_weight_palettized, x = input_49)[name = string("gate_5")]; string up_3_pad_type_0 = const()[name = string("up_3_pad_type_0"), val = string("valid")]; tensor up_3_strides_0 = const()[name = string("up_3_strides_0"), val = tensor([1, 1])]; tensor up_3_pad_0 = const()[name = string("up_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_3_dilations_0 = const()[name = string("up_3_dilations_0"), val = tensor([1, 1])]; int32 up_3_groups_0 = const()[name = string("up_3_groups_0"), val = int32(1)]; tensor up_3 = conv(dilations = up_3_dilations_0, groups = up_3_groups_0, pad = up_3_pad_0, pad_type = up_3_pad_type_0, strides = up_3_strides_0, weight = layers_1_mlp_up_proj_weight_palettized, x = input_49)[name = string("up_3")]; string gate_7_mode_0 = const()[name = string("gate_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_7 = gelu(mode = gate_7_mode_0, x = gate_5)[name = string("gate_7")]; tensor input_51 = mul(x = gate_7, y = up_3)[name = string("input_51")]; string mlp_out_3_pad_type_0 = const()[name = string("mlp_out_3_pad_type_0"), val = string("valid")]; tensor mlp_out_3_strides_0 = const()[name = string("mlp_out_3_strides_0"), val = tensor([1, 1])]; tensor mlp_out_3_pad_0 = const()[name = string("mlp_out_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_3_dilations_0 = const()[name = string("mlp_out_3_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_3_groups_0 = const()[name = string("mlp_out_3_groups_0"), val = int32(1)]; tensor mlp_out_3 = conv(dilations = mlp_out_3_dilations_0, groups = mlp_out_3_groups_0, pad = mlp_out_3_pad_0, pad_type = mlp_out_3_pad_type_0, strides = mlp_out_3_strides_0, weight = layers_1_mlp_down_proj_weight_palettized, x = input_51)[name = string("mlp_out_3")]; tensor var_2895_axes_0 = const()[name = string("op_2895_axes_0"), val = tensor([2])]; tensor var_2895 = squeeze(axes = var_2895_axes_0, x = mlp_out_3)[name = string("op_2895")]; tensor var_2899 = const()[name = string("op_2899"), val = tensor([0, 2, 1])]; int32 var_2905 = const()[name = string("op_2905"), val = int32(-1)]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_53 = transpose(perm = var_2899, x = var_2895)[name = string("transpose_337")]; tensor var_2911_cast_fp16 = mul(x = x_53, y = const_33_promoted_to_fp16)[name = string("op_2911_cast_fp16")]; bool input_53_interleave_0 = const()[name = string("input_53_interleave_0"), val = bool(false)]; tensor input_53_cast_fp16 = concat(axis = var_2905, interleave = input_53_interleave_0, values = (x_53, var_2911_cast_fp16))[name = string("input_53_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_2903_to_fp16 = const()[name = string("op_2903_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2903_to_fp16, x = input_53_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor var_2916_split_sizes_0 = const()[name = string("op_2916_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2916_axis_0 = const()[name = string("op_2916_axis_0"), val = int32(-1)]; tensor var_2916_cast_fp16_0, tensor var_2916_cast_fp16_1 = split(axis = var_2916_axis_0, split_sizes = var_2916_split_sizes_0, x = normed_49_cast_fp16)[name = string("op_2916_cast_fp16")]; tensor const_34_to_fp16 = const()[name = string("const_34_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065205568)))]; tensor var_2919_cast_fp16 = mul(x = var_2916_cast_fp16_0, y = const_34_to_fp16)[name = string("op_2919_cast_fp16")]; tensor hidden_states_21_cast_fp16 = add(x = x_49_cast_fp16, y = var_2919_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; tensor per_layer_slice_3_begin_0 = const()[name = string("per_layer_slice_3_begin_0"), val = tensor([0, 0, 256])]; tensor per_layer_slice_3_end_0 = const()[name = string("per_layer_slice_3_end_0"), val = tensor([1, 1, 512])]; tensor per_layer_slice_3_end_mask_0 = const()[name = string("per_layer_slice_3_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_3_cast_fp16 = slice_by_index(begin = per_layer_slice_3_begin_0, end = per_layer_slice_3_end_0, end_mask = per_layer_slice_3_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_3_cast_fp16")]; tensor gated_5 = linear(bias = linear_0_bias_0, weight = layers_1_per_layer_input_gate_weight_palettized, x = hidden_states_21_cast_fp16)[name = string("linear_2")]; string gated_7_mode_0 = const()[name = string("gated_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_7 = gelu(mode = gated_7_mode_0, x = gated_5)[name = string("gated_7")]; tensor input_57_cast_fp16 = mul(x = gated_7, y = per_layer_slice_3_cast_fp16)[name = string("input_57_cast_fp16")]; tensor layers_1_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065208704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065405376))))[name = string("layers_1_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_3_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_1_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_57_cast_fp16)[name = string("linear_3_cast_fp16")]; int32 var_2956 = const()[name = string("op_2956"), val = int32(-1)]; fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2962_cast_fp16 = mul(x = linear_3_cast_fp16, y = const_35_promoted_to_fp16)[name = string("op_2962_cast_fp16")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59_cast_fp16 = concat(axis = var_2956, interleave = input_59_interleave_0, values = (linear_3_cast_fp16, var_2962_cast_fp16))[name = string("input_59_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_2954_to_fp16 = const()[name = string("op_2954_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_2954_to_fp16, x = input_59_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor var_2967_split_sizes_0 = const()[name = string("op_2967_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2967_axis_0 = const()[name = string("op_2967_axis_0"), val = int32(-1)]; tensor var_2967_cast_fp16_0, tensor var_2967_cast_fp16_1 = split(axis = var_2967_axis_0, split_sizes = var_2967_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_2967_cast_fp16")]; tensor const_36_to_fp16 = const()[name = string("const_36_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065406976)))]; tensor var_2970_cast_fp16 = mul(x = var_2967_cast_fp16_0, y = const_36_to_fp16)[name = string("op_2970_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = var_2970_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; tensor layers_1_layer_scalar_to_fp16 = const()[name = string("layers_1_layer_scalar_to_fp16"), val = tensor([0x1.c8p-3])]; tensor x_61_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = layers_1_layer_scalar_to_fp16)[name = string("x_61_cast_fp16")]; int32 var_2978 = const()[name = string("op_2978"), val = int32(-1)]; fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2984_cast_fp16 = mul(x = x_61_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_2984_cast_fp16")]; bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; tensor input_61_cast_fp16 = concat(axis = var_2978, interleave = input_61_interleave_0, values = (x_61_cast_fp16, var_2984_cast_fp16))[name = string("input_61_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_2976_to_fp16 = const()[name = string("op_2976_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_2976_to_fp16, x = input_61_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_2989_split_sizes_0 = const()[name = string("op_2989_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2989_axis_0 = const()[name = string("op_2989_axis_0"), val = int32(-1)]; tensor var_2989_cast_fp16_0, tensor var_2989_cast_fp16_1 = split(axis = var_2989_axis_0, split_sizes = var_2989_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_2989_cast_fp16")]; tensor const_38_to_fp16 = const()[name = string("const_38_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065410112)))]; tensor var_2992_cast_fp16 = mul(x = var_2989_cast_fp16_0, y = const_38_to_fp16)[name = string("op_2992_cast_fp16")]; tensor var_3000 = const()[name = string("op_3000"), val = tensor([0, 2, 1])]; tensor var_3003_axes_0 = const()[name = string("op_3003_axes_0"), val = tensor([2])]; tensor var_3001_cast_fp16 = transpose(perm = var_3000, x = var_2992_cast_fp16)[name = string("transpose_336")]; tensor var_3003_cast_fp16 = expand_dims(axes = var_3003_axes_0, x = var_3001_cast_fp16)[name = string("op_3003_cast_fp16")]; string var_3019_pad_type_0 = const()[name = string("op_3019_pad_type_0"), val = string("valid")]; tensor var_3019_strides_0 = const()[name = string("op_3019_strides_0"), val = tensor([1, 1])]; tensor var_3019_pad_0 = const()[name = string("op_3019_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3019_dilations_0 = const()[name = string("op_3019_dilations_0"), val = tensor([1, 1])]; int32 var_3019_groups_0 = const()[name = string("op_3019_groups_0"), val = int32(1)]; tensor var_3019 = conv(dilations = var_3019_dilations_0, groups = var_3019_groups_0, pad = var_3019_pad_0, pad_type = var_3019_pad_type_0, strides = var_3019_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_3003_cast_fp16)[name = string("op_3019")]; tensor var_3024 = const()[name = string("op_3024"), val = tensor([1, 8, 256, 1])]; tensor var_3025 = reshape(shape = var_3024, x = var_3019)[name = string("op_3025")]; tensor var_3030 = const()[name = string("op_3030"), val = tensor([0, 1, 3, 2])]; tensor var_3040 = const()[name = string("op_3040"), val = tensor([1, 8, 256])]; tensor var_3031 = transpose(perm = var_3030, x = var_3025)[name = string("transpose_335")]; tensor x_65 = reshape(shape = var_3040, x = var_3031)[name = string("x_65")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-1)]; fp16 const_39_promoted_to_fp16 = const()[name = string("const_39_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3052_cast_fp16 = mul(x = x_65, y = const_39_promoted_to_fp16)[name = string("op_3052_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_3046, interleave = input_65_interleave_0, values = (x_65, var_3052_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_3044_to_fp16 = const()[name = string("op_3044_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_3044_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor var_3057_split_sizes_0 = const()[name = string("op_3057_split_sizes_0"), val = tensor([256, 256])]; int32 var_3057_axis_0 = const()[name = string("op_3057_axis_0"), val = int32(-1)]; tensor var_3057_cast_fp16_0, tensor var_3057_cast_fp16_1 = split(axis = var_3057_axis_0, split_sizes = var_3057_split_sizes_0, x = normed_61_cast_fp16)[name = string("op_3057_cast_fp16")]; tensor const_40_to_fp16 = const()[name = string("const_40_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065413248)))]; tensor var_3060_cast_fp16 = mul(x = var_3057_cast_fp16_0, y = const_40_to_fp16)[name = string("op_3060_cast_fp16")]; tensor var_3066 = const()[name = string("op_3066"), val = tensor([1, 8, 1, 256])]; tensor q_19 = reshape(shape = var_3066, x = var_3060_cast_fp16)[name = string("q_19")]; tensor var_3068 = mul(x = q_19, y = cos_1)[name = string("op_3068")]; tensor var_3069_split_sizes_0 = const()[name = string("op_3069_split_sizes_0"), val = tensor([128, 128])]; int32 var_3069_axis_0 = const()[name = string("op_3069_axis_0"), val = int32(-1)]; tensor var_3069_0, tensor var_3069_1 = split(axis = var_3069_axis_0, split_sizes = var_3069_split_sizes_0, x = q_19)[name = string("op_3069")]; fp16 const_41_promoted = const()[name = string("const_41_promoted"), val = fp16(-0x1p+0)]; tensor var_3071 = mul(x = var_3069_1, y = const_41_promoted)[name = string("op_3071")]; int32 var_3073 = const()[name = string("op_3073"), val = int32(-1)]; bool var_3074_interleave_0 = const()[name = string("op_3074_interleave_0"), val = bool(false)]; tensor var_3074 = concat(axis = var_3073, interleave = var_3074_interleave_0, values = (var_3071, var_3069_0))[name = string("op_3074")]; tensor var_3075 = mul(x = var_3074, y = sin_1)[name = string("op_3075")]; tensor q_23 = add(x = var_3068, y = var_3075)[name = string("q_23")]; string var_3088_pad_type_0 = const()[name = string("op_3088_pad_type_0"), val = string("valid")]; tensor var_3088_strides_0 = const()[name = string("op_3088_strides_0"), val = tensor([1, 1])]; tensor var_3088_pad_0 = const()[name = string("op_3088_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3088_dilations_0 = const()[name = string("op_3088_dilations_0"), val = tensor([1, 1])]; int32 var_3088_groups_0 = const()[name = string("op_3088_groups_0"), val = int32(1)]; tensor var_3088 = conv(dilations = var_3088_dilations_0, groups = var_3088_groups_0, pad = var_3088_pad_0, pad_type = var_3088_pad_type_0, strides = var_3088_strides_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_3003_cast_fp16)[name = string("op_3088")]; tensor var_3093 = const()[name = string("op_3093"), val = tensor([1, 1, 256, 1])]; tensor var_3094 = reshape(shape = var_3093, x = var_3088)[name = string("op_3094")]; tensor var_3099 = const()[name = string("op_3099"), val = tensor([0, 1, 3, 2])]; string var_3116_pad_type_0 = const()[name = string("op_3116_pad_type_0"), val = string("valid")]; tensor var_3116_strides_0 = const()[name = string("op_3116_strides_0"), val = tensor([1, 1])]; tensor var_3116_pad_0 = const()[name = string("op_3116_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3116_dilations_0 = const()[name = string("op_3116_dilations_0"), val = tensor([1, 1])]; int32 var_3116_groups_0 = const()[name = string("op_3116_groups_0"), val = int32(1)]; tensor var_3116 = conv(dilations = var_3116_dilations_0, groups = var_3116_groups_0, pad = var_3116_pad_0, pad_type = var_3116_pad_type_0, strides = var_3116_strides_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_3003_cast_fp16)[name = string("op_3116")]; tensor var_3121 = const()[name = string("op_3121"), val = tensor([1, 1, 256, 1])]; tensor var_3122 = reshape(shape = var_3121, x = var_3116)[name = string("op_3122")]; tensor var_3127 = const()[name = string("op_3127"), val = tensor([0, 1, 3, 2])]; tensor var_3137 = const()[name = string("op_3137"), val = tensor([1, 1, 256])]; tensor var_3100 = transpose(perm = var_3099, x = var_3094)[name = string("transpose_334")]; tensor x_69 = reshape(shape = var_3137, x = var_3100)[name = string("x_69")]; int32 var_3143 = const()[name = string("op_3143"), val = int32(-1)]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3149_cast_fp16 = mul(x = x_69, y = const_42_promoted_to_fp16)[name = string("op_3149_cast_fp16")]; bool input_67_interleave_0 = const()[name = string("input_67_interleave_0"), val = bool(false)]; tensor input_67_cast_fp16 = concat(axis = var_3143, interleave = input_67_interleave_0, values = (x_69, var_3149_cast_fp16))[name = string("input_67_cast_fp16")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_3141_to_fp16 = const()[name = string("op_3141_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_3141_to_fp16, x = input_67_cast_fp16)[name = string("normed_65_cast_fp16")]; tensor var_3154_split_sizes_0 = const()[name = string("op_3154_split_sizes_0"), val = tensor([256, 256])]; int32 var_3154_axis_0 = const()[name = string("op_3154_axis_0"), val = int32(-1)]; tensor var_3154_cast_fp16_0, tensor var_3154_cast_fp16_1 = split(axis = var_3154_axis_0, split_sizes = var_3154_split_sizes_0, x = normed_65_cast_fp16)[name = string("op_3154_cast_fp16")]; tensor const_43_to_fp16 = const()[name = string("const_43_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065413824)))]; tensor var_3157_cast_fp16 = mul(x = var_3154_cast_fp16_0, y = const_43_to_fp16)[name = string("op_3157_cast_fp16")]; tensor var_3163 = const()[name = string("op_3163"), val = tensor([1, 1, 1, 256])]; tensor q_21 = reshape(shape = var_3163, x = var_3157_cast_fp16)[name = string("q_21")]; fp16 var_3170_promoted_to_fp16 = const()[name = string("op_3170_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_3128 = transpose(perm = var_3127, x = var_3122)[name = string("transpose_333")]; tensor var_3171_cast_fp16 = pow(x = var_3128, y = var_3170_promoted_to_fp16)[name = string("op_3171_cast_fp16")]; tensor var_3176_axes_0 = const()[name = string("op_3176_axes_0"), val = tensor([-1])]; bool var_3176_keep_dims_0 = const()[name = string("op_3176_keep_dims_0"), val = bool(true)]; tensor var_3176_cast_fp16 = reduce_mean(axes = var_3176_axes_0, keep_dims = var_3176_keep_dims_0, x = var_3171_cast_fp16)[name = string("op_3176_cast_fp16")]; fp16 var_3178_to_fp16 = const()[name = string("op_3178_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_5_cast_fp16 = add(x = var_3176_cast_fp16, y = var_3178_to_fp16)[name = string("mean_sq_5_cast_fp16")]; fp16 var_3185_to_fp16 = const()[name = string("op_3185_to_fp16"), val = fp16(-0x1p-1)]; tensor var_3186_cast_fp16 = pow(x = mean_sq_5_cast_fp16, y = var_3185_to_fp16)[name = string("op_3186_cast_fp16")]; tensor var_3187_cast_fp16 = mul(x = var_3128, y = var_3186_cast_fp16)[name = string("op_3187_cast_fp16")]; tensor var_3193 = mul(x = q_21, y = cos_1)[name = string("op_3193")]; tensor var_3194_split_sizes_0 = const()[name = string("op_3194_split_sizes_0"), val = tensor([128, 128])]; int32 var_3194_axis_0 = const()[name = string("op_3194_axis_0"), val = int32(-1)]; tensor var_3194_0, tensor var_3194_1 = split(axis = var_3194_axis_0, split_sizes = var_3194_split_sizes_0, x = q_21)[name = string("op_3194")]; fp16 const_44_promoted = const()[name = string("const_44_promoted"), val = fp16(-0x1p+0)]; tensor var_3196 = mul(x = var_3194_1, y = const_44_promoted)[name = string("op_3196")]; int32 var_3198 = const()[name = string("op_3198"), val = int32(-1)]; bool var_3199_interleave_0 = const()[name = string("op_3199_interleave_0"), val = bool(false)]; tensor var_3199 = concat(axis = var_3198, interleave = var_3199_interleave_0, values = (var_3196, var_3194_0))[name = string("op_3199")]; tensor var_3200 = mul(x = var_3199, y = sin_1)[name = string("op_3200")]; tensor input_69 = add(x = var_3193, y = var_3200)[name = string("input_69")]; tensor var_3205_begin_0 = const()[name = string("op_3205_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_3205_end_0 = const()[name = string("op_3205_end_0"), val = tensor([3, 1, 512, 512])]; tensor var_3205_end_mask_0 = const()[name = string("op_3205_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3205_squeeze_mask_0 = const()[name = string("op_3205_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_3205_cast_fp16 = slice_by_index(begin = var_3205_begin_0, end = var_3205_end_0, end_mask = var_3205_end_mask_0, squeeze_mask = var_3205_squeeze_mask_0, x = coreml_update_state_33)[name = string("op_3205_cast_fp16")]; tensor K_cache_5_axes_0 = const()[name = string("K_cache_5_axes_0"), val = tensor([0])]; tensor K_cache_5_cast_fp16 = expand_dims(axes = K_cache_5_axes_0, x = var_3205_cast_fp16)[name = string("K_cache_5_cast_fp16")]; tensor var_3210_begin_0 = const()[name = string("op_3210_begin_0"), val = tensor([37, 0, 0, 0])]; tensor var_3210_end_0 = const()[name = string("op_3210_end_0"), val = tensor([38, 1, 512, 512])]; tensor var_3210_end_mask_0 = const()[name = string("op_3210_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3210_squeeze_mask_0 = const()[name = string("op_3210_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_3210_cast_fp16 = slice_by_index(begin = var_3210_begin_0, end = var_3210_end_0, end_mask = var_3210_end_mask_0, squeeze_mask = var_3210_squeeze_mask_0, x = coreml_update_state_33)[name = string("op_3210_cast_fp16")]; tensor V_cache_5_axes_0 = const()[name = string("V_cache_5_axes_0"), val = tensor([0])]; tensor V_cache_5_cast_fp16 = expand_dims(axes = V_cache_5_axes_0, x = var_3210_cast_fp16)[name = string("V_cache_5_cast_fp16")]; tensor k_padded_5_pad_0 = const()[name = string("k_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_5_mode_0 = const()[name = string("k_padded_5_mode_0"), val = string("constant")]; fp16 const_45_to_fp16 = const()[name = string("const_45_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_5_cast_fp16 = pad(constant_val = const_45_to_fp16, mode = k_padded_5_mode_0, pad = k_padded_5_pad_0, x = input_69)[name = string("k_padded_5_cast_fp16")]; tensor v_padded_5_pad_0 = const()[name = string("v_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_5_mode_0 = const()[name = string("v_padded_5_mode_0"), val = string("constant")]; fp16 const_46_to_fp16 = const()[name = string("const_46_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_5_cast_fp16 = pad(constant_val = const_46_to_fp16, mode = v_padded_5_mode_0, pad = v_padded_5_pad_0, x = var_3187_cast_fp16)[name = string("v_padded_5_cast_fp16")]; tensor var_3228_cast_fp16 = mul(x = K_cache_5_cast_fp16, y = var_2187_cast_fp16)[name = string("op_3228_cast_fp16")]; tensor var_3229_reps_0 = const()[name = string("op_3229_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_3229_cast_fp16 = tile(reps = var_3229_reps_0, x = k_padded_5_cast_fp16)[name = string("op_3229_cast_fp16")]; tensor var_3230_cast_fp16 = mul(x = var_3229_cast_fp16, y = update_mask)[name = string("op_3230_cast_fp16")]; tensor K_new_5_cast_fp16 = add(x = var_3228_cast_fp16, y = var_3230_cast_fp16)[name = string("K_new_5_cast_fp16")]; tensor var_3236_cast_fp16 = mul(x = V_cache_5_cast_fp16, y = var_2187_cast_fp16)[name = string("op_3236_cast_fp16")]; tensor var_3237_reps_0 = const()[name = string("op_3237_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_3237_cast_fp16 = tile(reps = var_3237_reps_0, x = v_padded_5_cast_fp16)[name = string("op_3237_cast_fp16")]; tensor var_3238_cast_fp16 = mul(x = var_3237_cast_fp16, y = update_mask)[name = string("op_3238_cast_fp16")]; tensor V_new_5_cast_fp16 = add(x = var_3236_cast_fp16, y = var_3238_cast_fp16)[name = string("V_new_5_cast_fp16")]; tensor var_3242_axes_0 = const()[name = string("op_3242_axes_0"), val = tensor([0])]; tensor var_3242_cast_fp16 = squeeze(axes = var_3242_axes_0, x = K_new_5_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor concat_16 = const()[name = string("concat_16"), val = tensor([2, 0, 0, 0])]; tensor concat_17 = const()[name = string("concat_17"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_16, begin_mask = kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_17, end_mask = kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_5_stride_0, update = var_3242_cast_fp16, x = coreml_update_state_33)[name = string("kv_cache_0_internal_tensor_assign_5_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_5_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_34_write_state")]; tensor coreml_update_state_34 = read_state(input = kv_cache_0)[name = string("coreml_update_state_34")]; tensor var_3249_axes_0 = const()[name = string("op_3249_axes_0"), val = tensor([0])]; tensor var_3249_cast_fp16 = squeeze(axes = var_3249_axes_0, x = V_new_5_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor concat_18 = const()[name = string("concat_18"), val = tensor([37, 0, 0, 0])]; tensor concat_19 = const()[name = string("concat_19"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_18, begin_mask = kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_19, end_mask = kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_6_stride_0, update = var_3249_cast_fp16, x = coreml_update_state_34)[name = string("kv_cache_0_internal_tensor_assign_6_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_6_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_35_write_state")]; tensor coreml_update_state_35 = read_state(input = kv_cache_0)[name = string("coreml_update_state_35")]; tensor K_for_attn_5_begin_0 = const()[name = string("K_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_5_end_0 = const()[name = string("K_for_attn_5_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_5_end_mask_0 = const()[name = string("K_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_5_cast_fp16 = slice_by_index(begin = K_for_attn_5_begin_0, end = K_for_attn_5_end_0, end_mask = K_for_attn_5_end_mask_0, x = K_new_5_cast_fp16)[name = string("K_for_attn_5_cast_fp16")]; tensor V_for_attn_5_begin_0 = const()[name = string("V_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_5_end_0 = const()[name = string("V_for_attn_5_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_5_end_mask_0 = const()[name = string("V_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_5_cast_fp16 = slice_by_index(begin = V_for_attn_5_begin_0, end = V_for_attn_5_end_0, end_mask = V_for_attn_5_end_mask_0, x = V_new_5_cast_fp16)[name = string("V_for_attn_5_cast_fp16")]; tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = K_for_attn_5_cast_fp16)[name = string("transpose_332")]; tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; tensor concat_20 = const()[name = string("concat_20"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_20, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_21 = const()[name = string("concat_21"), val = tensor([-1, 1, 512, 256])]; tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_331")]; tensor reshape_9_cast_fp16 = reshape(shape = concat_21, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor transpose_142_perm_0 = const()[name = string("transpose_142_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = V_for_attn_5_cast_fp16)[name = string("transpose_330")]; tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; tensor concat_22 = const()[name = string("concat_22"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_22, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_23 = const()[name = string("concat_23"), val = tensor([-1, 1, 512, 256])]; tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_329")]; tensor reshape_11_cast_fp16 = reshape(shape = concat_23, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor V_expanded_5_perm_0 = const()[name = string("V_expanded_5_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_3276_transpose_x_0 = const()[name = string("op_3276_transpose_x_0"), val = bool(false)]; bool var_3276_transpose_y_0 = const()[name = string("op_3276_transpose_y_0"), val = bool(false)]; tensor transpose_142_cast_fp16 = transpose(perm = transpose_142_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_328")]; tensor var_3276_cast_fp16 = matmul(transpose_x = var_3276_transpose_x_0, transpose_y = var_3276_transpose_y_0, x = q_23, y = transpose_142_cast_fp16)[name = string("op_3276_cast_fp16")]; tensor attn_weights_15_cast_fp16 = add(x = var_3276_cast_fp16, y = causal_mask)[name = string("attn_weights_15_cast_fp16")]; int32 var_3281 = const()[name = string("op_3281"), val = int32(-1)]; tensor attn_weights_17_cast_fp16 = softmax(axis = var_3281, x = attn_weights_15_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; tensor V_expanded_5_cast_fp16 = transpose(perm = V_expanded_5_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_327")]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = attn_weights_17_cast_fp16, y = V_expanded_5_cast_fp16)[name = string("attn_output_13_cast_fp16")]; tensor var_3289 = const()[name = string("op_3289"), val = tensor([0, 2, 1, 3])]; tensor var_3296 = const()[name = string("op_3296"), val = tensor([1, 1, -1])]; tensor var_3290_cast_fp16 = transpose(perm = var_3289, x = attn_output_13_cast_fp16)[name = string("transpose_326")]; tensor attn_output_15_cast_fp16 = reshape(shape = var_3296, x = var_3290_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor var_3301 = const()[name = string("op_3301"), val = tensor([0, 2, 1])]; string var_3317_pad_type_0 = const()[name = string("op_3317_pad_type_0"), val = string("valid")]; int32 var_3317_groups_0 = const()[name = string("op_3317_groups_0"), val = int32(1)]; tensor var_3317_strides_0 = const()[name = string("op_3317_strides_0"), val = tensor([1])]; tensor var_3317_pad_0 = const()[name = string("op_3317_pad_0"), val = tensor([0, 0])]; tensor var_3317_dilations_0 = const()[name = string("op_3317_dilations_0"), val = tensor([1])]; tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1065414400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1066987328))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3302_cast_fp16 = transpose(perm = var_3301, x = attn_output_15_cast_fp16)[name = string("transpose_325")]; tensor var_3317_cast_fp16 = conv(dilations = var_3317_dilations_0, groups = var_3317_groups_0, pad = var_3317_pad_0, pad_type = var_3317_pad_type_0, strides = var_3317_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_3302_cast_fp16)[name = string("op_3317_cast_fp16")]; tensor var_3321 = const()[name = string("op_3321"), val = tensor([0, 2, 1])]; int32 var_3327 = const()[name = string("op_3327"), val = int32(-1)]; fp16 const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_75_cast_fp16 = transpose(perm = var_3321, x = var_3317_cast_fp16)[name = string("transpose_324")]; tensor var_3333_cast_fp16 = mul(x = x_75_cast_fp16, y = const_47_promoted_to_fp16)[name = string("op_3333_cast_fp16")]; bool input_75_interleave_0 = const()[name = string("input_75_interleave_0"), val = bool(false)]; tensor input_75_cast_fp16 = concat(axis = var_3327, interleave = input_75_interleave_0, values = (x_75_cast_fp16, var_3333_cast_fp16))[name = string("input_75_cast_fp16")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_3325_to_fp16 = const()[name = string("op_3325_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_3325_to_fp16, x = input_75_cast_fp16)[name = string("normed_69_cast_fp16")]; tensor var_3338_split_sizes_0 = const()[name = string("op_3338_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3338_axis_0 = const()[name = string("op_3338_axis_0"), val = int32(-1)]; tensor var_3338_cast_fp16_0, tensor var_3338_cast_fp16_1 = split(axis = var_3338_axis_0, split_sizes = var_3338_split_sizes_0, x = normed_69_cast_fp16)[name = string("op_3338_cast_fp16")]; tensor const_48_to_fp16 = const()[name = string("const_48_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1066988928)))]; tensor var_3341_cast_fp16 = mul(x = var_3338_cast_fp16_0, y = const_48_to_fp16)[name = string("op_3341_cast_fp16")]; tensor x_79_cast_fp16 = add(x = x_61_cast_fp16, y = var_3341_cast_fp16)[name = string("x_79_cast_fp16")]; int32 var_3348 = const()[name = string("op_3348"), val = int32(-1)]; fp16 const_49_promoted_to_fp16 = const()[name = string("const_49_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3354_cast_fp16 = mul(x = x_79_cast_fp16, y = const_49_promoted_to_fp16)[name = string("op_3354_cast_fp16")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77_cast_fp16 = concat(axis = var_3348, interleave = input_77_interleave_0, values = (x_79_cast_fp16, var_3354_cast_fp16))[name = string("input_77_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_3346_to_fp16 = const()[name = string("op_3346_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_3346_to_fp16, x = input_77_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor var_3359_split_sizes_0 = const()[name = string("op_3359_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3359_axis_0 = const()[name = string("op_3359_axis_0"), val = int32(-1)]; tensor var_3359_cast_fp16_0, tensor var_3359_cast_fp16_1 = split(axis = var_3359_axis_0, split_sizes = var_3359_split_sizes_0, x = normed_73_cast_fp16)[name = string("op_3359_cast_fp16")]; tensor const_50_to_fp16 = const()[name = string("const_50_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1066992064)))]; tensor var_3362_cast_fp16 = mul(x = var_3359_cast_fp16_0, y = const_50_to_fp16)[name = string("op_3362_cast_fp16")]; tensor var_3375 = const()[name = string("op_3375"), val = tensor([0, 2, 1])]; tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; tensor var_3376 = transpose(perm = var_3375, x = var_3362_cast_fp16)[name = string("transpose_323")]; tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_3376)[name = string("input_79")]; string gate_9_pad_type_0 = const()[name = string("gate_9_pad_type_0"), val = string("valid")]; tensor gate_9_strides_0 = const()[name = string("gate_9_strides_0"), val = tensor([1, 1])]; tensor gate_9_pad_0 = const()[name = string("gate_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_9_dilations_0 = const()[name = string("gate_9_dilations_0"), val = tensor([1, 1])]; int32 gate_9_groups_0 = const()[name = string("gate_9_groups_0"), val = int32(1)]; tensor gate_9 = conv(dilations = gate_9_dilations_0, groups = gate_9_groups_0, pad = gate_9_pad_0, pad_type = gate_9_pad_type_0, strides = gate_9_strides_0, weight = layers_2_mlp_gate_proj_weight_palettized, x = input_79)[name = string("gate_9")]; string up_5_pad_type_0 = const()[name = string("up_5_pad_type_0"), val = string("valid")]; tensor up_5_strides_0 = const()[name = string("up_5_strides_0"), val = tensor([1, 1])]; tensor up_5_pad_0 = const()[name = string("up_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_5_dilations_0 = const()[name = string("up_5_dilations_0"), val = tensor([1, 1])]; int32 up_5_groups_0 = const()[name = string("up_5_groups_0"), val = int32(1)]; tensor up_5 = conv(dilations = up_5_dilations_0, groups = up_5_groups_0, pad = up_5_pad_0, pad_type = up_5_pad_type_0, strides = up_5_strides_0, weight = layers_2_mlp_up_proj_weight_palettized, x = input_79)[name = string("up_5")]; string gate_11_mode_0 = const()[name = string("gate_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_11 = gelu(mode = gate_11_mode_0, x = gate_9)[name = string("gate_11")]; tensor input_81 = mul(x = gate_11, y = up_5)[name = string("input_81")]; string mlp_out_5_pad_type_0 = const()[name = string("mlp_out_5_pad_type_0"), val = string("valid")]; tensor mlp_out_5_strides_0 = const()[name = string("mlp_out_5_strides_0"), val = tensor([1, 1])]; tensor mlp_out_5_pad_0 = const()[name = string("mlp_out_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_5_dilations_0 = const()[name = string("mlp_out_5_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_5_groups_0 = const()[name = string("mlp_out_5_groups_0"), val = int32(1)]; tensor mlp_out_5 = conv(dilations = mlp_out_5_dilations_0, groups = mlp_out_5_groups_0, pad = mlp_out_5_pad_0, pad_type = mlp_out_5_pad_type_0, strides = mlp_out_5_strides_0, weight = layers_2_mlp_down_proj_weight_palettized, x = input_81)[name = string("mlp_out_5")]; tensor var_3416_axes_0 = const()[name = string("op_3416_axes_0"), val = tensor([2])]; tensor var_3416 = squeeze(axes = var_3416_axes_0, x = mlp_out_5)[name = string("op_3416")]; tensor var_3420 = const()[name = string("op_3420"), val = tensor([0, 2, 1])]; int32 var_3426 = const()[name = string("op_3426"), val = int32(-1)]; fp16 const_51_promoted_to_fp16 = const()[name = string("const_51_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_83 = transpose(perm = var_3420, x = var_3416)[name = string("transpose_322")]; tensor var_3432_cast_fp16 = mul(x = x_83, y = const_51_promoted_to_fp16)[name = string("op_3432_cast_fp16")]; bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; tensor input_83_cast_fp16 = concat(axis = var_3426, interleave = input_83_interleave_0, values = (x_83, var_3432_cast_fp16))[name = string("input_83_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_3424_to_fp16 = const()[name = string("op_3424_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3424_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor var_3437_split_sizes_0 = const()[name = string("op_3437_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3437_axis_0 = const()[name = string("op_3437_axis_0"), val = int32(-1)]; tensor var_3437_cast_fp16_0, tensor var_3437_cast_fp16_1 = split(axis = var_3437_axis_0, split_sizes = var_3437_split_sizes_0, x = normed_77_cast_fp16)[name = string("op_3437_cast_fp16")]; tensor const_52_to_fp16 = const()[name = string("const_52_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1066995200)))]; tensor var_3440_cast_fp16 = mul(x = var_3437_cast_fp16_0, y = const_52_to_fp16)[name = string("op_3440_cast_fp16")]; tensor hidden_states_33_cast_fp16 = add(x = x_79_cast_fp16, y = var_3440_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor per_layer_slice_5_begin_0 = const()[name = string("per_layer_slice_5_begin_0"), val = tensor([0, 0, 512])]; tensor per_layer_slice_5_end_0 = const()[name = string("per_layer_slice_5_end_0"), val = tensor([1, 1, 768])]; tensor per_layer_slice_5_end_mask_0 = const()[name = string("per_layer_slice_5_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_5_cast_fp16 = slice_by_index(begin = per_layer_slice_5_begin_0, end = per_layer_slice_5_end_0, end_mask = per_layer_slice_5_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_5_cast_fp16")]; tensor gated_9 = linear(bias = linear_0_bias_0, weight = layers_2_per_layer_input_gate_weight_palettized, x = hidden_states_33_cast_fp16)[name = string("linear_4")]; string gated_11_mode_0 = const()[name = string("gated_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_11 = gelu(mode = gated_11_mode_0, x = gated_9)[name = string("gated_11")]; tensor input_87_cast_fp16 = mul(x = gated_11, y = per_layer_slice_5_cast_fp16)[name = string("input_87_cast_fp16")]; tensor layers_2_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1066998336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1067195008))))[name = string("layers_2_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_5_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_2_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_87_cast_fp16)[name = string("linear_5_cast_fp16")]; int32 var_3477 = const()[name = string("op_3477"), val = int32(-1)]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3483_cast_fp16 = mul(x = linear_5_cast_fp16, y = const_53_promoted_to_fp16)[name = string("op_3483_cast_fp16")]; bool input_89_interleave_0 = const()[name = string("input_89_interleave_0"), val = bool(false)]; tensor input_89_cast_fp16 = concat(axis = var_3477, interleave = input_89_interleave_0, values = (linear_5_cast_fp16, var_3483_cast_fp16))[name = string("input_89_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_3475_to_fp16 = const()[name = string("op_3475_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3475_to_fp16, x = input_89_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor var_3488_split_sizes_0 = const()[name = string("op_3488_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3488_axis_0 = const()[name = string("op_3488_axis_0"), val = int32(-1)]; tensor var_3488_cast_fp16_0, tensor var_3488_cast_fp16_1 = split(axis = var_3488_axis_0, split_sizes = var_3488_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_3488_cast_fp16")]; tensor const_54_to_fp16 = const()[name = string("const_54_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1067196608)))]; tensor var_3491_cast_fp16 = mul(x = var_3488_cast_fp16_0, y = const_54_to_fp16)[name = string("op_3491_cast_fp16")]; tensor hidden_states_37_cast_fp16 = add(x = hidden_states_33_cast_fp16, y = var_3491_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor layers_2_layer_scalar_to_fp16 = const()[name = string("layers_2_layer_scalar_to_fp16"), val = tensor([0x1.96p-1])]; tensor x_91_cast_fp16 = mul(x = hidden_states_37_cast_fp16, y = layers_2_layer_scalar_to_fp16)[name = string("x_91_cast_fp16")]; int32 var_3499 = const()[name = string("op_3499"), val = int32(-1)]; fp16 const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3505_cast_fp16 = mul(x = x_91_cast_fp16, y = const_55_promoted_to_fp16)[name = string("op_3505_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_3499, interleave = input_91_interleave_0, values = (x_91_cast_fp16, var_3505_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_3497_to_fp16 = const()[name = string("op_3497_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3497_to_fp16, x = input_91_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_3510_split_sizes_0 = const()[name = string("op_3510_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3510_axis_0 = const()[name = string("op_3510_axis_0"), val = int32(-1)]; tensor var_3510_cast_fp16_0, tensor var_3510_cast_fp16_1 = split(axis = var_3510_axis_0, split_sizes = var_3510_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_3510_cast_fp16")]; tensor const_56_to_fp16 = const()[name = string("const_56_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1067199744)))]; tensor var_3513_cast_fp16 = mul(x = var_3510_cast_fp16_0, y = const_56_to_fp16)[name = string("op_3513_cast_fp16")]; tensor var_3521 = const()[name = string("op_3521"), val = tensor([0, 2, 1])]; tensor var_3524_axes_0 = const()[name = string("op_3524_axes_0"), val = tensor([2])]; tensor var_3522_cast_fp16 = transpose(perm = var_3521, x = var_3513_cast_fp16)[name = string("transpose_321")]; tensor var_3524_cast_fp16 = expand_dims(axes = var_3524_axes_0, x = var_3522_cast_fp16)[name = string("op_3524_cast_fp16")]; string var_3540_pad_type_0 = const()[name = string("op_3540_pad_type_0"), val = string("valid")]; tensor var_3540_strides_0 = const()[name = string("op_3540_strides_0"), val = tensor([1, 1])]; tensor var_3540_pad_0 = const()[name = string("op_3540_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3540_dilations_0 = const()[name = string("op_3540_dilations_0"), val = tensor([1, 1])]; int32 var_3540_groups_0 = const()[name = string("op_3540_groups_0"), val = int32(1)]; tensor var_3540 = conv(dilations = var_3540_dilations_0, groups = var_3540_groups_0, pad = var_3540_pad_0, pad_type = var_3540_pad_type_0, strides = var_3540_strides_0, weight = layers_3_self_attn_q_proj_weight_palettized, x = var_3524_cast_fp16)[name = string("op_3540")]; tensor var_3545 = const()[name = string("op_3545"), val = tensor([1, 8, 256, 1])]; tensor var_3546 = reshape(shape = var_3545, x = var_3540)[name = string("op_3546")]; tensor var_3551 = const()[name = string("op_3551"), val = tensor([0, 1, 3, 2])]; tensor var_3561 = const()[name = string("op_3561"), val = tensor([1, 8, 256])]; tensor var_3552 = transpose(perm = var_3551, x = var_3546)[name = string("transpose_320")]; tensor x_95 = reshape(shape = var_3561, x = var_3552)[name = string("x_95")]; int32 var_3567 = const()[name = string("op_3567"), val = int32(-1)]; fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3573_cast_fp16 = mul(x = x_95, y = const_57_promoted_to_fp16)[name = string("op_3573_cast_fp16")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95_cast_fp16 = concat(axis = var_3567, interleave = input_95_interleave_0, values = (x_95, var_3573_cast_fp16))[name = string("input_95_cast_fp16")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_3565_to_fp16 = const()[name = string("op_3565_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3565_to_fp16, x = input_95_cast_fp16)[name = string("normed_89_cast_fp16")]; tensor var_3578_split_sizes_0 = const()[name = string("op_3578_split_sizes_0"), val = tensor([256, 256])]; int32 var_3578_axis_0 = const()[name = string("op_3578_axis_0"), val = int32(-1)]; tensor var_3578_cast_fp16_0, tensor var_3578_cast_fp16_1 = split(axis = var_3578_axis_0, split_sizes = var_3578_split_sizes_0, x = normed_89_cast_fp16)[name = string("op_3578_cast_fp16")]; tensor const_58_to_fp16 = const()[name = string("const_58_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1067202880)))]; tensor var_3581_cast_fp16 = mul(x = var_3578_cast_fp16_0, y = const_58_to_fp16)[name = string("op_3581_cast_fp16")]; tensor var_3587 = const()[name = string("op_3587"), val = tensor([1, 8, 1, 256])]; tensor q_27 = reshape(shape = var_3587, x = var_3581_cast_fp16)[name = string("q_27")]; tensor var_3589 = mul(x = q_27, y = cos_1)[name = string("op_3589")]; tensor var_3590_split_sizes_0 = const()[name = string("op_3590_split_sizes_0"), val = tensor([128, 128])]; int32 var_3590_axis_0 = const()[name = string("op_3590_axis_0"), val = int32(-1)]; tensor var_3590_0, tensor var_3590_1 = split(axis = var_3590_axis_0, split_sizes = var_3590_split_sizes_0, x = q_27)[name = string("op_3590")]; fp16 const_59_promoted = const()[name = string("const_59_promoted"), val = fp16(-0x1p+0)]; tensor var_3592 = mul(x = var_3590_1, y = const_59_promoted)[name = string("op_3592")]; int32 var_3594 = const()[name = string("op_3594"), val = int32(-1)]; bool var_3595_interleave_0 = const()[name = string("op_3595_interleave_0"), val = bool(false)]; tensor var_3595 = concat(axis = var_3594, interleave = var_3595_interleave_0, values = (var_3592, var_3590_0))[name = string("op_3595")]; tensor var_3596 = mul(x = var_3595, y = sin_1)[name = string("op_3596")]; tensor q_31 = add(x = var_3589, y = var_3596)[name = string("q_31")]; string var_3609_pad_type_0 = const()[name = string("op_3609_pad_type_0"), val = string("valid")]; tensor var_3609_strides_0 = const()[name = string("op_3609_strides_0"), val = tensor([1, 1])]; tensor var_3609_pad_0 = const()[name = string("op_3609_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3609_dilations_0 = const()[name = string("op_3609_dilations_0"), val = tensor([1, 1])]; int32 var_3609_groups_0 = const()[name = string("op_3609_groups_0"), val = int32(1)]; tensor var_3609 = conv(dilations = var_3609_dilations_0, groups = var_3609_groups_0, pad = var_3609_pad_0, pad_type = var_3609_pad_type_0, strides = var_3609_strides_0, weight = layers_3_self_attn_k_proj_weight_palettized, x = var_3524_cast_fp16)[name = string("op_3609")]; tensor var_3614 = const()[name = string("op_3614"), val = tensor([1, 1, 256, 1])]; tensor var_3615 = reshape(shape = var_3614, x = var_3609)[name = string("op_3615")]; tensor var_3620 = const()[name = string("op_3620"), val = tensor([0, 1, 3, 2])]; string var_3637_pad_type_0 = const()[name = string("op_3637_pad_type_0"), val = string("valid")]; tensor var_3637_strides_0 = const()[name = string("op_3637_strides_0"), val = tensor([1, 1])]; tensor var_3637_pad_0 = const()[name = string("op_3637_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3637_dilations_0 = const()[name = string("op_3637_dilations_0"), val = tensor([1, 1])]; int32 var_3637_groups_0 = const()[name = string("op_3637_groups_0"), val = int32(1)]; tensor var_3637 = conv(dilations = var_3637_dilations_0, groups = var_3637_groups_0, pad = var_3637_pad_0, pad_type = var_3637_pad_type_0, strides = var_3637_strides_0, weight = layers_3_self_attn_v_proj_weight_palettized, x = var_3524_cast_fp16)[name = string("op_3637")]; tensor var_3642 = const()[name = string("op_3642"), val = tensor([1, 1, 256, 1])]; tensor var_3643 = reshape(shape = var_3642, x = var_3637)[name = string("op_3643")]; tensor var_3648 = const()[name = string("op_3648"), val = tensor([0, 1, 3, 2])]; tensor var_3658 = const()[name = string("op_3658"), val = tensor([1, 1, 256])]; tensor var_3621 = transpose(perm = var_3620, x = var_3615)[name = string("transpose_319")]; tensor x_99 = reshape(shape = var_3658, x = var_3621)[name = string("x_99")]; int32 var_3664 = const()[name = string("op_3664"), val = int32(-1)]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3670_cast_fp16 = mul(x = x_99, y = const_60_promoted_to_fp16)[name = string("op_3670_cast_fp16")]; bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; tensor input_97_cast_fp16 = concat(axis = var_3664, interleave = input_97_interleave_0, values = (x_99, var_3670_cast_fp16))[name = string("input_97_cast_fp16")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_3662_to_fp16 = const()[name = string("op_3662_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_3662_to_fp16, x = input_97_cast_fp16)[name = string("normed_93_cast_fp16")]; tensor var_3675_split_sizes_0 = const()[name = string("op_3675_split_sizes_0"), val = tensor([256, 256])]; int32 var_3675_axis_0 = const()[name = string("op_3675_axis_0"), val = int32(-1)]; tensor var_3675_cast_fp16_0, tensor var_3675_cast_fp16_1 = split(axis = var_3675_axis_0, split_sizes = var_3675_split_sizes_0, x = normed_93_cast_fp16)[name = string("op_3675_cast_fp16")]; tensor const_61_to_fp16 = const()[name = string("const_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1067203456)))]; tensor var_3678_cast_fp16 = mul(x = var_3675_cast_fp16_0, y = const_61_to_fp16)[name = string("op_3678_cast_fp16")]; tensor var_3684 = const()[name = string("op_3684"), val = tensor([1, 1, 1, 256])]; tensor q_29 = reshape(shape = var_3684, x = var_3678_cast_fp16)[name = string("q_29")]; fp16 var_3691_promoted_to_fp16 = const()[name = string("op_3691_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_3649 = transpose(perm = var_3648, x = var_3643)[name = string("transpose_318")]; tensor var_3692_cast_fp16 = pow(x = var_3649, y = var_3691_promoted_to_fp16)[name = string("op_3692_cast_fp16")]; tensor var_3697_axes_0 = const()[name = string("op_3697_axes_0"), val = tensor([-1])]; bool var_3697_keep_dims_0 = const()[name = string("op_3697_keep_dims_0"), val = bool(true)]; tensor var_3697_cast_fp16 = reduce_mean(axes = var_3697_axes_0, keep_dims = var_3697_keep_dims_0, x = var_3692_cast_fp16)[name = string("op_3697_cast_fp16")]; fp16 var_3699_to_fp16 = const()[name = string("op_3699_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_7_cast_fp16 = add(x = var_3697_cast_fp16, y = var_3699_to_fp16)[name = string("mean_sq_7_cast_fp16")]; fp16 var_3706_to_fp16 = const()[name = string("op_3706_to_fp16"), val = fp16(-0x1p-1)]; tensor var_3707_cast_fp16 = pow(x = mean_sq_7_cast_fp16, y = var_3706_to_fp16)[name = string("op_3707_cast_fp16")]; tensor var_3708_cast_fp16 = mul(x = var_3649, y = var_3707_cast_fp16)[name = string("op_3708_cast_fp16")]; tensor var_3714 = mul(x = q_29, y = cos_1)[name = string("op_3714")]; tensor var_3715_split_sizes_0 = const()[name = string("op_3715_split_sizes_0"), val = tensor([128, 128])]; int32 var_3715_axis_0 = const()[name = string("op_3715_axis_0"), val = int32(-1)]; tensor var_3715_0, tensor var_3715_1 = split(axis = var_3715_axis_0, split_sizes = var_3715_split_sizes_0, x = q_29)[name = string("op_3715")]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_3717 = mul(x = var_3715_1, y = const_62_promoted)[name = string("op_3717")]; int32 var_3719 = const()[name = string("op_3719"), val = int32(-1)]; bool var_3720_interleave_0 = const()[name = string("op_3720_interleave_0"), val = bool(false)]; tensor var_3720 = concat(axis = var_3719, interleave = var_3720_interleave_0, values = (var_3717, var_3715_0))[name = string("op_3720")]; tensor var_3721 = mul(x = var_3720, y = sin_1)[name = string("op_3721")]; tensor input_99 = add(x = var_3714, y = var_3721)[name = string("input_99")]; tensor var_3726_begin_0 = const()[name = string("op_3726_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_3726_end_0 = const()[name = string("op_3726_end_0"), val = tensor([4, 1, 512, 512])]; tensor var_3726_end_mask_0 = const()[name = string("op_3726_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3726_squeeze_mask_0 = const()[name = string("op_3726_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_3726_cast_fp16 = slice_by_index(begin = var_3726_begin_0, end = var_3726_end_0, end_mask = var_3726_end_mask_0, squeeze_mask = var_3726_squeeze_mask_0, x = coreml_update_state_35)[name = string("op_3726_cast_fp16")]; tensor K_cache_7_axes_0 = const()[name = string("K_cache_7_axes_0"), val = tensor([0])]; tensor K_cache_7_cast_fp16 = expand_dims(axes = K_cache_7_axes_0, x = var_3726_cast_fp16)[name = string("K_cache_7_cast_fp16")]; tensor var_3731_begin_0 = const()[name = string("op_3731_begin_0"), val = tensor([38, 0, 0, 0])]; tensor var_3731_end_0 = const()[name = string("op_3731_end_0"), val = tensor([39, 1, 512, 512])]; tensor var_3731_end_mask_0 = const()[name = string("op_3731_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3731_squeeze_mask_0 = const()[name = string("op_3731_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_3731_cast_fp16 = slice_by_index(begin = var_3731_begin_0, end = var_3731_end_0, end_mask = var_3731_end_mask_0, squeeze_mask = var_3731_squeeze_mask_0, x = coreml_update_state_35)[name = string("op_3731_cast_fp16")]; tensor V_cache_7_axes_0 = const()[name = string("V_cache_7_axes_0"), val = tensor([0])]; tensor V_cache_7_cast_fp16 = expand_dims(axes = V_cache_7_axes_0, x = var_3731_cast_fp16)[name = string("V_cache_7_cast_fp16")]; tensor k_padded_7_pad_0 = const()[name = string("k_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_7_mode_0 = const()[name = string("k_padded_7_mode_0"), val = string("constant")]; fp16 const_63_to_fp16 = const()[name = string("const_63_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_7_cast_fp16 = pad(constant_val = const_63_to_fp16, mode = k_padded_7_mode_0, pad = k_padded_7_pad_0, x = input_99)[name = string("k_padded_7_cast_fp16")]; tensor v_padded_7_pad_0 = const()[name = string("v_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_7_mode_0 = const()[name = string("v_padded_7_mode_0"), val = string("constant")]; fp16 const_64_to_fp16 = const()[name = string("const_64_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_7_cast_fp16 = pad(constant_val = const_64_to_fp16, mode = v_padded_7_mode_0, pad = v_padded_7_pad_0, x = var_3708_cast_fp16)[name = string("v_padded_7_cast_fp16")]; tensor var_3749_cast_fp16 = mul(x = K_cache_7_cast_fp16, y = var_2187_cast_fp16)[name = string("op_3749_cast_fp16")]; tensor var_3750_reps_0 = const()[name = string("op_3750_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_3750_cast_fp16 = tile(reps = var_3750_reps_0, x = k_padded_7_cast_fp16)[name = string("op_3750_cast_fp16")]; tensor var_3751_cast_fp16 = mul(x = var_3750_cast_fp16, y = update_mask)[name = string("op_3751_cast_fp16")]; tensor K_new_7_cast_fp16 = add(x = var_3749_cast_fp16, y = var_3751_cast_fp16)[name = string("K_new_7_cast_fp16")]; tensor var_3757_cast_fp16 = mul(x = V_cache_7_cast_fp16, y = var_2187_cast_fp16)[name = string("op_3757_cast_fp16")]; tensor var_3758_reps_0 = const()[name = string("op_3758_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_3758_cast_fp16 = tile(reps = var_3758_reps_0, x = v_padded_7_cast_fp16)[name = string("op_3758_cast_fp16")]; tensor var_3759_cast_fp16 = mul(x = var_3758_cast_fp16, y = update_mask)[name = string("op_3759_cast_fp16")]; tensor V_new_7_cast_fp16 = add(x = var_3757_cast_fp16, y = var_3759_cast_fp16)[name = string("V_new_7_cast_fp16")]; tensor var_3763_axes_0 = const()[name = string("op_3763_axes_0"), val = tensor([0])]; tensor var_3763_cast_fp16 = squeeze(axes = var_3763_axes_0, x = K_new_7_cast_fp16)[name = string("op_3763_cast_fp16")]; tensor concat_24 = const()[name = string("concat_24"), val = tensor([3, 0, 0, 0])]; tensor concat_25 = const()[name = string("concat_25"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_24, begin_mask = kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_25, end_mask = kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_7_stride_0, update = var_3763_cast_fp16, x = coreml_update_state_35)[name = string("kv_cache_0_internal_tensor_assign_7_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_7_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_36_write_state")]; tensor coreml_update_state_36 = read_state(input = kv_cache_0)[name = string("coreml_update_state_36")]; tensor var_3770_axes_0 = const()[name = string("op_3770_axes_0"), val = tensor([0])]; tensor var_3770_cast_fp16 = squeeze(axes = var_3770_axes_0, x = V_new_7_cast_fp16)[name = string("op_3770_cast_fp16")]; tensor concat_26 = const()[name = string("concat_26"), val = tensor([38, 0, 0, 0])]; tensor concat_27 = const()[name = string("concat_27"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_26, begin_mask = kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_27, end_mask = kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_8_stride_0, update = var_3770_cast_fp16, x = coreml_update_state_36)[name = string("kv_cache_0_internal_tensor_assign_8_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_8_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_37_write_state")]; tensor coreml_update_state_37 = read_state(input = kv_cache_0)[name = string("coreml_update_state_37")]; tensor K_for_attn_7_begin_0 = const()[name = string("K_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_7_end_0 = const()[name = string("K_for_attn_7_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_7_end_mask_0 = const()[name = string("K_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_7_cast_fp16 = slice_by_index(begin = K_for_attn_7_begin_0, end = K_for_attn_7_end_0, end_mask = K_for_attn_7_end_mask_0, x = K_new_7_cast_fp16)[name = string("K_for_attn_7_cast_fp16")]; tensor V_for_attn_7_begin_0 = const()[name = string("V_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_7_end_0 = const()[name = string("V_for_attn_7_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_7_end_mask_0 = const()[name = string("V_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_7_cast_fp16 = slice_by_index(begin = V_for_attn_7_begin_0, end = V_for_attn_7_end_0, end_mask = V_for_attn_7_end_mask_0, x = V_new_7_cast_fp16)[name = string("V_for_attn_7_cast_fp16")]; tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = K_for_attn_7_cast_fp16)[name = string("transpose_317")]; tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; tensor concat_28 = const()[name = string("concat_28"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_28, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_29 = const()[name = string("concat_29"), val = tensor([-1, 1, 512, 256])]; tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_316")]; tensor reshape_13_cast_fp16 = reshape(shape = concat_29, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; tensor transpose_143_perm_0 = const()[name = string("transpose_143_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = V_for_attn_7_cast_fp16)[name = string("transpose_315")]; tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; tensor concat_30 = const()[name = string("concat_30"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_30, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_31 = const()[name = string("concat_31"), val = tensor([-1, 1, 512, 256])]; tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_314")]; tensor reshape_15_cast_fp16 = reshape(shape = concat_31, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor V_expanded_7_perm_0 = const()[name = string("V_expanded_7_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_3797_transpose_x_0 = const()[name = string("op_3797_transpose_x_0"), val = bool(false)]; bool var_3797_transpose_y_0 = const()[name = string("op_3797_transpose_y_0"), val = bool(false)]; tensor transpose_143_cast_fp16 = transpose(perm = transpose_143_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_313")]; tensor var_3797_cast_fp16 = matmul(transpose_x = var_3797_transpose_x_0, transpose_y = var_3797_transpose_y_0, x = q_31, y = transpose_143_cast_fp16)[name = string("op_3797_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = var_3797_cast_fp16, y = causal_mask)[name = string("attn_weights_21_cast_fp16")]; int32 var_3802 = const()[name = string("op_3802"), val = int32(-1)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_3802, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool attn_output_19_transpose_x_0 = const()[name = string("attn_output_19_transpose_x_0"), val = bool(false)]; bool attn_output_19_transpose_y_0 = const()[name = string("attn_output_19_transpose_y_0"), val = bool(false)]; tensor V_expanded_7_cast_fp16 = transpose(perm = V_expanded_7_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_312")]; tensor attn_output_19_cast_fp16 = matmul(transpose_x = attn_output_19_transpose_x_0, transpose_y = attn_output_19_transpose_y_0, x = attn_weights_23_cast_fp16, y = V_expanded_7_cast_fp16)[name = string("attn_output_19_cast_fp16")]; tensor var_3810 = const()[name = string("op_3810"), val = tensor([0, 2, 1, 3])]; tensor var_3817 = const()[name = string("op_3817"), val = tensor([1, 1, -1])]; tensor var_3811_cast_fp16 = transpose(perm = var_3810, x = attn_output_19_cast_fp16)[name = string("transpose_311")]; tensor attn_output_21_cast_fp16 = reshape(shape = var_3817, x = var_3811_cast_fp16)[name = string("attn_output_21_cast_fp16")]; tensor var_3822 = const()[name = string("op_3822"), val = tensor([0, 2, 1])]; string var_3838_pad_type_0 = const()[name = string("op_3838_pad_type_0"), val = string("valid")]; int32 var_3838_groups_0 = const()[name = string("op_3838_groups_0"), val = int32(1)]; tensor var_3838_strides_0 = const()[name = string("op_3838_strides_0"), val = tensor([1])]; tensor var_3838_pad_0 = const()[name = string("op_3838_pad_0"), val = tensor([0, 0])]; tensor var_3838_dilations_0 = const()[name = string("op_3838_dilations_0"), val = tensor([1])]; tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1067204032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068776960))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3823_cast_fp16 = transpose(perm = var_3822, x = attn_output_21_cast_fp16)[name = string("transpose_310")]; tensor var_3838_cast_fp16 = conv(dilations = var_3838_dilations_0, groups = var_3838_groups_0, pad = var_3838_pad_0, pad_type = var_3838_pad_type_0, strides = var_3838_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_3823_cast_fp16)[name = string("op_3838_cast_fp16")]; tensor var_3842 = const()[name = string("op_3842"), val = tensor([0, 2, 1])]; int32 var_3848 = const()[name = string("op_3848"), val = int32(-1)]; fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_105_cast_fp16 = transpose(perm = var_3842, x = var_3838_cast_fp16)[name = string("transpose_309")]; tensor var_3854_cast_fp16 = mul(x = x_105_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_3854_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_3848, interleave = input_105_interleave_0, values = (x_105_cast_fp16, var_3854_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_3846_to_fp16 = const()[name = string("op_3846_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_3846_to_fp16, x = input_105_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor var_3859_split_sizes_0 = const()[name = string("op_3859_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3859_axis_0 = const()[name = string("op_3859_axis_0"), val = int32(-1)]; tensor var_3859_cast_fp16_0, tensor var_3859_cast_fp16_1 = split(axis = var_3859_axis_0, split_sizes = var_3859_split_sizes_0, x = normed_97_cast_fp16)[name = string("op_3859_cast_fp16")]; tensor const_66_to_fp16 = const()[name = string("const_66_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068778560)))]; tensor var_3862_cast_fp16 = mul(x = var_3859_cast_fp16_0, y = const_66_to_fp16)[name = string("op_3862_cast_fp16")]; tensor x_109_cast_fp16 = add(x = x_91_cast_fp16, y = var_3862_cast_fp16)[name = string("x_109_cast_fp16")]; int32 var_3869 = const()[name = string("op_3869"), val = int32(-1)]; fp16 const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3875_cast_fp16 = mul(x = x_109_cast_fp16, y = const_67_promoted_to_fp16)[name = string("op_3875_cast_fp16")]; bool input_107_interleave_0 = const()[name = string("input_107_interleave_0"), val = bool(false)]; tensor input_107_cast_fp16 = concat(axis = var_3869, interleave = input_107_interleave_0, values = (x_109_cast_fp16, var_3875_cast_fp16))[name = string("input_107_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_3867_to_fp16 = const()[name = string("op_3867_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_3867_to_fp16, x = input_107_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor var_3880_split_sizes_0 = const()[name = string("op_3880_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3880_axis_0 = const()[name = string("op_3880_axis_0"), val = int32(-1)]; tensor var_3880_cast_fp16_0, tensor var_3880_cast_fp16_1 = split(axis = var_3880_axis_0, split_sizes = var_3880_split_sizes_0, x = normed_101_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor const_68_to_fp16 = const()[name = string("const_68_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068781696)))]; tensor var_3883_cast_fp16 = mul(x = var_3880_cast_fp16_0, y = const_68_to_fp16)[name = string("op_3883_cast_fp16")]; tensor var_3896 = const()[name = string("op_3896"), val = tensor([0, 2, 1])]; tensor input_109_axes_0 = const()[name = string("input_109_axes_0"), val = tensor([2])]; tensor var_3897 = transpose(perm = var_3896, x = var_3883_cast_fp16)[name = string("transpose_308")]; tensor input_109 = expand_dims(axes = input_109_axes_0, x = var_3897)[name = string("input_109")]; string gate_13_pad_type_0 = const()[name = string("gate_13_pad_type_0"), val = string("valid")]; tensor gate_13_strides_0 = const()[name = string("gate_13_strides_0"), val = tensor([1, 1])]; tensor gate_13_pad_0 = const()[name = string("gate_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_13_dilations_0 = const()[name = string("gate_13_dilations_0"), val = tensor([1, 1])]; int32 gate_13_groups_0 = const()[name = string("gate_13_groups_0"), val = int32(1)]; tensor gate_13 = conv(dilations = gate_13_dilations_0, groups = gate_13_groups_0, pad = gate_13_pad_0, pad_type = gate_13_pad_type_0, strides = gate_13_strides_0, weight = layers_3_mlp_gate_proj_weight_palettized, x = input_109)[name = string("gate_13")]; string up_7_pad_type_0 = const()[name = string("up_7_pad_type_0"), val = string("valid")]; tensor up_7_strides_0 = const()[name = string("up_7_strides_0"), val = tensor([1, 1])]; tensor up_7_pad_0 = const()[name = string("up_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_7_dilations_0 = const()[name = string("up_7_dilations_0"), val = tensor([1, 1])]; int32 up_7_groups_0 = const()[name = string("up_7_groups_0"), val = int32(1)]; tensor up_7 = conv(dilations = up_7_dilations_0, groups = up_7_groups_0, pad = up_7_pad_0, pad_type = up_7_pad_type_0, strides = up_7_strides_0, weight = layers_3_mlp_up_proj_weight_palettized, x = input_109)[name = string("up_7")]; string gate_15_mode_0 = const()[name = string("gate_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_15 = gelu(mode = gate_15_mode_0, x = gate_13)[name = string("gate_15")]; tensor input_111 = mul(x = gate_15, y = up_7)[name = string("input_111")]; string mlp_out_7_pad_type_0 = const()[name = string("mlp_out_7_pad_type_0"), val = string("valid")]; tensor mlp_out_7_strides_0 = const()[name = string("mlp_out_7_strides_0"), val = tensor([1, 1])]; tensor mlp_out_7_pad_0 = const()[name = string("mlp_out_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_7_dilations_0 = const()[name = string("mlp_out_7_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_7_groups_0 = const()[name = string("mlp_out_7_groups_0"), val = int32(1)]; tensor mlp_out_7 = conv(dilations = mlp_out_7_dilations_0, groups = mlp_out_7_groups_0, pad = mlp_out_7_pad_0, pad_type = mlp_out_7_pad_type_0, strides = mlp_out_7_strides_0, weight = layers_3_mlp_down_proj_weight_palettized, x = input_111)[name = string("mlp_out_7")]; tensor var_3937_axes_0 = const()[name = string("op_3937_axes_0"), val = tensor([2])]; tensor var_3937 = squeeze(axes = var_3937_axes_0, x = mlp_out_7)[name = string("op_3937")]; tensor var_3941 = const()[name = string("op_3941"), val = tensor([0, 2, 1])]; int32 var_3947 = const()[name = string("op_3947"), val = int32(-1)]; fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_113 = transpose(perm = var_3941, x = var_3937)[name = string("transpose_307")]; tensor var_3953_cast_fp16 = mul(x = x_113, y = const_69_promoted_to_fp16)[name = string("op_3953_cast_fp16")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113_cast_fp16 = concat(axis = var_3947, interleave = input_113_interleave_0, values = (x_113, var_3953_cast_fp16))[name = string("input_113_cast_fp16")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_3945_to_fp16 = const()[name = string("op_3945_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_3945_to_fp16, x = input_113_cast_fp16)[name = string("normed_105_cast_fp16")]; tensor var_3958_split_sizes_0 = const()[name = string("op_3958_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3958_axis_0 = const()[name = string("op_3958_axis_0"), val = int32(-1)]; tensor var_3958_cast_fp16_0, tensor var_3958_cast_fp16_1 = split(axis = var_3958_axis_0, split_sizes = var_3958_split_sizes_0, x = normed_105_cast_fp16)[name = string("op_3958_cast_fp16")]; tensor const_70_to_fp16 = const()[name = string("const_70_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068784832)))]; tensor var_3961_cast_fp16 = mul(x = var_3958_cast_fp16_0, y = const_70_to_fp16)[name = string("op_3961_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = x_109_cast_fp16, y = var_3961_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; tensor per_layer_slice_7_begin_0 = const()[name = string("per_layer_slice_7_begin_0"), val = tensor([0, 0, 768])]; tensor per_layer_slice_7_end_0 = const()[name = string("per_layer_slice_7_end_0"), val = tensor([1, 1, 1024])]; tensor per_layer_slice_7_end_mask_0 = const()[name = string("per_layer_slice_7_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_7_cast_fp16 = slice_by_index(begin = per_layer_slice_7_begin_0, end = per_layer_slice_7_end_0, end_mask = per_layer_slice_7_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_7_cast_fp16")]; tensor gated_13 = linear(bias = linear_0_bias_0, weight = layers_3_per_layer_input_gate_weight_palettized, x = hidden_states_45_cast_fp16)[name = string("linear_6")]; string gated_15_mode_0 = const()[name = string("gated_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_15 = gelu(mode = gated_15_mode_0, x = gated_13)[name = string("gated_15")]; tensor input_117_cast_fp16 = mul(x = gated_15, y = per_layer_slice_7_cast_fp16)[name = string("input_117_cast_fp16")]; tensor layers_3_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068787968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068984640))))[name = string("layers_3_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_7_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_3_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_117_cast_fp16)[name = string("linear_7_cast_fp16")]; int32 var_3998 = const()[name = string("op_3998"), val = int32(-1)]; fp16 const_71_promoted_to_fp16 = const()[name = string("const_71_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4004_cast_fp16 = mul(x = linear_7_cast_fp16, y = const_71_promoted_to_fp16)[name = string("op_4004_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_3998, interleave = input_119_interleave_0, values = (linear_7_cast_fp16, var_4004_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_3996_to_fp16 = const()[name = string("op_3996_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_3996_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor var_4009_split_sizes_0 = const()[name = string("op_4009_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4009_axis_0 = const()[name = string("op_4009_axis_0"), val = int32(-1)]; tensor var_4009_cast_fp16_0, tensor var_4009_cast_fp16_1 = split(axis = var_4009_axis_0, split_sizes = var_4009_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_4009_cast_fp16")]; tensor const_72_to_fp16 = const()[name = string("const_72_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068986240)))]; tensor var_4012_cast_fp16 = mul(x = var_4009_cast_fp16_0, y = const_72_to_fp16)[name = string("op_4012_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = var_4012_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; tensor layers_3_layer_scalar_to_fp16 = const()[name = string("layers_3_layer_scalar_to_fp16"), val = tensor([0x1.26p-2])]; tensor x_121_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = layers_3_layer_scalar_to_fp16)[name = string("x_121_cast_fp16")]; int32 var_4020 = const()[name = string("op_4020"), val = int32(-1)]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4026_cast_fp16 = mul(x = x_121_cast_fp16, y = const_73_promoted_to_fp16)[name = string("op_4026_cast_fp16")]; bool input_121_interleave_0 = const()[name = string("input_121_interleave_0"), val = bool(false)]; tensor input_121_cast_fp16 = concat(axis = var_4020, interleave = input_121_interleave_0, values = (x_121_cast_fp16, var_4026_cast_fp16))[name = string("input_121_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_4018_to_fp16 = const()[name = string("op_4018_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4018_to_fp16, x = input_121_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_4031_split_sizes_0 = const()[name = string("op_4031_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4031_axis_0 = const()[name = string("op_4031_axis_0"), val = int32(-1)]; tensor var_4031_cast_fp16_0, tensor var_4031_cast_fp16_1 = split(axis = var_4031_axis_0, split_sizes = var_4031_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_4031_cast_fp16")]; tensor const_74_to_fp16 = const()[name = string("const_74_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068989376)))]; tensor var_4034_cast_fp16 = mul(x = var_4031_cast_fp16_0, y = const_74_to_fp16)[name = string("op_4034_cast_fp16")]; tensor var_4042 = const()[name = string("op_4042"), val = tensor([0, 2, 1])]; tensor var_4045_axes_0 = const()[name = string("op_4045_axes_0"), val = tensor([2])]; tensor var_4043_cast_fp16 = transpose(perm = var_4042, x = var_4034_cast_fp16)[name = string("transpose_306")]; tensor var_4045_cast_fp16 = expand_dims(axes = var_4045_axes_0, x = var_4043_cast_fp16)[name = string("op_4045_cast_fp16")]; string var_4061_pad_type_0 = const()[name = string("op_4061_pad_type_0"), val = string("valid")]; tensor var_4061_strides_0 = const()[name = string("op_4061_strides_0"), val = tensor([1, 1])]; tensor var_4061_pad_0 = const()[name = string("op_4061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4061_dilations_0 = const()[name = string("op_4061_dilations_0"), val = tensor([1, 1])]; int32 var_4061_groups_0 = const()[name = string("op_4061_groups_0"), val = int32(1)]; tensor var_4061 = conv(dilations = var_4061_dilations_0, groups = var_4061_groups_0, pad = var_4061_pad_0, pad_type = var_4061_pad_type_0, strides = var_4061_strides_0, weight = layers_4_self_attn_q_proj_weight_palettized, x = var_4045_cast_fp16)[name = string("op_4061")]; tensor var_4066 = const()[name = string("op_4066"), val = tensor([1, 8, 512, 1])]; tensor var_4067 = reshape(shape = var_4066, x = var_4061)[name = string("op_4067")]; tensor var_4072 = const()[name = string("op_4072"), val = tensor([0, 1, 3, 2])]; tensor var_4082 = const()[name = string("op_4082"), val = tensor([1, 8, 512])]; tensor var_4073 = transpose(perm = var_4072, x = var_4067)[name = string("transpose_305")]; tensor x_125 = reshape(shape = var_4082, x = var_4073)[name = string("x_125")]; int32 var_4088 = const()[name = string("op_4088"), val = int32(-1)]; fp16 const_75_promoted_to_fp16 = const()[name = string("const_75_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4094_cast_fp16 = mul(x = x_125, y = const_75_promoted_to_fp16)[name = string("op_4094_cast_fp16")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125_cast_fp16 = concat(axis = var_4088, interleave = input_125_interleave_0, values = (x_125, var_4094_cast_fp16))[name = string("input_125_cast_fp16")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_4086_to_fp16 = const()[name = string("op_4086_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4086_to_fp16, x = input_125_cast_fp16)[name = string("normed_117_cast_fp16")]; tensor var_4099_split_sizes_0 = const()[name = string("op_4099_split_sizes_0"), val = tensor([512, 512])]; int32 var_4099_axis_0 = const()[name = string("op_4099_axis_0"), val = int32(-1)]; tensor var_4099_cast_fp16_0, tensor var_4099_cast_fp16_1 = split(axis = var_4099_axis_0, split_sizes = var_4099_split_sizes_0, x = normed_117_cast_fp16)[name = string("op_4099_cast_fp16")]; tensor const_76_to_fp16 = const()[name = string("const_76_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068992512)))]; tensor var_4102_cast_fp16 = mul(x = var_4099_cast_fp16_0, y = const_76_to_fp16)[name = string("op_4102_cast_fp16")]; tensor var_4108 = const()[name = string("op_4108"), val = tensor([1, 8, 1, 512])]; tensor q_35 = reshape(shape = var_4108, x = var_4102_cast_fp16)[name = string("q_35")]; tensor var_4110 = mul(x = q_35, y = cos)[name = string("op_4110")]; tensor var_4111_split_sizes_0 = const()[name = string("op_4111_split_sizes_0"), val = tensor([256, 256])]; int32 var_4111_axis_0 = const()[name = string("op_4111_axis_0"), val = int32(-1)]; tensor var_4111_0, tensor var_4111_1 = split(axis = var_4111_axis_0, split_sizes = var_4111_split_sizes_0, x = q_35)[name = string("op_4111")]; fp16 const_77_promoted = const()[name = string("const_77_promoted"), val = fp16(-0x1p+0)]; tensor var_4113 = mul(x = var_4111_1, y = const_77_promoted)[name = string("op_4113")]; int32 var_4115 = const()[name = string("op_4115"), val = int32(-1)]; bool var_4116_interleave_0 = const()[name = string("op_4116_interleave_0"), val = bool(false)]; tensor var_4116 = concat(axis = var_4115, interleave = var_4116_interleave_0, values = (var_4113, var_4111_0))[name = string("op_4116")]; tensor var_4117 = mul(x = var_4116, y = sin)[name = string("op_4117")]; tensor q_39 = add(x = var_4110, y = var_4117)[name = string("q_39")]; string var_4130_pad_type_0 = const()[name = string("op_4130_pad_type_0"), val = string("valid")]; tensor var_4130_strides_0 = const()[name = string("op_4130_strides_0"), val = tensor([1, 1])]; tensor var_4130_pad_0 = const()[name = string("op_4130_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4130_dilations_0 = const()[name = string("op_4130_dilations_0"), val = tensor([1, 1])]; int32 var_4130_groups_0 = const()[name = string("op_4130_groups_0"), val = int32(1)]; tensor var_4130 = conv(dilations = var_4130_dilations_0, groups = var_4130_groups_0, pad = var_4130_pad_0, pad_type = var_4130_pad_type_0, strides = var_4130_strides_0, weight = layers_4_self_attn_k_proj_weight_palettized, x = var_4045_cast_fp16)[name = string("op_4130")]; tensor var_4135 = const()[name = string("op_4135"), val = tensor([1, 1, 512, 1])]; tensor var_4136 = reshape(shape = var_4135, x = var_4130)[name = string("op_4136")]; tensor var_4141 = const()[name = string("op_4141"), val = tensor([0, 1, 3, 2])]; string var_4158_pad_type_0 = const()[name = string("op_4158_pad_type_0"), val = string("valid")]; tensor var_4158_strides_0 = const()[name = string("op_4158_strides_0"), val = tensor([1, 1])]; tensor var_4158_pad_0 = const()[name = string("op_4158_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4158_dilations_0 = const()[name = string("op_4158_dilations_0"), val = tensor([1, 1])]; int32 var_4158_groups_0 = const()[name = string("op_4158_groups_0"), val = int32(1)]; tensor var_4158 = conv(dilations = var_4158_dilations_0, groups = var_4158_groups_0, pad = var_4158_pad_0, pad_type = var_4158_pad_type_0, strides = var_4158_strides_0, weight = layers_4_self_attn_v_proj_weight_palettized, x = var_4045_cast_fp16)[name = string("op_4158")]; tensor var_4163 = const()[name = string("op_4163"), val = tensor([1, 1, 512, 1])]; tensor var_4164 = reshape(shape = var_4163, x = var_4158)[name = string("op_4164")]; tensor var_4169 = const()[name = string("op_4169"), val = tensor([0, 1, 3, 2])]; tensor var_4179 = const()[name = string("op_4179"), val = tensor([1, 1, 512])]; tensor var_4142 = transpose(perm = var_4141, x = var_4136)[name = string("transpose_304")]; tensor x_129 = reshape(shape = var_4179, x = var_4142)[name = string("x_129")]; int32 var_4185 = const()[name = string("op_4185"), val = int32(-1)]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4191_cast_fp16 = mul(x = x_129, y = const_78_promoted_to_fp16)[name = string("op_4191_cast_fp16")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127_cast_fp16 = concat(axis = var_4185, interleave = input_127_interleave_0, values = (x_129, var_4191_cast_fp16))[name = string("input_127_cast_fp16")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_4183_to_fp16 = const()[name = string("op_4183_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4183_to_fp16, x = input_127_cast_fp16)[name = string("normed_121_cast_fp16")]; tensor var_4196_split_sizes_0 = const()[name = string("op_4196_split_sizes_0"), val = tensor([512, 512])]; int32 var_4196_axis_0 = const()[name = string("op_4196_axis_0"), val = int32(-1)]; tensor var_4196_cast_fp16_0, tensor var_4196_cast_fp16_1 = split(axis = var_4196_axis_0, split_sizes = var_4196_split_sizes_0, x = normed_121_cast_fp16)[name = string("op_4196_cast_fp16")]; tensor const_79_to_fp16 = const()[name = string("const_79_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068993600)))]; tensor var_4199_cast_fp16 = mul(x = var_4196_cast_fp16_0, y = const_79_to_fp16)[name = string("op_4199_cast_fp16")]; tensor var_4205 = const()[name = string("op_4205"), val = tensor([1, 1, 1, 512])]; tensor q_37 = reshape(shape = var_4205, x = var_4199_cast_fp16)[name = string("q_37")]; fp16 var_4212_promoted_to_fp16 = const()[name = string("op_4212_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_4170 = transpose(perm = var_4169, x = var_4164)[name = string("transpose_303")]; tensor var_4213_cast_fp16 = pow(x = var_4170, y = var_4212_promoted_to_fp16)[name = string("op_4213_cast_fp16")]; tensor var_4218_axes_0 = const()[name = string("op_4218_axes_0"), val = tensor([-1])]; bool var_4218_keep_dims_0 = const()[name = string("op_4218_keep_dims_0"), val = bool(true)]; tensor var_4218_cast_fp16 = reduce_mean(axes = var_4218_axes_0, keep_dims = var_4218_keep_dims_0, x = var_4213_cast_fp16)[name = string("op_4218_cast_fp16")]; fp16 var_4220_to_fp16 = const()[name = string("op_4220_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_9_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4220_to_fp16)[name = string("mean_sq_9_cast_fp16")]; fp16 var_4227_to_fp16 = const()[name = string("op_4227_to_fp16"), val = fp16(-0x1p-1)]; tensor var_4228_cast_fp16 = pow(x = mean_sq_9_cast_fp16, y = var_4227_to_fp16)[name = string("op_4228_cast_fp16")]; tensor var_4229_cast_fp16 = mul(x = var_4170, y = var_4228_cast_fp16)[name = string("op_4229_cast_fp16")]; tensor var_4235 = mul(x = q_37, y = cos)[name = string("op_4235")]; tensor var_4236_split_sizes_0 = const()[name = string("op_4236_split_sizes_0"), val = tensor([256, 256])]; int32 var_4236_axis_0 = const()[name = string("op_4236_axis_0"), val = int32(-1)]; tensor var_4236_0, tensor var_4236_1 = split(axis = var_4236_axis_0, split_sizes = var_4236_split_sizes_0, x = q_37)[name = string("op_4236")]; fp16 const_80_promoted = const()[name = string("const_80_promoted"), val = fp16(-0x1p+0)]; tensor var_4238 = mul(x = var_4236_1, y = const_80_promoted)[name = string("op_4238")]; int32 var_4240 = const()[name = string("op_4240"), val = int32(-1)]; bool var_4241_interleave_0 = const()[name = string("op_4241_interleave_0"), val = bool(false)]; tensor var_4241 = concat(axis = var_4240, interleave = var_4241_interleave_0, values = (var_4238, var_4236_0))[name = string("op_4241")]; tensor var_4242 = mul(x = var_4241, y = sin)[name = string("op_4242")]; tensor k_11 = add(x = var_4235, y = var_4242)[name = string("k_11")]; tensor var_4247_begin_0 = const()[name = string("op_4247_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_4247_end_0 = const()[name = string("op_4247_end_0"), val = tensor([5, 1, 512, 512])]; tensor var_4247_end_mask_0 = const()[name = string("op_4247_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4247_squeeze_mask_0 = const()[name = string("op_4247_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4247_cast_fp16 = slice_by_index(begin = var_4247_begin_0, end = var_4247_end_0, end_mask = var_4247_end_mask_0, squeeze_mask = var_4247_squeeze_mask_0, x = coreml_update_state_37)[name = string("op_4247_cast_fp16")]; tensor K_cache_9_axes_0 = const()[name = string("K_cache_9_axes_0"), val = tensor([0])]; tensor K_cache_9_cast_fp16 = expand_dims(axes = K_cache_9_axes_0, x = var_4247_cast_fp16)[name = string("K_cache_9_cast_fp16")]; tensor var_4252_begin_0 = const()[name = string("op_4252_begin_0"), val = tensor([39, 0, 0, 0])]; tensor var_4252_end_0 = const()[name = string("op_4252_end_0"), val = tensor([40, 1, 512, 512])]; tensor var_4252_end_mask_0 = const()[name = string("op_4252_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4252_squeeze_mask_0 = const()[name = string("op_4252_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4252_cast_fp16 = slice_by_index(begin = var_4252_begin_0, end = var_4252_end_0, end_mask = var_4252_end_mask_0, squeeze_mask = var_4252_squeeze_mask_0, x = coreml_update_state_37)[name = string("op_4252_cast_fp16")]; tensor V_cache_9_axes_0 = const()[name = string("V_cache_9_axes_0"), val = tensor([0])]; tensor V_cache_9_cast_fp16 = expand_dims(axes = V_cache_9_axes_0, x = var_4252_cast_fp16)[name = string("V_cache_9_cast_fp16")]; tensor var_4258_cast_fp16 = mul(x = K_cache_9_cast_fp16, y = var_2187_cast_fp16)[name = string("op_4258_cast_fp16")]; tensor var_4259_reps_0 = const()[name = string("op_4259_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_4259 = tile(reps = var_4259_reps_0, x = k_11)[name = string("op_4259")]; tensor var_4260_cast_fp16 = mul(x = var_4259, y = update_mask)[name = string("op_4260_cast_fp16")]; tensor K_new_9_cast_fp16 = add(x = var_4258_cast_fp16, y = var_4260_cast_fp16)[name = string("K_new_9_cast_fp16")]; tensor var_4266_cast_fp16 = mul(x = V_cache_9_cast_fp16, y = var_2187_cast_fp16)[name = string("op_4266_cast_fp16")]; tensor var_4267_reps_0 = const()[name = string("op_4267_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_4267 = tile(reps = var_4267_reps_0, x = var_4229_cast_fp16)[name = string("op_4267")]; tensor var_4268_cast_fp16 = mul(x = var_4267, y = update_mask)[name = string("op_4268_cast_fp16")]; tensor V_new_9_cast_fp16 = add(x = var_4266_cast_fp16, y = var_4268_cast_fp16)[name = string("V_new_9_cast_fp16")]; tensor var_4272_axes_0 = const()[name = string("op_4272_axes_0"), val = tensor([0])]; tensor var_4272_cast_fp16 = squeeze(axes = var_4272_axes_0, x = K_new_9_cast_fp16)[name = string("op_4272_cast_fp16")]; tensor concat_32 = const()[name = string("concat_32"), val = tensor([4, 0, 0, 0])]; tensor concat_33 = const()[name = string("concat_33"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_32, begin_mask = kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_33, end_mask = kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_9_stride_0, update = var_4272_cast_fp16, x = coreml_update_state_37)[name = string("kv_cache_0_internal_tensor_assign_9_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_9_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_38_write_state")]; tensor coreml_update_state_38 = read_state(input = kv_cache_0)[name = string("coreml_update_state_38")]; tensor var_4279_axes_0 = const()[name = string("op_4279_axes_0"), val = tensor([0])]; tensor var_4279_cast_fp16 = squeeze(axes = var_4279_axes_0, x = V_new_9_cast_fp16)[name = string("op_4279_cast_fp16")]; tensor concat_34 = const()[name = string("concat_34"), val = tensor([39, 0, 0, 0])]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_34, begin_mask = kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_35, end_mask = kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_10_stride_0, update = var_4279_cast_fp16, x = coreml_update_state_38)[name = string("kv_cache_0_internal_tensor_assign_10_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_10_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_39_write_state")]; tensor coreml_update_state_39 = read_state(input = kv_cache_0)[name = string("coreml_update_state_39")]; tensor transpose_16_perm_0 = const()[name = string("transpose_16_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_8_reps_0 = const()[name = string("tile_8_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_16_cast_fp16 = transpose(perm = transpose_16_perm_0, x = K_new_9_cast_fp16)[name = string("transpose_302")]; tensor tile_8_cast_fp16 = tile(reps = tile_8_reps_0, x = transpose_16_cast_fp16)[name = string("tile_8_cast_fp16")]; tensor concat_36 = const()[name = string("concat_36"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_36, x = tile_8_cast_fp16)[name = string("reshape_16_cast_fp16")]; tensor transpose_17_perm_0 = const()[name = string("transpose_17_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_37 = const()[name = string("concat_37"), val = tensor([-1, 1, 512, 512])]; tensor transpose_17_cast_fp16 = transpose(perm = transpose_17_perm_0, x = reshape_16_cast_fp16)[name = string("transpose_301")]; tensor reshape_17_cast_fp16 = reshape(shape = concat_37, x = transpose_17_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor transpose_144_perm_0 = const()[name = string("transpose_144_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_18_perm_0 = const()[name = string("transpose_18_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_9_reps_0 = const()[name = string("tile_9_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_18_cast_fp16 = transpose(perm = transpose_18_perm_0, x = V_new_9_cast_fp16)[name = string("transpose_300")]; tensor tile_9_cast_fp16 = tile(reps = tile_9_reps_0, x = transpose_18_cast_fp16)[name = string("tile_9_cast_fp16")]; tensor concat_38 = const()[name = string("concat_38"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_38, x = tile_9_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor transpose_19_perm_0 = const()[name = string("transpose_19_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_39 = const()[name = string("concat_39"), val = tensor([-1, 1, 512, 512])]; tensor transpose_19_cast_fp16 = transpose(perm = transpose_19_perm_0, x = reshape_18_cast_fp16)[name = string("transpose_299")]; tensor reshape_19_cast_fp16 = reshape(shape = concat_39, x = transpose_19_cast_fp16)[name = string("reshape_19_cast_fp16")]; tensor V_expanded_9_perm_0 = const()[name = string("V_expanded_9_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_4306_transpose_x_0 = const()[name = string("op_4306_transpose_x_0"), val = bool(false)]; bool var_4306_transpose_y_0 = const()[name = string("op_4306_transpose_y_0"), val = bool(false)]; tensor transpose_144_cast_fp16 = transpose(perm = transpose_144_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_298")]; tensor var_4306_cast_fp16 = matmul(transpose_x = var_4306_transpose_x_0, transpose_y = var_4306_transpose_y_0, x = q_39, y = transpose_144_cast_fp16)[name = string("op_4306_cast_fp16")]; tensor attn_weights_27_cast_fp16 = add(x = var_4306_cast_fp16, y = causal_mask)[name = string("attn_weights_27_cast_fp16")]; int32 var_4311 = const()[name = string("op_4311"), val = int32(-1)]; tensor attn_weights_29_cast_fp16 = softmax(axis = var_4311, x = attn_weights_27_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; tensor V_expanded_9_cast_fp16 = transpose(perm = V_expanded_9_perm_0, x = reshape_19_cast_fp16)[name = string("transpose_297")]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = attn_weights_29_cast_fp16, y = V_expanded_9_cast_fp16)[name = string("attn_output_25_cast_fp16")]; tensor var_4319 = const()[name = string("op_4319"), val = tensor([0, 2, 1, 3])]; tensor var_4326 = const()[name = string("op_4326"), val = tensor([1, 1, -1])]; tensor var_4320_cast_fp16 = transpose(perm = var_4319, x = attn_output_25_cast_fp16)[name = string("transpose_296")]; tensor attn_output_27_cast_fp16 = reshape(shape = var_4326, x = var_4320_cast_fp16)[name = string("attn_output_27_cast_fp16")]; tensor var_4331 = const()[name = string("op_4331"), val = tensor([0, 2, 1])]; string var_4347_pad_type_0 = const()[name = string("op_4347_pad_type_0"), val = string("valid")]; int32 var_4347_groups_0 = const()[name = string("op_4347_groups_0"), val = int32(1)]; tensor var_4347_strides_0 = const()[name = string("op_4347_strides_0"), val = tensor([1])]; tensor var_4347_pad_0 = const()[name = string("op_4347_pad_0"), val = tensor([0, 0])]; tensor var_4347_dilations_0 = const()[name = string("op_4347_dilations_0"), val = tensor([1])]; tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1068994688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072140480))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4332_cast_fp16 = transpose(perm = var_4331, x = attn_output_27_cast_fp16)[name = string("transpose_295")]; tensor var_4347_cast_fp16 = conv(dilations = var_4347_dilations_0, groups = var_4347_groups_0, pad = var_4347_pad_0, pad_type = var_4347_pad_type_0, strides = var_4347_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_4332_cast_fp16)[name = string("op_4347_cast_fp16")]; tensor var_4351 = const()[name = string("op_4351"), val = tensor([0, 2, 1])]; int32 var_4357 = const()[name = string("op_4357"), val = int32(-1)]; fp16 const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_135_cast_fp16 = transpose(perm = var_4351, x = var_4347_cast_fp16)[name = string("transpose_294")]; tensor var_4363_cast_fp16 = mul(x = x_135_cast_fp16, y = const_81_promoted_to_fp16)[name = string("op_4363_cast_fp16")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131_cast_fp16 = concat(axis = var_4357, interleave = input_131_interleave_0, values = (x_135_cast_fp16, var_4363_cast_fp16))[name = string("input_131_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_4355_to_fp16 = const()[name = string("op_4355_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_4355_to_fp16, x = input_131_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor var_4368_split_sizes_0 = const()[name = string("op_4368_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4368_axis_0 = const()[name = string("op_4368_axis_0"), val = int32(-1)]; tensor var_4368_cast_fp16_0, tensor var_4368_cast_fp16_1 = split(axis = var_4368_axis_0, split_sizes = var_4368_split_sizes_0, x = normed_125_cast_fp16)[name = string("op_4368_cast_fp16")]; tensor const_82_to_fp16 = const()[name = string("const_82_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072142080)))]; tensor var_4371_cast_fp16 = mul(x = var_4368_cast_fp16_0, y = const_82_to_fp16)[name = string("op_4371_cast_fp16")]; tensor x_139_cast_fp16 = add(x = x_121_cast_fp16, y = var_4371_cast_fp16)[name = string("x_139_cast_fp16")]; int32 var_4378 = const()[name = string("op_4378"), val = int32(-1)]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4384_cast_fp16 = mul(x = x_139_cast_fp16, y = const_83_promoted_to_fp16)[name = string("op_4384_cast_fp16")]; bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; tensor input_133_cast_fp16 = concat(axis = var_4378, interleave = input_133_interleave_0, values = (x_139_cast_fp16, var_4384_cast_fp16))[name = string("input_133_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_4376_to_fp16 = const()[name = string("op_4376_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_4376_to_fp16, x = input_133_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor var_4389_split_sizes_0 = const()[name = string("op_4389_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4389_axis_0 = const()[name = string("op_4389_axis_0"), val = int32(-1)]; tensor var_4389_cast_fp16_0, tensor var_4389_cast_fp16_1 = split(axis = var_4389_axis_0, split_sizes = var_4389_split_sizes_0, x = normed_129_cast_fp16)[name = string("op_4389_cast_fp16")]; tensor const_84_to_fp16 = const()[name = string("const_84_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072145216)))]; tensor var_4392_cast_fp16 = mul(x = var_4389_cast_fp16_0, y = const_84_to_fp16)[name = string("op_4392_cast_fp16")]; tensor var_4405 = const()[name = string("op_4405"), val = tensor([0, 2, 1])]; tensor input_135_axes_0 = const()[name = string("input_135_axes_0"), val = tensor([2])]; tensor var_4406 = transpose(perm = var_4405, x = var_4392_cast_fp16)[name = string("transpose_293")]; tensor input_135 = expand_dims(axes = input_135_axes_0, x = var_4406)[name = string("input_135")]; string gate_17_pad_type_0 = const()[name = string("gate_17_pad_type_0"), val = string("valid")]; tensor gate_17_strides_0 = const()[name = string("gate_17_strides_0"), val = tensor([1, 1])]; tensor gate_17_pad_0 = const()[name = string("gate_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_17_dilations_0 = const()[name = string("gate_17_dilations_0"), val = tensor([1, 1])]; int32 gate_17_groups_0 = const()[name = string("gate_17_groups_0"), val = int32(1)]; tensor gate_17 = conv(dilations = gate_17_dilations_0, groups = gate_17_groups_0, pad = gate_17_pad_0, pad_type = gate_17_pad_type_0, strides = gate_17_strides_0, weight = layers_4_mlp_gate_proj_weight_palettized, x = input_135)[name = string("gate_17")]; string up_9_pad_type_0 = const()[name = string("up_9_pad_type_0"), val = string("valid")]; tensor up_9_strides_0 = const()[name = string("up_9_strides_0"), val = tensor([1, 1])]; tensor up_9_pad_0 = const()[name = string("up_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_9_dilations_0 = const()[name = string("up_9_dilations_0"), val = tensor([1, 1])]; int32 up_9_groups_0 = const()[name = string("up_9_groups_0"), val = int32(1)]; tensor up_9 = conv(dilations = up_9_dilations_0, groups = up_9_groups_0, pad = up_9_pad_0, pad_type = up_9_pad_type_0, strides = up_9_strides_0, weight = layers_4_mlp_up_proj_weight_palettized, x = input_135)[name = string("up_9")]; string gate_19_mode_0 = const()[name = string("gate_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_19 = gelu(mode = gate_19_mode_0, x = gate_17)[name = string("gate_19")]; tensor input_137 = mul(x = gate_19, y = up_9)[name = string("input_137")]; string mlp_out_9_pad_type_0 = const()[name = string("mlp_out_9_pad_type_0"), val = string("valid")]; tensor mlp_out_9_strides_0 = const()[name = string("mlp_out_9_strides_0"), val = tensor([1, 1])]; tensor mlp_out_9_pad_0 = const()[name = string("mlp_out_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_9_dilations_0 = const()[name = string("mlp_out_9_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_9_groups_0 = const()[name = string("mlp_out_9_groups_0"), val = int32(1)]; tensor mlp_out_9 = conv(dilations = mlp_out_9_dilations_0, groups = mlp_out_9_groups_0, pad = mlp_out_9_pad_0, pad_type = mlp_out_9_pad_type_0, strides = mlp_out_9_strides_0, weight = layers_4_mlp_down_proj_weight_palettized, x = input_137)[name = string("mlp_out_9")]; tensor var_4446_axes_0 = const()[name = string("op_4446_axes_0"), val = tensor([2])]; tensor var_4446 = squeeze(axes = var_4446_axes_0, x = mlp_out_9)[name = string("op_4446")]; tensor var_4450 = const()[name = string("op_4450"), val = tensor([0, 2, 1])]; int32 var_4456 = const()[name = string("op_4456"), val = int32(-1)]; fp16 const_85_promoted_to_fp16 = const()[name = string("const_85_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_143 = transpose(perm = var_4450, x = var_4446)[name = string("transpose_292")]; tensor var_4462_cast_fp16 = mul(x = x_143, y = const_85_promoted_to_fp16)[name = string("op_4462_cast_fp16")]; bool input_139_interleave_0 = const()[name = string("input_139_interleave_0"), val = bool(false)]; tensor input_139_cast_fp16 = concat(axis = var_4456, interleave = input_139_interleave_0, values = (x_143, var_4462_cast_fp16))[name = string("input_139_cast_fp16")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_4454_to_fp16 = const()[name = string("op_4454_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_4454_to_fp16, x = input_139_cast_fp16)[name = string("normed_133_cast_fp16")]; tensor var_4467_split_sizes_0 = const()[name = string("op_4467_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4467_axis_0 = const()[name = string("op_4467_axis_0"), val = int32(-1)]; tensor var_4467_cast_fp16_0, tensor var_4467_cast_fp16_1 = split(axis = var_4467_axis_0, split_sizes = var_4467_split_sizes_0, x = normed_133_cast_fp16)[name = string("op_4467_cast_fp16")]; tensor const_86_to_fp16 = const()[name = string("const_86_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072148352)))]; tensor var_4470_cast_fp16 = mul(x = var_4467_cast_fp16_0, y = const_86_to_fp16)[name = string("op_4470_cast_fp16")]; tensor hidden_states_57_cast_fp16 = add(x = x_139_cast_fp16, y = var_4470_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor per_layer_slice_9_begin_0 = const()[name = string("per_layer_slice_9_begin_0"), val = tensor([0, 0, 1024])]; tensor per_layer_slice_9_end_0 = const()[name = string("per_layer_slice_9_end_0"), val = tensor([1, 1, 1280])]; tensor per_layer_slice_9_end_mask_0 = const()[name = string("per_layer_slice_9_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_9_cast_fp16 = slice_by_index(begin = per_layer_slice_9_begin_0, end = per_layer_slice_9_end_0, end_mask = per_layer_slice_9_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_9_cast_fp16")]; tensor gated_17 = linear(bias = linear_0_bias_0, weight = layers_4_per_layer_input_gate_weight_palettized, x = hidden_states_57_cast_fp16)[name = string("linear_8")]; string gated_19_mode_0 = const()[name = string("gated_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_19 = gelu(mode = gated_19_mode_0, x = gated_17)[name = string("gated_19")]; tensor input_143_cast_fp16 = mul(x = gated_19, y = per_layer_slice_9_cast_fp16)[name = string("input_143_cast_fp16")]; tensor layers_4_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072151488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072348160))))[name = string("layers_4_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_9_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_4_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_143_cast_fp16)[name = string("linear_9_cast_fp16")]; int32 var_4507 = const()[name = string("op_4507"), val = int32(-1)]; fp16 const_87_promoted_to_fp16 = const()[name = string("const_87_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4513_cast_fp16 = mul(x = linear_9_cast_fp16, y = const_87_promoted_to_fp16)[name = string("op_4513_cast_fp16")]; bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; tensor input_145_cast_fp16 = concat(axis = var_4507, interleave = input_145_interleave_0, values = (linear_9_cast_fp16, var_4513_cast_fp16))[name = string("input_145_cast_fp16")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_4505_to_fp16 = const()[name = string("op_4505_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_4505_to_fp16, x = input_145_cast_fp16)[name = string("normed_137_cast_fp16")]; tensor var_4518_split_sizes_0 = const()[name = string("op_4518_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4518_axis_0 = const()[name = string("op_4518_axis_0"), val = int32(-1)]; tensor var_4518_cast_fp16_0, tensor var_4518_cast_fp16_1 = split(axis = var_4518_axis_0, split_sizes = var_4518_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_4518_cast_fp16")]; tensor const_88_to_fp16 = const()[name = string("const_88_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072349760)))]; tensor var_4521_cast_fp16 = mul(x = var_4518_cast_fp16_0, y = const_88_to_fp16)[name = string("op_4521_cast_fp16")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_57_cast_fp16, y = var_4521_cast_fp16)[name = string("hidden_states_61_cast_fp16")]; tensor layers_4_layer_scalar_to_fp16 = const()[name = string("layers_4_layer_scalar_to_fp16"), val = tensor([0x1.fep-2])]; tensor x_151_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = layers_4_layer_scalar_to_fp16)[name = string("x_151_cast_fp16")]; int32 var_4529 = const()[name = string("op_4529"), val = int32(-1)]; fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4535_cast_fp16 = mul(x = x_151_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_4535_cast_fp16")]; bool input_147_interleave_0 = const()[name = string("input_147_interleave_0"), val = bool(false)]; tensor input_147_cast_fp16 = concat(axis = var_4529, interleave = input_147_interleave_0, values = (x_151_cast_fp16, var_4535_cast_fp16))[name = string("input_147_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_4527_to_fp16 = const()[name = string("op_4527_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_4527_to_fp16, x = input_147_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-1)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_4540_cast_fp16")]; tensor const_90_to_fp16 = const()[name = string("const_90_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072352896)))]; tensor var_4543_cast_fp16 = mul(x = var_4540_cast_fp16_0, y = const_90_to_fp16)[name = string("op_4543_cast_fp16")]; tensor var_4551 = const()[name = string("op_4551"), val = tensor([0, 2, 1])]; tensor var_4554_axes_0 = const()[name = string("op_4554_axes_0"), val = tensor([2])]; tensor var_4552_cast_fp16 = transpose(perm = var_4551, x = var_4543_cast_fp16)[name = string("transpose_291")]; tensor var_4554_cast_fp16 = expand_dims(axes = var_4554_axes_0, x = var_4552_cast_fp16)[name = string("op_4554_cast_fp16")]; string var_4570_pad_type_0 = const()[name = string("op_4570_pad_type_0"), val = string("valid")]; tensor var_4570_strides_0 = const()[name = string("op_4570_strides_0"), val = tensor([1, 1])]; tensor var_4570_pad_0 = const()[name = string("op_4570_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4570_dilations_0 = const()[name = string("op_4570_dilations_0"), val = tensor([1, 1])]; int32 var_4570_groups_0 = const()[name = string("op_4570_groups_0"), val = int32(1)]; tensor var_4570 = conv(dilations = var_4570_dilations_0, groups = var_4570_groups_0, pad = var_4570_pad_0, pad_type = var_4570_pad_type_0, strides = var_4570_strides_0, weight = layers_5_self_attn_q_proj_weight_palettized, x = var_4554_cast_fp16)[name = string("op_4570")]; tensor var_4575 = const()[name = string("op_4575"), val = tensor([1, 8, 256, 1])]; tensor var_4576 = reshape(shape = var_4575, x = var_4570)[name = string("op_4576")]; tensor var_4581 = const()[name = string("op_4581"), val = tensor([0, 1, 3, 2])]; tensor var_4591 = const()[name = string("op_4591"), val = tensor([1, 8, 256])]; tensor var_4582 = transpose(perm = var_4581, x = var_4576)[name = string("transpose_290")]; tensor x_155 = reshape(shape = var_4591, x = var_4582)[name = string("x_155")]; int32 var_4597 = const()[name = string("op_4597"), val = int32(-1)]; fp16 const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4603_cast_fp16 = mul(x = x_155, y = const_91_promoted_to_fp16)[name = string("op_4603_cast_fp16")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151_cast_fp16 = concat(axis = var_4597, interleave = input_151_interleave_0, values = (x_155, var_4603_cast_fp16))[name = string("input_151_cast_fp16")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_4595_to_fp16 = const()[name = string("op_4595_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_4595_to_fp16, x = input_151_cast_fp16)[name = string("normed_145_cast_fp16")]; tensor var_4608_split_sizes_0 = const()[name = string("op_4608_split_sizes_0"), val = tensor([256, 256])]; int32 var_4608_axis_0 = const()[name = string("op_4608_axis_0"), val = int32(-1)]; tensor var_4608_cast_fp16_0, tensor var_4608_cast_fp16_1 = split(axis = var_4608_axis_0, split_sizes = var_4608_split_sizes_0, x = normed_145_cast_fp16)[name = string("op_4608_cast_fp16")]; tensor const_92_to_fp16 = const()[name = string("const_92_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072356032)))]; tensor var_4611_cast_fp16 = mul(x = var_4608_cast_fp16_0, y = const_92_to_fp16)[name = string("op_4611_cast_fp16")]; tensor var_4617 = const()[name = string("op_4617"), val = tensor([1, 8, 1, 256])]; tensor q_43 = reshape(shape = var_4617, x = var_4611_cast_fp16)[name = string("q_43")]; tensor var_4619 = mul(x = q_43, y = cos_1)[name = string("op_4619")]; tensor var_4620_split_sizes_0 = const()[name = string("op_4620_split_sizes_0"), val = tensor([128, 128])]; int32 var_4620_axis_0 = const()[name = string("op_4620_axis_0"), val = int32(-1)]; tensor var_4620_0, tensor var_4620_1 = split(axis = var_4620_axis_0, split_sizes = var_4620_split_sizes_0, x = q_43)[name = string("op_4620")]; fp16 const_93_promoted = const()[name = string("const_93_promoted"), val = fp16(-0x1p+0)]; tensor var_4622 = mul(x = var_4620_1, y = const_93_promoted)[name = string("op_4622")]; int32 var_4624 = const()[name = string("op_4624"), val = int32(-1)]; bool var_4625_interleave_0 = const()[name = string("op_4625_interleave_0"), val = bool(false)]; tensor var_4625 = concat(axis = var_4624, interleave = var_4625_interleave_0, values = (var_4622, var_4620_0))[name = string("op_4625")]; tensor var_4626 = mul(x = var_4625, y = sin_1)[name = string("op_4626")]; tensor q_47 = add(x = var_4619, y = var_4626)[name = string("q_47")]; string var_4639_pad_type_0 = const()[name = string("op_4639_pad_type_0"), val = string("valid")]; tensor var_4639_strides_0 = const()[name = string("op_4639_strides_0"), val = tensor([1, 1])]; tensor var_4639_pad_0 = const()[name = string("op_4639_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4639_dilations_0 = const()[name = string("op_4639_dilations_0"), val = tensor([1, 1])]; int32 var_4639_groups_0 = const()[name = string("op_4639_groups_0"), val = int32(1)]; tensor var_4639 = conv(dilations = var_4639_dilations_0, groups = var_4639_groups_0, pad = var_4639_pad_0, pad_type = var_4639_pad_type_0, strides = var_4639_strides_0, weight = layers_5_self_attn_k_proj_weight_palettized, x = var_4554_cast_fp16)[name = string("op_4639")]; tensor var_4644 = const()[name = string("op_4644"), val = tensor([1, 1, 256, 1])]; tensor var_4645 = reshape(shape = var_4644, x = var_4639)[name = string("op_4645")]; tensor var_4650 = const()[name = string("op_4650"), val = tensor([0, 1, 3, 2])]; string var_4667_pad_type_0 = const()[name = string("op_4667_pad_type_0"), val = string("valid")]; tensor var_4667_strides_0 = const()[name = string("op_4667_strides_0"), val = tensor([1, 1])]; tensor var_4667_pad_0 = const()[name = string("op_4667_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4667_dilations_0 = const()[name = string("op_4667_dilations_0"), val = tensor([1, 1])]; int32 var_4667_groups_0 = const()[name = string("op_4667_groups_0"), val = int32(1)]; tensor var_4667 = conv(dilations = var_4667_dilations_0, groups = var_4667_groups_0, pad = var_4667_pad_0, pad_type = var_4667_pad_type_0, strides = var_4667_strides_0, weight = layers_5_self_attn_v_proj_weight_palettized, x = var_4554_cast_fp16)[name = string("op_4667")]; tensor var_4672 = const()[name = string("op_4672"), val = tensor([1, 1, 256, 1])]; tensor var_4673 = reshape(shape = var_4672, x = var_4667)[name = string("op_4673")]; tensor var_4678 = const()[name = string("op_4678"), val = tensor([0, 1, 3, 2])]; tensor var_4688 = const()[name = string("op_4688"), val = tensor([1, 1, 256])]; tensor var_4651 = transpose(perm = var_4650, x = var_4645)[name = string("transpose_289")]; tensor x_159 = reshape(shape = var_4688, x = var_4651)[name = string("x_159")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(-1)]; fp16 const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4700_cast_fp16 = mul(x = x_159, y = const_94_promoted_to_fp16)[name = string("op_4700_cast_fp16")]; bool input_153_interleave_0 = const()[name = string("input_153_interleave_0"), val = bool(false)]; tensor input_153_cast_fp16 = concat(axis = var_4694, interleave = input_153_interleave_0, values = (x_159, var_4700_cast_fp16))[name = string("input_153_cast_fp16")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_4692_to_fp16 = const()[name = string("op_4692_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_4692_to_fp16, x = input_153_cast_fp16)[name = string("normed_149_cast_fp16")]; tensor var_4705_split_sizes_0 = const()[name = string("op_4705_split_sizes_0"), val = tensor([256, 256])]; int32 var_4705_axis_0 = const()[name = string("op_4705_axis_0"), val = int32(-1)]; tensor var_4705_cast_fp16_0, tensor var_4705_cast_fp16_1 = split(axis = var_4705_axis_0, split_sizes = var_4705_split_sizes_0, x = normed_149_cast_fp16)[name = string("op_4705_cast_fp16")]; tensor var_4708_cast_fp16 = mul(x = var_4705_cast_fp16_0, y = const_7_to_fp16)[name = string("op_4708_cast_fp16")]; tensor var_4714 = const()[name = string("op_4714"), val = tensor([1, 1, 1, 256])]; tensor q_45 = reshape(shape = var_4714, x = var_4708_cast_fp16)[name = string("q_45")]; fp16 var_4721_promoted_to_fp16 = const()[name = string("op_4721_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_4679 = transpose(perm = var_4678, x = var_4673)[name = string("transpose_288")]; tensor var_4722_cast_fp16 = pow(x = var_4679, y = var_4721_promoted_to_fp16)[name = string("op_4722_cast_fp16")]; tensor var_4727_axes_0 = const()[name = string("op_4727_axes_0"), val = tensor([-1])]; bool var_4727_keep_dims_0 = const()[name = string("op_4727_keep_dims_0"), val = bool(true)]; tensor var_4727_cast_fp16 = reduce_mean(axes = var_4727_axes_0, keep_dims = var_4727_keep_dims_0, x = var_4722_cast_fp16)[name = string("op_4727_cast_fp16")]; fp16 var_4729_to_fp16 = const()[name = string("op_4729_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_11_cast_fp16 = add(x = var_4727_cast_fp16, y = var_4729_to_fp16)[name = string("mean_sq_11_cast_fp16")]; fp16 var_4736_to_fp16 = const()[name = string("op_4736_to_fp16"), val = fp16(-0x1p-1)]; tensor var_4737_cast_fp16 = pow(x = mean_sq_11_cast_fp16, y = var_4736_to_fp16)[name = string("op_4737_cast_fp16")]; tensor var_4738_cast_fp16 = mul(x = var_4679, y = var_4737_cast_fp16)[name = string("op_4738_cast_fp16")]; tensor var_4744 = mul(x = q_45, y = cos_1)[name = string("op_4744")]; tensor var_4745_split_sizes_0 = const()[name = string("op_4745_split_sizes_0"), val = tensor([128, 128])]; int32 var_4745_axis_0 = const()[name = string("op_4745_axis_0"), val = int32(-1)]; tensor var_4745_0, tensor var_4745_1 = split(axis = var_4745_axis_0, split_sizes = var_4745_split_sizes_0, x = q_45)[name = string("op_4745")]; fp16 const_96_promoted = const()[name = string("const_96_promoted"), val = fp16(-0x1p+0)]; tensor var_4747 = mul(x = var_4745_1, y = const_96_promoted)[name = string("op_4747")]; int32 var_4749 = const()[name = string("op_4749"), val = int32(-1)]; bool var_4750_interleave_0 = const()[name = string("op_4750_interleave_0"), val = bool(false)]; tensor var_4750 = concat(axis = var_4749, interleave = var_4750_interleave_0, values = (var_4747, var_4745_0))[name = string("op_4750")]; tensor var_4751 = mul(x = var_4750, y = sin_1)[name = string("op_4751")]; tensor input_155 = add(x = var_4744, y = var_4751)[name = string("input_155")]; tensor var_4756_begin_0 = const()[name = string("op_4756_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_4756_end_0 = const()[name = string("op_4756_end_0"), val = tensor([6, 1, 512, 512])]; tensor var_4756_end_mask_0 = const()[name = string("op_4756_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4756_squeeze_mask_0 = const()[name = string("op_4756_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4756_cast_fp16 = slice_by_index(begin = var_4756_begin_0, end = var_4756_end_0, end_mask = var_4756_end_mask_0, squeeze_mask = var_4756_squeeze_mask_0, x = coreml_update_state_39)[name = string("op_4756_cast_fp16")]; tensor K_cache_11_axes_0 = const()[name = string("K_cache_11_axes_0"), val = tensor([0])]; tensor K_cache_11_cast_fp16 = expand_dims(axes = K_cache_11_axes_0, x = var_4756_cast_fp16)[name = string("K_cache_11_cast_fp16")]; tensor var_4761_begin_0 = const()[name = string("op_4761_begin_0"), val = tensor([40, 0, 0, 0])]; tensor var_4761_end_0 = const()[name = string("op_4761_end_0"), val = tensor([41, 1, 512, 512])]; tensor var_4761_end_mask_0 = const()[name = string("op_4761_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4761_squeeze_mask_0 = const()[name = string("op_4761_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4761_cast_fp16 = slice_by_index(begin = var_4761_begin_0, end = var_4761_end_0, end_mask = var_4761_end_mask_0, squeeze_mask = var_4761_squeeze_mask_0, x = coreml_update_state_39)[name = string("op_4761_cast_fp16")]; tensor V_cache_11_axes_0 = const()[name = string("V_cache_11_axes_0"), val = tensor([0])]; tensor V_cache_11_cast_fp16 = expand_dims(axes = V_cache_11_axes_0, x = var_4761_cast_fp16)[name = string("V_cache_11_cast_fp16")]; tensor k_padded_9_pad_0 = const()[name = string("k_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_9_mode_0 = const()[name = string("k_padded_9_mode_0"), val = string("constant")]; fp16 const_97_to_fp16 = const()[name = string("const_97_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_9_cast_fp16 = pad(constant_val = const_97_to_fp16, mode = k_padded_9_mode_0, pad = k_padded_9_pad_0, x = input_155)[name = string("k_padded_9_cast_fp16")]; tensor v_padded_9_pad_0 = const()[name = string("v_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_9_mode_0 = const()[name = string("v_padded_9_mode_0"), val = string("constant")]; fp16 const_98_to_fp16 = const()[name = string("const_98_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_9_cast_fp16 = pad(constant_val = const_98_to_fp16, mode = v_padded_9_mode_0, pad = v_padded_9_pad_0, x = var_4738_cast_fp16)[name = string("v_padded_9_cast_fp16")]; tensor var_4779_cast_fp16 = mul(x = K_cache_11_cast_fp16, y = var_2187_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor var_4780_reps_0 = const()[name = string("op_4780_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_4780_cast_fp16 = tile(reps = var_4780_reps_0, x = k_padded_9_cast_fp16)[name = string("op_4780_cast_fp16")]; tensor var_4781_cast_fp16 = mul(x = var_4780_cast_fp16, y = update_mask)[name = string("op_4781_cast_fp16")]; tensor K_new_11_cast_fp16 = add(x = var_4779_cast_fp16, y = var_4781_cast_fp16)[name = string("K_new_11_cast_fp16")]; tensor var_4787_cast_fp16 = mul(x = V_cache_11_cast_fp16, y = var_2187_cast_fp16)[name = string("op_4787_cast_fp16")]; tensor var_4788_reps_0 = const()[name = string("op_4788_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_4788_cast_fp16 = tile(reps = var_4788_reps_0, x = v_padded_9_cast_fp16)[name = string("op_4788_cast_fp16")]; tensor var_4789_cast_fp16 = mul(x = var_4788_cast_fp16, y = update_mask)[name = string("op_4789_cast_fp16")]; tensor V_new_11_cast_fp16 = add(x = var_4787_cast_fp16, y = var_4789_cast_fp16)[name = string("V_new_11_cast_fp16")]; tensor var_4793_axes_0 = const()[name = string("op_4793_axes_0"), val = tensor([0])]; tensor var_4793_cast_fp16 = squeeze(axes = var_4793_axes_0, x = K_new_11_cast_fp16)[name = string("op_4793_cast_fp16")]; tensor concat_40 = const()[name = string("concat_40"), val = tensor([5, 0, 0, 0])]; tensor concat_41 = const()[name = string("concat_41"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_40, begin_mask = kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_41, end_mask = kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_11_stride_0, update = var_4793_cast_fp16, x = coreml_update_state_39)[name = string("kv_cache_0_internal_tensor_assign_11_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_11_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_40_write_state")]; tensor coreml_update_state_40 = read_state(input = kv_cache_0)[name = string("coreml_update_state_40")]; tensor var_4800_axes_0 = const()[name = string("op_4800_axes_0"), val = tensor([0])]; tensor var_4800_cast_fp16 = squeeze(axes = var_4800_axes_0, x = V_new_11_cast_fp16)[name = string("op_4800_cast_fp16")]; tensor concat_42 = const()[name = string("concat_42"), val = tensor([40, 0, 0, 0])]; tensor concat_43 = const()[name = string("concat_43"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_42, begin_mask = kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_43, end_mask = kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_12_stride_0, update = var_4800_cast_fp16, x = coreml_update_state_40)[name = string("kv_cache_0_internal_tensor_assign_12_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_12_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_41_write_state")]; tensor coreml_update_state_41 = read_state(input = kv_cache_0)[name = string("coreml_update_state_41")]; tensor K_for_attn_11_begin_0 = const()[name = string("K_for_attn_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_11_end_0 = const()[name = string("K_for_attn_11_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_11_end_mask_0 = const()[name = string("K_for_attn_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_11_cast_fp16 = slice_by_index(begin = K_for_attn_11_begin_0, end = K_for_attn_11_end_0, end_mask = K_for_attn_11_end_mask_0, x = K_new_11_cast_fp16)[name = string("K_for_attn_11_cast_fp16")]; tensor V_for_attn_11_begin_0 = const()[name = string("V_for_attn_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_11_end_0 = const()[name = string("V_for_attn_11_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_11_end_mask_0 = const()[name = string("V_for_attn_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_11_cast_fp16 = slice_by_index(begin = V_for_attn_11_begin_0, end = V_for_attn_11_end_0, end_mask = V_for_attn_11_end_mask_0, x = V_new_11_cast_fp16)[name = string("V_for_attn_11_cast_fp16")]; tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = K_for_attn_11_cast_fp16)[name = string("transpose_287")]; tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; tensor concat_44 = const()[name = string("concat_44"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_44, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 512, 256])]; tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_286")]; tensor reshape_21_cast_fp16 = reshape(shape = concat_45, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor transpose_145_perm_0 = const()[name = string("transpose_145_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = V_for_attn_11_cast_fp16)[name = string("transpose_285")]; tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; tensor concat_46 = const()[name = string("concat_46"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_46, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_47 = const()[name = string("concat_47"), val = tensor([-1, 1, 512, 256])]; tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_284")]; tensor reshape_23_cast_fp16 = reshape(shape = concat_47, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor V_expanded_11_perm_0 = const()[name = string("V_expanded_11_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_4827_transpose_x_0 = const()[name = string("op_4827_transpose_x_0"), val = bool(false)]; bool var_4827_transpose_y_0 = const()[name = string("op_4827_transpose_y_0"), val = bool(false)]; tensor transpose_145_cast_fp16 = transpose(perm = transpose_145_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_283")]; tensor var_4827_cast_fp16 = matmul(transpose_x = var_4827_transpose_x_0, transpose_y = var_4827_transpose_y_0, x = q_47, y = transpose_145_cast_fp16)[name = string("op_4827_cast_fp16")]; tensor attn_weights_33_cast_fp16 = add(x = var_4827_cast_fp16, y = causal_mask)[name = string("attn_weights_33_cast_fp16")]; int32 var_4832 = const()[name = string("op_4832"), val = int32(-1)]; tensor attn_weights_35_cast_fp16 = softmax(axis = var_4832, x = attn_weights_33_cast_fp16)[name = string("attn_weights_35_cast_fp16")]; bool attn_output_31_transpose_x_0 = const()[name = string("attn_output_31_transpose_x_0"), val = bool(false)]; bool attn_output_31_transpose_y_0 = const()[name = string("attn_output_31_transpose_y_0"), val = bool(false)]; tensor V_expanded_11_cast_fp16 = transpose(perm = V_expanded_11_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_282")]; tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = attn_weights_35_cast_fp16, y = V_expanded_11_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor var_4840 = const()[name = string("op_4840"), val = tensor([0, 2, 1, 3])]; tensor var_4847 = const()[name = string("op_4847"), val = tensor([1, 1, -1])]; tensor var_4841_cast_fp16 = transpose(perm = var_4840, x = attn_output_31_cast_fp16)[name = string("transpose_281")]; tensor attn_output_33_cast_fp16 = reshape(shape = var_4847, x = var_4841_cast_fp16)[name = string("attn_output_33_cast_fp16")]; tensor var_4852 = const()[name = string("op_4852"), val = tensor([0, 2, 1])]; string var_4868_pad_type_0 = const()[name = string("op_4868_pad_type_0"), val = string("valid")]; int32 var_4868_groups_0 = const()[name = string("op_4868_groups_0"), val = int32(1)]; tensor var_4868_strides_0 = const()[name = string("op_4868_strides_0"), val = tensor([1])]; tensor var_4868_pad_0 = const()[name = string("op_4868_pad_0"), val = tensor([0, 0])]; tensor var_4868_dilations_0 = const()[name = string("op_4868_dilations_0"), val = tensor([1])]; tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1072356608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1073929536))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4853_cast_fp16 = transpose(perm = var_4852, x = attn_output_33_cast_fp16)[name = string("transpose_280")]; tensor var_4868_cast_fp16 = conv(dilations = var_4868_dilations_0, groups = var_4868_groups_0, pad = var_4868_pad_0, pad_type = var_4868_pad_type_0, strides = var_4868_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_4853_cast_fp16)[name = string("op_4868_cast_fp16")]; tensor var_4872 = const()[name = string("op_4872"), val = tensor([0, 2, 1])]; int32 var_4878 = const()[name = string("op_4878"), val = int32(-1)]; fp16 const_99_promoted_to_fp16 = const()[name = string("const_99_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_165_cast_fp16 = transpose(perm = var_4872, x = var_4868_cast_fp16)[name = string("transpose_279")]; tensor var_4884_cast_fp16 = mul(x = x_165_cast_fp16, y = const_99_promoted_to_fp16)[name = string("op_4884_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_4878, interleave = input_161_interleave_0, values = (x_165_cast_fp16, var_4884_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_4876_to_fp16 = const()[name = string("op_4876_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_4876_to_fp16, x = input_161_cast_fp16)[name = string("normed_153_cast_fp16")]; tensor var_4889_split_sizes_0 = const()[name = string("op_4889_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4889_axis_0 = const()[name = string("op_4889_axis_0"), val = int32(-1)]; tensor var_4889_cast_fp16_0, tensor var_4889_cast_fp16_1 = split(axis = var_4889_axis_0, split_sizes = var_4889_split_sizes_0, x = normed_153_cast_fp16)[name = string("op_4889_cast_fp16")]; tensor const_100_to_fp16 = const()[name = string("const_100_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1073931136)))]; tensor var_4892_cast_fp16 = mul(x = var_4889_cast_fp16_0, y = const_100_to_fp16)[name = string("op_4892_cast_fp16")]; tensor x_169_cast_fp16 = add(x = x_151_cast_fp16, y = var_4892_cast_fp16)[name = string("x_169_cast_fp16")]; int32 var_4899 = const()[name = string("op_4899"), val = int32(-1)]; fp16 const_101_promoted_to_fp16 = const()[name = string("const_101_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4905_cast_fp16 = mul(x = x_169_cast_fp16, y = const_101_promoted_to_fp16)[name = string("op_4905_cast_fp16")]; bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; tensor input_163_cast_fp16 = concat(axis = var_4899, interleave = input_163_interleave_0, values = (x_169_cast_fp16, var_4905_cast_fp16))[name = string("input_163_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_4897_to_fp16 = const()[name = string("op_4897_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_4897_to_fp16, x = input_163_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor var_4910_split_sizes_0 = const()[name = string("op_4910_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4910_axis_0 = const()[name = string("op_4910_axis_0"), val = int32(-1)]; tensor var_4910_cast_fp16_0, tensor var_4910_cast_fp16_1 = split(axis = var_4910_axis_0, split_sizes = var_4910_split_sizes_0, x = normed_157_cast_fp16)[name = string("op_4910_cast_fp16")]; tensor const_102_to_fp16 = const()[name = string("const_102_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1073934272)))]; tensor var_4913_cast_fp16 = mul(x = var_4910_cast_fp16_0, y = const_102_to_fp16)[name = string("op_4913_cast_fp16")]; tensor var_4926 = const()[name = string("op_4926"), val = tensor([0, 2, 1])]; tensor input_165_axes_0 = const()[name = string("input_165_axes_0"), val = tensor([2])]; tensor var_4927 = transpose(perm = var_4926, x = var_4913_cast_fp16)[name = string("transpose_278")]; tensor input_165 = expand_dims(axes = input_165_axes_0, x = var_4927)[name = string("input_165")]; string gate_21_pad_type_0 = const()[name = string("gate_21_pad_type_0"), val = string("valid")]; tensor gate_21_strides_0 = const()[name = string("gate_21_strides_0"), val = tensor([1, 1])]; tensor gate_21_pad_0 = const()[name = string("gate_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_21_dilations_0 = const()[name = string("gate_21_dilations_0"), val = tensor([1, 1])]; int32 gate_21_groups_0 = const()[name = string("gate_21_groups_0"), val = int32(1)]; tensor gate_21 = conv(dilations = gate_21_dilations_0, groups = gate_21_groups_0, pad = gate_21_pad_0, pad_type = gate_21_pad_type_0, strides = gate_21_strides_0, weight = layers_5_mlp_gate_proj_weight_palettized, x = input_165)[name = string("gate_21")]; string up_11_pad_type_0 = const()[name = string("up_11_pad_type_0"), val = string("valid")]; tensor up_11_strides_0 = const()[name = string("up_11_strides_0"), val = tensor([1, 1])]; tensor up_11_pad_0 = const()[name = string("up_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_11_dilations_0 = const()[name = string("up_11_dilations_0"), val = tensor([1, 1])]; int32 up_11_groups_0 = const()[name = string("up_11_groups_0"), val = int32(1)]; tensor up_11 = conv(dilations = up_11_dilations_0, groups = up_11_groups_0, pad = up_11_pad_0, pad_type = up_11_pad_type_0, strides = up_11_strides_0, weight = layers_5_mlp_up_proj_weight_palettized, x = input_165)[name = string("up_11")]; string gate_23_mode_0 = const()[name = string("gate_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_23 = gelu(mode = gate_23_mode_0, x = gate_21)[name = string("gate_23")]; tensor input_167 = mul(x = gate_23, y = up_11)[name = string("input_167")]; string mlp_out_11_pad_type_0 = const()[name = string("mlp_out_11_pad_type_0"), val = string("valid")]; tensor mlp_out_11_strides_0 = const()[name = string("mlp_out_11_strides_0"), val = tensor([1, 1])]; tensor mlp_out_11_pad_0 = const()[name = string("mlp_out_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_11_dilations_0 = const()[name = string("mlp_out_11_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_11_groups_0 = const()[name = string("mlp_out_11_groups_0"), val = int32(1)]; tensor mlp_out_11 = conv(dilations = mlp_out_11_dilations_0, groups = mlp_out_11_groups_0, pad = mlp_out_11_pad_0, pad_type = mlp_out_11_pad_type_0, strides = mlp_out_11_strides_0, weight = layers_5_mlp_down_proj_weight_palettized, x = input_167)[name = string("mlp_out_11")]; tensor var_4967_axes_0 = const()[name = string("op_4967_axes_0"), val = tensor([2])]; tensor var_4967 = squeeze(axes = var_4967_axes_0, x = mlp_out_11)[name = string("op_4967")]; tensor var_4971 = const()[name = string("op_4971"), val = tensor([0, 2, 1])]; int32 var_4977 = const()[name = string("op_4977"), val = int32(-1)]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_173 = transpose(perm = var_4971, x = var_4967)[name = string("transpose_277")]; tensor var_4983_cast_fp16 = mul(x = x_173, y = const_103_promoted_to_fp16)[name = string("op_4983_cast_fp16")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169_cast_fp16 = concat(axis = var_4977, interleave = input_169_interleave_0, values = (x_173, var_4983_cast_fp16))[name = string("input_169_cast_fp16")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_4975_to_fp16 = const()[name = string("op_4975_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_4975_to_fp16, x = input_169_cast_fp16)[name = string("normed_161_cast_fp16")]; tensor var_4988_split_sizes_0 = const()[name = string("op_4988_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4988_axis_0 = const()[name = string("op_4988_axis_0"), val = int32(-1)]; tensor var_4988_cast_fp16_0, tensor var_4988_cast_fp16_1 = split(axis = var_4988_axis_0, split_sizes = var_4988_split_sizes_0, x = normed_161_cast_fp16)[name = string("op_4988_cast_fp16")]; tensor const_104_to_fp16 = const()[name = string("const_104_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1073937408)))]; tensor var_4991_cast_fp16 = mul(x = var_4988_cast_fp16_0, y = const_104_to_fp16)[name = string("op_4991_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = x_169_cast_fp16, y = var_4991_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; tensor per_layer_slice_11_begin_0 = const()[name = string("per_layer_slice_11_begin_0"), val = tensor([0, 0, 1280])]; tensor per_layer_slice_11_end_0 = const()[name = string("per_layer_slice_11_end_0"), val = tensor([1, 1, 1536])]; tensor per_layer_slice_11_end_mask_0 = const()[name = string("per_layer_slice_11_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_11_cast_fp16 = slice_by_index(begin = per_layer_slice_11_begin_0, end = per_layer_slice_11_end_0, end_mask = per_layer_slice_11_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_11_cast_fp16")]; tensor gated_21 = linear(bias = linear_0_bias_0, weight = layers_5_per_layer_input_gate_weight_palettized, x = hidden_states_69_cast_fp16)[name = string("linear_10")]; string gated_23_mode_0 = const()[name = string("gated_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_23 = gelu(mode = gated_23_mode_0, x = gated_21)[name = string("gated_23")]; tensor input_173_cast_fp16 = mul(x = gated_23, y = per_layer_slice_11_cast_fp16)[name = string("input_173_cast_fp16")]; tensor layers_5_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1073940544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1074137216))))[name = string("layers_5_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_11_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_5_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_173_cast_fp16)[name = string("linear_11_cast_fp16")]; int32 var_5028 = const()[name = string("op_5028"), val = int32(-1)]; fp16 const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5034_cast_fp16 = mul(x = linear_11_cast_fp16, y = const_105_promoted_to_fp16)[name = string("op_5034_cast_fp16")]; bool input_175_interleave_0 = const()[name = string("input_175_interleave_0"), val = bool(false)]; tensor input_175_cast_fp16 = concat(axis = var_5028, interleave = input_175_interleave_0, values = (linear_11_cast_fp16, var_5034_cast_fp16))[name = string("input_175_cast_fp16")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_5026_to_fp16 = const()[name = string("op_5026_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_5026_to_fp16, x = input_175_cast_fp16)[name = string("normed_165_cast_fp16")]; tensor var_5039_split_sizes_0 = const()[name = string("op_5039_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5039_axis_0 = const()[name = string("op_5039_axis_0"), val = int32(-1)]; tensor var_5039_cast_fp16_0, tensor var_5039_cast_fp16_1 = split(axis = var_5039_axis_0, split_sizes = var_5039_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_5039_cast_fp16")]; tensor const_106_to_fp16 = const()[name = string("const_106_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1074138816)))]; tensor var_5042_cast_fp16 = mul(x = var_5039_cast_fp16_0, y = const_106_to_fp16)[name = string("op_5042_cast_fp16")]; tensor hidden_states_73_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_5042_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor layers_5_layer_scalar_to_fp16 = const()[name = string("layers_5_layer_scalar_to_fp16"), val = tensor([0x1.46p-1])]; tensor x_181_cast_fp16 = mul(x = hidden_states_73_cast_fp16, y = layers_5_layer_scalar_to_fp16)[name = string("x_181_cast_fp16")]; int32 var_5050 = const()[name = string("op_5050"), val = int32(-1)]; fp16 const_107_promoted_to_fp16 = const()[name = string("const_107_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5056_cast_fp16 = mul(x = x_181_cast_fp16, y = const_107_promoted_to_fp16)[name = string("op_5056_cast_fp16")]; bool input_177_interleave_0 = const()[name = string("input_177_interleave_0"), val = bool(false)]; tensor input_177_cast_fp16 = concat(axis = var_5050, interleave = input_177_interleave_0, values = (x_181_cast_fp16, var_5056_cast_fp16))[name = string("input_177_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_5048_to_fp16 = const()[name = string("op_5048_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_5048_to_fp16, x = input_177_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_5061_split_sizes_0 = const()[name = string("op_5061_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5061_axis_0 = const()[name = string("op_5061_axis_0"), val = int32(-1)]; tensor var_5061_cast_fp16_0, tensor var_5061_cast_fp16_1 = split(axis = var_5061_axis_0, split_sizes = var_5061_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_5061_cast_fp16")]; tensor const_108_to_fp16 = const()[name = string("const_108_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1074141952)))]; tensor var_5064_cast_fp16 = mul(x = var_5061_cast_fp16_0, y = const_108_to_fp16)[name = string("op_5064_cast_fp16")]; tensor var_5072 = const()[name = string("op_5072"), val = tensor([0, 2, 1])]; tensor var_5075_axes_0 = const()[name = string("op_5075_axes_0"), val = tensor([2])]; tensor var_5073_cast_fp16 = transpose(perm = var_5072, x = var_5064_cast_fp16)[name = string("transpose_276")]; tensor var_5075_cast_fp16 = expand_dims(axes = var_5075_axes_0, x = var_5073_cast_fp16)[name = string("op_5075_cast_fp16")]; string var_5091_pad_type_0 = const()[name = string("op_5091_pad_type_0"), val = string("valid")]; tensor var_5091_strides_0 = const()[name = string("op_5091_strides_0"), val = tensor([1, 1])]; tensor var_5091_pad_0 = const()[name = string("op_5091_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5091_dilations_0 = const()[name = string("op_5091_dilations_0"), val = tensor([1, 1])]; int32 var_5091_groups_0 = const()[name = string("op_5091_groups_0"), val = int32(1)]; tensor var_5091 = conv(dilations = var_5091_dilations_0, groups = var_5091_groups_0, pad = var_5091_pad_0, pad_type = var_5091_pad_type_0, strides = var_5091_strides_0, weight = layers_6_self_attn_q_proj_weight_palettized, x = var_5075_cast_fp16)[name = string("op_5091")]; tensor var_5096 = const()[name = string("op_5096"), val = tensor([1, 8, 256, 1])]; tensor var_5097 = reshape(shape = var_5096, x = var_5091)[name = string("op_5097")]; tensor var_5102 = const()[name = string("op_5102"), val = tensor([0, 1, 3, 2])]; tensor var_5112 = const()[name = string("op_5112"), val = tensor([1, 8, 256])]; tensor var_5103 = transpose(perm = var_5102, x = var_5097)[name = string("transpose_275")]; tensor x_185 = reshape(shape = var_5112, x = var_5103)[name = string("x_185")]; int32 var_5118 = const()[name = string("op_5118"), val = int32(-1)]; fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5124_cast_fp16 = mul(x = x_185, y = const_109_promoted_to_fp16)[name = string("op_5124_cast_fp16")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181_cast_fp16 = concat(axis = var_5118, interleave = input_181_interleave_0, values = (x_185, var_5124_cast_fp16))[name = string("input_181_cast_fp16")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_5116_to_fp16 = const()[name = string("op_5116_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_5116_to_fp16, x = input_181_cast_fp16)[name = string("normed_173_cast_fp16")]; tensor var_5129_split_sizes_0 = const()[name = string("op_5129_split_sizes_0"), val = tensor([256, 256])]; int32 var_5129_axis_0 = const()[name = string("op_5129_axis_0"), val = int32(-1)]; tensor var_5129_cast_fp16_0, tensor var_5129_cast_fp16_1 = split(axis = var_5129_axis_0, split_sizes = var_5129_split_sizes_0, x = normed_173_cast_fp16)[name = string("op_5129_cast_fp16")]; tensor var_5132_cast_fp16 = mul(x = var_5129_cast_fp16_0, y = const_22_to_fp16)[name = string("op_5132_cast_fp16")]; tensor var_5138 = const()[name = string("op_5138"), val = tensor([1, 8, 1, 256])]; tensor q_51 = reshape(shape = var_5138, x = var_5132_cast_fp16)[name = string("q_51")]; tensor var_5140 = mul(x = q_51, y = cos_1)[name = string("op_5140")]; tensor var_5141_split_sizes_0 = const()[name = string("op_5141_split_sizes_0"), val = tensor([128, 128])]; int32 var_5141_axis_0 = const()[name = string("op_5141_axis_0"), val = int32(-1)]; tensor var_5141_0, tensor var_5141_1 = split(axis = var_5141_axis_0, split_sizes = var_5141_split_sizes_0, x = q_51)[name = string("op_5141")]; fp16 const_111_promoted = const()[name = string("const_111_promoted"), val = fp16(-0x1p+0)]; tensor var_5143 = mul(x = var_5141_1, y = const_111_promoted)[name = string("op_5143")]; int32 var_5145 = const()[name = string("op_5145"), val = int32(-1)]; bool var_5146_interleave_0 = const()[name = string("op_5146_interleave_0"), val = bool(false)]; tensor var_5146 = concat(axis = var_5145, interleave = var_5146_interleave_0, values = (var_5143, var_5141_0))[name = string("op_5146")]; tensor var_5147 = mul(x = var_5146, y = sin_1)[name = string("op_5147")]; tensor q_55 = add(x = var_5140, y = var_5147)[name = string("q_55")]; string var_5160_pad_type_0 = const()[name = string("op_5160_pad_type_0"), val = string("valid")]; tensor var_5160_strides_0 = const()[name = string("op_5160_strides_0"), val = tensor([1, 1])]; tensor var_5160_pad_0 = const()[name = string("op_5160_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5160_dilations_0 = const()[name = string("op_5160_dilations_0"), val = tensor([1, 1])]; int32 var_5160_groups_0 = const()[name = string("op_5160_groups_0"), val = int32(1)]; tensor var_5160 = conv(dilations = var_5160_dilations_0, groups = var_5160_groups_0, pad = var_5160_pad_0, pad_type = var_5160_pad_type_0, strides = var_5160_strides_0, weight = layers_6_self_attn_k_proj_weight_palettized, x = var_5075_cast_fp16)[name = string("op_5160")]; tensor var_5165 = const()[name = string("op_5165"), val = tensor([1, 1, 256, 1])]; tensor var_5166 = reshape(shape = var_5165, x = var_5160)[name = string("op_5166")]; tensor var_5171 = const()[name = string("op_5171"), val = tensor([0, 1, 3, 2])]; string var_5188_pad_type_0 = const()[name = string("op_5188_pad_type_0"), val = string("valid")]; tensor var_5188_strides_0 = const()[name = string("op_5188_strides_0"), val = tensor([1, 1])]; tensor var_5188_pad_0 = const()[name = string("op_5188_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5188_dilations_0 = const()[name = string("op_5188_dilations_0"), val = tensor([1, 1])]; int32 var_5188_groups_0 = const()[name = string("op_5188_groups_0"), val = int32(1)]; tensor var_5188 = conv(dilations = var_5188_dilations_0, groups = var_5188_groups_0, pad = var_5188_pad_0, pad_type = var_5188_pad_type_0, strides = var_5188_strides_0, weight = layers_6_self_attn_v_proj_weight_palettized, x = var_5075_cast_fp16)[name = string("op_5188")]; tensor var_5193 = const()[name = string("op_5193"), val = tensor([1, 1, 256, 1])]; tensor var_5194 = reshape(shape = var_5193, x = var_5188)[name = string("op_5194")]; tensor var_5199 = const()[name = string("op_5199"), val = tensor([0, 1, 3, 2])]; tensor var_5209 = const()[name = string("op_5209"), val = tensor([1, 1, 256])]; tensor var_5172 = transpose(perm = var_5171, x = var_5166)[name = string("transpose_274")]; tensor x_189 = reshape(shape = var_5209, x = var_5172)[name = string("x_189")]; int32 var_5215 = const()[name = string("op_5215"), val = int32(-1)]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5221_cast_fp16 = mul(x = x_189, y = const_112_promoted_to_fp16)[name = string("op_5221_cast_fp16")]; bool input_183_interleave_0 = const()[name = string("input_183_interleave_0"), val = bool(false)]; tensor input_183_cast_fp16 = concat(axis = var_5215, interleave = input_183_interleave_0, values = (x_189, var_5221_cast_fp16))[name = string("input_183_cast_fp16")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_5213_to_fp16 = const()[name = string("op_5213_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_5213_to_fp16, x = input_183_cast_fp16)[name = string("normed_177_cast_fp16")]; tensor var_5226_split_sizes_0 = const()[name = string("op_5226_split_sizes_0"), val = tensor([256, 256])]; int32 var_5226_axis_0 = const()[name = string("op_5226_axis_0"), val = int32(-1)]; tensor var_5226_cast_fp16_0, tensor var_5226_cast_fp16_1 = split(axis = var_5226_axis_0, split_sizes = var_5226_split_sizes_0, x = normed_177_cast_fp16)[name = string("op_5226_cast_fp16")]; tensor var_5229_cast_fp16 = mul(x = var_5226_cast_fp16_0, y = const_25_to_fp16)[name = string("op_5229_cast_fp16")]; tensor var_5235 = const()[name = string("op_5235"), val = tensor([1, 1, 1, 256])]; tensor q_53 = reshape(shape = var_5235, x = var_5229_cast_fp16)[name = string("q_53")]; fp16 var_5242_promoted_to_fp16 = const()[name = string("op_5242_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_5200 = transpose(perm = var_5199, x = var_5194)[name = string("transpose_273")]; tensor var_5243_cast_fp16 = pow(x = var_5200, y = var_5242_promoted_to_fp16)[name = string("op_5243_cast_fp16")]; tensor var_5248_axes_0 = const()[name = string("op_5248_axes_0"), val = tensor([-1])]; bool var_5248_keep_dims_0 = const()[name = string("op_5248_keep_dims_0"), val = bool(true)]; tensor var_5248_cast_fp16 = reduce_mean(axes = var_5248_axes_0, keep_dims = var_5248_keep_dims_0, x = var_5243_cast_fp16)[name = string("op_5248_cast_fp16")]; fp16 var_5250_to_fp16 = const()[name = string("op_5250_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_13_cast_fp16 = add(x = var_5248_cast_fp16, y = var_5250_to_fp16)[name = string("mean_sq_13_cast_fp16")]; fp16 var_5257_to_fp16 = const()[name = string("op_5257_to_fp16"), val = fp16(-0x1p-1)]; tensor var_5258_cast_fp16 = pow(x = mean_sq_13_cast_fp16, y = var_5257_to_fp16)[name = string("op_5258_cast_fp16")]; tensor var_5259_cast_fp16 = mul(x = var_5200, y = var_5258_cast_fp16)[name = string("op_5259_cast_fp16")]; tensor var_5265 = mul(x = q_53, y = cos_1)[name = string("op_5265")]; tensor var_5266_split_sizes_0 = const()[name = string("op_5266_split_sizes_0"), val = tensor([128, 128])]; int32 var_5266_axis_0 = const()[name = string("op_5266_axis_0"), val = int32(-1)]; tensor var_5266_0, tensor var_5266_1 = split(axis = var_5266_axis_0, split_sizes = var_5266_split_sizes_0, x = q_53)[name = string("op_5266")]; fp16 const_114_promoted = const()[name = string("const_114_promoted"), val = fp16(-0x1p+0)]; tensor var_5268 = mul(x = var_5266_1, y = const_114_promoted)[name = string("op_5268")]; int32 var_5270 = const()[name = string("op_5270"), val = int32(-1)]; bool var_5271_interleave_0 = const()[name = string("op_5271_interleave_0"), val = bool(false)]; tensor var_5271 = concat(axis = var_5270, interleave = var_5271_interleave_0, values = (var_5268, var_5266_0))[name = string("op_5271")]; tensor var_5272 = mul(x = var_5271, y = sin_1)[name = string("op_5272")]; tensor input_185 = add(x = var_5265, y = var_5272)[name = string("input_185")]; tensor var_5277_begin_0 = const()[name = string("op_5277_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_5277_end_0 = const()[name = string("op_5277_end_0"), val = tensor([7, 1, 512, 512])]; tensor var_5277_end_mask_0 = const()[name = string("op_5277_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5277_squeeze_mask_0 = const()[name = string("op_5277_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5277_cast_fp16 = slice_by_index(begin = var_5277_begin_0, end = var_5277_end_0, end_mask = var_5277_end_mask_0, squeeze_mask = var_5277_squeeze_mask_0, x = coreml_update_state_41)[name = string("op_5277_cast_fp16")]; tensor K_cache_13_axes_0 = const()[name = string("K_cache_13_axes_0"), val = tensor([0])]; tensor K_cache_13_cast_fp16 = expand_dims(axes = K_cache_13_axes_0, x = var_5277_cast_fp16)[name = string("K_cache_13_cast_fp16")]; tensor var_5282_begin_0 = const()[name = string("op_5282_begin_0"), val = tensor([41, 0, 0, 0])]; tensor var_5282_end_0 = const()[name = string("op_5282_end_0"), val = tensor([42, 1, 512, 512])]; tensor var_5282_end_mask_0 = const()[name = string("op_5282_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5282_squeeze_mask_0 = const()[name = string("op_5282_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5282_cast_fp16 = slice_by_index(begin = var_5282_begin_0, end = var_5282_end_0, end_mask = var_5282_end_mask_0, squeeze_mask = var_5282_squeeze_mask_0, x = coreml_update_state_41)[name = string("op_5282_cast_fp16")]; tensor V_cache_13_axes_0 = const()[name = string("V_cache_13_axes_0"), val = tensor([0])]; tensor V_cache_13_cast_fp16 = expand_dims(axes = V_cache_13_axes_0, x = var_5282_cast_fp16)[name = string("V_cache_13_cast_fp16")]; tensor k_padded_11_pad_0 = const()[name = string("k_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_11_mode_0 = const()[name = string("k_padded_11_mode_0"), val = string("constant")]; fp16 const_115_to_fp16 = const()[name = string("const_115_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_11_cast_fp16 = pad(constant_val = const_115_to_fp16, mode = k_padded_11_mode_0, pad = k_padded_11_pad_0, x = input_185)[name = string("k_padded_11_cast_fp16")]; tensor v_padded_11_pad_0 = const()[name = string("v_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_11_mode_0 = const()[name = string("v_padded_11_mode_0"), val = string("constant")]; fp16 const_116_to_fp16 = const()[name = string("const_116_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_11_cast_fp16 = pad(constant_val = const_116_to_fp16, mode = v_padded_11_mode_0, pad = v_padded_11_pad_0, x = var_5259_cast_fp16)[name = string("v_padded_11_cast_fp16")]; tensor var_5300_cast_fp16 = mul(x = K_cache_13_cast_fp16, y = var_2187_cast_fp16)[name = string("op_5300_cast_fp16")]; tensor var_5301_reps_0 = const()[name = string("op_5301_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_5301_cast_fp16 = tile(reps = var_5301_reps_0, x = k_padded_11_cast_fp16)[name = string("op_5301_cast_fp16")]; tensor var_5302_cast_fp16 = mul(x = var_5301_cast_fp16, y = update_mask)[name = string("op_5302_cast_fp16")]; tensor K_new_13_cast_fp16 = add(x = var_5300_cast_fp16, y = var_5302_cast_fp16)[name = string("K_new_13_cast_fp16")]; tensor var_5308_cast_fp16 = mul(x = V_cache_13_cast_fp16, y = var_2187_cast_fp16)[name = string("op_5308_cast_fp16")]; tensor var_5309_reps_0 = const()[name = string("op_5309_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_5309_cast_fp16 = tile(reps = var_5309_reps_0, x = v_padded_11_cast_fp16)[name = string("op_5309_cast_fp16")]; tensor var_5310_cast_fp16 = mul(x = var_5309_cast_fp16, y = update_mask)[name = string("op_5310_cast_fp16")]; tensor V_new_13_cast_fp16 = add(x = var_5308_cast_fp16, y = var_5310_cast_fp16)[name = string("V_new_13_cast_fp16")]; tensor var_5314_axes_0 = const()[name = string("op_5314_axes_0"), val = tensor([0])]; tensor var_5314_cast_fp16 = squeeze(axes = var_5314_axes_0, x = K_new_13_cast_fp16)[name = string("op_5314_cast_fp16")]; tensor concat_48 = const()[name = string("concat_48"), val = tensor([6, 0, 0, 0])]; tensor concat_49 = const()[name = string("concat_49"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_48, begin_mask = kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_49, end_mask = kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_13_stride_0, update = var_5314_cast_fp16, x = coreml_update_state_41)[name = string("kv_cache_0_internal_tensor_assign_13_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_13_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_42_write_state")]; tensor coreml_update_state_42 = read_state(input = kv_cache_0)[name = string("coreml_update_state_42")]; tensor var_5321_axes_0 = const()[name = string("op_5321_axes_0"), val = tensor([0])]; tensor var_5321_cast_fp16 = squeeze(axes = var_5321_axes_0, x = V_new_13_cast_fp16)[name = string("op_5321_cast_fp16")]; tensor concat_50 = const()[name = string("concat_50"), val = tensor([41, 0, 0, 0])]; tensor concat_51 = const()[name = string("concat_51"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_50, begin_mask = kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_51, end_mask = kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_14_stride_0, update = var_5321_cast_fp16, x = coreml_update_state_42)[name = string("kv_cache_0_internal_tensor_assign_14_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_14_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_43_write_state")]; tensor coreml_update_state_43 = read_state(input = kv_cache_0)[name = string("coreml_update_state_43")]; tensor K_for_attn_13_begin_0 = const()[name = string("K_for_attn_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_13_end_0 = const()[name = string("K_for_attn_13_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_13_end_mask_0 = const()[name = string("K_for_attn_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_13_cast_fp16 = slice_by_index(begin = K_for_attn_13_begin_0, end = K_for_attn_13_end_0, end_mask = K_for_attn_13_end_mask_0, x = K_new_13_cast_fp16)[name = string("K_for_attn_13_cast_fp16")]; tensor V_for_attn_13_begin_0 = const()[name = string("V_for_attn_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_13_end_0 = const()[name = string("V_for_attn_13_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_13_end_mask_0 = const()[name = string("V_for_attn_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_13_cast_fp16 = slice_by_index(begin = V_for_attn_13_begin_0, end = V_for_attn_13_end_0, end_mask = V_for_attn_13_end_mask_0, x = V_new_13_cast_fp16)[name = string("V_for_attn_13_cast_fp16")]; tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = K_for_attn_13_cast_fp16)[name = string("transpose_272")]; tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; tensor concat_52 = const()[name = string("concat_52"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_52, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_53 = const()[name = string("concat_53"), val = tensor([-1, 1, 512, 256])]; tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_271")]; tensor reshape_25_cast_fp16 = reshape(shape = concat_53, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; tensor transpose_146_perm_0 = const()[name = string("transpose_146_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = V_for_attn_13_cast_fp16)[name = string("transpose_270")]; tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; tensor concat_54 = const()[name = string("concat_54"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_54, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_55 = const()[name = string("concat_55"), val = tensor([-1, 1, 512, 256])]; tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_269")]; tensor reshape_27_cast_fp16 = reshape(shape = concat_55, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor V_expanded_13_perm_0 = const()[name = string("V_expanded_13_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_5348_transpose_x_0 = const()[name = string("op_5348_transpose_x_0"), val = bool(false)]; bool var_5348_transpose_y_0 = const()[name = string("op_5348_transpose_y_0"), val = bool(false)]; tensor transpose_146_cast_fp16 = transpose(perm = transpose_146_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_268")]; tensor var_5348_cast_fp16 = matmul(transpose_x = var_5348_transpose_x_0, transpose_y = var_5348_transpose_y_0, x = q_55, y = transpose_146_cast_fp16)[name = string("op_5348_cast_fp16")]; tensor attn_weights_39_cast_fp16 = add(x = var_5348_cast_fp16, y = causal_mask)[name = string("attn_weights_39_cast_fp16")]; int32 var_5353 = const()[name = string("op_5353"), val = int32(-1)]; tensor attn_weights_41_cast_fp16 = softmax(axis = var_5353, x = attn_weights_39_cast_fp16)[name = string("attn_weights_41_cast_fp16")]; bool attn_output_37_transpose_x_0 = const()[name = string("attn_output_37_transpose_x_0"), val = bool(false)]; bool attn_output_37_transpose_y_0 = const()[name = string("attn_output_37_transpose_y_0"), val = bool(false)]; tensor V_expanded_13_cast_fp16 = transpose(perm = V_expanded_13_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_267")]; tensor attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = attn_weights_41_cast_fp16, y = V_expanded_13_cast_fp16)[name = string("attn_output_37_cast_fp16")]; tensor var_5361 = const()[name = string("op_5361"), val = tensor([0, 2, 1, 3])]; tensor var_5368 = const()[name = string("op_5368"), val = tensor([1, 1, -1])]; tensor var_5362_cast_fp16 = transpose(perm = var_5361, x = attn_output_37_cast_fp16)[name = string("transpose_266")]; tensor attn_output_39_cast_fp16 = reshape(shape = var_5368, x = var_5362_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor var_5373 = const()[name = string("op_5373"), val = tensor([0, 2, 1])]; string var_5389_pad_type_0 = const()[name = string("op_5389_pad_type_0"), val = string("valid")]; int32 var_5389_groups_0 = const()[name = string("op_5389_groups_0"), val = int32(1)]; tensor var_5389_strides_0 = const()[name = string("op_5389_strides_0"), val = tensor([1])]; tensor var_5389_pad_0 = const()[name = string("op_5389_pad_0"), val = tensor([0, 0])]; tensor var_5389_dilations_0 = const()[name = string("op_5389_dilations_0"), val = tensor([1])]; tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1074145088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075718016))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5374_cast_fp16 = transpose(perm = var_5373, x = attn_output_39_cast_fp16)[name = string("transpose_265")]; tensor var_5389_cast_fp16 = conv(dilations = var_5389_dilations_0, groups = var_5389_groups_0, pad = var_5389_pad_0, pad_type = var_5389_pad_type_0, strides = var_5389_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_5374_cast_fp16)[name = string("op_5389_cast_fp16")]; tensor var_5393 = const()[name = string("op_5393"), val = tensor([0, 2, 1])]; int32 var_5399 = const()[name = string("op_5399"), val = int32(-1)]; fp16 const_117_promoted_to_fp16 = const()[name = string("const_117_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_195_cast_fp16 = transpose(perm = var_5393, x = var_5389_cast_fp16)[name = string("transpose_264")]; tensor var_5405_cast_fp16 = mul(x = x_195_cast_fp16, y = const_117_promoted_to_fp16)[name = string("op_5405_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_5399, interleave = input_191_interleave_0, values = (x_195_cast_fp16, var_5405_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_5397_to_fp16 = const()[name = string("op_5397_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_5397_to_fp16, x = input_191_cast_fp16)[name = string("normed_181_cast_fp16")]; tensor var_5410_split_sizes_0 = const()[name = string("op_5410_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5410_axis_0 = const()[name = string("op_5410_axis_0"), val = int32(-1)]; tensor var_5410_cast_fp16_0, tensor var_5410_cast_fp16_1 = split(axis = var_5410_axis_0, split_sizes = var_5410_split_sizes_0, x = normed_181_cast_fp16)[name = string("op_5410_cast_fp16")]; tensor const_118_to_fp16 = const()[name = string("const_118_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075719616)))]; tensor var_5413_cast_fp16 = mul(x = var_5410_cast_fp16_0, y = const_118_to_fp16)[name = string("op_5413_cast_fp16")]; tensor x_199_cast_fp16 = add(x = x_181_cast_fp16, y = var_5413_cast_fp16)[name = string("x_199_cast_fp16")]; int32 var_5420 = const()[name = string("op_5420"), val = int32(-1)]; fp16 const_119_promoted_to_fp16 = const()[name = string("const_119_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5426_cast_fp16 = mul(x = x_199_cast_fp16, y = const_119_promoted_to_fp16)[name = string("op_5426_cast_fp16")]; bool input_193_interleave_0 = const()[name = string("input_193_interleave_0"), val = bool(false)]; tensor input_193_cast_fp16 = concat(axis = var_5420, interleave = input_193_interleave_0, values = (x_199_cast_fp16, var_5426_cast_fp16))[name = string("input_193_cast_fp16")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_5418_to_fp16 = const()[name = string("op_5418_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_5418_to_fp16, x = input_193_cast_fp16)[name = string("normed_185_cast_fp16")]; tensor var_5431_split_sizes_0 = const()[name = string("op_5431_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5431_axis_0 = const()[name = string("op_5431_axis_0"), val = int32(-1)]; tensor var_5431_cast_fp16_0, tensor var_5431_cast_fp16_1 = split(axis = var_5431_axis_0, split_sizes = var_5431_split_sizes_0, x = normed_185_cast_fp16)[name = string("op_5431_cast_fp16")]; tensor const_120_to_fp16 = const()[name = string("const_120_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075722752)))]; tensor var_5434_cast_fp16 = mul(x = var_5431_cast_fp16_0, y = const_120_to_fp16)[name = string("op_5434_cast_fp16")]; tensor var_5447 = const()[name = string("op_5447"), val = tensor([0, 2, 1])]; tensor input_195_axes_0 = const()[name = string("input_195_axes_0"), val = tensor([2])]; tensor var_5448 = transpose(perm = var_5447, x = var_5434_cast_fp16)[name = string("transpose_263")]; tensor input_195 = expand_dims(axes = input_195_axes_0, x = var_5448)[name = string("input_195")]; string gate_25_pad_type_0 = const()[name = string("gate_25_pad_type_0"), val = string("valid")]; tensor gate_25_strides_0 = const()[name = string("gate_25_strides_0"), val = tensor([1, 1])]; tensor gate_25_pad_0 = const()[name = string("gate_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_25_dilations_0 = const()[name = string("gate_25_dilations_0"), val = tensor([1, 1])]; int32 gate_25_groups_0 = const()[name = string("gate_25_groups_0"), val = int32(1)]; tensor gate_25 = conv(dilations = gate_25_dilations_0, groups = gate_25_groups_0, pad = gate_25_pad_0, pad_type = gate_25_pad_type_0, strides = gate_25_strides_0, weight = layers_6_mlp_gate_proj_weight_palettized, x = input_195)[name = string("gate_25")]; string up_13_pad_type_0 = const()[name = string("up_13_pad_type_0"), val = string("valid")]; tensor up_13_strides_0 = const()[name = string("up_13_strides_0"), val = tensor([1, 1])]; tensor up_13_pad_0 = const()[name = string("up_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_13_dilations_0 = const()[name = string("up_13_dilations_0"), val = tensor([1, 1])]; int32 up_13_groups_0 = const()[name = string("up_13_groups_0"), val = int32(1)]; tensor up_13 = conv(dilations = up_13_dilations_0, groups = up_13_groups_0, pad = up_13_pad_0, pad_type = up_13_pad_type_0, strides = up_13_strides_0, weight = layers_6_mlp_up_proj_weight_palettized, x = input_195)[name = string("up_13")]; string gate_27_mode_0 = const()[name = string("gate_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_27 = gelu(mode = gate_27_mode_0, x = gate_25)[name = string("gate_27")]; tensor input_197 = mul(x = gate_27, y = up_13)[name = string("input_197")]; string mlp_out_13_pad_type_0 = const()[name = string("mlp_out_13_pad_type_0"), val = string("valid")]; tensor mlp_out_13_strides_0 = const()[name = string("mlp_out_13_strides_0"), val = tensor([1, 1])]; tensor mlp_out_13_pad_0 = const()[name = string("mlp_out_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_13_dilations_0 = const()[name = string("mlp_out_13_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_13_groups_0 = const()[name = string("mlp_out_13_groups_0"), val = int32(1)]; tensor mlp_out_13 = conv(dilations = mlp_out_13_dilations_0, groups = mlp_out_13_groups_0, pad = mlp_out_13_pad_0, pad_type = mlp_out_13_pad_type_0, strides = mlp_out_13_strides_0, weight = layers_6_mlp_down_proj_weight_palettized, x = input_197)[name = string("mlp_out_13")]; tensor var_5488_axes_0 = const()[name = string("op_5488_axes_0"), val = tensor([2])]; tensor var_5488 = squeeze(axes = var_5488_axes_0, x = mlp_out_13)[name = string("op_5488")]; tensor var_5492 = const()[name = string("op_5492"), val = tensor([0, 2, 1])]; int32 var_5498 = const()[name = string("op_5498"), val = int32(-1)]; fp16 const_121_promoted_to_fp16 = const()[name = string("const_121_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_203 = transpose(perm = var_5492, x = var_5488)[name = string("transpose_262")]; tensor var_5504_cast_fp16 = mul(x = x_203, y = const_121_promoted_to_fp16)[name = string("op_5504_cast_fp16")]; bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; tensor input_199_cast_fp16 = concat(axis = var_5498, interleave = input_199_interleave_0, values = (x_203, var_5504_cast_fp16))[name = string("input_199_cast_fp16")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_5496_to_fp16 = const()[name = string("op_5496_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_5496_to_fp16, x = input_199_cast_fp16)[name = string("normed_189_cast_fp16")]; tensor var_5509_split_sizes_0 = const()[name = string("op_5509_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5509_axis_0 = const()[name = string("op_5509_axis_0"), val = int32(-1)]; tensor var_5509_cast_fp16_0, tensor var_5509_cast_fp16_1 = split(axis = var_5509_axis_0, split_sizes = var_5509_split_sizes_0, x = normed_189_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor const_122_to_fp16 = const()[name = string("const_122_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075725888)))]; tensor var_5512_cast_fp16 = mul(x = var_5509_cast_fp16_0, y = const_122_to_fp16)[name = string("op_5512_cast_fp16")]; tensor hidden_states_81_cast_fp16 = add(x = x_199_cast_fp16, y = var_5512_cast_fp16)[name = string("hidden_states_81_cast_fp16")]; tensor per_layer_slice_13_begin_0 = const()[name = string("per_layer_slice_13_begin_0"), val = tensor([0, 0, 1536])]; tensor per_layer_slice_13_end_0 = const()[name = string("per_layer_slice_13_end_0"), val = tensor([1, 1, 1792])]; tensor per_layer_slice_13_end_mask_0 = const()[name = string("per_layer_slice_13_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_13_cast_fp16 = slice_by_index(begin = per_layer_slice_13_begin_0, end = per_layer_slice_13_end_0, end_mask = per_layer_slice_13_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_13_cast_fp16")]; tensor gated_25 = linear(bias = linear_0_bias_0, weight = layers_6_per_layer_input_gate_weight_palettized, x = hidden_states_81_cast_fp16)[name = string("linear_12")]; string gated_27_mode_0 = const()[name = string("gated_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_27 = gelu(mode = gated_27_mode_0, x = gated_25)[name = string("gated_27")]; tensor input_203_cast_fp16 = mul(x = gated_27, y = per_layer_slice_13_cast_fp16)[name = string("input_203_cast_fp16")]; tensor layers_6_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075729024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075925696))))[name = string("layers_6_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_13_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_6_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_203_cast_fp16)[name = string("linear_13_cast_fp16")]; int32 var_5549 = const()[name = string("op_5549"), val = int32(-1)]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5555_cast_fp16 = mul(x = linear_13_cast_fp16, y = const_123_promoted_to_fp16)[name = string("op_5555_cast_fp16")]; bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; tensor input_205_cast_fp16 = concat(axis = var_5549, interleave = input_205_interleave_0, values = (linear_13_cast_fp16, var_5555_cast_fp16))[name = string("input_205_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_5547_to_fp16 = const()[name = string("op_5547_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_5547_to_fp16, x = input_205_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor var_5560_split_sizes_0 = const()[name = string("op_5560_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5560_axis_0 = const()[name = string("op_5560_axis_0"), val = int32(-1)]; tensor var_5560_cast_fp16_0, tensor var_5560_cast_fp16_1 = split(axis = var_5560_axis_0, split_sizes = var_5560_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_5560_cast_fp16")]; tensor const_124_to_fp16 = const()[name = string("const_124_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075927296)))]; tensor var_5563_cast_fp16 = mul(x = var_5560_cast_fp16_0, y = const_124_to_fp16)[name = string("op_5563_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = var_5563_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; tensor layers_6_layer_scalar_to_fp16 = const()[name = string("layers_6_layer_scalar_to_fp16"), val = tensor([0x1.fep-2])]; tensor x_211_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = layers_6_layer_scalar_to_fp16)[name = string("x_211_cast_fp16")]; int32 var_5571 = const()[name = string("op_5571"), val = int32(-1)]; fp16 const_125_promoted_to_fp16 = const()[name = string("const_125_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5577_cast_fp16 = mul(x = x_211_cast_fp16, y = const_125_promoted_to_fp16)[name = string("op_5577_cast_fp16")]; bool input_207_interleave_0 = const()[name = string("input_207_interleave_0"), val = bool(false)]; tensor input_207_cast_fp16 = concat(axis = var_5571, interleave = input_207_interleave_0, values = (x_211_cast_fp16, var_5577_cast_fp16))[name = string("input_207_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_5569_to_fp16 = const()[name = string("op_5569_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_5569_to_fp16, x = input_207_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_5582_split_sizes_0 = const()[name = string("op_5582_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(-1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = var_5582_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor const_126_to_fp16 = const()[name = string("const_126_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075930432)))]; tensor var_5585_cast_fp16 = mul(x = var_5582_cast_fp16_0, y = const_126_to_fp16)[name = string("op_5585_cast_fp16")]; tensor var_5593 = const()[name = string("op_5593"), val = tensor([0, 2, 1])]; tensor var_5596_axes_0 = const()[name = string("op_5596_axes_0"), val = tensor([2])]; tensor var_5594_cast_fp16 = transpose(perm = var_5593, x = var_5585_cast_fp16)[name = string("transpose_261")]; tensor var_5596_cast_fp16 = expand_dims(axes = var_5596_axes_0, x = var_5594_cast_fp16)[name = string("op_5596_cast_fp16")]; string var_5612_pad_type_0 = const()[name = string("op_5612_pad_type_0"), val = string("valid")]; tensor var_5612_strides_0 = const()[name = string("op_5612_strides_0"), val = tensor([1, 1])]; tensor var_5612_pad_0 = const()[name = string("op_5612_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5612_dilations_0 = const()[name = string("op_5612_dilations_0"), val = tensor([1, 1])]; int32 var_5612_groups_0 = const()[name = string("op_5612_groups_0"), val = int32(1)]; tensor var_5612 = conv(dilations = var_5612_dilations_0, groups = var_5612_groups_0, pad = var_5612_pad_0, pad_type = var_5612_pad_type_0, strides = var_5612_strides_0, weight = layers_7_self_attn_q_proj_weight_palettized, x = var_5596_cast_fp16)[name = string("op_5612")]; tensor var_5617 = const()[name = string("op_5617"), val = tensor([1, 8, 256, 1])]; tensor var_5618 = reshape(shape = var_5617, x = var_5612)[name = string("op_5618")]; tensor var_5623 = const()[name = string("op_5623"), val = tensor([0, 1, 3, 2])]; tensor var_5633 = const()[name = string("op_5633"), val = tensor([1, 8, 256])]; tensor var_5624 = transpose(perm = var_5623, x = var_5618)[name = string("transpose_260")]; tensor x_215 = reshape(shape = var_5633, x = var_5624)[name = string("x_215")]; int32 var_5639 = const()[name = string("op_5639"), val = int32(-1)]; fp16 const_127_promoted_to_fp16 = const()[name = string("const_127_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5645_cast_fp16 = mul(x = x_215, y = const_127_promoted_to_fp16)[name = string("op_5645_cast_fp16")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211_cast_fp16 = concat(axis = var_5639, interleave = input_211_interleave_0, values = (x_215, var_5645_cast_fp16))[name = string("input_211_cast_fp16")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_5637_to_fp16 = const()[name = string("op_5637_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_5637_to_fp16, x = input_211_cast_fp16)[name = string("normed_201_cast_fp16")]; tensor var_5650_split_sizes_0 = const()[name = string("op_5650_split_sizes_0"), val = tensor([256, 256])]; int32 var_5650_axis_0 = const()[name = string("op_5650_axis_0"), val = int32(-1)]; tensor var_5650_cast_fp16_0, tensor var_5650_cast_fp16_1 = split(axis = var_5650_axis_0, split_sizes = var_5650_split_sizes_0, x = normed_201_cast_fp16)[name = string("op_5650_cast_fp16")]; tensor const_128_to_fp16 = const()[name = string("const_128_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075933568)))]; tensor var_5653_cast_fp16 = mul(x = var_5650_cast_fp16_0, y = const_128_to_fp16)[name = string("op_5653_cast_fp16")]; tensor var_5659 = const()[name = string("op_5659"), val = tensor([1, 8, 1, 256])]; tensor q_59 = reshape(shape = var_5659, x = var_5653_cast_fp16)[name = string("q_59")]; tensor var_5661 = mul(x = q_59, y = cos_1)[name = string("op_5661")]; tensor var_5662_split_sizes_0 = const()[name = string("op_5662_split_sizes_0"), val = tensor([128, 128])]; int32 var_5662_axis_0 = const()[name = string("op_5662_axis_0"), val = int32(-1)]; tensor var_5662_0, tensor var_5662_1 = split(axis = var_5662_axis_0, split_sizes = var_5662_split_sizes_0, x = q_59)[name = string("op_5662")]; fp16 const_129_promoted = const()[name = string("const_129_promoted"), val = fp16(-0x1p+0)]; tensor var_5664 = mul(x = var_5662_1, y = const_129_promoted)[name = string("op_5664")]; int32 var_5666 = const()[name = string("op_5666"), val = int32(-1)]; bool var_5667_interleave_0 = const()[name = string("op_5667_interleave_0"), val = bool(false)]; tensor var_5667 = concat(axis = var_5666, interleave = var_5667_interleave_0, values = (var_5664, var_5662_0))[name = string("op_5667")]; tensor var_5668 = mul(x = var_5667, y = sin_1)[name = string("op_5668")]; tensor q_63 = add(x = var_5661, y = var_5668)[name = string("q_63")]; string var_5681_pad_type_0 = const()[name = string("op_5681_pad_type_0"), val = string("valid")]; tensor var_5681_strides_0 = const()[name = string("op_5681_strides_0"), val = tensor([1, 1])]; tensor var_5681_pad_0 = const()[name = string("op_5681_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5681_dilations_0 = const()[name = string("op_5681_dilations_0"), val = tensor([1, 1])]; int32 var_5681_groups_0 = const()[name = string("op_5681_groups_0"), val = int32(1)]; tensor var_5681 = conv(dilations = var_5681_dilations_0, groups = var_5681_groups_0, pad = var_5681_pad_0, pad_type = var_5681_pad_type_0, strides = var_5681_strides_0, weight = layers_7_self_attn_k_proj_weight_palettized, x = var_5596_cast_fp16)[name = string("op_5681")]; tensor var_5686 = const()[name = string("op_5686"), val = tensor([1, 1, 256, 1])]; tensor var_5687 = reshape(shape = var_5686, x = var_5681)[name = string("op_5687")]; tensor var_5692 = const()[name = string("op_5692"), val = tensor([0, 1, 3, 2])]; string var_5709_pad_type_0 = const()[name = string("op_5709_pad_type_0"), val = string("valid")]; tensor var_5709_strides_0 = const()[name = string("op_5709_strides_0"), val = tensor([1, 1])]; tensor var_5709_pad_0 = const()[name = string("op_5709_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5709_dilations_0 = const()[name = string("op_5709_dilations_0"), val = tensor([1, 1])]; int32 var_5709_groups_0 = const()[name = string("op_5709_groups_0"), val = int32(1)]; tensor var_5709 = conv(dilations = var_5709_dilations_0, groups = var_5709_groups_0, pad = var_5709_pad_0, pad_type = var_5709_pad_type_0, strides = var_5709_strides_0, weight = layers_7_self_attn_v_proj_weight_palettized, x = var_5596_cast_fp16)[name = string("op_5709")]; tensor var_5714 = const()[name = string("op_5714"), val = tensor([1, 1, 256, 1])]; tensor var_5715 = reshape(shape = var_5714, x = var_5709)[name = string("op_5715")]; tensor var_5720 = const()[name = string("op_5720"), val = tensor([0, 1, 3, 2])]; tensor var_5730 = const()[name = string("op_5730"), val = tensor([1, 1, 256])]; tensor var_5693 = transpose(perm = var_5692, x = var_5687)[name = string("transpose_259")]; tensor x_219 = reshape(shape = var_5730, x = var_5693)[name = string("x_219")]; int32 var_5736 = const()[name = string("op_5736"), val = int32(-1)]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5742_cast_fp16 = mul(x = x_219, y = const_130_promoted_to_fp16)[name = string("op_5742_cast_fp16")]; bool input_213_interleave_0 = const()[name = string("input_213_interleave_0"), val = bool(false)]; tensor input_213_cast_fp16 = concat(axis = var_5736, interleave = input_213_interleave_0, values = (x_219, var_5742_cast_fp16))[name = string("input_213_cast_fp16")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_5734_to_fp16 = const()[name = string("op_5734_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_5734_to_fp16, x = input_213_cast_fp16)[name = string("normed_205_cast_fp16")]; tensor var_5747_split_sizes_0 = const()[name = string("op_5747_split_sizes_0"), val = tensor([256, 256])]; int32 var_5747_axis_0 = const()[name = string("op_5747_axis_0"), val = int32(-1)]; tensor var_5747_cast_fp16_0, tensor var_5747_cast_fp16_1 = split(axis = var_5747_axis_0, split_sizes = var_5747_split_sizes_0, x = normed_205_cast_fp16)[name = string("op_5747_cast_fp16")]; tensor const_131_to_fp16 = const()[name = string("const_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075934144)))]; tensor var_5750_cast_fp16 = mul(x = var_5747_cast_fp16_0, y = const_131_to_fp16)[name = string("op_5750_cast_fp16")]; tensor var_5756 = const()[name = string("op_5756"), val = tensor([1, 1, 1, 256])]; tensor q_61 = reshape(shape = var_5756, x = var_5750_cast_fp16)[name = string("q_61")]; fp16 var_5763_promoted_to_fp16 = const()[name = string("op_5763_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_5721 = transpose(perm = var_5720, x = var_5715)[name = string("transpose_258")]; tensor var_5764_cast_fp16 = pow(x = var_5721, y = var_5763_promoted_to_fp16)[name = string("op_5764_cast_fp16")]; tensor var_5769_axes_0 = const()[name = string("op_5769_axes_0"), val = tensor([-1])]; bool var_5769_keep_dims_0 = const()[name = string("op_5769_keep_dims_0"), val = bool(true)]; tensor var_5769_cast_fp16 = reduce_mean(axes = var_5769_axes_0, keep_dims = var_5769_keep_dims_0, x = var_5764_cast_fp16)[name = string("op_5769_cast_fp16")]; fp16 var_5771_to_fp16 = const()[name = string("op_5771_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_15_cast_fp16 = add(x = var_5769_cast_fp16, y = var_5771_to_fp16)[name = string("mean_sq_15_cast_fp16")]; fp16 var_5778_to_fp16 = const()[name = string("op_5778_to_fp16"), val = fp16(-0x1p-1)]; tensor var_5779_cast_fp16 = pow(x = mean_sq_15_cast_fp16, y = var_5778_to_fp16)[name = string("op_5779_cast_fp16")]; tensor var_5780_cast_fp16 = mul(x = var_5721, y = var_5779_cast_fp16)[name = string("op_5780_cast_fp16")]; tensor var_5786 = mul(x = q_61, y = cos_1)[name = string("op_5786")]; tensor var_5787_split_sizes_0 = const()[name = string("op_5787_split_sizes_0"), val = tensor([128, 128])]; int32 var_5787_axis_0 = const()[name = string("op_5787_axis_0"), val = int32(-1)]; tensor var_5787_0, tensor var_5787_1 = split(axis = var_5787_axis_0, split_sizes = var_5787_split_sizes_0, x = q_61)[name = string("op_5787")]; fp16 const_132_promoted = const()[name = string("const_132_promoted"), val = fp16(-0x1p+0)]; tensor var_5789 = mul(x = var_5787_1, y = const_132_promoted)[name = string("op_5789")]; int32 var_5791 = const()[name = string("op_5791"), val = int32(-1)]; bool var_5792_interleave_0 = const()[name = string("op_5792_interleave_0"), val = bool(false)]; tensor var_5792 = concat(axis = var_5791, interleave = var_5792_interleave_0, values = (var_5789, var_5787_0))[name = string("op_5792")]; tensor var_5793 = mul(x = var_5792, y = sin_1)[name = string("op_5793")]; tensor input_215 = add(x = var_5786, y = var_5793)[name = string("input_215")]; tensor var_5798_begin_0 = const()[name = string("op_5798_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5798_end_0 = const()[name = string("op_5798_end_0"), val = tensor([8, 1, 512, 512])]; tensor var_5798_end_mask_0 = const()[name = string("op_5798_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5798_squeeze_mask_0 = const()[name = string("op_5798_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5798_cast_fp16 = slice_by_index(begin = var_5798_begin_0, end = var_5798_end_0, end_mask = var_5798_end_mask_0, squeeze_mask = var_5798_squeeze_mask_0, x = coreml_update_state_43)[name = string("op_5798_cast_fp16")]; tensor K_cache_15_axes_0 = const()[name = string("K_cache_15_axes_0"), val = tensor([0])]; tensor K_cache_15_cast_fp16 = expand_dims(axes = K_cache_15_axes_0, x = var_5798_cast_fp16)[name = string("K_cache_15_cast_fp16")]; tensor var_5803_begin_0 = const()[name = string("op_5803_begin_0"), val = tensor([42, 0, 0, 0])]; tensor var_5803_end_0 = const()[name = string("op_5803_end_0"), val = tensor([43, 1, 512, 512])]; tensor var_5803_end_mask_0 = const()[name = string("op_5803_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5803_squeeze_mask_0 = const()[name = string("op_5803_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5803_cast_fp16 = slice_by_index(begin = var_5803_begin_0, end = var_5803_end_0, end_mask = var_5803_end_mask_0, squeeze_mask = var_5803_squeeze_mask_0, x = coreml_update_state_43)[name = string("op_5803_cast_fp16")]; tensor V_cache_15_axes_0 = const()[name = string("V_cache_15_axes_0"), val = tensor([0])]; tensor V_cache_15_cast_fp16 = expand_dims(axes = V_cache_15_axes_0, x = var_5803_cast_fp16)[name = string("V_cache_15_cast_fp16")]; tensor k_padded_13_pad_0 = const()[name = string("k_padded_13_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_13_mode_0 = const()[name = string("k_padded_13_mode_0"), val = string("constant")]; fp16 const_133_to_fp16 = const()[name = string("const_133_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_13_cast_fp16 = pad(constant_val = const_133_to_fp16, mode = k_padded_13_mode_0, pad = k_padded_13_pad_0, x = input_215)[name = string("k_padded_13_cast_fp16")]; tensor v_padded_13_pad_0 = const()[name = string("v_padded_13_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_13_mode_0 = const()[name = string("v_padded_13_mode_0"), val = string("constant")]; fp16 const_134_to_fp16 = const()[name = string("const_134_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_13_cast_fp16 = pad(constant_val = const_134_to_fp16, mode = v_padded_13_mode_0, pad = v_padded_13_pad_0, x = var_5780_cast_fp16)[name = string("v_padded_13_cast_fp16")]; tensor var_5821_cast_fp16 = mul(x = K_cache_15_cast_fp16, y = var_2187_cast_fp16)[name = string("op_5821_cast_fp16")]; tensor var_5822_reps_0 = const()[name = string("op_5822_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_5822_cast_fp16 = tile(reps = var_5822_reps_0, x = k_padded_13_cast_fp16)[name = string("op_5822_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5822_cast_fp16, y = update_mask)[name = string("op_5823_cast_fp16")]; tensor K_new_15_cast_fp16 = add(x = var_5821_cast_fp16, y = var_5823_cast_fp16)[name = string("K_new_15_cast_fp16")]; tensor var_5829_cast_fp16 = mul(x = V_cache_15_cast_fp16, y = var_2187_cast_fp16)[name = string("op_5829_cast_fp16")]; tensor var_5830_reps_0 = const()[name = string("op_5830_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_5830_cast_fp16 = tile(reps = var_5830_reps_0, x = v_padded_13_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor var_5831_cast_fp16 = mul(x = var_5830_cast_fp16, y = update_mask)[name = string("op_5831_cast_fp16")]; tensor V_new_15_cast_fp16 = add(x = var_5829_cast_fp16, y = var_5831_cast_fp16)[name = string("V_new_15_cast_fp16")]; tensor var_5835_axes_0 = const()[name = string("op_5835_axes_0"), val = tensor([0])]; tensor var_5835_cast_fp16 = squeeze(axes = var_5835_axes_0, x = K_new_15_cast_fp16)[name = string("op_5835_cast_fp16")]; tensor concat_56 = const()[name = string("concat_56"), val = tensor([7, 0, 0, 0])]; tensor concat_57 = const()[name = string("concat_57"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_56, begin_mask = kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_57, end_mask = kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_15_stride_0, update = var_5835_cast_fp16, x = coreml_update_state_43)[name = string("kv_cache_0_internal_tensor_assign_15_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_15_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_44_write_state")]; tensor coreml_update_state_44 = read_state(input = kv_cache_0)[name = string("coreml_update_state_44")]; tensor var_5842_axes_0 = const()[name = string("op_5842_axes_0"), val = tensor([0])]; tensor var_5842_cast_fp16 = squeeze(axes = var_5842_axes_0, x = V_new_15_cast_fp16)[name = string("op_5842_cast_fp16")]; tensor concat_58 = const()[name = string("concat_58"), val = tensor([42, 0, 0, 0])]; tensor concat_59 = const()[name = string("concat_59"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_58, begin_mask = kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_59, end_mask = kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_16_stride_0, update = var_5842_cast_fp16, x = coreml_update_state_44)[name = string("kv_cache_0_internal_tensor_assign_16_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_16_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_45_write_state")]; tensor coreml_update_state_45 = read_state(input = kv_cache_0)[name = string("coreml_update_state_45")]; tensor K_for_attn_15_begin_0 = const()[name = string("K_for_attn_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_15_end_0 = const()[name = string("K_for_attn_15_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_15_end_mask_0 = const()[name = string("K_for_attn_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_15_cast_fp16 = slice_by_index(begin = K_for_attn_15_begin_0, end = K_for_attn_15_end_0, end_mask = K_for_attn_15_end_mask_0, x = K_new_15_cast_fp16)[name = string("K_for_attn_15_cast_fp16")]; tensor V_for_attn_15_begin_0 = const()[name = string("V_for_attn_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_15_end_0 = const()[name = string("V_for_attn_15_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_15_end_mask_0 = const()[name = string("V_for_attn_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_15_cast_fp16 = slice_by_index(begin = V_for_attn_15_begin_0, end = V_for_attn_15_end_0, end_mask = V_for_attn_15_end_mask_0, x = V_new_15_cast_fp16)[name = string("V_for_attn_15_cast_fp16")]; tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = K_for_attn_15_cast_fp16)[name = string("transpose_257")]; tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; tensor concat_60 = const()[name = string("concat_60"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_60, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_61 = const()[name = string("concat_61"), val = tensor([-1, 1, 512, 256])]; tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_256")]; tensor reshape_29_cast_fp16 = reshape(shape = concat_61, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor transpose_147_perm_0 = const()[name = string("transpose_147_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = V_for_attn_15_cast_fp16)[name = string("transpose_255")]; tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; tensor concat_62 = const()[name = string("concat_62"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_62, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_63 = const()[name = string("concat_63"), val = tensor([-1, 1, 512, 256])]; tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_254")]; tensor reshape_31_cast_fp16 = reshape(shape = concat_63, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; tensor V_expanded_15_perm_0 = const()[name = string("V_expanded_15_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_5869_transpose_x_0 = const()[name = string("op_5869_transpose_x_0"), val = bool(false)]; bool var_5869_transpose_y_0 = const()[name = string("op_5869_transpose_y_0"), val = bool(false)]; tensor transpose_147_cast_fp16 = transpose(perm = transpose_147_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_253")]; tensor var_5869_cast_fp16 = matmul(transpose_x = var_5869_transpose_x_0, transpose_y = var_5869_transpose_y_0, x = q_63, y = transpose_147_cast_fp16)[name = string("op_5869_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = var_5869_cast_fp16, y = causal_mask)[name = string("attn_weights_45_cast_fp16")]; int32 var_5874 = const()[name = string("op_5874"), val = int32(-1)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_5874, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_43_transpose_x_0 = const()[name = string("attn_output_43_transpose_x_0"), val = bool(false)]; bool attn_output_43_transpose_y_0 = const()[name = string("attn_output_43_transpose_y_0"), val = bool(false)]; tensor V_expanded_15_cast_fp16 = transpose(perm = V_expanded_15_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_252")]; tensor attn_output_43_cast_fp16 = matmul(transpose_x = attn_output_43_transpose_x_0, transpose_y = attn_output_43_transpose_y_0, x = attn_weights_47_cast_fp16, y = V_expanded_15_cast_fp16)[name = string("attn_output_43_cast_fp16")]; tensor var_5882 = const()[name = string("op_5882"), val = tensor([0, 2, 1, 3])]; tensor var_5889 = const()[name = string("op_5889"), val = tensor([1, 1, -1])]; tensor var_5883_cast_fp16 = transpose(perm = var_5882, x = attn_output_43_cast_fp16)[name = string("transpose_251")]; tensor attn_output_45_cast_fp16 = reshape(shape = var_5889, x = var_5883_cast_fp16)[name = string("attn_output_45_cast_fp16")]; tensor var_5894 = const()[name = string("op_5894"), val = tensor([0, 2, 1])]; string var_5910_pad_type_0 = const()[name = string("op_5910_pad_type_0"), val = string("valid")]; int32 var_5910_groups_0 = const()[name = string("op_5910_groups_0"), val = int32(1)]; tensor var_5910_strides_0 = const()[name = string("op_5910_strides_0"), val = tensor([1])]; tensor var_5910_pad_0 = const()[name = string("op_5910_pad_0"), val = tensor([0, 0])]; tensor var_5910_dilations_0 = const()[name = string("op_5910_dilations_0"), val = tensor([1])]; tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1075934720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077507648))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5895_cast_fp16 = transpose(perm = var_5894, x = attn_output_45_cast_fp16)[name = string("transpose_250")]; tensor var_5910_cast_fp16 = conv(dilations = var_5910_dilations_0, groups = var_5910_groups_0, pad = var_5910_pad_0, pad_type = var_5910_pad_type_0, strides = var_5910_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_5895_cast_fp16)[name = string("op_5910_cast_fp16")]; tensor var_5914 = const()[name = string("op_5914"), val = tensor([0, 2, 1])]; int32 var_5920 = const()[name = string("op_5920"), val = int32(-1)]; fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_225_cast_fp16 = transpose(perm = var_5914, x = var_5910_cast_fp16)[name = string("transpose_249")]; tensor var_5926_cast_fp16 = mul(x = x_225_cast_fp16, y = const_135_promoted_to_fp16)[name = string("op_5926_cast_fp16")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221_cast_fp16 = concat(axis = var_5920, interleave = input_221_interleave_0, values = (x_225_cast_fp16, var_5926_cast_fp16))[name = string("input_221_cast_fp16")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_5918_to_fp16 = const()[name = string("op_5918_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_5918_to_fp16, x = input_221_cast_fp16)[name = string("normed_209_cast_fp16")]; tensor var_5931_split_sizes_0 = const()[name = string("op_5931_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5931_axis_0 = const()[name = string("op_5931_axis_0"), val = int32(-1)]; tensor var_5931_cast_fp16_0, tensor var_5931_cast_fp16_1 = split(axis = var_5931_axis_0, split_sizes = var_5931_split_sizes_0, x = normed_209_cast_fp16)[name = string("op_5931_cast_fp16")]; tensor const_136_to_fp16 = const()[name = string("const_136_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077509248)))]; tensor var_5934_cast_fp16 = mul(x = var_5931_cast_fp16_0, y = const_136_to_fp16)[name = string("op_5934_cast_fp16")]; tensor x_229_cast_fp16 = add(x = x_211_cast_fp16, y = var_5934_cast_fp16)[name = string("x_229_cast_fp16")]; int32 var_5941 = const()[name = string("op_5941"), val = int32(-1)]; fp16 const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5947_cast_fp16 = mul(x = x_229_cast_fp16, y = const_137_promoted_to_fp16)[name = string("op_5947_cast_fp16")]; bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; tensor input_223_cast_fp16 = concat(axis = var_5941, interleave = input_223_interleave_0, values = (x_229_cast_fp16, var_5947_cast_fp16))[name = string("input_223_cast_fp16")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_5939_to_fp16 = const()[name = string("op_5939_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_5939_to_fp16, x = input_223_cast_fp16)[name = string("normed_213_cast_fp16")]; tensor var_5952_split_sizes_0 = const()[name = string("op_5952_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_5952_axis_0 = const()[name = string("op_5952_axis_0"), val = int32(-1)]; tensor var_5952_cast_fp16_0, tensor var_5952_cast_fp16_1 = split(axis = var_5952_axis_0, split_sizes = var_5952_split_sizes_0, x = normed_213_cast_fp16)[name = string("op_5952_cast_fp16")]; tensor const_138_to_fp16 = const()[name = string("const_138_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077512384)))]; tensor var_5955_cast_fp16 = mul(x = var_5952_cast_fp16_0, y = const_138_to_fp16)[name = string("op_5955_cast_fp16")]; tensor var_5968 = const()[name = string("op_5968"), val = tensor([0, 2, 1])]; tensor input_225_axes_0 = const()[name = string("input_225_axes_0"), val = tensor([2])]; tensor var_5969 = transpose(perm = var_5968, x = var_5955_cast_fp16)[name = string("transpose_248")]; tensor input_225 = expand_dims(axes = input_225_axes_0, x = var_5969)[name = string("input_225")]; string gate_29_pad_type_0 = const()[name = string("gate_29_pad_type_0"), val = string("valid")]; tensor gate_29_strides_0 = const()[name = string("gate_29_strides_0"), val = tensor([1, 1])]; tensor gate_29_pad_0 = const()[name = string("gate_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_29_dilations_0 = const()[name = string("gate_29_dilations_0"), val = tensor([1, 1])]; int32 gate_29_groups_0 = const()[name = string("gate_29_groups_0"), val = int32(1)]; tensor gate_29 = conv(dilations = gate_29_dilations_0, groups = gate_29_groups_0, pad = gate_29_pad_0, pad_type = gate_29_pad_type_0, strides = gate_29_strides_0, weight = layers_7_mlp_gate_proj_weight_palettized, x = input_225)[name = string("gate_29")]; string up_15_pad_type_0 = const()[name = string("up_15_pad_type_0"), val = string("valid")]; tensor up_15_strides_0 = const()[name = string("up_15_strides_0"), val = tensor([1, 1])]; tensor up_15_pad_0 = const()[name = string("up_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_15_dilations_0 = const()[name = string("up_15_dilations_0"), val = tensor([1, 1])]; int32 up_15_groups_0 = const()[name = string("up_15_groups_0"), val = int32(1)]; tensor up_15 = conv(dilations = up_15_dilations_0, groups = up_15_groups_0, pad = up_15_pad_0, pad_type = up_15_pad_type_0, strides = up_15_strides_0, weight = layers_7_mlp_up_proj_weight_palettized, x = input_225)[name = string("up_15")]; string gate_31_mode_0 = const()[name = string("gate_31_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_31 = gelu(mode = gate_31_mode_0, x = gate_29)[name = string("gate_31")]; tensor input_227 = mul(x = gate_31, y = up_15)[name = string("input_227")]; string mlp_out_15_pad_type_0 = const()[name = string("mlp_out_15_pad_type_0"), val = string("valid")]; tensor mlp_out_15_strides_0 = const()[name = string("mlp_out_15_strides_0"), val = tensor([1, 1])]; tensor mlp_out_15_pad_0 = const()[name = string("mlp_out_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_15_dilations_0 = const()[name = string("mlp_out_15_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_15_groups_0 = const()[name = string("mlp_out_15_groups_0"), val = int32(1)]; tensor mlp_out_15 = conv(dilations = mlp_out_15_dilations_0, groups = mlp_out_15_groups_0, pad = mlp_out_15_pad_0, pad_type = mlp_out_15_pad_type_0, strides = mlp_out_15_strides_0, weight = layers_7_mlp_down_proj_weight_palettized, x = input_227)[name = string("mlp_out_15")]; tensor var_6009_axes_0 = const()[name = string("op_6009_axes_0"), val = tensor([2])]; tensor var_6009 = squeeze(axes = var_6009_axes_0, x = mlp_out_15)[name = string("op_6009")]; tensor var_6013 = const()[name = string("op_6013"), val = tensor([0, 2, 1])]; int32 var_6019 = const()[name = string("op_6019"), val = int32(-1)]; fp16 const_139_promoted_to_fp16 = const()[name = string("const_139_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_233 = transpose(perm = var_6013, x = var_6009)[name = string("transpose_247")]; tensor var_6025_cast_fp16 = mul(x = x_233, y = const_139_promoted_to_fp16)[name = string("op_6025_cast_fp16")]; bool input_229_interleave_0 = const()[name = string("input_229_interleave_0"), val = bool(false)]; tensor input_229_cast_fp16 = concat(axis = var_6019, interleave = input_229_interleave_0, values = (x_233, var_6025_cast_fp16))[name = string("input_229_cast_fp16")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_6017_to_fp16 = const()[name = string("op_6017_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_6017_to_fp16, x = input_229_cast_fp16)[name = string("normed_217_cast_fp16")]; tensor var_6030_split_sizes_0 = const()[name = string("op_6030_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6030_axis_0 = const()[name = string("op_6030_axis_0"), val = int32(-1)]; tensor var_6030_cast_fp16_0, tensor var_6030_cast_fp16_1 = split(axis = var_6030_axis_0, split_sizes = var_6030_split_sizes_0, x = normed_217_cast_fp16)[name = string("op_6030_cast_fp16")]; tensor const_140_to_fp16 = const()[name = string("const_140_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077515520)))]; tensor var_6033_cast_fp16 = mul(x = var_6030_cast_fp16_0, y = const_140_to_fp16)[name = string("op_6033_cast_fp16")]; tensor hidden_states_93_cast_fp16 = add(x = x_229_cast_fp16, y = var_6033_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor per_layer_slice_15_begin_0 = const()[name = string("per_layer_slice_15_begin_0"), val = tensor([0, 0, 1792])]; tensor per_layer_slice_15_end_0 = const()[name = string("per_layer_slice_15_end_0"), val = tensor([1, 1, 2048])]; tensor per_layer_slice_15_end_mask_0 = const()[name = string("per_layer_slice_15_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_15_cast_fp16 = slice_by_index(begin = per_layer_slice_15_begin_0, end = per_layer_slice_15_end_0, end_mask = per_layer_slice_15_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_15_cast_fp16")]; tensor gated_29 = linear(bias = linear_0_bias_0, weight = layers_7_per_layer_input_gate_weight_palettized, x = hidden_states_93_cast_fp16)[name = string("linear_14")]; string gated_31_mode_0 = const()[name = string("gated_31_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_31 = gelu(mode = gated_31_mode_0, x = gated_29)[name = string("gated_31")]; tensor input_233_cast_fp16 = mul(x = gated_31, y = per_layer_slice_15_cast_fp16)[name = string("input_233_cast_fp16")]; tensor layers_7_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077518656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077715328))))[name = string("layers_7_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_15_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_7_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_233_cast_fp16)[name = string("linear_15_cast_fp16")]; int32 var_6070 = const()[name = string("op_6070"), val = int32(-1)]; fp16 const_141_promoted_to_fp16 = const()[name = string("const_141_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6076_cast_fp16 = mul(x = linear_15_cast_fp16, y = const_141_promoted_to_fp16)[name = string("op_6076_cast_fp16")]; bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; tensor input_235_cast_fp16 = concat(axis = var_6070, interleave = input_235_interleave_0, values = (linear_15_cast_fp16, var_6076_cast_fp16))[name = string("input_235_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_6068_to_fp16 = const()[name = string("op_6068_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_6068_to_fp16, x = input_235_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor var_6081_split_sizes_0 = const()[name = string("op_6081_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6081_axis_0 = const()[name = string("op_6081_axis_0"), val = int32(-1)]; tensor var_6081_cast_fp16_0, tensor var_6081_cast_fp16_1 = split(axis = var_6081_axis_0, split_sizes = var_6081_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_6081_cast_fp16")]; tensor const_142_to_fp16 = const()[name = string("const_142_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077716928)))]; tensor var_6084_cast_fp16 = mul(x = var_6081_cast_fp16_0, y = const_142_to_fp16)[name = string("op_6084_cast_fp16")]; tensor hidden_states_97_cast_fp16 = add(x = hidden_states_93_cast_fp16, y = var_6084_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor layers_7_layer_scalar_to_fp16 = const()[name = string("layers_7_layer_scalar_to_fp16"), val = tensor([0x1.38p-1])]; tensor x_241_cast_fp16 = mul(x = hidden_states_97_cast_fp16, y = layers_7_layer_scalar_to_fp16)[name = string("x_241_cast_fp16")]; int32 var_6092 = const()[name = string("op_6092"), val = int32(-1)]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6098_cast_fp16 = mul(x = x_241_cast_fp16, y = const_143_promoted_to_fp16)[name = string("op_6098_cast_fp16")]; bool input_237_interleave_0 = const()[name = string("input_237_interleave_0"), val = bool(false)]; tensor input_237_cast_fp16 = concat(axis = var_6092, interleave = input_237_interleave_0, values = (x_241_cast_fp16, var_6098_cast_fp16))[name = string("input_237_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_6090_to_fp16 = const()[name = string("op_6090_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_6090_to_fp16, x = input_237_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_6103_split_sizes_0 = const()[name = string("op_6103_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6103_axis_0 = const()[name = string("op_6103_axis_0"), val = int32(-1)]; tensor var_6103_cast_fp16_0, tensor var_6103_cast_fp16_1 = split(axis = var_6103_axis_0, split_sizes = var_6103_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_6103_cast_fp16")]; tensor const_144_to_fp16 = const()[name = string("const_144_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077720064)))]; tensor var_6106_cast_fp16 = mul(x = var_6103_cast_fp16_0, y = const_144_to_fp16)[name = string("op_6106_cast_fp16")]; tensor var_6114 = const()[name = string("op_6114"), val = tensor([0, 2, 1])]; tensor var_6117_axes_0 = const()[name = string("op_6117_axes_0"), val = tensor([2])]; tensor var_6115_cast_fp16 = transpose(perm = var_6114, x = var_6106_cast_fp16)[name = string("transpose_246")]; tensor var_6117_cast_fp16 = expand_dims(axes = var_6117_axes_0, x = var_6115_cast_fp16)[name = string("op_6117_cast_fp16")]; string var_6133_pad_type_0 = const()[name = string("op_6133_pad_type_0"), val = string("valid")]; tensor var_6133_strides_0 = const()[name = string("op_6133_strides_0"), val = tensor([1, 1])]; tensor var_6133_pad_0 = const()[name = string("op_6133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6133_dilations_0 = const()[name = string("op_6133_dilations_0"), val = tensor([1, 1])]; int32 var_6133_groups_0 = const()[name = string("op_6133_groups_0"), val = int32(1)]; tensor var_6133 = conv(dilations = var_6133_dilations_0, groups = var_6133_groups_0, pad = var_6133_pad_0, pad_type = var_6133_pad_type_0, strides = var_6133_strides_0, weight = layers_8_self_attn_q_proj_weight_palettized, x = var_6117_cast_fp16)[name = string("op_6133")]; tensor var_6138 = const()[name = string("op_6138"), val = tensor([1, 8, 256, 1])]; tensor var_6139 = reshape(shape = var_6138, x = var_6133)[name = string("op_6139")]; tensor var_6144 = const()[name = string("op_6144"), val = tensor([0, 1, 3, 2])]; tensor var_6154 = const()[name = string("op_6154"), val = tensor([1, 8, 256])]; tensor var_6145 = transpose(perm = var_6144, x = var_6139)[name = string("transpose_245")]; tensor x_245 = reshape(shape = var_6154, x = var_6145)[name = string("x_245")]; int32 var_6160 = const()[name = string("op_6160"), val = int32(-1)]; fp16 const_145_promoted_to_fp16 = const()[name = string("const_145_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6166_cast_fp16 = mul(x = x_245, y = const_145_promoted_to_fp16)[name = string("op_6166_cast_fp16")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241_cast_fp16 = concat(axis = var_6160, interleave = input_241_interleave_0, values = (x_245, var_6166_cast_fp16))[name = string("input_241_cast_fp16")]; tensor normed_229_axes_0 = const()[name = string("normed_229_axes_0"), val = tensor([-1])]; fp16 var_6158_to_fp16 = const()[name = string("op_6158_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_229_cast_fp16 = layer_norm(axes = normed_229_axes_0, epsilon = var_6158_to_fp16, x = input_241_cast_fp16)[name = string("normed_229_cast_fp16")]; tensor var_6171_split_sizes_0 = const()[name = string("op_6171_split_sizes_0"), val = tensor([256, 256])]; int32 var_6171_axis_0 = const()[name = string("op_6171_axis_0"), val = int32(-1)]; tensor var_6171_cast_fp16_0, tensor var_6171_cast_fp16_1 = split(axis = var_6171_axis_0, split_sizes = var_6171_split_sizes_0, x = normed_229_cast_fp16)[name = string("op_6171_cast_fp16")]; tensor const_146_to_fp16 = const()[name = string("const_146_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077723200)))]; tensor var_6174_cast_fp16 = mul(x = var_6171_cast_fp16_0, y = const_146_to_fp16)[name = string("op_6174_cast_fp16")]; tensor var_6180 = const()[name = string("op_6180"), val = tensor([1, 8, 1, 256])]; tensor q_67 = reshape(shape = var_6180, x = var_6174_cast_fp16)[name = string("q_67")]; tensor var_6182 = mul(x = q_67, y = cos_1)[name = string("op_6182")]; tensor var_6183_split_sizes_0 = const()[name = string("op_6183_split_sizes_0"), val = tensor([128, 128])]; int32 var_6183_axis_0 = const()[name = string("op_6183_axis_0"), val = int32(-1)]; tensor var_6183_0, tensor var_6183_1 = split(axis = var_6183_axis_0, split_sizes = var_6183_split_sizes_0, x = q_67)[name = string("op_6183")]; fp16 const_147_promoted = const()[name = string("const_147_promoted"), val = fp16(-0x1p+0)]; tensor var_6185 = mul(x = var_6183_1, y = const_147_promoted)[name = string("op_6185")]; int32 var_6187 = const()[name = string("op_6187"), val = int32(-1)]; bool var_6188_interleave_0 = const()[name = string("op_6188_interleave_0"), val = bool(false)]; tensor var_6188 = concat(axis = var_6187, interleave = var_6188_interleave_0, values = (var_6185, var_6183_0))[name = string("op_6188")]; tensor var_6189 = mul(x = var_6188, y = sin_1)[name = string("op_6189")]; tensor q_71 = add(x = var_6182, y = var_6189)[name = string("q_71")]; string var_6202_pad_type_0 = const()[name = string("op_6202_pad_type_0"), val = string("valid")]; tensor var_6202_strides_0 = const()[name = string("op_6202_strides_0"), val = tensor([1, 1])]; tensor var_6202_pad_0 = const()[name = string("op_6202_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6202_dilations_0 = const()[name = string("op_6202_dilations_0"), val = tensor([1, 1])]; int32 var_6202_groups_0 = const()[name = string("op_6202_groups_0"), val = int32(1)]; tensor var_6202 = conv(dilations = var_6202_dilations_0, groups = var_6202_groups_0, pad = var_6202_pad_0, pad_type = var_6202_pad_type_0, strides = var_6202_strides_0, weight = layers_8_self_attn_k_proj_weight_palettized, x = var_6117_cast_fp16)[name = string("op_6202")]; tensor var_6207 = const()[name = string("op_6207"), val = tensor([1, 1, 256, 1])]; tensor var_6208 = reshape(shape = var_6207, x = var_6202)[name = string("op_6208")]; tensor var_6213 = const()[name = string("op_6213"), val = tensor([0, 1, 3, 2])]; string var_6230_pad_type_0 = const()[name = string("op_6230_pad_type_0"), val = string("valid")]; tensor var_6230_strides_0 = const()[name = string("op_6230_strides_0"), val = tensor([1, 1])]; tensor var_6230_pad_0 = const()[name = string("op_6230_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6230_dilations_0 = const()[name = string("op_6230_dilations_0"), val = tensor([1, 1])]; int32 var_6230_groups_0 = const()[name = string("op_6230_groups_0"), val = int32(1)]; tensor var_6230 = conv(dilations = var_6230_dilations_0, groups = var_6230_groups_0, pad = var_6230_pad_0, pad_type = var_6230_pad_type_0, strides = var_6230_strides_0, weight = layers_8_self_attn_v_proj_weight_palettized, x = var_6117_cast_fp16)[name = string("op_6230")]; tensor var_6235 = const()[name = string("op_6235"), val = tensor([1, 1, 256, 1])]; tensor var_6236 = reshape(shape = var_6235, x = var_6230)[name = string("op_6236")]; tensor var_6241 = const()[name = string("op_6241"), val = tensor([0, 1, 3, 2])]; tensor var_6251 = const()[name = string("op_6251"), val = tensor([1, 1, 256])]; tensor var_6214 = transpose(perm = var_6213, x = var_6208)[name = string("transpose_244")]; tensor x_249 = reshape(shape = var_6251, x = var_6214)[name = string("x_249")]; int32 var_6257 = const()[name = string("op_6257"), val = int32(-1)]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6263_cast_fp16 = mul(x = x_249, y = const_148_promoted_to_fp16)[name = string("op_6263_cast_fp16")]; bool input_243_interleave_0 = const()[name = string("input_243_interleave_0"), val = bool(false)]; tensor input_243_cast_fp16 = concat(axis = var_6257, interleave = input_243_interleave_0, values = (x_249, var_6263_cast_fp16))[name = string("input_243_cast_fp16")]; tensor normed_233_axes_0 = const()[name = string("normed_233_axes_0"), val = tensor([-1])]; fp16 var_6255_to_fp16 = const()[name = string("op_6255_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_233_cast_fp16 = layer_norm(axes = normed_233_axes_0, epsilon = var_6255_to_fp16, x = input_243_cast_fp16)[name = string("normed_233_cast_fp16")]; tensor var_6268_split_sizes_0 = const()[name = string("op_6268_split_sizes_0"), val = tensor([256, 256])]; int32 var_6268_axis_0 = const()[name = string("op_6268_axis_0"), val = int32(-1)]; tensor var_6268_cast_fp16_0, tensor var_6268_cast_fp16_1 = split(axis = var_6268_axis_0, split_sizes = var_6268_split_sizes_0, x = normed_233_cast_fp16)[name = string("op_6268_cast_fp16")]; tensor const_149_to_fp16 = const()[name = string("const_149_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077723776)))]; tensor var_6271_cast_fp16 = mul(x = var_6268_cast_fp16_0, y = const_149_to_fp16)[name = string("op_6271_cast_fp16")]; tensor var_6277 = const()[name = string("op_6277"), val = tensor([1, 1, 1, 256])]; tensor q_69 = reshape(shape = var_6277, x = var_6271_cast_fp16)[name = string("q_69")]; fp16 var_6284_promoted_to_fp16 = const()[name = string("op_6284_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_6242 = transpose(perm = var_6241, x = var_6236)[name = string("transpose_243")]; tensor var_6285_cast_fp16 = pow(x = var_6242, y = var_6284_promoted_to_fp16)[name = string("op_6285_cast_fp16")]; tensor var_6290_axes_0 = const()[name = string("op_6290_axes_0"), val = tensor([-1])]; bool var_6290_keep_dims_0 = const()[name = string("op_6290_keep_dims_0"), val = bool(true)]; tensor var_6290_cast_fp16 = reduce_mean(axes = var_6290_axes_0, keep_dims = var_6290_keep_dims_0, x = var_6285_cast_fp16)[name = string("op_6290_cast_fp16")]; fp16 var_6292_to_fp16 = const()[name = string("op_6292_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_17_cast_fp16 = add(x = var_6290_cast_fp16, y = var_6292_to_fp16)[name = string("mean_sq_17_cast_fp16")]; fp16 var_6299_to_fp16 = const()[name = string("op_6299_to_fp16"), val = fp16(-0x1p-1)]; tensor var_6300_cast_fp16 = pow(x = mean_sq_17_cast_fp16, y = var_6299_to_fp16)[name = string("op_6300_cast_fp16")]; tensor var_6301_cast_fp16 = mul(x = var_6242, y = var_6300_cast_fp16)[name = string("op_6301_cast_fp16")]; tensor var_6307 = mul(x = q_69, y = cos_1)[name = string("op_6307")]; tensor var_6308_split_sizes_0 = const()[name = string("op_6308_split_sizes_0"), val = tensor([128, 128])]; int32 var_6308_axis_0 = const()[name = string("op_6308_axis_0"), val = int32(-1)]; tensor var_6308_0, tensor var_6308_1 = split(axis = var_6308_axis_0, split_sizes = var_6308_split_sizes_0, x = q_69)[name = string("op_6308")]; fp16 const_150_promoted = const()[name = string("const_150_promoted"), val = fp16(-0x1p+0)]; tensor var_6310 = mul(x = var_6308_1, y = const_150_promoted)[name = string("op_6310")]; int32 var_6312 = const()[name = string("op_6312"), val = int32(-1)]; bool var_6313_interleave_0 = const()[name = string("op_6313_interleave_0"), val = bool(false)]; tensor var_6313 = concat(axis = var_6312, interleave = var_6313_interleave_0, values = (var_6310, var_6308_0))[name = string("op_6313")]; tensor var_6314 = mul(x = var_6313, y = sin_1)[name = string("op_6314")]; tensor input_245 = add(x = var_6307, y = var_6314)[name = string("input_245")]; tensor var_6319_begin_0 = const()[name = string("op_6319_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_6319_end_0 = const()[name = string("op_6319_end_0"), val = tensor([9, 1, 512, 512])]; tensor var_6319_end_mask_0 = const()[name = string("op_6319_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6319_squeeze_mask_0 = const()[name = string("op_6319_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6319_cast_fp16 = slice_by_index(begin = var_6319_begin_0, end = var_6319_end_0, end_mask = var_6319_end_mask_0, squeeze_mask = var_6319_squeeze_mask_0, x = coreml_update_state_45)[name = string("op_6319_cast_fp16")]; tensor K_cache_17_axes_0 = const()[name = string("K_cache_17_axes_0"), val = tensor([0])]; tensor K_cache_17_cast_fp16 = expand_dims(axes = K_cache_17_axes_0, x = var_6319_cast_fp16)[name = string("K_cache_17_cast_fp16")]; tensor var_6324_begin_0 = const()[name = string("op_6324_begin_0"), val = tensor([43, 0, 0, 0])]; tensor var_6324_end_0 = const()[name = string("op_6324_end_0"), val = tensor([44, 1, 512, 512])]; tensor var_6324_end_mask_0 = const()[name = string("op_6324_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6324_squeeze_mask_0 = const()[name = string("op_6324_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6324_cast_fp16 = slice_by_index(begin = var_6324_begin_0, end = var_6324_end_0, end_mask = var_6324_end_mask_0, squeeze_mask = var_6324_squeeze_mask_0, x = coreml_update_state_45)[name = string("op_6324_cast_fp16")]; tensor V_cache_17_axes_0 = const()[name = string("V_cache_17_axes_0"), val = tensor([0])]; tensor V_cache_17_cast_fp16 = expand_dims(axes = V_cache_17_axes_0, x = var_6324_cast_fp16)[name = string("V_cache_17_cast_fp16")]; tensor k_padded_15_pad_0 = const()[name = string("k_padded_15_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_15_mode_0 = const()[name = string("k_padded_15_mode_0"), val = string("constant")]; fp16 const_151_to_fp16 = const()[name = string("const_151_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_15_cast_fp16 = pad(constant_val = const_151_to_fp16, mode = k_padded_15_mode_0, pad = k_padded_15_pad_0, x = input_245)[name = string("k_padded_15_cast_fp16")]; tensor v_padded_15_pad_0 = const()[name = string("v_padded_15_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_15_mode_0 = const()[name = string("v_padded_15_mode_0"), val = string("constant")]; fp16 const_152_to_fp16 = const()[name = string("const_152_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_15_cast_fp16 = pad(constant_val = const_152_to_fp16, mode = v_padded_15_mode_0, pad = v_padded_15_pad_0, x = var_6301_cast_fp16)[name = string("v_padded_15_cast_fp16")]; tensor var_6342_cast_fp16 = mul(x = K_cache_17_cast_fp16, y = var_2187_cast_fp16)[name = string("op_6342_cast_fp16")]; tensor var_6343_reps_0 = const()[name = string("op_6343_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_6343_cast_fp16 = tile(reps = var_6343_reps_0, x = k_padded_15_cast_fp16)[name = string("op_6343_cast_fp16")]; tensor var_6344_cast_fp16 = mul(x = var_6343_cast_fp16, y = update_mask)[name = string("op_6344_cast_fp16")]; tensor K_new_17_cast_fp16 = add(x = var_6342_cast_fp16, y = var_6344_cast_fp16)[name = string("K_new_17_cast_fp16")]; tensor var_6350_cast_fp16 = mul(x = V_cache_17_cast_fp16, y = var_2187_cast_fp16)[name = string("op_6350_cast_fp16")]; tensor var_6351_reps_0 = const()[name = string("op_6351_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_6351_cast_fp16 = tile(reps = var_6351_reps_0, x = v_padded_15_cast_fp16)[name = string("op_6351_cast_fp16")]; tensor var_6352_cast_fp16 = mul(x = var_6351_cast_fp16, y = update_mask)[name = string("op_6352_cast_fp16")]; tensor V_new_17_cast_fp16 = add(x = var_6350_cast_fp16, y = var_6352_cast_fp16)[name = string("V_new_17_cast_fp16")]; tensor var_6356_axes_0 = const()[name = string("op_6356_axes_0"), val = tensor([0])]; tensor var_6356_cast_fp16 = squeeze(axes = var_6356_axes_0, x = K_new_17_cast_fp16)[name = string("op_6356_cast_fp16")]; tensor concat_64 = const()[name = string("concat_64"), val = tensor([8, 0, 0, 0])]; tensor concat_65 = const()[name = string("concat_65"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_64, begin_mask = kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_65, end_mask = kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_17_stride_0, update = var_6356_cast_fp16, x = coreml_update_state_45)[name = string("kv_cache_0_internal_tensor_assign_17_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_17_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_46_write_state")]; tensor coreml_update_state_46 = read_state(input = kv_cache_0)[name = string("coreml_update_state_46")]; tensor var_6363_axes_0 = const()[name = string("op_6363_axes_0"), val = tensor([0])]; tensor var_6363_cast_fp16 = squeeze(axes = var_6363_axes_0, x = V_new_17_cast_fp16)[name = string("op_6363_cast_fp16")]; tensor concat_66 = const()[name = string("concat_66"), val = tensor([43, 0, 0, 0])]; tensor concat_67 = const()[name = string("concat_67"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_66, begin_mask = kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_67, end_mask = kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_18_stride_0, update = var_6363_cast_fp16, x = coreml_update_state_46)[name = string("kv_cache_0_internal_tensor_assign_18_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_18_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_47_write_state")]; tensor coreml_update_state_47 = read_state(input = kv_cache_0)[name = string("coreml_update_state_47")]; tensor K_for_attn_17_begin_0 = const()[name = string("K_for_attn_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_17_end_0 = const()[name = string("K_for_attn_17_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_17_end_mask_0 = const()[name = string("K_for_attn_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_17_cast_fp16 = slice_by_index(begin = K_for_attn_17_begin_0, end = K_for_attn_17_end_0, end_mask = K_for_attn_17_end_mask_0, x = K_new_17_cast_fp16)[name = string("K_for_attn_17_cast_fp16")]; tensor V_for_attn_17_begin_0 = const()[name = string("V_for_attn_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_17_end_0 = const()[name = string("V_for_attn_17_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_17_end_mask_0 = const()[name = string("V_for_attn_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_17_cast_fp16 = slice_by_index(begin = V_for_attn_17_begin_0, end = V_for_attn_17_end_0, end_mask = V_for_attn_17_end_mask_0, x = V_new_17_cast_fp16)[name = string("V_for_attn_17_cast_fp16")]; tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_16_reps_0 = const()[name = string("tile_16_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = K_for_attn_17_cast_fp16)[name = string("transpose_242")]; tensor tile_16_cast_fp16 = tile(reps = tile_16_reps_0, x = transpose_32_cast_fp16)[name = string("tile_16_cast_fp16")]; tensor concat_68 = const()[name = string("concat_68"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_32_cast_fp16 = reshape(shape = concat_68, x = tile_16_cast_fp16)[name = string("reshape_32_cast_fp16")]; tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_69 = const()[name = string("concat_69"), val = tensor([-1, 1, 512, 256])]; tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_241")]; tensor reshape_33_cast_fp16 = reshape(shape = concat_69, x = transpose_33_cast_fp16)[name = string("reshape_33_cast_fp16")]; tensor transpose_148_perm_0 = const()[name = string("transpose_148_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_17_reps_0 = const()[name = string("tile_17_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = V_for_attn_17_cast_fp16)[name = string("transpose_240")]; tensor tile_17_cast_fp16 = tile(reps = tile_17_reps_0, x = transpose_34_cast_fp16)[name = string("tile_17_cast_fp16")]; tensor concat_70 = const()[name = string("concat_70"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_34_cast_fp16 = reshape(shape = concat_70, x = tile_17_cast_fp16)[name = string("reshape_34_cast_fp16")]; tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_71 = const()[name = string("concat_71"), val = tensor([-1, 1, 512, 256])]; tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_34_cast_fp16)[name = string("transpose_239")]; tensor reshape_35_cast_fp16 = reshape(shape = concat_71, x = transpose_35_cast_fp16)[name = string("reshape_35_cast_fp16")]; tensor V_expanded_17_perm_0 = const()[name = string("V_expanded_17_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_6390_transpose_x_0 = const()[name = string("op_6390_transpose_x_0"), val = bool(false)]; bool var_6390_transpose_y_0 = const()[name = string("op_6390_transpose_y_0"), val = bool(false)]; tensor transpose_148_cast_fp16 = transpose(perm = transpose_148_perm_0, x = reshape_33_cast_fp16)[name = string("transpose_238")]; tensor var_6390_cast_fp16 = matmul(transpose_x = var_6390_transpose_x_0, transpose_y = var_6390_transpose_y_0, x = q_71, y = transpose_148_cast_fp16)[name = string("op_6390_cast_fp16")]; tensor attn_weights_51_cast_fp16 = add(x = var_6390_cast_fp16, y = causal_mask)[name = string("attn_weights_51_cast_fp16")]; int32 var_6395 = const()[name = string("op_6395"), val = int32(-1)]; tensor attn_weights_53_cast_fp16 = softmax(axis = var_6395, x = attn_weights_51_cast_fp16)[name = string("attn_weights_53_cast_fp16")]; bool attn_output_49_transpose_x_0 = const()[name = string("attn_output_49_transpose_x_0"), val = bool(false)]; bool attn_output_49_transpose_y_0 = const()[name = string("attn_output_49_transpose_y_0"), val = bool(false)]; tensor V_expanded_17_cast_fp16 = transpose(perm = V_expanded_17_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_237")]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = attn_weights_53_cast_fp16, y = V_expanded_17_cast_fp16)[name = string("attn_output_49_cast_fp16")]; tensor var_6403 = const()[name = string("op_6403"), val = tensor([0, 2, 1, 3])]; tensor var_6410 = const()[name = string("op_6410"), val = tensor([1, 1, -1])]; tensor var_6404_cast_fp16 = transpose(perm = var_6403, x = attn_output_49_cast_fp16)[name = string("transpose_236")]; tensor attn_output_51_cast_fp16 = reshape(shape = var_6410, x = var_6404_cast_fp16)[name = string("attn_output_51_cast_fp16")]; tensor var_6415 = const()[name = string("op_6415"), val = tensor([0, 2, 1])]; string var_6431_pad_type_0 = const()[name = string("op_6431_pad_type_0"), val = string("valid")]; int32 var_6431_groups_0 = const()[name = string("op_6431_groups_0"), val = int32(1)]; tensor var_6431_strides_0 = const()[name = string("op_6431_strides_0"), val = tensor([1])]; tensor var_6431_pad_0 = const()[name = string("op_6431_pad_0"), val = tensor([0, 0])]; tensor var_6431_dilations_0 = const()[name = string("op_6431_dilations_0"), val = tensor([1])]; tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077724352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079297280))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6416_cast_fp16 = transpose(perm = var_6415, x = attn_output_51_cast_fp16)[name = string("transpose_235")]; tensor var_6431_cast_fp16 = conv(dilations = var_6431_dilations_0, groups = var_6431_groups_0, pad = var_6431_pad_0, pad_type = var_6431_pad_type_0, strides = var_6431_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_6416_cast_fp16)[name = string("op_6431_cast_fp16")]; tensor var_6435 = const()[name = string("op_6435"), val = tensor([0, 2, 1])]; int32 var_6441 = const()[name = string("op_6441"), val = int32(-1)]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_255_cast_fp16 = transpose(perm = var_6435, x = var_6431_cast_fp16)[name = string("transpose_234")]; tensor var_6447_cast_fp16 = mul(x = x_255_cast_fp16, y = const_153_promoted_to_fp16)[name = string("op_6447_cast_fp16")]; bool input_251_interleave_0 = const()[name = string("input_251_interleave_0"), val = bool(false)]; tensor input_251_cast_fp16 = concat(axis = var_6441, interleave = input_251_interleave_0, values = (x_255_cast_fp16, var_6447_cast_fp16))[name = string("input_251_cast_fp16")]; tensor normed_237_axes_0 = const()[name = string("normed_237_axes_0"), val = tensor([-1])]; fp16 var_6439_to_fp16 = const()[name = string("op_6439_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_237_cast_fp16 = layer_norm(axes = normed_237_axes_0, epsilon = var_6439_to_fp16, x = input_251_cast_fp16)[name = string("normed_237_cast_fp16")]; tensor var_6452_split_sizes_0 = const()[name = string("op_6452_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6452_axis_0 = const()[name = string("op_6452_axis_0"), val = int32(-1)]; tensor var_6452_cast_fp16_0, tensor var_6452_cast_fp16_1 = split(axis = var_6452_axis_0, split_sizes = var_6452_split_sizes_0, x = normed_237_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor const_154_to_fp16 = const()[name = string("const_154_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079298880)))]; tensor var_6455_cast_fp16 = mul(x = var_6452_cast_fp16_0, y = const_154_to_fp16)[name = string("op_6455_cast_fp16")]; tensor x_259_cast_fp16 = add(x = x_241_cast_fp16, y = var_6455_cast_fp16)[name = string("x_259_cast_fp16")]; int32 var_6462 = const()[name = string("op_6462"), val = int32(-1)]; fp16 const_155_promoted_to_fp16 = const()[name = string("const_155_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = x_259_cast_fp16, y = const_155_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; bool input_253_interleave_0 = const()[name = string("input_253_interleave_0"), val = bool(false)]; tensor input_253_cast_fp16 = concat(axis = var_6462, interleave = input_253_interleave_0, values = (x_259_cast_fp16, var_6468_cast_fp16))[name = string("input_253_cast_fp16")]; tensor normed_241_axes_0 = const()[name = string("normed_241_axes_0"), val = tensor([-1])]; fp16 var_6460_to_fp16 = const()[name = string("op_6460_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_241_cast_fp16 = layer_norm(axes = normed_241_axes_0, epsilon = var_6460_to_fp16, x = input_253_cast_fp16)[name = string("normed_241_cast_fp16")]; tensor var_6473_split_sizes_0 = const()[name = string("op_6473_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6473_axis_0 = const()[name = string("op_6473_axis_0"), val = int32(-1)]; tensor var_6473_cast_fp16_0, tensor var_6473_cast_fp16_1 = split(axis = var_6473_axis_0, split_sizes = var_6473_split_sizes_0, x = normed_241_cast_fp16)[name = string("op_6473_cast_fp16")]; tensor const_156_to_fp16 = const()[name = string("const_156_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079302016)))]; tensor var_6476_cast_fp16 = mul(x = var_6473_cast_fp16_0, y = const_156_to_fp16)[name = string("op_6476_cast_fp16")]; tensor var_6489 = const()[name = string("op_6489"), val = tensor([0, 2, 1])]; tensor input_255_axes_0 = const()[name = string("input_255_axes_0"), val = tensor([2])]; tensor var_6490 = transpose(perm = var_6489, x = var_6476_cast_fp16)[name = string("transpose_233")]; tensor input_255 = expand_dims(axes = input_255_axes_0, x = var_6490)[name = string("input_255")]; string gate_33_pad_type_0 = const()[name = string("gate_33_pad_type_0"), val = string("valid")]; tensor gate_33_strides_0 = const()[name = string("gate_33_strides_0"), val = tensor([1, 1])]; tensor gate_33_pad_0 = const()[name = string("gate_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_33_dilations_0 = const()[name = string("gate_33_dilations_0"), val = tensor([1, 1])]; int32 gate_33_groups_0 = const()[name = string("gate_33_groups_0"), val = int32(1)]; tensor gate_33 = conv(dilations = gate_33_dilations_0, groups = gate_33_groups_0, pad = gate_33_pad_0, pad_type = gate_33_pad_type_0, strides = gate_33_strides_0, weight = layers_8_mlp_gate_proj_weight_palettized, x = input_255)[name = string("gate_33")]; string up_17_pad_type_0 = const()[name = string("up_17_pad_type_0"), val = string("valid")]; tensor up_17_strides_0 = const()[name = string("up_17_strides_0"), val = tensor([1, 1])]; tensor up_17_pad_0 = const()[name = string("up_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_17_dilations_0 = const()[name = string("up_17_dilations_0"), val = tensor([1, 1])]; int32 up_17_groups_0 = const()[name = string("up_17_groups_0"), val = int32(1)]; tensor up_17 = conv(dilations = up_17_dilations_0, groups = up_17_groups_0, pad = up_17_pad_0, pad_type = up_17_pad_type_0, strides = up_17_strides_0, weight = layers_8_mlp_up_proj_weight_palettized, x = input_255)[name = string("up_17")]; string gate_35_mode_0 = const()[name = string("gate_35_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_35 = gelu(mode = gate_35_mode_0, x = gate_33)[name = string("gate_35")]; tensor input_257 = mul(x = gate_35, y = up_17)[name = string("input_257")]; string mlp_out_17_pad_type_0 = const()[name = string("mlp_out_17_pad_type_0"), val = string("valid")]; tensor mlp_out_17_strides_0 = const()[name = string("mlp_out_17_strides_0"), val = tensor([1, 1])]; tensor mlp_out_17_pad_0 = const()[name = string("mlp_out_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_17_dilations_0 = const()[name = string("mlp_out_17_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_17_groups_0 = const()[name = string("mlp_out_17_groups_0"), val = int32(1)]; tensor mlp_out_17 = conv(dilations = mlp_out_17_dilations_0, groups = mlp_out_17_groups_0, pad = mlp_out_17_pad_0, pad_type = mlp_out_17_pad_type_0, strides = mlp_out_17_strides_0, weight = layers_8_mlp_down_proj_weight_palettized, x = input_257)[name = string("mlp_out_17")]; tensor var_6530_axes_0 = const()[name = string("op_6530_axes_0"), val = tensor([2])]; tensor var_6530 = squeeze(axes = var_6530_axes_0, x = mlp_out_17)[name = string("op_6530")]; tensor var_6534 = const()[name = string("op_6534"), val = tensor([0, 2, 1])]; int32 var_6540 = const()[name = string("op_6540"), val = int32(-1)]; fp16 const_157_promoted_to_fp16 = const()[name = string("const_157_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_263 = transpose(perm = var_6534, x = var_6530)[name = string("transpose_232")]; tensor var_6546_cast_fp16 = mul(x = x_263, y = const_157_promoted_to_fp16)[name = string("op_6546_cast_fp16")]; bool input_259_interleave_0 = const()[name = string("input_259_interleave_0"), val = bool(false)]; tensor input_259_cast_fp16 = concat(axis = var_6540, interleave = input_259_interleave_0, values = (x_263, var_6546_cast_fp16))[name = string("input_259_cast_fp16")]; tensor normed_245_axes_0 = const()[name = string("normed_245_axes_0"), val = tensor([-1])]; fp16 var_6538_to_fp16 = const()[name = string("op_6538_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_245_cast_fp16 = layer_norm(axes = normed_245_axes_0, epsilon = var_6538_to_fp16, x = input_259_cast_fp16)[name = string("normed_245_cast_fp16")]; tensor var_6551_split_sizes_0 = const()[name = string("op_6551_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6551_axis_0 = const()[name = string("op_6551_axis_0"), val = int32(-1)]; tensor var_6551_cast_fp16_0, tensor var_6551_cast_fp16_1 = split(axis = var_6551_axis_0, split_sizes = var_6551_split_sizes_0, x = normed_245_cast_fp16)[name = string("op_6551_cast_fp16")]; tensor const_158_to_fp16 = const()[name = string("const_158_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079305152)))]; tensor var_6554_cast_fp16 = mul(x = var_6551_cast_fp16_0, y = const_158_to_fp16)[name = string("op_6554_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = x_259_cast_fp16, y = var_6554_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; tensor per_layer_slice_17_begin_0 = const()[name = string("per_layer_slice_17_begin_0"), val = tensor([0, 0, 2048])]; tensor per_layer_slice_17_end_0 = const()[name = string("per_layer_slice_17_end_0"), val = tensor([1, 1, 2304])]; tensor per_layer_slice_17_end_mask_0 = const()[name = string("per_layer_slice_17_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_17_cast_fp16 = slice_by_index(begin = per_layer_slice_17_begin_0, end = per_layer_slice_17_end_0, end_mask = per_layer_slice_17_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_17_cast_fp16")]; tensor gated_33 = linear(bias = linear_0_bias_0, weight = layers_8_per_layer_input_gate_weight_palettized, x = hidden_states_105_cast_fp16)[name = string("linear_16")]; string gated_35_mode_0 = const()[name = string("gated_35_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_35 = gelu(mode = gated_35_mode_0, x = gated_33)[name = string("gated_35")]; tensor input_263_cast_fp16 = mul(x = gated_35, y = per_layer_slice_17_cast_fp16)[name = string("input_263_cast_fp16")]; tensor layers_8_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079308288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079504960))))[name = string("layers_8_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_17_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_8_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_263_cast_fp16)[name = string("linear_17_cast_fp16")]; int32 var_6591 = const()[name = string("op_6591"), val = int32(-1)]; fp16 const_159_promoted_to_fp16 = const()[name = string("const_159_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6597_cast_fp16 = mul(x = linear_17_cast_fp16, y = const_159_promoted_to_fp16)[name = string("op_6597_cast_fp16")]; bool input_265_interleave_0 = const()[name = string("input_265_interleave_0"), val = bool(false)]; tensor input_265_cast_fp16 = concat(axis = var_6591, interleave = input_265_interleave_0, values = (linear_17_cast_fp16, var_6597_cast_fp16))[name = string("input_265_cast_fp16")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; fp16 var_6589_to_fp16 = const()[name = string("op_6589_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_6589_to_fp16, x = input_265_cast_fp16)[name = string("normed_249_cast_fp16")]; tensor var_6602_split_sizes_0 = const()[name = string("op_6602_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6602_axis_0 = const()[name = string("op_6602_axis_0"), val = int32(-1)]; tensor var_6602_cast_fp16_0, tensor var_6602_cast_fp16_1 = split(axis = var_6602_axis_0, split_sizes = var_6602_split_sizes_0, x = normed_249_cast_fp16)[name = string("op_6602_cast_fp16")]; tensor const_160_to_fp16 = const()[name = string("const_160_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079506560)))]; tensor var_6605_cast_fp16 = mul(x = var_6602_cast_fp16_0, y = const_160_to_fp16)[name = string("op_6605_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = var_6605_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; tensor layers_8_layer_scalar_to_fp16 = const()[name = string("layers_8_layer_scalar_to_fp16"), val = tensor([0x1.82p-2])]; tensor x_271_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = layers_8_layer_scalar_to_fp16)[name = string("x_271_cast_fp16")]; int32 var_6613 = const()[name = string("op_6613"), val = int32(-1)]; fp16 const_161_promoted_to_fp16 = const()[name = string("const_161_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6619_cast_fp16 = mul(x = x_271_cast_fp16, y = const_161_promoted_to_fp16)[name = string("op_6619_cast_fp16")]; bool input_267_interleave_0 = const()[name = string("input_267_interleave_0"), val = bool(false)]; tensor input_267_cast_fp16 = concat(axis = var_6613, interleave = input_267_interleave_0, values = (x_271_cast_fp16, var_6619_cast_fp16))[name = string("input_267_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; fp16 var_6611_to_fp16 = const()[name = string("op_6611_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_6611_to_fp16, x = input_267_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor var_6624_split_sizes_0 = const()[name = string("op_6624_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6624_axis_0 = const()[name = string("op_6624_axis_0"), val = int32(-1)]; tensor var_6624_cast_fp16_0, tensor var_6624_cast_fp16_1 = split(axis = var_6624_axis_0, split_sizes = var_6624_split_sizes_0, x = normed_253_cast_fp16)[name = string("op_6624_cast_fp16")]; tensor const_162_to_fp16 = const()[name = string("const_162_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079509696)))]; tensor var_6627_cast_fp16 = mul(x = var_6624_cast_fp16_0, y = const_162_to_fp16)[name = string("op_6627_cast_fp16")]; tensor var_6635 = const()[name = string("op_6635"), val = tensor([0, 2, 1])]; tensor var_6638_axes_0 = const()[name = string("op_6638_axes_0"), val = tensor([2])]; tensor var_6636_cast_fp16 = transpose(perm = var_6635, x = var_6627_cast_fp16)[name = string("transpose_231")]; tensor var_6638_cast_fp16 = expand_dims(axes = var_6638_axes_0, x = var_6636_cast_fp16)[name = string("op_6638_cast_fp16")]; string var_6654_pad_type_0 = const()[name = string("op_6654_pad_type_0"), val = string("valid")]; tensor var_6654_strides_0 = const()[name = string("op_6654_strides_0"), val = tensor([1, 1])]; tensor var_6654_pad_0 = const()[name = string("op_6654_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6654_dilations_0 = const()[name = string("op_6654_dilations_0"), val = tensor([1, 1])]; int32 var_6654_groups_0 = const()[name = string("op_6654_groups_0"), val = int32(1)]; tensor var_6654 = conv(dilations = var_6654_dilations_0, groups = var_6654_groups_0, pad = var_6654_pad_0, pad_type = var_6654_pad_type_0, strides = var_6654_strides_0, weight = layers_9_self_attn_q_proj_weight_palettized, x = var_6638_cast_fp16)[name = string("op_6654")]; tensor var_6659 = const()[name = string("op_6659"), val = tensor([1, 8, 512, 1])]; tensor var_6660 = reshape(shape = var_6659, x = var_6654)[name = string("op_6660")]; tensor var_6665 = const()[name = string("op_6665"), val = tensor([0, 1, 3, 2])]; tensor var_6675 = const()[name = string("op_6675"), val = tensor([1, 8, 512])]; tensor var_6666 = transpose(perm = var_6665, x = var_6660)[name = string("transpose_230")]; tensor x_275 = reshape(shape = var_6675, x = var_6666)[name = string("x_275")]; int32 var_6681 = const()[name = string("op_6681"), val = int32(-1)]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6687_cast_fp16 = mul(x = x_275, y = const_163_promoted_to_fp16)[name = string("op_6687_cast_fp16")]; bool input_271_interleave_0 = const()[name = string("input_271_interleave_0"), val = bool(false)]; tensor input_271_cast_fp16 = concat(axis = var_6681, interleave = input_271_interleave_0, values = (x_275, var_6687_cast_fp16))[name = string("input_271_cast_fp16")]; tensor normed_257_axes_0 = const()[name = string("normed_257_axes_0"), val = tensor([-1])]; fp16 var_6679_to_fp16 = const()[name = string("op_6679_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_257_cast_fp16 = layer_norm(axes = normed_257_axes_0, epsilon = var_6679_to_fp16, x = input_271_cast_fp16)[name = string("normed_257_cast_fp16")]; tensor var_6692_split_sizes_0 = const()[name = string("op_6692_split_sizes_0"), val = tensor([512, 512])]; int32 var_6692_axis_0 = const()[name = string("op_6692_axis_0"), val = int32(-1)]; tensor var_6692_cast_fp16_0, tensor var_6692_cast_fp16_1 = split(axis = var_6692_axis_0, split_sizes = var_6692_split_sizes_0, x = normed_257_cast_fp16)[name = string("op_6692_cast_fp16")]; tensor const_164_to_fp16 = const()[name = string("const_164_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079512832)))]; tensor var_6695_cast_fp16 = mul(x = var_6692_cast_fp16_0, y = const_164_to_fp16)[name = string("op_6695_cast_fp16")]; tensor var_6701 = const()[name = string("op_6701"), val = tensor([1, 8, 1, 512])]; tensor q_75 = reshape(shape = var_6701, x = var_6695_cast_fp16)[name = string("q_75")]; tensor var_6703 = mul(x = q_75, y = cos)[name = string("op_6703")]; tensor var_6704_split_sizes_0 = const()[name = string("op_6704_split_sizes_0"), val = tensor([256, 256])]; int32 var_6704_axis_0 = const()[name = string("op_6704_axis_0"), val = int32(-1)]; tensor var_6704_0, tensor var_6704_1 = split(axis = var_6704_axis_0, split_sizes = var_6704_split_sizes_0, x = q_75)[name = string("op_6704")]; fp16 const_165_promoted = const()[name = string("const_165_promoted"), val = fp16(-0x1p+0)]; tensor var_6706 = mul(x = var_6704_1, y = const_165_promoted)[name = string("op_6706")]; int32 var_6708 = const()[name = string("op_6708"), val = int32(-1)]; bool var_6709_interleave_0 = const()[name = string("op_6709_interleave_0"), val = bool(false)]; tensor var_6709 = concat(axis = var_6708, interleave = var_6709_interleave_0, values = (var_6706, var_6704_0))[name = string("op_6709")]; tensor var_6710 = mul(x = var_6709, y = sin)[name = string("op_6710")]; tensor q_79 = add(x = var_6703, y = var_6710)[name = string("q_79")]; string var_6723_pad_type_0 = const()[name = string("op_6723_pad_type_0"), val = string("valid")]; tensor var_6723_strides_0 = const()[name = string("op_6723_strides_0"), val = tensor([1, 1])]; tensor var_6723_pad_0 = const()[name = string("op_6723_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6723_dilations_0 = const()[name = string("op_6723_dilations_0"), val = tensor([1, 1])]; int32 var_6723_groups_0 = const()[name = string("op_6723_groups_0"), val = int32(1)]; tensor var_6723 = conv(dilations = var_6723_dilations_0, groups = var_6723_groups_0, pad = var_6723_pad_0, pad_type = var_6723_pad_type_0, strides = var_6723_strides_0, weight = layers_9_self_attn_k_proj_weight_palettized, x = var_6638_cast_fp16)[name = string("op_6723")]; tensor var_6728 = const()[name = string("op_6728"), val = tensor([1, 1, 512, 1])]; tensor var_6729 = reshape(shape = var_6728, x = var_6723)[name = string("op_6729")]; tensor var_6734 = const()[name = string("op_6734"), val = tensor([0, 1, 3, 2])]; string var_6751_pad_type_0 = const()[name = string("op_6751_pad_type_0"), val = string("valid")]; tensor var_6751_strides_0 = const()[name = string("op_6751_strides_0"), val = tensor([1, 1])]; tensor var_6751_pad_0 = const()[name = string("op_6751_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6751_dilations_0 = const()[name = string("op_6751_dilations_0"), val = tensor([1, 1])]; int32 var_6751_groups_0 = const()[name = string("op_6751_groups_0"), val = int32(1)]; tensor var_6751 = conv(dilations = var_6751_dilations_0, groups = var_6751_groups_0, pad = var_6751_pad_0, pad_type = var_6751_pad_type_0, strides = var_6751_strides_0, weight = layers_9_self_attn_v_proj_weight_palettized, x = var_6638_cast_fp16)[name = string("op_6751")]; tensor var_6756 = const()[name = string("op_6756"), val = tensor([1, 1, 512, 1])]; tensor var_6757 = reshape(shape = var_6756, x = var_6751)[name = string("op_6757")]; tensor var_6762 = const()[name = string("op_6762"), val = tensor([0, 1, 3, 2])]; tensor var_6772 = const()[name = string("op_6772"), val = tensor([1, 1, 512])]; tensor var_6735 = transpose(perm = var_6734, x = var_6729)[name = string("transpose_229")]; tensor x_279 = reshape(shape = var_6772, x = var_6735)[name = string("x_279")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-1)]; fp16 const_166_promoted_to_fp16 = const()[name = string("const_166_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6784_cast_fp16 = mul(x = x_279, y = const_166_promoted_to_fp16)[name = string("op_6784_cast_fp16")]; bool input_273_interleave_0 = const()[name = string("input_273_interleave_0"), val = bool(false)]; tensor input_273_cast_fp16 = concat(axis = var_6778, interleave = input_273_interleave_0, values = (x_279, var_6784_cast_fp16))[name = string("input_273_cast_fp16")]; tensor normed_261_axes_0 = const()[name = string("normed_261_axes_0"), val = tensor([-1])]; fp16 var_6776_to_fp16 = const()[name = string("op_6776_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_261_cast_fp16 = layer_norm(axes = normed_261_axes_0, epsilon = var_6776_to_fp16, x = input_273_cast_fp16)[name = string("normed_261_cast_fp16")]; tensor var_6789_split_sizes_0 = const()[name = string("op_6789_split_sizes_0"), val = tensor([512, 512])]; int32 var_6789_axis_0 = const()[name = string("op_6789_axis_0"), val = int32(-1)]; tensor var_6789_cast_fp16_0, tensor var_6789_cast_fp16_1 = split(axis = var_6789_axis_0, split_sizes = var_6789_split_sizes_0, x = normed_261_cast_fp16)[name = string("op_6789_cast_fp16")]; tensor const_167_to_fp16 = const()[name = string("const_167_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079513920)))]; tensor var_6792_cast_fp16 = mul(x = var_6789_cast_fp16_0, y = const_167_to_fp16)[name = string("op_6792_cast_fp16")]; tensor var_6798 = const()[name = string("op_6798"), val = tensor([1, 1, 1, 512])]; tensor q_77 = reshape(shape = var_6798, x = var_6792_cast_fp16)[name = string("q_77")]; fp16 var_6805_promoted_to_fp16 = const()[name = string("op_6805_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_6763 = transpose(perm = var_6762, x = var_6757)[name = string("transpose_228")]; tensor var_6806_cast_fp16 = pow(x = var_6763, y = var_6805_promoted_to_fp16)[name = string("op_6806_cast_fp16")]; tensor var_6811_axes_0 = const()[name = string("op_6811_axes_0"), val = tensor([-1])]; bool var_6811_keep_dims_0 = const()[name = string("op_6811_keep_dims_0"), val = bool(true)]; tensor var_6811_cast_fp16 = reduce_mean(axes = var_6811_axes_0, keep_dims = var_6811_keep_dims_0, x = var_6806_cast_fp16)[name = string("op_6811_cast_fp16")]; fp16 var_6813_to_fp16 = const()[name = string("op_6813_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_19_cast_fp16 = add(x = var_6811_cast_fp16, y = var_6813_to_fp16)[name = string("mean_sq_19_cast_fp16")]; fp16 var_6820_to_fp16 = const()[name = string("op_6820_to_fp16"), val = fp16(-0x1p-1)]; tensor var_6821_cast_fp16 = pow(x = mean_sq_19_cast_fp16, y = var_6820_to_fp16)[name = string("op_6821_cast_fp16")]; tensor var_6822_cast_fp16 = mul(x = var_6763, y = var_6821_cast_fp16)[name = string("op_6822_cast_fp16")]; tensor var_6828 = mul(x = q_77, y = cos)[name = string("op_6828")]; tensor var_6829_split_sizes_0 = const()[name = string("op_6829_split_sizes_0"), val = tensor([256, 256])]; int32 var_6829_axis_0 = const()[name = string("op_6829_axis_0"), val = int32(-1)]; tensor var_6829_0, tensor var_6829_1 = split(axis = var_6829_axis_0, split_sizes = var_6829_split_sizes_0, x = q_77)[name = string("op_6829")]; fp16 const_168_promoted = const()[name = string("const_168_promoted"), val = fp16(-0x1p+0)]; tensor var_6831 = mul(x = var_6829_1, y = const_168_promoted)[name = string("op_6831")]; int32 var_6833 = const()[name = string("op_6833"), val = int32(-1)]; bool var_6834_interleave_0 = const()[name = string("op_6834_interleave_0"), val = bool(false)]; tensor var_6834 = concat(axis = var_6833, interleave = var_6834_interleave_0, values = (var_6831, var_6829_0))[name = string("op_6834")]; tensor var_6835 = mul(x = var_6834, y = sin)[name = string("op_6835")]; tensor k_23 = add(x = var_6828, y = var_6835)[name = string("k_23")]; tensor var_6840_begin_0 = const()[name = string("op_6840_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_6840_end_0 = const()[name = string("op_6840_end_0"), val = tensor([10, 1, 512, 512])]; tensor var_6840_end_mask_0 = const()[name = string("op_6840_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6840_squeeze_mask_0 = const()[name = string("op_6840_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6840_cast_fp16 = slice_by_index(begin = var_6840_begin_0, end = var_6840_end_0, end_mask = var_6840_end_mask_0, squeeze_mask = var_6840_squeeze_mask_0, x = coreml_update_state_47)[name = string("op_6840_cast_fp16")]; tensor K_cache_19_axes_0 = const()[name = string("K_cache_19_axes_0"), val = tensor([0])]; tensor K_cache_19_cast_fp16 = expand_dims(axes = K_cache_19_axes_0, x = var_6840_cast_fp16)[name = string("K_cache_19_cast_fp16")]; tensor var_6845_begin_0 = const()[name = string("op_6845_begin_0"), val = tensor([44, 0, 0, 0])]; tensor var_6845_end_0 = const()[name = string("op_6845_end_0"), val = tensor([45, 1, 512, 512])]; tensor var_6845_end_mask_0 = const()[name = string("op_6845_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6845_squeeze_mask_0 = const()[name = string("op_6845_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6845_cast_fp16 = slice_by_index(begin = var_6845_begin_0, end = var_6845_end_0, end_mask = var_6845_end_mask_0, squeeze_mask = var_6845_squeeze_mask_0, x = coreml_update_state_47)[name = string("op_6845_cast_fp16")]; tensor V_cache_19_axes_0 = const()[name = string("V_cache_19_axes_0"), val = tensor([0])]; tensor V_cache_19_cast_fp16 = expand_dims(axes = V_cache_19_axes_0, x = var_6845_cast_fp16)[name = string("V_cache_19_cast_fp16")]; tensor var_6851_cast_fp16 = mul(x = K_cache_19_cast_fp16, y = var_2187_cast_fp16)[name = string("op_6851_cast_fp16")]; tensor var_6852_reps_0 = const()[name = string("op_6852_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_6852 = tile(reps = var_6852_reps_0, x = k_23)[name = string("op_6852")]; tensor var_6853_cast_fp16 = mul(x = var_6852, y = update_mask)[name = string("op_6853_cast_fp16")]; tensor K_new_19_cast_fp16 = add(x = var_6851_cast_fp16, y = var_6853_cast_fp16)[name = string("K_new_19_cast_fp16")]; tensor var_6859_cast_fp16 = mul(x = V_cache_19_cast_fp16, y = var_2187_cast_fp16)[name = string("op_6859_cast_fp16")]; tensor var_6860_reps_0 = const()[name = string("op_6860_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_6860 = tile(reps = var_6860_reps_0, x = var_6822_cast_fp16)[name = string("op_6860")]; tensor var_6861_cast_fp16 = mul(x = var_6860, y = update_mask)[name = string("op_6861_cast_fp16")]; tensor V_new_19_cast_fp16 = add(x = var_6859_cast_fp16, y = var_6861_cast_fp16)[name = string("V_new_19_cast_fp16")]; tensor var_6865_axes_0 = const()[name = string("op_6865_axes_0"), val = tensor([0])]; tensor var_6865_cast_fp16 = squeeze(axes = var_6865_axes_0, x = K_new_19_cast_fp16)[name = string("op_6865_cast_fp16")]; tensor concat_72 = const()[name = string("concat_72"), val = tensor([9, 0, 0, 0])]; tensor concat_73 = const()[name = string("concat_73"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_72, begin_mask = kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_73, end_mask = kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_19_stride_0, update = var_6865_cast_fp16, x = coreml_update_state_47)[name = string("kv_cache_0_internal_tensor_assign_19_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_19_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_48_write_state")]; tensor coreml_update_state_48 = read_state(input = kv_cache_0)[name = string("coreml_update_state_48")]; tensor var_6872_axes_0 = const()[name = string("op_6872_axes_0"), val = tensor([0])]; tensor var_6872_cast_fp16 = squeeze(axes = var_6872_axes_0, x = V_new_19_cast_fp16)[name = string("op_6872_cast_fp16")]; tensor concat_74 = const()[name = string("concat_74"), val = tensor([44, 0, 0, 0])]; tensor concat_75 = const()[name = string("concat_75"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_74, begin_mask = kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_75, end_mask = kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_20_stride_0, update = var_6872_cast_fp16, x = coreml_update_state_48)[name = string("kv_cache_0_internal_tensor_assign_20_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_20_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_49_write_state")]; tensor coreml_update_state_49 = read_state(input = kv_cache_0)[name = string("coreml_update_state_49")]; tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_18_reps_0 = const()[name = string("tile_18_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = K_new_19_cast_fp16)[name = string("transpose_227")]; tensor tile_18_cast_fp16 = tile(reps = tile_18_reps_0, x = transpose_36_cast_fp16)[name = string("tile_18_cast_fp16")]; tensor concat_76 = const()[name = string("concat_76"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_36_cast_fp16 = reshape(shape = concat_76, x = tile_18_cast_fp16)[name = string("reshape_36_cast_fp16")]; tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_77 = const()[name = string("concat_77"), val = tensor([-1, 1, 512, 512])]; tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_36_cast_fp16)[name = string("transpose_226")]; tensor reshape_37_cast_fp16 = reshape(shape = concat_77, x = transpose_37_cast_fp16)[name = string("reshape_37_cast_fp16")]; tensor transpose_149_perm_0 = const()[name = string("transpose_149_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_19_reps_0 = const()[name = string("tile_19_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = V_new_19_cast_fp16)[name = string("transpose_225")]; tensor tile_19_cast_fp16 = tile(reps = tile_19_reps_0, x = transpose_38_cast_fp16)[name = string("tile_19_cast_fp16")]; tensor concat_78 = const()[name = string("concat_78"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_38_cast_fp16 = reshape(shape = concat_78, x = tile_19_cast_fp16)[name = string("reshape_38_cast_fp16")]; tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_79 = const()[name = string("concat_79"), val = tensor([-1, 1, 512, 512])]; tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_224")]; tensor reshape_39_cast_fp16 = reshape(shape = concat_79, x = transpose_39_cast_fp16)[name = string("reshape_39_cast_fp16")]; tensor V_expanded_19_perm_0 = const()[name = string("V_expanded_19_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_6899_transpose_x_0 = const()[name = string("op_6899_transpose_x_0"), val = bool(false)]; bool var_6899_transpose_y_0 = const()[name = string("op_6899_transpose_y_0"), val = bool(false)]; tensor transpose_149_cast_fp16 = transpose(perm = transpose_149_perm_0, x = reshape_37_cast_fp16)[name = string("transpose_223")]; tensor var_6899_cast_fp16 = matmul(transpose_x = var_6899_transpose_x_0, transpose_y = var_6899_transpose_y_0, x = q_79, y = transpose_149_cast_fp16)[name = string("op_6899_cast_fp16")]; tensor attn_weights_57_cast_fp16 = add(x = var_6899_cast_fp16, y = causal_mask)[name = string("attn_weights_57_cast_fp16")]; int32 var_6904 = const()[name = string("op_6904"), val = int32(-1)]; tensor attn_weights_59_cast_fp16 = softmax(axis = var_6904, x = attn_weights_57_cast_fp16)[name = string("attn_weights_59_cast_fp16")]; bool attn_output_55_transpose_x_0 = const()[name = string("attn_output_55_transpose_x_0"), val = bool(false)]; bool attn_output_55_transpose_y_0 = const()[name = string("attn_output_55_transpose_y_0"), val = bool(false)]; tensor V_expanded_19_cast_fp16 = transpose(perm = V_expanded_19_perm_0, x = reshape_39_cast_fp16)[name = string("transpose_222")]; tensor attn_output_55_cast_fp16 = matmul(transpose_x = attn_output_55_transpose_x_0, transpose_y = attn_output_55_transpose_y_0, x = attn_weights_59_cast_fp16, y = V_expanded_19_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor var_6912 = const()[name = string("op_6912"), val = tensor([0, 2, 1, 3])]; tensor var_6919 = const()[name = string("op_6919"), val = tensor([1, 1, -1])]; tensor var_6913_cast_fp16 = transpose(perm = var_6912, x = attn_output_55_cast_fp16)[name = string("transpose_221")]; tensor attn_output_57_cast_fp16 = reshape(shape = var_6919, x = var_6913_cast_fp16)[name = string("attn_output_57_cast_fp16")]; tensor var_6924 = const()[name = string("op_6924"), val = tensor([0, 2, 1])]; string var_6940_pad_type_0 = const()[name = string("op_6940_pad_type_0"), val = string("valid")]; int32 var_6940_groups_0 = const()[name = string("op_6940_groups_0"), val = int32(1)]; tensor var_6940_strides_0 = const()[name = string("op_6940_strides_0"), val = tensor([1])]; tensor var_6940_pad_0 = const()[name = string("op_6940_pad_0"), val = tensor([0, 0])]; tensor var_6940_dilations_0 = const()[name = string("op_6940_dilations_0"), val = tensor([1])]; tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1079515008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082660800))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6925_cast_fp16 = transpose(perm = var_6924, x = attn_output_57_cast_fp16)[name = string("transpose_220")]; tensor var_6940_cast_fp16 = conv(dilations = var_6940_dilations_0, groups = var_6940_groups_0, pad = var_6940_pad_0, pad_type = var_6940_pad_type_0, strides = var_6940_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_6925_cast_fp16)[name = string("op_6940_cast_fp16")]; tensor var_6944 = const()[name = string("op_6944"), val = tensor([0, 2, 1])]; int32 var_6950 = const()[name = string("op_6950"), val = int32(-1)]; fp16 const_169_promoted_to_fp16 = const()[name = string("const_169_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_285_cast_fp16 = transpose(perm = var_6944, x = var_6940_cast_fp16)[name = string("transpose_219")]; tensor var_6956_cast_fp16 = mul(x = x_285_cast_fp16, y = const_169_promoted_to_fp16)[name = string("op_6956_cast_fp16")]; bool input_277_interleave_0 = const()[name = string("input_277_interleave_0"), val = bool(false)]; tensor input_277_cast_fp16 = concat(axis = var_6950, interleave = input_277_interleave_0, values = (x_285_cast_fp16, var_6956_cast_fp16))[name = string("input_277_cast_fp16")]; tensor normed_265_axes_0 = const()[name = string("normed_265_axes_0"), val = tensor([-1])]; fp16 var_6948_to_fp16 = const()[name = string("op_6948_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_265_cast_fp16 = layer_norm(axes = normed_265_axes_0, epsilon = var_6948_to_fp16, x = input_277_cast_fp16)[name = string("normed_265_cast_fp16")]; tensor var_6961_split_sizes_0 = const()[name = string("op_6961_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6961_axis_0 = const()[name = string("op_6961_axis_0"), val = int32(-1)]; tensor var_6961_cast_fp16_0, tensor var_6961_cast_fp16_1 = split(axis = var_6961_axis_0, split_sizes = var_6961_split_sizes_0, x = normed_265_cast_fp16)[name = string("op_6961_cast_fp16")]; tensor const_170_to_fp16 = const()[name = string("const_170_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082662400)))]; tensor var_6964_cast_fp16 = mul(x = var_6961_cast_fp16_0, y = const_170_to_fp16)[name = string("op_6964_cast_fp16")]; tensor x_289_cast_fp16 = add(x = x_271_cast_fp16, y = var_6964_cast_fp16)[name = string("x_289_cast_fp16")]; int32 var_6971 = const()[name = string("op_6971"), val = int32(-1)]; fp16 const_171_promoted_to_fp16 = const()[name = string("const_171_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6977_cast_fp16 = mul(x = x_289_cast_fp16, y = const_171_promoted_to_fp16)[name = string("op_6977_cast_fp16")]; bool input_279_interleave_0 = const()[name = string("input_279_interleave_0"), val = bool(false)]; tensor input_279_cast_fp16 = concat(axis = var_6971, interleave = input_279_interleave_0, values = (x_289_cast_fp16, var_6977_cast_fp16))[name = string("input_279_cast_fp16")]; tensor normed_269_axes_0 = const()[name = string("normed_269_axes_0"), val = tensor([-1])]; fp16 var_6969_to_fp16 = const()[name = string("op_6969_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_269_cast_fp16 = layer_norm(axes = normed_269_axes_0, epsilon = var_6969_to_fp16, x = input_279_cast_fp16)[name = string("normed_269_cast_fp16")]; tensor var_6982_split_sizes_0 = const()[name = string("op_6982_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_6982_axis_0 = const()[name = string("op_6982_axis_0"), val = int32(-1)]; tensor var_6982_cast_fp16_0, tensor var_6982_cast_fp16_1 = split(axis = var_6982_axis_0, split_sizes = var_6982_split_sizes_0, x = normed_269_cast_fp16)[name = string("op_6982_cast_fp16")]; tensor const_172_to_fp16 = const()[name = string("const_172_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082665536)))]; tensor var_6985_cast_fp16 = mul(x = var_6982_cast_fp16_0, y = const_172_to_fp16)[name = string("op_6985_cast_fp16")]; tensor var_6998 = const()[name = string("op_6998"), val = tensor([0, 2, 1])]; tensor input_281_axes_0 = const()[name = string("input_281_axes_0"), val = tensor([2])]; tensor var_6999 = transpose(perm = var_6998, x = var_6985_cast_fp16)[name = string("transpose_218")]; tensor input_281 = expand_dims(axes = input_281_axes_0, x = var_6999)[name = string("input_281")]; string gate_37_pad_type_0 = const()[name = string("gate_37_pad_type_0"), val = string("valid")]; tensor gate_37_strides_0 = const()[name = string("gate_37_strides_0"), val = tensor([1, 1])]; tensor gate_37_pad_0 = const()[name = string("gate_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_37_dilations_0 = const()[name = string("gate_37_dilations_0"), val = tensor([1, 1])]; int32 gate_37_groups_0 = const()[name = string("gate_37_groups_0"), val = int32(1)]; tensor gate_37 = conv(dilations = gate_37_dilations_0, groups = gate_37_groups_0, pad = gate_37_pad_0, pad_type = gate_37_pad_type_0, strides = gate_37_strides_0, weight = layers_9_mlp_gate_proj_weight_palettized, x = input_281)[name = string("gate_37")]; string up_19_pad_type_0 = const()[name = string("up_19_pad_type_0"), val = string("valid")]; tensor up_19_strides_0 = const()[name = string("up_19_strides_0"), val = tensor([1, 1])]; tensor up_19_pad_0 = const()[name = string("up_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_19_dilations_0 = const()[name = string("up_19_dilations_0"), val = tensor([1, 1])]; int32 up_19_groups_0 = const()[name = string("up_19_groups_0"), val = int32(1)]; tensor up_19 = conv(dilations = up_19_dilations_0, groups = up_19_groups_0, pad = up_19_pad_0, pad_type = up_19_pad_type_0, strides = up_19_strides_0, weight = layers_9_mlp_up_proj_weight_palettized, x = input_281)[name = string("up_19")]; string gate_39_mode_0 = const()[name = string("gate_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_39 = gelu(mode = gate_39_mode_0, x = gate_37)[name = string("gate_39")]; tensor input_283 = mul(x = gate_39, y = up_19)[name = string("input_283")]; string mlp_out_19_pad_type_0 = const()[name = string("mlp_out_19_pad_type_0"), val = string("valid")]; tensor mlp_out_19_strides_0 = const()[name = string("mlp_out_19_strides_0"), val = tensor([1, 1])]; tensor mlp_out_19_pad_0 = const()[name = string("mlp_out_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_19_dilations_0 = const()[name = string("mlp_out_19_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_19_groups_0 = const()[name = string("mlp_out_19_groups_0"), val = int32(1)]; tensor mlp_out_19 = conv(dilations = mlp_out_19_dilations_0, groups = mlp_out_19_groups_0, pad = mlp_out_19_pad_0, pad_type = mlp_out_19_pad_type_0, strides = mlp_out_19_strides_0, weight = layers_9_mlp_down_proj_weight_palettized, x = input_283)[name = string("mlp_out_19")]; tensor var_7039_axes_0 = const()[name = string("op_7039_axes_0"), val = tensor([2])]; tensor var_7039 = squeeze(axes = var_7039_axes_0, x = mlp_out_19)[name = string("op_7039")]; tensor var_7043 = const()[name = string("op_7043"), val = tensor([0, 2, 1])]; int32 var_7049 = const()[name = string("op_7049"), val = int32(-1)]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_293 = transpose(perm = var_7043, x = var_7039)[name = string("transpose_217")]; tensor var_7055_cast_fp16 = mul(x = x_293, y = const_173_promoted_to_fp16)[name = string("op_7055_cast_fp16")]; bool input_285_interleave_0 = const()[name = string("input_285_interleave_0"), val = bool(false)]; tensor input_285_cast_fp16 = concat(axis = var_7049, interleave = input_285_interleave_0, values = (x_293, var_7055_cast_fp16))[name = string("input_285_cast_fp16")]; tensor normed_273_axes_0 = const()[name = string("normed_273_axes_0"), val = tensor([-1])]; fp16 var_7047_to_fp16 = const()[name = string("op_7047_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_273_cast_fp16 = layer_norm(axes = normed_273_axes_0, epsilon = var_7047_to_fp16, x = input_285_cast_fp16)[name = string("normed_273_cast_fp16")]; tensor var_7060_split_sizes_0 = const()[name = string("op_7060_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7060_axis_0 = const()[name = string("op_7060_axis_0"), val = int32(-1)]; tensor var_7060_cast_fp16_0, tensor var_7060_cast_fp16_1 = split(axis = var_7060_axis_0, split_sizes = var_7060_split_sizes_0, x = normed_273_cast_fp16)[name = string("op_7060_cast_fp16")]; tensor const_174_to_fp16 = const()[name = string("const_174_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082668672)))]; tensor var_7063_cast_fp16 = mul(x = var_7060_cast_fp16_0, y = const_174_to_fp16)[name = string("op_7063_cast_fp16")]; tensor hidden_states_117_cast_fp16 = add(x = x_289_cast_fp16, y = var_7063_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor per_layer_slice_19_begin_0 = const()[name = string("per_layer_slice_19_begin_0"), val = tensor([0, 0, 2304])]; tensor per_layer_slice_19_end_0 = const()[name = string("per_layer_slice_19_end_0"), val = tensor([1, 1, 2560])]; tensor per_layer_slice_19_end_mask_0 = const()[name = string("per_layer_slice_19_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_19_cast_fp16 = slice_by_index(begin = per_layer_slice_19_begin_0, end = per_layer_slice_19_end_0, end_mask = per_layer_slice_19_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_19_cast_fp16")]; tensor gated_37 = linear(bias = linear_0_bias_0, weight = layers_9_per_layer_input_gate_weight_palettized, x = hidden_states_117_cast_fp16)[name = string("linear_18")]; string gated_39_mode_0 = const()[name = string("gated_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_39 = gelu(mode = gated_39_mode_0, x = gated_37)[name = string("gated_39")]; tensor input_289_cast_fp16 = mul(x = gated_39, y = per_layer_slice_19_cast_fp16)[name = string("input_289_cast_fp16")]; tensor layers_9_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082671808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082868480))))[name = string("layers_9_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_19_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_9_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_289_cast_fp16)[name = string("linear_19_cast_fp16")]; int32 var_7100 = const()[name = string("op_7100"), val = int32(-1)]; fp16 const_175_promoted_to_fp16 = const()[name = string("const_175_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7106_cast_fp16 = mul(x = linear_19_cast_fp16, y = const_175_promoted_to_fp16)[name = string("op_7106_cast_fp16")]; bool input_291_interleave_0 = const()[name = string("input_291_interleave_0"), val = bool(false)]; tensor input_291_cast_fp16 = concat(axis = var_7100, interleave = input_291_interleave_0, values = (linear_19_cast_fp16, var_7106_cast_fp16))[name = string("input_291_cast_fp16")]; tensor normed_277_axes_0 = const()[name = string("normed_277_axes_0"), val = tensor([-1])]; fp16 var_7098_to_fp16 = const()[name = string("op_7098_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_277_cast_fp16 = layer_norm(axes = normed_277_axes_0, epsilon = var_7098_to_fp16, x = input_291_cast_fp16)[name = string("normed_277_cast_fp16")]; tensor var_7111_split_sizes_0 = const()[name = string("op_7111_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7111_axis_0 = const()[name = string("op_7111_axis_0"), val = int32(-1)]; tensor var_7111_cast_fp16_0, tensor var_7111_cast_fp16_1 = split(axis = var_7111_axis_0, split_sizes = var_7111_split_sizes_0, x = normed_277_cast_fp16)[name = string("op_7111_cast_fp16")]; tensor const_176_to_fp16 = const()[name = string("const_176_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082870080)))]; tensor var_7114_cast_fp16 = mul(x = var_7111_cast_fp16_0, y = const_176_to_fp16)[name = string("op_7114_cast_fp16")]; tensor hidden_states_121_cast_fp16 = add(x = hidden_states_117_cast_fp16, y = var_7114_cast_fp16)[name = string("hidden_states_121_cast_fp16")]; tensor layers_9_layer_scalar_to_fp16 = const()[name = string("layers_9_layer_scalar_to_fp16"), val = tensor([0x1.dcp-2])]; tensor x_301_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = layers_9_layer_scalar_to_fp16)[name = string("x_301_cast_fp16")]; int32 var_7122 = const()[name = string("op_7122"), val = int32(-1)]; fp16 const_177_promoted_to_fp16 = const()[name = string("const_177_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7128_cast_fp16 = mul(x = x_301_cast_fp16, y = const_177_promoted_to_fp16)[name = string("op_7128_cast_fp16")]; bool input_293_interleave_0 = const()[name = string("input_293_interleave_0"), val = bool(false)]; tensor input_293_cast_fp16 = concat(axis = var_7122, interleave = input_293_interleave_0, values = (x_301_cast_fp16, var_7128_cast_fp16))[name = string("input_293_cast_fp16")]; tensor normed_281_axes_0 = const()[name = string("normed_281_axes_0"), val = tensor([-1])]; fp16 var_7120_to_fp16 = const()[name = string("op_7120_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_281_cast_fp16 = layer_norm(axes = normed_281_axes_0, epsilon = var_7120_to_fp16, x = input_293_cast_fp16)[name = string("normed_281_cast_fp16")]; tensor var_7133_split_sizes_0 = const()[name = string("op_7133_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7133_axis_0 = const()[name = string("op_7133_axis_0"), val = int32(-1)]; tensor var_7133_cast_fp16_0, tensor var_7133_cast_fp16_1 = split(axis = var_7133_axis_0, split_sizes = var_7133_split_sizes_0, x = normed_281_cast_fp16)[name = string("op_7133_cast_fp16")]; tensor const_178_to_fp16 = const()[name = string("const_178_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082873216)))]; tensor var_7136_cast_fp16 = mul(x = var_7133_cast_fp16_0, y = const_178_to_fp16)[name = string("op_7136_cast_fp16")]; tensor var_7144 = const()[name = string("op_7144"), val = tensor([0, 2, 1])]; tensor var_7147_axes_0 = const()[name = string("op_7147_axes_0"), val = tensor([2])]; tensor var_7145_cast_fp16 = transpose(perm = var_7144, x = var_7136_cast_fp16)[name = string("transpose_216")]; tensor var_7147_cast_fp16 = expand_dims(axes = var_7147_axes_0, x = var_7145_cast_fp16)[name = string("op_7147_cast_fp16")]; string var_7163_pad_type_0 = const()[name = string("op_7163_pad_type_0"), val = string("valid")]; tensor var_7163_strides_0 = const()[name = string("op_7163_strides_0"), val = tensor([1, 1])]; tensor var_7163_pad_0 = const()[name = string("op_7163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7163_dilations_0 = const()[name = string("op_7163_dilations_0"), val = tensor([1, 1])]; int32 var_7163_groups_0 = const()[name = string("op_7163_groups_0"), val = int32(1)]; tensor var_7163 = conv(dilations = var_7163_dilations_0, groups = var_7163_groups_0, pad = var_7163_pad_0, pad_type = var_7163_pad_type_0, strides = var_7163_strides_0, weight = layers_10_self_attn_q_proj_weight_palettized, x = var_7147_cast_fp16)[name = string("op_7163")]; tensor var_7168 = const()[name = string("op_7168"), val = tensor([1, 8, 256, 1])]; tensor var_7169 = reshape(shape = var_7168, x = var_7163)[name = string("op_7169")]; tensor var_7174 = const()[name = string("op_7174"), val = tensor([0, 1, 3, 2])]; tensor var_7184 = const()[name = string("op_7184"), val = tensor([1, 8, 256])]; tensor var_7175 = transpose(perm = var_7174, x = var_7169)[name = string("transpose_215")]; tensor x_305 = reshape(shape = var_7184, x = var_7175)[name = string("x_305")]; int32 var_7190 = const()[name = string("op_7190"), val = int32(-1)]; fp16 const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7196_cast_fp16 = mul(x = x_305, y = const_179_promoted_to_fp16)[name = string("op_7196_cast_fp16")]; bool input_297_interleave_0 = const()[name = string("input_297_interleave_0"), val = bool(false)]; tensor input_297_cast_fp16 = concat(axis = var_7190, interleave = input_297_interleave_0, values = (x_305, var_7196_cast_fp16))[name = string("input_297_cast_fp16")]; tensor normed_285_axes_0 = const()[name = string("normed_285_axes_0"), val = tensor([-1])]; fp16 var_7188_to_fp16 = const()[name = string("op_7188_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_285_cast_fp16 = layer_norm(axes = normed_285_axes_0, epsilon = var_7188_to_fp16, x = input_297_cast_fp16)[name = string("normed_285_cast_fp16")]; tensor var_7201_split_sizes_0 = const()[name = string("op_7201_split_sizes_0"), val = tensor([256, 256])]; int32 var_7201_axis_0 = const()[name = string("op_7201_axis_0"), val = int32(-1)]; tensor var_7201_cast_fp16_0, tensor var_7201_cast_fp16_1 = split(axis = var_7201_axis_0, split_sizes = var_7201_split_sizes_0, x = normed_285_cast_fp16)[name = string("op_7201_cast_fp16")]; tensor var_7204_cast_fp16 = mul(x = var_7201_cast_fp16_0, y = const_58_to_fp16)[name = string("op_7204_cast_fp16")]; tensor var_7210 = const()[name = string("op_7210"), val = tensor([1, 8, 1, 256])]; tensor q_83 = reshape(shape = var_7210, x = var_7204_cast_fp16)[name = string("q_83")]; tensor var_7212 = mul(x = q_83, y = cos_1)[name = string("op_7212")]; tensor var_7213_split_sizes_0 = const()[name = string("op_7213_split_sizes_0"), val = tensor([128, 128])]; int32 var_7213_axis_0 = const()[name = string("op_7213_axis_0"), val = int32(-1)]; tensor var_7213_0, tensor var_7213_1 = split(axis = var_7213_axis_0, split_sizes = var_7213_split_sizes_0, x = q_83)[name = string("op_7213")]; fp16 const_181_promoted = const()[name = string("const_181_promoted"), val = fp16(-0x1p+0)]; tensor var_7215 = mul(x = var_7213_1, y = const_181_promoted)[name = string("op_7215")]; int32 var_7217 = const()[name = string("op_7217"), val = int32(-1)]; bool var_7218_interleave_0 = const()[name = string("op_7218_interleave_0"), val = bool(false)]; tensor var_7218 = concat(axis = var_7217, interleave = var_7218_interleave_0, values = (var_7215, var_7213_0))[name = string("op_7218")]; tensor var_7219 = mul(x = var_7218, y = sin_1)[name = string("op_7219")]; tensor q_87 = add(x = var_7212, y = var_7219)[name = string("q_87")]; string var_7232_pad_type_0 = const()[name = string("op_7232_pad_type_0"), val = string("valid")]; tensor var_7232_strides_0 = const()[name = string("op_7232_strides_0"), val = tensor([1, 1])]; tensor var_7232_pad_0 = const()[name = string("op_7232_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7232_dilations_0 = const()[name = string("op_7232_dilations_0"), val = tensor([1, 1])]; int32 var_7232_groups_0 = const()[name = string("op_7232_groups_0"), val = int32(1)]; tensor var_7232 = conv(dilations = var_7232_dilations_0, groups = var_7232_groups_0, pad = var_7232_pad_0, pad_type = var_7232_pad_type_0, strides = var_7232_strides_0, weight = layers_10_self_attn_k_proj_weight_palettized, x = var_7147_cast_fp16)[name = string("op_7232")]; tensor var_7237 = const()[name = string("op_7237"), val = tensor([1, 1, 256, 1])]; tensor var_7238 = reshape(shape = var_7237, x = var_7232)[name = string("op_7238")]; tensor var_7243 = const()[name = string("op_7243"), val = tensor([0, 1, 3, 2])]; string var_7260_pad_type_0 = const()[name = string("op_7260_pad_type_0"), val = string("valid")]; tensor var_7260_strides_0 = const()[name = string("op_7260_strides_0"), val = tensor([1, 1])]; tensor var_7260_pad_0 = const()[name = string("op_7260_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7260_dilations_0 = const()[name = string("op_7260_dilations_0"), val = tensor([1, 1])]; int32 var_7260_groups_0 = const()[name = string("op_7260_groups_0"), val = int32(1)]; tensor var_7260 = conv(dilations = var_7260_dilations_0, groups = var_7260_groups_0, pad = var_7260_pad_0, pad_type = var_7260_pad_type_0, strides = var_7260_strides_0, weight = layers_10_self_attn_v_proj_weight_palettized, x = var_7147_cast_fp16)[name = string("op_7260")]; tensor var_7265 = const()[name = string("op_7265"), val = tensor([1, 1, 256, 1])]; tensor var_7266 = reshape(shape = var_7265, x = var_7260)[name = string("op_7266")]; tensor var_7271 = const()[name = string("op_7271"), val = tensor([0, 1, 3, 2])]; tensor var_7281 = const()[name = string("op_7281"), val = tensor([1, 1, 256])]; tensor var_7244 = transpose(perm = var_7243, x = var_7238)[name = string("transpose_214")]; tensor x_309 = reshape(shape = var_7281, x = var_7244)[name = string("x_309")]; int32 var_7287 = const()[name = string("op_7287"), val = int32(-1)]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7293_cast_fp16 = mul(x = x_309, y = const_182_promoted_to_fp16)[name = string("op_7293_cast_fp16")]; bool input_299_interleave_0 = const()[name = string("input_299_interleave_0"), val = bool(false)]; tensor input_299_cast_fp16 = concat(axis = var_7287, interleave = input_299_interleave_0, values = (x_309, var_7293_cast_fp16))[name = string("input_299_cast_fp16")]; tensor normed_289_axes_0 = const()[name = string("normed_289_axes_0"), val = tensor([-1])]; fp16 var_7285_to_fp16 = const()[name = string("op_7285_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_289_cast_fp16 = layer_norm(axes = normed_289_axes_0, epsilon = var_7285_to_fp16, x = input_299_cast_fp16)[name = string("normed_289_cast_fp16")]; tensor var_7298_split_sizes_0 = const()[name = string("op_7298_split_sizes_0"), val = tensor([256, 256])]; int32 var_7298_axis_0 = const()[name = string("op_7298_axis_0"), val = int32(-1)]; tensor var_7298_cast_fp16_0, tensor var_7298_cast_fp16_1 = split(axis = var_7298_axis_0, split_sizes = var_7298_split_sizes_0, x = normed_289_cast_fp16)[name = string("op_7298_cast_fp16")]; tensor const_183_to_fp16 = const()[name = string("const_183_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082876352)))]; tensor var_7301_cast_fp16 = mul(x = var_7298_cast_fp16_0, y = const_183_to_fp16)[name = string("op_7301_cast_fp16")]; tensor var_7307 = const()[name = string("op_7307"), val = tensor([1, 1, 1, 256])]; tensor q_85 = reshape(shape = var_7307, x = var_7301_cast_fp16)[name = string("q_85")]; fp16 var_7314_promoted_to_fp16 = const()[name = string("op_7314_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_7272 = transpose(perm = var_7271, x = var_7266)[name = string("transpose_213")]; tensor var_7315_cast_fp16 = pow(x = var_7272, y = var_7314_promoted_to_fp16)[name = string("op_7315_cast_fp16")]; tensor var_7320_axes_0 = const()[name = string("op_7320_axes_0"), val = tensor([-1])]; bool var_7320_keep_dims_0 = const()[name = string("op_7320_keep_dims_0"), val = bool(true)]; tensor var_7320_cast_fp16 = reduce_mean(axes = var_7320_axes_0, keep_dims = var_7320_keep_dims_0, x = var_7315_cast_fp16)[name = string("op_7320_cast_fp16")]; fp16 var_7322_to_fp16 = const()[name = string("op_7322_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_21_cast_fp16 = add(x = var_7320_cast_fp16, y = var_7322_to_fp16)[name = string("mean_sq_21_cast_fp16")]; fp16 var_7329_to_fp16 = const()[name = string("op_7329_to_fp16"), val = fp16(-0x1p-1)]; tensor var_7330_cast_fp16 = pow(x = mean_sq_21_cast_fp16, y = var_7329_to_fp16)[name = string("op_7330_cast_fp16")]; tensor var_7331_cast_fp16 = mul(x = var_7272, y = var_7330_cast_fp16)[name = string("op_7331_cast_fp16")]; tensor var_7337 = mul(x = q_85, y = cos_1)[name = string("op_7337")]; tensor var_7338_split_sizes_0 = const()[name = string("op_7338_split_sizes_0"), val = tensor([128, 128])]; int32 var_7338_axis_0 = const()[name = string("op_7338_axis_0"), val = int32(-1)]; tensor var_7338_0, tensor var_7338_1 = split(axis = var_7338_axis_0, split_sizes = var_7338_split_sizes_0, x = q_85)[name = string("op_7338")]; fp16 const_184_promoted = const()[name = string("const_184_promoted"), val = fp16(-0x1p+0)]; tensor var_7340 = mul(x = var_7338_1, y = const_184_promoted)[name = string("op_7340")]; int32 var_7342 = const()[name = string("op_7342"), val = int32(-1)]; bool var_7343_interleave_0 = const()[name = string("op_7343_interleave_0"), val = bool(false)]; tensor var_7343 = concat(axis = var_7342, interleave = var_7343_interleave_0, values = (var_7340, var_7338_0))[name = string("op_7343")]; tensor var_7344 = mul(x = var_7343, y = sin_1)[name = string("op_7344")]; tensor input_301 = add(x = var_7337, y = var_7344)[name = string("input_301")]; tensor var_7349_begin_0 = const()[name = string("op_7349_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_7349_end_0 = const()[name = string("op_7349_end_0"), val = tensor([11, 1, 512, 512])]; tensor var_7349_end_mask_0 = const()[name = string("op_7349_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7349_squeeze_mask_0 = const()[name = string("op_7349_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_7349_cast_fp16 = slice_by_index(begin = var_7349_begin_0, end = var_7349_end_0, end_mask = var_7349_end_mask_0, squeeze_mask = var_7349_squeeze_mask_0, x = coreml_update_state_49)[name = string("op_7349_cast_fp16")]; tensor K_cache_21_axes_0 = const()[name = string("K_cache_21_axes_0"), val = tensor([0])]; tensor K_cache_21_cast_fp16 = expand_dims(axes = K_cache_21_axes_0, x = var_7349_cast_fp16)[name = string("K_cache_21_cast_fp16")]; tensor var_7354_begin_0 = const()[name = string("op_7354_begin_0"), val = tensor([45, 0, 0, 0])]; tensor var_7354_end_0 = const()[name = string("op_7354_end_0"), val = tensor([46, 1, 512, 512])]; tensor var_7354_end_mask_0 = const()[name = string("op_7354_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7354_squeeze_mask_0 = const()[name = string("op_7354_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_7354_cast_fp16 = slice_by_index(begin = var_7354_begin_0, end = var_7354_end_0, end_mask = var_7354_end_mask_0, squeeze_mask = var_7354_squeeze_mask_0, x = coreml_update_state_49)[name = string("op_7354_cast_fp16")]; tensor V_cache_21_axes_0 = const()[name = string("V_cache_21_axes_0"), val = tensor([0])]; tensor V_cache_21_cast_fp16 = expand_dims(axes = V_cache_21_axes_0, x = var_7354_cast_fp16)[name = string("V_cache_21_cast_fp16")]; tensor k_padded_17_pad_0 = const()[name = string("k_padded_17_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_17_mode_0 = const()[name = string("k_padded_17_mode_0"), val = string("constant")]; fp16 const_185_to_fp16 = const()[name = string("const_185_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_17_cast_fp16 = pad(constant_val = const_185_to_fp16, mode = k_padded_17_mode_0, pad = k_padded_17_pad_0, x = input_301)[name = string("k_padded_17_cast_fp16")]; tensor v_padded_17_pad_0 = const()[name = string("v_padded_17_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_17_mode_0 = const()[name = string("v_padded_17_mode_0"), val = string("constant")]; fp16 const_186_to_fp16 = const()[name = string("const_186_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_17_cast_fp16 = pad(constant_val = const_186_to_fp16, mode = v_padded_17_mode_0, pad = v_padded_17_pad_0, x = var_7331_cast_fp16)[name = string("v_padded_17_cast_fp16")]; tensor var_7372_cast_fp16 = mul(x = K_cache_21_cast_fp16, y = var_2187_cast_fp16)[name = string("op_7372_cast_fp16")]; tensor var_7373_reps_0 = const()[name = string("op_7373_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_7373_cast_fp16 = tile(reps = var_7373_reps_0, x = k_padded_17_cast_fp16)[name = string("op_7373_cast_fp16")]; tensor var_7374_cast_fp16 = mul(x = var_7373_cast_fp16, y = update_mask)[name = string("op_7374_cast_fp16")]; tensor K_new_21_cast_fp16 = add(x = var_7372_cast_fp16, y = var_7374_cast_fp16)[name = string("K_new_21_cast_fp16")]; tensor var_7380_cast_fp16 = mul(x = V_cache_21_cast_fp16, y = var_2187_cast_fp16)[name = string("op_7380_cast_fp16")]; tensor var_7381_reps_0 = const()[name = string("op_7381_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_7381_cast_fp16 = tile(reps = var_7381_reps_0, x = v_padded_17_cast_fp16)[name = string("op_7381_cast_fp16")]; tensor var_7382_cast_fp16 = mul(x = var_7381_cast_fp16, y = update_mask)[name = string("op_7382_cast_fp16")]; tensor V_new_21_cast_fp16 = add(x = var_7380_cast_fp16, y = var_7382_cast_fp16)[name = string("V_new_21_cast_fp16")]; tensor var_7386_axes_0 = const()[name = string("op_7386_axes_0"), val = tensor([0])]; tensor var_7386_cast_fp16 = squeeze(axes = var_7386_axes_0, x = K_new_21_cast_fp16)[name = string("op_7386_cast_fp16")]; tensor concat_80 = const()[name = string("concat_80"), val = tensor([10, 0, 0, 0])]; tensor concat_81 = const()[name = string("concat_81"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_80, begin_mask = kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_81, end_mask = kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_21_stride_0, update = var_7386_cast_fp16, x = coreml_update_state_49)[name = string("kv_cache_0_internal_tensor_assign_21_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_21_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_50_write_state")]; tensor coreml_update_state_50 = read_state(input = kv_cache_0)[name = string("coreml_update_state_50")]; tensor var_7393_axes_0 = const()[name = string("op_7393_axes_0"), val = tensor([0])]; tensor var_7393_cast_fp16 = squeeze(axes = var_7393_axes_0, x = V_new_21_cast_fp16)[name = string("op_7393_cast_fp16")]; tensor concat_82 = const()[name = string("concat_82"), val = tensor([45, 0, 0, 0])]; tensor concat_83 = const()[name = string("concat_83"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_82, begin_mask = kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_83, end_mask = kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_22_stride_0, update = var_7393_cast_fp16, x = coreml_update_state_50)[name = string("kv_cache_0_internal_tensor_assign_22_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_22_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_51_write_state")]; tensor coreml_update_state_51 = read_state(input = kv_cache_0)[name = string("coreml_update_state_51")]; tensor K_for_attn_21_begin_0 = const()[name = string("K_for_attn_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_21_end_0 = const()[name = string("K_for_attn_21_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_21_end_mask_0 = const()[name = string("K_for_attn_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_21_cast_fp16 = slice_by_index(begin = K_for_attn_21_begin_0, end = K_for_attn_21_end_0, end_mask = K_for_attn_21_end_mask_0, x = K_new_21_cast_fp16)[name = string("K_for_attn_21_cast_fp16")]; tensor V_for_attn_21_begin_0 = const()[name = string("V_for_attn_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_21_end_0 = const()[name = string("V_for_attn_21_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_21_end_mask_0 = const()[name = string("V_for_attn_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_21_cast_fp16 = slice_by_index(begin = V_for_attn_21_begin_0, end = V_for_attn_21_end_0, end_mask = V_for_attn_21_end_mask_0, x = V_new_21_cast_fp16)[name = string("V_for_attn_21_cast_fp16")]; tensor transpose_40_perm_0 = const()[name = string("transpose_40_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_20_reps_0 = const()[name = string("tile_20_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_40_cast_fp16 = transpose(perm = transpose_40_perm_0, x = K_for_attn_21_cast_fp16)[name = string("transpose_212")]; tensor tile_20_cast_fp16 = tile(reps = tile_20_reps_0, x = transpose_40_cast_fp16)[name = string("tile_20_cast_fp16")]; tensor concat_84 = const()[name = string("concat_84"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_40_cast_fp16 = reshape(shape = concat_84, x = tile_20_cast_fp16)[name = string("reshape_40_cast_fp16")]; tensor transpose_41_perm_0 = const()[name = string("transpose_41_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_85 = const()[name = string("concat_85"), val = tensor([-1, 1, 512, 256])]; tensor transpose_41_cast_fp16 = transpose(perm = transpose_41_perm_0, x = reshape_40_cast_fp16)[name = string("transpose_211")]; tensor reshape_41_cast_fp16 = reshape(shape = concat_85, x = transpose_41_cast_fp16)[name = string("reshape_41_cast_fp16")]; tensor transpose_150_perm_0 = const()[name = string("transpose_150_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_42_perm_0 = const()[name = string("transpose_42_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_21_reps_0 = const()[name = string("tile_21_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_42_cast_fp16 = transpose(perm = transpose_42_perm_0, x = V_for_attn_21_cast_fp16)[name = string("transpose_210")]; tensor tile_21_cast_fp16 = tile(reps = tile_21_reps_0, x = transpose_42_cast_fp16)[name = string("tile_21_cast_fp16")]; tensor concat_86 = const()[name = string("concat_86"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_42_cast_fp16 = reshape(shape = concat_86, x = tile_21_cast_fp16)[name = string("reshape_42_cast_fp16")]; tensor transpose_43_perm_0 = const()[name = string("transpose_43_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_87 = const()[name = string("concat_87"), val = tensor([-1, 1, 512, 256])]; tensor transpose_43_cast_fp16 = transpose(perm = transpose_43_perm_0, x = reshape_42_cast_fp16)[name = string("transpose_209")]; tensor reshape_43_cast_fp16 = reshape(shape = concat_87, x = transpose_43_cast_fp16)[name = string("reshape_43_cast_fp16")]; tensor V_expanded_21_perm_0 = const()[name = string("V_expanded_21_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_7420_transpose_x_0 = const()[name = string("op_7420_transpose_x_0"), val = bool(false)]; bool var_7420_transpose_y_0 = const()[name = string("op_7420_transpose_y_0"), val = bool(false)]; tensor transpose_150_cast_fp16 = transpose(perm = transpose_150_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_208")]; tensor var_7420_cast_fp16 = matmul(transpose_x = var_7420_transpose_x_0, transpose_y = var_7420_transpose_y_0, x = q_87, y = transpose_150_cast_fp16)[name = string("op_7420_cast_fp16")]; tensor attn_weights_63_cast_fp16 = add(x = var_7420_cast_fp16, y = causal_mask)[name = string("attn_weights_63_cast_fp16")]; int32 var_7425 = const()[name = string("op_7425"), val = int32(-1)]; tensor attn_weights_65_cast_fp16 = softmax(axis = var_7425, x = attn_weights_63_cast_fp16)[name = string("attn_weights_65_cast_fp16")]; bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; tensor V_expanded_21_cast_fp16 = transpose(perm = V_expanded_21_perm_0, x = reshape_43_cast_fp16)[name = string("transpose_207")]; tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = attn_weights_65_cast_fp16, y = V_expanded_21_cast_fp16)[name = string("attn_output_61_cast_fp16")]; tensor var_7433 = const()[name = string("op_7433"), val = tensor([0, 2, 1, 3])]; tensor var_7440 = const()[name = string("op_7440"), val = tensor([1, 1, -1])]; tensor var_7434_cast_fp16 = transpose(perm = var_7433, x = attn_output_61_cast_fp16)[name = string("transpose_206")]; tensor attn_output_63_cast_fp16 = reshape(shape = var_7440, x = var_7434_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor var_7445 = const()[name = string("op_7445"), val = tensor([0, 2, 1])]; string var_7461_pad_type_0 = const()[name = string("op_7461_pad_type_0"), val = string("valid")]; int32 var_7461_groups_0 = const()[name = string("op_7461_groups_0"), val = int32(1)]; tensor var_7461_strides_0 = const()[name = string("op_7461_strides_0"), val = tensor([1])]; tensor var_7461_pad_0 = const()[name = string("op_7461_pad_0"), val = tensor([0, 0])]; tensor var_7461_dilations_0 = const()[name = string("op_7461_dilations_0"), val = tensor([1])]; tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1082876928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084449856))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7446_cast_fp16 = transpose(perm = var_7445, x = attn_output_63_cast_fp16)[name = string("transpose_205")]; tensor var_7461_cast_fp16 = conv(dilations = var_7461_dilations_0, groups = var_7461_groups_0, pad = var_7461_pad_0, pad_type = var_7461_pad_type_0, strides = var_7461_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_7446_cast_fp16)[name = string("op_7461_cast_fp16")]; tensor var_7465 = const()[name = string("op_7465"), val = tensor([0, 2, 1])]; int32 var_7471 = const()[name = string("op_7471"), val = int32(-1)]; fp16 const_187_promoted_to_fp16 = const()[name = string("const_187_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_315_cast_fp16 = transpose(perm = var_7465, x = var_7461_cast_fp16)[name = string("transpose_204")]; tensor var_7477_cast_fp16 = mul(x = x_315_cast_fp16, y = const_187_promoted_to_fp16)[name = string("op_7477_cast_fp16")]; bool input_307_interleave_0 = const()[name = string("input_307_interleave_0"), val = bool(false)]; tensor input_307_cast_fp16 = concat(axis = var_7471, interleave = input_307_interleave_0, values = (x_315_cast_fp16, var_7477_cast_fp16))[name = string("input_307_cast_fp16")]; tensor normed_293_axes_0 = const()[name = string("normed_293_axes_0"), val = tensor([-1])]; fp16 var_7469_to_fp16 = const()[name = string("op_7469_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_293_cast_fp16 = layer_norm(axes = normed_293_axes_0, epsilon = var_7469_to_fp16, x = input_307_cast_fp16)[name = string("normed_293_cast_fp16")]; tensor var_7482_split_sizes_0 = const()[name = string("op_7482_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7482_axis_0 = const()[name = string("op_7482_axis_0"), val = int32(-1)]; tensor var_7482_cast_fp16_0, tensor var_7482_cast_fp16_1 = split(axis = var_7482_axis_0, split_sizes = var_7482_split_sizes_0, x = normed_293_cast_fp16)[name = string("op_7482_cast_fp16")]; tensor const_188_to_fp16 = const()[name = string("const_188_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084451456)))]; tensor var_7485_cast_fp16 = mul(x = var_7482_cast_fp16_0, y = const_188_to_fp16)[name = string("op_7485_cast_fp16")]; tensor x_319_cast_fp16 = add(x = x_301_cast_fp16, y = var_7485_cast_fp16)[name = string("x_319_cast_fp16")]; int32 var_7492 = const()[name = string("op_7492"), val = int32(-1)]; fp16 const_189_promoted_to_fp16 = const()[name = string("const_189_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7498_cast_fp16 = mul(x = x_319_cast_fp16, y = const_189_promoted_to_fp16)[name = string("op_7498_cast_fp16")]; bool input_309_interleave_0 = const()[name = string("input_309_interleave_0"), val = bool(false)]; tensor input_309_cast_fp16 = concat(axis = var_7492, interleave = input_309_interleave_0, values = (x_319_cast_fp16, var_7498_cast_fp16))[name = string("input_309_cast_fp16")]; tensor normed_297_axes_0 = const()[name = string("normed_297_axes_0"), val = tensor([-1])]; fp16 var_7490_to_fp16 = const()[name = string("op_7490_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_297_cast_fp16 = layer_norm(axes = normed_297_axes_0, epsilon = var_7490_to_fp16, x = input_309_cast_fp16)[name = string("normed_297_cast_fp16")]; tensor var_7503_split_sizes_0 = const()[name = string("op_7503_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7503_axis_0 = const()[name = string("op_7503_axis_0"), val = int32(-1)]; tensor var_7503_cast_fp16_0, tensor var_7503_cast_fp16_1 = split(axis = var_7503_axis_0, split_sizes = var_7503_split_sizes_0, x = normed_297_cast_fp16)[name = string("op_7503_cast_fp16")]; tensor const_190_to_fp16 = const()[name = string("const_190_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084454592)))]; tensor var_7506_cast_fp16 = mul(x = var_7503_cast_fp16_0, y = const_190_to_fp16)[name = string("op_7506_cast_fp16")]; tensor var_7519 = const()[name = string("op_7519"), val = tensor([0, 2, 1])]; tensor input_311_axes_0 = const()[name = string("input_311_axes_0"), val = tensor([2])]; tensor var_7520 = transpose(perm = var_7519, x = var_7506_cast_fp16)[name = string("transpose_203")]; tensor input_311 = expand_dims(axes = input_311_axes_0, x = var_7520)[name = string("input_311")]; string gate_41_pad_type_0 = const()[name = string("gate_41_pad_type_0"), val = string("valid")]; tensor gate_41_strides_0 = const()[name = string("gate_41_strides_0"), val = tensor([1, 1])]; tensor gate_41_pad_0 = const()[name = string("gate_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_41_dilations_0 = const()[name = string("gate_41_dilations_0"), val = tensor([1, 1])]; int32 gate_41_groups_0 = const()[name = string("gate_41_groups_0"), val = int32(1)]; tensor gate_41 = conv(dilations = gate_41_dilations_0, groups = gate_41_groups_0, pad = gate_41_pad_0, pad_type = gate_41_pad_type_0, strides = gate_41_strides_0, weight = layers_10_mlp_gate_proj_weight_palettized, x = input_311)[name = string("gate_41")]; string up_21_pad_type_0 = const()[name = string("up_21_pad_type_0"), val = string("valid")]; tensor up_21_strides_0 = const()[name = string("up_21_strides_0"), val = tensor([1, 1])]; tensor up_21_pad_0 = const()[name = string("up_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_21_dilations_0 = const()[name = string("up_21_dilations_0"), val = tensor([1, 1])]; int32 up_21_groups_0 = const()[name = string("up_21_groups_0"), val = int32(1)]; tensor up_21 = conv(dilations = up_21_dilations_0, groups = up_21_groups_0, pad = up_21_pad_0, pad_type = up_21_pad_type_0, strides = up_21_strides_0, weight = layers_10_mlp_up_proj_weight_palettized, x = input_311)[name = string("up_21")]; string gate_43_mode_0 = const()[name = string("gate_43_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_43 = gelu(mode = gate_43_mode_0, x = gate_41)[name = string("gate_43")]; tensor input_313 = mul(x = gate_43, y = up_21)[name = string("input_313")]; string mlp_out_21_pad_type_0 = const()[name = string("mlp_out_21_pad_type_0"), val = string("valid")]; tensor mlp_out_21_strides_0 = const()[name = string("mlp_out_21_strides_0"), val = tensor([1, 1])]; tensor mlp_out_21_pad_0 = const()[name = string("mlp_out_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_21_dilations_0 = const()[name = string("mlp_out_21_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_21_groups_0 = const()[name = string("mlp_out_21_groups_0"), val = int32(1)]; tensor mlp_out_21 = conv(dilations = mlp_out_21_dilations_0, groups = mlp_out_21_groups_0, pad = mlp_out_21_pad_0, pad_type = mlp_out_21_pad_type_0, strides = mlp_out_21_strides_0, weight = layers_10_mlp_down_proj_weight_palettized, x = input_313)[name = string("mlp_out_21")]; tensor var_7560_axes_0 = const()[name = string("op_7560_axes_0"), val = tensor([2])]; tensor var_7560 = squeeze(axes = var_7560_axes_0, x = mlp_out_21)[name = string("op_7560")]; tensor var_7564 = const()[name = string("op_7564"), val = tensor([0, 2, 1])]; int32 var_7570 = const()[name = string("op_7570"), val = int32(-1)]; fp16 const_191_promoted_to_fp16 = const()[name = string("const_191_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_323 = transpose(perm = var_7564, x = var_7560)[name = string("transpose_202")]; tensor var_7576_cast_fp16 = mul(x = x_323, y = const_191_promoted_to_fp16)[name = string("op_7576_cast_fp16")]; bool input_315_interleave_0 = const()[name = string("input_315_interleave_0"), val = bool(false)]; tensor input_315_cast_fp16 = concat(axis = var_7570, interleave = input_315_interleave_0, values = (x_323, var_7576_cast_fp16))[name = string("input_315_cast_fp16")]; tensor normed_301_axes_0 = const()[name = string("normed_301_axes_0"), val = tensor([-1])]; fp16 var_7568_to_fp16 = const()[name = string("op_7568_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_301_cast_fp16 = layer_norm(axes = normed_301_axes_0, epsilon = var_7568_to_fp16, x = input_315_cast_fp16)[name = string("normed_301_cast_fp16")]; tensor var_7581_split_sizes_0 = const()[name = string("op_7581_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7581_axis_0 = const()[name = string("op_7581_axis_0"), val = int32(-1)]; tensor var_7581_cast_fp16_0, tensor var_7581_cast_fp16_1 = split(axis = var_7581_axis_0, split_sizes = var_7581_split_sizes_0, x = normed_301_cast_fp16)[name = string("op_7581_cast_fp16")]; tensor const_192_to_fp16 = const()[name = string("const_192_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084457728)))]; tensor var_7584_cast_fp16 = mul(x = var_7581_cast_fp16_0, y = const_192_to_fp16)[name = string("op_7584_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = x_319_cast_fp16, y = var_7584_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; tensor per_layer_slice_21_begin_0 = const()[name = string("per_layer_slice_21_begin_0"), val = tensor([0, 0, 2560])]; tensor per_layer_slice_21_end_0 = const()[name = string("per_layer_slice_21_end_0"), val = tensor([1, 1, 2816])]; tensor per_layer_slice_21_end_mask_0 = const()[name = string("per_layer_slice_21_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_21_cast_fp16 = slice_by_index(begin = per_layer_slice_21_begin_0, end = per_layer_slice_21_end_0, end_mask = per_layer_slice_21_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_21_cast_fp16")]; tensor gated_41 = linear(bias = linear_0_bias_0, weight = layers_10_per_layer_input_gate_weight_palettized, x = hidden_states_129_cast_fp16)[name = string("linear_20")]; string gated_43_mode_0 = const()[name = string("gated_43_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_43 = gelu(mode = gated_43_mode_0, x = gated_41)[name = string("gated_43")]; tensor input_319_cast_fp16 = mul(x = gated_43, y = per_layer_slice_21_cast_fp16)[name = string("input_319_cast_fp16")]; tensor layers_10_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084460864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084657536))))[name = string("layers_10_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_21_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_10_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_319_cast_fp16)[name = string("linear_21_cast_fp16")]; int32 var_7621 = const()[name = string("op_7621"), val = int32(-1)]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7627_cast_fp16 = mul(x = linear_21_cast_fp16, y = const_193_promoted_to_fp16)[name = string("op_7627_cast_fp16")]; bool input_321_interleave_0 = const()[name = string("input_321_interleave_0"), val = bool(false)]; tensor input_321_cast_fp16 = concat(axis = var_7621, interleave = input_321_interleave_0, values = (linear_21_cast_fp16, var_7627_cast_fp16))[name = string("input_321_cast_fp16")]; tensor normed_305_axes_0 = const()[name = string("normed_305_axes_0"), val = tensor([-1])]; fp16 var_7619_to_fp16 = const()[name = string("op_7619_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_305_cast_fp16 = layer_norm(axes = normed_305_axes_0, epsilon = var_7619_to_fp16, x = input_321_cast_fp16)[name = string("normed_305_cast_fp16")]; tensor var_7632_split_sizes_0 = const()[name = string("op_7632_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7632_axis_0 = const()[name = string("op_7632_axis_0"), val = int32(-1)]; tensor var_7632_cast_fp16_0, tensor var_7632_cast_fp16_1 = split(axis = var_7632_axis_0, split_sizes = var_7632_split_sizes_0, x = normed_305_cast_fp16)[name = string("op_7632_cast_fp16")]; tensor const_194_to_fp16 = const()[name = string("const_194_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084659136)))]; tensor var_7635_cast_fp16 = mul(x = var_7632_cast_fp16_0, y = const_194_to_fp16)[name = string("op_7635_cast_fp16")]; tensor hidden_states_133_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_7635_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor layers_10_layer_scalar_to_fp16 = const()[name = string("layers_10_layer_scalar_to_fp16"), val = tensor([0x1.c6p-2])]; tensor x_331_cast_fp16 = mul(x = hidden_states_133_cast_fp16, y = layers_10_layer_scalar_to_fp16)[name = string("x_331_cast_fp16")]; int32 var_7643 = const()[name = string("op_7643"), val = int32(-1)]; fp16 const_195_promoted_to_fp16 = const()[name = string("const_195_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7649_cast_fp16 = mul(x = x_331_cast_fp16, y = const_195_promoted_to_fp16)[name = string("op_7649_cast_fp16")]; bool input_323_interleave_0 = const()[name = string("input_323_interleave_0"), val = bool(false)]; tensor input_323_cast_fp16 = concat(axis = var_7643, interleave = input_323_interleave_0, values = (x_331_cast_fp16, var_7649_cast_fp16))[name = string("input_323_cast_fp16")]; tensor normed_309_axes_0 = const()[name = string("normed_309_axes_0"), val = tensor([-1])]; fp16 var_7641_to_fp16 = const()[name = string("op_7641_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_309_cast_fp16 = layer_norm(axes = normed_309_axes_0, epsilon = var_7641_to_fp16, x = input_323_cast_fp16)[name = string("normed_309_cast_fp16")]; tensor var_7654_split_sizes_0 = const()[name = string("op_7654_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_7654_axis_0 = const()[name = string("op_7654_axis_0"), val = int32(-1)]; tensor var_7654_cast_fp16_0, tensor var_7654_cast_fp16_1 = split(axis = var_7654_axis_0, split_sizes = var_7654_split_sizes_0, x = normed_309_cast_fp16)[name = string("op_7654_cast_fp16")]; tensor const_196_to_fp16 = const()[name = string("const_196_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084662272)))]; tensor var_7657_cast_fp16 = mul(x = var_7654_cast_fp16_0, y = const_196_to_fp16)[name = string("op_7657_cast_fp16")]; tensor var_7665 = const()[name = string("op_7665"), val = tensor([0, 2, 1])]; tensor var_7668_axes_0 = const()[name = string("op_7668_axes_0"), val = tensor([2])]; tensor var_7666_cast_fp16 = transpose(perm = var_7665, x = var_7657_cast_fp16)[name = string("transpose_201")]; tensor var_7668_cast_fp16 = expand_dims(axes = var_7668_axes_0, x = var_7666_cast_fp16)[name = string("op_7668_cast_fp16")]; string var_7684_pad_type_0 = const()[name = string("op_7684_pad_type_0"), val = string("valid")]; tensor var_7684_strides_0 = const()[name = string("op_7684_strides_0"), val = tensor([1, 1])]; tensor var_7684_pad_0 = const()[name = string("op_7684_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7684_dilations_0 = const()[name = string("op_7684_dilations_0"), val = tensor([1, 1])]; int32 var_7684_groups_0 = const()[name = string("op_7684_groups_0"), val = int32(1)]; tensor var_7684 = conv(dilations = var_7684_dilations_0, groups = var_7684_groups_0, pad = var_7684_pad_0, pad_type = var_7684_pad_type_0, strides = var_7684_strides_0, weight = layers_11_self_attn_q_proj_weight_palettized, x = var_7668_cast_fp16)[name = string("op_7684")]; tensor var_7689 = const()[name = string("op_7689"), val = tensor([1, 8, 256, 1])]; tensor var_7690 = reshape(shape = var_7689, x = var_7684)[name = string("op_7690")]; tensor var_7695 = const()[name = string("op_7695"), val = tensor([0, 1, 3, 2])]; tensor var_7705 = const()[name = string("op_7705"), val = tensor([1, 8, 256])]; tensor var_7696 = transpose(perm = var_7695, x = var_7690)[name = string("transpose_200")]; tensor x_335 = reshape(shape = var_7705, x = var_7696)[name = string("x_335")]; int32 var_7711 = const()[name = string("op_7711"), val = int32(-1)]; fp16 const_197_promoted_to_fp16 = const()[name = string("const_197_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7717_cast_fp16 = mul(x = x_335, y = const_197_promoted_to_fp16)[name = string("op_7717_cast_fp16")]; bool input_327_interleave_0 = const()[name = string("input_327_interleave_0"), val = bool(false)]; tensor input_327_cast_fp16 = concat(axis = var_7711, interleave = input_327_interleave_0, values = (x_335, var_7717_cast_fp16))[name = string("input_327_cast_fp16")]; tensor normed_313_axes_0 = const()[name = string("normed_313_axes_0"), val = tensor([-1])]; fp16 var_7709_to_fp16 = const()[name = string("op_7709_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_313_cast_fp16 = layer_norm(axes = normed_313_axes_0, epsilon = var_7709_to_fp16, x = input_327_cast_fp16)[name = string("normed_313_cast_fp16")]; tensor var_7722_split_sizes_0 = const()[name = string("op_7722_split_sizes_0"), val = tensor([256, 256])]; int32 var_7722_axis_0 = const()[name = string("op_7722_axis_0"), val = int32(-1)]; tensor var_7722_cast_fp16_0, tensor var_7722_cast_fp16_1 = split(axis = var_7722_axis_0, split_sizes = var_7722_split_sizes_0, x = normed_313_cast_fp16)[name = string("op_7722_cast_fp16")]; tensor var_7731 = const()[name = string("op_7731"), val = tensor([1, 8, 1, 256])]; tensor q_91 = reshape(shape = var_7731, x = var_7722_cast_fp16_0)[name = string("q_91")]; tensor var_7733 = mul(x = q_91, y = cos_1)[name = string("op_7733")]; tensor var_7734_split_sizes_0 = const()[name = string("op_7734_split_sizes_0"), val = tensor([128, 128])]; int32 var_7734_axis_0 = const()[name = string("op_7734_axis_0"), val = int32(-1)]; tensor var_7734_0, tensor var_7734_1 = split(axis = var_7734_axis_0, split_sizes = var_7734_split_sizes_0, x = q_91)[name = string("op_7734")]; fp16 const_199_promoted = const()[name = string("const_199_promoted"), val = fp16(-0x1p+0)]; tensor var_7736 = mul(x = var_7734_1, y = const_199_promoted)[name = string("op_7736")]; int32 var_7738 = const()[name = string("op_7738"), val = int32(-1)]; bool var_7739_interleave_0 = const()[name = string("op_7739_interleave_0"), val = bool(false)]; tensor var_7739 = concat(axis = var_7738, interleave = var_7739_interleave_0, values = (var_7736, var_7734_0))[name = string("op_7739")]; tensor var_7740 = mul(x = var_7739, y = sin_1)[name = string("op_7740")]; tensor q_95 = add(x = var_7733, y = var_7740)[name = string("q_95")]; string var_7753_pad_type_0 = const()[name = string("op_7753_pad_type_0"), val = string("valid")]; tensor var_7753_strides_0 = const()[name = string("op_7753_strides_0"), val = tensor([1, 1])]; tensor var_7753_pad_0 = const()[name = string("op_7753_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7753_dilations_0 = const()[name = string("op_7753_dilations_0"), val = tensor([1, 1])]; int32 var_7753_groups_0 = const()[name = string("op_7753_groups_0"), val = int32(1)]; tensor var_7753 = conv(dilations = var_7753_dilations_0, groups = var_7753_groups_0, pad = var_7753_pad_0, pad_type = var_7753_pad_type_0, strides = var_7753_strides_0, weight = layers_11_self_attn_k_proj_weight_palettized, x = var_7668_cast_fp16)[name = string("op_7753")]; tensor var_7758 = const()[name = string("op_7758"), val = tensor([1, 1, 256, 1])]; tensor var_7759 = reshape(shape = var_7758, x = var_7753)[name = string("op_7759")]; tensor var_7764 = const()[name = string("op_7764"), val = tensor([0, 1, 3, 2])]; string var_7781_pad_type_0 = const()[name = string("op_7781_pad_type_0"), val = string("valid")]; tensor var_7781_strides_0 = const()[name = string("op_7781_strides_0"), val = tensor([1, 1])]; tensor var_7781_pad_0 = const()[name = string("op_7781_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7781_dilations_0 = const()[name = string("op_7781_dilations_0"), val = tensor([1, 1])]; int32 var_7781_groups_0 = const()[name = string("op_7781_groups_0"), val = int32(1)]; tensor var_7781 = conv(dilations = var_7781_dilations_0, groups = var_7781_groups_0, pad = var_7781_pad_0, pad_type = var_7781_pad_type_0, strides = var_7781_strides_0, weight = layers_11_self_attn_v_proj_weight_palettized, x = var_7668_cast_fp16)[name = string("op_7781")]; tensor var_7786 = const()[name = string("op_7786"), val = tensor([1, 1, 256, 1])]; tensor var_7787 = reshape(shape = var_7786, x = var_7781)[name = string("op_7787")]; tensor var_7792 = const()[name = string("op_7792"), val = tensor([0, 1, 3, 2])]; tensor var_7802 = const()[name = string("op_7802"), val = tensor([1, 1, 256])]; tensor var_7765 = transpose(perm = var_7764, x = var_7759)[name = string("transpose_199")]; tensor x_339 = reshape(shape = var_7802, x = var_7765)[name = string("x_339")]; int32 var_7808 = const()[name = string("op_7808"), val = int32(-1)]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7814_cast_fp16 = mul(x = x_339, y = const_200_promoted_to_fp16)[name = string("op_7814_cast_fp16")]; bool input_329_interleave_0 = const()[name = string("input_329_interleave_0"), val = bool(false)]; tensor input_329_cast_fp16 = concat(axis = var_7808, interleave = input_329_interleave_0, values = (x_339, var_7814_cast_fp16))[name = string("input_329_cast_fp16")]; tensor normed_317_axes_0 = const()[name = string("normed_317_axes_0"), val = tensor([-1])]; fp16 var_7806_to_fp16 = const()[name = string("op_7806_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_317_cast_fp16 = layer_norm(axes = normed_317_axes_0, epsilon = var_7806_to_fp16, x = input_329_cast_fp16)[name = string("normed_317_cast_fp16")]; tensor var_7819_split_sizes_0 = const()[name = string("op_7819_split_sizes_0"), val = tensor([256, 256])]; int32 var_7819_axis_0 = const()[name = string("op_7819_axis_0"), val = int32(-1)]; tensor var_7819_cast_fp16_0, tensor var_7819_cast_fp16_1 = split(axis = var_7819_axis_0, split_sizes = var_7819_split_sizes_0, x = normed_317_cast_fp16)[name = string("op_7819_cast_fp16")]; tensor const_201_to_fp16 = const()[name = string("const_201_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084665408)))]; tensor var_7822_cast_fp16 = mul(x = var_7819_cast_fp16_0, y = const_201_to_fp16)[name = string("op_7822_cast_fp16")]; tensor var_7828 = const()[name = string("op_7828"), val = tensor([1, 1, 1, 256])]; tensor q_93 = reshape(shape = var_7828, x = var_7822_cast_fp16)[name = string("q_93")]; fp16 var_7835_promoted_to_fp16 = const()[name = string("op_7835_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_7793 = transpose(perm = var_7792, x = var_7787)[name = string("transpose_198")]; tensor var_7836_cast_fp16 = pow(x = var_7793, y = var_7835_promoted_to_fp16)[name = string("op_7836_cast_fp16")]; tensor var_7841_axes_0 = const()[name = string("op_7841_axes_0"), val = tensor([-1])]; bool var_7841_keep_dims_0 = const()[name = string("op_7841_keep_dims_0"), val = bool(true)]; tensor var_7841_cast_fp16 = reduce_mean(axes = var_7841_axes_0, keep_dims = var_7841_keep_dims_0, x = var_7836_cast_fp16)[name = string("op_7841_cast_fp16")]; fp16 var_7843_to_fp16 = const()[name = string("op_7843_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_23_cast_fp16 = add(x = var_7841_cast_fp16, y = var_7843_to_fp16)[name = string("mean_sq_23_cast_fp16")]; fp16 var_7850_to_fp16 = const()[name = string("op_7850_to_fp16"), val = fp16(-0x1p-1)]; tensor var_7851_cast_fp16 = pow(x = mean_sq_23_cast_fp16, y = var_7850_to_fp16)[name = string("op_7851_cast_fp16")]; tensor var_7852_cast_fp16 = mul(x = var_7793, y = var_7851_cast_fp16)[name = string("op_7852_cast_fp16")]; tensor var_7858 = mul(x = q_93, y = cos_1)[name = string("op_7858")]; tensor var_7859_split_sizes_0 = const()[name = string("op_7859_split_sizes_0"), val = tensor([128, 128])]; int32 var_7859_axis_0 = const()[name = string("op_7859_axis_0"), val = int32(-1)]; tensor var_7859_0, tensor var_7859_1 = split(axis = var_7859_axis_0, split_sizes = var_7859_split_sizes_0, x = q_93)[name = string("op_7859")]; fp16 const_202_promoted = const()[name = string("const_202_promoted"), val = fp16(-0x1p+0)]; tensor var_7861 = mul(x = var_7859_1, y = const_202_promoted)[name = string("op_7861")]; int32 var_7863 = const()[name = string("op_7863"), val = int32(-1)]; bool var_7864_interleave_0 = const()[name = string("op_7864_interleave_0"), val = bool(false)]; tensor var_7864 = concat(axis = var_7863, interleave = var_7864_interleave_0, values = (var_7861, var_7859_0))[name = string("op_7864")]; tensor var_7865 = mul(x = var_7864, y = sin_1)[name = string("op_7865")]; tensor input_331 = add(x = var_7858, y = var_7865)[name = string("input_331")]; tensor var_7870_begin_0 = const()[name = string("op_7870_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_7870_end_0 = const()[name = string("op_7870_end_0"), val = tensor([12, 1, 512, 512])]; tensor var_7870_end_mask_0 = const()[name = string("op_7870_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7870_squeeze_mask_0 = const()[name = string("op_7870_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_7870_cast_fp16 = slice_by_index(begin = var_7870_begin_0, end = var_7870_end_0, end_mask = var_7870_end_mask_0, squeeze_mask = var_7870_squeeze_mask_0, x = coreml_update_state_51)[name = string("op_7870_cast_fp16")]; tensor K_cache_23_axes_0 = const()[name = string("K_cache_23_axes_0"), val = tensor([0])]; tensor K_cache_23_cast_fp16 = expand_dims(axes = K_cache_23_axes_0, x = var_7870_cast_fp16)[name = string("K_cache_23_cast_fp16")]; tensor var_7875_begin_0 = const()[name = string("op_7875_begin_0"), val = tensor([46, 0, 0, 0])]; tensor var_7875_end_0 = const()[name = string("op_7875_end_0"), val = tensor([47, 1, 512, 512])]; tensor var_7875_end_mask_0 = const()[name = string("op_7875_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7875_squeeze_mask_0 = const()[name = string("op_7875_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_7875_cast_fp16 = slice_by_index(begin = var_7875_begin_0, end = var_7875_end_0, end_mask = var_7875_end_mask_0, squeeze_mask = var_7875_squeeze_mask_0, x = coreml_update_state_51)[name = string("op_7875_cast_fp16")]; tensor V_cache_23_axes_0 = const()[name = string("V_cache_23_axes_0"), val = tensor([0])]; tensor V_cache_23_cast_fp16 = expand_dims(axes = V_cache_23_axes_0, x = var_7875_cast_fp16)[name = string("V_cache_23_cast_fp16")]; tensor k_padded_19_pad_0 = const()[name = string("k_padded_19_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_19_mode_0 = const()[name = string("k_padded_19_mode_0"), val = string("constant")]; fp16 const_203_to_fp16 = const()[name = string("const_203_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_19_cast_fp16 = pad(constant_val = const_203_to_fp16, mode = k_padded_19_mode_0, pad = k_padded_19_pad_0, x = input_331)[name = string("k_padded_19_cast_fp16")]; tensor v_padded_19_pad_0 = const()[name = string("v_padded_19_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_19_mode_0 = const()[name = string("v_padded_19_mode_0"), val = string("constant")]; fp16 const_204_to_fp16 = const()[name = string("const_204_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_19_cast_fp16 = pad(constant_val = const_204_to_fp16, mode = v_padded_19_mode_0, pad = v_padded_19_pad_0, x = var_7852_cast_fp16)[name = string("v_padded_19_cast_fp16")]; tensor var_7893_cast_fp16 = mul(x = K_cache_23_cast_fp16, y = var_2187_cast_fp16)[name = string("op_7893_cast_fp16")]; tensor var_7894_reps_0 = const()[name = string("op_7894_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_7894_cast_fp16 = tile(reps = var_7894_reps_0, x = k_padded_19_cast_fp16)[name = string("op_7894_cast_fp16")]; tensor var_7895_cast_fp16 = mul(x = var_7894_cast_fp16, y = update_mask)[name = string("op_7895_cast_fp16")]; tensor K_new_23_cast_fp16 = add(x = var_7893_cast_fp16, y = var_7895_cast_fp16)[name = string("K_new_23_cast_fp16")]; tensor var_7901_cast_fp16 = mul(x = V_cache_23_cast_fp16, y = var_2187_cast_fp16)[name = string("op_7901_cast_fp16")]; tensor var_7902_reps_0 = const()[name = string("op_7902_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_7902_cast_fp16 = tile(reps = var_7902_reps_0, x = v_padded_19_cast_fp16)[name = string("op_7902_cast_fp16")]; tensor var_7903_cast_fp16 = mul(x = var_7902_cast_fp16, y = update_mask)[name = string("op_7903_cast_fp16")]; tensor V_new_23_cast_fp16 = add(x = var_7901_cast_fp16, y = var_7903_cast_fp16)[name = string("V_new_23_cast_fp16")]; tensor var_7907_axes_0 = const()[name = string("op_7907_axes_0"), val = tensor([0])]; tensor var_7907_cast_fp16 = squeeze(axes = var_7907_axes_0, x = K_new_23_cast_fp16)[name = string("op_7907_cast_fp16")]; tensor concat_88 = const()[name = string("concat_88"), val = tensor([11, 0, 0, 0])]; tensor concat_89 = const()[name = string("concat_89"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_88, begin_mask = kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_89, end_mask = kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_23_stride_0, update = var_7907_cast_fp16, x = coreml_update_state_51)[name = string("kv_cache_0_internal_tensor_assign_23_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_23_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_52_write_state")]; tensor coreml_update_state_52 = read_state(input = kv_cache_0)[name = string("coreml_update_state_52")]; tensor var_7914_axes_0 = const()[name = string("op_7914_axes_0"), val = tensor([0])]; tensor var_7914_cast_fp16 = squeeze(axes = var_7914_axes_0, x = V_new_23_cast_fp16)[name = string("op_7914_cast_fp16")]; tensor concat_90 = const()[name = string("concat_90"), val = tensor([46, 0, 0, 0])]; tensor concat_91 = const()[name = string("concat_91"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_90, begin_mask = kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_91, end_mask = kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_24_stride_0, update = var_7914_cast_fp16, x = coreml_update_state_52)[name = string("kv_cache_0_internal_tensor_assign_24_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_24_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_53_write_state")]; tensor coreml_update_state_53 = read_state(input = kv_cache_0)[name = string("coreml_update_state_53")]; tensor K_for_attn_23_begin_0 = const()[name = string("K_for_attn_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_23_end_0 = const()[name = string("K_for_attn_23_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_23_end_mask_0 = const()[name = string("K_for_attn_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_23_cast_fp16 = slice_by_index(begin = K_for_attn_23_begin_0, end = K_for_attn_23_end_0, end_mask = K_for_attn_23_end_mask_0, x = K_new_23_cast_fp16)[name = string("K_for_attn_23_cast_fp16")]; tensor V_for_attn_23_begin_0 = const()[name = string("V_for_attn_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_23_end_0 = const()[name = string("V_for_attn_23_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_23_end_mask_0 = const()[name = string("V_for_attn_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_23_cast_fp16 = slice_by_index(begin = V_for_attn_23_begin_0, end = V_for_attn_23_end_0, end_mask = V_for_attn_23_end_mask_0, x = V_new_23_cast_fp16)[name = string("V_for_attn_23_cast_fp16")]; tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_22_reps_0 = const()[name = string("tile_22_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = K_for_attn_23_cast_fp16)[name = string("transpose_197")]; tensor tile_22_cast_fp16 = tile(reps = tile_22_reps_0, x = transpose_44_cast_fp16)[name = string("tile_22_cast_fp16")]; tensor concat_92 = const()[name = string("concat_92"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_44_cast_fp16 = reshape(shape = concat_92, x = tile_22_cast_fp16)[name = string("reshape_44_cast_fp16")]; tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_93 = const()[name = string("concat_93"), val = tensor([-1, 1, 512, 256])]; tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_196")]; tensor reshape_45_cast_fp16 = reshape(shape = concat_93, x = transpose_45_cast_fp16)[name = string("reshape_45_cast_fp16")]; tensor transpose_151_perm_0 = const()[name = string("transpose_151_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_46_perm_0 = const()[name = string("transpose_46_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_23_reps_0 = const()[name = string("tile_23_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_46_cast_fp16 = transpose(perm = transpose_46_perm_0, x = V_for_attn_23_cast_fp16)[name = string("transpose_195")]; tensor tile_23_cast_fp16 = tile(reps = tile_23_reps_0, x = transpose_46_cast_fp16)[name = string("tile_23_cast_fp16")]; tensor concat_94 = const()[name = string("concat_94"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_46_cast_fp16 = reshape(shape = concat_94, x = tile_23_cast_fp16)[name = string("reshape_46_cast_fp16")]; tensor transpose_47_perm_0 = const()[name = string("transpose_47_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_95 = const()[name = string("concat_95"), val = tensor([-1, 1, 512, 256])]; tensor transpose_47_cast_fp16 = transpose(perm = transpose_47_perm_0, x = reshape_46_cast_fp16)[name = string("transpose_194")]; tensor reshape_47_cast_fp16 = reshape(shape = concat_95, x = transpose_47_cast_fp16)[name = string("reshape_47_cast_fp16")]; tensor V_expanded_23_perm_0 = const()[name = string("V_expanded_23_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_7941_transpose_x_0 = const()[name = string("op_7941_transpose_x_0"), val = bool(false)]; bool var_7941_transpose_y_0 = const()[name = string("op_7941_transpose_y_0"), val = bool(false)]; tensor transpose_151_cast_fp16 = transpose(perm = transpose_151_perm_0, x = reshape_45_cast_fp16)[name = string("transpose_193")]; tensor var_7941_cast_fp16 = matmul(transpose_x = var_7941_transpose_x_0, transpose_y = var_7941_transpose_y_0, x = q_95, y = transpose_151_cast_fp16)[name = string("op_7941_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = var_7941_cast_fp16, y = causal_mask)[name = string("attn_weights_69_cast_fp16")]; int32 var_7946 = const()[name = string("op_7946"), val = int32(-1)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_7946, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool attn_output_67_transpose_x_0 = const()[name = string("attn_output_67_transpose_x_0"), val = bool(false)]; bool attn_output_67_transpose_y_0 = const()[name = string("attn_output_67_transpose_y_0"), val = bool(false)]; tensor V_expanded_23_cast_fp16 = transpose(perm = V_expanded_23_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_192")]; tensor attn_output_67_cast_fp16 = matmul(transpose_x = attn_output_67_transpose_x_0, transpose_y = attn_output_67_transpose_y_0, x = attn_weights_71_cast_fp16, y = V_expanded_23_cast_fp16)[name = string("attn_output_67_cast_fp16")]; tensor var_7954 = const()[name = string("op_7954"), val = tensor([0, 2, 1, 3])]; tensor var_7961 = const()[name = string("op_7961"), val = tensor([1, 1, -1])]; tensor var_7955_cast_fp16 = transpose(perm = var_7954, x = attn_output_67_cast_fp16)[name = string("transpose_191")]; tensor attn_output_69_cast_fp16 = reshape(shape = var_7961, x = var_7955_cast_fp16)[name = string("attn_output_69_cast_fp16")]; tensor var_7966 = const()[name = string("op_7966"), val = tensor([0, 2, 1])]; string var_7982_pad_type_0 = const()[name = string("op_7982_pad_type_0"), val = string("valid")]; int32 var_7982_groups_0 = const()[name = string("op_7982_groups_0"), val = int32(1)]; tensor var_7982_strides_0 = const()[name = string("op_7982_strides_0"), val = tensor([1])]; tensor var_7982_pad_0 = const()[name = string("op_7982_pad_0"), val = tensor([0, 0])]; tensor var_7982_dilations_0 = const()[name = string("op_7982_dilations_0"), val = tensor([1])]; tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1084665984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086238912))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7967_cast_fp16 = transpose(perm = var_7966, x = attn_output_69_cast_fp16)[name = string("transpose_190")]; tensor var_7982_cast_fp16 = conv(dilations = var_7982_dilations_0, groups = var_7982_groups_0, pad = var_7982_pad_0, pad_type = var_7982_pad_type_0, strides = var_7982_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_7967_cast_fp16)[name = string("op_7982_cast_fp16")]; tensor var_7986 = const()[name = string("op_7986"), val = tensor([0, 2, 1])]; int32 var_7992 = const()[name = string("op_7992"), val = int32(-1)]; fp16 const_205_promoted_to_fp16 = const()[name = string("const_205_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_345_cast_fp16 = transpose(perm = var_7986, x = var_7982_cast_fp16)[name = string("transpose_189")]; tensor var_7998_cast_fp16 = mul(x = x_345_cast_fp16, y = const_205_promoted_to_fp16)[name = string("op_7998_cast_fp16")]; bool input_337_interleave_0 = const()[name = string("input_337_interleave_0"), val = bool(false)]; tensor input_337_cast_fp16 = concat(axis = var_7992, interleave = input_337_interleave_0, values = (x_345_cast_fp16, var_7998_cast_fp16))[name = string("input_337_cast_fp16")]; tensor normed_321_axes_0 = const()[name = string("normed_321_axes_0"), val = tensor([-1])]; fp16 var_7990_to_fp16 = const()[name = string("op_7990_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_321_cast_fp16 = layer_norm(axes = normed_321_axes_0, epsilon = var_7990_to_fp16, x = input_337_cast_fp16)[name = string("normed_321_cast_fp16")]; tensor var_8003_split_sizes_0 = const()[name = string("op_8003_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8003_axis_0 = const()[name = string("op_8003_axis_0"), val = int32(-1)]; tensor var_8003_cast_fp16_0, tensor var_8003_cast_fp16_1 = split(axis = var_8003_axis_0, split_sizes = var_8003_split_sizes_0, x = normed_321_cast_fp16)[name = string("op_8003_cast_fp16")]; tensor const_206_to_fp16 = const()[name = string("const_206_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086240512)))]; tensor var_8006_cast_fp16 = mul(x = var_8003_cast_fp16_0, y = const_206_to_fp16)[name = string("op_8006_cast_fp16")]; tensor x_349_cast_fp16 = add(x = x_331_cast_fp16, y = var_8006_cast_fp16)[name = string("x_349_cast_fp16")]; int32 var_8013 = const()[name = string("op_8013"), val = int32(-1)]; fp16 const_207_promoted_to_fp16 = const()[name = string("const_207_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8019_cast_fp16 = mul(x = x_349_cast_fp16, y = const_207_promoted_to_fp16)[name = string("op_8019_cast_fp16")]; bool input_339_interleave_0 = const()[name = string("input_339_interleave_0"), val = bool(false)]; tensor input_339_cast_fp16 = concat(axis = var_8013, interleave = input_339_interleave_0, values = (x_349_cast_fp16, var_8019_cast_fp16))[name = string("input_339_cast_fp16")]; tensor normed_325_axes_0 = const()[name = string("normed_325_axes_0"), val = tensor([-1])]; fp16 var_8011_to_fp16 = const()[name = string("op_8011_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_325_cast_fp16 = layer_norm(axes = normed_325_axes_0, epsilon = var_8011_to_fp16, x = input_339_cast_fp16)[name = string("normed_325_cast_fp16")]; tensor var_8024_split_sizes_0 = const()[name = string("op_8024_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8024_axis_0 = const()[name = string("op_8024_axis_0"), val = int32(-1)]; tensor var_8024_cast_fp16_0, tensor var_8024_cast_fp16_1 = split(axis = var_8024_axis_0, split_sizes = var_8024_split_sizes_0, x = normed_325_cast_fp16)[name = string("op_8024_cast_fp16")]; tensor const_208_to_fp16 = const()[name = string("const_208_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086243648)))]; tensor var_8027_cast_fp16 = mul(x = var_8024_cast_fp16_0, y = const_208_to_fp16)[name = string("op_8027_cast_fp16")]; tensor var_8040 = const()[name = string("op_8040"), val = tensor([0, 2, 1])]; tensor input_341_axes_0 = const()[name = string("input_341_axes_0"), val = tensor([2])]; tensor var_8041 = transpose(perm = var_8040, x = var_8027_cast_fp16)[name = string("transpose_188")]; tensor input_341 = expand_dims(axes = input_341_axes_0, x = var_8041)[name = string("input_341")]; string gate_45_pad_type_0 = const()[name = string("gate_45_pad_type_0"), val = string("valid")]; tensor gate_45_strides_0 = const()[name = string("gate_45_strides_0"), val = tensor([1, 1])]; tensor gate_45_pad_0 = const()[name = string("gate_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_45_dilations_0 = const()[name = string("gate_45_dilations_0"), val = tensor([1, 1])]; int32 gate_45_groups_0 = const()[name = string("gate_45_groups_0"), val = int32(1)]; tensor gate_45 = conv(dilations = gate_45_dilations_0, groups = gate_45_groups_0, pad = gate_45_pad_0, pad_type = gate_45_pad_type_0, strides = gate_45_strides_0, weight = layers_11_mlp_gate_proj_weight_palettized, x = input_341)[name = string("gate_45")]; string up_23_pad_type_0 = const()[name = string("up_23_pad_type_0"), val = string("valid")]; tensor up_23_strides_0 = const()[name = string("up_23_strides_0"), val = tensor([1, 1])]; tensor up_23_pad_0 = const()[name = string("up_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_23_dilations_0 = const()[name = string("up_23_dilations_0"), val = tensor([1, 1])]; int32 up_23_groups_0 = const()[name = string("up_23_groups_0"), val = int32(1)]; tensor up_23 = conv(dilations = up_23_dilations_0, groups = up_23_groups_0, pad = up_23_pad_0, pad_type = up_23_pad_type_0, strides = up_23_strides_0, weight = layers_11_mlp_up_proj_weight_palettized, x = input_341)[name = string("up_23")]; string gate_47_mode_0 = const()[name = string("gate_47_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_47 = gelu(mode = gate_47_mode_0, x = gate_45)[name = string("gate_47")]; tensor input_343 = mul(x = gate_47, y = up_23)[name = string("input_343")]; string mlp_out_23_pad_type_0 = const()[name = string("mlp_out_23_pad_type_0"), val = string("valid")]; tensor mlp_out_23_strides_0 = const()[name = string("mlp_out_23_strides_0"), val = tensor([1, 1])]; tensor mlp_out_23_pad_0 = const()[name = string("mlp_out_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_23_dilations_0 = const()[name = string("mlp_out_23_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_23_groups_0 = const()[name = string("mlp_out_23_groups_0"), val = int32(1)]; tensor mlp_out_23 = conv(dilations = mlp_out_23_dilations_0, groups = mlp_out_23_groups_0, pad = mlp_out_23_pad_0, pad_type = mlp_out_23_pad_type_0, strides = mlp_out_23_strides_0, weight = layers_11_mlp_down_proj_weight_palettized, x = input_343)[name = string("mlp_out_23")]; tensor var_8081_axes_0 = const()[name = string("op_8081_axes_0"), val = tensor([2])]; tensor var_8081 = squeeze(axes = var_8081_axes_0, x = mlp_out_23)[name = string("op_8081")]; tensor var_8085 = const()[name = string("op_8085"), val = tensor([0, 2, 1])]; int32 var_8091 = const()[name = string("op_8091"), val = int32(-1)]; fp16 const_209_promoted_to_fp16 = const()[name = string("const_209_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_353 = transpose(perm = var_8085, x = var_8081)[name = string("transpose_187")]; tensor var_8097_cast_fp16 = mul(x = x_353, y = const_209_promoted_to_fp16)[name = string("op_8097_cast_fp16")]; bool input_345_interleave_0 = const()[name = string("input_345_interleave_0"), val = bool(false)]; tensor input_345_cast_fp16 = concat(axis = var_8091, interleave = input_345_interleave_0, values = (x_353, var_8097_cast_fp16))[name = string("input_345_cast_fp16")]; tensor normed_329_axes_0 = const()[name = string("normed_329_axes_0"), val = tensor([-1])]; fp16 var_8089_to_fp16 = const()[name = string("op_8089_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_329_cast_fp16 = layer_norm(axes = normed_329_axes_0, epsilon = var_8089_to_fp16, x = input_345_cast_fp16)[name = string("normed_329_cast_fp16")]; tensor var_8102_split_sizes_0 = const()[name = string("op_8102_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8102_axis_0 = const()[name = string("op_8102_axis_0"), val = int32(-1)]; tensor var_8102_cast_fp16_0, tensor var_8102_cast_fp16_1 = split(axis = var_8102_axis_0, split_sizes = var_8102_split_sizes_0, x = normed_329_cast_fp16)[name = string("op_8102_cast_fp16")]; tensor const_210_to_fp16 = const()[name = string("const_210_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086246784)))]; tensor var_8105_cast_fp16 = mul(x = var_8102_cast_fp16_0, y = const_210_to_fp16)[name = string("op_8105_cast_fp16")]; tensor hidden_states_141_cast_fp16 = add(x = x_349_cast_fp16, y = var_8105_cast_fp16)[name = string("hidden_states_141_cast_fp16")]; tensor per_layer_slice_23_begin_0 = const()[name = string("per_layer_slice_23_begin_0"), val = tensor([0, 0, 2816])]; tensor per_layer_slice_23_end_0 = const()[name = string("per_layer_slice_23_end_0"), val = tensor([1, 1, 3072])]; tensor per_layer_slice_23_end_mask_0 = const()[name = string("per_layer_slice_23_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_23_cast_fp16 = slice_by_index(begin = per_layer_slice_23_begin_0, end = per_layer_slice_23_end_0, end_mask = per_layer_slice_23_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_23_cast_fp16")]; tensor gated_45 = linear(bias = linear_0_bias_0, weight = layers_11_per_layer_input_gate_weight_palettized, x = hidden_states_141_cast_fp16)[name = string("linear_22")]; string gated_47_mode_0 = const()[name = string("gated_47_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_47 = gelu(mode = gated_47_mode_0, x = gated_45)[name = string("gated_47")]; tensor input_349_cast_fp16 = mul(x = gated_47, y = per_layer_slice_23_cast_fp16)[name = string("input_349_cast_fp16")]; tensor layers_11_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086249920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086446592))))[name = string("layers_11_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_23_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_11_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_349_cast_fp16)[name = string("linear_23_cast_fp16")]; int32 var_8142 = const()[name = string("op_8142"), val = int32(-1)]; fp16 const_211_promoted_to_fp16 = const()[name = string("const_211_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8148_cast_fp16 = mul(x = linear_23_cast_fp16, y = const_211_promoted_to_fp16)[name = string("op_8148_cast_fp16")]; bool input_351_interleave_0 = const()[name = string("input_351_interleave_0"), val = bool(false)]; tensor input_351_cast_fp16 = concat(axis = var_8142, interleave = input_351_interleave_0, values = (linear_23_cast_fp16, var_8148_cast_fp16))[name = string("input_351_cast_fp16")]; tensor normed_333_axes_0 = const()[name = string("normed_333_axes_0"), val = tensor([-1])]; fp16 var_8140_to_fp16 = const()[name = string("op_8140_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_333_cast_fp16 = layer_norm(axes = normed_333_axes_0, epsilon = var_8140_to_fp16, x = input_351_cast_fp16)[name = string("normed_333_cast_fp16")]; tensor var_8153_split_sizes_0 = const()[name = string("op_8153_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8153_axis_0 = const()[name = string("op_8153_axis_0"), val = int32(-1)]; tensor var_8153_cast_fp16_0, tensor var_8153_cast_fp16_1 = split(axis = var_8153_axis_0, split_sizes = var_8153_split_sizes_0, x = normed_333_cast_fp16)[name = string("op_8153_cast_fp16")]; tensor const_212_to_fp16 = const()[name = string("const_212_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086448192)))]; tensor var_8156_cast_fp16 = mul(x = var_8153_cast_fp16_0, y = const_212_to_fp16)[name = string("op_8156_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_141_cast_fp16, y = var_8156_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; tensor layers_11_layer_scalar_to_fp16 = const()[name = string("layers_11_layer_scalar_to_fp16"), val = tensor([0x1.7ap-2])]; tensor x_361_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = layers_11_layer_scalar_to_fp16)[name = string("x_361_cast_fp16")]; int32 var_8164 = const()[name = string("op_8164"), val = int32(-1)]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8170_cast_fp16 = mul(x = x_361_cast_fp16, y = const_213_promoted_to_fp16)[name = string("op_8170_cast_fp16")]; bool input_353_interleave_0 = const()[name = string("input_353_interleave_0"), val = bool(false)]; tensor input_353_cast_fp16 = concat(axis = var_8164, interleave = input_353_interleave_0, values = (x_361_cast_fp16, var_8170_cast_fp16))[name = string("input_353_cast_fp16")]; tensor normed_337_axes_0 = const()[name = string("normed_337_axes_0"), val = tensor([-1])]; fp16 var_8162_to_fp16 = const()[name = string("op_8162_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_337_cast_fp16 = layer_norm(axes = normed_337_axes_0, epsilon = var_8162_to_fp16, x = input_353_cast_fp16)[name = string("normed_337_cast_fp16")]; tensor var_8175_split_sizes_0 = const()[name = string("op_8175_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8175_axis_0 = const()[name = string("op_8175_axis_0"), val = int32(-1)]; tensor var_8175_cast_fp16_0, tensor var_8175_cast_fp16_1 = split(axis = var_8175_axis_0, split_sizes = var_8175_split_sizes_0, x = normed_337_cast_fp16)[name = string("op_8175_cast_fp16")]; tensor const_214_to_fp16 = const()[name = string("const_214_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086451328)))]; tensor var_8178_cast_fp16 = mul(x = var_8175_cast_fp16_0, y = const_214_to_fp16)[name = string("op_8178_cast_fp16")]; tensor var_8186 = const()[name = string("op_8186"), val = tensor([0, 2, 1])]; tensor var_8189_axes_0 = const()[name = string("op_8189_axes_0"), val = tensor([2])]; tensor var_8187_cast_fp16 = transpose(perm = var_8186, x = var_8178_cast_fp16)[name = string("transpose_186")]; tensor var_8189_cast_fp16 = expand_dims(axes = var_8189_axes_0, x = var_8187_cast_fp16)[name = string("op_8189_cast_fp16")]; string var_8205_pad_type_0 = const()[name = string("op_8205_pad_type_0"), val = string("valid")]; tensor var_8205_strides_0 = const()[name = string("op_8205_strides_0"), val = tensor([1, 1])]; tensor var_8205_pad_0 = const()[name = string("op_8205_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8205_dilations_0 = const()[name = string("op_8205_dilations_0"), val = tensor([1, 1])]; int32 var_8205_groups_0 = const()[name = string("op_8205_groups_0"), val = int32(1)]; tensor var_8205 = conv(dilations = var_8205_dilations_0, groups = var_8205_groups_0, pad = var_8205_pad_0, pad_type = var_8205_pad_type_0, strides = var_8205_strides_0, weight = layers_12_self_attn_q_proj_weight_palettized, x = var_8189_cast_fp16)[name = string("op_8205")]; tensor var_8210 = const()[name = string("op_8210"), val = tensor([1, 8, 256, 1])]; tensor var_8211 = reshape(shape = var_8210, x = var_8205)[name = string("op_8211")]; tensor var_8216 = const()[name = string("op_8216"), val = tensor([0, 1, 3, 2])]; tensor var_8226 = const()[name = string("op_8226"), val = tensor([1, 8, 256])]; tensor var_8217 = transpose(perm = var_8216, x = var_8211)[name = string("transpose_185")]; tensor x_365 = reshape(shape = var_8226, x = var_8217)[name = string("x_365")]; int32 var_8232 = const()[name = string("op_8232"), val = int32(-1)]; fp16 const_215_promoted_to_fp16 = const()[name = string("const_215_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8238_cast_fp16 = mul(x = x_365, y = const_215_promoted_to_fp16)[name = string("op_8238_cast_fp16")]; bool input_357_interleave_0 = const()[name = string("input_357_interleave_0"), val = bool(false)]; tensor input_357_cast_fp16 = concat(axis = var_8232, interleave = input_357_interleave_0, values = (x_365, var_8238_cast_fp16))[name = string("input_357_cast_fp16")]; tensor normed_341_axes_0 = const()[name = string("normed_341_axes_0"), val = tensor([-1])]; fp16 var_8230_to_fp16 = const()[name = string("op_8230_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_341_cast_fp16 = layer_norm(axes = normed_341_axes_0, epsilon = var_8230_to_fp16, x = input_357_cast_fp16)[name = string("normed_341_cast_fp16")]; tensor var_8243_split_sizes_0 = const()[name = string("op_8243_split_sizes_0"), val = tensor([256, 256])]; int32 var_8243_axis_0 = const()[name = string("op_8243_axis_0"), val = int32(-1)]; tensor var_8243_cast_fp16_0, tensor var_8243_cast_fp16_1 = split(axis = var_8243_axis_0, split_sizes = var_8243_split_sizes_0, x = normed_341_cast_fp16)[name = string("op_8243_cast_fp16")]; tensor var_8246_cast_fp16 = mul(x = var_8243_cast_fp16_0, y = const_22_to_fp16)[name = string("op_8246_cast_fp16")]; tensor var_8252 = const()[name = string("op_8252"), val = tensor([1, 8, 1, 256])]; tensor q_99 = reshape(shape = var_8252, x = var_8246_cast_fp16)[name = string("q_99")]; tensor var_8254 = mul(x = q_99, y = cos_1)[name = string("op_8254")]; tensor var_8255_split_sizes_0 = const()[name = string("op_8255_split_sizes_0"), val = tensor([128, 128])]; int32 var_8255_axis_0 = const()[name = string("op_8255_axis_0"), val = int32(-1)]; tensor var_8255_0, tensor var_8255_1 = split(axis = var_8255_axis_0, split_sizes = var_8255_split_sizes_0, x = q_99)[name = string("op_8255")]; fp16 const_217_promoted = const()[name = string("const_217_promoted"), val = fp16(-0x1p+0)]; tensor var_8257 = mul(x = var_8255_1, y = const_217_promoted)[name = string("op_8257")]; int32 var_8259 = const()[name = string("op_8259"), val = int32(-1)]; bool var_8260_interleave_0 = const()[name = string("op_8260_interleave_0"), val = bool(false)]; tensor var_8260 = concat(axis = var_8259, interleave = var_8260_interleave_0, values = (var_8257, var_8255_0))[name = string("op_8260")]; tensor var_8261 = mul(x = var_8260, y = sin_1)[name = string("op_8261")]; tensor q_103 = add(x = var_8254, y = var_8261)[name = string("q_103")]; string var_8274_pad_type_0 = const()[name = string("op_8274_pad_type_0"), val = string("valid")]; tensor var_8274_strides_0 = const()[name = string("op_8274_strides_0"), val = tensor([1, 1])]; tensor var_8274_pad_0 = const()[name = string("op_8274_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8274_dilations_0 = const()[name = string("op_8274_dilations_0"), val = tensor([1, 1])]; int32 var_8274_groups_0 = const()[name = string("op_8274_groups_0"), val = int32(1)]; tensor var_8274 = conv(dilations = var_8274_dilations_0, groups = var_8274_groups_0, pad = var_8274_pad_0, pad_type = var_8274_pad_type_0, strides = var_8274_strides_0, weight = layers_12_self_attn_k_proj_weight_palettized, x = var_8189_cast_fp16)[name = string("op_8274")]; tensor var_8279 = const()[name = string("op_8279"), val = tensor([1, 1, 256, 1])]; tensor var_8280 = reshape(shape = var_8279, x = var_8274)[name = string("op_8280")]; tensor var_8285 = const()[name = string("op_8285"), val = tensor([0, 1, 3, 2])]; string var_8302_pad_type_0 = const()[name = string("op_8302_pad_type_0"), val = string("valid")]; tensor var_8302_strides_0 = const()[name = string("op_8302_strides_0"), val = tensor([1, 1])]; tensor var_8302_pad_0 = const()[name = string("op_8302_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8302_dilations_0 = const()[name = string("op_8302_dilations_0"), val = tensor([1, 1])]; int32 var_8302_groups_0 = const()[name = string("op_8302_groups_0"), val = int32(1)]; tensor var_8302 = conv(dilations = var_8302_dilations_0, groups = var_8302_groups_0, pad = var_8302_pad_0, pad_type = var_8302_pad_type_0, strides = var_8302_strides_0, weight = layers_12_self_attn_v_proj_weight_palettized, x = var_8189_cast_fp16)[name = string("op_8302")]; tensor var_8307 = const()[name = string("op_8307"), val = tensor([1, 1, 256, 1])]; tensor var_8308 = reshape(shape = var_8307, x = var_8302)[name = string("op_8308")]; tensor var_8313 = const()[name = string("op_8313"), val = tensor([0, 1, 3, 2])]; tensor var_8323 = const()[name = string("op_8323"), val = tensor([1, 1, 256])]; tensor var_8286 = transpose(perm = var_8285, x = var_8280)[name = string("transpose_184")]; tensor x_369 = reshape(shape = var_8323, x = var_8286)[name = string("x_369")]; int32 var_8329 = const()[name = string("op_8329"), val = int32(-1)]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8335_cast_fp16 = mul(x = x_369, y = const_218_promoted_to_fp16)[name = string("op_8335_cast_fp16")]; bool input_359_interleave_0 = const()[name = string("input_359_interleave_0"), val = bool(false)]; tensor input_359_cast_fp16 = concat(axis = var_8329, interleave = input_359_interleave_0, values = (x_369, var_8335_cast_fp16))[name = string("input_359_cast_fp16")]; tensor normed_345_axes_0 = const()[name = string("normed_345_axes_0"), val = tensor([-1])]; fp16 var_8327_to_fp16 = const()[name = string("op_8327_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_345_cast_fp16 = layer_norm(axes = normed_345_axes_0, epsilon = var_8327_to_fp16, x = input_359_cast_fp16)[name = string("normed_345_cast_fp16")]; tensor var_8340_split_sizes_0 = const()[name = string("op_8340_split_sizes_0"), val = tensor([256, 256])]; int32 var_8340_axis_0 = const()[name = string("op_8340_axis_0"), val = int32(-1)]; tensor var_8340_cast_fp16_0, tensor var_8340_cast_fp16_1 = split(axis = var_8340_axis_0, split_sizes = var_8340_split_sizes_0, x = normed_345_cast_fp16)[name = string("op_8340_cast_fp16")]; tensor const_219_to_fp16 = const()[name = string("const_219_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086454464)))]; tensor var_8343_cast_fp16 = mul(x = var_8340_cast_fp16_0, y = const_219_to_fp16)[name = string("op_8343_cast_fp16")]; tensor var_8349 = const()[name = string("op_8349"), val = tensor([1, 1, 1, 256])]; tensor q_101 = reshape(shape = var_8349, x = var_8343_cast_fp16)[name = string("q_101")]; fp16 var_8356_promoted_to_fp16 = const()[name = string("op_8356_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_8314 = transpose(perm = var_8313, x = var_8308)[name = string("transpose_183")]; tensor var_8357_cast_fp16 = pow(x = var_8314, y = var_8356_promoted_to_fp16)[name = string("op_8357_cast_fp16")]; tensor var_8362_axes_0 = const()[name = string("op_8362_axes_0"), val = tensor([-1])]; bool var_8362_keep_dims_0 = const()[name = string("op_8362_keep_dims_0"), val = bool(true)]; tensor var_8362_cast_fp16 = reduce_mean(axes = var_8362_axes_0, keep_dims = var_8362_keep_dims_0, x = var_8357_cast_fp16)[name = string("op_8362_cast_fp16")]; fp16 var_8364_to_fp16 = const()[name = string("op_8364_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_25_cast_fp16 = add(x = var_8362_cast_fp16, y = var_8364_to_fp16)[name = string("mean_sq_25_cast_fp16")]; fp16 var_8371_to_fp16 = const()[name = string("op_8371_to_fp16"), val = fp16(-0x1p-1)]; tensor var_8372_cast_fp16 = pow(x = mean_sq_25_cast_fp16, y = var_8371_to_fp16)[name = string("op_8372_cast_fp16")]; tensor var_8373_cast_fp16 = mul(x = var_8314, y = var_8372_cast_fp16)[name = string("op_8373_cast_fp16")]; tensor var_8379 = mul(x = q_101, y = cos_1)[name = string("op_8379")]; tensor var_8380_split_sizes_0 = const()[name = string("op_8380_split_sizes_0"), val = tensor([128, 128])]; int32 var_8380_axis_0 = const()[name = string("op_8380_axis_0"), val = int32(-1)]; tensor var_8380_0, tensor var_8380_1 = split(axis = var_8380_axis_0, split_sizes = var_8380_split_sizes_0, x = q_101)[name = string("op_8380")]; fp16 const_220_promoted = const()[name = string("const_220_promoted"), val = fp16(-0x1p+0)]; tensor var_8382 = mul(x = var_8380_1, y = const_220_promoted)[name = string("op_8382")]; int32 var_8384 = const()[name = string("op_8384"), val = int32(-1)]; bool var_8385_interleave_0 = const()[name = string("op_8385_interleave_0"), val = bool(false)]; tensor var_8385 = concat(axis = var_8384, interleave = var_8385_interleave_0, values = (var_8382, var_8380_0))[name = string("op_8385")]; tensor var_8386 = mul(x = var_8385, y = sin_1)[name = string("op_8386")]; tensor input_361 = add(x = var_8379, y = var_8386)[name = string("input_361")]; tensor var_8391_begin_0 = const()[name = string("op_8391_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_8391_end_0 = const()[name = string("op_8391_end_0"), val = tensor([13, 1, 512, 512])]; tensor var_8391_end_mask_0 = const()[name = string("op_8391_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8391_squeeze_mask_0 = const()[name = string("op_8391_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_8391_cast_fp16 = slice_by_index(begin = var_8391_begin_0, end = var_8391_end_0, end_mask = var_8391_end_mask_0, squeeze_mask = var_8391_squeeze_mask_0, x = coreml_update_state_53)[name = string("op_8391_cast_fp16")]; tensor K_cache_25_axes_0 = const()[name = string("K_cache_25_axes_0"), val = tensor([0])]; tensor K_cache_25_cast_fp16 = expand_dims(axes = K_cache_25_axes_0, x = var_8391_cast_fp16)[name = string("K_cache_25_cast_fp16")]; tensor var_8396_begin_0 = const()[name = string("op_8396_begin_0"), val = tensor([47, 0, 0, 0])]; tensor var_8396_end_0 = const()[name = string("op_8396_end_0"), val = tensor([48, 1, 512, 512])]; tensor var_8396_end_mask_0 = const()[name = string("op_8396_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8396_squeeze_mask_0 = const()[name = string("op_8396_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_8396_cast_fp16 = slice_by_index(begin = var_8396_begin_0, end = var_8396_end_0, end_mask = var_8396_end_mask_0, squeeze_mask = var_8396_squeeze_mask_0, x = coreml_update_state_53)[name = string("op_8396_cast_fp16")]; tensor V_cache_25_axes_0 = const()[name = string("V_cache_25_axes_0"), val = tensor([0])]; tensor V_cache_25_cast_fp16 = expand_dims(axes = V_cache_25_axes_0, x = var_8396_cast_fp16)[name = string("V_cache_25_cast_fp16")]; tensor k_padded_21_pad_0 = const()[name = string("k_padded_21_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_21_mode_0 = const()[name = string("k_padded_21_mode_0"), val = string("constant")]; fp16 const_221_to_fp16 = const()[name = string("const_221_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_21_cast_fp16 = pad(constant_val = const_221_to_fp16, mode = k_padded_21_mode_0, pad = k_padded_21_pad_0, x = input_361)[name = string("k_padded_21_cast_fp16")]; tensor v_padded_21_pad_0 = const()[name = string("v_padded_21_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_21_mode_0 = const()[name = string("v_padded_21_mode_0"), val = string("constant")]; fp16 const_222_to_fp16 = const()[name = string("const_222_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_21_cast_fp16 = pad(constant_val = const_222_to_fp16, mode = v_padded_21_mode_0, pad = v_padded_21_pad_0, x = var_8373_cast_fp16)[name = string("v_padded_21_cast_fp16")]; tensor var_8414_cast_fp16 = mul(x = K_cache_25_cast_fp16, y = var_2187_cast_fp16)[name = string("op_8414_cast_fp16")]; tensor var_8415_reps_0 = const()[name = string("op_8415_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_8415_cast_fp16 = tile(reps = var_8415_reps_0, x = k_padded_21_cast_fp16)[name = string("op_8415_cast_fp16")]; tensor var_8416_cast_fp16 = mul(x = var_8415_cast_fp16, y = update_mask)[name = string("op_8416_cast_fp16")]; tensor K_new_25_cast_fp16 = add(x = var_8414_cast_fp16, y = var_8416_cast_fp16)[name = string("K_new_25_cast_fp16")]; tensor var_8422_cast_fp16 = mul(x = V_cache_25_cast_fp16, y = var_2187_cast_fp16)[name = string("op_8422_cast_fp16")]; tensor var_8423_reps_0 = const()[name = string("op_8423_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_8423_cast_fp16 = tile(reps = var_8423_reps_0, x = v_padded_21_cast_fp16)[name = string("op_8423_cast_fp16")]; tensor var_8424_cast_fp16 = mul(x = var_8423_cast_fp16, y = update_mask)[name = string("op_8424_cast_fp16")]; tensor V_new_25_cast_fp16 = add(x = var_8422_cast_fp16, y = var_8424_cast_fp16)[name = string("V_new_25_cast_fp16")]; tensor var_8428_axes_0 = const()[name = string("op_8428_axes_0"), val = tensor([0])]; tensor var_8428_cast_fp16 = squeeze(axes = var_8428_axes_0, x = K_new_25_cast_fp16)[name = string("op_8428_cast_fp16")]; tensor concat_96 = const()[name = string("concat_96"), val = tensor([12, 0, 0, 0])]; tensor concat_97 = const()[name = string("concat_97"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_96, begin_mask = kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_97, end_mask = kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_25_stride_0, update = var_8428_cast_fp16, x = coreml_update_state_53)[name = string("kv_cache_0_internal_tensor_assign_25_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_25_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_54_write_state")]; tensor coreml_update_state_54 = read_state(input = kv_cache_0)[name = string("coreml_update_state_54")]; tensor var_8435_axes_0 = const()[name = string("op_8435_axes_0"), val = tensor([0])]; tensor var_8435_cast_fp16 = squeeze(axes = var_8435_axes_0, x = V_new_25_cast_fp16)[name = string("op_8435_cast_fp16")]; tensor concat_98 = const()[name = string("concat_98"), val = tensor([47, 0, 0, 0])]; tensor concat_99 = const()[name = string("concat_99"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_98, begin_mask = kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_99, end_mask = kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_26_stride_0, update = var_8435_cast_fp16, x = coreml_update_state_54)[name = string("kv_cache_0_internal_tensor_assign_26_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_26_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_55_write_state")]; tensor coreml_update_state_55 = read_state(input = kv_cache_0)[name = string("coreml_update_state_55")]; tensor K_for_attn_25_begin_0 = const()[name = string("K_for_attn_25_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_25_end_0 = const()[name = string("K_for_attn_25_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_25_end_mask_0 = const()[name = string("K_for_attn_25_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_25_cast_fp16 = slice_by_index(begin = K_for_attn_25_begin_0, end = K_for_attn_25_end_0, end_mask = K_for_attn_25_end_mask_0, x = K_new_25_cast_fp16)[name = string("K_for_attn_25_cast_fp16")]; tensor V_for_attn_25_begin_0 = const()[name = string("V_for_attn_25_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_25_end_0 = const()[name = string("V_for_attn_25_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_25_end_mask_0 = const()[name = string("V_for_attn_25_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_25_cast_fp16 = slice_by_index(begin = V_for_attn_25_begin_0, end = V_for_attn_25_end_0, end_mask = V_for_attn_25_end_mask_0, x = V_new_25_cast_fp16)[name = string("V_for_attn_25_cast_fp16")]; tensor transpose_48_perm_0 = const()[name = string("transpose_48_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_24_reps_0 = const()[name = string("tile_24_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_48_cast_fp16 = transpose(perm = transpose_48_perm_0, x = K_for_attn_25_cast_fp16)[name = string("transpose_182")]; tensor tile_24_cast_fp16 = tile(reps = tile_24_reps_0, x = transpose_48_cast_fp16)[name = string("tile_24_cast_fp16")]; tensor concat_100 = const()[name = string("concat_100"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_48_cast_fp16 = reshape(shape = concat_100, x = tile_24_cast_fp16)[name = string("reshape_48_cast_fp16")]; tensor transpose_49_perm_0 = const()[name = string("transpose_49_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_101 = const()[name = string("concat_101"), val = tensor([-1, 1, 512, 256])]; tensor transpose_49_cast_fp16 = transpose(perm = transpose_49_perm_0, x = reshape_48_cast_fp16)[name = string("transpose_181")]; tensor reshape_49_cast_fp16 = reshape(shape = concat_101, x = transpose_49_cast_fp16)[name = string("reshape_49_cast_fp16")]; tensor transpose_152_perm_0 = const()[name = string("transpose_152_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_50_perm_0 = const()[name = string("transpose_50_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_25_reps_0 = const()[name = string("tile_25_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_50_cast_fp16 = transpose(perm = transpose_50_perm_0, x = V_for_attn_25_cast_fp16)[name = string("transpose_180")]; tensor tile_25_cast_fp16 = tile(reps = tile_25_reps_0, x = transpose_50_cast_fp16)[name = string("tile_25_cast_fp16")]; tensor concat_102 = const()[name = string("concat_102"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_50_cast_fp16 = reshape(shape = concat_102, x = tile_25_cast_fp16)[name = string("reshape_50_cast_fp16")]; tensor transpose_51_perm_0 = const()[name = string("transpose_51_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_103 = const()[name = string("concat_103"), val = tensor([-1, 1, 512, 256])]; tensor transpose_51_cast_fp16 = transpose(perm = transpose_51_perm_0, x = reshape_50_cast_fp16)[name = string("transpose_179")]; tensor reshape_51_cast_fp16 = reshape(shape = concat_103, x = transpose_51_cast_fp16)[name = string("reshape_51_cast_fp16")]; tensor V_expanded_25_perm_0 = const()[name = string("V_expanded_25_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_8462_transpose_x_0 = const()[name = string("op_8462_transpose_x_0"), val = bool(false)]; bool var_8462_transpose_y_0 = const()[name = string("op_8462_transpose_y_0"), val = bool(false)]; tensor transpose_152_cast_fp16 = transpose(perm = transpose_152_perm_0, x = reshape_49_cast_fp16)[name = string("transpose_178")]; tensor var_8462_cast_fp16 = matmul(transpose_x = var_8462_transpose_x_0, transpose_y = var_8462_transpose_y_0, x = q_103, y = transpose_152_cast_fp16)[name = string("op_8462_cast_fp16")]; tensor attn_weights_75_cast_fp16 = add(x = var_8462_cast_fp16, y = causal_mask)[name = string("attn_weights_75_cast_fp16")]; int32 var_8467 = const()[name = string("op_8467"), val = int32(-1)]; tensor attn_weights_77_cast_fp16 = softmax(axis = var_8467, x = attn_weights_75_cast_fp16)[name = string("attn_weights_77_cast_fp16")]; bool attn_output_73_transpose_x_0 = const()[name = string("attn_output_73_transpose_x_0"), val = bool(false)]; bool attn_output_73_transpose_y_0 = const()[name = string("attn_output_73_transpose_y_0"), val = bool(false)]; tensor V_expanded_25_cast_fp16 = transpose(perm = V_expanded_25_perm_0, x = reshape_51_cast_fp16)[name = string("transpose_177")]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = attn_weights_77_cast_fp16, y = V_expanded_25_cast_fp16)[name = string("attn_output_73_cast_fp16")]; tensor var_8475 = const()[name = string("op_8475"), val = tensor([0, 2, 1, 3])]; tensor var_8482 = const()[name = string("op_8482"), val = tensor([1, 1, -1])]; tensor var_8476_cast_fp16 = transpose(perm = var_8475, x = attn_output_73_cast_fp16)[name = string("transpose_176")]; tensor attn_output_75_cast_fp16 = reshape(shape = var_8482, x = var_8476_cast_fp16)[name = string("attn_output_75_cast_fp16")]; tensor var_8487 = const()[name = string("op_8487"), val = tensor([0, 2, 1])]; string var_8503_pad_type_0 = const()[name = string("op_8503_pad_type_0"), val = string("valid")]; int32 var_8503_groups_0 = const()[name = string("op_8503_groups_0"), val = int32(1)]; tensor var_8503_strides_0 = const()[name = string("op_8503_strides_0"), val = tensor([1])]; tensor var_8503_pad_0 = const()[name = string("op_8503_pad_0"), val = tensor([0, 0])]; tensor var_8503_dilations_0 = const()[name = string("op_8503_dilations_0"), val = tensor([1])]; tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086455040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088027968))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_8488_cast_fp16 = transpose(perm = var_8487, x = attn_output_75_cast_fp16)[name = string("transpose_175")]; tensor var_8503_cast_fp16 = conv(dilations = var_8503_dilations_0, groups = var_8503_groups_0, pad = var_8503_pad_0, pad_type = var_8503_pad_type_0, strides = var_8503_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_8488_cast_fp16)[name = string("op_8503_cast_fp16")]; tensor var_8507 = const()[name = string("op_8507"), val = tensor([0, 2, 1])]; int32 var_8513 = const()[name = string("op_8513"), val = int32(-1)]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_375_cast_fp16 = transpose(perm = var_8507, x = var_8503_cast_fp16)[name = string("transpose_174")]; tensor var_8519_cast_fp16 = mul(x = x_375_cast_fp16, y = const_223_promoted_to_fp16)[name = string("op_8519_cast_fp16")]; bool input_367_interleave_0 = const()[name = string("input_367_interleave_0"), val = bool(false)]; tensor input_367_cast_fp16 = concat(axis = var_8513, interleave = input_367_interleave_0, values = (x_375_cast_fp16, var_8519_cast_fp16))[name = string("input_367_cast_fp16")]; tensor normed_349_axes_0 = const()[name = string("normed_349_axes_0"), val = tensor([-1])]; fp16 var_8511_to_fp16 = const()[name = string("op_8511_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_349_cast_fp16 = layer_norm(axes = normed_349_axes_0, epsilon = var_8511_to_fp16, x = input_367_cast_fp16)[name = string("normed_349_cast_fp16")]; tensor var_8524_split_sizes_0 = const()[name = string("op_8524_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8524_axis_0 = const()[name = string("op_8524_axis_0"), val = int32(-1)]; tensor var_8524_cast_fp16_0, tensor var_8524_cast_fp16_1 = split(axis = var_8524_axis_0, split_sizes = var_8524_split_sizes_0, x = normed_349_cast_fp16)[name = string("op_8524_cast_fp16")]; tensor const_224_to_fp16 = const()[name = string("const_224_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088029568)))]; tensor var_8527_cast_fp16 = mul(x = var_8524_cast_fp16_0, y = const_224_to_fp16)[name = string("op_8527_cast_fp16")]; tensor x_379_cast_fp16 = add(x = x_361_cast_fp16, y = var_8527_cast_fp16)[name = string("x_379_cast_fp16")]; int32 var_8534 = const()[name = string("op_8534"), val = int32(-1)]; fp16 const_225_promoted_to_fp16 = const()[name = string("const_225_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8540_cast_fp16 = mul(x = x_379_cast_fp16, y = const_225_promoted_to_fp16)[name = string("op_8540_cast_fp16")]; bool input_369_interleave_0 = const()[name = string("input_369_interleave_0"), val = bool(false)]; tensor input_369_cast_fp16 = concat(axis = var_8534, interleave = input_369_interleave_0, values = (x_379_cast_fp16, var_8540_cast_fp16))[name = string("input_369_cast_fp16")]; tensor normed_353_axes_0 = const()[name = string("normed_353_axes_0"), val = tensor([-1])]; fp16 var_8532_to_fp16 = const()[name = string("op_8532_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_353_cast_fp16 = layer_norm(axes = normed_353_axes_0, epsilon = var_8532_to_fp16, x = input_369_cast_fp16)[name = string("normed_353_cast_fp16")]; tensor var_8545_split_sizes_0 = const()[name = string("op_8545_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8545_axis_0 = const()[name = string("op_8545_axis_0"), val = int32(-1)]; tensor var_8545_cast_fp16_0, tensor var_8545_cast_fp16_1 = split(axis = var_8545_axis_0, split_sizes = var_8545_split_sizes_0, x = normed_353_cast_fp16)[name = string("op_8545_cast_fp16")]; tensor const_226_to_fp16 = const()[name = string("const_226_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088032704)))]; tensor var_8548_cast_fp16 = mul(x = var_8545_cast_fp16_0, y = const_226_to_fp16)[name = string("op_8548_cast_fp16")]; tensor var_8561 = const()[name = string("op_8561"), val = tensor([0, 2, 1])]; tensor input_371_axes_0 = const()[name = string("input_371_axes_0"), val = tensor([2])]; tensor var_8562 = transpose(perm = var_8561, x = var_8548_cast_fp16)[name = string("transpose_173")]; tensor input_371 = expand_dims(axes = input_371_axes_0, x = var_8562)[name = string("input_371")]; string gate_49_pad_type_0 = const()[name = string("gate_49_pad_type_0"), val = string("valid")]; tensor gate_49_strides_0 = const()[name = string("gate_49_strides_0"), val = tensor([1, 1])]; tensor gate_49_pad_0 = const()[name = string("gate_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_49_dilations_0 = const()[name = string("gate_49_dilations_0"), val = tensor([1, 1])]; int32 gate_49_groups_0 = const()[name = string("gate_49_groups_0"), val = int32(1)]; tensor gate_49 = conv(dilations = gate_49_dilations_0, groups = gate_49_groups_0, pad = gate_49_pad_0, pad_type = gate_49_pad_type_0, strides = gate_49_strides_0, weight = layers_12_mlp_gate_proj_weight_palettized, x = input_371)[name = string("gate_49")]; string up_25_pad_type_0 = const()[name = string("up_25_pad_type_0"), val = string("valid")]; tensor up_25_strides_0 = const()[name = string("up_25_strides_0"), val = tensor([1, 1])]; tensor up_25_pad_0 = const()[name = string("up_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_25_dilations_0 = const()[name = string("up_25_dilations_0"), val = tensor([1, 1])]; int32 up_25_groups_0 = const()[name = string("up_25_groups_0"), val = int32(1)]; tensor up_25 = conv(dilations = up_25_dilations_0, groups = up_25_groups_0, pad = up_25_pad_0, pad_type = up_25_pad_type_0, strides = up_25_strides_0, weight = layers_12_mlp_up_proj_weight_palettized, x = input_371)[name = string("up_25")]; string gate_51_mode_0 = const()[name = string("gate_51_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_51 = gelu(mode = gate_51_mode_0, x = gate_49)[name = string("gate_51")]; tensor input_373 = mul(x = gate_51, y = up_25)[name = string("input_373")]; string mlp_out_25_pad_type_0 = const()[name = string("mlp_out_25_pad_type_0"), val = string("valid")]; tensor mlp_out_25_strides_0 = const()[name = string("mlp_out_25_strides_0"), val = tensor([1, 1])]; tensor mlp_out_25_pad_0 = const()[name = string("mlp_out_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_25_dilations_0 = const()[name = string("mlp_out_25_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_25_groups_0 = const()[name = string("mlp_out_25_groups_0"), val = int32(1)]; tensor mlp_out_25 = conv(dilations = mlp_out_25_dilations_0, groups = mlp_out_25_groups_0, pad = mlp_out_25_pad_0, pad_type = mlp_out_25_pad_type_0, strides = mlp_out_25_strides_0, weight = layers_12_mlp_down_proj_weight_palettized, x = input_373)[name = string("mlp_out_25")]; tensor var_8602_axes_0 = const()[name = string("op_8602_axes_0"), val = tensor([2])]; tensor var_8602 = squeeze(axes = var_8602_axes_0, x = mlp_out_25)[name = string("op_8602")]; tensor var_8606 = const()[name = string("op_8606"), val = tensor([0, 2, 1])]; int32 var_8612 = const()[name = string("op_8612"), val = int32(-1)]; fp16 const_227_promoted_to_fp16 = const()[name = string("const_227_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_383 = transpose(perm = var_8606, x = var_8602)[name = string("transpose_172")]; tensor var_8618_cast_fp16 = mul(x = x_383, y = const_227_promoted_to_fp16)[name = string("op_8618_cast_fp16")]; bool input_375_interleave_0 = const()[name = string("input_375_interleave_0"), val = bool(false)]; tensor input_375_cast_fp16 = concat(axis = var_8612, interleave = input_375_interleave_0, values = (x_383, var_8618_cast_fp16))[name = string("input_375_cast_fp16")]; tensor normed_357_axes_0 = const()[name = string("normed_357_axes_0"), val = tensor([-1])]; fp16 var_8610_to_fp16 = const()[name = string("op_8610_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_357_cast_fp16 = layer_norm(axes = normed_357_axes_0, epsilon = var_8610_to_fp16, x = input_375_cast_fp16)[name = string("normed_357_cast_fp16")]; tensor var_8623_split_sizes_0 = const()[name = string("op_8623_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8623_axis_0 = const()[name = string("op_8623_axis_0"), val = int32(-1)]; tensor var_8623_cast_fp16_0, tensor var_8623_cast_fp16_1 = split(axis = var_8623_axis_0, split_sizes = var_8623_split_sizes_0, x = normed_357_cast_fp16)[name = string("op_8623_cast_fp16")]; tensor const_228_to_fp16 = const()[name = string("const_228_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088035840)))]; tensor var_8626_cast_fp16 = mul(x = var_8623_cast_fp16_0, y = const_228_to_fp16)[name = string("op_8626_cast_fp16")]; tensor hidden_states_153_cast_fp16 = add(x = x_379_cast_fp16, y = var_8626_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor per_layer_slice_25_begin_0 = const()[name = string("per_layer_slice_25_begin_0"), val = tensor([0, 0, 3072])]; tensor per_layer_slice_25_end_0 = const()[name = string("per_layer_slice_25_end_0"), val = tensor([1, 1, 3328])]; tensor per_layer_slice_25_end_mask_0 = const()[name = string("per_layer_slice_25_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_25_cast_fp16 = slice_by_index(begin = per_layer_slice_25_begin_0, end = per_layer_slice_25_end_0, end_mask = per_layer_slice_25_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_25_cast_fp16")]; tensor gated_49 = linear(bias = linear_0_bias_0, weight = layers_12_per_layer_input_gate_weight_palettized, x = hidden_states_153_cast_fp16)[name = string("linear_24")]; string gated_51_mode_0 = const()[name = string("gated_51_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_51 = gelu(mode = gated_51_mode_0, x = gated_49)[name = string("gated_51")]; tensor input_379_cast_fp16 = mul(x = gated_51, y = per_layer_slice_25_cast_fp16)[name = string("input_379_cast_fp16")]; tensor layers_12_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088038976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088235648))))[name = string("layers_12_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_25_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_12_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_379_cast_fp16)[name = string("linear_25_cast_fp16")]; int32 var_8663 = const()[name = string("op_8663"), val = int32(-1)]; fp16 const_229_promoted_to_fp16 = const()[name = string("const_229_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8669_cast_fp16 = mul(x = linear_25_cast_fp16, y = const_229_promoted_to_fp16)[name = string("op_8669_cast_fp16")]; bool input_381_interleave_0 = const()[name = string("input_381_interleave_0"), val = bool(false)]; tensor input_381_cast_fp16 = concat(axis = var_8663, interleave = input_381_interleave_0, values = (linear_25_cast_fp16, var_8669_cast_fp16))[name = string("input_381_cast_fp16")]; tensor normed_361_axes_0 = const()[name = string("normed_361_axes_0"), val = tensor([-1])]; fp16 var_8661_to_fp16 = const()[name = string("op_8661_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_361_cast_fp16 = layer_norm(axes = normed_361_axes_0, epsilon = var_8661_to_fp16, x = input_381_cast_fp16)[name = string("normed_361_cast_fp16")]; tensor var_8674_split_sizes_0 = const()[name = string("op_8674_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8674_axis_0 = const()[name = string("op_8674_axis_0"), val = int32(-1)]; tensor var_8674_cast_fp16_0, tensor var_8674_cast_fp16_1 = split(axis = var_8674_axis_0, split_sizes = var_8674_split_sizes_0, x = normed_361_cast_fp16)[name = string("op_8674_cast_fp16")]; tensor const_230_to_fp16 = const()[name = string("const_230_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088237248)))]; tensor var_8677_cast_fp16 = mul(x = var_8674_cast_fp16_0, y = const_230_to_fp16)[name = string("op_8677_cast_fp16")]; tensor hidden_states_157_cast_fp16 = add(x = hidden_states_153_cast_fp16, y = var_8677_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor layers_12_layer_scalar_to_fp16 = const()[name = string("layers_12_layer_scalar_to_fp16"), val = tensor([0x1.4cp-2])]; tensor x_391_cast_fp16 = mul(x = hidden_states_157_cast_fp16, y = layers_12_layer_scalar_to_fp16)[name = string("x_391_cast_fp16")]; int32 var_8685 = const()[name = string("op_8685"), val = int32(-1)]; fp16 const_231_promoted_to_fp16 = const()[name = string("const_231_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8691_cast_fp16 = mul(x = x_391_cast_fp16, y = const_231_promoted_to_fp16)[name = string("op_8691_cast_fp16")]; bool input_383_interleave_0 = const()[name = string("input_383_interleave_0"), val = bool(false)]; tensor input_383_cast_fp16 = concat(axis = var_8685, interleave = input_383_interleave_0, values = (x_391_cast_fp16, var_8691_cast_fp16))[name = string("input_383_cast_fp16")]; tensor normed_365_axes_0 = const()[name = string("normed_365_axes_0"), val = tensor([-1])]; fp16 var_8683_to_fp16 = const()[name = string("op_8683_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_365_cast_fp16 = layer_norm(axes = normed_365_axes_0, epsilon = var_8683_to_fp16, x = input_383_cast_fp16)[name = string("normed_365_cast_fp16")]; tensor var_8696_split_sizes_0 = const()[name = string("op_8696_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_8696_axis_0 = const()[name = string("op_8696_axis_0"), val = int32(-1)]; tensor var_8696_cast_fp16_0, tensor var_8696_cast_fp16_1 = split(axis = var_8696_axis_0, split_sizes = var_8696_split_sizes_0, x = normed_365_cast_fp16)[name = string("op_8696_cast_fp16")]; tensor const_232_to_fp16 = const()[name = string("const_232_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088240384)))]; tensor var_8699_cast_fp16 = mul(x = var_8696_cast_fp16_0, y = const_232_to_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8707 = const()[name = string("op_8707"), val = tensor([0, 2, 1])]; tensor var_8710_axes_0 = const()[name = string("op_8710_axes_0"), val = tensor([2])]; tensor var_8708_cast_fp16 = transpose(perm = var_8707, x = var_8699_cast_fp16)[name = string("transpose_171")]; tensor var_8710_cast_fp16 = expand_dims(axes = var_8710_axes_0, x = var_8708_cast_fp16)[name = string("op_8710_cast_fp16")]; string var_8726_pad_type_0 = const()[name = string("op_8726_pad_type_0"), val = string("valid")]; tensor var_8726_strides_0 = const()[name = string("op_8726_strides_0"), val = tensor([1, 1])]; tensor var_8726_pad_0 = const()[name = string("op_8726_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8726_dilations_0 = const()[name = string("op_8726_dilations_0"), val = tensor([1, 1])]; int32 var_8726_groups_0 = const()[name = string("op_8726_groups_0"), val = int32(1)]; tensor var_8726 = conv(dilations = var_8726_dilations_0, groups = var_8726_groups_0, pad = var_8726_pad_0, pad_type = var_8726_pad_type_0, strides = var_8726_strides_0, weight = layers_13_self_attn_q_proj_weight_palettized, x = var_8710_cast_fp16)[name = string("op_8726")]; tensor var_8731 = const()[name = string("op_8731"), val = tensor([1, 8, 256, 1])]; tensor var_8732 = reshape(shape = var_8731, x = var_8726)[name = string("op_8732")]; tensor var_8737 = const()[name = string("op_8737"), val = tensor([0, 1, 3, 2])]; tensor var_8747 = const()[name = string("op_8747"), val = tensor([1, 8, 256])]; tensor var_8738 = transpose(perm = var_8737, x = var_8732)[name = string("transpose_170")]; tensor x_395 = reshape(shape = var_8747, x = var_8738)[name = string("x_395")]; int32 var_8753 = const()[name = string("op_8753"), val = int32(-1)]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8759_cast_fp16 = mul(x = x_395, y = const_233_promoted_to_fp16)[name = string("op_8759_cast_fp16")]; bool input_387_interleave_0 = const()[name = string("input_387_interleave_0"), val = bool(false)]; tensor input_387_cast_fp16 = concat(axis = var_8753, interleave = input_387_interleave_0, values = (x_395, var_8759_cast_fp16))[name = string("input_387_cast_fp16")]; tensor normed_369_axes_0 = const()[name = string("normed_369_axes_0"), val = tensor([-1])]; fp16 var_8751_to_fp16 = const()[name = string("op_8751_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_369_cast_fp16 = layer_norm(axes = normed_369_axes_0, epsilon = var_8751_to_fp16, x = input_387_cast_fp16)[name = string("normed_369_cast_fp16")]; tensor var_8764_split_sizes_0 = const()[name = string("op_8764_split_sizes_0"), val = tensor([256, 256])]; int32 var_8764_axis_0 = const()[name = string("op_8764_axis_0"), val = int32(-1)]; tensor var_8764_cast_fp16_0, tensor var_8764_cast_fp16_1 = split(axis = var_8764_axis_0, split_sizes = var_8764_split_sizes_0, x = normed_369_cast_fp16)[name = string("op_8764_cast_fp16")]; tensor const_234_to_fp16 = const()[name = string("const_234_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088243520)))]; tensor var_8767_cast_fp16 = mul(x = var_8764_cast_fp16_0, y = const_234_to_fp16)[name = string("op_8767_cast_fp16")]; tensor var_8773 = const()[name = string("op_8773"), val = tensor([1, 8, 1, 256])]; tensor q_107 = reshape(shape = var_8773, x = var_8767_cast_fp16)[name = string("q_107")]; tensor var_8775 = mul(x = q_107, y = cos_1)[name = string("op_8775")]; tensor var_8776_split_sizes_0 = const()[name = string("op_8776_split_sizes_0"), val = tensor([128, 128])]; int32 var_8776_axis_0 = const()[name = string("op_8776_axis_0"), val = int32(-1)]; tensor var_8776_0, tensor var_8776_1 = split(axis = var_8776_axis_0, split_sizes = var_8776_split_sizes_0, x = q_107)[name = string("op_8776")]; fp16 const_235_promoted = const()[name = string("const_235_promoted"), val = fp16(-0x1p+0)]; tensor var_8778 = mul(x = var_8776_1, y = const_235_promoted)[name = string("op_8778")]; int32 var_8780 = const()[name = string("op_8780"), val = int32(-1)]; bool var_8781_interleave_0 = const()[name = string("op_8781_interleave_0"), val = bool(false)]; tensor var_8781 = concat(axis = var_8780, interleave = var_8781_interleave_0, values = (var_8778, var_8776_0))[name = string("op_8781")]; tensor var_8782 = mul(x = var_8781, y = sin_1)[name = string("op_8782")]; tensor q_111 = add(x = var_8775, y = var_8782)[name = string("q_111")]; string var_8795_pad_type_0 = const()[name = string("op_8795_pad_type_0"), val = string("valid")]; tensor var_8795_strides_0 = const()[name = string("op_8795_strides_0"), val = tensor([1, 1])]; tensor var_8795_pad_0 = const()[name = string("op_8795_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8795_dilations_0 = const()[name = string("op_8795_dilations_0"), val = tensor([1, 1])]; int32 var_8795_groups_0 = const()[name = string("op_8795_groups_0"), val = int32(1)]; tensor var_8795 = conv(dilations = var_8795_dilations_0, groups = var_8795_groups_0, pad = var_8795_pad_0, pad_type = var_8795_pad_type_0, strides = var_8795_strides_0, weight = layers_13_self_attn_k_proj_weight_palettized, x = var_8710_cast_fp16)[name = string("op_8795")]; tensor var_8800 = const()[name = string("op_8800"), val = tensor([1, 1, 256, 1])]; tensor var_8801 = reshape(shape = var_8800, x = var_8795)[name = string("op_8801")]; tensor var_8806 = const()[name = string("op_8806"), val = tensor([0, 1, 3, 2])]; string var_8823_pad_type_0 = const()[name = string("op_8823_pad_type_0"), val = string("valid")]; tensor var_8823_strides_0 = const()[name = string("op_8823_strides_0"), val = tensor([1, 1])]; tensor var_8823_pad_0 = const()[name = string("op_8823_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8823_dilations_0 = const()[name = string("op_8823_dilations_0"), val = tensor([1, 1])]; int32 var_8823_groups_0 = const()[name = string("op_8823_groups_0"), val = int32(1)]; tensor var_8823 = conv(dilations = var_8823_dilations_0, groups = var_8823_groups_0, pad = var_8823_pad_0, pad_type = var_8823_pad_type_0, strides = var_8823_strides_0, weight = layers_13_self_attn_v_proj_weight_palettized, x = var_8710_cast_fp16)[name = string("op_8823")]; tensor var_8828 = const()[name = string("op_8828"), val = tensor([1, 1, 256, 1])]; tensor var_8829 = reshape(shape = var_8828, x = var_8823)[name = string("op_8829")]; tensor var_8834 = const()[name = string("op_8834"), val = tensor([0, 1, 3, 2])]; tensor var_8844 = const()[name = string("op_8844"), val = tensor([1, 1, 256])]; tensor var_8807 = transpose(perm = var_8806, x = var_8801)[name = string("transpose_169")]; tensor x_399 = reshape(shape = var_8844, x = var_8807)[name = string("x_399")]; int32 var_8850 = const()[name = string("op_8850"), val = int32(-1)]; fp16 const_236_promoted_to_fp16 = const()[name = string("const_236_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8856_cast_fp16 = mul(x = x_399, y = const_236_promoted_to_fp16)[name = string("op_8856_cast_fp16")]; bool input_389_interleave_0 = const()[name = string("input_389_interleave_0"), val = bool(false)]; tensor input_389_cast_fp16 = concat(axis = var_8850, interleave = input_389_interleave_0, values = (x_399, var_8856_cast_fp16))[name = string("input_389_cast_fp16")]; tensor normed_373_axes_0 = const()[name = string("normed_373_axes_0"), val = tensor([-1])]; fp16 var_8848_to_fp16 = const()[name = string("op_8848_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_373_cast_fp16 = layer_norm(axes = normed_373_axes_0, epsilon = var_8848_to_fp16, x = input_389_cast_fp16)[name = string("normed_373_cast_fp16")]; tensor var_8861_split_sizes_0 = const()[name = string("op_8861_split_sizes_0"), val = tensor([256, 256])]; int32 var_8861_axis_0 = const()[name = string("op_8861_axis_0"), val = int32(-1)]; tensor var_8861_cast_fp16_0, tensor var_8861_cast_fp16_1 = split(axis = var_8861_axis_0, split_sizes = var_8861_split_sizes_0, x = normed_373_cast_fp16)[name = string("op_8861_cast_fp16")]; tensor const_237_to_fp16 = const()[name = string("const_237_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088244096)))]; tensor var_8864_cast_fp16 = mul(x = var_8861_cast_fp16_0, y = const_237_to_fp16)[name = string("op_8864_cast_fp16")]; tensor var_8870 = const()[name = string("op_8870"), val = tensor([1, 1, 1, 256])]; tensor q_109 = reshape(shape = var_8870, x = var_8864_cast_fp16)[name = string("q_109")]; fp16 var_8877_promoted_to_fp16 = const()[name = string("op_8877_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_8835 = transpose(perm = var_8834, x = var_8829)[name = string("transpose_168")]; tensor var_8878_cast_fp16 = pow(x = var_8835, y = var_8877_promoted_to_fp16)[name = string("op_8878_cast_fp16")]; tensor var_8883_axes_0 = const()[name = string("op_8883_axes_0"), val = tensor([-1])]; bool var_8883_keep_dims_0 = const()[name = string("op_8883_keep_dims_0"), val = bool(true)]; tensor var_8883_cast_fp16 = reduce_mean(axes = var_8883_axes_0, keep_dims = var_8883_keep_dims_0, x = var_8878_cast_fp16)[name = string("op_8883_cast_fp16")]; fp16 var_8885_to_fp16 = const()[name = string("op_8885_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_27_cast_fp16 = add(x = var_8883_cast_fp16, y = var_8885_to_fp16)[name = string("mean_sq_27_cast_fp16")]; fp16 var_8892_to_fp16 = const()[name = string("op_8892_to_fp16"), val = fp16(-0x1p-1)]; tensor var_8893_cast_fp16 = pow(x = mean_sq_27_cast_fp16, y = var_8892_to_fp16)[name = string("op_8893_cast_fp16")]; tensor var_8894_cast_fp16 = mul(x = var_8835, y = var_8893_cast_fp16)[name = string("op_8894_cast_fp16")]; tensor var_8900 = mul(x = q_109, y = cos_1)[name = string("op_8900")]; tensor var_8901_split_sizes_0 = const()[name = string("op_8901_split_sizes_0"), val = tensor([128, 128])]; int32 var_8901_axis_0 = const()[name = string("op_8901_axis_0"), val = int32(-1)]; tensor var_8901_0, tensor var_8901_1 = split(axis = var_8901_axis_0, split_sizes = var_8901_split_sizes_0, x = q_109)[name = string("op_8901")]; fp16 const_238_promoted = const()[name = string("const_238_promoted"), val = fp16(-0x1p+0)]; tensor var_8903 = mul(x = var_8901_1, y = const_238_promoted)[name = string("op_8903")]; int32 var_8905 = const()[name = string("op_8905"), val = int32(-1)]; bool var_8906_interleave_0 = const()[name = string("op_8906_interleave_0"), val = bool(false)]; tensor var_8906 = concat(axis = var_8905, interleave = var_8906_interleave_0, values = (var_8903, var_8901_0))[name = string("op_8906")]; tensor var_8907 = mul(x = var_8906, y = sin_1)[name = string("op_8907")]; tensor input_391 = add(x = var_8900, y = var_8907)[name = string("input_391")]; tensor var_8912_begin_0 = const()[name = string("op_8912_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_8912_end_0 = const()[name = string("op_8912_end_0"), val = tensor([14, 1, 512, 512])]; tensor var_8912_end_mask_0 = const()[name = string("op_8912_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8912_squeeze_mask_0 = const()[name = string("op_8912_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_8912_cast_fp16 = slice_by_index(begin = var_8912_begin_0, end = var_8912_end_0, end_mask = var_8912_end_mask_0, squeeze_mask = var_8912_squeeze_mask_0, x = coreml_update_state_55)[name = string("op_8912_cast_fp16")]; tensor K_cache_27_axes_0 = const()[name = string("K_cache_27_axes_0"), val = tensor([0])]; tensor K_cache_27_cast_fp16 = expand_dims(axes = K_cache_27_axes_0, x = var_8912_cast_fp16)[name = string("K_cache_27_cast_fp16")]; tensor var_8917_begin_0 = const()[name = string("op_8917_begin_0"), val = tensor([48, 0, 0, 0])]; tensor var_8917_end_0 = const()[name = string("op_8917_end_0"), val = tensor([49, 1, 512, 512])]; tensor var_8917_end_mask_0 = const()[name = string("op_8917_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8917_squeeze_mask_0 = const()[name = string("op_8917_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_8917_cast_fp16 = slice_by_index(begin = var_8917_begin_0, end = var_8917_end_0, end_mask = var_8917_end_mask_0, squeeze_mask = var_8917_squeeze_mask_0, x = coreml_update_state_55)[name = string("op_8917_cast_fp16")]; tensor V_cache_27_axes_0 = const()[name = string("V_cache_27_axes_0"), val = tensor([0])]; tensor V_cache_27_cast_fp16 = expand_dims(axes = V_cache_27_axes_0, x = var_8917_cast_fp16)[name = string("V_cache_27_cast_fp16")]; tensor k_padded_pad_0 = const()[name = string("k_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_mode_0 = const()[name = string("k_padded_mode_0"), val = string("constant")]; fp16 const_239_to_fp16 = const()[name = string("const_239_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_cast_fp16 = pad(constant_val = const_239_to_fp16, mode = k_padded_mode_0, pad = k_padded_pad_0, x = input_391)[name = string("k_padded_cast_fp16")]; tensor v_padded_pad_0 = const()[name = string("v_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_mode_0 = const()[name = string("v_padded_mode_0"), val = string("constant")]; fp16 const_240_to_fp16 = const()[name = string("const_240_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_cast_fp16 = pad(constant_val = const_240_to_fp16, mode = v_padded_mode_0, pad = v_padded_pad_0, x = var_8894_cast_fp16)[name = string("v_padded_cast_fp16")]; tensor var_8935_cast_fp16 = mul(x = K_cache_27_cast_fp16, y = var_2187_cast_fp16)[name = string("op_8935_cast_fp16")]; tensor var_8936_reps_0 = const()[name = string("op_8936_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_8936_cast_fp16 = tile(reps = var_8936_reps_0, x = k_padded_cast_fp16)[name = string("op_8936_cast_fp16")]; tensor var_8937_cast_fp16 = mul(x = var_8936_cast_fp16, y = update_mask)[name = string("op_8937_cast_fp16")]; tensor K_new_27_cast_fp16 = add(x = var_8935_cast_fp16, y = var_8937_cast_fp16)[name = string("K_new_27_cast_fp16")]; tensor var_8943_cast_fp16 = mul(x = V_cache_27_cast_fp16, y = var_2187_cast_fp16)[name = string("op_8943_cast_fp16")]; tensor var_8944_reps_0 = const()[name = string("op_8944_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_8944_cast_fp16 = tile(reps = var_8944_reps_0, x = v_padded_cast_fp16)[name = string("op_8944_cast_fp16")]; tensor var_8945_cast_fp16 = mul(x = var_8944_cast_fp16, y = update_mask)[name = string("op_8945_cast_fp16")]; tensor V_new_27_cast_fp16 = add(x = var_8943_cast_fp16, y = var_8945_cast_fp16)[name = string("V_new_27_cast_fp16")]; tensor var_8949_axes_0 = const()[name = string("op_8949_axes_0"), val = tensor([0])]; tensor var_8949_cast_fp16 = squeeze(axes = var_8949_axes_0, x = K_new_27_cast_fp16)[name = string("op_8949_cast_fp16")]; tensor concat_104 = const()[name = string("concat_104"), val = tensor([13, 0, 0, 0])]; tensor concat_105 = const()[name = string("concat_105"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_104, begin_mask = kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_105, end_mask = kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_27_stride_0, update = var_8949_cast_fp16, x = coreml_update_state_55)[name = string("kv_cache_0_internal_tensor_assign_27_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_27_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_56_write_state")]; tensor coreml_update_state_56 = read_state(input = kv_cache_0)[name = string("coreml_update_state_56")]; tensor var_8956_axes_0 = const()[name = string("op_8956_axes_0"), val = tensor([0])]; tensor var_8956_cast_fp16 = squeeze(axes = var_8956_axes_0, x = V_new_27_cast_fp16)[name = string("op_8956_cast_fp16")]; tensor concat_106 = const()[name = string("concat_106"), val = tensor([48, 0, 0, 0])]; tensor concat_107 = const()[name = string("concat_107"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_106, begin_mask = kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_107, end_mask = kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_28_stride_0, update = var_8956_cast_fp16, x = coreml_update_state_56)[name = string("kv_cache_0_internal_tensor_assign_28_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_28_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_57_write_state")]; tensor coreml_update_state_57 = read_state(input = kv_cache_0)[name = string("coreml_update_state_57")]; tensor K_for_attn_27_begin_0 = const()[name = string("K_for_attn_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_27_end_0 = const()[name = string("K_for_attn_27_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_for_attn_27_end_mask_0 = const()[name = string("K_for_attn_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_27_cast_fp16 = slice_by_index(begin = K_for_attn_27_begin_0, end = K_for_attn_27_end_0, end_mask = K_for_attn_27_end_mask_0, x = K_new_27_cast_fp16)[name = string("K_for_attn_27_cast_fp16")]; tensor V_for_attn_27_begin_0 = const()[name = string("V_for_attn_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_27_end_0 = const()[name = string("V_for_attn_27_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_27_end_mask_0 = const()[name = string("V_for_attn_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_27_cast_fp16 = slice_by_index(begin = V_for_attn_27_begin_0, end = V_for_attn_27_end_0, end_mask = V_for_attn_27_end_mask_0, x = V_new_27_cast_fp16)[name = string("V_for_attn_27_cast_fp16")]; tensor transpose_52_perm_0 = const()[name = string("transpose_52_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_26_reps_0 = const()[name = string("tile_26_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_52_cast_fp16 = transpose(perm = transpose_52_perm_0, x = K_for_attn_27_cast_fp16)[name = string("transpose_167")]; tensor tile_26_cast_fp16 = tile(reps = tile_26_reps_0, x = transpose_52_cast_fp16)[name = string("tile_26_cast_fp16")]; tensor concat_108 = const()[name = string("concat_108"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_52_cast_fp16 = reshape(shape = concat_108, x = tile_26_cast_fp16)[name = string("reshape_52_cast_fp16")]; tensor transpose_53_perm_0 = const()[name = string("transpose_53_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_109 = const()[name = string("concat_109"), val = tensor([-1, 1, 512, 256])]; tensor transpose_53_cast_fp16 = transpose(perm = transpose_53_perm_0, x = reshape_52_cast_fp16)[name = string("transpose_166")]; tensor reshape_53_cast_fp16 = reshape(shape = concat_109, x = transpose_53_cast_fp16)[name = string("reshape_53_cast_fp16")]; tensor transpose_153_perm_0 = const()[name = string("transpose_153_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_54_perm_0 = const()[name = string("transpose_54_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_27_reps_0 = const()[name = string("tile_27_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_54_cast_fp16 = transpose(perm = transpose_54_perm_0, x = V_for_attn_27_cast_fp16)[name = string("transpose_165")]; tensor tile_27_cast_fp16 = tile(reps = tile_27_reps_0, x = transpose_54_cast_fp16)[name = string("tile_27_cast_fp16")]; tensor concat_110 = const()[name = string("concat_110"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_54_cast_fp16 = reshape(shape = concat_110, x = tile_27_cast_fp16)[name = string("reshape_54_cast_fp16")]; tensor transpose_55_perm_0 = const()[name = string("transpose_55_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_111 = const()[name = string("concat_111"), val = tensor([-1, 1, 512, 256])]; tensor transpose_55_cast_fp16 = transpose(perm = transpose_55_perm_0, x = reshape_54_cast_fp16)[name = string("transpose_164")]; tensor reshape_55_cast_fp16 = reshape(shape = concat_111, x = transpose_55_cast_fp16)[name = string("reshape_55_cast_fp16")]; tensor V_expanded_27_perm_0 = const()[name = string("V_expanded_27_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_8993_transpose_x_0 = const()[name = string("op_8993_transpose_x_0"), val = bool(false)]; bool var_8993_transpose_y_0 = const()[name = string("op_8993_transpose_y_0"), val = bool(false)]; tensor transpose_153_cast_fp16 = transpose(perm = transpose_153_perm_0, x = reshape_53_cast_fp16)[name = string("transpose_163")]; tensor var_8993_cast_fp16 = matmul(transpose_x = var_8993_transpose_x_0, transpose_y = var_8993_transpose_y_0, x = q_111, y = transpose_153_cast_fp16)[name = string("op_8993_cast_fp16")]; tensor attn_weights_81_cast_fp16 = add(x = var_8993_cast_fp16, y = causal_mask)[name = string("attn_weights_81_cast_fp16")]; int32 var_8998 = const()[name = string("op_8998"), val = int32(-1)]; tensor attn_weights_83_cast_fp16 = softmax(axis = var_8998, x = attn_weights_81_cast_fp16)[name = string("attn_weights_83_cast_fp16")]; bool attn_output_79_transpose_x_0 = const()[name = string("attn_output_79_transpose_x_0"), val = bool(false)]; bool attn_output_79_transpose_y_0 = const()[name = string("attn_output_79_transpose_y_0"), val = bool(false)]; tensor V_expanded_27_cast_fp16 = transpose(perm = V_expanded_27_perm_0, x = reshape_55_cast_fp16)[name = string("transpose_162")]; tensor attn_output_79_cast_fp16 = matmul(transpose_x = attn_output_79_transpose_x_0, transpose_y = attn_output_79_transpose_y_0, x = attn_weights_83_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor var_9006 = const()[name = string("op_9006"), val = tensor([0, 2, 1, 3])]; tensor var_9013 = const()[name = string("op_9013"), val = tensor([1, 1, -1])]; tensor var_9007_cast_fp16 = transpose(perm = var_9006, x = attn_output_79_cast_fp16)[name = string("transpose_161")]; tensor attn_output_81_cast_fp16 = reshape(shape = var_9013, x = var_9007_cast_fp16)[name = string("attn_output_81_cast_fp16")]; tensor var_9018 = const()[name = string("op_9018"), val = tensor([0, 2, 1])]; string var_9034_pad_type_0 = const()[name = string("op_9034_pad_type_0"), val = string("valid")]; int32 var_9034_groups_0 = const()[name = string("op_9034_groups_0"), val = int32(1)]; tensor var_9034_strides_0 = const()[name = string("op_9034_strides_0"), val = tensor([1])]; tensor var_9034_pad_0 = const()[name = string("op_9034_pad_0"), val = tensor([0, 0])]; tensor var_9034_dilations_0 = const()[name = string("op_9034_dilations_0"), val = tensor([1])]; tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1088244672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1089817600))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9019_cast_fp16 = transpose(perm = var_9018, x = attn_output_81_cast_fp16)[name = string("transpose_160")]; tensor var_9034_cast_fp16 = conv(dilations = var_9034_dilations_0, groups = var_9034_groups_0, pad = var_9034_pad_0, pad_type = var_9034_pad_type_0, strides = var_9034_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_9019_cast_fp16)[name = string("op_9034_cast_fp16")]; tensor var_9038 = const()[name = string("op_9038"), val = tensor([0, 2, 1])]; int32 var_9044 = const()[name = string("op_9044"), val = int32(-1)]; fp16 const_241_promoted_to_fp16 = const()[name = string("const_241_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_405_cast_fp16 = transpose(perm = var_9038, x = var_9034_cast_fp16)[name = string("transpose_159")]; tensor var_9050_cast_fp16 = mul(x = x_405_cast_fp16, y = const_241_promoted_to_fp16)[name = string("op_9050_cast_fp16")]; bool input_397_interleave_0 = const()[name = string("input_397_interleave_0"), val = bool(false)]; tensor input_397_cast_fp16 = concat(axis = var_9044, interleave = input_397_interleave_0, values = (x_405_cast_fp16, var_9050_cast_fp16))[name = string("input_397_cast_fp16")]; tensor normed_377_axes_0 = const()[name = string("normed_377_axes_0"), val = tensor([-1])]; fp16 var_9042_to_fp16 = const()[name = string("op_9042_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_377_cast_fp16 = layer_norm(axes = normed_377_axes_0, epsilon = var_9042_to_fp16, x = input_397_cast_fp16)[name = string("normed_377_cast_fp16")]; tensor var_9055_split_sizes_0 = const()[name = string("op_9055_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9055_axis_0 = const()[name = string("op_9055_axis_0"), val = int32(-1)]; tensor var_9055_cast_fp16_0, tensor var_9055_cast_fp16_1 = split(axis = var_9055_axis_0, split_sizes = var_9055_split_sizes_0, x = normed_377_cast_fp16)[name = string("op_9055_cast_fp16")]; tensor const_242_to_fp16 = const()[name = string("const_242_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1089819200)))]; tensor var_9058_cast_fp16 = mul(x = var_9055_cast_fp16_0, y = const_242_to_fp16)[name = string("op_9058_cast_fp16")]; tensor x_409_cast_fp16 = add(x = x_391_cast_fp16, y = var_9058_cast_fp16)[name = string("x_409_cast_fp16")]; int32 var_9065 = const()[name = string("op_9065"), val = int32(-1)]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9071_cast_fp16 = mul(x = x_409_cast_fp16, y = const_243_promoted_to_fp16)[name = string("op_9071_cast_fp16")]; bool input_399_interleave_0 = const()[name = string("input_399_interleave_0"), val = bool(false)]; tensor input_399_cast_fp16 = concat(axis = var_9065, interleave = input_399_interleave_0, values = (x_409_cast_fp16, var_9071_cast_fp16))[name = string("input_399_cast_fp16")]; tensor normed_381_axes_0 = const()[name = string("normed_381_axes_0"), val = tensor([-1])]; fp16 var_9063_to_fp16 = const()[name = string("op_9063_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_381_cast_fp16 = layer_norm(axes = normed_381_axes_0, epsilon = var_9063_to_fp16, x = input_399_cast_fp16)[name = string("normed_381_cast_fp16")]; tensor var_9076_split_sizes_0 = const()[name = string("op_9076_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9076_axis_0 = const()[name = string("op_9076_axis_0"), val = int32(-1)]; tensor var_9076_cast_fp16_0, tensor var_9076_cast_fp16_1 = split(axis = var_9076_axis_0, split_sizes = var_9076_split_sizes_0, x = normed_381_cast_fp16)[name = string("op_9076_cast_fp16")]; tensor const_244_to_fp16 = const()[name = string("const_244_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1089822336)))]; tensor var_9079_cast_fp16 = mul(x = var_9076_cast_fp16_0, y = const_244_to_fp16)[name = string("op_9079_cast_fp16")]; tensor var_9092 = const()[name = string("op_9092"), val = tensor([0, 2, 1])]; tensor input_401_axes_0 = const()[name = string("input_401_axes_0"), val = tensor([2])]; tensor var_9093 = transpose(perm = var_9092, x = var_9079_cast_fp16)[name = string("transpose_158")]; tensor input_401 = expand_dims(axes = input_401_axes_0, x = var_9093)[name = string("input_401")]; string gate_53_pad_type_0 = const()[name = string("gate_53_pad_type_0"), val = string("valid")]; tensor gate_53_strides_0 = const()[name = string("gate_53_strides_0"), val = tensor([1, 1])]; tensor gate_53_pad_0 = const()[name = string("gate_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_53_dilations_0 = const()[name = string("gate_53_dilations_0"), val = tensor([1, 1])]; int32 gate_53_groups_0 = const()[name = string("gate_53_groups_0"), val = int32(1)]; tensor gate_53 = conv(dilations = gate_53_dilations_0, groups = gate_53_groups_0, pad = gate_53_pad_0, pad_type = gate_53_pad_type_0, strides = gate_53_strides_0, weight = layers_13_mlp_gate_proj_weight_palettized, x = input_401)[name = string("gate_53")]; string up_27_pad_type_0 = const()[name = string("up_27_pad_type_0"), val = string("valid")]; tensor up_27_strides_0 = const()[name = string("up_27_strides_0"), val = tensor([1, 1])]; tensor up_27_pad_0 = const()[name = string("up_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_27_dilations_0 = const()[name = string("up_27_dilations_0"), val = tensor([1, 1])]; int32 up_27_groups_0 = const()[name = string("up_27_groups_0"), val = int32(1)]; tensor up_27 = conv(dilations = up_27_dilations_0, groups = up_27_groups_0, pad = up_27_pad_0, pad_type = up_27_pad_type_0, strides = up_27_strides_0, weight = layers_13_mlp_up_proj_weight_palettized, x = input_401)[name = string("up_27")]; string gate_55_mode_0 = const()[name = string("gate_55_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_55 = gelu(mode = gate_55_mode_0, x = gate_53)[name = string("gate_55")]; tensor input_403 = mul(x = gate_55, y = up_27)[name = string("input_403")]; string mlp_out_27_pad_type_0 = const()[name = string("mlp_out_27_pad_type_0"), val = string("valid")]; tensor mlp_out_27_strides_0 = const()[name = string("mlp_out_27_strides_0"), val = tensor([1, 1])]; tensor mlp_out_27_pad_0 = const()[name = string("mlp_out_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_27_dilations_0 = const()[name = string("mlp_out_27_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_27_groups_0 = const()[name = string("mlp_out_27_groups_0"), val = int32(1)]; tensor mlp_out_27 = conv(dilations = mlp_out_27_dilations_0, groups = mlp_out_27_groups_0, pad = mlp_out_27_pad_0, pad_type = mlp_out_27_pad_type_0, strides = mlp_out_27_strides_0, weight = layers_13_mlp_down_proj_weight_palettized, x = input_403)[name = string("mlp_out_27")]; tensor var_9133_axes_0 = const()[name = string("op_9133_axes_0"), val = tensor([2])]; tensor var_9133 = squeeze(axes = var_9133_axes_0, x = mlp_out_27)[name = string("op_9133")]; tensor var_9137 = const()[name = string("op_9137"), val = tensor([0, 2, 1])]; int32 var_9143 = const()[name = string("op_9143"), val = int32(-1)]; fp16 const_245_promoted_to_fp16 = const()[name = string("const_245_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_413 = transpose(perm = var_9137, x = var_9133)[name = string("transpose_157")]; tensor var_9149_cast_fp16 = mul(x = x_413, y = const_245_promoted_to_fp16)[name = string("op_9149_cast_fp16")]; bool input_405_interleave_0 = const()[name = string("input_405_interleave_0"), val = bool(false)]; tensor input_405_cast_fp16 = concat(axis = var_9143, interleave = input_405_interleave_0, values = (x_413, var_9149_cast_fp16))[name = string("input_405_cast_fp16")]; tensor normed_385_axes_0 = const()[name = string("normed_385_axes_0"), val = tensor([-1])]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_385_cast_fp16 = layer_norm(axes = normed_385_axes_0, epsilon = var_9141_to_fp16, x = input_405_cast_fp16)[name = string("normed_385_cast_fp16")]; tensor var_9154_split_sizes_0 = const()[name = string("op_9154_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9154_axis_0 = const()[name = string("op_9154_axis_0"), val = int32(-1)]; tensor var_9154_cast_fp16_0, tensor var_9154_cast_fp16_1 = split(axis = var_9154_axis_0, split_sizes = var_9154_split_sizes_0, x = normed_385_cast_fp16)[name = string("op_9154_cast_fp16")]; tensor const_246_to_fp16 = const()[name = string("const_246_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1089825472)))]; tensor var_9157_cast_fp16 = mul(x = var_9154_cast_fp16_0, y = const_246_to_fp16)[name = string("op_9157_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = x_409_cast_fp16, y = var_9157_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; tensor per_layer_slice_27_begin_0 = const()[name = string("per_layer_slice_27_begin_0"), val = tensor([0, 0, 3328])]; tensor per_layer_slice_27_end_0 = const()[name = string("per_layer_slice_27_end_0"), val = tensor([1, 1, 3584])]; tensor per_layer_slice_27_end_mask_0 = const()[name = string("per_layer_slice_27_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_27_cast_fp16 = slice_by_index(begin = per_layer_slice_27_begin_0, end = per_layer_slice_27_end_0, end_mask = per_layer_slice_27_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_27_cast_fp16")]; tensor gated_53 = linear(bias = linear_0_bias_0, weight = layers_13_per_layer_input_gate_weight_palettized, x = hidden_states_165_cast_fp16)[name = string("linear_26")]; string gated_55_mode_0 = const()[name = string("gated_55_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_55 = gelu(mode = gated_55_mode_0, x = gated_53)[name = string("gated_55")]; tensor input_409_cast_fp16 = mul(x = gated_55, y = per_layer_slice_27_cast_fp16)[name = string("input_409_cast_fp16")]; tensor layers_13_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1089828608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1090025280))))[name = string("layers_13_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_27_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_13_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_409_cast_fp16)[name = string("linear_27_cast_fp16")]; int32 var_9194 = const()[name = string("op_9194"), val = int32(-1)]; fp16 const_247_promoted_to_fp16 = const()[name = string("const_247_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9200_cast_fp16 = mul(x = linear_27_cast_fp16, y = const_247_promoted_to_fp16)[name = string("op_9200_cast_fp16")]; bool input_411_interleave_0 = const()[name = string("input_411_interleave_0"), val = bool(false)]; tensor input_411_cast_fp16 = concat(axis = var_9194, interleave = input_411_interleave_0, values = (linear_27_cast_fp16, var_9200_cast_fp16))[name = string("input_411_cast_fp16")]; tensor normed_389_axes_0 = const()[name = string("normed_389_axes_0"), val = tensor([-1])]; fp16 var_9192_to_fp16 = const()[name = string("op_9192_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_389_cast_fp16 = layer_norm(axes = normed_389_axes_0, epsilon = var_9192_to_fp16, x = input_411_cast_fp16)[name = string("normed_389_cast_fp16")]; tensor var_9205_split_sizes_0 = const()[name = string("op_9205_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9205_axis_0 = const()[name = string("op_9205_axis_0"), val = int32(-1)]; tensor var_9205_cast_fp16_0, tensor var_9205_cast_fp16_1 = split(axis = var_9205_axis_0, split_sizes = var_9205_split_sizes_0, x = normed_389_cast_fp16)[name = string("op_9205_cast_fp16")]; tensor const_248_to_fp16 = const()[name = string("const_248_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1090026880)))]; tensor var_9208_cast_fp16 = mul(x = var_9205_cast_fp16_0, y = const_248_to_fp16)[name = string("op_9208_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = var_9208_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; tensor layers_13_layer_scalar_to_fp16 = const()[name = string("layers_13_layer_scalar_to_fp16"), val = tensor([0x1.6ap-4])]; tensor x_421_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = layers_13_layer_scalar_to_fp16)[name = string("x_421_cast_fp16")]; int32 var_9216 = const()[name = string("op_9216"), val = int32(-1)]; fp16 const_249_promoted_to_fp16 = const()[name = string("const_249_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9222_cast_fp16 = mul(x = x_421_cast_fp16, y = const_249_promoted_to_fp16)[name = string("op_9222_cast_fp16")]; bool input_413_interleave_0 = const()[name = string("input_413_interleave_0"), val = bool(false)]; tensor input_413_cast_fp16 = concat(axis = var_9216, interleave = input_413_interleave_0, values = (x_421_cast_fp16, var_9222_cast_fp16))[name = string("input_413_cast_fp16")]; tensor normed_393_axes_0 = const()[name = string("normed_393_axes_0"), val = tensor([-1])]; fp16 var_9214_to_fp16 = const()[name = string("op_9214_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_393_cast_fp16 = layer_norm(axes = normed_393_axes_0, epsilon = var_9214_to_fp16, x = input_413_cast_fp16)[name = string("normed_393_cast_fp16")]; tensor var_9227_split_sizes_0 = const()[name = string("op_9227_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9227_axis_0 = const()[name = string("op_9227_axis_0"), val = int32(-1)]; tensor var_9227_cast_fp16_0, tensor var_9227_cast_fp16_1 = split(axis = var_9227_axis_0, split_sizes = var_9227_split_sizes_0, x = normed_393_cast_fp16)[name = string("op_9227_cast_fp16")]; tensor const_250_to_fp16 = const()[name = string("const_250_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1090030016)))]; tensor var_9230_cast_fp16 = mul(x = var_9227_cast_fp16_0, y = const_250_to_fp16)[name = string("op_9230_cast_fp16")]; tensor var_9238 = const()[name = string("op_9238"), val = tensor([0, 2, 1])]; tensor var_9241_axes_0 = const()[name = string("op_9241_axes_0"), val = tensor([2])]; tensor var_9239_cast_fp16 = transpose(perm = var_9238, x = var_9230_cast_fp16)[name = string("transpose_156")]; tensor var_9241_cast_fp16 = expand_dims(axes = var_9241_axes_0, x = var_9239_cast_fp16)[name = string("op_9241_cast_fp16")]; string var_9257_pad_type_0 = const()[name = string("op_9257_pad_type_0"), val = string("valid")]; tensor var_9257_strides_0 = const()[name = string("op_9257_strides_0"), val = tensor([1, 1])]; tensor var_9257_pad_0 = const()[name = string("op_9257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9257_dilations_0 = const()[name = string("op_9257_dilations_0"), val = tensor([1, 1])]; int32 var_9257_groups_0 = const()[name = string("op_9257_groups_0"), val = int32(1)]; tensor var_9257 = conv(dilations = var_9257_dilations_0, groups = var_9257_groups_0, pad = var_9257_pad_0, pad_type = var_9257_pad_type_0, strides = var_9257_strides_0, weight = layers_14_self_attn_q_proj_weight_palettized, x = var_9241_cast_fp16)[name = string("op_9257")]; tensor var_9262 = const()[name = string("op_9262"), val = tensor([1, 8, 512, 1])]; tensor var_9263 = reshape(shape = var_9262, x = var_9257)[name = string("op_9263")]; tensor var_9268 = const()[name = string("op_9268"), val = tensor([0, 1, 3, 2])]; tensor var_9278 = const()[name = string("op_9278"), val = tensor([1, 8, 512])]; tensor var_9269 = transpose(perm = var_9268, x = var_9263)[name = string("transpose_155")]; tensor x_425 = reshape(shape = var_9278, x = var_9269)[name = string("x_425")]; int32 var_9284 = const()[name = string("op_9284"), val = int32(-1)]; fp16 const_251_promoted_to_fp16 = const()[name = string("const_251_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9290_cast_fp16 = mul(x = x_425, y = const_251_promoted_to_fp16)[name = string("op_9290_cast_fp16")]; bool input_417_interleave_0 = const()[name = string("input_417_interleave_0"), val = bool(false)]; tensor input_417_cast_fp16 = concat(axis = var_9284, interleave = input_417_interleave_0, values = (x_425, var_9290_cast_fp16))[name = string("input_417_cast_fp16")]; tensor normed_397_axes_0 = const()[name = string("normed_397_axes_0"), val = tensor([-1])]; fp16 var_9282_to_fp16 = const()[name = string("op_9282_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_397_cast_fp16 = layer_norm(axes = normed_397_axes_0, epsilon = var_9282_to_fp16, x = input_417_cast_fp16)[name = string("normed_397_cast_fp16")]; tensor var_9295_split_sizes_0 = const()[name = string("op_9295_split_sizes_0"), val = tensor([512, 512])]; int32 var_9295_axis_0 = const()[name = string("op_9295_axis_0"), val = int32(-1)]; tensor var_9295_cast_fp16_0, tensor var_9295_cast_fp16_1 = split(axis = var_9295_axis_0, split_sizes = var_9295_split_sizes_0, x = normed_397_cast_fp16)[name = string("op_9295_cast_fp16")]; tensor const_252_to_fp16 = const()[name = string("const_252_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1090033152)))]; tensor var_9298_cast_fp16 = mul(x = var_9295_cast_fp16_0, y = const_252_to_fp16)[name = string("op_9298_cast_fp16")]; tensor var_9304 = const()[name = string("op_9304"), val = tensor([1, 8, 1, 512])]; tensor q_115 = reshape(shape = var_9304, x = var_9298_cast_fp16)[name = string("q_115")]; tensor var_9306 = mul(x = q_115, y = cos)[name = string("op_9306")]; tensor var_9307_split_sizes_0 = const()[name = string("op_9307_split_sizes_0"), val = tensor([256, 256])]; int32 var_9307_axis_0 = const()[name = string("op_9307_axis_0"), val = int32(-1)]; tensor var_9307_0, tensor var_9307_1 = split(axis = var_9307_axis_0, split_sizes = var_9307_split_sizes_0, x = q_115)[name = string("op_9307")]; fp16 const_253_promoted = const()[name = string("const_253_promoted"), val = fp16(-0x1p+0)]; tensor var_9309 = mul(x = var_9307_1, y = const_253_promoted)[name = string("op_9309")]; int32 var_9311 = const()[name = string("op_9311"), val = int32(-1)]; bool var_9312_interleave_0 = const()[name = string("op_9312_interleave_0"), val = bool(false)]; tensor var_9312 = concat(axis = var_9311, interleave = var_9312_interleave_0, values = (var_9309, var_9307_0))[name = string("op_9312")]; tensor var_9313 = mul(x = var_9312, y = sin)[name = string("op_9313")]; tensor q_119 = add(x = var_9306, y = var_9313)[name = string("q_119")]; string var_9326_pad_type_0 = const()[name = string("op_9326_pad_type_0"), val = string("valid")]; tensor var_9326_strides_0 = const()[name = string("op_9326_strides_0"), val = tensor([1, 1])]; tensor var_9326_pad_0 = const()[name = string("op_9326_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9326_dilations_0 = const()[name = string("op_9326_dilations_0"), val = tensor([1, 1])]; int32 var_9326_groups_0 = const()[name = string("op_9326_groups_0"), val = int32(1)]; tensor var_9326 = conv(dilations = var_9326_dilations_0, groups = var_9326_groups_0, pad = var_9326_pad_0, pad_type = var_9326_pad_type_0, strides = var_9326_strides_0, weight = layers_14_self_attn_k_proj_weight_palettized, x = var_9241_cast_fp16)[name = string("op_9326")]; tensor var_9331 = const()[name = string("op_9331"), val = tensor([1, 1, 512, 1])]; tensor var_9332 = reshape(shape = var_9331, x = var_9326)[name = string("op_9332")]; tensor var_9337 = const()[name = string("op_9337"), val = tensor([0, 1, 3, 2])]; string var_9354_pad_type_0 = const()[name = string("op_9354_pad_type_0"), val = string("valid")]; tensor var_9354_strides_0 = const()[name = string("op_9354_strides_0"), val = tensor([1, 1])]; tensor var_9354_pad_0 = const()[name = string("op_9354_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9354_dilations_0 = const()[name = string("op_9354_dilations_0"), val = tensor([1, 1])]; int32 var_9354_groups_0 = const()[name = string("op_9354_groups_0"), val = int32(1)]; tensor var_9354 = conv(dilations = var_9354_dilations_0, groups = var_9354_groups_0, pad = var_9354_pad_0, pad_type = var_9354_pad_type_0, strides = var_9354_strides_0, weight = layers_14_self_attn_v_proj_weight_palettized, x = var_9241_cast_fp16)[name = string("op_9354")]; tensor var_9359 = const()[name = string("op_9359"), val = tensor([1, 1, 512, 1])]; tensor var_9360 = reshape(shape = var_9359, x = var_9354)[name = string("op_9360")]; tensor var_9365 = const()[name = string("op_9365"), val = tensor([0, 1, 3, 2])]; tensor var_9375 = const()[name = string("op_9375"), val = tensor([1, 1, 512])]; tensor var_9338 = transpose(perm = var_9337, x = var_9332)[name = string("transpose_154")]; tensor x_429 = reshape(shape = var_9375, x = var_9338)[name = string("x_429")]; int32 var_9381 = const()[name = string("op_9381"), val = int32(-1)]; fp16 const_254_promoted_to_fp16 = const()[name = string("const_254_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9387_cast_fp16 = mul(x = x_429, y = const_254_promoted_to_fp16)[name = string("op_9387_cast_fp16")]; bool input_419_interleave_0 = const()[name = string("input_419_interleave_0"), val = bool(false)]; tensor input_419_cast_fp16 = concat(axis = var_9381, interleave = input_419_interleave_0, values = (x_429, var_9387_cast_fp16))[name = string("input_419_cast_fp16")]; tensor normed_401_axes_0 = const()[name = string("normed_401_axes_0"), val = tensor([-1])]; fp16 var_9379_to_fp16 = const()[name = string("op_9379_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_401_cast_fp16 = layer_norm(axes = normed_401_axes_0, epsilon = var_9379_to_fp16, x = input_419_cast_fp16)[name = string("normed_401_cast_fp16")]; tensor var_9392_split_sizes_0 = const()[name = string("op_9392_split_sizes_0"), val = tensor([512, 512])]; int32 var_9392_axis_0 = const()[name = string("op_9392_axis_0"), val = int32(-1)]; tensor var_9392_cast_fp16_0, tensor var_9392_cast_fp16_1 = split(axis = var_9392_axis_0, split_sizes = var_9392_split_sizes_0, x = normed_401_cast_fp16)[name = string("op_9392_cast_fp16")]; tensor const_255_to_fp16 = const()[name = string("const_255_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1090034240)))]; tensor var_9395_cast_fp16 = mul(x = var_9392_cast_fp16_0, y = const_255_to_fp16)[name = string("op_9395_cast_fp16")]; tensor var_9401 = const()[name = string("op_9401"), val = tensor([1, 1, 1, 512])]; tensor q_117 = reshape(shape = var_9401, x = var_9395_cast_fp16)[name = string("q_117")]; fp16 var_9408_promoted_to_fp16 = const()[name = string("op_9408_promoted_to_fp16"), val = fp16(0x1p+1)]; tensor var_9366 = transpose(perm = var_9365, x = var_9360)[name = string("transpose_153")]; tensor var_9409_cast_fp16 = pow(x = var_9366, y = var_9408_promoted_to_fp16)[name = string("op_9409_cast_fp16")]; tensor var_9414_axes_0 = const()[name = string("op_9414_axes_0"), val = tensor([-1])]; bool var_9414_keep_dims_0 = const()[name = string("op_9414_keep_dims_0"), val = bool(true)]; tensor var_9414_cast_fp16 = reduce_mean(axes = var_9414_axes_0, keep_dims = var_9414_keep_dims_0, x = var_9409_cast_fp16)[name = string("op_9414_cast_fp16")]; fp16 var_9416_to_fp16 = const()[name = string("op_9416_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_cast_fp16 = add(x = var_9414_cast_fp16, y = var_9416_to_fp16)[name = string("mean_sq_cast_fp16")]; fp16 var_9423_to_fp16 = const()[name = string("op_9423_to_fp16"), val = fp16(-0x1p-1)]; tensor var_9424_cast_fp16 = pow(x = mean_sq_cast_fp16, y = var_9423_to_fp16)[name = string("op_9424_cast_fp16")]; tensor var_9425_cast_fp16 = mul(x = var_9366, y = var_9424_cast_fp16)[name = string("op_9425_cast_fp16")]; tensor var_9431 = mul(x = q_117, y = cos)[name = string("op_9431")]; tensor var_9432_split_sizes_0 = const()[name = string("op_9432_split_sizes_0"), val = tensor([256, 256])]; int32 var_9432_axis_0 = const()[name = string("op_9432_axis_0"), val = int32(-1)]; tensor var_9432_0, tensor var_9432_1 = split(axis = var_9432_axis_0, split_sizes = var_9432_split_sizes_0, x = q_117)[name = string("op_9432")]; fp16 const_256_promoted = const()[name = string("const_256_promoted"), val = fp16(-0x1p+0)]; tensor var_9434 = mul(x = var_9432_1, y = const_256_promoted)[name = string("op_9434")]; int32 var_9436 = const()[name = string("op_9436"), val = int32(-1)]; bool var_9437_interleave_0 = const()[name = string("op_9437_interleave_0"), val = bool(false)]; tensor var_9437 = concat(axis = var_9436, interleave = var_9437_interleave_0, values = (var_9434, var_9432_0))[name = string("op_9437")]; tensor var_9438 = mul(x = var_9437, y = sin)[name = string("op_9438")]; tensor k = add(x = var_9431, y = var_9438)[name = string("k")]; tensor var_9443_begin_0 = const()[name = string("op_9443_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_9443_end_0 = const()[name = string("op_9443_end_0"), val = tensor([15, 1, 512, 512])]; tensor var_9443_end_mask_0 = const()[name = string("op_9443_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9443_squeeze_mask_0 = const()[name = string("op_9443_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_9443_cast_fp16 = slice_by_index(begin = var_9443_begin_0, end = var_9443_end_0, end_mask = var_9443_end_mask_0, squeeze_mask = var_9443_squeeze_mask_0, x = coreml_update_state_57)[name = string("op_9443_cast_fp16")]; tensor K_cache_axes_0 = const()[name = string("K_cache_axes_0"), val = tensor([0])]; tensor K_cache_cast_fp16 = expand_dims(axes = K_cache_axes_0, x = var_9443_cast_fp16)[name = string("K_cache_cast_fp16")]; tensor var_9448_begin_0 = const()[name = string("op_9448_begin_0"), val = tensor([49, 0, 0, 0])]; tensor var_9448_end_0 = const()[name = string("op_9448_end_0"), val = tensor([50, 1, 512, 512])]; tensor var_9448_end_mask_0 = const()[name = string("op_9448_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9448_squeeze_mask_0 = const()[name = string("op_9448_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_9448_cast_fp16 = slice_by_index(begin = var_9448_begin_0, end = var_9448_end_0, end_mask = var_9448_end_mask_0, squeeze_mask = var_9448_squeeze_mask_0, x = coreml_update_state_57)[name = string("op_9448_cast_fp16")]; tensor V_cache_axes_0 = const()[name = string("V_cache_axes_0"), val = tensor([0])]; tensor V_cache_cast_fp16 = expand_dims(axes = V_cache_axes_0, x = var_9448_cast_fp16)[name = string("V_cache_cast_fp16")]; tensor var_9454_cast_fp16 = mul(x = K_cache_cast_fp16, y = var_2187_cast_fp16)[name = string("op_9454_cast_fp16")]; tensor var_9455_reps_0 = const()[name = string("op_9455_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_9455 = tile(reps = var_9455_reps_0, x = k)[name = string("op_9455")]; tensor var_9456_cast_fp16 = mul(x = var_9455, y = update_mask)[name = string("op_9456_cast_fp16")]; tensor K_new_cast_fp16 = add(x = var_9454_cast_fp16, y = var_9456_cast_fp16)[name = string("K_new_cast_fp16")]; tensor var_9462_cast_fp16 = mul(x = V_cache_cast_fp16, y = var_2187_cast_fp16)[name = string("op_9462_cast_fp16")]; tensor var_9463_reps_0 = const()[name = string("op_9463_reps_0"), val = tensor([1, 1, 512, 1])]; tensor var_9463 = tile(reps = var_9463_reps_0, x = var_9425_cast_fp16)[name = string("op_9463")]; tensor var_9464_cast_fp16 = mul(x = var_9463, y = update_mask)[name = string("op_9464_cast_fp16")]; tensor V_new_cast_fp16 = add(x = var_9462_cast_fp16, y = var_9464_cast_fp16)[name = string("V_new_cast_fp16")]; tensor var_9468_axes_0 = const()[name = string("op_9468_axes_0"), val = tensor([0])]; tensor var_9468_cast_fp16 = squeeze(axes = var_9468_axes_0, x = K_new_cast_fp16)[name = string("op_9468_cast_fp16")]; tensor concat_112 = const()[name = string("concat_112"), val = tensor([14, 0, 0, 0])]; tensor concat_113 = const()[name = string("concat_113"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_29_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_29_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_29_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_29_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_29_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_29_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_29_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_29_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_29_cast_fp16 = slice_update(begin = concat_112, begin_mask = kv_cache_0_internal_tensor_assign_29_begin_mask_0, end = concat_113, end_mask = kv_cache_0_internal_tensor_assign_29_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_29_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_29_stride_0, update = var_9468_cast_fp16, x = coreml_update_state_57)[name = string("kv_cache_0_internal_tensor_assign_29_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_29_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_58_write_state")]; tensor coreml_update_state_58 = read_state(input = kv_cache_0)[name = string("coreml_update_state_58")]; tensor var_9475_axes_0 = const()[name = string("op_9475_axes_0"), val = tensor([0])]; tensor var_9475_cast_fp16 = squeeze(axes = var_9475_axes_0, x = V_new_cast_fp16)[name = string("op_9475_cast_fp16")]; tensor concat_114 = const()[name = string("concat_114"), val = tensor([49, 0, 0, 0])]; tensor concat_115 = const()[name = string("concat_115"), val = tensor([0, 0, 0, 0])]; tensor kv_cache_0_internal_tensor_assign_30_stride_0 = const()[name = string("kv_cache_0_internal_tensor_assign_30_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_0_internal_tensor_assign_30_begin_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_30_begin_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_30_end_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_30_end_mask_0"), val = tensor([false, true, true, true])]; tensor kv_cache_0_internal_tensor_assign_30_squeeze_mask_0 = const()[name = string("kv_cache_0_internal_tensor_assign_30_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor kv_cache_0_internal_tensor_assign_30_cast_fp16 = slice_update(begin = concat_114, begin_mask = kv_cache_0_internal_tensor_assign_30_begin_mask_0, end = concat_115, end_mask = kv_cache_0_internal_tensor_assign_30_end_mask_0, squeeze_mask = kv_cache_0_internal_tensor_assign_30_squeeze_mask_0, stride = kv_cache_0_internal_tensor_assign_30_stride_0, update = var_9475_cast_fp16, x = coreml_update_state_58)[name = string("kv_cache_0_internal_tensor_assign_30_cast_fp16")]; write_state(data = kv_cache_0_internal_tensor_assign_30_cast_fp16, input = kv_cache_0)[name = string("coreml_update_state_59_write_state")]; tensor transpose_56_perm_0 = const()[name = string("transpose_56_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_28_reps_0 = const()[name = string("tile_28_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_56_cast_fp16 = transpose(perm = transpose_56_perm_0, x = K_new_cast_fp16)[name = string("transpose_152")]; tensor tile_28_cast_fp16 = tile(reps = tile_28_reps_0, x = transpose_56_cast_fp16)[name = string("tile_28_cast_fp16")]; tensor concat_116 = const()[name = string("concat_116"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_56_cast_fp16 = reshape(shape = concat_116, x = tile_28_cast_fp16)[name = string("reshape_56_cast_fp16")]; tensor transpose_57_perm_0 = const()[name = string("transpose_57_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_117 = const()[name = string("concat_117"), val = tensor([-1, 1, 512, 512])]; tensor transpose_57_cast_fp16 = transpose(perm = transpose_57_perm_0, x = reshape_56_cast_fp16)[name = string("transpose_151")]; tensor reshape_57_cast_fp16 = reshape(shape = concat_117, x = transpose_57_cast_fp16)[name = string("reshape_57_cast_fp16")]; tensor transpose_154_perm_0 = const()[name = string("transpose_154_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_58_perm_0 = const()[name = string("transpose_58_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_29_reps_0 = const()[name = string("tile_29_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_58_cast_fp16 = transpose(perm = transpose_58_perm_0, x = V_new_cast_fp16)[name = string("transpose_150")]; tensor tile_29_cast_fp16 = tile(reps = tile_29_reps_0, x = transpose_58_cast_fp16)[name = string("tile_29_cast_fp16")]; tensor concat_118 = const()[name = string("concat_118"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_58_cast_fp16 = reshape(shape = concat_118, x = tile_29_cast_fp16)[name = string("reshape_58_cast_fp16")]; tensor transpose_59_perm_0 = const()[name = string("transpose_59_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_119 = const()[name = string("concat_119"), val = tensor([-1, 1, 512, 512])]; tensor transpose_59_cast_fp16 = transpose(perm = transpose_59_perm_0, x = reshape_58_cast_fp16)[name = string("transpose_149")]; tensor reshape_59_cast_fp16 = reshape(shape = concat_119, x = transpose_59_cast_fp16)[name = string("reshape_59_cast_fp16")]; tensor V_expanded_29_perm_0 = const()[name = string("V_expanded_29_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_9512_transpose_x_0 = const()[name = string("op_9512_transpose_x_0"), val = bool(false)]; bool var_9512_transpose_y_0 = const()[name = string("op_9512_transpose_y_0"), val = bool(false)]; tensor transpose_154_cast_fp16 = transpose(perm = transpose_154_perm_0, x = reshape_57_cast_fp16)[name = string("transpose_148")]; tensor var_9512_cast_fp16 = matmul(transpose_x = var_9512_transpose_x_0, transpose_y = var_9512_transpose_y_0, x = q_119, y = transpose_154_cast_fp16)[name = string("op_9512_cast_fp16")]; tensor attn_weights_87_cast_fp16 = add(x = var_9512_cast_fp16, y = causal_mask)[name = string("attn_weights_87_cast_fp16")]; int32 var_9517 = const()[name = string("op_9517"), val = int32(-1)]; tensor attn_weights_89_cast_fp16 = softmax(axis = var_9517, x = attn_weights_87_cast_fp16)[name = string("attn_weights_89_cast_fp16")]; bool attn_output_85_transpose_x_0 = const()[name = string("attn_output_85_transpose_x_0"), val = bool(false)]; bool attn_output_85_transpose_y_0 = const()[name = string("attn_output_85_transpose_y_0"), val = bool(false)]; tensor V_expanded_29_cast_fp16 = transpose(perm = V_expanded_29_perm_0, x = reshape_59_cast_fp16)[name = string("transpose_147")]; tensor attn_output_85_cast_fp16 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = attn_weights_89_cast_fp16, y = V_expanded_29_cast_fp16)[name = string("attn_output_85_cast_fp16")]; tensor var_9525 = const()[name = string("op_9525"), val = tensor([0, 2, 1, 3])]; tensor var_9532 = const()[name = string("op_9532"), val = tensor([1, 1, -1])]; tensor var_9526_cast_fp16 = transpose(perm = var_9525, x = attn_output_85_cast_fp16)[name = string("transpose_146")]; tensor attn_output_87_cast_fp16 = reshape(shape = var_9532, x = var_9526_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor var_9537 = const()[name = string("op_9537"), val = tensor([0, 2, 1])]; string var_9553_pad_type_0 = const()[name = string("op_9553_pad_type_0"), val = string("valid")]; int32 var_9553_groups_0 = const()[name = string("op_9553_groups_0"), val = int32(1)]; tensor var_9553_strides_0 = const()[name = string("op_9553_strides_0"), val = tensor([1])]; tensor var_9553_pad_0 = const()[name = string("op_9553_pad_0"), val = tensor([0, 0])]; tensor var_9553_dilations_0 = const()[name = string("op_9553_dilations_0"), val = tensor([1])]; tensor squeeze_14_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1090035328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093181120))))[name = string("squeeze_14_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9538_cast_fp16 = transpose(perm = var_9537, x = attn_output_87_cast_fp16)[name = string("transpose_145")]; tensor var_9553_cast_fp16 = conv(dilations = var_9553_dilations_0, groups = var_9553_groups_0, pad = var_9553_pad_0, pad_type = var_9553_pad_type_0, strides = var_9553_strides_0, weight = squeeze_14_cast_fp16_to_fp32_to_fp16_palettized, x = var_9538_cast_fp16)[name = string("op_9553_cast_fp16")]; tensor var_9557 = const()[name = string("op_9557"), val = tensor([0, 2, 1])]; int32 var_9563 = const()[name = string("op_9563"), val = int32(-1)]; fp16 const_257_promoted_to_fp16 = const()[name = string("const_257_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_435_cast_fp16 = transpose(perm = var_9557, x = var_9553_cast_fp16)[name = string("transpose_144")]; tensor var_9569_cast_fp16 = mul(x = x_435_cast_fp16, y = const_257_promoted_to_fp16)[name = string("op_9569_cast_fp16")]; bool input_423_interleave_0 = const()[name = string("input_423_interleave_0"), val = bool(false)]; tensor input_423_cast_fp16 = concat(axis = var_9563, interleave = input_423_interleave_0, values = (x_435_cast_fp16, var_9569_cast_fp16))[name = string("input_423_cast_fp16")]; tensor normed_405_axes_0 = const()[name = string("normed_405_axes_0"), val = tensor([-1])]; fp16 var_9561_to_fp16 = const()[name = string("op_9561_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_405_cast_fp16 = layer_norm(axes = normed_405_axes_0, epsilon = var_9561_to_fp16, x = input_423_cast_fp16)[name = string("normed_405_cast_fp16")]; tensor var_9574_split_sizes_0 = const()[name = string("op_9574_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9574_axis_0 = const()[name = string("op_9574_axis_0"), val = int32(-1)]; tensor var_9574_cast_fp16_0, tensor var_9574_cast_fp16_1 = split(axis = var_9574_axis_0, split_sizes = var_9574_split_sizes_0, x = normed_405_cast_fp16)[name = string("op_9574_cast_fp16")]; tensor const_258_to_fp16 = const()[name = string("const_258_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093182720)))]; tensor var_9577_cast_fp16 = mul(x = var_9574_cast_fp16_0, y = const_258_to_fp16)[name = string("op_9577_cast_fp16")]; tensor x_439_cast_fp16 = add(x = x_421_cast_fp16, y = var_9577_cast_fp16)[name = string("x_439_cast_fp16")]; int32 var_9584 = const()[name = string("op_9584"), val = int32(-1)]; fp16 const_259_promoted_to_fp16 = const()[name = string("const_259_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9590_cast_fp16 = mul(x = x_439_cast_fp16, y = const_259_promoted_to_fp16)[name = string("op_9590_cast_fp16")]; bool input_425_interleave_0 = const()[name = string("input_425_interleave_0"), val = bool(false)]; tensor input_425_cast_fp16 = concat(axis = var_9584, interleave = input_425_interleave_0, values = (x_439_cast_fp16, var_9590_cast_fp16))[name = string("input_425_cast_fp16")]; tensor normed_409_axes_0 = const()[name = string("normed_409_axes_0"), val = tensor([-1])]; fp16 var_9582_to_fp16 = const()[name = string("op_9582_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_409_cast_fp16 = layer_norm(axes = normed_409_axes_0, epsilon = var_9582_to_fp16, x = input_425_cast_fp16)[name = string("normed_409_cast_fp16")]; tensor var_9595_split_sizes_0 = const()[name = string("op_9595_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9595_axis_0 = const()[name = string("op_9595_axis_0"), val = int32(-1)]; tensor var_9595_cast_fp16_0, tensor var_9595_cast_fp16_1 = split(axis = var_9595_axis_0, split_sizes = var_9595_split_sizes_0, x = normed_409_cast_fp16)[name = string("op_9595_cast_fp16")]; tensor const_260_to_fp16 = const()[name = string("const_260_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093185856)))]; tensor var_9598_cast_fp16 = mul(x = var_9595_cast_fp16_0, y = const_260_to_fp16)[name = string("op_9598_cast_fp16")]; tensor var_9611 = const()[name = string("op_9611"), val = tensor([0, 2, 1])]; tensor input_427_axes_0 = const()[name = string("input_427_axes_0"), val = tensor([2])]; tensor var_9612 = transpose(perm = var_9611, x = var_9598_cast_fp16)[name = string("transpose_143")]; tensor input_427 = expand_dims(axes = input_427_axes_0, x = var_9612)[name = string("input_427")]; string gate_57_pad_type_0 = const()[name = string("gate_57_pad_type_0"), val = string("valid")]; tensor gate_57_strides_0 = const()[name = string("gate_57_strides_0"), val = tensor([1, 1])]; tensor gate_57_pad_0 = const()[name = string("gate_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_57_dilations_0 = const()[name = string("gate_57_dilations_0"), val = tensor([1, 1])]; int32 gate_57_groups_0 = const()[name = string("gate_57_groups_0"), val = int32(1)]; tensor gate_57 = conv(dilations = gate_57_dilations_0, groups = gate_57_groups_0, pad = gate_57_pad_0, pad_type = gate_57_pad_type_0, strides = gate_57_strides_0, weight = layers_14_mlp_gate_proj_weight_palettized, x = input_427)[name = string("gate_57")]; string up_29_pad_type_0 = const()[name = string("up_29_pad_type_0"), val = string("valid")]; tensor up_29_strides_0 = const()[name = string("up_29_strides_0"), val = tensor([1, 1])]; tensor up_29_pad_0 = const()[name = string("up_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_29_dilations_0 = const()[name = string("up_29_dilations_0"), val = tensor([1, 1])]; int32 up_29_groups_0 = const()[name = string("up_29_groups_0"), val = int32(1)]; tensor up_29 = conv(dilations = up_29_dilations_0, groups = up_29_groups_0, pad = up_29_pad_0, pad_type = up_29_pad_type_0, strides = up_29_strides_0, weight = layers_14_mlp_up_proj_weight_palettized, x = input_427)[name = string("up_29")]; string gate_59_mode_0 = const()[name = string("gate_59_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_59 = gelu(mode = gate_59_mode_0, x = gate_57)[name = string("gate_59")]; tensor input_429 = mul(x = gate_59, y = up_29)[name = string("input_429")]; string mlp_out_29_pad_type_0 = const()[name = string("mlp_out_29_pad_type_0"), val = string("valid")]; tensor mlp_out_29_strides_0 = const()[name = string("mlp_out_29_strides_0"), val = tensor([1, 1])]; tensor mlp_out_29_pad_0 = const()[name = string("mlp_out_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_29_dilations_0 = const()[name = string("mlp_out_29_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_29_groups_0 = const()[name = string("mlp_out_29_groups_0"), val = int32(1)]; tensor mlp_out_29 = conv(dilations = mlp_out_29_dilations_0, groups = mlp_out_29_groups_0, pad = mlp_out_29_pad_0, pad_type = mlp_out_29_pad_type_0, strides = mlp_out_29_strides_0, weight = layers_14_mlp_down_proj_weight_palettized, x = input_429)[name = string("mlp_out_29")]; tensor var_9652_axes_0 = const()[name = string("op_9652_axes_0"), val = tensor([2])]; tensor var_9652 = squeeze(axes = var_9652_axes_0, x = mlp_out_29)[name = string("op_9652")]; tensor var_9656 = const()[name = string("op_9656"), val = tensor([0, 2, 1])]; int32 var_9662 = const()[name = string("op_9662"), val = int32(-1)]; fp16 const_261_promoted_to_fp16 = const()[name = string("const_261_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_443 = transpose(perm = var_9656, x = var_9652)[name = string("transpose_142")]; tensor var_9668_cast_fp16 = mul(x = x_443, y = const_261_promoted_to_fp16)[name = string("op_9668_cast_fp16")]; bool input_431_interleave_0 = const()[name = string("input_431_interleave_0"), val = bool(false)]; tensor input_431_cast_fp16 = concat(axis = var_9662, interleave = input_431_interleave_0, values = (x_443, var_9668_cast_fp16))[name = string("input_431_cast_fp16")]; tensor normed_413_axes_0 = const()[name = string("normed_413_axes_0"), val = tensor([-1])]; fp16 var_9660_to_fp16 = const()[name = string("op_9660_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_413_cast_fp16 = layer_norm(axes = normed_413_axes_0, epsilon = var_9660_to_fp16, x = input_431_cast_fp16)[name = string("normed_413_cast_fp16")]; tensor var_9673_split_sizes_0 = const()[name = string("op_9673_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9673_axis_0 = const()[name = string("op_9673_axis_0"), val = int32(-1)]; tensor var_9673_cast_fp16_0, tensor var_9673_cast_fp16_1 = split(axis = var_9673_axis_0, split_sizes = var_9673_split_sizes_0, x = normed_413_cast_fp16)[name = string("op_9673_cast_fp16")]; tensor const_262_to_fp16 = const()[name = string("const_262_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093188992)))]; tensor var_9676_cast_fp16 = mul(x = var_9673_cast_fp16_0, y = const_262_to_fp16)[name = string("op_9676_cast_fp16")]; tensor hidden_states_177_cast_fp16 = add(x = x_439_cast_fp16, y = var_9676_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor per_layer_slice_29_begin_0 = const()[name = string("per_layer_slice_29_begin_0"), val = tensor([0, 0, 3584])]; tensor per_layer_slice_29_end_0 = const()[name = string("per_layer_slice_29_end_0"), val = tensor([1, 1, 3840])]; tensor per_layer_slice_29_end_mask_0 = const()[name = string("per_layer_slice_29_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_29_cast_fp16 = slice_by_index(begin = per_layer_slice_29_begin_0, end = per_layer_slice_29_end_0, end_mask = per_layer_slice_29_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_29_cast_fp16")]; tensor gated_57 = linear(bias = linear_0_bias_0, weight = layers_14_per_layer_input_gate_weight_palettized, x = hidden_states_177_cast_fp16)[name = string("linear_28")]; string gated_59_mode_0 = const()[name = string("gated_59_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_59 = gelu(mode = gated_59_mode_0, x = gated_57)[name = string("gated_59")]; tensor input_435_cast_fp16 = mul(x = gated_59, y = per_layer_slice_29_cast_fp16)[name = string("input_435_cast_fp16")]; tensor layers_14_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093192128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093388800))))[name = string("layers_14_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_29_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_14_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_435_cast_fp16)[name = string("linear_29_cast_fp16")]; int32 var_9713 = const()[name = string("op_9713"), val = int32(-1)]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9719_cast_fp16 = mul(x = linear_29_cast_fp16, y = const_263_promoted_to_fp16)[name = string("op_9719_cast_fp16")]; bool input_437_interleave_0 = const()[name = string("input_437_interleave_0"), val = bool(false)]; tensor input_437_cast_fp16 = concat(axis = var_9713, interleave = input_437_interleave_0, values = (linear_29_cast_fp16, var_9719_cast_fp16))[name = string("input_437_cast_fp16")]; tensor normed_417_axes_0 = const()[name = string("normed_417_axes_0"), val = tensor([-1])]; fp16 var_9711_to_fp16 = const()[name = string("op_9711_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_417_cast_fp16 = layer_norm(axes = normed_417_axes_0, epsilon = var_9711_to_fp16, x = input_437_cast_fp16)[name = string("normed_417_cast_fp16")]; tensor var_9724_split_sizes_0 = const()[name = string("op_9724_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9724_axis_0 = const()[name = string("op_9724_axis_0"), val = int32(-1)]; tensor var_9724_cast_fp16_0, tensor var_9724_cast_fp16_1 = split(axis = var_9724_axis_0, split_sizes = var_9724_split_sizes_0, x = normed_417_cast_fp16)[name = string("op_9724_cast_fp16")]; tensor const_264_to_fp16 = const()[name = string("const_264_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093390400)))]; tensor var_9727_cast_fp16 = mul(x = var_9724_cast_fp16_0, y = const_264_to_fp16)[name = string("op_9727_cast_fp16")]; tensor hidden_states_181_cast_fp16 = add(x = hidden_states_177_cast_fp16, y = var_9727_cast_fp16)[name = string("hidden_states_181_cast_fp16")]; tensor layers_14_layer_scalar_to_fp16 = const()[name = string("layers_14_layer_scalar_to_fp16"), val = tensor([0x1.d4p-6])]; tensor x_451_cast_fp16 = mul(x = hidden_states_181_cast_fp16, y = layers_14_layer_scalar_to_fp16)[name = string("x_451_cast_fp16")]; int32 var_9735 = const()[name = string("op_9735"), val = int32(-1)]; fp16 const_265_promoted_to_fp16 = const()[name = string("const_265_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9741_cast_fp16 = mul(x = x_451_cast_fp16, y = const_265_promoted_to_fp16)[name = string("op_9741_cast_fp16")]; bool input_439_interleave_0 = const()[name = string("input_439_interleave_0"), val = bool(false)]; tensor input_439_cast_fp16 = concat(axis = var_9735, interleave = input_439_interleave_0, values = (x_451_cast_fp16, var_9741_cast_fp16))[name = string("input_439_cast_fp16")]; tensor normed_421_axes_0 = const()[name = string("normed_421_axes_0"), val = tensor([-1])]; fp16 var_9733_to_fp16 = const()[name = string("op_9733_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_421_cast_fp16 = layer_norm(axes = normed_421_axes_0, epsilon = var_9733_to_fp16, x = input_439_cast_fp16)[name = string("normed_421_cast_fp16")]; tensor var_9746_split_sizes_0 = const()[name = string("op_9746_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9746_axis_0 = const()[name = string("op_9746_axis_0"), val = int32(-1)]; tensor var_9746_cast_fp16_0, tensor var_9746_cast_fp16_1 = split(axis = var_9746_axis_0, split_sizes = var_9746_split_sizes_0, x = normed_421_cast_fp16)[name = string("op_9746_cast_fp16")]; tensor const_266_to_fp16 = const()[name = string("const_266_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093393536)))]; tensor var_9749_cast_fp16 = mul(x = var_9746_cast_fp16_0, y = const_266_to_fp16)[name = string("op_9749_cast_fp16")]; tensor var_9757 = const()[name = string("op_9757"), val = tensor([0, 2, 1])]; tensor var_9760_axes_0 = const()[name = string("op_9760_axes_0"), val = tensor([2])]; tensor var_9758_cast_fp16 = transpose(perm = var_9757, x = var_9749_cast_fp16)[name = string("transpose_141")]; tensor var_9760_cast_fp16 = expand_dims(axes = var_9760_axes_0, x = var_9758_cast_fp16)[name = string("op_9760_cast_fp16")]; string var_9776_pad_type_0 = const()[name = string("op_9776_pad_type_0"), val = string("valid")]; tensor var_9776_strides_0 = const()[name = string("op_9776_strides_0"), val = tensor([1, 1])]; tensor var_9776_pad_0 = const()[name = string("op_9776_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9776_dilations_0 = const()[name = string("op_9776_dilations_0"), val = tensor([1, 1])]; int32 var_9776_groups_0 = const()[name = string("op_9776_groups_0"), val = int32(1)]; tensor var_9776 = conv(dilations = var_9776_dilations_0, groups = var_9776_groups_0, pad = var_9776_pad_0, pad_type = var_9776_pad_type_0, strides = var_9776_strides_0, weight = layers_15_self_attn_q_proj_weight_palettized, x = var_9760_cast_fp16)[name = string("op_9776")]; tensor var_9781 = const()[name = string("op_9781"), val = tensor([1, 8, 256, 1])]; tensor var_9782 = reshape(shape = var_9781, x = var_9776)[name = string("op_9782")]; tensor var_9787 = const()[name = string("op_9787"), val = tensor([0, 1, 3, 2])]; tensor var_9797 = const()[name = string("op_9797"), val = tensor([1, 8, 256])]; tensor var_9788 = transpose(perm = var_9787, x = var_9782)[name = string("transpose_140")]; tensor x_455 = reshape(shape = var_9797, x = var_9788)[name = string("x_455")]; int32 var_9803 = const()[name = string("op_9803"), val = int32(-1)]; fp16 const_267_promoted_to_fp16 = const()[name = string("const_267_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9809_cast_fp16 = mul(x = x_455, y = const_267_promoted_to_fp16)[name = string("op_9809_cast_fp16")]; bool input_443_interleave_0 = const()[name = string("input_443_interleave_0"), val = bool(false)]; tensor input_443_cast_fp16 = concat(axis = var_9803, interleave = input_443_interleave_0, values = (x_455, var_9809_cast_fp16))[name = string("input_443_cast_fp16")]; tensor normed_425_axes_0 = const()[name = string("normed_425_axes_0"), val = tensor([-1])]; fp16 var_9801_to_fp16 = const()[name = string("op_9801_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_425_cast_fp16 = layer_norm(axes = normed_425_axes_0, epsilon = var_9801_to_fp16, x = input_443_cast_fp16)[name = string("normed_425_cast_fp16")]; tensor var_9814_split_sizes_0 = const()[name = string("op_9814_split_sizes_0"), val = tensor([256, 256])]; int32 var_9814_axis_0 = const()[name = string("op_9814_axis_0"), val = int32(-1)]; tensor var_9814_cast_fp16_0, tensor var_9814_cast_fp16_1 = split(axis = var_9814_axis_0, split_sizes = var_9814_split_sizes_0, x = normed_425_cast_fp16)[name = string("op_9814_cast_fp16")]; tensor var_9817_cast_fp16 = mul(x = var_9814_cast_fp16_0, y = const_234_to_fp16)[name = string("op_9817_cast_fp16")]; tensor var_9823 = const()[name = string("op_9823"), val = tensor([1, 8, 1, 256])]; tensor q_123 = reshape(shape = var_9823, x = var_9817_cast_fp16)[name = string("q_123")]; tensor var_9825 = mul(x = q_123, y = cos_1)[name = string("op_9825")]; tensor var_9826_split_sizes_0 = const()[name = string("op_9826_split_sizes_0"), val = tensor([128, 128])]; int32 var_9826_axis_0 = const()[name = string("op_9826_axis_0"), val = int32(-1)]; tensor var_9826_0, tensor var_9826_1 = split(axis = var_9826_axis_0, split_sizes = var_9826_split_sizes_0, x = q_123)[name = string("op_9826")]; fp16 const_269_promoted = const()[name = string("const_269_promoted"), val = fp16(-0x1p+0)]; tensor var_9828 = mul(x = var_9826_1, y = const_269_promoted)[name = string("op_9828")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(-1)]; bool var_9831_interleave_0 = const()[name = string("op_9831_interleave_0"), val = bool(false)]; tensor var_9831 = concat(axis = var_9830, interleave = var_9831_interleave_0, values = (var_9828, var_9826_0))[name = string("op_9831")]; tensor var_9832 = mul(x = var_9831, y = sin_1)[name = string("op_9832")]; tensor q_125 = add(x = var_9825, y = var_9832)[name = string("q_125")]; bool var_9846_transpose_x_0 = const()[name = string("op_9846_transpose_x_0"), val = bool(false)]; bool var_9846_transpose_y_0 = const()[name = string("op_9846_transpose_y_0"), val = bool(false)]; tensor var_9846_cast_fp16 = matmul(transpose_x = var_9846_transpose_x_0, transpose_y = var_9846_transpose_y_0, x = q_125, y = transpose_153_cast_fp16)[name = string("op_9846_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = var_9846_cast_fp16, y = causal_mask)[name = string("attn_weights_93_cast_fp16")]; int32 var_9851 = const()[name = string("op_9851"), val = int32(-1)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_9851, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_91_transpose_x_0 = const()[name = string("attn_output_91_transpose_x_0"), val = bool(false)]; bool attn_output_91_transpose_y_0 = const()[name = string("attn_output_91_transpose_y_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = matmul(transpose_x = attn_output_91_transpose_x_0, transpose_y = attn_output_91_transpose_y_0, x = attn_weights_95_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_91_cast_fp16")]; tensor var_9859 = const()[name = string("op_9859"), val = tensor([0, 2, 1, 3])]; tensor var_9866 = const()[name = string("op_9866"), val = tensor([1, 1, -1])]; tensor var_9860_cast_fp16 = transpose(perm = var_9859, x = attn_output_91_cast_fp16)[name = string("transpose_139")]; tensor attn_output_93_cast_fp16 = reshape(shape = var_9866, x = var_9860_cast_fp16)[name = string("attn_output_93_cast_fp16")]; tensor var_9871 = const()[name = string("op_9871"), val = tensor([0, 2, 1])]; string var_9887_pad_type_0 = const()[name = string("op_9887_pad_type_0"), val = string("valid")]; int32 var_9887_groups_0 = const()[name = string("op_9887_groups_0"), val = int32(1)]; tensor var_9887_strides_0 = const()[name = string("op_9887_strides_0"), val = tensor([1])]; tensor var_9887_pad_0 = const()[name = string("op_9887_pad_0"), val = tensor([0, 0])]; tensor var_9887_dilations_0 = const()[name = string("op_9887_dilations_0"), val = tensor([1])]; tensor squeeze_15_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1093396672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1094969600))))[name = string("squeeze_15_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9872_cast_fp16 = transpose(perm = var_9871, x = attn_output_93_cast_fp16)[name = string("transpose_138")]; tensor var_9887_cast_fp16 = conv(dilations = var_9887_dilations_0, groups = var_9887_groups_0, pad = var_9887_pad_0, pad_type = var_9887_pad_type_0, strides = var_9887_strides_0, weight = squeeze_15_cast_fp16_to_fp32_to_fp16_palettized, x = var_9872_cast_fp16)[name = string("op_9887_cast_fp16")]; tensor var_9891 = const()[name = string("op_9891"), val = tensor([0, 2, 1])]; int32 var_9897 = const()[name = string("op_9897"), val = int32(-1)]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_459_cast_fp16 = transpose(perm = var_9891, x = var_9887_cast_fp16)[name = string("transpose_137")]; tensor var_9903_cast_fp16 = mul(x = x_459_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9903_cast_fp16")]; bool input_447_interleave_0 = const()[name = string("input_447_interleave_0"), val = bool(false)]; tensor input_447_cast_fp16 = concat(axis = var_9897, interleave = input_447_interleave_0, values = (x_459_cast_fp16, var_9903_cast_fp16))[name = string("input_447_cast_fp16")]; tensor normed_429_axes_0 = const()[name = string("normed_429_axes_0"), val = tensor([-1])]; fp16 var_9895_to_fp16 = const()[name = string("op_9895_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_429_cast_fp16 = layer_norm(axes = normed_429_axes_0, epsilon = var_9895_to_fp16, x = input_447_cast_fp16)[name = string("normed_429_cast_fp16")]; tensor var_9908_split_sizes_0 = const()[name = string("op_9908_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9908_axis_0 = const()[name = string("op_9908_axis_0"), val = int32(-1)]; tensor var_9908_cast_fp16_0, tensor var_9908_cast_fp16_1 = split(axis = var_9908_axis_0, split_sizes = var_9908_split_sizes_0, x = normed_429_cast_fp16)[name = string("op_9908_cast_fp16")]; tensor const_271_to_fp16 = const()[name = string("const_271_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1094971200)))]; tensor var_9911_cast_fp16 = mul(x = var_9908_cast_fp16_0, y = const_271_to_fp16)[name = string("op_9911_cast_fp16")]; tensor x_463_cast_fp16 = add(x = x_451_cast_fp16, y = var_9911_cast_fp16)[name = string("x_463_cast_fp16")]; int32 var_9918 = const()[name = string("op_9918"), val = int32(-1)]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9924_cast_fp16 = mul(x = x_463_cast_fp16, y = const_272_promoted_to_fp16)[name = string("op_9924_cast_fp16")]; bool input_449_interleave_0 = const()[name = string("input_449_interleave_0"), val = bool(false)]; tensor input_449_cast_fp16 = concat(axis = var_9918, interleave = input_449_interleave_0, values = (x_463_cast_fp16, var_9924_cast_fp16))[name = string("input_449_cast_fp16")]; tensor normed_433_axes_0 = const()[name = string("normed_433_axes_0"), val = tensor([-1])]; fp16 var_9916_to_fp16 = const()[name = string("op_9916_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_433_cast_fp16 = layer_norm(axes = normed_433_axes_0, epsilon = var_9916_to_fp16, x = input_449_cast_fp16)[name = string("normed_433_cast_fp16")]; tensor var_9929_split_sizes_0 = const()[name = string("op_9929_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_9929_axis_0 = const()[name = string("op_9929_axis_0"), val = int32(-1)]; tensor var_9929_cast_fp16_0, tensor var_9929_cast_fp16_1 = split(axis = var_9929_axis_0, split_sizes = var_9929_split_sizes_0, x = normed_433_cast_fp16)[name = string("op_9929_cast_fp16")]; tensor const_273_to_fp16 = const()[name = string("const_273_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1094974336)))]; tensor var_9932_cast_fp16 = mul(x = var_9929_cast_fp16_0, y = const_273_to_fp16)[name = string("op_9932_cast_fp16")]; tensor var_9945 = const()[name = string("op_9945"), val = tensor([0, 2, 1])]; tensor input_451_axes_0 = const()[name = string("input_451_axes_0"), val = tensor([2])]; tensor var_9946 = transpose(perm = var_9945, x = var_9932_cast_fp16)[name = string("transpose_136")]; tensor input_451 = expand_dims(axes = input_451_axes_0, x = var_9946)[name = string("input_451")]; string gate_61_pad_type_0 = const()[name = string("gate_61_pad_type_0"), val = string("valid")]; tensor gate_61_strides_0 = const()[name = string("gate_61_strides_0"), val = tensor([1, 1])]; tensor gate_61_pad_0 = const()[name = string("gate_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_61_dilations_0 = const()[name = string("gate_61_dilations_0"), val = tensor([1, 1])]; int32 gate_61_groups_0 = const()[name = string("gate_61_groups_0"), val = int32(1)]; tensor gate_61 = conv(dilations = gate_61_dilations_0, groups = gate_61_groups_0, pad = gate_61_pad_0, pad_type = gate_61_pad_type_0, strides = gate_61_strides_0, weight = layers_15_mlp_gate_proj_weight_palettized, x = input_451)[name = string("gate_61")]; string up_31_pad_type_0 = const()[name = string("up_31_pad_type_0"), val = string("valid")]; tensor up_31_strides_0 = const()[name = string("up_31_strides_0"), val = tensor([1, 1])]; tensor up_31_pad_0 = const()[name = string("up_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_31_dilations_0 = const()[name = string("up_31_dilations_0"), val = tensor([1, 1])]; int32 up_31_groups_0 = const()[name = string("up_31_groups_0"), val = int32(1)]; tensor up_31 = conv(dilations = up_31_dilations_0, groups = up_31_groups_0, pad = up_31_pad_0, pad_type = up_31_pad_type_0, strides = up_31_strides_0, weight = layers_15_mlp_up_proj_weight_palettized, x = input_451)[name = string("up_31")]; string gate_63_mode_0 = const()[name = string("gate_63_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_63 = gelu(mode = gate_63_mode_0, x = gate_61)[name = string("gate_63")]; tensor input_453 = mul(x = gate_63, y = up_31)[name = string("input_453")]; string mlp_out_31_pad_type_0 = const()[name = string("mlp_out_31_pad_type_0"), val = string("valid")]; tensor mlp_out_31_strides_0 = const()[name = string("mlp_out_31_strides_0"), val = tensor([1, 1])]; tensor mlp_out_31_pad_0 = const()[name = string("mlp_out_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_31_dilations_0 = const()[name = string("mlp_out_31_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_31_groups_0 = const()[name = string("mlp_out_31_groups_0"), val = int32(1)]; tensor mlp_out_31 = conv(dilations = mlp_out_31_dilations_0, groups = mlp_out_31_groups_0, pad = mlp_out_31_pad_0, pad_type = mlp_out_31_pad_type_0, strides = mlp_out_31_strides_0, weight = layers_15_mlp_down_proj_weight_palettized, x = input_453)[name = string("mlp_out_31")]; tensor var_9986_axes_0 = const()[name = string("op_9986_axes_0"), val = tensor([2])]; tensor var_9986 = squeeze(axes = var_9986_axes_0, x = mlp_out_31)[name = string("op_9986")]; tensor var_9990 = const()[name = string("op_9990"), val = tensor([0, 2, 1])]; int32 var_9996 = const()[name = string("op_9996"), val = int32(-1)]; fp16 const_274_promoted_to_fp16 = const()[name = string("const_274_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_467 = transpose(perm = var_9990, x = var_9986)[name = string("transpose_135")]; tensor var_10002_cast_fp16 = mul(x = x_467, y = const_274_promoted_to_fp16)[name = string("op_10002_cast_fp16")]; bool input_455_interleave_0 = const()[name = string("input_455_interleave_0"), val = bool(false)]; tensor input_455_cast_fp16 = concat(axis = var_9996, interleave = input_455_interleave_0, values = (x_467, var_10002_cast_fp16))[name = string("input_455_cast_fp16")]; tensor normed_437_axes_0 = const()[name = string("normed_437_axes_0"), val = tensor([-1])]; fp16 var_9994_to_fp16 = const()[name = string("op_9994_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_437_cast_fp16 = layer_norm(axes = normed_437_axes_0, epsilon = var_9994_to_fp16, x = input_455_cast_fp16)[name = string("normed_437_cast_fp16")]; tensor var_10007_split_sizes_0 = const()[name = string("op_10007_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10007_axis_0 = const()[name = string("op_10007_axis_0"), val = int32(-1)]; tensor var_10007_cast_fp16_0, tensor var_10007_cast_fp16_1 = split(axis = var_10007_axis_0, split_sizes = var_10007_split_sizes_0, x = normed_437_cast_fp16)[name = string("op_10007_cast_fp16")]; tensor const_275_to_fp16 = const()[name = string("const_275_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1094977472)))]; tensor var_10010_cast_fp16 = mul(x = var_10007_cast_fp16_0, y = const_275_to_fp16)[name = string("op_10010_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = x_463_cast_fp16, y = var_10010_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; tensor per_layer_slice_31_begin_0 = const()[name = string("per_layer_slice_31_begin_0"), val = tensor([0, 0, 3840])]; tensor per_layer_slice_31_end_0 = const()[name = string("per_layer_slice_31_end_0"), val = tensor([1, 1, 4096])]; tensor per_layer_slice_31_end_mask_0 = const()[name = string("per_layer_slice_31_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_31_cast_fp16 = slice_by_index(begin = per_layer_slice_31_begin_0, end = per_layer_slice_31_end_0, end_mask = per_layer_slice_31_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_31_cast_fp16")]; tensor gated_61 = linear(bias = linear_0_bias_0, weight = layers_15_per_layer_input_gate_weight_palettized, x = hidden_states_189_cast_fp16)[name = string("linear_30")]; string gated_63_mode_0 = const()[name = string("gated_63_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_63 = gelu(mode = gated_63_mode_0, x = gated_61)[name = string("gated_63")]; tensor input_459_cast_fp16 = mul(x = gated_63, y = per_layer_slice_31_cast_fp16)[name = string("input_459_cast_fp16")]; tensor layers_15_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1094980608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095177280))))[name = string("layers_15_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_31_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_15_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_459_cast_fp16)[name = string("linear_31_cast_fp16")]; int32 var_10047 = const()[name = string("op_10047"), val = int32(-1)]; fp16 const_276_promoted_to_fp16 = const()[name = string("const_276_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10053_cast_fp16 = mul(x = linear_31_cast_fp16, y = const_276_promoted_to_fp16)[name = string("op_10053_cast_fp16")]; bool input_461_interleave_0 = const()[name = string("input_461_interleave_0"), val = bool(false)]; tensor input_461_cast_fp16 = concat(axis = var_10047, interleave = input_461_interleave_0, values = (linear_31_cast_fp16, var_10053_cast_fp16))[name = string("input_461_cast_fp16")]; tensor normed_441_axes_0 = const()[name = string("normed_441_axes_0"), val = tensor([-1])]; fp16 var_10045_to_fp16 = const()[name = string("op_10045_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_441_cast_fp16 = layer_norm(axes = normed_441_axes_0, epsilon = var_10045_to_fp16, x = input_461_cast_fp16)[name = string("normed_441_cast_fp16")]; tensor var_10058_split_sizes_0 = const()[name = string("op_10058_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10058_axis_0 = const()[name = string("op_10058_axis_0"), val = int32(-1)]; tensor var_10058_cast_fp16_0, tensor var_10058_cast_fp16_1 = split(axis = var_10058_axis_0, split_sizes = var_10058_split_sizes_0, x = normed_441_cast_fp16)[name = string("op_10058_cast_fp16")]; tensor const_277_to_fp16 = const()[name = string("const_277_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095178880)))]; tensor var_10061_cast_fp16 = mul(x = var_10058_cast_fp16_0, y = const_277_to_fp16)[name = string("op_10061_cast_fp16")]; tensor hidden_states_193_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = var_10061_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor layers_15_layer_scalar_to_fp16 = const()[name = string("layers_15_layer_scalar_to_fp16"), val = tensor([0x1.04p-2])]; tensor x_475_cast_fp16 = mul(x = hidden_states_193_cast_fp16, y = layers_15_layer_scalar_to_fp16)[name = string("x_475_cast_fp16")]; int32 var_10069 = const()[name = string("op_10069"), val = int32(-1)]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10075_cast_fp16 = mul(x = x_475_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_10075_cast_fp16")]; bool input_463_interleave_0 = const()[name = string("input_463_interleave_0"), val = bool(false)]; tensor input_463_cast_fp16 = concat(axis = var_10069, interleave = input_463_interleave_0, values = (x_475_cast_fp16, var_10075_cast_fp16))[name = string("input_463_cast_fp16")]; tensor normed_445_axes_0 = const()[name = string("normed_445_axes_0"), val = tensor([-1])]; fp16 var_10067_to_fp16 = const()[name = string("op_10067_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_445_cast_fp16 = layer_norm(axes = normed_445_axes_0, epsilon = var_10067_to_fp16, x = input_463_cast_fp16)[name = string("normed_445_cast_fp16")]; tensor var_10080_split_sizes_0 = const()[name = string("op_10080_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10080_axis_0 = const()[name = string("op_10080_axis_0"), val = int32(-1)]; tensor var_10080_cast_fp16_0, tensor var_10080_cast_fp16_1 = split(axis = var_10080_axis_0, split_sizes = var_10080_split_sizes_0, x = normed_445_cast_fp16)[name = string("op_10080_cast_fp16")]; tensor const_279_to_fp16 = const()[name = string("const_279_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095182016)))]; tensor var_10083_cast_fp16 = mul(x = var_10080_cast_fp16_0, y = const_279_to_fp16)[name = string("op_10083_cast_fp16")]; tensor var_10091 = const()[name = string("op_10091"), val = tensor([0, 2, 1])]; tensor var_10094_axes_0 = const()[name = string("op_10094_axes_0"), val = tensor([2])]; tensor var_10092_cast_fp16 = transpose(perm = var_10091, x = var_10083_cast_fp16)[name = string("transpose_134")]; tensor var_10094_cast_fp16 = expand_dims(axes = var_10094_axes_0, x = var_10092_cast_fp16)[name = string("op_10094_cast_fp16")]; string var_10110_pad_type_0 = const()[name = string("op_10110_pad_type_0"), val = string("valid")]; tensor var_10110_strides_0 = const()[name = string("op_10110_strides_0"), val = tensor([1, 1])]; tensor var_10110_pad_0 = const()[name = string("op_10110_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10110_dilations_0 = const()[name = string("op_10110_dilations_0"), val = tensor([1, 1])]; int32 var_10110_groups_0 = const()[name = string("op_10110_groups_0"), val = int32(1)]; tensor var_10110 = conv(dilations = var_10110_dilations_0, groups = var_10110_groups_0, pad = var_10110_pad_0, pad_type = var_10110_pad_type_0, strides = var_10110_strides_0, weight = layers_16_self_attn_q_proj_weight_palettized, x = var_10094_cast_fp16)[name = string("op_10110")]; tensor var_10115 = const()[name = string("op_10115"), val = tensor([1, 8, 256, 1])]; tensor var_10116 = reshape(shape = var_10115, x = var_10110)[name = string("op_10116")]; tensor var_10121 = const()[name = string("op_10121"), val = tensor([0, 1, 3, 2])]; tensor var_10131 = const()[name = string("op_10131"), val = tensor([1, 8, 256])]; tensor var_10122 = transpose(perm = var_10121, x = var_10116)[name = string("transpose_133")]; tensor x_479 = reshape(shape = var_10131, x = var_10122)[name = string("x_479")]; int32 var_10137 = const()[name = string("op_10137"), val = int32(-1)]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10143_cast_fp16 = mul(x = x_479, y = const_280_promoted_to_fp16)[name = string("op_10143_cast_fp16")]; bool input_467_interleave_0 = const()[name = string("input_467_interleave_0"), val = bool(false)]; tensor input_467_cast_fp16 = concat(axis = var_10137, interleave = input_467_interleave_0, values = (x_479, var_10143_cast_fp16))[name = string("input_467_cast_fp16")]; tensor normed_449_axes_0 = const()[name = string("normed_449_axes_0"), val = tensor([-1])]; fp16 var_10135_to_fp16 = const()[name = string("op_10135_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_449_cast_fp16 = layer_norm(axes = normed_449_axes_0, epsilon = var_10135_to_fp16, x = input_467_cast_fp16)[name = string("normed_449_cast_fp16")]; tensor var_10148_split_sizes_0 = const()[name = string("op_10148_split_sizes_0"), val = tensor([256, 256])]; int32 var_10148_axis_0 = const()[name = string("op_10148_axis_0"), val = int32(-1)]; tensor var_10148_cast_fp16_0, tensor var_10148_cast_fp16_1 = split(axis = var_10148_axis_0, split_sizes = var_10148_split_sizes_0, x = normed_449_cast_fp16)[name = string("op_10148_cast_fp16")]; tensor var_10151_cast_fp16 = mul(x = var_10148_cast_fp16_0, y = const_234_to_fp16)[name = string("op_10151_cast_fp16")]; tensor var_10157 = const()[name = string("op_10157"), val = tensor([1, 8, 1, 256])]; tensor q_129 = reshape(shape = var_10157, x = var_10151_cast_fp16)[name = string("q_129")]; tensor var_10159 = mul(x = q_129, y = cos_1)[name = string("op_10159")]; tensor var_10160_split_sizes_0 = const()[name = string("op_10160_split_sizes_0"), val = tensor([128, 128])]; int32 var_10160_axis_0 = const()[name = string("op_10160_axis_0"), val = int32(-1)]; tensor var_10160_0, tensor var_10160_1 = split(axis = var_10160_axis_0, split_sizes = var_10160_split_sizes_0, x = q_129)[name = string("op_10160")]; fp16 const_282_promoted = const()[name = string("const_282_promoted"), val = fp16(-0x1p+0)]; tensor var_10162 = mul(x = var_10160_1, y = const_282_promoted)[name = string("op_10162")]; int32 var_10164 = const()[name = string("op_10164"), val = int32(-1)]; bool var_10165_interleave_0 = const()[name = string("op_10165_interleave_0"), val = bool(false)]; tensor var_10165 = concat(axis = var_10164, interleave = var_10165_interleave_0, values = (var_10162, var_10160_0))[name = string("op_10165")]; tensor var_10166 = mul(x = var_10165, y = sin_1)[name = string("op_10166")]; tensor q_131 = add(x = var_10159, y = var_10166)[name = string("q_131")]; bool var_10180_transpose_x_0 = const()[name = string("op_10180_transpose_x_0"), val = bool(false)]; bool var_10180_transpose_y_0 = const()[name = string("op_10180_transpose_y_0"), val = bool(false)]; tensor var_10180_cast_fp16 = matmul(transpose_x = var_10180_transpose_x_0, transpose_y = var_10180_transpose_y_0, x = q_131, y = transpose_153_cast_fp16)[name = string("op_10180_cast_fp16")]; tensor attn_weights_99_cast_fp16 = add(x = var_10180_cast_fp16, y = causal_mask)[name = string("attn_weights_99_cast_fp16")]; int32 var_10185 = const()[name = string("op_10185"), val = int32(-1)]; tensor attn_weights_101_cast_fp16 = softmax(axis = var_10185, x = attn_weights_99_cast_fp16)[name = string("attn_weights_101_cast_fp16")]; bool attn_output_97_transpose_x_0 = const()[name = string("attn_output_97_transpose_x_0"), val = bool(false)]; bool attn_output_97_transpose_y_0 = const()[name = string("attn_output_97_transpose_y_0"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_0, transpose_y = attn_output_97_transpose_y_0, x = attn_weights_101_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_97_cast_fp16")]; tensor var_10193 = const()[name = string("op_10193"), val = tensor([0, 2, 1, 3])]; tensor var_10200 = const()[name = string("op_10200"), val = tensor([1, 1, -1])]; tensor var_10194_cast_fp16 = transpose(perm = var_10193, x = attn_output_97_cast_fp16)[name = string("transpose_132")]; tensor attn_output_99_cast_fp16 = reshape(shape = var_10200, x = var_10194_cast_fp16)[name = string("attn_output_99_cast_fp16")]; tensor var_10205 = const()[name = string("op_10205"), val = tensor([0, 2, 1])]; string var_10221_pad_type_0 = const()[name = string("op_10221_pad_type_0"), val = string("valid")]; int32 var_10221_groups_0 = const()[name = string("op_10221_groups_0"), val = int32(1)]; tensor var_10221_strides_0 = const()[name = string("op_10221_strides_0"), val = tensor([1])]; tensor var_10221_pad_0 = const()[name = string("op_10221_pad_0"), val = tensor([0, 0])]; tensor var_10221_dilations_0 = const()[name = string("op_10221_dilations_0"), val = tensor([1])]; tensor squeeze_16_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095185152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096758080))))[name = string("squeeze_16_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10206_cast_fp16 = transpose(perm = var_10205, x = attn_output_99_cast_fp16)[name = string("transpose_131")]; tensor var_10221_cast_fp16 = conv(dilations = var_10221_dilations_0, groups = var_10221_groups_0, pad = var_10221_pad_0, pad_type = var_10221_pad_type_0, strides = var_10221_strides_0, weight = squeeze_16_cast_fp16_to_fp32_to_fp16_palettized, x = var_10206_cast_fp16)[name = string("op_10221_cast_fp16")]; tensor var_10225 = const()[name = string("op_10225"), val = tensor([0, 2, 1])]; int32 var_10231 = const()[name = string("op_10231"), val = int32(-1)]; fp16 const_283_promoted_to_fp16 = const()[name = string("const_283_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_483_cast_fp16 = transpose(perm = var_10225, x = var_10221_cast_fp16)[name = string("transpose_130")]; tensor var_10237_cast_fp16 = mul(x = x_483_cast_fp16, y = const_283_promoted_to_fp16)[name = string("op_10237_cast_fp16")]; bool input_471_interleave_0 = const()[name = string("input_471_interleave_0"), val = bool(false)]; tensor input_471_cast_fp16 = concat(axis = var_10231, interleave = input_471_interleave_0, values = (x_483_cast_fp16, var_10237_cast_fp16))[name = string("input_471_cast_fp16")]; tensor normed_453_axes_0 = const()[name = string("normed_453_axes_0"), val = tensor([-1])]; fp16 var_10229_to_fp16 = const()[name = string("op_10229_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_453_cast_fp16 = layer_norm(axes = normed_453_axes_0, epsilon = var_10229_to_fp16, x = input_471_cast_fp16)[name = string("normed_453_cast_fp16")]; tensor var_10242_split_sizes_0 = const()[name = string("op_10242_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10242_axis_0 = const()[name = string("op_10242_axis_0"), val = int32(-1)]; tensor var_10242_cast_fp16_0, tensor var_10242_cast_fp16_1 = split(axis = var_10242_axis_0, split_sizes = var_10242_split_sizes_0, x = normed_453_cast_fp16)[name = string("op_10242_cast_fp16")]; tensor const_284_to_fp16 = const()[name = string("const_284_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096759680)))]; tensor var_10245_cast_fp16 = mul(x = var_10242_cast_fp16_0, y = const_284_to_fp16)[name = string("op_10245_cast_fp16")]; tensor x_487_cast_fp16 = add(x = x_475_cast_fp16, y = var_10245_cast_fp16)[name = string("x_487_cast_fp16")]; int32 var_10252 = const()[name = string("op_10252"), val = int32(-1)]; fp16 const_285_promoted_to_fp16 = const()[name = string("const_285_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10258_cast_fp16 = mul(x = x_487_cast_fp16, y = const_285_promoted_to_fp16)[name = string("op_10258_cast_fp16")]; bool input_473_interleave_0 = const()[name = string("input_473_interleave_0"), val = bool(false)]; tensor input_473_cast_fp16 = concat(axis = var_10252, interleave = input_473_interleave_0, values = (x_487_cast_fp16, var_10258_cast_fp16))[name = string("input_473_cast_fp16")]; tensor normed_457_axes_0 = const()[name = string("normed_457_axes_0"), val = tensor([-1])]; fp16 var_10250_to_fp16 = const()[name = string("op_10250_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_457_cast_fp16 = layer_norm(axes = normed_457_axes_0, epsilon = var_10250_to_fp16, x = input_473_cast_fp16)[name = string("normed_457_cast_fp16")]; tensor var_10263_split_sizes_0 = const()[name = string("op_10263_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10263_axis_0 = const()[name = string("op_10263_axis_0"), val = int32(-1)]; tensor var_10263_cast_fp16_0, tensor var_10263_cast_fp16_1 = split(axis = var_10263_axis_0, split_sizes = var_10263_split_sizes_0, x = normed_457_cast_fp16)[name = string("op_10263_cast_fp16")]; tensor const_286_to_fp16 = const()[name = string("const_286_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096762816)))]; tensor var_10266_cast_fp16 = mul(x = var_10263_cast_fp16_0, y = const_286_to_fp16)[name = string("op_10266_cast_fp16")]; tensor var_10279 = const()[name = string("op_10279"), val = tensor([0, 2, 1])]; tensor input_475_axes_0 = const()[name = string("input_475_axes_0"), val = tensor([2])]; tensor var_10280 = transpose(perm = var_10279, x = var_10266_cast_fp16)[name = string("transpose_129")]; tensor input_475 = expand_dims(axes = input_475_axes_0, x = var_10280)[name = string("input_475")]; string gate_65_pad_type_0 = const()[name = string("gate_65_pad_type_0"), val = string("valid")]; tensor gate_65_strides_0 = const()[name = string("gate_65_strides_0"), val = tensor([1, 1])]; tensor gate_65_pad_0 = const()[name = string("gate_65_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_65_dilations_0 = const()[name = string("gate_65_dilations_0"), val = tensor([1, 1])]; int32 gate_65_groups_0 = const()[name = string("gate_65_groups_0"), val = int32(1)]; tensor gate_65 = conv(dilations = gate_65_dilations_0, groups = gate_65_groups_0, pad = gate_65_pad_0, pad_type = gate_65_pad_type_0, strides = gate_65_strides_0, weight = layers_16_mlp_gate_proj_weight_palettized, x = input_475)[name = string("gate_65")]; string up_33_pad_type_0 = const()[name = string("up_33_pad_type_0"), val = string("valid")]; tensor up_33_strides_0 = const()[name = string("up_33_strides_0"), val = tensor([1, 1])]; tensor up_33_pad_0 = const()[name = string("up_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_33_dilations_0 = const()[name = string("up_33_dilations_0"), val = tensor([1, 1])]; int32 up_33_groups_0 = const()[name = string("up_33_groups_0"), val = int32(1)]; tensor up_33 = conv(dilations = up_33_dilations_0, groups = up_33_groups_0, pad = up_33_pad_0, pad_type = up_33_pad_type_0, strides = up_33_strides_0, weight = layers_16_mlp_up_proj_weight_palettized, x = input_475)[name = string("up_33")]; string gate_67_mode_0 = const()[name = string("gate_67_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_67 = gelu(mode = gate_67_mode_0, x = gate_65)[name = string("gate_67")]; tensor input_477 = mul(x = gate_67, y = up_33)[name = string("input_477")]; string mlp_out_33_pad_type_0 = const()[name = string("mlp_out_33_pad_type_0"), val = string("valid")]; tensor mlp_out_33_strides_0 = const()[name = string("mlp_out_33_strides_0"), val = tensor([1, 1])]; tensor mlp_out_33_pad_0 = const()[name = string("mlp_out_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_33_dilations_0 = const()[name = string("mlp_out_33_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_33_groups_0 = const()[name = string("mlp_out_33_groups_0"), val = int32(1)]; tensor mlp_out_33 = conv(dilations = mlp_out_33_dilations_0, groups = mlp_out_33_groups_0, pad = mlp_out_33_pad_0, pad_type = mlp_out_33_pad_type_0, strides = mlp_out_33_strides_0, weight = layers_16_mlp_down_proj_weight_palettized, x = input_477)[name = string("mlp_out_33")]; tensor var_10320_axes_0 = const()[name = string("op_10320_axes_0"), val = tensor([2])]; tensor var_10320 = squeeze(axes = var_10320_axes_0, x = mlp_out_33)[name = string("op_10320")]; tensor var_10324 = const()[name = string("op_10324"), val = tensor([0, 2, 1])]; int32 var_10330 = const()[name = string("op_10330"), val = int32(-1)]; fp16 const_287_promoted_to_fp16 = const()[name = string("const_287_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_491 = transpose(perm = var_10324, x = var_10320)[name = string("transpose_128")]; tensor var_10336_cast_fp16 = mul(x = x_491, y = const_287_promoted_to_fp16)[name = string("op_10336_cast_fp16")]; bool input_479_interleave_0 = const()[name = string("input_479_interleave_0"), val = bool(false)]; tensor input_479_cast_fp16 = concat(axis = var_10330, interleave = input_479_interleave_0, values = (x_491, var_10336_cast_fp16))[name = string("input_479_cast_fp16")]; tensor normed_461_axes_0 = const()[name = string("normed_461_axes_0"), val = tensor([-1])]; fp16 var_10328_to_fp16 = const()[name = string("op_10328_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_461_cast_fp16 = layer_norm(axes = normed_461_axes_0, epsilon = var_10328_to_fp16, x = input_479_cast_fp16)[name = string("normed_461_cast_fp16")]; tensor var_10341_split_sizes_0 = const()[name = string("op_10341_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10341_axis_0 = const()[name = string("op_10341_axis_0"), val = int32(-1)]; tensor var_10341_cast_fp16_0, tensor var_10341_cast_fp16_1 = split(axis = var_10341_axis_0, split_sizes = var_10341_split_sizes_0, x = normed_461_cast_fp16)[name = string("op_10341_cast_fp16")]; tensor const_288_to_fp16 = const()[name = string("const_288_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096765952)))]; tensor var_10344_cast_fp16 = mul(x = var_10341_cast_fp16_0, y = const_288_to_fp16)[name = string("op_10344_cast_fp16")]; tensor hidden_states_201_cast_fp16 = add(x = x_487_cast_fp16, y = var_10344_cast_fp16)[name = string("hidden_states_201_cast_fp16")]; tensor per_layer_slice_33_begin_0 = const()[name = string("per_layer_slice_33_begin_0"), val = tensor([0, 0, 4096])]; tensor per_layer_slice_33_end_0 = const()[name = string("per_layer_slice_33_end_0"), val = tensor([1, 1, 4352])]; tensor per_layer_slice_33_end_mask_0 = const()[name = string("per_layer_slice_33_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_33_cast_fp16 = slice_by_index(begin = per_layer_slice_33_begin_0, end = per_layer_slice_33_end_0, end_mask = per_layer_slice_33_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_33_cast_fp16")]; tensor gated_65 = linear(bias = linear_0_bias_0, weight = layers_16_per_layer_input_gate_weight_palettized, x = hidden_states_201_cast_fp16)[name = string("linear_32")]; string gated_67_mode_0 = const()[name = string("gated_67_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_67 = gelu(mode = gated_67_mode_0, x = gated_65)[name = string("gated_67")]; tensor input_483_cast_fp16 = mul(x = gated_67, y = per_layer_slice_33_cast_fp16)[name = string("input_483_cast_fp16")]; tensor layers_16_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096769088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096965760))))[name = string("layers_16_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_33_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_16_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_483_cast_fp16)[name = string("linear_33_cast_fp16")]; int32 var_10381 = const()[name = string("op_10381"), val = int32(-1)]; fp16 const_289_promoted_to_fp16 = const()[name = string("const_289_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10387_cast_fp16 = mul(x = linear_33_cast_fp16, y = const_289_promoted_to_fp16)[name = string("op_10387_cast_fp16")]; bool input_485_interleave_0 = const()[name = string("input_485_interleave_0"), val = bool(false)]; tensor input_485_cast_fp16 = concat(axis = var_10381, interleave = input_485_interleave_0, values = (linear_33_cast_fp16, var_10387_cast_fp16))[name = string("input_485_cast_fp16")]; tensor normed_465_axes_0 = const()[name = string("normed_465_axes_0"), val = tensor([-1])]; fp16 var_10379_to_fp16 = const()[name = string("op_10379_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_465_cast_fp16 = layer_norm(axes = normed_465_axes_0, epsilon = var_10379_to_fp16, x = input_485_cast_fp16)[name = string("normed_465_cast_fp16")]; tensor var_10392_split_sizes_0 = const()[name = string("op_10392_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10392_axis_0 = const()[name = string("op_10392_axis_0"), val = int32(-1)]; tensor var_10392_cast_fp16_0, tensor var_10392_cast_fp16_1 = split(axis = var_10392_axis_0, split_sizes = var_10392_split_sizes_0, x = normed_465_cast_fp16)[name = string("op_10392_cast_fp16")]; tensor const_290_to_fp16 = const()[name = string("const_290_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096967360)))]; tensor var_10395_cast_fp16 = mul(x = var_10392_cast_fp16_0, y = const_290_to_fp16)[name = string("op_10395_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_201_cast_fp16, y = var_10395_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; tensor layers_16_layer_scalar_to_fp16 = const()[name = string("layers_16_layer_scalar_to_fp16"), val = tensor([0x1.2cp-1])]; tensor x_499_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = layers_16_layer_scalar_to_fp16)[name = string("x_499_cast_fp16")]; int32 var_10403 = const()[name = string("op_10403"), val = int32(-1)]; fp16 const_291_promoted_to_fp16 = const()[name = string("const_291_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10409_cast_fp16 = mul(x = x_499_cast_fp16, y = const_291_promoted_to_fp16)[name = string("op_10409_cast_fp16")]; bool input_487_interleave_0 = const()[name = string("input_487_interleave_0"), val = bool(false)]; tensor input_487_cast_fp16 = concat(axis = var_10403, interleave = input_487_interleave_0, values = (x_499_cast_fp16, var_10409_cast_fp16))[name = string("input_487_cast_fp16")]; tensor normed_469_axes_0 = const()[name = string("normed_469_axes_0"), val = tensor([-1])]; fp16 var_10401_to_fp16 = const()[name = string("op_10401_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_469_cast_fp16 = layer_norm(axes = normed_469_axes_0, epsilon = var_10401_to_fp16, x = input_487_cast_fp16)[name = string("normed_469_cast_fp16")]; tensor var_10414_split_sizes_0 = const()[name = string("op_10414_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10414_axis_0 = const()[name = string("op_10414_axis_0"), val = int32(-1)]; tensor var_10414_cast_fp16_0, tensor var_10414_cast_fp16_1 = split(axis = var_10414_axis_0, split_sizes = var_10414_split_sizes_0, x = normed_469_cast_fp16)[name = string("op_10414_cast_fp16")]; tensor const_292_to_fp16 = const()[name = string("const_292_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096970496)))]; tensor var_10417_cast_fp16 = mul(x = var_10414_cast_fp16_0, y = const_292_to_fp16)[name = string("op_10417_cast_fp16")]; tensor var_10425 = const()[name = string("op_10425"), val = tensor([0, 2, 1])]; tensor var_10428_axes_0 = const()[name = string("op_10428_axes_0"), val = tensor([2])]; tensor var_10426_cast_fp16 = transpose(perm = var_10425, x = var_10417_cast_fp16)[name = string("transpose_127")]; tensor var_10428_cast_fp16 = expand_dims(axes = var_10428_axes_0, x = var_10426_cast_fp16)[name = string("op_10428_cast_fp16")]; string var_10444_pad_type_0 = const()[name = string("op_10444_pad_type_0"), val = string("valid")]; tensor var_10444_strides_0 = const()[name = string("op_10444_strides_0"), val = tensor([1, 1])]; tensor var_10444_pad_0 = const()[name = string("op_10444_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10444_dilations_0 = const()[name = string("op_10444_dilations_0"), val = tensor([1, 1])]; int32 var_10444_groups_0 = const()[name = string("op_10444_groups_0"), val = int32(1)]; tensor var_10444 = conv(dilations = var_10444_dilations_0, groups = var_10444_groups_0, pad = var_10444_pad_0, pad_type = var_10444_pad_type_0, strides = var_10444_strides_0, weight = layers_17_self_attn_q_proj_weight_palettized, x = var_10428_cast_fp16)[name = string("op_10444")]; tensor var_10449 = const()[name = string("op_10449"), val = tensor([1, 8, 256, 1])]; tensor var_10450 = reshape(shape = var_10449, x = var_10444)[name = string("op_10450")]; tensor var_10455 = const()[name = string("op_10455"), val = tensor([0, 1, 3, 2])]; tensor var_10465 = const()[name = string("op_10465"), val = tensor([1, 8, 256])]; tensor var_10456 = transpose(perm = var_10455, x = var_10450)[name = string("transpose_126")]; tensor x_503 = reshape(shape = var_10465, x = var_10456)[name = string("x_503")]; int32 var_10471 = const()[name = string("op_10471"), val = int32(-1)]; fp16 const_293_promoted_to_fp16 = const()[name = string("const_293_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10477_cast_fp16 = mul(x = x_503, y = const_293_promoted_to_fp16)[name = string("op_10477_cast_fp16")]; bool input_491_interleave_0 = const()[name = string("input_491_interleave_0"), val = bool(false)]; tensor input_491_cast_fp16 = concat(axis = var_10471, interleave = input_491_interleave_0, values = (x_503, var_10477_cast_fp16))[name = string("input_491_cast_fp16")]; tensor normed_473_axes_0 = const()[name = string("normed_473_axes_0"), val = tensor([-1])]; fp16 var_10469_to_fp16 = const()[name = string("op_10469_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_473_cast_fp16 = layer_norm(axes = normed_473_axes_0, epsilon = var_10469_to_fp16, x = input_491_cast_fp16)[name = string("normed_473_cast_fp16")]; tensor var_10482_split_sizes_0 = const()[name = string("op_10482_split_sizes_0"), val = tensor([256, 256])]; int32 var_10482_axis_0 = const()[name = string("op_10482_axis_0"), val = int32(-1)]; tensor var_10482_cast_fp16_0, tensor var_10482_cast_fp16_1 = split(axis = var_10482_axis_0, split_sizes = var_10482_split_sizes_0, x = normed_473_cast_fp16)[name = string("op_10482_cast_fp16")]; tensor var_10485_cast_fp16 = mul(x = var_10482_cast_fp16_0, y = const_234_to_fp16)[name = string("op_10485_cast_fp16")]; tensor var_10491 = const()[name = string("op_10491"), val = tensor([1, 8, 1, 256])]; tensor q_135 = reshape(shape = var_10491, x = var_10485_cast_fp16)[name = string("q_135")]; tensor var_10493 = mul(x = q_135, y = cos_1)[name = string("op_10493")]; tensor var_10494_split_sizes_0 = const()[name = string("op_10494_split_sizes_0"), val = tensor([128, 128])]; int32 var_10494_axis_0 = const()[name = string("op_10494_axis_0"), val = int32(-1)]; tensor var_10494_0, tensor var_10494_1 = split(axis = var_10494_axis_0, split_sizes = var_10494_split_sizes_0, x = q_135)[name = string("op_10494")]; fp16 const_295_promoted = const()[name = string("const_295_promoted"), val = fp16(-0x1p+0)]; tensor var_10496 = mul(x = var_10494_1, y = const_295_promoted)[name = string("op_10496")]; int32 var_10498 = const()[name = string("op_10498"), val = int32(-1)]; bool var_10499_interleave_0 = const()[name = string("op_10499_interleave_0"), val = bool(false)]; tensor var_10499 = concat(axis = var_10498, interleave = var_10499_interleave_0, values = (var_10496, var_10494_0))[name = string("op_10499")]; tensor var_10500 = mul(x = var_10499, y = sin_1)[name = string("op_10500")]; tensor q_137 = add(x = var_10493, y = var_10500)[name = string("q_137")]; bool var_10514_transpose_x_0 = const()[name = string("op_10514_transpose_x_0"), val = bool(false)]; bool var_10514_transpose_y_0 = const()[name = string("op_10514_transpose_y_0"), val = bool(false)]; tensor var_10514_cast_fp16 = matmul(transpose_x = var_10514_transpose_x_0, transpose_y = var_10514_transpose_y_0, x = q_137, y = transpose_153_cast_fp16)[name = string("op_10514_cast_fp16")]; tensor attn_weights_105_cast_fp16 = add(x = var_10514_cast_fp16, y = causal_mask)[name = string("attn_weights_105_cast_fp16")]; int32 var_10519 = const()[name = string("op_10519"), val = int32(-1)]; tensor attn_weights_107_cast_fp16 = softmax(axis = var_10519, x = attn_weights_105_cast_fp16)[name = string("attn_weights_107_cast_fp16")]; bool attn_output_103_transpose_x_0 = const()[name = string("attn_output_103_transpose_x_0"), val = bool(false)]; bool attn_output_103_transpose_y_0 = const()[name = string("attn_output_103_transpose_y_0"), val = bool(false)]; tensor attn_output_103_cast_fp16 = matmul(transpose_x = attn_output_103_transpose_x_0, transpose_y = attn_output_103_transpose_y_0, x = attn_weights_107_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor var_10527 = const()[name = string("op_10527"), val = tensor([0, 2, 1, 3])]; tensor var_10534 = const()[name = string("op_10534"), val = tensor([1, 1, -1])]; tensor var_10528_cast_fp16 = transpose(perm = var_10527, x = attn_output_103_cast_fp16)[name = string("transpose_125")]; tensor attn_output_105_cast_fp16 = reshape(shape = var_10534, x = var_10528_cast_fp16)[name = string("attn_output_105_cast_fp16")]; tensor var_10539 = const()[name = string("op_10539"), val = tensor([0, 2, 1])]; string var_10555_pad_type_0 = const()[name = string("op_10555_pad_type_0"), val = string("valid")]; int32 var_10555_groups_0 = const()[name = string("op_10555_groups_0"), val = int32(1)]; tensor var_10555_strides_0 = const()[name = string("op_10555_strides_0"), val = tensor([1])]; tensor var_10555_pad_0 = const()[name = string("op_10555_pad_0"), val = tensor([0, 0])]; tensor var_10555_dilations_0 = const()[name = string("op_10555_dilations_0"), val = tensor([1])]; tensor squeeze_17_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1096973632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098546560))))[name = string("squeeze_17_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10540_cast_fp16 = transpose(perm = var_10539, x = attn_output_105_cast_fp16)[name = string("transpose_124")]; tensor var_10555_cast_fp16 = conv(dilations = var_10555_dilations_0, groups = var_10555_groups_0, pad = var_10555_pad_0, pad_type = var_10555_pad_type_0, strides = var_10555_strides_0, weight = squeeze_17_cast_fp16_to_fp32_to_fp16_palettized, x = var_10540_cast_fp16)[name = string("op_10555_cast_fp16")]; tensor var_10559 = const()[name = string("op_10559"), val = tensor([0, 2, 1])]; int32 var_10565 = const()[name = string("op_10565"), val = int32(-1)]; fp16 const_296_promoted_to_fp16 = const()[name = string("const_296_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_507_cast_fp16 = transpose(perm = var_10559, x = var_10555_cast_fp16)[name = string("transpose_123")]; tensor var_10571_cast_fp16 = mul(x = x_507_cast_fp16, y = const_296_promoted_to_fp16)[name = string("op_10571_cast_fp16")]; bool input_495_interleave_0 = const()[name = string("input_495_interleave_0"), val = bool(false)]; tensor input_495_cast_fp16 = concat(axis = var_10565, interleave = input_495_interleave_0, values = (x_507_cast_fp16, var_10571_cast_fp16))[name = string("input_495_cast_fp16")]; tensor normed_477_axes_0 = const()[name = string("normed_477_axes_0"), val = tensor([-1])]; fp16 var_10563_to_fp16 = const()[name = string("op_10563_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_477_cast_fp16 = layer_norm(axes = normed_477_axes_0, epsilon = var_10563_to_fp16, x = input_495_cast_fp16)[name = string("normed_477_cast_fp16")]; tensor var_10576_split_sizes_0 = const()[name = string("op_10576_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10576_axis_0 = const()[name = string("op_10576_axis_0"), val = int32(-1)]; tensor var_10576_cast_fp16_0, tensor var_10576_cast_fp16_1 = split(axis = var_10576_axis_0, split_sizes = var_10576_split_sizes_0, x = normed_477_cast_fp16)[name = string("op_10576_cast_fp16")]; tensor const_297_to_fp16 = const()[name = string("const_297_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098548160)))]; tensor var_10579_cast_fp16 = mul(x = var_10576_cast_fp16_0, y = const_297_to_fp16)[name = string("op_10579_cast_fp16")]; tensor x_511_cast_fp16 = add(x = x_499_cast_fp16, y = var_10579_cast_fp16)[name = string("x_511_cast_fp16")]; int32 var_10586 = const()[name = string("op_10586"), val = int32(-1)]; fp16 const_298_promoted_to_fp16 = const()[name = string("const_298_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10592_cast_fp16 = mul(x = x_511_cast_fp16, y = const_298_promoted_to_fp16)[name = string("op_10592_cast_fp16")]; bool input_497_interleave_0 = const()[name = string("input_497_interleave_0"), val = bool(false)]; tensor input_497_cast_fp16 = concat(axis = var_10586, interleave = input_497_interleave_0, values = (x_511_cast_fp16, var_10592_cast_fp16))[name = string("input_497_cast_fp16")]; tensor normed_481_axes_0 = const()[name = string("normed_481_axes_0"), val = tensor([-1])]; fp16 var_10584_to_fp16 = const()[name = string("op_10584_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_481_cast_fp16 = layer_norm(axes = normed_481_axes_0, epsilon = var_10584_to_fp16, x = input_497_cast_fp16)[name = string("normed_481_cast_fp16")]; tensor var_10597_split_sizes_0 = const()[name = string("op_10597_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10597_axis_0 = const()[name = string("op_10597_axis_0"), val = int32(-1)]; tensor var_10597_cast_fp16_0, tensor var_10597_cast_fp16_1 = split(axis = var_10597_axis_0, split_sizes = var_10597_split_sizes_0, x = normed_481_cast_fp16)[name = string("op_10597_cast_fp16")]; tensor const_299_to_fp16 = const()[name = string("const_299_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098551296)))]; tensor var_10600_cast_fp16 = mul(x = var_10597_cast_fp16_0, y = const_299_to_fp16)[name = string("op_10600_cast_fp16")]; tensor var_10613 = const()[name = string("op_10613"), val = tensor([0, 2, 1])]; tensor input_499_axes_0 = const()[name = string("input_499_axes_0"), val = tensor([2])]; tensor var_10614 = transpose(perm = var_10613, x = var_10600_cast_fp16)[name = string("transpose_122")]; tensor input_499 = expand_dims(axes = input_499_axes_0, x = var_10614)[name = string("input_499")]; string gate_69_pad_type_0 = const()[name = string("gate_69_pad_type_0"), val = string("valid")]; tensor gate_69_strides_0 = const()[name = string("gate_69_strides_0"), val = tensor([1, 1])]; tensor gate_69_pad_0 = const()[name = string("gate_69_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_69_dilations_0 = const()[name = string("gate_69_dilations_0"), val = tensor([1, 1])]; int32 gate_69_groups_0 = const()[name = string("gate_69_groups_0"), val = int32(1)]; tensor gate_69 = conv(dilations = gate_69_dilations_0, groups = gate_69_groups_0, pad = gate_69_pad_0, pad_type = gate_69_pad_type_0, strides = gate_69_strides_0, weight = layers_17_mlp_gate_proj_weight_palettized, x = input_499)[name = string("gate_69")]; string up_35_pad_type_0 = const()[name = string("up_35_pad_type_0"), val = string("valid")]; tensor up_35_strides_0 = const()[name = string("up_35_strides_0"), val = tensor([1, 1])]; tensor up_35_pad_0 = const()[name = string("up_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_35_dilations_0 = const()[name = string("up_35_dilations_0"), val = tensor([1, 1])]; int32 up_35_groups_0 = const()[name = string("up_35_groups_0"), val = int32(1)]; tensor up_35 = conv(dilations = up_35_dilations_0, groups = up_35_groups_0, pad = up_35_pad_0, pad_type = up_35_pad_type_0, strides = up_35_strides_0, weight = layers_17_mlp_up_proj_weight_palettized, x = input_499)[name = string("up_35")]; string gate_71_mode_0 = const()[name = string("gate_71_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_71 = gelu(mode = gate_71_mode_0, x = gate_69)[name = string("gate_71")]; tensor input_501 = mul(x = gate_71, y = up_35)[name = string("input_501")]; string mlp_out_35_pad_type_0 = const()[name = string("mlp_out_35_pad_type_0"), val = string("valid")]; tensor mlp_out_35_strides_0 = const()[name = string("mlp_out_35_strides_0"), val = tensor([1, 1])]; tensor mlp_out_35_pad_0 = const()[name = string("mlp_out_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_35_dilations_0 = const()[name = string("mlp_out_35_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_35_groups_0 = const()[name = string("mlp_out_35_groups_0"), val = int32(1)]; tensor mlp_out_35 = conv(dilations = mlp_out_35_dilations_0, groups = mlp_out_35_groups_0, pad = mlp_out_35_pad_0, pad_type = mlp_out_35_pad_type_0, strides = mlp_out_35_strides_0, weight = layers_17_mlp_down_proj_weight_palettized, x = input_501)[name = string("mlp_out_35")]; tensor var_10654_axes_0 = const()[name = string("op_10654_axes_0"), val = tensor([2])]; tensor var_10654 = squeeze(axes = var_10654_axes_0, x = mlp_out_35)[name = string("op_10654")]; tensor var_10658 = const()[name = string("op_10658"), val = tensor([0, 2, 1])]; int32 var_10664 = const()[name = string("op_10664"), val = int32(-1)]; fp16 const_300_promoted_to_fp16 = const()[name = string("const_300_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_515 = transpose(perm = var_10658, x = var_10654)[name = string("transpose_121")]; tensor var_10670_cast_fp16 = mul(x = x_515, y = const_300_promoted_to_fp16)[name = string("op_10670_cast_fp16")]; bool input_503_interleave_0 = const()[name = string("input_503_interleave_0"), val = bool(false)]; tensor input_503_cast_fp16 = concat(axis = var_10664, interleave = input_503_interleave_0, values = (x_515, var_10670_cast_fp16))[name = string("input_503_cast_fp16")]; tensor normed_485_axes_0 = const()[name = string("normed_485_axes_0"), val = tensor([-1])]; fp16 var_10662_to_fp16 = const()[name = string("op_10662_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_485_cast_fp16 = layer_norm(axes = normed_485_axes_0, epsilon = var_10662_to_fp16, x = input_503_cast_fp16)[name = string("normed_485_cast_fp16")]; tensor var_10675_split_sizes_0 = const()[name = string("op_10675_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10675_axis_0 = const()[name = string("op_10675_axis_0"), val = int32(-1)]; tensor var_10675_cast_fp16_0, tensor var_10675_cast_fp16_1 = split(axis = var_10675_axis_0, split_sizes = var_10675_split_sizes_0, x = normed_485_cast_fp16)[name = string("op_10675_cast_fp16")]; tensor const_301_to_fp16 = const()[name = string("const_301_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098554432)))]; tensor var_10678_cast_fp16 = mul(x = var_10675_cast_fp16_0, y = const_301_to_fp16)[name = string("op_10678_cast_fp16")]; tensor hidden_states_213_cast_fp16 = add(x = x_511_cast_fp16, y = var_10678_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor per_layer_slice_35_begin_0 = const()[name = string("per_layer_slice_35_begin_0"), val = tensor([0, 0, 4352])]; tensor per_layer_slice_35_end_0 = const()[name = string("per_layer_slice_35_end_0"), val = tensor([1, 1, 4608])]; tensor per_layer_slice_35_end_mask_0 = const()[name = string("per_layer_slice_35_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_35_cast_fp16 = slice_by_index(begin = per_layer_slice_35_begin_0, end = per_layer_slice_35_end_0, end_mask = per_layer_slice_35_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_35_cast_fp16")]; tensor gated_69 = linear(bias = linear_0_bias_0, weight = layers_17_per_layer_input_gate_weight_palettized, x = hidden_states_213_cast_fp16)[name = string("linear_34")]; string gated_71_mode_0 = const()[name = string("gated_71_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_71 = gelu(mode = gated_71_mode_0, x = gated_69)[name = string("gated_71")]; tensor input_507_cast_fp16 = mul(x = gated_71, y = per_layer_slice_35_cast_fp16)[name = string("input_507_cast_fp16")]; tensor layers_17_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098557568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098754240))))[name = string("layers_17_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_35_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_17_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_507_cast_fp16)[name = string("linear_35_cast_fp16")]; int32 var_10715 = const()[name = string("op_10715"), val = int32(-1)]; fp16 const_302_promoted_to_fp16 = const()[name = string("const_302_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10721_cast_fp16 = mul(x = linear_35_cast_fp16, y = const_302_promoted_to_fp16)[name = string("op_10721_cast_fp16")]; bool input_509_interleave_0 = const()[name = string("input_509_interleave_0"), val = bool(false)]; tensor input_509_cast_fp16 = concat(axis = var_10715, interleave = input_509_interleave_0, values = (linear_35_cast_fp16, var_10721_cast_fp16))[name = string("input_509_cast_fp16")]; tensor normed_489_axes_0 = const()[name = string("normed_489_axes_0"), val = tensor([-1])]; fp16 var_10713_to_fp16 = const()[name = string("op_10713_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_489_cast_fp16 = layer_norm(axes = normed_489_axes_0, epsilon = var_10713_to_fp16, x = input_509_cast_fp16)[name = string("normed_489_cast_fp16")]; tensor var_10726_split_sizes_0 = const()[name = string("op_10726_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10726_axis_0 = const()[name = string("op_10726_axis_0"), val = int32(-1)]; tensor var_10726_cast_fp16_0, tensor var_10726_cast_fp16_1 = split(axis = var_10726_axis_0, split_sizes = var_10726_split_sizes_0, x = normed_489_cast_fp16)[name = string("op_10726_cast_fp16")]; tensor const_303_to_fp16 = const()[name = string("const_303_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098755840)))]; tensor var_10729_cast_fp16 = mul(x = var_10726_cast_fp16_0, y = const_303_to_fp16)[name = string("op_10729_cast_fp16")]; tensor hidden_states_217_cast_fp16 = add(x = hidden_states_213_cast_fp16, y = var_10729_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor layers_17_layer_scalar_to_fp16 = const()[name = string("layers_17_layer_scalar_to_fp16"), val = tensor([0x1.5p-1])]; tensor x_523_cast_fp16 = mul(x = hidden_states_217_cast_fp16, y = layers_17_layer_scalar_to_fp16)[name = string("x_523_cast_fp16")]; int32 var_10737 = const()[name = string("op_10737"), val = int32(-1)]; fp16 const_304_promoted_to_fp16 = const()[name = string("const_304_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10743_cast_fp16 = mul(x = x_523_cast_fp16, y = const_304_promoted_to_fp16)[name = string("op_10743_cast_fp16")]; bool input_511_interleave_0 = const()[name = string("input_511_interleave_0"), val = bool(false)]; tensor input_511_cast_fp16 = concat(axis = var_10737, interleave = input_511_interleave_0, values = (x_523_cast_fp16, var_10743_cast_fp16))[name = string("input_511_cast_fp16")]; tensor normed_493_axes_0 = const()[name = string("normed_493_axes_0"), val = tensor([-1])]; fp16 var_10735_to_fp16 = const()[name = string("op_10735_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_493_cast_fp16 = layer_norm(axes = normed_493_axes_0, epsilon = var_10735_to_fp16, x = input_511_cast_fp16)[name = string("normed_493_cast_fp16")]; tensor var_10748_split_sizes_0 = const()[name = string("op_10748_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10748_axis_0 = const()[name = string("op_10748_axis_0"), val = int32(-1)]; tensor var_10748_cast_fp16_0, tensor var_10748_cast_fp16_1 = split(axis = var_10748_axis_0, split_sizes = var_10748_split_sizes_0, x = normed_493_cast_fp16)[name = string("op_10748_cast_fp16")]; tensor const_305_to_fp16 = const()[name = string("const_305_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098758976)))]; tensor var_10751_cast_fp16 = mul(x = var_10748_cast_fp16_0, y = const_305_to_fp16)[name = string("op_10751_cast_fp16")]; tensor var_10759 = const()[name = string("op_10759"), val = tensor([0, 2, 1])]; tensor var_10762_axes_0 = const()[name = string("op_10762_axes_0"), val = tensor([2])]; tensor var_10760_cast_fp16 = transpose(perm = var_10759, x = var_10751_cast_fp16)[name = string("transpose_120")]; tensor var_10762_cast_fp16 = expand_dims(axes = var_10762_axes_0, x = var_10760_cast_fp16)[name = string("op_10762_cast_fp16")]; string var_10778_pad_type_0 = const()[name = string("op_10778_pad_type_0"), val = string("valid")]; tensor var_10778_strides_0 = const()[name = string("op_10778_strides_0"), val = tensor([1, 1])]; tensor var_10778_pad_0 = const()[name = string("op_10778_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10778_dilations_0 = const()[name = string("op_10778_dilations_0"), val = tensor([1, 1])]; int32 var_10778_groups_0 = const()[name = string("op_10778_groups_0"), val = int32(1)]; tensor var_10778 = conv(dilations = var_10778_dilations_0, groups = var_10778_groups_0, pad = var_10778_pad_0, pad_type = var_10778_pad_type_0, strides = var_10778_strides_0, weight = layers_18_self_attn_q_proj_weight_palettized, x = var_10762_cast_fp16)[name = string("op_10778")]; tensor var_10783 = const()[name = string("op_10783"), val = tensor([1, 8, 256, 1])]; tensor var_10784 = reshape(shape = var_10783, x = var_10778)[name = string("op_10784")]; tensor var_10789 = const()[name = string("op_10789"), val = tensor([0, 1, 3, 2])]; tensor var_10799 = const()[name = string("op_10799"), val = tensor([1, 8, 256])]; tensor var_10790 = transpose(perm = var_10789, x = var_10784)[name = string("transpose_119")]; tensor x_527 = reshape(shape = var_10799, x = var_10790)[name = string("x_527")]; int32 var_10805 = const()[name = string("op_10805"), val = int32(-1)]; fp16 const_306_promoted_to_fp16 = const()[name = string("const_306_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10811_cast_fp16 = mul(x = x_527, y = const_306_promoted_to_fp16)[name = string("op_10811_cast_fp16")]; bool input_515_interleave_0 = const()[name = string("input_515_interleave_0"), val = bool(false)]; tensor input_515_cast_fp16 = concat(axis = var_10805, interleave = input_515_interleave_0, values = (x_527, var_10811_cast_fp16))[name = string("input_515_cast_fp16")]; tensor normed_497_axes_0 = const()[name = string("normed_497_axes_0"), val = tensor([-1])]; fp16 var_10803_to_fp16 = const()[name = string("op_10803_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_497_cast_fp16 = layer_norm(axes = normed_497_axes_0, epsilon = var_10803_to_fp16, x = input_515_cast_fp16)[name = string("normed_497_cast_fp16")]; tensor var_10816_split_sizes_0 = const()[name = string("op_10816_split_sizes_0"), val = tensor([256, 256])]; int32 var_10816_axis_0 = const()[name = string("op_10816_axis_0"), val = int32(-1)]; tensor var_10816_cast_fp16_0, tensor var_10816_cast_fp16_1 = split(axis = var_10816_axis_0, split_sizes = var_10816_split_sizes_0, x = normed_497_cast_fp16)[name = string("op_10816_cast_fp16")]; tensor var_10819_cast_fp16 = mul(x = var_10816_cast_fp16_0, y = const_234_to_fp16)[name = string("op_10819_cast_fp16")]; tensor var_10825 = const()[name = string("op_10825"), val = tensor([1, 8, 1, 256])]; tensor q_141 = reshape(shape = var_10825, x = var_10819_cast_fp16)[name = string("q_141")]; tensor var_10827 = mul(x = q_141, y = cos_1)[name = string("op_10827")]; tensor var_10828_split_sizes_0 = const()[name = string("op_10828_split_sizes_0"), val = tensor([128, 128])]; int32 var_10828_axis_0 = const()[name = string("op_10828_axis_0"), val = int32(-1)]; tensor var_10828_0, tensor var_10828_1 = split(axis = var_10828_axis_0, split_sizes = var_10828_split_sizes_0, x = q_141)[name = string("op_10828")]; fp16 const_308_promoted = const()[name = string("const_308_promoted"), val = fp16(-0x1p+0)]; tensor var_10830 = mul(x = var_10828_1, y = const_308_promoted)[name = string("op_10830")]; int32 var_10832 = const()[name = string("op_10832"), val = int32(-1)]; bool var_10833_interleave_0 = const()[name = string("op_10833_interleave_0"), val = bool(false)]; tensor var_10833 = concat(axis = var_10832, interleave = var_10833_interleave_0, values = (var_10830, var_10828_0))[name = string("op_10833")]; tensor var_10834 = mul(x = var_10833, y = sin_1)[name = string("op_10834")]; tensor q_143 = add(x = var_10827, y = var_10834)[name = string("q_143")]; bool var_10848_transpose_x_0 = const()[name = string("op_10848_transpose_x_0"), val = bool(false)]; bool var_10848_transpose_y_0 = const()[name = string("op_10848_transpose_y_0"), val = bool(false)]; tensor var_10848_cast_fp16 = matmul(transpose_x = var_10848_transpose_x_0, transpose_y = var_10848_transpose_y_0, x = q_143, y = transpose_153_cast_fp16)[name = string("op_10848_cast_fp16")]; tensor attn_weights_111_cast_fp16 = add(x = var_10848_cast_fp16, y = causal_mask)[name = string("attn_weights_111_cast_fp16")]; int32 var_10853 = const()[name = string("op_10853"), val = int32(-1)]; tensor attn_weights_113_cast_fp16 = softmax(axis = var_10853, x = attn_weights_111_cast_fp16)[name = string("attn_weights_113_cast_fp16")]; bool attn_output_109_transpose_x_0 = const()[name = string("attn_output_109_transpose_x_0"), val = bool(false)]; bool attn_output_109_transpose_y_0 = const()[name = string("attn_output_109_transpose_y_0"), val = bool(false)]; tensor attn_output_109_cast_fp16 = matmul(transpose_x = attn_output_109_transpose_x_0, transpose_y = attn_output_109_transpose_y_0, x = attn_weights_113_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_109_cast_fp16")]; tensor var_10861 = const()[name = string("op_10861"), val = tensor([0, 2, 1, 3])]; tensor var_10868 = const()[name = string("op_10868"), val = tensor([1, 1, -1])]; tensor var_10862_cast_fp16 = transpose(perm = var_10861, x = attn_output_109_cast_fp16)[name = string("transpose_118")]; tensor attn_output_111_cast_fp16 = reshape(shape = var_10868, x = var_10862_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor var_10873 = const()[name = string("op_10873"), val = tensor([0, 2, 1])]; string var_10889_pad_type_0 = const()[name = string("op_10889_pad_type_0"), val = string("valid")]; int32 var_10889_groups_0 = const()[name = string("op_10889_groups_0"), val = int32(1)]; tensor var_10889_strides_0 = const()[name = string("op_10889_strides_0"), val = tensor([1])]; tensor var_10889_pad_0 = const()[name = string("op_10889_pad_0"), val = tensor([0, 0])]; tensor var_10889_dilations_0 = const()[name = string("op_10889_dilations_0"), val = tensor([1])]; tensor squeeze_18_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1098762112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100335040))))[name = string("squeeze_18_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10874_cast_fp16 = transpose(perm = var_10873, x = attn_output_111_cast_fp16)[name = string("transpose_117")]; tensor var_10889_cast_fp16 = conv(dilations = var_10889_dilations_0, groups = var_10889_groups_0, pad = var_10889_pad_0, pad_type = var_10889_pad_type_0, strides = var_10889_strides_0, weight = squeeze_18_cast_fp16_to_fp32_to_fp16_palettized, x = var_10874_cast_fp16)[name = string("op_10889_cast_fp16")]; tensor var_10893 = const()[name = string("op_10893"), val = tensor([0, 2, 1])]; int32 var_10899 = const()[name = string("op_10899"), val = int32(-1)]; fp16 const_309_promoted_to_fp16 = const()[name = string("const_309_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_531_cast_fp16 = transpose(perm = var_10893, x = var_10889_cast_fp16)[name = string("transpose_116")]; tensor var_10905_cast_fp16 = mul(x = x_531_cast_fp16, y = const_309_promoted_to_fp16)[name = string("op_10905_cast_fp16")]; bool input_519_interleave_0 = const()[name = string("input_519_interleave_0"), val = bool(false)]; tensor input_519_cast_fp16 = concat(axis = var_10899, interleave = input_519_interleave_0, values = (x_531_cast_fp16, var_10905_cast_fp16))[name = string("input_519_cast_fp16")]; tensor normed_501_axes_0 = const()[name = string("normed_501_axes_0"), val = tensor([-1])]; fp16 var_10897_to_fp16 = const()[name = string("op_10897_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_501_cast_fp16 = layer_norm(axes = normed_501_axes_0, epsilon = var_10897_to_fp16, x = input_519_cast_fp16)[name = string("normed_501_cast_fp16")]; tensor var_10910_split_sizes_0 = const()[name = string("op_10910_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10910_axis_0 = const()[name = string("op_10910_axis_0"), val = int32(-1)]; tensor var_10910_cast_fp16_0, tensor var_10910_cast_fp16_1 = split(axis = var_10910_axis_0, split_sizes = var_10910_split_sizes_0, x = normed_501_cast_fp16)[name = string("op_10910_cast_fp16")]; tensor const_310_to_fp16 = const()[name = string("const_310_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100336640)))]; tensor var_10913_cast_fp16 = mul(x = var_10910_cast_fp16_0, y = const_310_to_fp16)[name = string("op_10913_cast_fp16")]; tensor x_535_cast_fp16 = add(x = x_523_cast_fp16, y = var_10913_cast_fp16)[name = string("x_535_cast_fp16")]; int32 var_10920 = const()[name = string("op_10920"), val = int32(-1)]; fp16 const_311_promoted_to_fp16 = const()[name = string("const_311_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10926_cast_fp16 = mul(x = x_535_cast_fp16, y = const_311_promoted_to_fp16)[name = string("op_10926_cast_fp16")]; bool input_521_interleave_0 = const()[name = string("input_521_interleave_0"), val = bool(false)]; tensor input_521_cast_fp16 = concat(axis = var_10920, interleave = input_521_interleave_0, values = (x_535_cast_fp16, var_10926_cast_fp16))[name = string("input_521_cast_fp16")]; tensor normed_505_axes_0 = const()[name = string("normed_505_axes_0"), val = tensor([-1])]; fp16 var_10918_to_fp16 = const()[name = string("op_10918_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_505_cast_fp16 = layer_norm(axes = normed_505_axes_0, epsilon = var_10918_to_fp16, x = input_521_cast_fp16)[name = string("normed_505_cast_fp16")]; tensor var_10931_split_sizes_0 = const()[name = string("op_10931_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_10931_axis_0 = const()[name = string("op_10931_axis_0"), val = int32(-1)]; tensor var_10931_cast_fp16_0, tensor var_10931_cast_fp16_1 = split(axis = var_10931_axis_0, split_sizes = var_10931_split_sizes_0, x = normed_505_cast_fp16)[name = string("op_10931_cast_fp16")]; tensor const_312_to_fp16 = const()[name = string("const_312_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100339776)))]; tensor var_10934_cast_fp16 = mul(x = var_10931_cast_fp16_0, y = const_312_to_fp16)[name = string("op_10934_cast_fp16")]; tensor var_10947 = const()[name = string("op_10947"), val = tensor([0, 2, 1])]; tensor input_523_axes_0 = const()[name = string("input_523_axes_0"), val = tensor([2])]; tensor var_10948 = transpose(perm = var_10947, x = var_10934_cast_fp16)[name = string("transpose_115")]; tensor input_523 = expand_dims(axes = input_523_axes_0, x = var_10948)[name = string("input_523")]; string gate_73_pad_type_0 = const()[name = string("gate_73_pad_type_0"), val = string("valid")]; tensor gate_73_strides_0 = const()[name = string("gate_73_strides_0"), val = tensor([1, 1])]; tensor gate_73_pad_0 = const()[name = string("gate_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_73_dilations_0 = const()[name = string("gate_73_dilations_0"), val = tensor([1, 1])]; int32 gate_73_groups_0 = const()[name = string("gate_73_groups_0"), val = int32(1)]; tensor gate_73 = conv(dilations = gate_73_dilations_0, groups = gate_73_groups_0, pad = gate_73_pad_0, pad_type = gate_73_pad_type_0, strides = gate_73_strides_0, weight = layers_18_mlp_gate_proj_weight_palettized, x = input_523)[name = string("gate_73")]; string up_37_pad_type_0 = const()[name = string("up_37_pad_type_0"), val = string("valid")]; tensor up_37_strides_0 = const()[name = string("up_37_strides_0"), val = tensor([1, 1])]; tensor up_37_pad_0 = const()[name = string("up_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_37_dilations_0 = const()[name = string("up_37_dilations_0"), val = tensor([1, 1])]; int32 up_37_groups_0 = const()[name = string("up_37_groups_0"), val = int32(1)]; tensor up_37 = conv(dilations = up_37_dilations_0, groups = up_37_groups_0, pad = up_37_pad_0, pad_type = up_37_pad_type_0, strides = up_37_strides_0, weight = layers_18_mlp_up_proj_weight_palettized, x = input_523)[name = string("up_37")]; string gate_75_mode_0 = const()[name = string("gate_75_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_75 = gelu(mode = gate_75_mode_0, x = gate_73)[name = string("gate_75")]; tensor input_525 = mul(x = gate_75, y = up_37)[name = string("input_525")]; string mlp_out_37_pad_type_0 = const()[name = string("mlp_out_37_pad_type_0"), val = string("valid")]; tensor mlp_out_37_strides_0 = const()[name = string("mlp_out_37_strides_0"), val = tensor([1, 1])]; tensor mlp_out_37_pad_0 = const()[name = string("mlp_out_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_37_dilations_0 = const()[name = string("mlp_out_37_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_37_groups_0 = const()[name = string("mlp_out_37_groups_0"), val = int32(1)]; tensor mlp_out_37 = conv(dilations = mlp_out_37_dilations_0, groups = mlp_out_37_groups_0, pad = mlp_out_37_pad_0, pad_type = mlp_out_37_pad_type_0, strides = mlp_out_37_strides_0, weight = layers_18_mlp_down_proj_weight_palettized, x = input_525)[name = string("mlp_out_37")]; tensor var_10988_axes_0 = const()[name = string("op_10988_axes_0"), val = tensor([2])]; tensor var_10988 = squeeze(axes = var_10988_axes_0, x = mlp_out_37)[name = string("op_10988")]; tensor var_10992 = const()[name = string("op_10992"), val = tensor([0, 2, 1])]; int32 var_10998 = const()[name = string("op_10998"), val = int32(-1)]; fp16 const_313_promoted_to_fp16 = const()[name = string("const_313_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_539 = transpose(perm = var_10992, x = var_10988)[name = string("transpose_114")]; tensor var_11004_cast_fp16 = mul(x = x_539, y = const_313_promoted_to_fp16)[name = string("op_11004_cast_fp16")]; bool input_527_interleave_0 = const()[name = string("input_527_interleave_0"), val = bool(false)]; tensor input_527_cast_fp16 = concat(axis = var_10998, interleave = input_527_interleave_0, values = (x_539, var_11004_cast_fp16))[name = string("input_527_cast_fp16")]; tensor normed_509_axes_0 = const()[name = string("normed_509_axes_0"), val = tensor([-1])]; fp16 var_10996_to_fp16 = const()[name = string("op_10996_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_509_cast_fp16 = layer_norm(axes = normed_509_axes_0, epsilon = var_10996_to_fp16, x = input_527_cast_fp16)[name = string("normed_509_cast_fp16")]; tensor var_11009_split_sizes_0 = const()[name = string("op_11009_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11009_axis_0 = const()[name = string("op_11009_axis_0"), val = int32(-1)]; tensor var_11009_cast_fp16_0, tensor var_11009_cast_fp16_1 = split(axis = var_11009_axis_0, split_sizes = var_11009_split_sizes_0, x = normed_509_cast_fp16)[name = string("op_11009_cast_fp16")]; tensor const_314_to_fp16 = const()[name = string("const_314_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100342912)))]; tensor var_11012_cast_fp16 = mul(x = var_11009_cast_fp16_0, y = const_314_to_fp16)[name = string("op_11012_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = x_535_cast_fp16, y = var_11012_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; tensor per_layer_slice_37_begin_0 = const()[name = string("per_layer_slice_37_begin_0"), val = tensor([0, 0, 4608])]; tensor per_layer_slice_37_end_0 = const()[name = string("per_layer_slice_37_end_0"), val = tensor([1, 1, 4864])]; tensor per_layer_slice_37_end_mask_0 = const()[name = string("per_layer_slice_37_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_37_cast_fp16 = slice_by_index(begin = per_layer_slice_37_begin_0, end = per_layer_slice_37_end_0, end_mask = per_layer_slice_37_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_37_cast_fp16")]; tensor gated_73 = linear(bias = linear_0_bias_0, weight = layers_18_per_layer_input_gate_weight_palettized, x = hidden_states_225_cast_fp16)[name = string("linear_36")]; string gated_75_mode_0 = const()[name = string("gated_75_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_75 = gelu(mode = gated_75_mode_0, x = gated_73)[name = string("gated_75")]; tensor input_531_cast_fp16 = mul(x = gated_75, y = per_layer_slice_37_cast_fp16)[name = string("input_531_cast_fp16")]; tensor layers_18_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100346048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100542720))))[name = string("layers_18_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_37_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_18_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_531_cast_fp16)[name = string("linear_37_cast_fp16")]; int32 var_11049 = const()[name = string("op_11049"), val = int32(-1)]; fp16 const_315_promoted_to_fp16 = const()[name = string("const_315_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11055_cast_fp16 = mul(x = linear_37_cast_fp16, y = const_315_promoted_to_fp16)[name = string("op_11055_cast_fp16")]; bool input_533_interleave_0 = const()[name = string("input_533_interleave_0"), val = bool(false)]; tensor input_533_cast_fp16 = concat(axis = var_11049, interleave = input_533_interleave_0, values = (linear_37_cast_fp16, var_11055_cast_fp16))[name = string("input_533_cast_fp16")]; tensor normed_513_axes_0 = const()[name = string("normed_513_axes_0"), val = tensor([-1])]; fp16 var_11047_to_fp16 = const()[name = string("op_11047_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_513_cast_fp16 = layer_norm(axes = normed_513_axes_0, epsilon = var_11047_to_fp16, x = input_533_cast_fp16)[name = string("normed_513_cast_fp16")]; tensor var_11060_split_sizes_0 = const()[name = string("op_11060_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11060_axis_0 = const()[name = string("op_11060_axis_0"), val = int32(-1)]; tensor var_11060_cast_fp16_0, tensor var_11060_cast_fp16_1 = split(axis = var_11060_axis_0, split_sizes = var_11060_split_sizes_0, x = normed_513_cast_fp16)[name = string("op_11060_cast_fp16")]; tensor const_316_to_fp16 = const()[name = string("const_316_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100544320)))]; tensor var_11063_cast_fp16 = mul(x = var_11060_cast_fp16_0, y = const_316_to_fp16)[name = string("op_11063_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = var_11063_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; tensor layers_18_layer_scalar_to_fp16 = const()[name = string("layers_18_layer_scalar_to_fp16"), val = tensor([0x1.34p-1])]; tensor x_547_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = layers_18_layer_scalar_to_fp16)[name = string("x_547_cast_fp16")]; int32 var_11071 = const()[name = string("op_11071"), val = int32(-1)]; fp16 const_317_promoted_to_fp16 = const()[name = string("const_317_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11077_cast_fp16 = mul(x = x_547_cast_fp16, y = const_317_promoted_to_fp16)[name = string("op_11077_cast_fp16")]; bool input_535_interleave_0 = const()[name = string("input_535_interleave_0"), val = bool(false)]; tensor input_535_cast_fp16 = concat(axis = var_11071, interleave = input_535_interleave_0, values = (x_547_cast_fp16, var_11077_cast_fp16))[name = string("input_535_cast_fp16")]; tensor normed_517_axes_0 = const()[name = string("normed_517_axes_0"), val = tensor([-1])]; fp16 var_11069_to_fp16 = const()[name = string("op_11069_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_517_cast_fp16 = layer_norm(axes = normed_517_axes_0, epsilon = var_11069_to_fp16, x = input_535_cast_fp16)[name = string("normed_517_cast_fp16")]; tensor var_11082_split_sizes_0 = const()[name = string("op_11082_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11082_axis_0 = const()[name = string("op_11082_axis_0"), val = int32(-1)]; tensor var_11082_cast_fp16_0, tensor var_11082_cast_fp16_1 = split(axis = var_11082_axis_0, split_sizes = var_11082_split_sizes_0, x = normed_517_cast_fp16)[name = string("op_11082_cast_fp16")]; tensor const_318_to_fp16 = const()[name = string("const_318_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100547456)))]; tensor var_11085_cast_fp16 = mul(x = var_11082_cast_fp16_0, y = const_318_to_fp16)[name = string("op_11085_cast_fp16")]; tensor var_11093 = const()[name = string("op_11093"), val = tensor([0, 2, 1])]; tensor var_11096_axes_0 = const()[name = string("op_11096_axes_0"), val = tensor([2])]; tensor var_11094_cast_fp16 = transpose(perm = var_11093, x = var_11085_cast_fp16)[name = string("transpose_113")]; tensor var_11096_cast_fp16 = expand_dims(axes = var_11096_axes_0, x = var_11094_cast_fp16)[name = string("op_11096_cast_fp16")]; string var_11112_pad_type_0 = const()[name = string("op_11112_pad_type_0"), val = string("valid")]; tensor var_11112_strides_0 = const()[name = string("op_11112_strides_0"), val = tensor([1, 1])]; tensor var_11112_pad_0 = const()[name = string("op_11112_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11112_dilations_0 = const()[name = string("op_11112_dilations_0"), val = tensor([1, 1])]; int32 var_11112_groups_0 = const()[name = string("op_11112_groups_0"), val = int32(1)]; tensor var_11112 = conv(dilations = var_11112_dilations_0, groups = var_11112_groups_0, pad = var_11112_pad_0, pad_type = var_11112_pad_type_0, strides = var_11112_strides_0, weight = layers_19_self_attn_q_proj_weight_palettized, x = var_11096_cast_fp16)[name = string("op_11112")]; tensor var_11117 = const()[name = string("op_11117"), val = tensor([1, 8, 512, 1])]; tensor var_11118 = reshape(shape = var_11117, x = var_11112)[name = string("op_11118")]; tensor var_11123 = const()[name = string("op_11123"), val = tensor([0, 1, 3, 2])]; tensor var_11133 = const()[name = string("op_11133"), val = tensor([1, 8, 512])]; tensor var_11124 = transpose(perm = var_11123, x = var_11118)[name = string("transpose_112")]; tensor x_551 = reshape(shape = var_11133, x = var_11124)[name = string("x_551")]; int32 var_11139 = const()[name = string("op_11139"), val = int32(-1)]; fp16 const_319_promoted_to_fp16 = const()[name = string("const_319_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11145_cast_fp16 = mul(x = x_551, y = const_319_promoted_to_fp16)[name = string("op_11145_cast_fp16")]; bool input_539_interleave_0 = const()[name = string("input_539_interleave_0"), val = bool(false)]; tensor input_539_cast_fp16 = concat(axis = var_11139, interleave = input_539_interleave_0, values = (x_551, var_11145_cast_fp16))[name = string("input_539_cast_fp16")]; tensor normed_521_axes_0 = const()[name = string("normed_521_axes_0"), val = tensor([-1])]; fp16 var_11137_to_fp16 = const()[name = string("op_11137_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_521_cast_fp16 = layer_norm(axes = normed_521_axes_0, epsilon = var_11137_to_fp16, x = input_539_cast_fp16)[name = string("normed_521_cast_fp16")]; tensor var_11150_split_sizes_0 = const()[name = string("op_11150_split_sizes_0"), val = tensor([512, 512])]; int32 var_11150_axis_0 = const()[name = string("op_11150_axis_0"), val = int32(-1)]; tensor var_11150_cast_fp16_0, tensor var_11150_cast_fp16_1 = split(axis = var_11150_axis_0, split_sizes = var_11150_split_sizes_0, x = normed_521_cast_fp16)[name = string("op_11150_cast_fp16")]; tensor var_11153_cast_fp16 = mul(x = var_11150_cast_fp16_0, y = const_252_to_fp16)[name = string("op_11153_cast_fp16")]; tensor var_11159 = const()[name = string("op_11159"), val = tensor([1, 8, 1, 512])]; tensor q_147 = reshape(shape = var_11159, x = var_11153_cast_fp16)[name = string("q_147")]; tensor var_11161 = mul(x = q_147, y = cos)[name = string("op_11161")]; tensor var_11162_split_sizes_0 = const()[name = string("op_11162_split_sizes_0"), val = tensor([256, 256])]; int32 var_11162_axis_0 = const()[name = string("op_11162_axis_0"), val = int32(-1)]; tensor var_11162_0, tensor var_11162_1 = split(axis = var_11162_axis_0, split_sizes = var_11162_split_sizes_0, x = q_147)[name = string("op_11162")]; fp16 const_321_promoted = const()[name = string("const_321_promoted"), val = fp16(-0x1p+0)]; tensor var_11164 = mul(x = var_11162_1, y = const_321_promoted)[name = string("op_11164")]; int32 var_11166 = const()[name = string("op_11166"), val = int32(-1)]; bool var_11167_interleave_0 = const()[name = string("op_11167_interleave_0"), val = bool(false)]; tensor var_11167 = concat(axis = var_11166, interleave = var_11167_interleave_0, values = (var_11164, var_11162_0))[name = string("op_11167")]; tensor var_11168 = mul(x = var_11167, y = sin)[name = string("op_11168")]; tensor q_149 = add(x = var_11161, y = var_11168)[name = string("q_149")]; bool var_11182_transpose_x_0 = const()[name = string("op_11182_transpose_x_0"), val = bool(false)]; bool var_11182_transpose_y_0 = const()[name = string("op_11182_transpose_y_0"), val = bool(false)]; tensor var_11182_cast_fp16 = matmul(transpose_x = var_11182_transpose_x_0, transpose_y = var_11182_transpose_y_0, x = q_149, y = transpose_154_cast_fp16)[name = string("op_11182_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = var_11182_cast_fp16, y = causal_mask)[name = string("attn_weights_117_cast_fp16")]; int32 var_11187 = const()[name = string("op_11187"), val = int32(-1)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_11187, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool attn_output_115_transpose_x_0 = const()[name = string("attn_output_115_transpose_x_0"), val = bool(false)]; bool attn_output_115_transpose_y_0 = const()[name = string("attn_output_115_transpose_y_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = matmul(transpose_x = attn_output_115_transpose_x_0, transpose_y = attn_output_115_transpose_y_0, x = attn_weights_119_cast_fp16, y = V_expanded_29_cast_fp16)[name = string("attn_output_115_cast_fp16")]; tensor var_11195 = const()[name = string("op_11195"), val = tensor([0, 2, 1, 3])]; tensor var_11202 = const()[name = string("op_11202"), val = tensor([1, 1, -1])]; tensor var_11196_cast_fp16 = transpose(perm = var_11195, x = attn_output_115_cast_fp16)[name = string("transpose_111")]; tensor attn_output_117_cast_fp16 = reshape(shape = var_11202, x = var_11196_cast_fp16)[name = string("attn_output_117_cast_fp16")]; tensor var_11207 = const()[name = string("op_11207"), val = tensor([0, 2, 1])]; string var_11223_pad_type_0 = const()[name = string("op_11223_pad_type_0"), val = string("valid")]; int32 var_11223_groups_0 = const()[name = string("op_11223_groups_0"), val = int32(1)]; tensor var_11223_strides_0 = const()[name = string("op_11223_strides_0"), val = tensor([1])]; tensor var_11223_pad_0 = const()[name = string("op_11223_pad_0"), val = tensor([0, 0])]; tensor var_11223_dilations_0 = const()[name = string("op_11223_dilations_0"), val = tensor([1])]; tensor squeeze_19_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1100550592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103696384))))[name = string("squeeze_19_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11208_cast_fp16 = transpose(perm = var_11207, x = attn_output_117_cast_fp16)[name = string("transpose_110")]; tensor var_11223_cast_fp16 = conv(dilations = var_11223_dilations_0, groups = var_11223_groups_0, pad = var_11223_pad_0, pad_type = var_11223_pad_type_0, strides = var_11223_strides_0, weight = squeeze_19_cast_fp16_to_fp32_to_fp16_palettized, x = var_11208_cast_fp16)[name = string("op_11223_cast_fp16")]; tensor var_11227 = const()[name = string("op_11227"), val = tensor([0, 2, 1])]; int32 var_11233 = const()[name = string("op_11233"), val = int32(-1)]; fp16 const_322_promoted_to_fp16 = const()[name = string("const_322_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_555_cast_fp16 = transpose(perm = var_11227, x = var_11223_cast_fp16)[name = string("transpose_109")]; tensor var_11239_cast_fp16 = mul(x = x_555_cast_fp16, y = const_322_promoted_to_fp16)[name = string("op_11239_cast_fp16")]; bool input_543_interleave_0 = const()[name = string("input_543_interleave_0"), val = bool(false)]; tensor input_543_cast_fp16 = concat(axis = var_11233, interleave = input_543_interleave_0, values = (x_555_cast_fp16, var_11239_cast_fp16))[name = string("input_543_cast_fp16")]; tensor normed_525_axes_0 = const()[name = string("normed_525_axes_0"), val = tensor([-1])]; fp16 var_11231_to_fp16 = const()[name = string("op_11231_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_525_cast_fp16 = layer_norm(axes = normed_525_axes_0, epsilon = var_11231_to_fp16, x = input_543_cast_fp16)[name = string("normed_525_cast_fp16")]; tensor var_11244_split_sizes_0 = const()[name = string("op_11244_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11244_axis_0 = const()[name = string("op_11244_axis_0"), val = int32(-1)]; tensor var_11244_cast_fp16_0, tensor var_11244_cast_fp16_1 = split(axis = var_11244_axis_0, split_sizes = var_11244_split_sizes_0, x = normed_525_cast_fp16)[name = string("op_11244_cast_fp16")]; tensor const_323_to_fp16 = const()[name = string("const_323_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103697984)))]; tensor var_11247_cast_fp16 = mul(x = var_11244_cast_fp16_0, y = const_323_to_fp16)[name = string("op_11247_cast_fp16")]; tensor x_559_cast_fp16 = add(x = x_547_cast_fp16, y = var_11247_cast_fp16)[name = string("x_559_cast_fp16")]; int32 var_11254 = const()[name = string("op_11254"), val = int32(-1)]; fp16 const_324_promoted_to_fp16 = const()[name = string("const_324_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11260_cast_fp16 = mul(x = x_559_cast_fp16, y = const_324_promoted_to_fp16)[name = string("op_11260_cast_fp16")]; bool input_545_interleave_0 = const()[name = string("input_545_interleave_0"), val = bool(false)]; tensor input_545_cast_fp16 = concat(axis = var_11254, interleave = input_545_interleave_0, values = (x_559_cast_fp16, var_11260_cast_fp16))[name = string("input_545_cast_fp16")]; tensor normed_529_axes_0 = const()[name = string("normed_529_axes_0"), val = tensor([-1])]; fp16 var_11252_to_fp16 = const()[name = string("op_11252_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_529_cast_fp16 = layer_norm(axes = normed_529_axes_0, epsilon = var_11252_to_fp16, x = input_545_cast_fp16)[name = string("normed_529_cast_fp16")]; tensor var_11265_split_sizes_0 = const()[name = string("op_11265_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11265_axis_0 = const()[name = string("op_11265_axis_0"), val = int32(-1)]; tensor var_11265_cast_fp16_0, tensor var_11265_cast_fp16_1 = split(axis = var_11265_axis_0, split_sizes = var_11265_split_sizes_0, x = normed_529_cast_fp16)[name = string("op_11265_cast_fp16")]; tensor const_325_to_fp16 = const()[name = string("const_325_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103701120)))]; tensor var_11268_cast_fp16 = mul(x = var_11265_cast_fp16_0, y = const_325_to_fp16)[name = string("op_11268_cast_fp16")]; tensor var_11281 = const()[name = string("op_11281"), val = tensor([0, 2, 1])]; tensor input_547_axes_0 = const()[name = string("input_547_axes_0"), val = tensor([2])]; tensor var_11282 = transpose(perm = var_11281, x = var_11268_cast_fp16)[name = string("transpose_108")]; tensor input_547 = expand_dims(axes = input_547_axes_0, x = var_11282)[name = string("input_547")]; string gate_77_pad_type_0 = const()[name = string("gate_77_pad_type_0"), val = string("valid")]; tensor gate_77_strides_0 = const()[name = string("gate_77_strides_0"), val = tensor([1, 1])]; tensor gate_77_pad_0 = const()[name = string("gate_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_77_dilations_0 = const()[name = string("gate_77_dilations_0"), val = tensor([1, 1])]; int32 gate_77_groups_0 = const()[name = string("gate_77_groups_0"), val = int32(1)]; tensor gate_77 = conv(dilations = gate_77_dilations_0, groups = gate_77_groups_0, pad = gate_77_pad_0, pad_type = gate_77_pad_type_0, strides = gate_77_strides_0, weight = layers_19_mlp_gate_proj_weight_palettized, x = input_547)[name = string("gate_77")]; string up_39_pad_type_0 = const()[name = string("up_39_pad_type_0"), val = string("valid")]; tensor up_39_strides_0 = const()[name = string("up_39_strides_0"), val = tensor([1, 1])]; tensor up_39_pad_0 = const()[name = string("up_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_39_dilations_0 = const()[name = string("up_39_dilations_0"), val = tensor([1, 1])]; int32 up_39_groups_0 = const()[name = string("up_39_groups_0"), val = int32(1)]; tensor up_39 = conv(dilations = up_39_dilations_0, groups = up_39_groups_0, pad = up_39_pad_0, pad_type = up_39_pad_type_0, strides = up_39_strides_0, weight = layers_19_mlp_up_proj_weight_palettized, x = input_547)[name = string("up_39")]; string gate_79_mode_0 = const()[name = string("gate_79_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_79 = gelu(mode = gate_79_mode_0, x = gate_77)[name = string("gate_79")]; tensor input_549 = mul(x = gate_79, y = up_39)[name = string("input_549")]; string mlp_out_39_pad_type_0 = const()[name = string("mlp_out_39_pad_type_0"), val = string("valid")]; tensor mlp_out_39_strides_0 = const()[name = string("mlp_out_39_strides_0"), val = tensor([1, 1])]; tensor mlp_out_39_pad_0 = const()[name = string("mlp_out_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_39_dilations_0 = const()[name = string("mlp_out_39_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_39_groups_0 = const()[name = string("mlp_out_39_groups_0"), val = int32(1)]; tensor mlp_out_39 = conv(dilations = mlp_out_39_dilations_0, groups = mlp_out_39_groups_0, pad = mlp_out_39_pad_0, pad_type = mlp_out_39_pad_type_0, strides = mlp_out_39_strides_0, weight = layers_19_mlp_down_proj_weight_palettized, x = input_549)[name = string("mlp_out_39")]; tensor var_11322_axes_0 = const()[name = string("op_11322_axes_0"), val = tensor([2])]; tensor var_11322 = squeeze(axes = var_11322_axes_0, x = mlp_out_39)[name = string("op_11322")]; tensor var_11326 = const()[name = string("op_11326"), val = tensor([0, 2, 1])]; int32 var_11332 = const()[name = string("op_11332"), val = int32(-1)]; fp16 const_326_promoted_to_fp16 = const()[name = string("const_326_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_563 = transpose(perm = var_11326, x = var_11322)[name = string("transpose_107")]; tensor var_11338_cast_fp16 = mul(x = x_563, y = const_326_promoted_to_fp16)[name = string("op_11338_cast_fp16")]; bool input_551_interleave_0 = const()[name = string("input_551_interleave_0"), val = bool(false)]; tensor input_551_cast_fp16 = concat(axis = var_11332, interleave = input_551_interleave_0, values = (x_563, var_11338_cast_fp16))[name = string("input_551_cast_fp16")]; tensor normed_533_axes_0 = const()[name = string("normed_533_axes_0"), val = tensor([-1])]; fp16 var_11330_to_fp16 = const()[name = string("op_11330_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_533_cast_fp16 = layer_norm(axes = normed_533_axes_0, epsilon = var_11330_to_fp16, x = input_551_cast_fp16)[name = string("normed_533_cast_fp16")]; tensor var_11343_split_sizes_0 = const()[name = string("op_11343_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11343_axis_0 = const()[name = string("op_11343_axis_0"), val = int32(-1)]; tensor var_11343_cast_fp16_0, tensor var_11343_cast_fp16_1 = split(axis = var_11343_axis_0, split_sizes = var_11343_split_sizes_0, x = normed_533_cast_fp16)[name = string("op_11343_cast_fp16")]; tensor const_327_to_fp16 = const()[name = string("const_327_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103704256)))]; tensor var_11346_cast_fp16 = mul(x = var_11343_cast_fp16_0, y = const_327_to_fp16)[name = string("op_11346_cast_fp16")]; tensor hidden_states_237_cast_fp16 = add(x = x_559_cast_fp16, y = var_11346_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor per_layer_slice_39_begin_0 = const()[name = string("per_layer_slice_39_begin_0"), val = tensor([0, 0, 4864])]; tensor per_layer_slice_39_end_0 = const()[name = string("per_layer_slice_39_end_0"), val = tensor([1, 1, 5120])]; tensor per_layer_slice_39_end_mask_0 = const()[name = string("per_layer_slice_39_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_39_cast_fp16 = slice_by_index(begin = per_layer_slice_39_begin_0, end = per_layer_slice_39_end_0, end_mask = per_layer_slice_39_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_39_cast_fp16")]; tensor gated_77 = linear(bias = linear_0_bias_0, weight = layers_19_per_layer_input_gate_weight_palettized, x = hidden_states_237_cast_fp16)[name = string("linear_38")]; string gated_79_mode_0 = const()[name = string("gated_79_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_79 = gelu(mode = gated_79_mode_0, x = gated_77)[name = string("gated_79")]; tensor input_555_cast_fp16 = mul(x = gated_79, y = per_layer_slice_39_cast_fp16)[name = string("input_555_cast_fp16")]; tensor layers_19_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103707392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103904064))))[name = string("layers_19_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_39_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_19_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_555_cast_fp16)[name = string("linear_39_cast_fp16")]; int32 var_11383 = const()[name = string("op_11383"), val = int32(-1)]; fp16 const_328_promoted_to_fp16 = const()[name = string("const_328_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11389_cast_fp16 = mul(x = linear_39_cast_fp16, y = const_328_promoted_to_fp16)[name = string("op_11389_cast_fp16")]; bool input_557_interleave_0 = const()[name = string("input_557_interleave_0"), val = bool(false)]; tensor input_557_cast_fp16 = concat(axis = var_11383, interleave = input_557_interleave_0, values = (linear_39_cast_fp16, var_11389_cast_fp16))[name = string("input_557_cast_fp16")]; tensor normed_537_axes_0 = const()[name = string("normed_537_axes_0"), val = tensor([-1])]; fp16 var_11381_to_fp16 = const()[name = string("op_11381_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_537_cast_fp16 = layer_norm(axes = normed_537_axes_0, epsilon = var_11381_to_fp16, x = input_557_cast_fp16)[name = string("normed_537_cast_fp16")]; tensor var_11394_split_sizes_0 = const()[name = string("op_11394_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11394_axis_0 = const()[name = string("op_11394_axis_0"), val = int32(-1)]; tensor var_11394_cast_fp16_0, tensor var_11394_cast_fp16_1 = split(axis = var_11394_axis_0, split_sizes = var_11394_split_sizes_0, x = normed_537_cast_fp16)[name = string("op_11394_cast_fp16")]; tensor const_329_to_fp16 = const()[name = string("const_329_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103905664)))]; tensor var_11397_cast_fp16 = mul(x = var_11394_cast_fp16_0, y = const_329_to_fp16)[name = string("op_11397_cast_fp16")]; tensor hidden_states_241_cast_fp16 = add(x = hidden_states_237_cast_fp16, y = var_11397_cast_fp16)[name = string("hidden_states_241_cast_fp16")]; tensor layers_19_layer_scalar_to_fp16 = const()[name = string("layers_19_layer_scalar_to_fp16"), val = tensor([0x1.14p-1])]; tensor x_571_cast_fp16 = mul(x = hidden_states_241_cast_fp16, y = layers_19_layer_scalar_to_fp16)[name = string("x_571_cast_fp16")]; int32 var_11405 = const()[name = string("op_11405"), val = int32(-1)]; fp16 const_330_promoted_to_fp16 = const()[name = string("const_330_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11411_cast_fp16 = mul(x = x_571_cast_fp16, y = const_330_promoted_to_fp16)[name = string("op_11411_cast_fp16")]; bool input_559_interleave_0 = const()[name = string("input_559_interleave_0"), val = bool(false)]; tensor input_559_cast_fp16 = concat(axis = var_11405, interleave = input_559_interleave_0, values = (x_571_cast_fp16, var_11411_cast_fp16))[name = string("input_559_cast_fp16")]; tensor normed_541_axes_0 = const()[name = string("normed_541_axes_0"), val = tensor([-1])]; fp16 var_11403_to_fp16 = const()[name = string("op_11403_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_541_cast_fp16 = layer_norm(axes = normed_541_axes_0, epsilon = var_11403_to_fp16, x = input_559_cast_fp16)[name = string("normed_541_cast_fp16")]; tensor var_11416_split_sizes_0 = const()[name = string("op_11416_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11416_axis_0 = const()[name = string("op_11416_axis_0"), val = int32(-1)]; tensor var_11416_cast_fp16_0, tensor var_11416_cast_fp16_1 = split(axis = var_11416_axis_0, split_sizes = var_11416_split_sizes_0, x = normed_541_cast_fp16)[name = string("op_11416_cast_fp16")]; tensor const_331_to_fp16 = const()[name = string("const_331_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103908800)))]; tensor var_11419_cast_fp16 = mul(x = var_11416_cast_fp16_0, y = const_331_to_fp16)[name = string("op_11419_cast_fp16")]; tensor var_11427 = const()[name = string("op_11427"), val = tensor([0, 2, 1])]; tensor var_11430_axes_0 = const()[name = string("op_11430_axes_0"), val = tensor([2])]; tensor var_11428_cast_fp16 = transpose(perm = var_11427, x = var_11419_cast_fp16)[name = string("transpose_106")]; tensor var_11430_cast_fp16 = expand_dims(axes = var_11430_axes_0, x = var_11428_cast_fp16)[name = string("op_11430_cast_fp16")]; string var_11446_pad_type_0 = const()[name = string("op_11446_pad_type_0"), val = string("valid")]; tensor var_11446_strides_0 = const()[name = string("op_11446_strides_0"), val = tensor([1, 1])]; tensor var_11446_pad_0 = const()[name = string("op_11446_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11446_dilations_0 = const()[name = string("op_11446_dilations_0"), val = tensor([1, 1])]; int32 var_11446_groups_0 = const()[name = string("op_11446_groups_0"), val = int32(1)]; tensor var_11446 = conv(dilations = var_11446_dilations_0, groups = var_11446_groups_0, pad = var_11446_pad_0, pad_type = var_11446_pad_type_0, strides = var_11446_strides_0, weight = layers_20_self_attn_q_proj_weight_palettized, x = var_11430_cast_fp16)[name = string("op_11446")]; tensor var_11451 = const()[name = string("op_11451"), val = tensor([1, 8, 256, 1])]; tensor var_11452 = reshape(shape = var_11451, x = var_11446)[name = string("op_11452")]; tensor var_11457 = const()[name = string("op_11457"), val = tensor([0, 1, 3, 2])]; tensor var_11467 = const()[name = string("op_11467"), val = tensor([1, 8, 256])]; tensor var_11458 = transpose(perm = var_11457, x = var_11452)[name = string("transpose_105")]; tensor x_575 = reshape(shape = var_11467, x = var_11458)[name = string("x_575")]; int32 var_11473 = const()[name = string("op_11473"), val = int32(-1)]; fp16 const_332_promoted_to_fp16 = const()[name = string("const_332_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11479_cast_fp16 = mul(x = x_575, y = const_332_promoted_to_fp16)[name = string("op_11479_cast_fp16")]; bool input_563_interleave_0 = const()[name = string("input_563_interleave_0"), val = bool(false)]; tensor input_563_cast_fp16 = concat(axis = var_11473, interleave = input_563_interleave_0, values = (x_575, var_11479_cast_fp16))[name = string("input_563_cast_fp16")]; tensor normed_545_axes_0 = const()[name = string("normed_545_axes_0"), val = tensor([-1])]; fp16 var_11471_to_fp16 = const()[name = string("op_11471_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_545_cast_fp16 = layer_norm(axes = normed_545_axes_0, epsilon = var_11471_to_fp16, x = input_563_cast_fp16)[name = string("normed_545_cast_fp16")]; tensor var_11484_split_sizes_0 = const()[name = string("op_11484_split_sizes_0"), val = tensor([256, 256])]; int32 var_11484_axis_0 = const()[name = string("op_11484_axis_0"), val = int32(-1)]; tensor var_11484_cast_fp16_0, tensor var_11484_cast_fp16_1 = split(axis = var_11484_axis_0, split_sizes = var_11484_split_sizes_0, x = normed_545_cast_fp16)[name = string("op_11484_cast_fp16")]; tensor var_11487_cast_fp16 = mul(x = var_11484_cast_fp16_0, y = const_234_to_fp16)[name = string("op_11487_cast_fp16")]; tensor var_11493 = const()[name = string("op_11493"), val = tensor([1, 8, 1, 256])]; tensor q_153 = reshape(shape = var_11493, x = var_11487_cast_fp16)[name = string("q_153")]; tensor var_11495 = mul(x = q_153, y = cos_1)[name = string("op_11495")]; tensor var_11496_split_sizes_0 = const()[name = string("op_11496_split_sizes_0"), val = tensor([128, 128])]; int32 var_11496_axis_0 = const()[name = string("op_11496_axis_0"), val = int32(-1)]; tensor var_11496_0, tensor var_11496_1 = split(axis = var_11496_axis_0, split_sizes = var_11496_split_sizes_0, x = q_153)[name = string("op_11496")]; fp16 const_334_promoted = const()[name = string("const_334_promoted"), val = fp16(-0x1p+0)]; tensor var_11498 = mul(x = var_11496_1, y = const_334_promoted)[name = string("op_11498")]; int32 var_11500 = const()[name = string("op_11500"), val = int32(-1)]; bool var_11501_interleave_0 = const()[name = string("op_11501_interleave_0"), val = bool(false)]; tensor var_11501 = concat(axis = var_11500, interleave = var_11501_interleave_0, values = (var_11498, var_11496_0))[name = string("op_11501")]; tensor var_11502 = mul(x = var_11501, y = sin_1)[name = string("op_11502")]; tensor q_155 = add(x = var_11495, y = var_11502)[name = string("q_155")]; bool var_11516_transpose_x_0 = const()[name = string("op_11516_transpose_x_0"), val = bool(false)]; bool var_11516_transpose_y_0 = const()[name = string("op_11516_transpose_y_0"), val = bool(false)]; tensor var_11516_cast_fp16 = matmul(transpose_x = var_11516_transpose_x_0, transpose_y = var_11516_transpose_y_0, x = q_155, y = transpose_153_cast_fp16)[name = string("op_11516_cast_fp16")]; tensor attn_weights_123_cast_fp16 = add(x = var_11516_cast_fp16, y = causal_mask)[name = string("attn_weights_123_cast_fp16")]; int32 var_11521 = const()[name = string("op_11521"), val = int32(-1)]; tensor attn_weights_125_cast_fp16 = softmax(axis = var_11521, x = attn_weights_123_cast_fp16)[name = string("attn_weights_125_cast_fp16")]; bool attn_output_121_transpose_x_0 = const()[name = string("attn_output_121_transpose_x_0"), val = bool(false)]; bool attn_output_121_transpose_y_0 = const()[name = string("attn_output_121_transpose_y_0"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_0, transpose_y = attn_output_121_transpose_y_0, x = attn_weights_125_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_121_cast_fp16")]; tensor var_11529 = const()[name = string("op_11529"), val = tensor([0, 2, 1, 3])]; tensor var_11536 = const()[name = string("op_11536"), val = tensor([1, 1, -1])]; tensor var_11530_cast_fp16 = transpose(perm = var_11529, x = attn_output_121_cast_fp16)[name = string("transpose_104")]; tensor attn_output_123_cast_fp16 = reshape(shape = var_11536, x = var_11530_cast_fp16)[name = string("attn_output_123_cast_fp16")]; tensor var_11541 = const()[name = string("op_11541"), val = tensor([0, 2, 1])]; string var_11557_pad_type_0 = const()[name = string("op_11557_pad_type_0"), val = string("valid")]; int32 var_11557_groups_0 = const()[name = string("op_11557_groups_0"), val = int32(1)]; tensor var_11557_strides_0 = const()[name = string("op_11557_strides_0"), val = tensor([1])]; tensor var_11557_pad_0 = const()[name = string("op_11557_pad_0"), val = tensor([0, 0])]; tensor var_11557_dilations_0 = const()[name = string("op_11557_dilations_0"), val = tensor([1])]; tensor squeeze_20_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1103911936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105484864))))[name = string("squeeze_20_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11542_cast_fp16 = transpose(perm = var_11541, x = attn_output_123_cast_fp16)[name = string("transpose_103")]; tensor var_11557_cast_fp16 = conv(dilations = var_11557_dilations_0, groups = var_11557_groups_0, pad = var_11557_pad_0, pad_type = var_11557_pad_type_0, strides = var_11557_strides_0, weight = squeeze_20_cast_fp16_to_fp32_to_fp16_palettized, x = var_11542_cast_fp16)[name = string("op_11557_cast_fp16")]; tensor var_11561 = const()[name = string("op_11561"), val = tensor([0, 2, 1])]; int32 var_11567 = const()[name = string("op_11567"), val = int32(-1)]; fp16 const_335_promoted_to_fp16 = const()[name = string("const_335_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_579_cast_fp16 = transpose(perm = var_11561, x = var_11557_cast_fp16)[name = string("transpose_102")]; tensor var_11573_cast_fp16 = mul(x = x_579_cast_fp16, y = const_335_promoted_to_fp16)[name = string("op_11573_cast_fp16")]; bool input_567_interleave_0 = const()[name = string("input_567_interleave_0"), val = bool(false)]; tensor input_567_cast_fp16 = concat(axis = var_11567, interleave = input_567_interleave_0, values = (x_579_cast_fp16, var_11573_cast_fp16))[name = string("input_567_cast_fp16")]; tensor normed_549_axes_0 = const()[name = string("normed_549_axes_0"), val = tensor([-1])]; fp16 var_11565_to_fp16 = const()[name = string("op_11565_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_549_cast_fp16 = layer_norm(axes = normed_549_axes_0, epsilon = var_11565_to_fp16, x = input_567_cast_fp16)[name = string("normed_549_cast_fp16")]; tensor var_11578_split_sizes_0 = const()[name = string("op_11578_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11578_axis_0 = const()[name = string("op_11578_axis_0"), val = int32(-1)]; tensor var_11578_cast_fp16_0, tensor var_11578_cast_fp16_1 = split(axis = var_11578_axis_0, split_sizes = var_11578_split_sizes_0, x = normed_549_cast_fp16)[name = string("op_11578_cast_fp16")]; tensor const_336_to_fp16 = const()[name = string("const_336_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105486464)))]; tensor var_11581_cast_fp16 = mul(x = var_11578_cast_fp16_0, y = const_336_to_fp16)[name = string("op_11581_cast_fp16")]; tensor x_583_cast_fp16 = add(x = x_571_cast_fp16, y = var_11581_cast_fp16)[name = string("x_583_cast_fp16")]; int32 var_11588 = const()[name = string("op_11588"), val = int32(-1)]; fp16 const_337_promoted_to_fp16 = const()[name = string("const_337_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11594_cast_fp16 = mul(x = x_583_cast_fp16, y = const_337_promoted_to_fp16)[name = string("op_11594_cast_fp16")]; bool input_569_interleave_0 = const()[name = string("input_569_interleave_0"), val = bool(false)]; tensor input_569_cast_fp16 = concat(axis = var_11588, interleave = input_569_interleave_0, values = (x_583_cast_fp16, var_11594_cast_fp16))[name = string("input_569_cast_fp16")]; tensor normed_553_axes_0 = const()[name = string("normed_553_axes_0"), val = tensor([-1])]; fp16 var_11586_to_fp16 = const()[name = string("op_11586_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_553_cast_fp16 = layer_norm(axes = normed_553_axes_0, epsilon = var_11586_to_fp16, x = input_569_cast_fp16)[name = string("normed_553_cast_fp16")]; tensor var_11599_split_sizes_0 = const()[name = string("op_11599_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11599_axis_0 = const()[name = string("op_11599_axis_0"), val = int32(-1)]; tensor var_11599_cast_fp16_0, tensor var_11599_cast_fp16_1 = split(axis = var_11599_axis_0, split_sizes = var_11599_split_sizes_0, x = normed_553_cast_fp16)[name = string("op_11599_cast_fp16")]; tensor const_338_to_fp16 = const()[name = string("const_338_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105489600)))]; tensor var_11602_cast_fp16 = mul(x = var_11599_cast_fp16_0, y = const_338_to_fp16)[name = string("op_11602_cast_fp16")]; tensor var_11615 = const()[name = string("op_11615"), val = tensor([0, 2, 1])]; tensor input_571_axes_0 = const()[name = string("input_571_axes_0"), val = tensor([2])]; tensor var_11616 = transpose(perm = var_11615, x = var_11602_cast_fp16)[name = string("transpose_101")]; tensor input_571 = expand_dims(axes = input_571_axes_0, x = var_11616)[name = string("input_571")]; string gate_81_pad_type_0 = const()[name = string("gate_81_pad_type_0"), val = string("valid")]; tensor gate_81_strides_0 = const()[name = string("gate_81_strides_0"), val = tensor([1, 1])]; tensor gate_81_pad_0 = const()[name = string("gate_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_81_dilations_0 = const()[name = string("gate_81_dilations_0"), val = tensor([1, 1])]; int32 gate_81_groups_0 = const()[name = string("gate_81_groups_0"), val = int32(1)]; tensor gate_81 = conv(dilations = gate_81_dilations_0, groups = gate_81_groups_0, pad = gate_81_pad_0, pad_type = gate_81_pad_type_0, strides = gate_81_strides_0, weight = layers_20_mlp_gate_proj_weight_palettized, x = input_571)[name = string("gate_81")]; string up_41_pad_type_0 = const()[name = string("up_41_pad_type_0"), val = string("valid")]; tensor up_41_strides_0 = const()[name = string("up_41_strides_0"), val = tensor([1, 1])]; tensor up_41_pad_0 = const()[name = string("up_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_41_dilations_0 = const()[name = string("up_41_dilations_0"), val = tensor([1, 1])]; int32 up_41_groups_0 = const()[name = string("up_41_groups_0"), val = int32(1)]; tensor up_41 = conv(dilations = up_41_dilations_0, groups = up_41_groups_0, pad = up_41_pad_0, pad_type = up_41_pad_type_0, strides = up_41_strides_0, weight = layers_20_mlp_up_proj_weight_palettized, x = input_571)[name = string("up_41")]; string gate_83_mode_0 = const()[name = string("gate_83_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_83 = gelu(mode = gate_83_mode_0, x = gate_81)[name = string("gate_83")]; tensor input_573 = mul(x = gate_83, y = up_41)[name = string("input_573")]; string mlp_out_41_pad_type_0 = const()[name = string("mlp_out_41_pad_type_0"), val = string("valid")]; tensor mlp_out_41_strides_0 = const()[name = string("mlp_out_41_strides_0"), val = tensor([1, 1])]; tensor mlp_out_41_pad_0 = const()[name = string("mlp_out_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_41_dilations_0 = const()[name = string("mlp_out_41_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_41_groups_0 = const()[name = string("mlp_out_41_groups_0"), val = int32(1)]; tensor mlp_out_41 = conv(dilations = mlp_out_41_dilations_0, groups = mlp_out_41_groups_0, pad = mlp_out_41_pad_0, pad_type = mlp_out_41_pad_type_0, strides = mlp_out_41_strides_0, weight = layers_20_mlp_down_proj_weight_palettized, x = input_573)[name = string("mlp_out_41")]; tensor var_11656_axes_0 = const()[name = string("op_11656_axes_0"), val = tensor([2])]; tensor var_11656 = squeeze(axes = var_11656_axes_0, x = mlp_out_41)[name = string("op_11656")]; tensor var_11660 = const()[name = string("op_11660"), val = tensor([0, 2, 1])]; int32 var_11666 = const()[name = string("op_11666"), val = int32(-1)]; fp16 const_339_promoted_to_fp16 = const()[name = string("const_339_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_587 = transpose(perm = var_11660, x = var_11656)[name = string("transpose_100")]; tensor var_11672_cast_fp16 = mul(x = x_587, y = const_339_promoted_to_fp16)[name = string("op_11672_cast_fp16")]; bool input_575_interleave_0 = const()[name = string("input_575_interleave_0"), val = bool(false)]; tensor input_575_cast_fp16 = concat(axis = var_11666, interleave = input_575_interleave_0, values = (x_587, var_11672_cast_fp16))[name = string("input_575_cast_fp16")]; tensor normed_557_axes_0 = const()[name = string("normed_557_axes_0"), val = tensor([-1])]; fp16 var_11664_to_fp16 = const()[name = string("op_11664_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_557_cast_fp16 = layer_norm(axes = normed_557_axes_0, epsilon = var_11664_to_fp16, x = input_575_cast_fp16)[name = string("normed_557_cast_fp16")]; tensor var_11677_split_sizes_0 = const()[name = string("op_11677_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11677_axis_0 = const()[name = string("op_11677_axis_0"), val = int32(-1)]; tensor var_11677_cast_fp16_0, tensor var_11677_cast_fp16_1 = split(axis = var_11677_axis_0, split_sizes = var_11677_split_sizes_0, x = normed_557_cast_fp16)[name = string("op_11677_cast_fp16")]; tensor const_340_to_fp16 = const()[name = string("const_340_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105492736)))]; tensor var_11680_cast_fp16 = mul(x = var_11677_cast_fp16_0, y = const_340_to_fp16)[name = string("op_11680_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = x_583_cast_fp16, y = var_11680_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; tensor per_layer_slice_41_begin_0 = const()[name = string("per_layer_slice_41_begin_0"), val = tensor([0, 0, 5120])]; tensor per_layer_slice_41_end_0 = const()[name = string("per_layer_slice_41_end_0"), val = tensor([1, 1, 5376])]; tensor per_layer_slice_41_end_mask_0 = const()[name = string("per_layer_slice_41_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_41_cast_fp16 = slice_by_index(begin = per_layer_slice_41_begin_0, end = per_layer_slice_41_end_0, end_mask = per_layer_slice_41_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_41_cast_fp16")]; tensor gated_81 = linear(bias = linear_0_bias_0, weight = layers_20_per_layer_input_gate_weight_palettized, x = hidden_states_249_cast_fp16)[name = string("linear_40")]; string gated_83_mode_0 = const()[name = string("gated_83_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_83 = gelu(mode = gated_83_mode_0, x = gated_81)[name = string("gated_83")]; tensor input_579_cast_fp16 = mul(x = gated_83, y = per_layer_slice_41_cast_fp16)[name = string("input_579_cast_fp16")]; tensor layers_20_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105495872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105692544))))[name = string("layers_20_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_41_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_20_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_579_cast_fp16)[name = string("linear_41_cast_fp16")]; int32 var_11717 = const()[name = string("op_11717"), val = int32(-1)]; fp16 const_341_promoted_to_fp16 = const()[name = string("const_341_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11723_cast_fp16 = mul(x = linear_41_cast_fp16, y = const_341_promoted_to_fp16)[name = string("op_11723_cast_fp16")]; bool input_581_interleave_0 = const()[name = string("input_581_interleave_0"), val = bool(false)]; tensor input_581_cast_fp16 = concat(axis = var_11717, interleave = input_581_interleave_0, values = (linear_41_cast_fp16, var_11723_cast_fp16))[name = string("input_581_cast_fp16")]; tensor normed_561_axes_0 = const()[name = string("normed_561_axes_0"), val = tensor([-1])]; fp16 var_11715_to_fp16 = const()[name = string("op_11715_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_561_cast_fp16 = layer_norm(axes = normed_561_axes_0, epsilon = var_11715_to_fp16, x = input_581_cast_fp16)[name = string("normed_561_cast_fp16")]; tensor var_11728_split_sizes_0 = const()[name = string("op_11728_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11728_axis_0 = const()[name = string("op_11728_axis_0"), val = int32(-1)]; tensor var_11728_cast_fp16_0, tensor var_11728_cast_fp16_1 = split(axis = var_11728_axis_0, split_sizes = var_11728_split_sizes_0, x = normed_561_cast_fp16)[name = string("op_11728_cast_fp16")]; tensor const_342_to_fp16 = const()[name = string("const_342_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105694144)))]; tensor var_11731_cast_fp16 = mul(x = var_11728_cast_fp16_0, y = const_342_to_fp16)[name = string("op_11731_cast_fp16")]; tensor hidden_states_253_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = var_11731_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor layers_20_layer_scalar_to_fp16 = const()[name = string("layers_20_layer_scalar_to_fp16"), val = tensor([0x1.fap-2])]; tensor x_595_cast_fp16 = mul(x = hidden_states_253_cast_fp16, y = layers_20_layer_scalar_to_fp16)[name = string("x_595_cast_fp16")]; int32 var_11739 = const()[name = string("op_11739"), val = int32(-1)]; fp16 const_343_promoted_to_fp16 = const()[name = string("const_343_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11745_cast_fp16 = mul(x = x_595_cast_fp16, y = const_343_promoted_to_fp16)[name = string("op_11745_cast_fp16")]; bool input_583_interleave_0 = const()[name = string("input_583_interleave_0"), val = bool(false)]; tensor input_583_cast_fp16 = concat(axis = var_11739, interleave = input_583_interleave_0, values = (x_595_cast_fp16, var_11745_cast_fp16))[name = string("input_583_cast_fp16")]; tensor normed_565_axes_0 = const()[name = string("normed_565_axes_0"), val = tensor([-1])]; fp16 var_11737_to_fp16 = const()[name = string("op_11737_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_565_cast_fp16 = layer_norm(axes = normed_565_axes_0, epsilon = var_11737_to_fp16, x = input_583_cast_fp16)[name = string("normed_565_cast_fp16")]; tensor var_11750_split_sizes_0 = const()[name = string("op_11750_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11750_axis_0 = const()[name = string("op_11750_axis_0"), val = int32(-1)]; tensor var_11750_cast_fp16_0, tensor var_11750_cast_fp16_1 = split(axis = var_11750_axis_0, split_sizes = var_11750_split_sizes_0, x = normed_565_cast_fp16)[name = string("op_11750_cast_fp16")]; tensor const_344_to_fp16 = const()[name = string("const_344_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105697280)))]; tensor var_11753_cast_fp16 = mul(x = var_11750_cast_fp16_0, y = const_344_to_fp16)[name = string("op_11753_cast_fp16")]; tensor var_11761 = const()[name = string("op_11761"), val = tensor([0, 2, 1])]; tensor var_11764_axes_0 = const()[name = string("op_11764_axes_0"), val = tensor([2])]; tensor var_11762_cast_fp16 = transpose(perm = var_11761, x = var_11753_cast_fp16)[name = string("transpose_99")]; tensor var_11764_cast_fp16 = expand_dims(axes = var_11764_axes_0, x = var_11762_cast_fp16)[name = string("op_11764_cast_fp16")]; string var_11780_pad_type_0 = const()[name = string("op_11780_pad_type_0"), val = string("valid")]; tensor var_11780_strides_0 = const()[name = string("op_11780_strides_0"), val = tensor([1, 1])]; tensor var_11780_pad_0 = const()[name = string("op_11780_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11780_dilations_0 = const()[name = string("op_11780_dilations_0"), val = tensor([1, 1])]; int32 var_11780_groups_0 = const()[name = string("op_11780_groups_0"), val = int32(1)]; tensor var_11780 = conv(dilations = var_11780_dilations_0, groups = var_11780_groups_0, pad = var_11780_pad_0, pad_type = var_11780_pad_type_0, strides = var_11780_strides_0, weight = layers_21_self_attn_q_proj_weight_palettized, x = var_11764_cast_fp16)[name = string("op_11780")]; tensor var_11785 = const()[name = string("op_11785"), val = tensor([1, 8, 256, 1])]; tensor var_11786 = reshape(shape = var_11785, x = var_11780)[name = string("op_11786")]; tensor var_11791 = const()[name = string("op_11791"), val = tensor([0, 1, 3, 2])]; tensor var_11801 = const()[name = string("op_11801"), val = tensor([1, 8, 256])]; tensor var_11792 = transpose(perm = var_11791, x = var_11786)[name = string("transpose_98")]; tensor x_599 = reshape(shape = var_11801, x = var_11792)[name = string("x_599")]; int32 var_11807 = const()[name = string("op_11807"), val = int32(-1)]; fp16 const_345_promoted_to_fp16 = const()[name = string("const_345_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11813_cast_fp16 = mul(x = x_599, y = const_345_promoted_to_fp16)[name = string("op_11813_cast_fp16")]; bool input_587_interleave_0 = const()[name = string("input_587_interleave_0"), val = bool(false)]; tensor input_587_cast_fp16 = concat(axis = var_11807, interleave = input_587_interleave_0, values = (x_599, var_11813_cast_fp16))[name = string("input_587_cast_fp16")]; tensor normed_569_axes_0 = const()[name = string("normed_569_axes_0"), val = tensor([-1])]; fp16 var_11805_to_fp16 = const()[name = string("op_11805_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_569_cast_fp16 = layer_norm(axes = normed_569_axes_0, epsilon = var_11805_to_fp16, x = input_587_cast_fp16)[name = string("normed_569_cast_fp16")]; tensor var_11818_split_sizes_0 = const()[name = string("op_11818_split_sizes_0"), val = tensor([256, 256])]; int32 var_11818_axis_0 = const()[name = string("op_11818_axis_0"), val = int32(-1)]; tensor var_11818_cast_fp16_0, tensor var_11818_cast_fp16_1 = split(axis = var_11818_axis_0, split_sizes = var_11818_split_sizes_0, x = normed_569_cast_fp16)[name = string("op_11818_cast_fp16")]; tensor var_11821_cast_fp16 = mul(x = var_11818_cast_fp16_0, y = const_234_to_fp16)[name = string("op_11821_cast_fp16")]; tensor var_11827 = const()[name = string("op_11827"), val = tensor([1, 8, 1, 256])]; tensor q_159 = reshape(shape = var_11827, x = var_11821_cast_fp16)[name = string("q_159")]; tensor var_11829 = mul(x = q_159, y = cos_1)[name = string("op_11829")]; tensor var_11830_split_sizes_0 = const()[name = string("op_11830_split_sizes_0"), val = tensor([128, 128])]; int32 var_11830_axis_0 = const()[name = string("op_11830_axis_0"), val = int32(-1)]; tensor var_11830_0, tensor var_11830_1 = split(axis = var_11830_axis_0, split_sizes = var_11830_split_sizes_0, x = q_159)[name = string("op_11830")]; fp16 const_347_promoted = const()[name = string("const_347_promoted"), val = fp16(-0x1p+0)]; tensor var_11832 = mul(x = var_11830_1, y = const_347_promoted)[name = string("op_11832")]; int32 var_11834 = const()[name = string("op_11834"), val = int32(-1)]; bool var_11835_interleave_0 = const()[name = string("op_11835_interleave_0"), val = bool(false)]; tensor var_11835 = concat(axis = var_11834, interleave = var_11835_interleave_0, values = (var_11832, var_11830_0))[name = string("op_11835")]; tensor var_11836 = mul(x = var_11835, y = sin_1)[name = string("op_11836")]; tensor q_161 = add(x = var_11829, y = var_11836)[name = string("q_161")]; bool var_11850_transpose_x_0 = const()[name = string("op_11850_transpose_x_0"), val = bool(false)]; bool var_11850_transpose_y_0 = const()[name = string("op_11850_transpose_y_0"), val = bool(false)]; tensor var_11850_cast_fp16 = matmul(transpose_x = var_11850_transpose_x_0, transpose_y = var_11850_transpose_y_0, x = q_161, y = transpose_153_cast_fp16)[name = string("op_11850_cast_fp16")]; tensor attn_weights_129_cast_fp16 = add(x = var_11850_cast_fp16, y = causal_mask)[name = string("attn_weights_129_cast_fp16")]; int32 var_11855 = const()[name = string("op_11855"), val = int32(-1)]; tensor attn_weights_131_cast_fp16 = softmax(axis = var_11855, x = attn_weights_129_cast_fp16)[name = string("attn_weights_131_cast_fp16")]; bool attn_output_127_transpose_x_0 = const()[name = string("attn_output_127_transpose_x_0"), val = bool(false)]; bool attn_output_127_transpose_y_0 = const()[name = string("attn_output_127_transpose_y_0"), val = bool(false)]; tensor attn_output_127_cast_fp16 = matmul(transpose_x = attn_output_127_transpose_x_0, transpose_y = attn_output_127_transpose_y_0, x = attn_weights_131_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor var_11863 = const()[name = string("op_11863"), val = tensor([0, 2, 1, 3])]; tensor var_11870 = const()[name = string("op_11870"), val = tensor([1, 1, -1])]; tensor var_11864_cast_fp16 = transpose(perm = var_11863, x = attn_output_127_cast_fp16)[name = string("transpose_97")]; tensor attn_output_129_cast_fp16 = reshape(shape = var_11870, x = var_11864_cast_fp16)[name = string("attn_output_129_cast_fp16")]; tensor var_11875 = const()[name = string("op_11875"), val = tensor([0, 2, 1])]; string var_11891_pad_type_0 = const()[name = string("op_11891_pad_type_0"), val = string("valid")]; int32 var_11891_groups_0 = const()[name = string("op_11891_groups_0"), val = int32(1)]; tensor var_11891_strides_0 = const()[name = string("op_11891_strides_0"), val = tensor([1])]; tensor var_11891_pad_0 = const()[name = string("op_11891_pad_0"), val = tensor([0, 0])]; tensor var_11891_dilations_0 = const()[name = string("op_11891_dilations_0"), val = tensor([1])]; tensor squeeze_21_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1105700416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107273344))))[name = string("squeeze_21_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11876_cast_fp16 = transpose(perm = var_11875, x = attn_output_129_cast_fp16)[name = string("transpose_96")]; tensor var_11891_cast_fp16 = conv(dilations = var_11891_dilations_0, groups = var_11891_groups_0, pad = var_11891_pad_0, pad_type = var_11891_pad_type_0, strides = var_11891_strides_0, weight = squeeze_21_cast_fp16_to_fp32_to_fp16_palettized, x = var_11876_cast_fp16)[name = string("op_11891_cast_fp16")]; tensor var_11895 = const()[name = string("op_11895"), val = tensor([0, 2, 1])]; int32 var_11901 = const()[name = string("op_11901"), val = int32(-1)]; fp16 const_348_promoted_to_fp16 = const()[name = string("const_348_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_603_cast_fp16 = transpose(perm = var_11895, x = var_11891_cast_fp16)[name = string("transpose_95")]; tensor var_11907_cast_fp16 = mul(x = x_603_cast_fp16, y = const_348_promoted_to_fp16)[name = string("op_11907_cast_fp16")]; bool input_591_interleave_0 = const()[name = string("input_591_interleave_0"), val = bool(false)]; tensor input_591_cast_fp16 = concat(axis = var_11901, interleave = input_591_interleave_0, values = (x_603_cast_fp16, var_11907_cast_fp16))[name = string("input_591_cast_fp16")]; tensor normed_573_axes_0 = const()[name = string("normed_573_axes_0"), val = tensor([-1])]; fp16 var_11899_to_fp16 = const()[name = string("op_11899_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_573_cast_fp16 = layer_norm(axes = normed_573_axes_0, epsilon = var_11899_to_fp16, x = input_591_cast_fp16)[name = string("normed_573_cast_fp16")]; tensor var_11912_split_sizes_0 = const()[name = string("op_11912_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11912_axis_0 = const()[name = string("op_11912_axis_0"), val = int32(-1)]; tensor var_11912_cast_fp16_0, tensor var_11912_cast_fp16_1 = split(axis = var_11912_axis_0, split_sizes = var_11912_split_sizes_0, x = normed_573_cast_fp16)[name = string("op_11912_cast_fp16")]; tensor const_349_to_fp16 = const()[name = string("const_349_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107274944)))]; tensor var_11915_cast_fp16 = mul(x = var_11912_cast_fp16_0, y = const_349_to_fp16)[name = string("op_11915_cast_fp16")]; tensor x_607_cast_fp16 = add(x = x_595_cast_fp16, y = var_11915_cast_fp16)[name = string("x_607_cast_fp16")]; int32 var_11922 = const()[name = string("op_11922"), val = int32(-1)]; fp16 const_350_promoted_to_fp16 = const()[name = string("const_350_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11928_cast_fp16 = mul(x = x_607_cast_fp16, y = const_350_promoted_to_fp16)[name = string("op_11928_cast_fp16")]; bool input_593_interleave_0 = const()[name = string("input_593_interleave_0"), val = bool(false)]; tensor input_593_cast_fp16 = concat(axis = var_11922, interleave = input_593_interleave_0, values = (x_607_cast_fp16, var_11928_cast_fp16))[name = string("input_593_cast_fp16")]; tensor normed_577_axes_0 = const()[name = string("normed_577_axes_0"), val = tensor([-1])]; fp16 var_11920_to_fp16 = const()[name = string("op_11920_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_577_cast_fp16 = layer_norm(axes = normed_577_axes_0, epsilon = var_11920_to_fp16, x = input_593_cast_fp16)[name = string("normed_577_cast_fp16")]; tensor var_11933_split_sizes_0 = const()[name = string("op_11933_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_11933_axis_0 = const()[name = string("op_11933_axis_0"), val = int32(-1)]; tensor var_11933_cast_fp16_0, tensor var_11933_cast_fp16_1 = split(axis = var_11933_axis_0, split_sizes = var_11933_split_sizes_0, x = normed_577_cast_fp16)[name = string("op_11933_cast_fp16")]; tensor const_351_to_fp16 = const()[name = string("const_351_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107278080)))]; tensor var_11936_cast_fp16 = mul(x = var_11933_cast_fp16_0, y = const_351_to_fp16)[name = string("op_11936_cast_fp16")]; tensor var_11949 = const()[name = string("op_11949"), val = tensor([0, 2, 1])]; tensor input_595_axes_0 = const()[name = string("input_595_axes_0"), val = tensor([2])]; tensor var_11950 = transpose(perm = var_11949, x = var_11936_cast_fp16)[name = string("transpose_94")]; tensor input_595 = expand_dims(axes = input_595_axes_0, x = var_11950)[name = string("input_595")]; string gate_85_pad_type_0 = const()[name = string("gate_85_pad_type_0"), val = string("valid")]; tensor gate_85_strides_0 = const()[name = string("gate_85_strides_0"), val = tensor([1, 1])]; tensor gate_85_pad_0 = const()[name = string("gate_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_85_dilations_0 = const()[name = string("gate_85_dilations_0"), val = tensor([1, 1])]; int32 gate_85_groups_0 = const()[name = string("gate_85_groups_0"), val = int32(1)]; tensor gate_85 = conv(dilations = gate_85_dilations_0, groups = gate_85_groups_0, pad = gate_85_pad_0, pad_type = gate_85_pad_type_0, strides = gate_85_strides_0, weight = layers_21_mlp_gate_proj_weight_palettized, x = input_595)[name = string("gate_85")]; string up_43_pad_type_0 = const()[name = string("up_43_pad_type_0"), val = string("valid")]; tensor up_43_strides_0 = const()[name = string("up_43_strides_0"), val = tensor([1, 1])]; tensor up_43_pad_0 = const()[name = string("up_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_43_dilations_0 = const()[name = string("up_43_dilations_0"), val = tensor([1, 1])]; int32 up_43_groups_0 = const()[name = string("up_43_groups_0"), val = int32(1)]; tensor up_43 = conv(dilations = up_43_dilations_0, groups = up_43_groups_0, pad = up_43_pad_0, pad_type = up_43_pad_type_0, strides = up_43_strides_0, weight = layers_21_mlp_up_proj_weight_palettized, x = input_595)[name = string("up_43")]; string gate_87_mode_0 = const()[name = string("gate_87_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_87 = gelu(mode = gate_87_mode_0, x = gate_85)[name = string("gate_87")]; tensor input_597 = mul(x = gate_87, y = up_43)[name = string("input_597")]; string mlp_out_43_pad_type_0 = const()[name = string("mlp_out_43_pad_type_0"), val = string("valid")]; tensor mlp_out_43_strides_0 = const()[name = string("mlp_out_43_strides_0"), val = tensor([1, 1])]; tensor mlp_out_43_pad_0 = const()[name = string("mlp_out_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_43_dilations_0 = const()[name = string("mlp_out_43_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_43_groups_0 = const()[name = string("mlp_out_43_groups_0"), val = int32(1)]; tensor mlp_out_43 = conv(dilations = mlp_out_43_dilations_0, groups = mlp_out_43_groups_0, pad = mlp_out_43_pad_0, pad_type = mlp_out_43_pad_type_0, strides = mlp_out_43_strides_0, weight = layers_21_mlp_down_proj_weight_palettized, x = input_597)[name = string("mlp_out_43")]; tensor var_11990_axes_0 = const()[name = string("op_11990_axes_0"), val = tensor([2])]; tensor var_11990 = squeeze(axes = var_11990_axes_0, x = mlp_out_43)[name = string("op_11990")]; tensor var_11994 = const()[name = string("op_11994"), val = tensor([0, 2, 1])]; int32 var_12000 = const()[name = string("op_12000"), val = int32(-1)]; fp16 const_352_promoted_to_fp16 = const()[name = string("const_352_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_611 = transpose(perm = var_11994, x = var_11990)[name = string("transpose_93")]; tensor var_12006_cast_fp16 = mul(x = x_611, y = const_352_promoted_to_fp16)[name = string("op_12006_cast_fp16")]; bool input_599_interleave_0 = const()[name = string("input_599_interleave_0"), val = bool(false)]; tensor input_599_cast_fp16 = concat(axis = var_12000, interleave = input_599_interleave_0, values = (x_611, var_12006_cast_fp16))[name = string("input_599_cast_fp16")]; tensor normed_581_axes_0 = const()[name = string("normed_581_axes_0"), val = tensor([-1])]; fp16 var_11998_to_fp16 = const()[name = string("op_11998_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_581_cast_fp16 = layer_norm(axes = normed_581_axes_0, epsilon = var_11998_to_fp16, x = input_599_cast_fp16)[name = string("normed_581_cast_fp16")]; tensor var_12011_split_sizes_0 = const()[name = string("op_12011_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12011_axis_0 = const()[name = string("op_12011_axis_0"), val = int32(-1)]; tensor var_12011_cast_fp16_0, tensor var_12011_cast_fp16_1 = split(axis = var_12011_axis_0, split_sizes = var_12011_split_sizes_0, x = normed_581_cast_fp16)[name = string("op_12011_cast_fp16")]; tensor const_353_to_fp16 = const()[name = string("const_353_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107281216)))]; tensor var_12014_cast_fp16 = mul(x = var_12011_cast_fp16_0, y = const_353_to_fp16)[name = string("op_12014_cast_fp16")]; tensor hidden_states_261_cast_fp16 = add(x = x_607_cast_fp16, y = var_12014_cast_fp16)[name = string("hidden_states_261_cast_fp16")]; tensor per_layer_slice_43_begin_0 = const()[name = string("per_layer_slice_43_begin_0"), val = tensor([0, 0, 5376])]; tensor per_layer_slice_43_end_0 = const()[name = string("per_layer_slice_43_end_0"), val = tensor([1, 1, 5632])]; tensor per_layer_slice_43_end_mask_0 = const()[name = string("per_layer_slice_43_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_43_cast_fp16 = slice_by_index(begin = per_layer_slice_43_begin_0, end = per_layer_slice_43_end_0, end_mask = per_layer_slice_43_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_43_cast_fp16")]; tensor gated_85 = linear(bias = linear_0_bias_0, weight = layers_21_per_layer_input_gate_weight_palettized, x = hidden_states_261_cast_fp16)[name = string("linear_42")]; string gated_87_mode_0 = const()[name = string("gated_87_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_87 = gelu(mode = gated_87_mode_0, x = gated_85)[name = string("gated_87")]; tensor input_603_cast_fp16 = mul(x = gated_87, y = per_layer_slice_43_cast_fp16)[name = string("input_603_cast_fp16")]; tensor layers_21_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107284352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107481024))))[name = string("layers_21_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_43_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_21_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_603_cast_fp16)[name = string("linear_43_cast_fp16")]; int32 var_12051 = const()[name = string("op_12051"), val = int32(-1)]; fp16 const_354_promoted_to_fp16 = const()[name = string("const_354_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12057_cast_fp16 = mul(x = linear_43_cast_fp16, y = const_354_promoted_to_fp16)[name = string("op_12057_cast_fp16")]; bool input_605_interleave_0 = const()[name = string("input_605_interleave_0"), val = bool(false)]; tensor input_605_cast_fp16 = concat(axis = var_12051, interleave = input_605_interleave_0, values = (linear_43_cast_fp16, var_12057_cast_fp16))[name = string("input_605_cast_fp16")]; tensor normed_585_axes_0 = const()[name = string("normed_585_axes_0"), val = tensor([-1])]; fp16 var_12049_to_fp16 = const()[name = string("op_12049_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_585_cast_fp16 = layer_norm(axes = normed_585_axes_0, epsilon = var_12049_to_fp16, x = input_605_cast_fp16)[name = string("normed_585_cast_fp16")]; tensor var_12062_split_sizes_0 = const()[name = string("op_12062_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12062_axis_0 = const()[name = string("op_12062_axis_0"), val = int32(-1)]; tensor var_12062_cast_fp16_0, tensor var_12062_cast_fp16_1 = split(axis = var_12062_axis_0, split_sizes = var_12062_split_sizes_0, x = normed_585_cast_fp16)[name = string("op_12062_cast_fp16")]; tensor const_355_to_fp16 = const()[name = string("const_355_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107482624)))]; tensor var_12065_cast_fp16 = mul(x = var_12062_cast_fp16_0, y = const_355_to_fp16)[name = string("op_12065_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_261_cast_fp16, y = var_12065_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; tensor layers_21_layer_scalar_to_fp16 = const()[name = string("layers_21_layer_scalar_to_fp16"), val = tensor([0x1.4ap-1])]; tensor x_619_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = layers_21_layer_scalar_to_fp16)[name = string("x_619_cast_fp16")]; int32 var_12073 = const()[name = string("op_12073"), val = int32(-1)]; fp16 const_356_promoted_to_fp16 = const()[name = string("const_356_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12079_cast_fp16 = mul(x = x_619_cast_fp16, y = const_356_promoted_to_fp16)[name = string("op_12079_cast_fp16")]; bool input_607_interleave_0 = const()[name = string("input_607_interleave_0"), val = bool(false)]; tensor input_607_cast_fp16 = concat(axis = var_12073, interleave = input_607_interleave_0, values = (x_619_cast_fp16, var_12079_cast_fp16))[name = string("input_607_cast_fp16")]; tensor normed_589_axes_0 = const()[name = string("normed_589_axes_0"), val = tensor([-1])]; fp16 var_12071_to_fp16 = const()[name = string("op_12071_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_589_cast_fp16 = layer_norm(axes = normed_589_axes_0, epsilon = var_12071_to_fp16, x = input_607_cast_fp16)[name = string("normed_589_cast_fp16")]; tensor var_12084_split_sizes_0 = const()[name = string("op_12084_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12084_axis_0 = const()[name = string("op_12084_axis_0"), val = int32(-1)]; tensor var_12084_cast_fp16_0, tensor var_12084_cast_fp16_1 = split(axis = var_12084_axis_0, split_sizes = var_12084_split_sizes_0, x = normed_589_cast_fp16)[name = string("op_12084_cast_fp16")]; tensor const_357_to_fp16 = const()[name = string("const_357_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107485760)))]; tensor var_12087_cast_fp16 = mul(x = var_12084_cast_fp16_0, y = const_357_to_fp16)[name = string("op_12087_cast_fp16")]; tensor var_12095 = const()[name = string("op_12095"), val = tensor([0, 2, 1])]; tensor var_12098_axes_0 = const()[name = string("op_12098_axes_0"), val = tensor([2])]; tensor var_12096_cast_fp16 = transpose(perm = var_12095, x = var_12087_cast_fp16)[name = string("transpose_92")]; tensor var_12098_cast_fp16 = expand_dims(axes = var_12098_axes_0, x = var_12096_cast_fp16)[name = string("op_12098_cast_fp16")]; string var_12114_pad_type_0 = const()[name = string("op_12114_pad_type_0"), val = string("valid")]; tensor var_12114_strides_0 = const()[name = string("op_12114_strides_0"), val = tensor([1, 1])]; tensor var_12114_pad_0 = const()[name = string("op_12114_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12114_dilations_0 = const()[name = string("op_12114_dilations_0"), val = tensor([1, 1])]; int32 var_12114_groups_0 = const()[name = string("op_12114_groups_0"), val = int32(1)]; tensor var_12114 = conv(dilations = var_12114_dilations_0, groups = var_12114_groups_0, pad = var_12114_pad_0, pad_type = var_12114_pad_type_0, strides = var_12114_strides_0, weight = layers_22_self_attn_q_proj_weight_palettized, x = var_12098_cast_fp16)[name = string("op_12114")]; tensor var_12119 = const()[name = string("op_12119"), val = tensor([1, 8, 256, 1])]; tensor var_12120 = reshape(shape = var_12119, x = var_12114)[name = string("op_12120")]; tensor var_12125 = const()[name = string("op_12125"), val = tensor([0, 1, 3, 2])]; tensor var_12135 = const()[name = string("op_12135"), val = tensor([1, 8, 256])]; tensor var_12126 = transpose(perm = var_12125, x = var_12120)[name = string("transpose_91")]; tensor x_623 = reshape(shape = var_12135, x = var_12126)[name = string("x_623")]; int32 var_12141 = const()[name = string("op_12141"), val = int32(-1)]; fp16 const_358_promoted_to_fp16 = const()[name = string("const_358_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12147_cast_fp16 = mul(x = x_623, y = const_358_promoted_to_fp16)[name = string("op_12147_cast_fp16")]; bool input_611_interleave_0 = const()[name = string("input_611_interleave_0"), val = bool(false)]; tensor input_611_cast_fp16 = concat(axis = var_12141, interleave = input_611_interleave_0, values = (x_623, var_12147_cast_fp16))[name = string("input_611_cast_fp16")]; tensor normed_593_axes_0 = const()[name = string("normed_593_axes_0"), val = tensor([-1])]; fp16 var_12139_to_fp16 = const()[name = string("op_12139_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_593_cast_fp16 = layer_norm(axes = normed_593_axes_0, epsilon = var_12139_to_fp16, x = input_611_cast_fp16)[name = string("normed_593_cast_fp16")]; tensor var_12152_split_sizes_0 = const()[name = string("op_12152_split_sizes_0"), val = tensor([256, 256])]; int32 var_12152_axis_0 = const()[name = string("op_12152_axis_0"), val = int32(-1)]; tensor var_12152_cast_fp16_0, tensor var_12152_cast_fp16_1 = split(axis = var_12152_axis_0, split_sizes = var_12152_split_sizes_0, x = normed_593_cast_fp16)[name = string("op_12152_cast_fp16")]; tensor var_12155_cast_fp16 = mul(x = var_12152_cast_fp16_0, y = const_234_to_fp16)[name = string("op_12155_cast_fp16")]; tensor var_12161 = const()[name = string("op_12161"), val = tensor([1, 8, 1, 256])]; tensor q_165 = reshape(shape = var_12161, x = var_12155_cast_fp16)[name = string("q_165")]; tensor var_12163 = mul(x = q_165, y = cos_1)[name = string("op_12163")]; tensor var_12164_split_sizes_0 = const()[name = string("op_12164_split_sizes_0"), val = tensor([128, 128])]; int32 var_12164_axis_0 = const()[name = string("op_12164_axis_0"), val = int32(-1)]; tensor var_12164_0, tensor var_12164_1 = split(axis = var_12164_axis_0, split_sizes = var_12164_split_sizes_0, x = q_165)[name = string("op_12164")]; fp16 const_360_promoted = const()[name = string("const_360_promoted"), val = fp16(-0x1p+0)]; tensor var_12166 = mul(x = var_12164_1, y = const_360_promoted)[name = string("op_12166")]; int32 var_12168 = const()[name = string("op_12168"), val = int32(-1)]; bool var_12169_interleave_0 = const()[name = string("op_12169_interleave_0"), val = bool(false)]; tensor var_12169 = concat(axis = var_12168, interleave = var_12169_interleave_0, values = (var_12166, var_12164_0))[name = string("op_12169")]; tensor var_12170 = mul(x = var_12169, y = sin_1)[name = string("op_12170")]; tensor q_167 = add(x = var_12163, y = var_12170)[name = string("q_167")]; bool var_12184_transpose_x_0 = const()[name = string("op_12184_transpose_x_0"), val = bool(false)]; bool var_12184_transpose_y_0 = const()[name = string("op_12184_transpose_y_0"), val = bool(false)]; tensor var_12184_cast_fp16 = matmul(transpose_x = var_12184_transpose_x_0, transpose_y = var_12184_transpose_y_0, x = q_167, y = transpose_153_cast_fp16)[name = string("op_12184_cast_fp16")]; tensor attn_weights_135_cast_fp16 = add(x = var_12184_cast_fp16, y = causal_mask)[name = string("attn_weights_135_cast_fp16")]; int32 var_12189 = const()[name = string("op_12189"), val = int32(-1)]; tensor attn_weights_137_cast_fp16 = softmax(axis = var_12189, x = attn_weights_135_cast_fp16)[name = string("attn_weights_137_cast_fp16")]; bool attn_output_133_transpose_x_0 = const()[name = string("attn_output_133_transpose_x_0"), val = bool(false)]; bool attn_output_133_transpose_y_0 = const()[name = string("attn_output_133_transpose_y_0"), val = bool(false)]; tensor attn_output_133_cast_fp16 = matmul(transpose_x = attn_output_133_transpose_x_0, transpose_y = attn_output_133_transpose_y_0, x = attn_weights_137_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_133_cast_fp16")]; tensor var_12197 = const()[name = string("op_12197"), val = tensor([0, 2, 1, 3])]; tensor var_12204 = const()[name = string("op_12204"), val = tensor([1, 1, -1])]; tensor var_12198_cast_fp16 = transpose(perm = var_12197, x = attn_output_133_cast_fp16)[name = string("transpose_90")]; tensor attn_output_135_cast_fp16 = reshape(shape = var_12204, x = var_12198_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor var_12209 = const()[name = string("op_12209"), val = tensor([0, 2, 1])]; string var_12225_pad_type_0 = const()[name = string("op_12225_pad_type_0"), val = string("valid")]; int32 var_12225_groups_0 = const()[name = string("op_12225_groups_0"), val = int32(1)]; tensor var_12225_strides_0 = const()[name = string("op_12225_strides_0"), val = tensor([1])]; tensor var_12225_pad_0 = const()[name = string("op_12225_pad_0"), val = tensor([0, 0])]; tensor var_12225_dilations_0 = const()[name = string("op_12225_dilations_0"), val = tensor([1])]; tensor squeeze_22_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1107488896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109061824))))[name = string("squeeze_22_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12210_cast_fp16 = transpose(perm = var_12209, x = attn_output_135_cast_fp16)[name = string("transpose_89")]; tensor var_12225_cast_fp16 = conv(dilations = var_12225_dilations_0, groups = var_12225_groups_0, pad = var_12225_pad_0, pad_type = var_12225_pad_type_0, strides = var_12225_strides_0, weight = squeeze_22_cast_fp16_to_fp32_to_fp16_palettized, x = var_12210_cast_fp16)[name = string("op_12225_cast_fp16")]; tensor var_12229 = const()[name = string("op_12229"), val = tensor([0, 2, 1])]; int32 var_12235 = const()[name = string("op_12235"), val = int32(-1)]; fp16 const_361_promoted_to_fp16 = const()[name = string("const_361_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_627_cast_fp16 = transpose(perm = var_12229, x = var_12225_cast_fp16)[name = string("transpose_88")]; tensor var_12241_cast_fp16 = mul(x = x_627_cast_fp16, y = const_361_promoted_to_fp16)[name = string("op_12241_cast_fp16")]; bool input_615_interleave_0 = const()[name = string("input_615_interleave_0"), val = bool(false)]; tensor input_615_cast_fp16 = concat(axis = var_12235, interleave = input_615_interleave_0, values = (x_627_cast_fp16, var_12241_cast_fp16))[name = string("input_615_cast_fp16")]; tensor normed_597_axes_0 = const()[name = string("normed_597_axes_0"), val = tensor([-1])]; fp16 var_12233_to_fp16 = const()[name = string("op_12233_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_597_cast_fp16 = layer_norm(axes = normed_597_axes_0, epsilon = var_12233_to_fp16, x = input_615_cast_fp16)[name = string("normed_597_cast_fp16")]; tensor var_12246_split_sizes_0 = const()[name = string("op_12246_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12246_axis_0 = const()[name = string("op_12246_axis_0"), val = int32(-1)]; tensor var_12246_cast_fp16_0, tensor var_12246_cast_fp16_1 = split(axis = var_12246_axis_0, split_sizes = var_12246_split_sizes_0, x = normed_597_cast_fp16)[name = string("op_12246_cast_fp16")]; tensor const_362_to_fp16 = const()[name = string("const_362_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109063424)))]; tensor var_12249_cast_fp16 = mul(x = var_12246_cast_fp16_0, y = const_362_to_fp16)[name = string("op_12249_cast_fp16")]; tensor x_631_cast_fp16 = add(x = x_619_cast_fp16, y = var_12249_cast_fp16)[name = string("x_631_cast_fp16")]; int32 var_12256 = const()[name = string("op_12256"), val = int32(-1)]; fp16 const_363_promoted_to_fp16 = const()[name = string("const_363_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12262_cast_fp16 = mul(x = x_631_cast_fp16, y = const_363_promoted_to_fp16)[name = string("op_12262_cast_fp16")]; bool input_617_interleave_0 = const()[name = string("input_617_interleave_0"), val = bool(false)]; tensor input_617_cast_fp16 = concat(axis = var_12256, interleave = input_617_interleave_0, values = (x_631_cast_fp16, var_12262_cast_fp16))[name = string("input_617_cast_fp16")]; tensor normed_601_axes_0 = const()[name = string("normed_601_axes_0"), val = tensor([-1])]; fp16 var_12254_to_fp16 = const()[name = string("op_12254_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_601_cast_fp16 = layer_norm(axes = normed_601_axes_0, epsilon = var_12254_to_fp16, x = input_617_cast_fp16)[name = string("normed_601_cast_fp16")]; tensor var_12267_split_sizes_0 = const()[name = string("op_12267_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12267_axis_0 = const()[name = string("op_12267_axis_0"), val = int32(-1)]; tensor var_12267_cast_fp16_0, tensor var_12267_cast_fp16_1 = split(axis = var_12267_axis_0, split_sizes = var_12267_split_sizes_0, x = normed_601_cast_fp16)[name = string("op_12267_cast_fp16")]; tensor const_364_to_fp16 = const()[name = string("const_364_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109066560)))]; tensor var_12270_cast_fp16 = mul(x = var_12267_cast_fp16_0, y = const_364_to_fp16)[name = string("op_12270_cast_fp16")]; tensor var_12283 = const()[name = string("op_12283"), val = tensor([0, 2, 1])]; tensor input_619_axes_0 = const()[name = string("input_619_axes_0"), val = tensor([2])]; tensor var_12284 = transpose(perm = var_12283, x = var_12270_cast_fp16)[name = string("transpose_87")]; tensor input_619 = expand_dims(axes = input_619_axes_0, x = var_12284)[name = string("input_619")]; string gate_89_pad_type_0 = const()[name = string("gate_89_pad_type_0"), val = string("valid")]; tensor gate_89_strides_0 = const()[name = string("gate_89_strides_0"), val = tensor([1, 1])]; tensor gate_89_pad_0 = const()[name = string("gate_89_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_89_dilations_0 = const()[name = string("gate_89_dilations_0"), val = tensor([1, 1])]; int32 gate_89_groups_0 = const()[name = string("gate_89_groups_0"), val = int32(1)]; tensor gate_89 = conv(dilations = gate_89_dilations_0, groups = gate_89_groups_0, pad = gate_89_pad_0, pad_type = gate_89_pad_type_0, strides = gate_89_strides_0, weight = layers_22_mlp_gate_proj_weight_palettized, x = input_619)[name = string("gate_89")]; string up_45_pad_type_0 = const()[name = string("up_45_pad_type_0"), val = string("valid")]; tensor up_45_strides_0 = const()[name = string("up_45_strides_0"), val = tensor([1, 1])]; tensor up_45_pad_0 = const()[name = string("up_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_45_dilations_0 = const()[name = string("up_45_dilations_0"), val = tensor([1, 1])]; int32 up_45_groups_0 = const()[name = string("up_45_groups_0"), val = int32(1)]; tensor up_45 = conv(dilations = up_45_dilations_0, groups = up_45_groups_0, pad = up_45_pad_0, pad_type = up_45_pad_type_0, strides = up_45_strides_0, weight = layers_22_mlp_up_proj_weight_palettized, x = input_619)[name = string("up_45")]; string gate_91_mode_0 = const()[name = string("gate_91_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_91 = gelu(mode = gate_91_mode_0, x = gate_89)[name = string("gate_91")]; tensor input_621 = mul(x = gate_91, y = up_45)[name = string("input_621")]; string mlp_out_45_pad_type_0 = const()[name = string("mlp_out_45_pad_type_0"), val = string("valid")]; tensor mlp_out_45_strides_0 = const()[name = string("mlp_out_45_strides_0"), val = tensor([1, 1])]; tensor mlp_out_45_pad_0 = const()[name = string("mlp_out_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_45_dilations_0 = const()[name = string("mlp_out_45_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_45_groups_0 = const()[name = string("mlp_out_45_groups_0"), val = int32(1)]; tensor mlp_out_45 = conv(dilations = mlp_out_45_dilations_0, groups = mlp_out_45_groups_0, pad = mlp_out_45_pad_0, pad_type = mlp_out_45_pad_type_0, strides = mlp_out_45_strides_0, weight = layers_22_mlp_down_proj_weight_palettized, x = input_621)[name = string("mlp_out_45")]; tensor var_12324_axes_0 = const()[name = string("op_12324_axes_0"), val = tensor([2])]; tensor var_12324 = squeeze(axes = var_12324_axes_0, x = mlp_out_45)[name = string("op_12324")]; tensor var_12328 = const()[name = string("op_12328"), val = tensor([0, 2, 1])]; int32 var_12334 = const()[name = string("op_12334"), val = int32(-1)]; fp16 const_365_promoted_to_fp16 = const()[name = string("const_365_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_635 = transpose(perm = var_12328, x = var_12324)[name = string("transpose_86")]; tensor var_12340_cast_fp16 = mul(x = x_635, y = const_365_promoted_to_fp16)[name = string("op_12340_cast_fp16")]; bool input_623_interleave_0 = const()[name = string("input_623_interleave_0"), val = bool(false)]; tensor input_623_cast_fp16 = concat(axis = var_12334, interleave = input_623_interleave_0, values = (x_635, var_12340_cast_fp16))[name = string("input_623_cast_fp16")]; tensor normed_605_axes_0 = const()[name = string("normed_605_axes_0"), val = tensor([-1])]; fp16 var_12332_to_fp16 = const()[name = string("op_12332_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_605_cast_fp16 = layer_norm(axes = normed_605_axes_0, epsilon = var_12332_to_fp16, x = input_623_cast_fp16)[name = string("normed_605_cast_fp16")]; tensor var_12345_split_sizes_0 = const()[name = string("op_12345_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12345_axis_0 = const()[name = string("op_12345_axis_0"), val = int32(-1)]; tensor var_12345_cast_fp16_0, tensor var_12345_cast_fp16_1 = split(axis = var_12345_axis_0, split_sizes = var_12345_split_sizes_0, x = normed_605_cast_fp16)[name = string("op_12345_cast_fp16")]; tensor const_366_to_fp16 = const()[name = string("const_366_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109069696)))]; tensor var_12348_cast_fp16 = mul(x = var_12345_cast_fp16_0, y = const_366_to_fp16)[name = string("op_12348_cast_fp16")]; tensor hidden_states_273_cast_fp16 = add(x = x_631_cast_fp16, y = var_12348_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor per_layer_slice_45_begin_0 = const()[name = string("per_layer_slice_45_begin_0"), val = tensor([0, 0, 5632])]; tensor per_layer_slice_45_end_0 = const()[name = string("per_layer_slice_45_end_0"), val = tensor([1, 1, 5888])]; tensor per_layer_slice_45_end_mask_0 = const()[name = string("per_layer_slice_45_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_45_cast_fp16 = slice_by_index(begin = per_layer_slice_45_begin_0, end = per_layer_slice_45_end_0, end_mask = per_layer_slice_45_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_45_cast_fp16")]; tensor gated_89 = linear(bias = linear_0_bias_0, weight = layers_22_per_layer_input_gate_weight_palettized, x = hidden_states_273_cast_fp16)[name = string("linear_44")]; string gated_91_mode_0 = const()[name = string("gated_91_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_91 = gelu(mode = gated_91_mode_0, x = gated_89)[name = string("gated_91")]; tensor input_627_cast_fp16 = mul(x = gated_91, y = per_layer_slice_45_cast_fp16)[name = string("input_627_cast_fp16")]; tensor layers_22_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109072832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109269504))))[name = string("layers_22_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_45_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_22_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_627_cast_fp16)[name = string("linear_45_cast_fp16")]; int32 var_12385 = const()[name = string("op_12385"), val = int32(-1)]; fp16 const_367_promoted_to_fp16 = const()[name = string("const_367_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12391_cast_fp16 = mul(x = linear_45_cast_fp16, y = const_367_promoted_to_fp16)[name = string("op_12391_cast_fp16")]; bool input_629_interleave_0 = const()[name = string("input_629_interleave_0"), val = bool(false)]; tensor input_629_cast_fp16 = concat(axis = var_12385, interleave = input_629_interleave_0, values = (linear_45_cast_fp16, var_12391_cast_fp16))[name = string("input_629_cast_fp16")]; tensor normed_609_axes_0 = const()[name = string("normed_609_axes_0"), val = tensor([-1])]; fp16 var_12383_to_fp16 = const()[name = string("op_12383_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_609_cast_fp16 = layer_norm(axes = normed_609_axes_0, epsilon = var_12383_to_fp16, x = input_629_cast_fp16)[name = string("normed_609_cast_fp16")]; tensor var_12396_split_sizes_0 = const()[name = string("op_12396_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12396_axis_0 = const()[name = string("op_12396_axis_0"), val = int32(-1)]; tensor var_12396_cast_fp16_0, tensor var_12396_cast_fp16_1 = split(axis = var_12396_axis_0, split_sizes = var_12396_split_sizes_0, x = normed_609_cast_fp16)[name = string("op_12396_cast_fp16")]; tensor const_368_to_fp16 = const()[name = string("const_368_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109271104)))]; tensor var_12399_cast_fp16 = mul(x = var_12396_cast_fp16_0, y = const_368_to_fp16)[name = string("op_12399_cast_fp16")]; tensor hidden_states_277_cast_fp16 = add(x = hidden_states_273_cast_fp16, y = var_12399_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor layers_22_layer_scalar_to_fp16 = const()[name = string("layers_22_layer_scalar_to_fp16"), val = tensor([0x1.44p-1])]; tensor x_643_cast_fp16 = mul(x = hidden_states_277_cast_fp16, y = layers_22_layer_scalar_to_fp16)[name = string("x_643_cast_fp16")]; int32 var_12407 = const()[name = string("op_12407"), val = int32(-1)]; fp16 const_369_promoted_to_fp16 = const()[name = string("const_369_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12413_cast_fp16 = mul(x = x_643_cast_fp16, y = const_369_promoted_to_fp16)[name = string("op_12413_cast_fp16")]; bool input_631_interleave_0 = const()[name = string("input_631_interleave_0"), val = bool(false)]; tensor input_631_cast_fp16 = concat(axis = var_12407, interleave = input_631_interleave_0, values = (x_643_cast_fp16, var_12413_cast_fp16))[name = string("input_631_cast_fp16")]; tensor normed_613_axes_0 = const()[name = string("normed_613_axes_0"), val = tensor([-1])]; fp16 var_12405_to_fp16 = const()[name = string("op_12405_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_613_cast_fp16 = layer_norm(axes = normed_613_axes_0, epsilon = var_12405_to_fp16, x = input_631_cast_fp16)[name = string("normed_613_cast_fp16")]; tensor var_12418_split_sizes_0 = const()[name = string("op_12418_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12418_axis_0 = const()[name = string("op_12418_axis_0"), val = int32(-1)]; tensor var_12418_cast_fp16_0, tensor var_12418_cast_fp16_1 = split(axis = var_12418_axis_0, split_sizes = var_12418_split_sizes_0, x = normed_613_cast_fp16)[name = string("op_12418_cast_fp16")]; tensor const_370_to_fp16 = const()[name = string("const_370_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109274240)))]; tensor var_12421_cast_fp16 = mul(x = var_12418_cast_fp16_0, y = const_370_to_fp16)[name = string("op_12421_cast_fp16")]; tensor var_12429 = const()[name = string("op_12429"), val = tensor([0, 2, 1])]; tensor var_12432_axes_0 = const()[name = string("op_12432_axes_0"), val = tensor([2])]; tensor var_12430_cast_fp16 = transpose(perm = var_12429, x = var_12421_cast_fp16)[name = string("transpose_85")]; tensor var_12432_cast_fp16 = expand_dims(axes = var_12432_axes_0, x = var_12430_cast_fp16)[name = string("op_12432_cast_fp16")]; string var_12448_pad_type_0 = const()[name = string("op_12448_pad_type_0"), val = string("valid")]; tensor var_12448_strides_0 = const()[name = string("op_12448_strides_0"), val = tensor([1, 1])]; tensor var_12448_pad_0 = const()[name = string("op_12448_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12448_dilations_0 = const()[name = string("op_12448_dilations_0"), val = tensor([1, 1])]; int32 var_12448_groups_0 = const()[name = string("op_12448_groups_0"), val = int32(1)]; tensor var_12448 = conv(dilations = var_12448_dilations_0, groups = var_12448_groups_0, pad = var_12448_pad_0, pad_type = var_12448_pad_type_0, strides = var_12448_strides_0, weight = layers_23_self_attn_q_proj_weight_palettized, x = var_12432_cast_fp16)[name = string("op_12448")]; tensor var_12453 = const()[name = string("op_12453"), val = tensor([1, 8, 256, 1])]; tensor var_12454 = reshape(shape = var_12453, x = var_12448)[name = string("op_12454")]; tensor var_12459 = const()[name = string("op_12459"), val = tensor([0, 1, 3, 2])]; tensor var_12469 = const()[name = string("op_12469"), val = tensor([1, 8, 256])]; tensor var_12460 = transpose(perm = var_12459, x = var_12454)[name = string("transpose_84")]; tensor x_647 = reshape(shape = var_12469, x = var_12460)[name = string("x_647")]; int32 var_12475 = const()[name = string("op_12475"), val = int32(-1)]; fp16 const_371_promoted_to_fp16 = const()[name = string("const_371_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12481_cast_fp16 = mul(x = x_647, y = const_371_promoted_to_fp16)[name = string("op_12481_cast_fp16")]; bool input_635_interleave_0 = const()[name = string("input_635_interleave_0"), val = bool(false)]; tensor input_635_cast_fp16 = concat(axis = var_12475, interleave = input_635_interleave_0, values = (x_647, var_12481_cast_fp16))[name = string("input_635_cast_fp16")]; tensor normed_617_axes_0 = const()[name = string("normed_617_axes_0"), val = tensor([-1])]; fp16 var_12473_to_fp16 = const()[name = string("op_12473_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_617_cast_fp16 = layer_norm(axes = normed_617_axes_0, epsilon = var_12473_to_fp16, x = input_635_cast_fp16)[name = string("normed_617_cast_fp16")]; tensor var_12486_split_sizes_0 = const()[name = string("op_12486_split_sizes_0"), val = tensor([256, 256])]; int32 var_12486_axis_0 = const()[name = string("op_12486_axis_0"), val = int32(-1)]; tensor var_12486_cast_fp16_0, tensor var_12486_cast_fp16_1 = split(axis = var_12486_axis_0, split_sizes = var_12486_split_sizes_0, x = normed_617_cast_fp16)[name = string("op_12486_cast_fp16")]; tensor var_12489_cast_fp16 = mul(x = var_12486_cast_fp16_0, y = const_234_to_fp16)[name = string("op_12489_cast_fp16")]; tensor var_12495 = const()[name = string("op_12495"), val = tensor([1, 8, 1, 256])]; tensor q_171 = reshape(shape = var_12495, x = var_12489_cast_fp16)[name = string("q_171")]; tensor var_12497 = mul(x = q_171, y = cos_1)[name = string("op_12497")]; tensor var_12498_split_sizes_0 = const()[name = string("op_12498_split_sizes_0"), val = tensor([128, 128])]; int32 var_12498_axis_0 = const()[name = string("op_12498_axis_0"), val = int32(-1)]; tensor var_12498_0, tensor var_12498_1 = split(axis = var_12498_axis_0, split_sizes = var_12498_split_sizes_0, x = q_171)[name = string("op_12498")]; fp16 const_373_promoted = const()[name = string("const_373_promoted"), val = fp16(-0x1p+0)]; tensor var_12500 = mul(x = var_12498_1, y = const_373_promoted)[name = string("op_12500")]; int32 var_12502 = const()[name = string("op_12502"), val = int32(-1)]; bool var_12503_interleave_0 = const()[name = string("op_12503_interleave_0"), val = bool(false)]; tensor var_12503 = concat(axis = var_12502, interleave = var_12503_interleave_0, values = (var_12500, var_12498_0))[name = string("op_12503")]; tensor var_12504 = mul(x = var_12503, y = sin_1)[name = string("op_12504")]; tensor q_173 = add(x = var_12497, y = var_12504)[name = string("q_173")]; bool var_12518_transpose_x_0 = const()[name = string("op_12518_transpose_x_0"), val = bool(false)]; bool var_12518_transpose_y_0 = const()[name = string("op_12518_transpose_y_0"), val = bool(false)]; tensor var_12518_cast_fp16 = matmul(transpose_x = var_12518_transpose_x_0, transpose_y = var_12518_transpose_y_0, x = q_173, y = transpose_153_cast_fp16)[name = string("op_12518_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = var_12518_cast_fp16, y = causal_mask)[name = string("attn_weights_141_cast_fp16")]; int32 var_12523 = const()[name = string("op_12523"), val = int32(-1)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_12523, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_139_transpose_x_0 = const()[name = string("attn_output_139_transpose_x_0"), val = bool(false)]; bool attn_output_139_transpose_y_0 = const()[name = string("attn_output_139_transpose_y_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = matmul(transpose_x = attn_output_139_transpose_x_0, transpose_y = attn_output_139_transpose_y_0, x = attn_weights_143_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_139_cast_fp16")]; tensor var_12531 = const()[name = string("op_12531"), val = tensor([0, 2, 1, 3])]; tensor var_12538 = const()[name = string("op_12538"), val = tensor([1, 1, -1])]; tensor var_12532_cast_fp16 = transpose(perm = var_12531, x = attn_output_139_cast_fp16)[name = string("transpose_83")]; tensor attn_output_141_cast_fp16 = reshape(shape = var_12538, x = var_12532_cast_fp16)[name = string("attn_output_141_cast_fp16")]; tensor var_12543 = const()[name = string("op_12543"), val = tensor([0, 2, 1])]; string var_12559_pad_type_0 = const()[name = string("op_12559_pad_type_0"), val = string("valid")]; int32 var_12559_groups_0 = const()[name = string("op_12559_groups_0"), val = int32(1)]; tensor var_12559_strides_0 = const()[name = string("op_12559_strides_0"), val = tensor([1])]; tensor var_12559_pad_0 = const()[name = string("op_12559_pad_0"), val = tensor([0, 0])]; tensor var_12559_dilations_0 = const()[name = string("op_12559_dilations_0"), val = tensor([1])]; tensor squeeze_23_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1109277376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1110850304))))[name = string("squeeze_23_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12544_cast_fp16 = transpose(perm = var_12543, x = attn_output_141_cast_fp16)[name = string("transpose_82")]; tensor var_12559_cast_fp16 = conv(dilations = var_12559_dilations_0, groups = var_12559_groups_0, pad = var_12559_pad_0, pad_type = var_12559_pad_type_0, strides = var_12559_strides_0, weight = squeeze_23_cast_fp16_to_fp32_to_fp16_palettized, x = var_12544_cast_fp16)[name = string("op_12559_cast_fp16")]; tensor var_12563 = const()[name = string("op_12563"), val = tensor([0, 2, 1])]; int32 var_12569 = const()[name = string("op_12569"), val = int32(-1)]; fp16 const_374_promoted_to_fp16 = const()[name = string("const_374_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_651_cast_fp16 = transpose(perm = var_12563, x = var_12559_cast_fp16)[name = string("transpose_81")]; tensor var_12575_cast_fp16 = mul(x = x_651_cast_fp16, y = const_374_promoted_to_fp16)[name = string("op_12575_cast_fp16")]; bool input_639_interleave_0 = const()[name = string("input_639_interleave_0"), val = bool(false)]; tensor input_639_cast_fp16 = concat(axis = var_12569, interleave = input_639_interleave_0, values = (x_651_cast_fp16, var_12575_cast_fp16))[name = string("input_639_cast_fp16")]; tensor normed_621_axes_0 = const()[name = string("normed_621_axes_0"), val = tensor([-1])]; fp16 var_12567_to_fp16 = const()[name = string("op_12567_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_621_cast_fp16 = layer_norm(axes = normed_621_axes_0, epsilon = var_12567_to_fp16, x = input_639_cast_fp16)[name = string("normed_621_cast_fp16")]; tensor var_12580_split_sizes_0 = const()[name = string("op_12580_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12580_axis_0 = const()[name = string("op_12580_axis_0"), val = int32(-1)]; tensor var_12580_cast_fp16_0, tensor var_12580_cast_fp16_1 = split(axis = var_12580_axis_0, split_sizes = var_12580_split_sizes_0, x = normed_621_cast_fp16)[name = string("op_12580_cast_fp16")]; tensor const_375_to_fp16 = const()[name = string("const_375_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1110851904)))]; tensor var_12583_cast_fp16 = mul(x = var_12580_cast_fp16_0, y = const_375_to_fp16)[name = string("op_12583_cast_fp16")]; tensor x_655_cast_fp16 = add(x = x_643_cast_fp16, y = var_12583_cast_fp16)[name = string("x_655_cast_fp16")]; int32 var_12590 = const()[name = string("op_12590"), val = int32(-1)]; fp16 const_376_promoted_to_fp16 = const()[name = string("const_376_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12596_cast_fp16 = mul(x = x_655_cast_fp16, y = const_376_promoted_to_fp16)[name = string("op_12596_cast_fp16")]; bool input_641_interleave_0 = const()[name = string("input_641_interleave_0"), val = bool(false)]; tensor input_641_cast_fp16 = concat(axis = var_12590, interleave = input_641_interleave_0, values = (x_655_cast_fp16, var_12596_cast_fp16))[name = string("input_641_cast_fp16")]; tensor normed_625_axes_0 = const()[name = string("normed_625_axes_0"), val = tensor([-1])]; fp16 var_12588_to_fp16 = const()[name = string("op_12588_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_625_cast_fp16 = layer_norm(axes = normed_625_axes_0, epsilon = var_12588_to_fp16, x = input_641_cast_fp16)[name = string("normed_625_cast_fp16")]; tensor var_12601_split_sizes_0 = const()[name = string("op_12601_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12601_axis_0 = const()[name = string("op_12601_axis_0"), val = int32(-1)]; tensor var_12601_cast_fp16_0, tensor var_12601_cast_fp16_1 = split(axis = var_12601_axis_0, split_sizes = var_12601_split_sizes_0, x = normed_625_cast_fp16)[name = string("op_12601_cast_fp16")]; tensor const_377_to_fp16 = const()[name = string("const_377_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1110855040)))]; tensor var_12604_cast_fp16 = mul(x = var_12601_cast_fp16_0, y = const_377_to_fp16)[name = string("op_12604_cast_fp16")]; tensor var_12617 = const()[name = string("op_12617"), val = tensor([0, 2, 1])]; tensor input_643_axes_0 = const()[name = string("input_643_axes_0"), val = tensor([2])]; tensor var_12618 = transpose(perm = var_12617, x = var_12604_cast_fp16)[name = string("transpose_80")]; tensor input_643 = expand_dims(axes = input_643_axes_0, x = var_12618)[name = string("input_643")]; string gate_93_pad_type_0 = const()[name = string("gate_93_pad_type_0"), val = string("valid")]; tensor gate_93_strides_0 = const()[name = string("gate_93_strides_0"), val = tensor([1, 1])]; tensor gate_93_pad_0 = const()[name = string("gate_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_93_dilations_0 = const()[name = string("gate_93_dilations_0"), val = tensor([1, 1])]; int32 gate_93_groups_0 = const()[name = string("gate_93_groups_0"), val = int32(1)]; tensor gate_93 = conv(dilations = gate_93_dilations_0, groups = gate_93_groups_0, pad = gate_93_pad_0, pad_type = gate_93_pad_type_0, strides = gate_93_strides_0, weight = layers_23_mlp_gate_proj_weight_palettized, x = input_643)[name = string("gate_93")]; string up_47_pad_type_0 = const()[name = string("up_47_pad_type_0"), val = string("valid")]; tensor up_47_strides_0 = const()[name = string("up_47_strides_0"), val = tensor([1, 1])]; tensor up_47_pad_0 = const()[name = string("up_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_47_dilations_0 = const()[name = string("up_47_dilations_0"), val = tensor([1, 1])]; int32 up_47_groups_0 = const()[name = string("up_47_groups_0"), val = int32(1)]; tensor up_47 = conv(dilations = up_47_dilations_0, groups = up_47_groups_0, pad = up_47_pad_0, pad_type = up_47_pad_type_0, strides = up_47_strides_0, weight = layers_23_mlp_up_proj_weight_palettized, x = input_643)[name = string("up_47")]; string gate_95_mode_0 = const()[name = string("gate_95_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_95 = gelu(mode = gate_95_mode_0, x = gate_93)[name = string("gate_95")]; tensor input_645 = mul(x = gate_95, y = up_47)[name = string("input_645")]; string mlp_out_47_pad_type_0 = const()[name = string("mlp_out_47_pad_type_0"), val = string("valid")]; tensor mlp_out_47_strides_0 = const()[name = string("mlp_out_47_strides_0"), val = tensor([1, 1])]; tensor mlp_out_47_pad_0 = const()[name = string("mlp_out_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_47_dilations_0 = const()[name = string("mlp_out_47_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_47_groups_0 = const()[name = string("mlp_out_47_groups_0"), val = int32(1)]; tensor mlp_out_47 = conv(dilations = mlp_out_47_dilations_0, groups = mlp_out_47_groups_0, pad = mlp_out_47_pad_0, pad_type = mlp_out_47_pad_type_0, strides = mlp_out_47_strides_0, weight = layers_23_mlp_down_proj_weight_palettized, x = input_645)[name = string("mlp_out_47")]; tensor var_12658_axes_0 = const()[name = string("op_12658_axes_0"), val = tensor([2])]; tensor var_12658 = squeeze(axes = var_12658_axes_0, x = mlp_out_47)[name = string("op_12658")]; tensor var_12662 = const()[name = string("op_12662"), val = tensor([0, 2, 1])]; int32 var_12668 = const()[name = string("op_12668"), val = int32(-1)]; fp16 const_378_promoted_to_fp16 = const()[name = string("const_378_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_659 = transpose(perm = var_12662, x = var_12658)[name = string("transpose_79")]; tensor var_12674_cast_fp16 = mul(x = x_659, y = const_378_promoted_to_fp16)[name = string("op_12674_cast_fp16")]; bool input_647_interleave_0 = const()[name = string("input_647_interleave_0"), val = bool(false)]; tensor input_647_cast_fp16 = concat(axis = var_12668, interleave = input_647_interleave_0, values = (x_659, var_12674_cast_fp16))[name = string("input_647_cast_fp16")]; tensor normed_629_axes_0 = const()[name = string("normed_629_axes_0"), val = tensor([-1])]; fp16 var_12666_to_fp16 = const()[name = string("op_12666_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_629_cast_fp16 = layer_norm(axes = normed_629_axes_0, epsilon = var_12666_to_fp16, x = input_647_cast_fp16)[name = string("normed_629_cast_fp16")]; tensor var_12679_split_sizes_0 = const()[name = string("op_12679_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12679_axis_0 = const()[name = string("op_12679_axis_0"), val = int32(-1)]; tensor var_12679_cast_fp16_0, tensor var_12679_cast_fp16_1 = split(axis = var_12679_axis_0, split_sizes = var_12679_split_sizes_0, x = normed_629_cast_fp16)[name = string("op_12679_cast_fp16")]; tensor const_379_to_fp16 = const()[name = string("const_379_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1110858176)))]; tensor var_12682_cast_fp16 = mul(x = var_12679_cast_fp16_0, y = const_379_to_fp16)[name = string("op_12682_cast_fp16")]; tensor hidden_states_285_cast_fp16 = add(x = x_655_cast_fp16, y = var_12682_cast_fp16)[name = string("hidden_states_285_cast_fp16")]; tensor per_layer_slice_47_begin_0 = const()[name = string("per_layer_slice_47_begin_0"), val = tensor([0, 0, 5888])]; tensor per_layer_slice_47_end_0 = const()[name = string("per_layer_slice_47_end_0"), val = tensor([1, 1, 6144])]; tensor per_layer_slice_47_end_mask_0 = const()[name = string("per_layer_slice_47_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_47_cast_fp16 = slice_by_index(begin = per_layer_slice_47_begin_0, end = per_layer_slice_47_end_0, end_mask = per_layer_slice_47_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_47_cast_fp16")]; tensor gated_93 = linear(bias = linear_0_bias_0, weight = layers_23_per_layer_input_gate_weight_palettized, x = hidden_states_285_cast_fp16)[name = string("linear_46")]; string gated_95_mode_0 = const()[name = string("gated_95_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_95 = gelu(mode = gated_95_mode_0, x = gated_93)[name = string("gated_95")]; tensor input_651_cast_fp16 = mul(x = gated_95, y = per_layer_slice_47_cast_fp16)[name = string("input_651_cast_fp16")]; tensor layers_23_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1110861312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1111057984))))[name = string("layers_23_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_47_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_23_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_651_cast_fp16)[name = string("linear_47_cast_fp16")]; int32 var_12719 = const()[name = string("op_12719"), val = int32(-1)]; fp16 const_380_promoted_to_fp16 = const()[name = string("const_380_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12725_cast_fp16 = mul(x = linear_47_cast_fp16, y = const_380_promoted_to_fp16)[name = string("op_12725_cast_fp16")]; bool input_653_interleave_0 = const()[name = string("input_653_interleave_0"), val = bool(false)]; tensor input_653_cast_fp16 = concat(axis = var_12719, interleave = input_653_interleave_0, values = (linear_47_cast_fp16, var_12725_cast_fp16))[name = string("input_653_cast_fp16")]; tensor normed_633_axes_0 = const()[name = string("normed_633_axes_0"), val = tensor([-1])]; fp16 var_12717_to_fp16 = const()[name = string("op_12717_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_633_cast_fp16 = layer_norm(axes = normed_633_axes_0, epsilon = var_12717_to_fp16, x = input_653_cast_fp16)[name = string("normed_633_cast_fp16")]; tensor var_12730_split_sizes_0 = const()[name = string("op_12730_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12730_axis_0 = const()[name = string("op_12730_axis_0"), val = int32(-1)]; tensor var_12730_cast_fp16_0, tensor var_12730_cast_fp16_1 = split(axis = var_12730_axis_0, split_sizes = var_12730_split_sizes_0, x = normed_633_cast_fp16)[name = string("op_12730_cast_fp16")]; tensor const_381_to_fp16 = const()[name = string("const_381_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1111059584)))]; tensor var_12733_cast_fp16 = mul(x = var_12730_cast_fp16_0, y = const_381_to_fp16)[name = string("op_12733_cast_fp16")]; tensor hidden_states_289_cast_fp16 = add(x = hidden_states_285_cast_fp16, y = var_12733_cast_fp16)[name = string("hidden_states_289_cast_fp16")]; tensor layers_23_layer_scalar_to_fp16 = const()[name = string("layers_23_layer_scalar_to_fp16"), val = tensor([0x1.bap-2])]; tensor x_667_cast_fp16 = mul(x = hidden_states_289_cast_fp16, y = layers_23_layer_scalar_to_fp16)[name = string("x_667_cast_fp16")]; int32 var_12741 = const()[name = string("op_12741"), val = int32(-1)]; fp16 const_382_promoted_to_fp16 = const()[name = string("const_382_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12747_cast_fp16 = mul(x = x_667_cast_fp16, y = const_382_promoted_to_fp16)[name = string("op_12747_cast_fp16")]; bool input_655_interleave_0 = const()[name = string("input_655_interleave_0"), val = bool(false)]; tensor input_655_cast_fp16 = concat(axis = var_12741, interleave = input_655_interleave_0, values = (x_667_cast_fp16, var_12747_cast_fp16))[name = string("input_655_cast_fp16")]; tensor normed_637_axes_0 = const()[name = string("normed_637_axes_0"), val = tensor([-1])]; fp16 var_12739_to_fp16 = const()[name = string("op_12739_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_637_cast_fp16 = layer_norm(axes = normed_637_axes_0, epsilon = var_12739_to_fp16, x = input_655_cast_fp16)[name = string("normed_637_cast_fp16")]; tensor var_12752_split_sizes_0 = const()[name = string("op_12752_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12752_axis_0 = const()[name = string("op_12752_axis_0"), val = int32(-1)]; tensor var_12752_cast_fp16_0, tensor var_12752_cast_fp16_1 = split(axis = var_12752_axis_0, split_sizes = var_12752_split_sizes_0, x = normed_637_cast_fp16)[name = string("op_12752_cast_fp16")]; tensor const_383_to_fp16 = const()[name = string("const_383_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1111062720)))]; tensor var_12755_cast_fp16 = mul(x = var_12752_cast_fp16_0, y = const_383_to_fp16)[name = string("op_12755_cast_fp16")]; tensor var_12763 = const()[name = string("op_12763"), val = tensor([0, 2, 1])]; tensor var_12766_axes_0 = const()[name = string("op_12766_axes_0"), val = tensor([2])]; tensor var_12764_cast_fp16 = transpose(perm = var_12763, x = var_12755_cast_fp16)[name = string("transpose_78")]; tensor var_12766_cast_fp16 = expand_dims(axes = var_12766_axes_0, x = var_12764_cast_fp16)[name = string("op_12766_cast_fp16")]; string var_12782_pad_type_0 = const()[name = string("op_12782_pad_type_0"), val = string("valid")]; tensor var_12782_strides_0 = const()[name = string("op_12782_strides_0"), val = tensor([1, 1])]; tensor var_12782_pad_0 = const()[name = string("op_12782_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12782_dilations_0 = const()[name = string("op_12782_dilations_0"), val = tensor([1, 1])]; int32 var_12782_groups_0 = const()[name = string("op_12782_groups_0"), val = int32(1)]; tensor var_12782 = conv(dilations = var_12782_dilations_0, groups = var_12782_groups_0, pad = var_12782_pad_0, pad_type = var_12782_pad_type_0, strides = var_12782_strides_0, weight = layers_24_self_attn_q_proj_weight_palettized, x = var_12766_cast_fp16)[name = string("op_12782")]; tensor var_12787 = const()[name = string("op_12787"), val = tensor([1, 8, 512, 1])]; tensor var_12788 = reshape(shape = var_12787, x = var_12782)[name = string("op_12788")]; tensor var_12793 = const()[name = string("op_12793"), val = tensor([0, 1, 3, 2])]; tensor var_12803 = const()[name = string("op_12803"), val = tensor([1, 8, 512])]; tensor var_12794 = transpose(perm = var_12793, x = var_12788)[name = string("transpose_77")]; tensor x_671 = reshape(shape = var_12803, x = var_12794)[name = string("x_671")]; int32 var_12809 = const()[name = string("op_12809"), val = int32(-1)]; fp16 const_384_promoted_to_fp16 = const()[name = string("const_384_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12815_cast_fp16 = mul(x = x_671, y = const_384_promoted_to_fp16)[name = string("op_12815_cast_fp16")]; bool input_659_interleave_0 = const()[name = string("input_659_interleave_0"), val = bool(false)]; tensor input_659_cast_fp16 = concat(axis = var_12809, interleave = input_659_interleave_0, values = (x_671, var_12815_cast_fp16))[name = string("input_659_cast_fp16")]; tensor normed_641_axes_0 = const()[name = string("normed_641_axes_0"), val = tensor([-1])]; fp16 var_12807_to_fp16 = const()[name = string("op_12807_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_641_cast_fp16 = layer_norm(axes = normed_641_axes_0, epsilon = var_12807_to_fp16, x = input_659_cast_fp16)[name = string("normed_641_cast_fp16")]; tensor var_12820_split_sizes_0 = const()[name = string("op_12820_split_sizes_0"), val = tensor([512, 512])]; int32 var_12820_axis_0 = const()[name = string("op_12820_axis_0"), val = int32(-1)]; tensor var_12820_cast_fp16_0, tensor var_12820_cast_fp16_1 = split(axis = var_12820_axis_0, split_sizes = var_12820_split_sizes_0, x = normed_641_cast_fp16)[name = string("op_12820_cast_fp16")]; tensor var_12823_cast_fp16 = mul(x = var_12820_cast_fp16_0, y = const_252_to_fp16)[name = string("op_12823_cast_fp16")]; tensor var_12829 = const()[name = string("op_12829"), val = tensor([1, 8, 1, 512])]; tensor q_177 = reshape(shape = var_12829, x = var_12823_cast_fp16)[name = string("q_177")]; tensor var_12831 = mul(x = q_177, y = cos)[name = string("op_12831")]; tensor var_12832_split_sizes_0 = const()[name = string("op_12832_split_sizes_0"), val = tensor([256, 256])]; int32 var_12832_axis_0 = const()[name = string("op_12832_axis_0"), val = int32(-1)]; tensor var_12832_0, tensor var_12832_1 = split(axis = var_12832_axis_0, split_sizes = var_12832_split_sizes_0, x = q_177)[name = string("op_12832")]; fp16 const_386_promoted = const()[name = string("const_386_promoted"), val = fp16(-0x1p+0)]; tensor var_12834 = mul(x = var_12832_1, y = const_386_promoted)[name = string("op_12834")]; int32 var_12836 = const()[name = string("op_12836"), val = int32(-1)]; bool var_12837_interleave_0 = const()[name = string("op_12837_interleave_0"), val = bool(false)]; tensor var_12837 = concat(axis = var_12836, interleave = var_12837_interleave_0, values = (var_12834, var_12832_0))[name = string("op_12837")]; tensor var_12838 = mul(x = var_12837, y = sin)[name = string("op_12838")]; tensor q_179 = add(x = var_12831, y = var_12838)[name = string("q_179")]; bool var_12852_transpose_x_0 = const()[name = string("op_12852_transpose_x_0"), val = bool(false)]; bool var_12852_transpose_y_0 = const()[name = string("op_12852_transpose_y_0"), val = bool(false)]; tensor var_12852_cast_fp16 = matmul(transpose_x = var_12852_transpose_x_0, transpose_y = var_12852_transpose_y_0, x = q_179, y = transpose_154_cast_fp16)[name = string("op_12852_cast_fp16")]; tensor attn_weights_147_cast_fp16 = add(x = var_12852_cast_fp16, y = causal_mask)[name = string("attn_weights_147_cast_fp16")]; int32 var_12857 = const()[name = string("op_12857"), val = int32(-1)]; tensor attn_weights_149_cast_fp16 = softmax(axis = var_12857, x = attn_weights_147_cast_fp16)[name = string("attn_weights_149_cast_fp16")]; bool attn_output_145_transpose_x_0 = const()[name = string("attn_output_145_transpose_x_0"), val = bool(false)]; bool attn_output_145_transpose_y_0 = const()[name = string("attn_output_145_transpose_y_0"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_0, transpose_y = attn_output_145_transpose_y_0, x = attn_weights_149_cast_fp16, y = V_expanded_29_cast_fp16)[name = string("attn_output_145_cast_fp16")]; tensor var_12865 = const()[name = string("op_12865"), val = tensor([0, 2, 1, 3])]; tensor var_12872 = const()[name = string("op_12872"), val = tensor([1, 1, -1])]; tensor var_12866_cast_fp16 = transpose(perm = var_12865, x = attn_output_145_cast_fp16)[name = string("transpose_76")]; tensor attn_output_147_cast_fp16 = reshape(shape = var_12872, x = var_12866_cast_fp16)[name = string("attn_output_147_cast_fp16")]; tensor var_12877 = const()[name = string("op_12877"), val = tensor([0, 2, 1])]; string var_12893_pad_type_0 = const()[name = string("op_12893_pad_type_0"), val = string("valid")]; int32 var_12893_groups_0 = const()[name = string("op_12893_groups_0"), val = int32(1)]; tensor var_12893_strides_0 = const()[name = string("op_12893_strides_0"), val = tensor([1])]; tensor var_12893_pad_0 = const()[name = string("op_12893_pad_0"), val = tensor([0, 0])]; tensor var_12893_dilations_0 = const()[name = string("op_12893_dilations_0"), val = tensor([1])]; tensor squeeze_24_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1111065856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114211648))))[name = string("squeeze_24_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12878_cast_fp16 = transpose(perm = var_12877, x = attn_output_147_cast_fp16)[name = string("transpose_75")]; tensor var_12893_cast_fp16 = conv(dilations = var_12893_dilations_0, groups = var_12893_groups_0, pad = var_12893_pad_0, pad_type = var_12893_pad_type_0, strides = var_12893_strides_0, weight = squeeze_24_cast_fp16_to_fp32_to_fp16_palettized, x = var_12878_cast_fp16)[name = string("op_12893_cast_fp16")]; tensor var_12897 = const()[name = string("op_12897"), val = tensor([0, 2, 1])]; int32 var_12903 = const()[name = string("op_12903"), val = int32(-1)]; fp16 const_387_promoted_to_fp16 = const()[name = string("const_387_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_675_cast_fp16 = transpose(perm = var_12897, x = var_12893_cast_fp16)[name = string("transpose_74")]; tensor var_12909_cast_fp16 = mul(x = x_675_cast_fp16, y = const_387_promoted_to_fp16)[name = string("op_12909_cast_fp16")]; bool input_663_interleave_0 = const()[name = string("input_663_interleave_0"), val = bool(false)]; tensor input_663_cast_fp16 = concat(axis = var_12903, interleave = input_663_interleave_0, values = (x_675_cast_fp16, var_12909_cast_fp16))[name = string("input_663_cast_fp16")]; tensor normed_645_axes_0 = const()[name = string("normed_645_axes_0"), val = tensor([-1])]; fp16 var_12901_to_fp16 = const()[name = string("op_12901_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_645_cast_fp16 = layer_norm(axes = normed_645_axes_0, epsilon = var_12901_to_fp16, x = input_663_cast_fp16)[name = string("normed_645_cast_fp16")]; tensor var_12914_split_sizes_0 = const()[name = string("op_12914_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12914_axis_0 = const()[name = string("op_12914_axis_0"), val = int32(-1)]; tensor var_12914_cast_fp16_0, tensor var_12914_cast_fp16_1 = split(axis = var_12914_axis_0, split_sizes = var_12914_split_sizes_0, x = normed_645_cast_fp16)[name = string("op_12914_cast_fp16")]; tensor const_388_to_fp16 = const()[name = string("const_388_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114213248)))]; tensor var_12917_cast_fp16 = mul(x = var_12914_cast_fp16_0, y = const_388_to_fp16)[name = string("op_12917_cast_fp16")]; tensor x_679_cast_fp16 = add(x = x_667_cast_fp16, y = var_12917_cast_fp16)[name = string("x_679_cast_fp16")]; int32 var_12924 = const()[name = string("op_12924"), val = int32(-1)]; fp16 const_389_promoted_to_fp16 = const()[name = string("const_389_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12930_cast_fp16 = mul(x = x_679_cast_fp16, y = const_389_promoted_to_fp16)[name = string("op_12930_cast_fp16")]; bool input_665_interleave_0 = const()[name = string("input_665_interleave_0"), val = bool(false)]; tensor input_665_cast_fp16 = concat(axis = var_12924, interleave = input_665_interleave_0, values = (x_679_cast_fp16, var_12930_cast_fp16))[name = string("input_665_cast_fp16")]; tensor normed_649_axes_0 = const()[name = string("normed_649_axes_0"), val = tensor([-1])]; fp16 var_12922_to_fp16 = const()[name = string("op_12922_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_649_cast_fp16 = layer_norm(axes = normed_649_axes_0, epsilon = var_12922_to_fp16, x = input_665_cast_fp16)[name = string("normed_649_cast_fp16")]; tensor var_12935_split_sizes_0 = const()[name = string("op_12935_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_12935_axis_0 = const()[name = string("op_12935_axis_0"), val = int32(-1)]; tensor var_12935_cast_fp16_0, tensor var_12935_cast_fp16_1 = split(axis = var_12935_axis_0, split_sizes = var_12935_split_sizes_0, x = normed_649_cast_fp16)[name = string("op_12935_cast_fp16")]; tensor const_390_to_fp16 = const()[name = string("const_390_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114216384)))]; tensor var_12938_cast_fp16 = mul(x = var_12935_cast_fp16_0, y = const_390_to_fp16)[name = string("op_12938_cast_fp16")]; tensor var_12951 = const()[name = string("op_12951"), val = tensor([0, 2, 1])]; tensor input_667_axes_0 = const()[name = string("input_667_axes_0"), val = tensor([2])]; tensor var_12952 = transpose(perm = var_12951, x = var_12938_cast_fp16)[name = string("transpose_73")]; tensor input_667 = expand_dims(axes = input_667_axes_0, x = var_12952)[name = string("input_667")]; string gate_97_pad_type_0 = const()[name = string("gate_97_pad_type_0"), val = string("valid")]; tensor gate_97_strides_0 = const()[name = string("gate_97_strides_0"), val = tensor([1, 1])]; tensor gate_97_pad_0 = const()[name = string("gate_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_97_dilations_0 = const()[name = string("gate_97_dilations_0"), val = tensor([1, 1])]; int32 gate_97_groups_0 = const()[name = string("gate_97_groups_0"), val = int32(1)]; tensor gate_97 = conv(dilations = gate_97_dilations_0, groups = gate_97_groups_0, pad = gate_97_pad_0, pad_type = gate_97_pad_type_0, strides = gate_97_strides_0, weight = layers_24_mlp_gate_proj_weight_palettized, x = input_667)[name = string("gate_97")]; string up_49_pad_type_0 = const()[name = string("up_49_pad_type_0"), val = string("valid")]; tensor up_49_strides_0 = const()[name = string("up_49_strides_0"), val = tensor([1, 1])]; tensor up_49_pad_0 = const()[name = string("up_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_49_dilations_0 = const()[name = string("up_49_dilations_0"), val = tensor([1, 1])]; int32 up_49_groups_0 = const()[name = string("up_49_groups_0"), val = int32(1)]; tensor up_49 = conv(dilations = up_49_dilations_0, groups = up_49_groups_0, pad = up_49_pad_0, pad_type = up_49_pad_type_0, strides = up_49_strides_0, weight = layers_24_mlp_up_proj_weight_palettized, x = input_667)[name = string("up_49")]; string gate_99_mode_0 = const()[name = string("gate_99_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_99 = gelu(mode = gate_99_mode_0, x = gate_97)[name = string("gate_99")]; tensor input_669 = mul(x = gate_99, y = up_49)[name = string("input_669")]; string mlp_out_49_pad_type_0 = const()[name = string("mlp_out_49_pad_type_0"), val = string("valid")]; tensor mlp_out_49_strides_0 = const()[name = string("mlp_out_49_strides_0"), val = tensor([1, 1])]; tensor mlp_out_49_pad_0 = const()[name = string("mlp_out_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_49_dilations_0 = const()[name = string("mlp_out_49_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_49_groups_0 = const()[name = string("mlp_out_49_groups_0"), val = int32(1)]; tensor mlp_out_49 = conv(dilations = mlp_out_49_dilations_0, groups = mlp_out_49_groups_0, pad = mlp_out_49_pad_0, pad_type = mlp_out_49_pad_type_0, strides = mlp_out_49_strides_0, weight = layers_24_mlp_down_proj_weight_palettized, x = input_669)[name = string("mlp_out_49")]; tensor var_12992_axes_0 = const()[name = string("op_12992_axes_0"), val = tensor([2])]; tensor var_12992 = squeeze(axes = var_12992_axes_0, x = mlp_out_49)[name = string("op_12992")]; tensor var_12996 = const()[name = string("op_12996"), val = tensor([0, 2, 1])]; int32 var_13002 = const()[name = string("op_13002"), val = int32(-1)]; fp16 const_391_promoted_to_fp16 = const()[name = string("const_391_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_683 = transpose(perm = var_12996, x = var_12992)[name = string("transpose_72")]; tensor var_13008_cast_fp16 = mul(x = x_683, y = const_391_promoted_to_fp16)[name = string("op_13008_cast_fp16")]; bool input_671_interleave_0 = const()[name = string("input_671_interleave_0"), val = bool(false)]; tensor input_671_cast_fp16 = concat(axis = var_13002, interleave = input_671_interleave_0, values = (x_683, var_13008_cast_fp16))[name = string("input_671_cast_fp16")]; tensor normed_653_axes_0 = const()[name = string("normed_653_axes_0"), val = tensor([-1])]; fp16 var_13000_to_fp16 = const()[name = string("op_13000_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_653_cast_fp16 = layer_norm(axes = normed_653_axes_0, epsilon = var_13000_to_fp16, x = input_671_cast_fp16)[name = string("normed_653_cast_fp16")]; tensor var_13013_split_sizes_0 = const()[name = string("op_13013_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13013_axis_0 = const()[name = string("op_13013_axis_0"), val = int32(-1)]; tensor var_13013_cast_fp16_0, tensor var_13013_cast_fp16_1 = split(axis = var_13013_axis_0, split_sizes = var_13013_split_sizes_0, x = normed_653_cast_fp16)[name = string("op_13013_cast_fp16")]; tensor const_392_to_fp16 = const()[name = string("const_392_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114219520)))]; tensor var_13016_cast_fp16 = mul(x = var_13013_cast_fp16_0, y = const_392_to_fp16)[name = string("op_13016_cast_fp16")]; tensor hidden_states_297_cast_fp16 = add(x = x_679_cast_fp16, y = var_13016_cast_fp16)[name = string("hidden_states_297_cast_fp16")]; tensor per_layer_slice_49_begin_0 = const()[name = string("per_layer_slice_49_begin_0"), val = tensor([0, 0, 6144])]; tensor per_layer_slice_49_end_0 = const()[name = string("per_layer_slice_49_end_0"), val = tensor([1, 1, 6400])]; tensor per_layer_slice_49_end_mask_0 = const()[name = string("per_layer_slice_49_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_49_cast_fp16 = slice_by_index(begin = per_layer_slice_49_begin_0, end = per_layer_slice_49_end_0, end_mask = per_layer_slice_49_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_49_cast_fp16")]; tensor gated_97 = linear(bias = linear_0_bias_0, weight = layers_24_per_layer_input_gate_weight_palettized, x = hidden_states_297_cast_fp16)[name = string("linear_48")]; string gated_99_mode_0 = const()[name = string("gated_99_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_99 = gelu(mode = gated_99_mode_0, x = gated_97)[name = string("gated_99")]; tensor input_675_cast_fp16 = mul(x = gated_99, y = per_layer_slice_49_cast_fp16)[name = string("input_675_cast_fp16")]; tensor layers_24_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114222656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114419328))))[name = string("layers_24_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_49_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_24_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_675_cast_fp16)[name = string("linear_49_cast_fp16")]; int32 var_13053 = const()[name = string("op_13053"), val = int32(-1)]; fp16 const_393_promoted_to_fp16 = const()[name = string("const_393_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13059_cast_fp16 = mul(x = linear_49_cast_fp16, y = const_393_promoted_to_fp16)[name = string("op_13059_cast_fp16")]; bool input_677_interleave_0 = const()[name = string("input_677_interleave_0"), val = bool(false)]; tensor input_677_cast_fp16 = concat(axis = var_13053, interleave = input_677_interleave_0, values = (linear_49_cast_fp16, var_13059_cast_fp16))[name = string("input_677_cast_fp16")]; tensor normed_657_axes_0 = const()[name = string("normed_657_axes_0"), val = tensor([-1])]; fp16 var_13051_to_fp16 = const()[name = string("op_13051_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_657_cast_fp16 = layer_norm(axes = normed_657_axes_0, epsilon = var_13051_to_fp16, x = input_677_cast_fp16)[name = string("normed_657_cast_fp16")]; tensor var_13064_split_sizes_0 = const()[name = string("op_13064_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13064_axis_0 = const()[name = string("op_13064_axis_0"), val = int32(-1)]; tensor var_13064_cast_fp16_0, tensor var_13064_cast_fp16_1 = split(axis = var_13064_axis_0, split_sizes = var_13064_split_sizes_0, x = normed_657_cast_fp16)[name = string("op_13064_cast_fp16")]; tensor const_394_to_fp16 = const()[name = string("const_394_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114420928)))]; tensor var_13067_cast_fp16 = mul(x = var_13064_cast_fp16_0, y = const_394_to_fp16)[name = string("op_13067_cast_fp16")]; tensor hidden_states_301_cast_fp16 = add(x = hidden_states_297_cast_fp16, y = var_13067_cast_fp16)[name = string("hidden_states_301_cast_fp16")]; tensor layers_24_layer_scalar_to_fp16 = const()[name = string("layers_24_layer_scalar_to_fp16"), val = tensor([0x1.cp-2])]; tensor x_691_cast_fp16 = mul(x = hidden_states_301_cast_fp16, y = layers_24_layer_scalar_to_fp16)[name = string("x_691_cast_fp16")]; int32 var_13075 = const()[name = string("op_13075"), val = int32(-1)]; fp16 const_395_promoted_to_fp16 = const()[name = string("const_395_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13081_cast_fp16 = mul(x = x_691_cast_fp16, y = const_395_promoted_to_fp16)[name = string("op_13081_cast_fp16")]; bool input_679_interleave_0 = const()[name = string("input_679_interleave_0"), val = bool(false)]; tensor input_679_cast_fp16 = concat(axis = var_13075, interleave = input_679_interleave_0, values = (x_691_cast_fp16, var_13081_cast_fp16))[name = string("input_679_cast_fp16")]; tensor normed_661_axes_0 = const()[name = string("normed_661_axes_0"), val = tensor([-1])]; fp16 var_13073_to_fp16 = const()[name = string("op_13073_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_661_cast_fp16 = layer_norm(axes = normed_661_axes_0, epsilon = var_13073_to_fp16, x = input_679_cast_fp16)[name = string("normed_661_cast_fp16")]; tensor var_13086_split_sizes_0 = const()[name = string("op_13086_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13086_axis_0 = const()[name = string("op_13086_axis_0"), val = int32(-1)]; tensor var_13086_cast_fp16_0, tensor var_13086_cast_fp16_1 = split(axis = var_13086_axis_0, split_sizes = var_13086_split_sizes_0, x = normed_661_cast_fp16)[name = string("op_13086_cast_fp16")]; tensor const_396_to_fp16 = const()[name = string("const_396_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114424064)))]; tensor var_13089_cast_fp16 = mul(x = var_13086_cast_fp16_0, y = const_396_to_fp16)[name = string("op_13089_cast_fp16")]; tensor var_13097 = const()[name = string("op_13097"), val = tensor([0, 2, 1])]; tensor var_13100_axes_0 = const()[name = string("op_13100_axes_0"), val = tensor([2])]; tensor var_13098_cast_fp16 = transpose(perm = var_13097, x = var_13089_cast_fp16)[name = string("transpose_71")]; tensor var_13100_cast_fp16 = expand_dims(axes = var_13100_axes_0, x = var_13098_cast_fp16)[name = string("op_13100_cast_fp16")]; string var_13116_pad_type_0 = const()[name = string("op_13116_pad_type_0"), val = string("valid")]; tensor var_13116_strides_0 = const()[name = string("op_13116_strides_0"), val = tensor([1, 1])]; tensor var_13116_pad_0 = const()[name = string("op_13116_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13116_dilations_0 = const()[name = string("op_13116_dilations_0"), val = tensor([1, 1])]; int32 var_13116_groups_0 = const()[name = string("op_13116_groups_0"), val = int32(1)]; tensor var_13116 = conv(dilations = var_13116_dilations_0, groups = var_13116_groups_0, pad = var_13116_pad_0, pad_type = var_13116_pad_type_0, strides = var_13116_strides_0, weight = layers_25_self_attn_q_proj_weight_palettized, x = var_13100_cast_fp16)[name = string("op_13116")]; tensor var_13121 = const()[name = string("op_13121"), val = tensor([1, 8, 256, 1])]; tensor var_13122 = reshape(shape = var_13121, x = var_13116)[name = string("op_13122")]; tensor var_13127 = const()[name = string("op_13127"), val = tensor([0, 1, 3, 2])]; tensor var_13137 = const()[name = string("op_13137"), val = tensor([1, 8, 256])]; tensor var_13128 = transpose(perm = var_13127, x = var_13122)[name = string("transpose_70")]; tensor x_695 = reshape(shape = var_13137, x = var_13128)[name = string("x_695")]; int32 var_13143 = const()[name = string("op_13143"), val = int32(-1)]; fp16 const_397_promoted_to_fp16 = const()[name = string("const_397_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13149_cast_fp16 = mul(x = x_695, y = const_397_promoted_to_fp16)[name = string("op_13149_cast_fp16")]; bool input_683_interleave_0 = const()[name = string("input_683_interleave_0"), val = bool(false)]; tensor input_683_cast_fp16 = concat(axis = var_13143, interleave = input_683_interleave_0, values = (x_695, var_13149_cast_fp16))[name = string("input_683_cast_fp16")]; tensor normed_665_axes_0 = const()[name = string("normed_665_axes_0"), val = tensor([-1])]; fp16 var_13141_to_fp16 = const()[name = string("op_13141_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_665_cast_fp16 = layer_norm(axes = normed_665_axes_0, epsilon = var_13141_to_fp16, x = input_683_cast_fp16)[name = string("normed_665_cast_fp16")]; tensor var_13154_split_sizes_0 = const()[name = string("op_13154_split_sizes_0"), val = tensor([256, 256])]; int32 var_13154_axis_0 = const()[name = string("op_13154_axis_0"), val = int32(-1)]; tensor var_13154_cast_fp16_0, tensor var_13154_cast_fp16_1 = split(axis = var_13154_axis_0, split_sizes = var_13154_split_sizes_0, x = normed_665_cast_fp16)[name = string("op_13154_cast_fp16")]; tensor var_13157_cast_fp16 = mul(x = var_13154_cast_fp16_0, y = const_234_to_fp16)[name = string("op_13157_cast_fp16")]; tensor var_13163 = const()[name = string("op_13163"), val = tensor([1, 8, 1, 256])]; tensor q_183 = reshape(shape = var_13163, x = var_13157_cast_fp16)[name = string("q_183")]; tensor var_13165 = mul(x = q_183, y = cos_1)[name = string("op_13165")]; tensor var_13166_split_sizes_0 = const()[name = string("op_13166_split_sizes_0"), val = tensor([128, 128])]; int32 var_13166_axis_0 = const()[name = string("op_13166_axis_0"), val = int32(-1)]; tensor var_13166_0, tensor var_13166_1 = split(axis = var_13166_axis_0, split_sizes = var_13166_split_sizes_0, x = q_183)[name = string("op_13166")]; fp16 const_399_promoted = const()[name = string("const_399_promoted"), val = fp16(-0x1p+0)]; tensor var_13168 = mul(x = var_13166_1, y = const_399_promoted)[name = string("op_13168")]; int32 var_13170 = const()[name = string("op_13170"), val = int32(-1)]; bool var_13171_interleave_0 = const()[name = string("op_13171_interleave_0"), val = bool(false)]; tensor var_13171 = concat(axis = var_13170, interleave = var_13171_interleave_0, values = (var_13168, var_13166_0))[name = string("op_13171")]; tensor var_13172 = mul(x = var_13171, y = sin_1)[name = string("op_13172")]; tensor q_185 = add(x = var_13165, y = var_13172)[name = string("q_185")]; bool var_13186_transpose_x_0 = const()[name = string("op_13186_transpose_x_0"), val = bool(false)]; bool var_13186_transpose_y_0 = const()[name = string("op_13186_transpose_y_0"), val = bool(false)]; tensor var_13186_cast_fp16 = matmul(transpose_x = var_13186_transpose_x_0, transpose_y = var_13186_transpose_y_0, x = q_185, y = transpose_153_cast_fp16)[name = string("op_13186_cast_fp16")]; tensor attn_weights_153_cast_fp16 = add(x = var_13186_cast_fp16, y = causal_mask)[name = string("attn_weights_153_cast_fp16")]; int32 var_13191 = const()[name = string("op_13191"), val = int32(-1)]; tensor attn_weights_155_cast_fp16 = softmax(axis = var_13191, x = attn_weights_153_cast_fp16)[name = string("attn_weights_155_cast_fp16")]; bool attn_output_151_transpose_x_0 = const()[name = string("attn_output_151_transpose_x_0"), val = bool(false)]; bool attn_output_151_transpose_y_0 = const()[name = string("attn_output_151_transpose_y_0"), val = bool(false)]; tensor attn_output_151_cast_fp16 = matmul(transpose_x = attn_output_151_transpose_x_0, transpose_y = attn_output_151_transpose_y_0, x = attn_weights_155_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor var_13199 = const()[name = string("op_13199"), val = tensor([0, 2, 1, 3])]; tensor var_13206 = const()[name = string("op_13206"), val = tensor([1, 1, -1])]; tensor var_13200_cast_fp16 = transpose(perm = var_13199, x = attn_output_151_cast_fp16)[name = string("transpose_69")]; tensor attn_output_153_cast_fp16 = reshape(shape = var_13206, x = var_13200_cast_fp16)[name = string("attn_output_153_cast_fp16")]; tensor var_13211 = const()[name = string("op_13211"), val = tensor([0, 2, 1])]; string var_13227_pad_type_0 = const()[name = string("op_13227_pad_type_0"), val = string("valid")]; int32 var_13227_groups_0 = const()[name = string("op_13227_groups_0"), val = int32(1)]; tensor var_13227_strides_0 = const()[name = string("op_13227_strides_0"), val = tensor([1])]; tensor var_13227_pad_0 = const()[name = string("op_13227_pad_0"), val = tensor([0, 0])]; tensor var_13227_dilations_0 = const()[name = string("op_13227_dilations_0"), val = tensor([1])]; tensor squeeze_25_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1114427200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116000128))))[name = string("squeeze_25_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_13212_cast_fp16 = transpose(perm = var_13211, x = attn_output_153_cast_fp16)[name = string("transpose_68")]; tensor var_13227_cast_fp16 = conv(dilations = var_13227_dilations_0, groups = var_13227_groups_0, pad = var_13227_pad_0, pad_type = var_13227_pad_type_0, strides = var_13227_strides_0, weight = squeeze_25_cast_fp16_to_fp32_to_fp16_palettized, x = var_13212_cast_fp16)[name = string("op_13227_cast_fp16")]; tensor var_13231 = const()[name = string("op_13231"), val = tensor([0, 2, 1])]; int32 var_13237 = const()[name = string("op_13237"), val = int32(-1)]; fp16 const_400_promoted_to_fp16 = const()[name = string("const_400_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_699_cast_fp16 = transpose(perm = var_13231, x = var_13227_cast_fp16)[name = string("transpose_67")]; tensor var_13243_cast_fp16 = mul(x = x_699_cast_fp16, y = const_400_promoted_to_fp16)[name = string("op_13243_cast_fp16")]; bool input_687_interleave_0 = const()[name = string("input_687_interleave_0"), val = bool(false)]; tensor input_687_cast_fp16 = concat(axis = var_13237, interleave = input_687_interleave_0, values = (x_699_cast_fp16, var_13243_cast_fp16))[name = string("input_687_cast_fp16")]; tensor normed_669_axes_0 = const()[name = string("normed_669_axes_0"), val = tensor([-1])]; fp16 var_13235_to_fp16 = const()[name = string("op_13235_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_669_cast_fp16 = layer_norm(axes = normed_669_axes_0, epsilon = var_13235_to_fp16, x = input_687_cast_fp16)[name = string("normed_669_cast_fp16")]; tensor var_13248_split_sizes_0 = const()[name = string("op_13248_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13248_axis_0 = const()[name = string("op_13248_axis_0"), val = int32(-1)]; tensor var_13248_cast_fp16_0, tensor var_13248_cast_fp16_1 = split(axis = var_13248_axis_0, split_sizes = var_13248_split_sizes_0, x = normed_669_cast_fp16)[name = string("op_13248_cast_fp16")]; tensor const_401_to_fp16 = const()[name = string("const_401_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116001728)))]; tensor var_13251_cast_fp16 = mul(x = var_13248_cast_fp16_0, y = const_401_to_fp16)[name = string("op_13251_cast_fp16")]; tensor x_703_cast_fp16 = add(x = x_691_cast_fp16, y = var_13251_cast_fp16)[name = string("x_703_cast_fp16")]; int32 var_13258 = const()[name = string("op_13258"), val = int32(-1)]; fp16 const_402_promoted_to_fp16 = const()[name = string("const_402_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13264_cast_fp16 = mul(x = x_703_cast_fp16, y = const_402_promoted_to_fp16)[name = string("op_13264_cast_fp16")]; bool input_689_interleave_0 = const()[name = string("input_689_interleave_0"), val = bool(false)]; tensor input_689_cast_fp16 = concat(axis = var_13258, interleave = input_689_interleave_0, values = (x_703_cast_fp16, var_13264_cast_fp16))[name = string("input_689_cast_fp16")]; tensor normed_673_axes_0 = const()[name = string("normed_673_axes_0"), val = tensor([-1])]; fp16 var_13256_to_fp16 = const()[name = string("op_13256_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_673_cast_fp16 = layer_norm(axes = normed_673_axes_0, epsilon = var_13256_to_fp16, x = input_689_cast_fp16)[name = string("normed_673_cast_fp16")]; tensor var_13269_split_sizes_0 = const()[name = string("op_13269_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13269_axis_0 = const()[name = string("op_13269_axis_0"), val = int32(-1)]; tensor var_13269_cast_fp16_0, tensor var_13269_cast_fp16_1 = split(axis = var_13269_axis_0, split_sizes = var_13269_split_sizes_0, x = normed_673_cast_fp16)[name = string("op_13269_cast_fp16")]; tensor const_403_to_fp16 = const()[name = string("const_403_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116004864)))]; tensor var_13272_cast_fp16 = mul(x = var_13269_cast_fp16_0, y = const_403_to_fp16)[name = string("op_13272_cast_fp16")]; tensor var_13285 = const()[name = string("op_13285"), val = tensor([0, 2, 1])]; tensor input_691_axes_0 = const()[name = string("input_691_axes_0"), val = tensor([2])]; tensor var_13286 = transpose(perm = var_13285, x = var_13272_cast_fp16)[name = string("transpose_66")]; tensor input_691 = expand_dims(axes = input_691_axes_0, x = var_13286)[name = string("input_691")]; string gate_101_pad_type_0 = const()[name = string("gate_101_pad_type_0"), val = string("valid")]; tensor gate_101_strides_0 = const()[name = string("gate_101_strides_0"), val = tensor([1, 1])]; tensor gate_101_pad_0 = const()[name = string("gate_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_101_dilations_0 = const()[name = string("gate_101_dilations_0"), val = tensor([1, 1])]; int32 gate_101_groups_0 = const()[name = string("gate_101_groups_0"), val = int32(1)]; tensor gate_101 = conv(dilations = gate_101_dilations_0, groups = gate_101_groups_0, pad = gate_101_pad_0, pad_type = gate_101_pad_type_0, strides = gate_101_strides_0, weight = layers_25_mlp_gate_proj_weight_palettized, x = input_691)[name = string("gate_101")]; string up_51_pad_type_0 = const()[name = string("up_51_pad_type_0"), val = string("valid")]; tensor up_51_strides_0 = const()[name = string("up_51_strides_0"), val = tensor([1, 1])]; tensor up_51_pad_0 = const()[name = string("up_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_51_dilations_0 = const()[name = string("up_51_dilations_0"), val = tensor([1, 1])]; int32 up_51_groups_0 = const()[name = string("up_51_groups_0"), val = int32(1)]; tensor up_51 = conv(dilations = up_51_dilations_0, groups = up_51_groups_0, pad = up_51_pad_0, pad_type = up_51_pad_type_0, strides = up_51_strides_0, weight = layers_25_mlp_up_proj_weight_palettized, x = input_691)[name = string("up_51")]; string gate_103_mode_0 = const()[name = string("gate_103_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_103 = gelu(mode = gate_103_mode_0, x = gate_101)[name = string("gate_103")]; tensor input_693 = mul(x = gate_103, y = up_51)[name = string("input_693")]; string mlp_out_51_pad_type_0 = const()[name = string("mlp_out_51_pad_type_0"), val = string("valid")]; tensor mlp_out_51_strides_0 = const()[name = string("mlp_out_51_strides_0"), val = tensor([1, 1])]; tensor mlp_out_51_pad_0 = const()[name = string("mlp_out_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_51_dilations_0 = const()[name = string("mlp_out_51_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_51_groups_0 = const()[name = string("mlp_out_51_groups_0"), val = int32(1)]; tensor mlp_out_51 = conv(dilations = mlp_out_51_dilations_0, groups = mlp_out_51_groups_0, pad = mlp_out_51_pad_0, pad_type = mlp_out_51_pad_type_0, strides = mlp_out_51_strides_0, weight = layers_25_mlp_down_proj_weight_palettized, x = input_693)[name = string("mlp_out_51")]; tensor var_13326_axes_0 = const()[name = string("op_13326_axes_0"), val = tensor([2])]; tensor var_13326 = squeeze(axes = var_13326_axes_0, x = mlp_out_51)[name = string("op_13326")]; tensor var_13330 = const()[name = string("op_13330"), val = tensor([0, 2, 1])]; int32 var_13336 = const()[name = string("op_13336"), val = int32(-1)]; fp16 const_404_promoted_to_fp16 = const()[name = string("const_404_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_707 = transpose(perm = var_13330, x = var_13326)[name = string("transpose_65")]; tensor var_13342_cast_fp16 = mul(x = x_707, y = const_404_promoted_to_fp16)[name = string("op_13342_cast_fp16")]; bool input_695_interleave_0 = const()[name = string("input_695_interleave_0"), val = bool(false)]; tensor input_695_cast_fp16 = concat(axis = var_13336, interleave = input_695_interleave_0, values = (x_707, var_13342_cast_fp16))[name = string("input_695_cast_fp16")]; tensor normed_677_axes_0 = const()[name = string("normed_677_axes_0"), val = tensor([-1])]; fp16 var_13334_to_fp16 = const()[name = string("op_13334_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_677_cast_fp16 = layer_norm(axes = normed_677_axes_0, epsilon = var_13334_to_fp16, x = input_695_cast_fp16)[name = string("normed_677_cast_fp16")]; tensor var_13347_split_sizes_0 = const()[name = string("op_13347_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13347_axis_0 = const()[name = string("op_13347_axis_0"), val = int32(-1)]; tensor var_13347_cast_fp16_0, tensor var_13347_cast_fp16_1 = split(axis = var_13347_axis_0, split_sizes = var_13347_split_sizes_0, x = normed_677_cast_fp16)[name = string("op_13347_cast_fp16")]; tensor const_405_to_fp16 = const()[name = string("const_405_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116008000)))]; tensor var_13350_cast_fp16 = mul(x = var_13347_cast_fp16_0, y = const_405_to_fp16)[name = string("op_13350_cast_fp16")]; tensor hidden_states_309_cast_fp16 = add(x = x_703_cast_fp16, y = var_13350_cast_fp16)[name = string("hidden_states_309_cast_fp16")]; tensor per_layer_slice_51_begin_0 = const()[name = string("per_layer_slice_51_begin_0"), val = tensor([0, 0, 6400])]; tensor per_layer_slice_51_end_0 = const()[name = string("per_layer_slice_51_end_0"), val = tensor([1, 1, 6656])]; tensor per_layer_slice_51_end_mask_0 = const()[name = string("per_layer_slice_51_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_51_cast_fp16 = slice_by_index(begin = per_layer_slice_51_begin_0, end = per_layer_slice_51_end_0, end_mask = per_layer_slice_51_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_51_cast_fp16")]; tensor gated_101 = linear(bias = linear_0_bias_0, weight = layers_25_per_layer_input_gate_weight_palettized, x = hidden_states_309_cast_fp16)[name = string("linear_50")]; string gated_103_mode_0 = const()[name = string("gated_103_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_103 = gelu(mode = gated_103_mode_0, x = gated_101)[name = string("gated_103")]; tensor input_699_cast_fp16 = mul(x = gated_103, y = per_layer_slice_51_cast_fp16)[name = string("input_699_cast_fp16")]; tensor layers_25_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116011136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116207808))))[name = string("layers_25_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_51_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_25_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_699_cast_fp16)[name = string("linear_51_cast_fp16")]; int32 var_13387 = const()[name = string("op_13387"), val = int32(-1)]; fp16 const_406_promoted_to_fp16 = const()[name = string("const_406_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13393_cast_fp16 = mul(x = linear_51_cast_fp16, y = const_406_promoted_to_fp16)[name = string("op_13393_cast_fp16")]; bool input_701_interleave_0 = const()[name = string("input_701_interleave_0"), val = bool(false)]; tensor input_701_cast_fp16 = concat(axis = var_13387, interleave = input_701_interleave_0, values = (linear_51_cast_fp16, var_13393_cast_fp16))[name = string("input_701_cast_fp16")]; tensor normed_681_axes_0 = const()[name = string("normed_681_axes_0"), val = tensor([-1])]; fp16 var_13385_to_fp16 = const()[name = string("op_13385_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_681_cast_fp16 = layer_norm(axes = normed_681_axes_0, epsilon = var_13385_to_fp16, x = input_701_cast_fp16)[name = string("normed_681_cast_fp16")]; tensor var_13398_split_sizes_0 = const()[name = string("op_13398_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13398_axis_0 = const()[name = string("op_13398_axis_0"), val = int32(-1)]; tensor var_13398_cast_fp16_0, tensor var_13398_cast_fp16_1 = split(axis = var_13398_axis_0, split_sizes = var_13398_split_sizes_0, x = normed_681_cast_fp16)[name = string("op_13398_cast_fp16")]; tensor const_407_to_fp16 = const()[name = string("const_407_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116209408)))]; tensor var_13401_cast_fp16 = mul(x = var_13398_cast_fp16_0, y = const_407_to_fp16)[name = string("op_13401_cast_fp16")]; tensor hidden_states_313_cast_fp16 = add(x = hidden_states_309_cast_fp16, y = var_13401_cast_fp16)[name = string("hidden_states_313_cast_fp16")]; tensor layers_25_layer_scalar_to_fp16 = const()[name = string("layers_25_layer_scalar_to_fp16"), val = tensor([0x1.92p-1])]; tensor x_715_cast_fp16 = mul(x = hidden_states_313_cast_fp16, y = layers_25_layer_scalar_to_fp16)[name = string("x_715_cast_fp16")]; int32 var_13409 = const()[name = string("op_13409"), val = int32(-1)]; fp16 const_408_promoted_to_fp16 = const()[name = string("const_408_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13415_cast_fp16 = mul(x = x_715_cast_fp16, y = const_408_promoted_to_fp16)[name = string("op_13415_cast_fp16")]; bool input_703_interleave_0 = const()[name = string("input_703_interleave_0"), val = bool(false)]; tensor input_703_cast_fp16 = concat(axis = var_13409, interleave = input_703_interleave_0, values = (x_715_cast_fp16, var_13415_cast_fp16))[name = string("input_703_cast_fp16")]; tensor normed_685_axes_0 = const()[name = string("normed_685_axes_0"), val = tensor([-1])]; fp16 var_13407_to_fp16 = const()[name = string("op_13407_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_685_cast_fp16 = layer_norm(axes = normed_685_axes_0, epsilon = var_13407_to_fp16, x = input_703_cast_fp16)[name = string("normed_685_cast_fp16")]; tensor var_13420_split_sizes_0 = const()[name = string("op_13420_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13420_axis_0 = const()[name = string("op_13420_axis_0"), val = int32(-1)]; tensor var_13420_cast_fp16_0, tensor var_13420_cast_fp16_1 = split(axis = var_13420_axis_0, split_sizes = var_13420_split_sizes_0, x = normed_685_cast_fp16)[name = string("op_13420_cast_fp16")]; tensor const_409_to_fp16 = const()[name = string("const_409_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116212544)))]; tensor var_13423_cast_fp16 = mul(x = var_13420_cast_fp16_0, y = const_409_to_fp16)[name = string("op_13423_cast_fp16")]; tensor var_13431 = const()[name = string("op_13431"), val = tensor([0, 2, 1])]; tensor var_13434_axes_0 = const()[name = string("op_13434_axes_0"), val = tensor([2])]; tensor var_13432_cast_fp16 = transpose(perm = var_13431, x = var_13423_cast_fp16)[name = string("transpose_64")]; tensor var_13434_cast_fp16 = expand_dims(axes = var_13434_axes_0, x = var_13432_cast_fp16)[name = string("op_13434_cast_fp16")]; string var_13450_pad_type_0 = const()[name = string("op_13450_pad_type_0"), val = string("valid")]; tensor var_13450_strides_0 = const()[name = string("op_13450_strides_0"), val = tensor([1, 1])]; tensor var_13450_pad_0 = const()[name = string("op_13450_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13450_dilations_0 = const()[name = string("op_13450_dilations_0"), val = tensor([1, 1])]; int32 var_13450_groups_0 = const()[name = string("op_13450_groups_0"), val = int32(1)]; tensor var_13450 = conv(dilations = var_13450_dilations_0, groups = var_13450_groups_0, pad = var_13450_pad_0, pad_type = var_13450_pad_type_0, strides = var_13450_strides_0, weight = layers_26_self_attn_q_proj_weight_palettized, x = var_13434_cast_fp16)[name = string("op_13450")]; tensor var_13455 = const()[name = string("op_13455"), val = tensor([1, 8, 256, 1])]; tensor var_13456 = reshape(shape = var_13455, x = var_13450)[name = string("op_13456")]; tensor var_13461 = const()[name = string("op_13461"), val = tensor([0, 1, 3, 2])]; tensor var_13471 = const()[name = string("op_13471"), val = tensor([1, 8, 256])]; tensor var_13462 = transpose(perm = var_13461, x = var_13456)[name = string("transpose_63")]; tensor x_719 = reshape(shape = var_13471, x = var_13462)[name = string("x_719")]; int32 var_13477 = const()[name = string("op_13477"), val = int32(-1)]; fp16 const_410_promoted_to_fp16 = const()[name = string("const_410_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13483_cast_fp16 = mul(x = x_719, y = const_410_promoted_to_fp16)[name = string("op_13483_cast_fp16")]; bool input_707_interleave_0 = const()[name = string("input_707_interleave_0"), val = bool(false)]; tensor input_707_cast_fp16 = concat(axis = var_13477, interleave = input_707_interleave_0, values = (x_719, var_13483_cast_fp16))[name = string("input_707_cast_fp16")]; tensor normed_689_axes_0 = const()[name = string("normed_689_axes_0"), val = tensor([-1])]; fp16 var_13475_to_fp16 = const()[name = string("op_13475_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_689_cast_fp16 = layer_norm(axes = normed_689_axes_0, epsilon = var_13475_to_fp16, x = input_707_cast_fp16)[name = string("normed_689_cast_fp16")]; tensor var_13488_split_sizes_0 = const()[name = string("op_13488_split_sizes_0"), val = tensor([256, 256])]; int32 var_13488_axis_0 = const()[name = string("op_13488_axis_0"), val = int32(-1)]; tensor var_13488_cast_fp16_0, tensor var_13488_cast_fp16_1 = split(axis = var_13488_axis_0, split_sizes = var_13488_split_sizes_0, x = normed_689_cast_fp16)[name = string("op_13488_cast_fp16")]; tensor var_13491_cast_fp16 = mul(x = var_13488_cast_fp16_0, y = const_234_to_fp16)[name = string("op_13491_cast_fp16")]; tensor var_13497 = const()[name = string("op_13497"), val = tensor([1, 8, 1, 256])]; tensor q_189 = reshape(shape = var_13497, x = var_13491_cast_fp16)[name = string("q_189")]; tensor var_13499 = mul(x = q_189, y = cos_1)[name = string("op_13499")]; tensor var_13500_split_sizes_0 = const()[name = string("op_13500_split_sizes_0"), val = tensor([128, 128])]; int32 var_13500_axis_0 = const()[name = string("op_13500_axis_0"), val = int32(-1)]; tensor var_13500_0, tensor var_13500_1 = split(axis = var_13500_axis_0, split_sizes = var_13500_split_sizes_0, x = q_189)[name = string("op_13500")]; fp16 const_412_promoted = const()[name = string("const_412_promoted"), val = fp16(-0x1p+0)]; tensor var_13502 = mul(x = var_13500_1, y = const_412_promoted)[name = string("op_13502")]; int32 var_13504 = const()[name = string("op_13504"), val = int32(-1)]; bool var_13505_interleave_0 = const()[name = string("op_13505_interleave_0"), val = bool(false)]; tensor var_13505 = concat(axis = var_13504, interleave = var_13505_interleave_0, values = (var_13502, var_13500_0))[name = string("op_13505")]; tensor var_13506 = mul(x = var_13505, y = sin_1)[name = string("op_13506")]; tensor q_191 = add(x = var_13499, y = var_13506)[name = string("q_191")]; bool var_13520_transpose_x_0 = const()[name = string("op_13520_transpose_x_0"), val = bool(false)]; bool var_13520_transpose_y_0 = const()[name = string("op_13520_transpose_y_0"), val = bool(false)]; tensor var_13520_cast_fp16 = matmul(transpose_x = var_13520_transpose_x_0, transpose_y = var_13520_transpose_y_0, x = q_191, y = transpose_153_cast_fp16)[name = string("op_13520_cast_fp16")]; tensor attn_weights_159_cast_fp16 = add(x = var_13520_cast_fp16, y = causal_mask)[name = string("attn_weights_159_cast_fp16")]; int32 var_13525 = const()[name = string("op_13525"), val = int32(-1)]; tensor attn_weights_161_cast_fp16 = softmax(axis = var_13525, x = attn_weights_159_cast_fp16)[name = string("attn_weights_161_cast_fp16")]; bool attn_output_157_transpose_x_0 = const()[name = string("attn_output_157_transpose_x_0"), val = bool(false)]; bool attn_output_157_transpose_y_0 = const()[name = string("attn_output_157_transpose_y_0"), val = bool(false)]; tensor attn_output_157_cast_fp16 = matmul(transpose_x = attn_output_157_transpose_x_0, transpose_y = attn_output_157_transpose_y_0, x = attn_weights_161_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_157_cast_fp16")]; tensor var_13533 = const()[name = string("op_13533"), val = tensor([0, 2, 1, 3])]; tensor var_13540 = const()[name = string("op_13540"), val = tensor([1, 1, -1])]; tensor var_13534_cast_fp16 = transpose(perm = var_13533, x = attn_output_157_cast_fp16)[name = string("transpose_62")]; tensor attn_output_159_cast_fp16 = reshape(shape = var_13540, x = var_13534_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor var_13545 = const()[name = string("op_13545"), val = tensor([0, 2, 1])]; string var_13561_pad_type_0 = const()[name = string("op_13561_pad_type_0"), val = string("valid")]; int32 var_13561_groups_0 = const()[name = string("op_13561_groups_0"), val = int32(1)]; tensor var_13561_strides_0 = const()[name = string("op_13561_strides_0"), val = tensor([1])]; tensor var_13561_pad_0 = const()[name = string("op_13561_pad_0"), val = tensor([0, 0])]; tensor var_13561_dilations_0 = const()[name = string("op_13561_dilations_0"), val = tensor([1])]; tensor squeeze_26_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1116215680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117788608))))[name = string("squeeze_26_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_13546_cast_fp16 = transpose(perm = var_13545, x = attn_output_159_cast_fp16)[name = string("transpose_61")]; tensor var_13561_cast_fp16 = conv(dilations = var_13561_dilations_0, groups = var_13561_groups_0, pad = var_13561_pad_0, pad_type = var_13561_pad_type_0, strides = var_13561_strides_0, weight = squeeze_26_cast_fp16_to_fp32_to_fp16_palettized, x = var_13546_cast_fp16)[name = string("op_13561_cast_fp16")]; tensor var_13565 = const()[name = string("op_13565"), val = tensor([0, 2, 1])]; int32 var_13571 = const()[name = string("op_13571"), val = int32(-1)]; fp16 const_413_promoted_to_fp16 = const()[name = string("const_413_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_723_cast_fp16 = transpose(perm = var_13565, x = var_13561_cast_fp16)[name = string("transpose_60")]; tensor var_13577_cast_fp16 = mul(x = x_723_cast_fp16, y = const_413_promoted_to_fp16)[name = string("op_13577_cast_fp16")]; bool input_711_interleave_0 = const()[name = string("input_711_interleave_0"), val = bool(false)]; tensor input_711_cast_fp16 = concat(axis = var_13571, interleave = input_711_interleave_0, values = (x_723_cast_fp16, var_13577_cast_fp16))[name = string("input_711_cast_fp16")]; tensor normed_693_axes_0 = const()[name = string("normed_693_axes_0"), val = tensor([-1])]; fp16 var_13569_to_fp16 = const()[name = string("op_13569_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_693_cast_fp16 = layer_norm(axes = normed_693_axes_0, epsilon = var_13569_to_fp16, x = input_711_cast_fp16)[name = string("normed_693_cast_fp16")]; tensor var_13582_split_sizes_0 = const()[name = string("op_13582_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13582_axis_0 = const()[name = string("op_13582_axis_0"), val = int32(-1)]; tensor var_13582_cast_fp16_0, tensor var_13582_cast_fp16_1 = split(axis = var_13582_axis_0, split_sizes = var_13582_split_sizes_0, x = normed_693_cast_fp16)[name = string("op_13582_cast_fp16")]; tensor const_414_to_fp16 = const()[name = string("const_414_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117790208)))]; tensor var_13585_cast_fp16 = mul(x = var_13582_cast_fp16_0, y = const_414_to_fp16)[name = string("op_13585_cast_fp16")]; tensor x_727_cast_fp16 = add(x = x_715_cast_fp16, y = var_13585_cast_fp16)[name = string("x_727_cast_fp16")]; int32 var_13592 = const()[name = string("op_13592"), val = int32(-1)]; fp16 const_415_promoted_to_fp16 = const()[name = string("const_415_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13598_cast_fp16 = mul(x = x_727_cast_fp16, y = const_415_promoted_to_fp16)[name = string("op_13598_cast_fp16")]; bool input_713_interleave_0 = const()[name = string("input_713_interleave_0"), val = bool(false)]; tensor input_713_cast_fp16 = concat(axis = var_13592, interleave = input_713_interleave_0, values = (x_727_cast_fp16, var_13598_cast_fp16))[name = string("input_713_cast_fp16")]; tensor normed_697_axes_0 = const()[name = string("normed_697_axes_0"), val = tensor([-1])]; fp16 var_13590_to_fp16 = const()[name = string("op_13590_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_697_cast_fp16 = layer_norm(axes = normed_697_axes_0, epsilon = var_13590_to_fp16, x = input_713_cast_fp16)[name = string("normed_697_cast_fp16")]; tensor var_13603_split_sizes_0 = const()[name = string("op_13603_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13603_axis_0 = const()[name = string("op_13603_axis_0"), val = int32(-1)]; tensor var_13603_cast_fp16_0, tensor var_13603_cast_fp16_1 = split(axis = var_13603_axis_0, split_sizes = var_13603_split_sizes_0, x = normed_697_cast_fp16)[name = string("op_13603_cast_fp16")]; tensor const_416_to_fp16 = const()[name = string("const_416_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117793344)))]; tensor var_13606_cast_fp16 = mul(x = var_13603_cast_fp16_0, y = const_416_to_fp16)[name = string("op_13606_cast_fp16")]; tensor var_13619 = const()[name = string("op_13619"), val = tensor([0, 2, 1])]; tensor input_715_axes_0 = const()[name = string("input_715_axes_0"), val = tensor([2])]; tensor var_13620 = transpose(perm = var_13619, x = var_13606_cast_fp16)[name = string("transpose_59")]; tensor input_715 = expand_dims(axes = input_715_axes_0, x = var_13620)[name = string("input_715")]; string gate_105_pad_type_0 = const()[name = string("gate_105_pad_type_0"), val = string("valid")]; tensor gate_105_strides_0 = const()[name = string("gate_105_strides_0"), val = tensor([1, 1])]; tensor gate_105_pad_0 = const()[name = string("gate_105_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_105_dilations_0 = const()[name = string("gate_105_dilations_0"), val = tensor([1, 1])]; int32 gate_105_groups_0 = const()[name = string("gate_105_groups_0"), val = int32(1)]; tensor gate_105 = conv(dilations = gate_105_dilations_0, groups = gate_105_groups_0, pad = gate_105_pad_0, pad_type = gate_105_pad_type_0, strides = gate_105_strides_0, weight = layers_26_mlp_gate_proj_weight_palettized, x = input_715)[name = string("gate_105")]; string up_53_pad_type_0 = const()[name = string("up_53_pad_type_0"), val = string("valid")]; tensor up_53_strides_0 = const()[name = string("up_53_strides_0"), val = tensor([1, 1])]; tensor up_53_pad_0 = const()[name = string("up_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_53_dilations_0 = const()[name = string("up_53_dilations_0"), val = tensor([1, 1])]; int32 up_53_groups_0 = const()[name = string("up_53_groups_0"), val = int32(1)]; tensor up_53 = conv(dilations = up_53_dilations_0, groups = up_53_groups_0, pad = up_53_pad_0, pad_type = up_53_pad_type_0, strides = up_53_strides_0, weight = layers_26_mlp_up_proj_weight_palettized, x = input_715)[name = string("up_53")]; string gate_107_mode_0 = const()[name = string("gate_107_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_107 = gelu(mode = gate_107_mode_0, x = gate_105)[name = string("gate_107")]; tensor input_717 = mul(x = gate_107, y = up_53)[name = string("input_717")]; string mlp_out_53_pad_type_0 = const()[name = string("mlp_out_53_pad_type_0"), val = string("valid")]; tensor mlp_out_53_strides_0 = const()[name = string("mlp_out_53_strides_0"), val = tensor([1, 1])]; tensor mlp_out_53_pad_0 = const()[name = string("mlp_out_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_53_dilations_0 = const()[name = string("mlp_out_53_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_53_groups_0 = const()[name = string("mlp_out_53_groups_0"), val = int32(1)]; tensor mlp_out_53 = conv(dilations = mlp_out_53_dilations_0, groups = mlp_out_53_groups_0, pad = mlp_out_53_pad_0, pad_type = mlp_out_53_pad_type_0, strides = mlp_out_53_strides_0, weight = layers_26_mlp_down_proj_weight_palettized, x = input_717)[name = string("mlp_out_53")]; tensor var_13660_axes_0 = const()[name = string("op_13660_axes_0"), val = tensor([2])]; tensor var_13660 = squeeze(axes = var_13660_axes_0, x = mlp_out_53)[name = string("op_13660")]; tensor var_13664 = const()[name = string("op_13664"), val = tensor([0, 2, 1])]; int32 var_13670 = const()[name = string("op_13670"), val = int32(-1)]; fp16 const_417_promoted_to_fp16 = const()[name = string("const_417_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_731 = transpose(perm = var_13664, x = var_13660)[name = string("transpose_58")]; tensor var_13676_cast_fp16 = mul(x = x_731, y = const_417_promoted_to_fp16)[name = string("op_13676_cast_fp16")]; bool input_719_interleave_0 = const()[name = string("input_719_interleave_0"), val = bool(false)]; tensor input_719_cast_fp16 = concat(axis = var_13670, interleave = input_719_interleave_0, values = (x_731, var_13676_cast_fp16))[name = string("input_719_cast_fp16")]; tensor normed_701_axes_0 = const()[name = string("normed_701_axes_0"), val = tensor([-1])]; fp16 var_13668_to_fp16 = const()[name = string("op_13668_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_701_cast_fp16 = layer_norm(axes = normed_701_axes_0, epsilon = var_13668_to_fp16, x = input_719_cast_fp16)[name = string("normed_701_cast_fp16")]; tensor var_13681_split_sizes_0 = const()[name = string("op_13681_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13681_axis_0 = const()[name = string("op_13681_axis_0"), val = int32(-1)]; tensor var_13681_cast_fp16_0, tensor var_13681_cast_fp16_1 = split(axis = var_13681_axis_0, split_sizes = var_13681_split_sizes_0, x = normed_701_cast_fp16)[name = string("op_13681_cast_fp16")]; tensor const_418_to_fp16 = const()[name = string("const_418_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117796480)))]; tensor var_13684_cast_fp16 = mul(x = var_13681_cast_fp16_0, y = const_418_to_fp16)[name = string("op_13684_cast_fp16")]; tensor hidden_states_321_cast_fp16 = add(x = x_727_cast_fp16, y = var_13684_cast_fp16)[name = string("hidden_states_321_cast_fp16")]; tensor per_layer_slice_53_begin_0 = const()[name = string("per_layer_slice_53_begin_0"), val = tensor([0, 0, 6656])]; tensor per_layer_slice_53_end_0 = const()[name = string("per_layer_slice_53_end_0"), val = tensor([1, 1, 6912])]; tensor per_layer_slice_53_end_mask_0 = const()[name = string("per_layer_slice_53_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_53_cast_fp16 = slice_by_index(begin = per_layer_slice_53_begin_0, end = per_layer_slice_53_end_0, end_mask = per_layer_slice_53_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_53_cast_fp16")]; tensor gated_105 = linear(bias = linear_0_bias_0, weight = layers_26_per_layer_input_gate_weight_palettized, x = hidden_states_321_cast_fp16)[name = string("linear_52")]; string gated_107_mode_0 = const()[name = string("gated_107_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_107 = gelu(mode = gated_107_mode_0, x = gated_105)[name = string("gated_107")]; tensor input_723_cast_fp16 = mul(x = gated_107, y = per_layer_slice_53_cast_fp16)[name = string("input_723_cast_fp16")]; tensor layers_26_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117799616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117996288))))[name = string("layers_26_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_53_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_26_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_723_cast_fp16)[name = string("linear_53_cast_fp16")]; int32 var_13721 = const()[name = string("op_13721"), val = int32(-1)]; fp16 const_419_promoted_to_fp16 = const()[name = string("const_419_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13727_cast_fp16 = mul(x = linear_53_cast_fp16, y = const_419_promoted_to_fp16)[name = string("op_13727_cast_fp16")]; bool input_725_interleave_0 = const()[name = string("input_725_interleave_0"), val = bool(false)]; tensor input_725_cast_fp16 = concat(axis = var_13721, interleave = input_725_interleave_0, values = (linear_53_cast_fp16, var_13727_cast_fp16))[name = string("input_725_cast_fp16")]; tensor normed_705_axes_0 = const()[name = string("normed_705_axes_0"), val = tensor([-1])]; fp16 var_13719_to_fp16 = const()[name = string("op_13719_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_705_cast_fp16 = layer_norm(axes = normed_705_axes_0, epsilon = var_13719_to_fp16, x = input_725_cast_fp16)[name = string("normed_705_cast_fp16")]; tensor var_13732_split_sizes_0 = const()[name = string("op_13732_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13732_axis_0 = const()[name = string("op_13732_axis_0"), val = int32(-1)]; tensor var_13732_cast_fp16_0, tensor var_13732_cast_fp16_1 = split(axis = var_13732_axis_0, split_sizes = var_13732_split_sizes_0, x = normed_705_cast_fp16)[name = string("op_13732_cast_fp16")]; tensor const_420_to_fp16 = const()[name = string("const_420_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1117997888)))]; tensor var_13735_cast_fp16 = mul(x = var_13732_cast_fp16_0, y = const_420_to_fp16)[name = string("op_13735_cast_fp16")]; tensor hidden_states_325_cast_fp16 = add(x = hidden_states_321_cast_fp16, y = var_13735_cast_fp16)[name = string("hidden_states_325_cast_fp16")]; tensor layers_26_layer_scalar_to_fp16 = const()[name = string("layers_26_layer_scalar_to_fp16"), val = tensor([0x1.a6p-1])]; tensor x_739_cast_fp16 = mul(x = hidden_states_325_cast_fp16, y = layers_26_layer_scalar_to_fp16)[name = string("x_739_cast_fp16")]; int32 var_13743 = const()[name = string("op_13743"), val = int32(-1)]; fp16 const_421_promoted_to_fp16 = const()[name = string("const_421_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13749_cast_fp16 = mul(x = x_739_cast_fp16, y = const_421_promoted_to_fp16)[name = string("op_13749_cast_fp16")]; bool input_727_interleave_0 = const()[name = string("input_727_interleave_0"), val = bool(false)]; tensor input_727_cast_fp16 = concat(axis = var_13743, interleave = input_727_interleave_0, values = (x_739_cast_fp16, var_13749_cast_fp16))[name = string("input_727_cast_fp16")]; tensor normed_709_axes_0 = const()[name = string("normed_709_axes_0"), val = tensor([-1])]; fp16 var_13741_to_fp16 = const()[name = string("op_13741_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_709_cast_fp16 = layer_norm(axes = normed_709_axes_0, epsilon = var_13741_to_fp16, x = input_727_cast_fp16)[name = string("normed_709_cast_fp16")]; tensor var_13754_split_sizes_0 = const()[name = string("op_13754_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13754_axis_0 = const()[name = string("op_13754_axis_0"), val = int32(-1)]; tensor var_13754_cast_fp16_0, tensor var_13754_cast_fp16_1 = split(axis = var_13754_axis_0, split_sizes = var_13754_split_sizes_0, x = normed_709_cast_fp16)[name = string("op_13754_cast_fp16")]; tensor const_422_to_fp16 = const()[name = string("const_422_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1118001024)))]; tensor var_13757_cast_fp16 = mul(x = var_13754_cast_fp16_0, y = const_422_to_fp16)[name = string("op_13757_cast_fp16")]; tensor var_13765 = const()[name = string("op_13765"), val = tensor([0, 2, 1])]; tensor var_13768_axes_0 = const()[name = string("op_13768_axes_0"), val = tensor([2])]; tensor var_13766_cast_fp16 = transpose(perm = var_13765, x = var_13757_cast_fp16)[name = string("transpose_57")]; tensor var_13768_cast_fp16 = expand_dims(axes = var_13768_axes_0, x = var_13766_cast_fp16)[name = string("op_13768_cast_fp16")]; string var_13784_pad_type_0 = const()[name = string("op_13784_pad_type_0"), val = string("valid")]; tensor var_13784_strides_0 = const()[name = string("op_13784_strides_0"), val = tensor([1, 1])]; tensor var_13784_pad_0 = const()[name = string("op_13784_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13784_dilations_0 = const()[name = string("op_13784_dilations_0"), val = tensor([1, 1])]; int32 var_13784_groups_0 = const()[name = string("op_13784_groups_0"), val = int32(1)]; tensor var_13784 = conv(dilations = var_13784_dilations_0, groups = var_13784_groups_0, pad = var_13784_pad_0, pad_type = var_13784_pad_type_0, strides = var_13784_strides_0, weight = layers_27_self_attn_q_proj_weight_palettized, x = var_13768_cast_fp16)[name = string("op_13784")]; tensor var_13789 = const()[name = string("op_13789"), val = tensor([1, 8, 256, 1])]; tensor var_13790 = reshape(shape = var_13789, x = var_13784)[name = string("op_13790")]; tensor var_13795 = const()[name = string("op_13795"), val = tensor([0, 1, 3, 2])]; tensor var_13805 = const()[name = string("op_13805"), val = tensor([1, 8, 256])]; tensor var_13796 = transpose(perm = var_13795, x = var_13790)[name = string("transpose_56")]; tensor x_743 = reshape(shape = var_13805, x = var_13796)[name = string("x_743")]; int32 var_13811 = const()[name = string("op_13811"), val = int32(-1)]; fp16 const_423_promoted_to_fp16 = const()[name = string("const_423_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13817_cast_fp16 = mul(x = x_743, y = const_423_promoted_to_fp16)[name = string("op_13817_cast_fp16")]; bool input_731_interleave_0 = const()[name = string("input_731_interleave_0"), val = bool(false)]; tensor input_731_cast_fp16 = concat(axis = var_13811, interleave = input_731_interleave_0, values = (x_743, var_13817_cast_fp16))[name = string("input_731_cast_fp16")]; tensor normed_713_axes_0 = const()[name = string("normed_713_axes_0"), val = tensor([-1])]; fp16 var_13809_to_fp16 = const()[name = string("op_13809_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_713_cast_fp16 = layer_norm(axes = normed_713_axes_0, epsilon = var_13809_to_fp16, x = input_731_cast_fp16)[name = string("normed_713_cast_fp16")]; tensor var_13822_split_sizes_0 = const()[name = string("op_13822_split_sizes_0"), val = tensor([256, 256])]; int32 var_13822_axis_0 = const()[name = string("op_13822_axis_0"), val = int32(-1)]; tensor var_13822_cast_fp16_0, tensor var_13822_cast_fp16_1 = split(axis = var_13822_axis_0, split_sizes = var_13822_split_sizes_0, x = normed_713_cast_fp16)[name = string("op_13822_cast_fp16")]; tensor var_13825_cast_fp16 = mul(x = var_13822_cast_fp16_0, y = const_234_to_fp16)[name = string("op_13825_cast_fp16")]; tensor var_13831 = const()[name = string("op_13831"), val = tensor([1, 8, 1, 256])]; tensor q_195 = reshape(shape = var_13831, x = var_13825_cast_fp16)[name = string("q_195")]; tensor var_13833 = mul(x = q_195, y = cos_1)[name = string("op_13833")]; tensor var_13834_split_sizes_0 = const()[name = string("op_13834_split_sizes_0"), val = tensor([128, 128])]; int32 var_13834_axis_0 = const()[name = string("op_13834_axis_0"), val = int32(-1)]; tensor var_13834_0, tensor var_13834_1 = split(axis = var_13834_axis_0, split_sizes = var_13834_split_sizes_0, x = q_195)[name = string("op_13834")]; fp16 const_425_promoted = const()[name = string("const_425_promoted"), val = fp16(-0x1p+0)]; tensor var_13836 = mul(x = var_13834_1, y = const_425_promoted)[name = string("op_13836")]; int32 var_13838 = const()[name = string("op_13838"), val = int32(-1)]; bool var_13839_interleave_0 = const()[name = string("op_13839_interleave_0"), val = bool(false)]; tensor var_13839 = concat(axis = var_13838, interleave = var_13839_interleave_0, values = (var_13836, var_13834_0))[name = string("op_13839")]; tensor var_13840 = mul(x = var_13839, y = sin_1)[name = string("op_13840")]; tensor q_197 = add(x = var_13833, y = var_13840)[name = string("q_197")]; bool var_13854_transpose_x_0 = const()[name = string("op_13854_transpose_x_0"), val = bool(false)]; bool var_13854_transpose_y_0 = const()[name = string("op_13854_transpose_y_0"), val = bool(false)]; tensor var_13854_cast_fp16 = matmul(transpose_x = var_13854_transpose_x_0, transpose_y = var_13854_transpose_y_0, x = q_197, y = transpose_153_cast_fp16)[name = string("op_13854_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = var_13854_cast_fp16, y = causal_mask)[name = string("attn_weights_165_cast_fp16")]; int32 var_13859 = const()[name = string("op_13859"), val = int32(-1)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_13859, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool attn_output_163_transpose_x_0 = const()[name = string("attn_output_163_transpose_x_0"), val = bool(false)]; bool attn_output_163_transpose_y_0 = const()[name = string("attn_output_163_transpose_y_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = matmul(transpose_x = attn_output_163_transpose_x_0, transpose_y = attn_output_163_transpose_y_0, x = attn_weights_167_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_163_cast_fp16")]; tensor var_13867 = const()[name = string("op_13867"), val = tensor([0, 2, 1, 3])]; tensor var_13874 = const()[name = string("op_13874"), val = tensor([1, 1, -1])]; tensor var_13868_cast_fp16 = transpose(perm = var_13867, x = attn_output_163_cast_fp16)[name = string("transpose_55")]; tensor attn_output_165_cast_fp16 = reshape(shape = var_13874, x = var_13868_cast_fp16)[name = string("attn_output_165_cast_fp16")]; tensor var_13879 = const()[name = string("op_13879"), val = tensor([0, 2, 1])]; string var_13895_pad_type_0 = const()[name = string("op_13895_pad_type_0"), val = string("valid")]; int32 var_13895_groups_0 = const()[name = string("op_13895_groups_0"), val = int32(1)]; tensor var_13895_strides_0 = const()[name = string("op_13895_strides_0"), val = tensor([1])]; tensor var_13895_pad_0 = const()[name = string("op_13895_pad_0"), val = tensor([0, 0])]; tensor var_13895_dilations_0 = const()[name = string("op_13895_dilations_0"), val = tensor([1])]; tensor squeeze_27_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1118004160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119577088))))[name = string("squeeze_27_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_13880_cast_fp16 = transpose(perm = var_13879, x = attn_output_165_cast_fp16)[name = string("transpose_54")]; tensor var_13895_cast_fp16 = conv(dilations = var_13895_dilations_0, groups = var_13895_groups_0, pad = var_13895_pad_0, pad_type = var_13895_pad_type_0, strides = var_13895_strides_0, weight = squeeze_27_cast_fp16_to_fp32_to_fp16_palettized, x = var_13880_cast_fp16)[name = string("op_13895_cast_fp16")]; tensor var_13899 = const()[name = string("op_13899"), val = tensor([0, 2, 1])]; int32 var_13905 = const()[name = string("op_13905"), val = int32(-1)]; fp16 const_426_promoted_to_fp16 = const()[name = string("const_426_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_747_cast_fp16 = transpose(perm = var_13899, x = var_13895_cast_fp16)[name = string("transpose_53")]; tensor var_13911_cast_fp16 = mul(x = x_747_cast_fp16, y = const_426_promoted_to_fp16)[name = string("op_13911_cast_fp16")]; bool input_735_interleave_0 = const()[name = string("input_735_interleave_0"), val = bool(false)]; tensor input_735_cast_fp16 = concat(axis = var_13905, interleave = input_735_interleave_0, values = (x_747_cast_fp16, var_13911_cast_fp16))[name = string("input_735_cast_fp16")]; tensor normed_717_axes_0 = const()[name = string("normed_717_axes_0"), val = tensor([-1])]; fp16 var_13903_to_fp16 = const()[name = string("op_13903_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_717_cast_fp16 = layer_norm(axes = normed_717_axes_0, epsilon = var_13903_to_fp16, x = input_735_cast_fp16)[name = string("normed_717_cast_fp16")]; tensor var_13916_split_sizes_0 = const()[name = string("op_13916_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13916_axis_0 = const()[name = string("op_13916_axis_0"), val = int32(-1)]; tensor var_13916_cast_fp16_0, tensor var_13916_cast_fp16_1 = split(axis = var_13916_axis_0, split_sizes = var_13916_split_sizes_0, x = normed_717_cast_fp16)[name = string("op_13916_cast_fp16")]; tensor const_427_to_fp16 = const()[name = string("const_427_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119578688)))]; tensor var_13919_cast_fp16 = mul(x = var_13916_cast_fp16_0, y = const_427_to_fp16)[name = string("op_13919_cast_fp16")]; tensor x_751_cast_fp16 = add(x = x_739_cast_fp16, y = var_13919_cast_fp16)[name = string("x_751_cast_fp16")]; int32 var_13926 = const()[name = string("op_13926"), val = int32(-1)]; fp16 const_428_promoted_to_fp16 = const()[name = string("const_428_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13932_cast_fp16 = mul(x = x_751_cast_fp16, y = const_428_promoted_to_fp16)[name = string("op_13932_cast_fp16")]; bool input_737_interleave_0 = const()[name = string("input_737_interleave_0"), val = bool(false)]; tensor input_737_cast_fp16 = concat(axis = var_13926, interleave = input_737_interleave_0, values = (x_751_cast_fp16, var_13932_cast_fp16))[name = string("input_737_cast_fp16")]; tensor normed_721_axes_0 = const()[name = string("normed_721_axes_0"), val = tensor([-1])]; fp16 var_13924_to_fp16 = const()[name = string("op_13924_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_721_cast_fp16 = layer_norm(axes = normed_721_axes_0, epsilon = var_13924_to_fp16, x = input_737_cast_fp16)[name = string("normed_721_cast_fp16")]; tensor var_13937_split_sizes_0 = const()[name = string("op_13937_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_13937_axis_0 = const()[name = string("op_13937_axis_0"), val = int32(-1)]; tensor var_13937_cast_fp16_0, tensor var_13937_cast_fp16_1 = split(axis = var_13937_axis_0, split_sizes = var_13937_split_sizes_0, x = normed_721_cast_fp16)[name = string("op_13937_cast_fp16")]; tensor const_429_to_fp16 = const()[name = string("const_429_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119581824)))]; tensor var_13940_cast_fp16 = mul(x = var_13937_cast_fp16_0, y = const_429_to_fp16)[name = string("op_13940_cast_fp16")]; tensor var_13953 = const()[name = string("op_13953"), val = tensor([0, 2, 1])]; tensor input_739_axes_0 = const()[name = string("input_739_axes_0"), val = tensor([2])]; tensor var_13954 = transpose(perm = var_13953, x = var_13940_cast_fp16)[name = string("transpose_52")]; tensor input_739 = expand_dims(axes = input_739_axes_0, x = var_13954)[name = string("input_739")]; string gate_109_pad_type_0 = const()[name = string("gate_109_pad_type_0"), val = string("valid")]; tensor gate_109_strides_0 = const()[name = string("gate_109_strides_0"), val = tensor([1, 1])]; tensor gate_109_pad_0 = const()[name = string("gate_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_109_dilations_0 = const()[name = string("gate_109_dilations_0"), val = tensor([1, 1])]; int32 gate_109_groups_0 = const()[name = string("gate_109_groups_0"), val = int32(1)]; tensor gate_109 = conv(dilations = gate_109_dilations_0, groups = gate_109_groups_0, pad = gate_109_pad_0, pad_type = gate_109_pad_type_0, strides = gate_109_strides_0, weight = layers_27_mlp_gate_proj_weight_palettized, x = input_739)[name = string("gate_109")]; string up_55_pad_type_0 = const()[name = string("up_55_pad_type_0"), val = string("valid")]; tensor up_55_strides_0 = const()[name = string("up_55_strides_0"), val = tensor([1, 1])]; tensor up_55_pad_0 = const()[name = string("up_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_55_dilations_0 = const()[name = string("up_55_dilations_0"), val = tensor([1, 1])]; int32 up_55_groups_0 = const()[name = string("up_55_groups_0"), val = int32(1)]; tensor up_55 = conv(dilations = up_55_dilations_0, groups = up_55_groups_0, pad = up_55_pad_0, pad_type = up_55_pad_type_0, strides = up_55_strides_0, weight = layers_27_mlp_up_proj_weight_palettized, x = input_739)[name = string("up_55")]; string gate_111_mode_0 = const()[name = string("gate_111_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_111 = gelu(mode = gate_111_mode_0, x = gate_109)[name = string("gate_111")]; tensor input_741 = mul(x = gate_111, y = up_55)[name = string("input_741")]; string mlp_out_55_pad_type_0 = const()[name = string("mlp_out_55_pad_type_0"), val = string("valid")]; tensor mlp_out_55_strides_0 = const()[name = string("mlp_out_55_strides_0"), val = tensor([1, 1])]; tensor mlp_out_55_pad_0 = const()[name = string("mlp_out_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_55_dilations_0 = const()[name = string("mlp_out_55_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_55_groups_0 = const()[name = string("mlp_out_55_groups_0"), val = int32(1)]; tensor mlp_out_55 = conv(dilations = mlp_out_55_dilations_0, groups = mlp_out_55_groups_0, pad = mlp_out_55_pad_0, pad_type = mlp_out_55_pad_type_0, strides = mlp_out_55_strides_0, weight = layers_27_mlp_down_proj_weight_palettized, x = input_741)[name = string("mlp_out_55")]; tensor var_13994_axes_0 = const()[name = string("op_13994_axes_0"), val = tensor([2])]; tensor var_13994 = squeeze(axes = var_13994_axes_0, x = mlp_out_55)[name = string("op_13994")]; tensor var_13998 = const()[name = string("op_13998"), val = tensor([0, 2, 1])]; int32 var_14004 = const()[name = string("op_14004"), val = int32(-1)]; fp16 const_430_promoted_to_fp16 = const()[name = string("const_430_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_755 = transpose(perm = var_13998, x = var_13994)[name = string("transpose_51")]; tensor var_14010_cast_fp16 = mul(x = x_755, y = const_430_promoted_to_fp16)[name = string("op_14010_cast_fp16")]; bool input_743_interleave_0 = const()[name = string("input_743_interleave_0"), val = bool(false)]; tensor input_743_cast_fp16 = concat(axis = var_14004, interleave = input_743_interleave_0, values = (x_755, var_14010_cast_fp16))[name = string("input_743_cast_fp16")]; tensor normed_725_axes_0 = const()[name = string("normed_725_axes_0"), val = tensor([-1])]; fp16 var_14002_to_fp16 = const()[name = string("op_14002_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_725_cast_fp16 = layer_norm(axes = normed_725_axes_0, epsilon = var_14002_to_fp16, x = input_743_cast_fp16)[name = string("normed_725_cast_fp16")]; tensor var_14015_split_sizes_0 = const()[name = string("op_14015_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14015_axis_0 = const()[name = string("op_14015_axis_0"), val = int32(-1)]; tensor var_14015_cast_fp16_0, tensor var_14015_cast_fp16_1 = split(axis = var_14015_axis_0, split_sizes = var_14015_split_sizes_0, x = normed_725_cast_fp16)[name = string("op_14015_cast_fp16")]; tensor const_431_to_fp16 = const()[name = string("const_431_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119584960)))]; tensor var_14018_cast_fp16 = mul(x = var_14015_cast_fp16_0, y = const_431_to_fp16)[name = string("op_14018_cast_fp16")]; tensor hidden_states_333_cast_fp16 = add(x = x_751_cast_fp16, y = var_14018_cast_fp16)[name = string("hidden_states_333_cast_fp16")]; tensor per_layer_slice_55_begin_0 = const()[name = string("per_layer_slice_55_begin_0"), val = tensor([0, 0, 6912])]; tensor per_layer_slice_55_end_0 = const()[name = string("per_layer_slice_55_end_0"), val = tensor([1, 1, 7168])]; tensor per_layer_slice_55_end_mask_0 = const()[name = string("per_layer_slice_55_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_55_cast_fp16 = slice_by_index(begin = per_layer_slice_55_begin_0, end = per_layer_slice_55_end_0, end_mask = per_layer_slice_55_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_55_cast_fp16")]; tensor gated_109 = linear(bias = linear_0_bias_0, weight = layers_27_per_layer_input_gate_weight_palettized, x = hidden_states_333_cast_fp16)[name = string("linear_54")]; string gated_111_mode_0 = const()[name = string("gated_111_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_111 = gelu(mode = gated_111_mode_0, x = gated_109)[name = string("gated_111")]; tensor input_747_cast_fp16 = mul(x = gated_111, y = per_layer_slice_55_cast_fp16)[name = string("input_747_cast_fp16")]; tensor layers_27_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119588096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119784768))))[name = string("layers_27_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_55_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_27_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_747_cast_fp16)[name = string("linear_55_cast_fp16")]; int32 var_14055 = const()[name = string("op_14055"), val = int32(-1)]; fp16 const_432_promoted_to_fp16 = const()[name = string("const_432_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14061_cast_fp16 = mul(x = linear_55_cast_fp16, y = const_432_promoted_to_fp16)[name = string("op_14061_cast_fp16")]; bool input_749_interleave_0 = const()[name = string("input_749_interleave_0"), val = bool(false)]; tensor input_749_cast_fp16 = concat(axis = var_14055, interleave = input_749_interleave_0, values = (linear_55_cast_fp16, var_14061_cast_fp16))[name = string("input_749_cast_fp16")]; tensor normed_729_axes_0 = const()[name = string("normed_729_axes_0"), val = tensor([-1])]; fp16 var_14053_to_fp16 = const()[name = string("op_14053_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_729_cast_fp16 = layer_norm(axes = normed_729_axes_0, epsilon = var_14053_to_fp16, x = input_749_cast_fp16)[name = string("normed_729_cast_fp16")]; tensor var_14066_split_sizes_0 = const()[name = string("op_14066_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14066_axis_0 = const()[name = string("op_14066_axis_0"), val = int32(-1)]; tensor var_14066_cast_fp16_0, tensor var_14066_cast_fp16_1 = split(axis = var_14066_axis_0, split_sizes = var_14066_split_sizes_0, x = normed_729_cast_fp16)[name = string("op_14066_cast_fp16")]; tensor const_433_to_fp16 = const()[name = string("const_433_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119786368)))]; tensor var_14069_cast_fp16 = mul(x = var_14066_cast_fp16_0, y = const_433_to_fp16)[name = string("op_14069_cast_fp16")]; tensor hidden_states_337_cast_fp16 = add(x = hidden_states_333_cast_fp16, y = var_14069_cast_fp16)[name = string("hidden_states_337_cast_fp16")]; tensor layers_27_layer_scalar_to_fp16 = const()[name = string("layers_27_layer_scalar_to_fp16"), val = tensor([0x1.a4p-1])]; tensor x_763_cast_fp16 = mul(x = hidden_states_337_cast_fp16, y = layers_27_layer_scalar_to_fp16)[name = string("x_763_cast_fp16")]; int32 var_14077 = const()[name = string("op_14077"), val = int32(-1)]; fp16 const_434_promoted_to_fp16 = const()[name = string("const_434_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14083_cast_fp16 = mul(x = x_763_cast_fp16, y = const_434_promoted_to_fp16)[name = string("op_14083_cast_fp16")]; bool input_751_interleave_0 = const()[name = string("input_751_interleave_0"), val = bool(false)]; tensor input_751_cast_fp16 = concat(axis = var_14077, interleave = input_751_interleave_0, values = (x_763_cast_fp16, var_14083_cast_fp16))[name = string("input_751_cast_fp16")]; tensor normed_733_axes_0 = const()[name = string("normed_733_axes_0"), val = tensor([-1])]; fp16 var_14075_to_fp16 = const()[name = string("op_14075_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_733_cast_fp16 = layer_norm(axes = normed_733_axes_0, epsilon = var_14075_to_fp16, x = input_751_cast_fp16)[name = string("normed_733_cast_fp16")]; tensor var_14088_split_sizes_0 = const()[name = string("op_14088_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14088_axis_0 = const()[name = string("op_14088_axis_0"), val = int32(-1)]; tensor var_14088_cast_fp16_0, tensor var_14088_cast_fp16_1 = split(axis = var_14088_axis_0, split_sizes = var_14088_split_sizes_0, x = normed_733_cast_fp16)[name = string("op_14088_cast_fp16")]; tensor const_435_to_fp16 = const()[name = string("const_435_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119789504)))]; tensor var_14091_cast_fp16 = mul(x = var_14088_cast_fp16_0, y = const_435_to_fp16)[name = string("op_14091_cast_fp16")]; tensor var_14099 = const()[name = string("op_14099"), val = tensor([0, 2, 1])]; tensor var_14102_axes_0 = const()[name = string("op_14102_axes_0"), val = tensor([2])]; tensor var_14100_cast_fp16 = transpose(perm = var_14099, x = var_14091_cast_fp16)[name = string("transpose_50")]; tensor var_14102_cast_fp16 = expand_dims(axes = var_14102_axes_0, x = var_14100_cast_fp16)[name = string("op_14102_cast_fp16")]; string var_14118_pad_type_0 = const()[name = string("op_14118_pad_type_0"), val = string("valid")]; tensor var_14118_strides_0 = const()[name = string("op_14118_strides_0"), val = tensor([1, 1])]; tensor var_14118_pad_0 = const()[name = string("op_14118_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_14118_dilations_0 = const()[name = string("op_14118_dilations_0"), val = tensor([1, 1])]; int32 var_14118_groups_0 = const()[name = string("op_14118_groups_0"), val = int32(1)]; tensor var_14118 = conv(dilations = var_14118_dilations_0, groups = var_14118_groups_0, pad = var_14118_pad_0, pad_type = var_14118_pad_type_0, strides = var_14118_strides_0, weight = layers_28_self_attn_q_proj_weight_palettized, x = var_14102_cast_fp16)[name = string("op_14118")]; tensor var_14123 = const()[name = string("op_14123"), val = tensor([1, 8, 256, 1])]; tensor var_14124 = reshape(shape = var_14123, x = var_14118)[name = string("op_14124")]; tensor var_14129 = const()[name = string("op_14129"), val = tensor([0, 1, 3, 2])]; tensor var_14139 = const()[name = string("op_14139"), val = tensor([1, 8, 256])]; tensor var_14130 = transpose(perm = var_14129, x = var_14124)[name = string("transpose_49")]; tensor x_767 = reshape(shape = var_14139, x = var_14130)[name = string("x_767")]; int32 var_14145 = const()[name = string("op_14145"), val = int32(-1)]; fp16 const_436_promoted_to_fp16 = const()[name = string("const_436_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14151_cast_fp16 = mul(x = x_767, y = const_436_promoted_to_fp16)[name = string("op_14151_cast_fp16")]; bool input_755_interleave_0 = const()[name = string("input_755_interleave_0"), val = bool(false)]; tensor input_755_cast_fp16 = concat(axis = var_14145, interleave = input_755_interleave_0, values = (x_767, var_14151_cast_fp16))[name = string("input_755_cast_fp16")]; tensor normed_737_axes_0 = const()[name = string("normed_737_axes_0"), val = tensor([-1])]; fp16 var_14143_to_fp16 = const()[name = string("op_14143_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_737_cast_fp16 = layer_norm(axes = normed_737_axes_0, epsilon = var_14143_to_fp16, x = input_755_cast_fp16)[name = string("normed_737_cast_fp16")]; tensor var_14156_split_sizes_0 = const()[name = string("op_14156_split_sizes_0"), val = tensor([256, 256])]; int32 var_14156_axis_0 = const()[name = string("op_14156_axis_0"), val = int32(-1)]; tensor var_14156_cast_fp16_0, tensor var_14156_cast_fp16_1 = split(axis = var_14156_axis_0, split_sizes = var_14156_split_sizes_0, x = normed_737_cast_fp16)[name = string("op_14156_cast_fp16")]; tensor var_14159_cast_fp16 = mul(x = var_14156_cast_fp16_0, y = const_234_to_fp16)[name = string("op_14159_cast_fp16")]; tensor var_14165 = const()[name = string("op_14165"), val = tensor([1, 8, 1, 256])]; tensor q_201 = reshape(shape = var_14165, x = var_14159_cast_fp16)[name = string("q_201")]; tensor var_14167 = mul(x = q_201, y = cos_1)[name = string("op_14167")]; tensor var_14168_split_sizes_0 = const()[name = string("op_14168_split_sizes_0"), val = tensor([128, 128])]; int32 var_14168_axis_0 = const()[name = string("op_14168_axis_0"), val = int32(-1)]; tensor var_14168_0, tensor var_14168_1 = split(axis = var_14168_axis_0, split_sizes = var_14168_split_sizes_0, x = q_201)[name = string("op_14168")]; fp16 const_438_promoted = const()[name = string("const_438_promoted"), val = fp16(-0x1p+0)]; tensor var_14170 = mul(x = var_14168_1, y = const_438_promoted)[name = string("op_14170")]; int32 var_14172 = const()[name = string("op_14172"), val = int32(-1)]; bool var_14173_interleave_0 = const()[name = string("op_14173_interleave_0"), val = bool(false)]; tensor var_14173 = concat(axis = var_14172, interleave = var_14173_interleave_0, values = (var_14170, var_14168_0))[name = string("op_14173")]; tensor var_14174 = mul(x = var_14173, y = sin_1)[name = string("op_14174")]; tensor q_203 = add(x = var_14167, y = var_14174)[name = string("q_203")]; bool var_14188_transpose_x_0 = const()[name = string("op_14188_transpose_x_0"), val = bool(false)]; bool var_14188_transpose_y_0 = const()[name = string("op_14188_transpose_y_0"), val = bool(false)]; tensor var_14188_cast_fp16 = matmul(transpose_x = var_14188_transpose_x_0, transpose_y = var_14188_transpose_y_0, x = q_203, y = transpose_153_cast_fp16)[name = string("op_14188_cast_fp16")]; tensor attn_weights_171_cast_fp16 = add(x = var_14188_cast_fp16, y = causal_mask)[name = string("attn_weights_171_cast_fp16")]; int32 var_14193 = const()[name = string("op_14193"), val = int32(-1)]; tensor attn_weights_173_cast_fp16 = softmax(axis = var_14193, x = attn_weights_171_cast_fp16)[name = string("attn_weights_173_cast_fp16")]; bool attn_output_169_transpose_x_0 = const()[name = string("attn_output_169_transpose_x_0"), val = bool(false)]; bool attn_output_169_transpose_y_0 = const()[name = string("attn_output_169_transpose_y_0"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_0, transpose_y = attn_output_169_transpose_y_0, x = attn_weights_173_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_169_cast_fp16")]; tensor var_14201 = const()[name = string("op_14201"), val = tensor([0, 2, 1, 3])]; tensor var_14208 = const()[name = string("op_14208"), val = tensor([1, 1, -1])]; tensor var_14202_cast_fp16 = transpose(perm = var_14201, x = attn_output_169_cast_fp16)[name = string("transpose_48")]; tensor attn_output_171_cast_fp16 = reshape(shape = var_14208, x = var_14202_cast_fp16)[name = string("attn_output_171_cast_fp16")]; tensor var_14213 = const()[name = string("op_14213"), val = tensor([0, 2, 1])]; string var_14229_pad_type_0 = const()[name = string("op_14229_pad_type_0"), val = string("valid")]; int32 var_14229_groups_0 = const()[name = string("op_14229_groups_0"), val = int32(1)]; tensor var_14229_strides_0 = const()[name = string("op_14229_strides_0"), val = tensor([1])]; tensor var_14229_pad_0 = const()[name = string("op_14229_pad_0"), val = tensor([0, 0])]; tensor var_14229_dilations_0 = const()[name = string("op_14229_dilations_0"), val = tensor([1])]; tensor squeeze_28_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1119792640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121365568))))[name = string("squeeze_28_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_14214_cast_fp16 = transpose(perm = var_14213, x = attn_output_171_cast_fp16)[name = string("transpose_47")]; tensor var_14229_cast_fp16 = conv(dilations = var_14229_dilations_0, groups = var_14229_groups_0, pad = var_14229_pad_0, pad_type = var_14229_pad_type_0, strides = var_14229_strides_0, weight = squeeze_28_cast_fp16_to_fp32_to_fp16_palettized, x = var_14214_cast_fp16)[name = string("op_14229_cast_fp16")]; tensor var_14233 = const()[name = string("op_14233"), val = tensor([0, 2, 1])]; int32 var_14239 = const()[name = string("op_14239"), val = int32(-1)]; fp16 const_439_promoted_to_fp16 = const()[name = string("const_439_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_771_cast_fp16 = transpose(perm = var_14233, x = var_14229_cast_fp16)[name = string("transpose_46")]; tensor var_14245_cast_fp16 = mul(x = x_771_cast_fp16, y = const_439_promoted_to_fp16)[name = string("op_14245_cast_fp16")]; bool input_759_interleave_0 = const()[name = string("input_759_interleave_0"), val = bool(false)]; tensor input_759_cast_fp16 = concat(axis = var_14239, interleave = input_759_interleave_0, values = (x_771_cast_fp16, var_14245_cast_fp16))[name = string("input_759_cast_fp16")]; tensor normed_741_axes_0 = const()[name = string("normed_741_axes_0"), val = tensor([-1])]; fp16 var_14237_to_fp16 = const()[name = string("op_14237_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_741_cast_fp16 = layer_norm(axes = normed_741_axes_0, epsilon = var_14237_to_fp16, x = input_759_cast_fp16)[name = string("normed_741_cast_fp16")]; tensor var_14250_split_sizes_0 = const()[name = string("op_14250_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14250_axis_0 = const()[name = string("op_14250_axis_0"), val = int32(-1)]; tensor var_14250_cast_fp16_0, tensor var_14250_cast_fp16_1 = split(axis = var_14250_axis_0, split_sizes = var_14250_split_sizes_0, x = normed_741_cast_fp16)[name = string("op_14250_cast_fp16")]; tensor const_440_to_fp16 = const()[name = string("const_440_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121367168)))]; tensor var_14253_cast_fp16 = mul(x = var_14250_cast_fp16_0, y = const_440_to_fp16)[name = string("op_14253_cast_fp16")]; tensor x_775_cast_fp16 = add(x = x_763_cast_fp16, y = var_14253_cast_fp16)[name = string("x_775_cast_fp16")]; int32 var_14260 = const()[name = string("op_14260"), val = int32(-1)]; fp16 const_441_promoted_to_fp16 = const()[name = string("const_441_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14266_cast_fp16 = mul(x = x_775_cast_fp16, y = const_441_promoted_to_fp16)[name = string("op_14266_cast_fp16")]; bool input_761_interleave_0 = const()[name = string("input_761_interleave_0"), val = bool(false)]; tensor input_761_cast_fp16 = concat(axis = var_14260, interleave = input_761_interleave_0, values = (x_775_cast_fp16, var_14266_cast_fp16))[name = string("input_761_cast_fp16")]; tensor normed_745_axes_0 = const()[name = string("normed_745_axes_0"), val = tensor([-1])]; fp16 var_14258_to_fp16 = const()[name = string("op_14258_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_745_cast_fp16 = layer_norm(axes = normed_745_axes_0, epsilon = var_14258_to_fp16, x = input_761_cast_fp16)[name = string("normed_745_cast_fp16")]; tensor var_14271_split_sizes_0 = const()[name = string("op_14271_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14271_axis_0 = const()[name = string("op_14271_axis_0"), val = int32(-1)]; tensor var_14271_cast_fp16_0, tensor var_14271_cast_fp16_1 = split(axis = var_14271_axis_0, split_sizes = var_14271_split_sizes_0, x = normed_745_cast_fp16)[name = string("op_14271_cast_fp16")]; tensor const_442_to_fp16 = const()[name = string("const_442_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121370304)))]; tensor var_14274_cast_fp16 = mul(x = var_14271_cast_fp16_0, y = const_442_to_fp16)[name = string("op_14274_cast_fp16")]; tensor var_14287 = const()[name = string("op_14287"), val = tensor([0, 2, 1])]; tensor input_763_axes_0 = const()[name = string("input_763_axes_0"), val = tensor([2])]; tensor var_14288 = transpose(perm = var_14287, x = var_14274_cast_fp16)[name = string("transpose_45")]; tensor input_763 = expand_dims(axes = input_763_axes_0, x = var_14288)[name = string("input_763")]; string gate_113_pad_type_0 = const()[name = string("gate_113_pad_type_0"), val = string("valid")]; tensor gate_113_strides_0 = const()[name = string("gate_113_strides_0"), val = tensor([1, 1])]; tensor gate_113_pad_0 = const()[name = string("gate_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_113_dilations_0 = const()[name = string("gate_113_dilations_0"), val = tensor([1, 1])]; int32 gate_113_groups_0 = const()[name = string("gate_113_groups_0"), val = int32(1)]; tensor gate_113 = conv(dilations = gate_113_dilations_0, groups = gate_113_groups_0, pad = gate_113_pad_0, pad_type = gate_113_pad_type_0, strides = gate_113_strides_0, weight = layers_28_mlp_gate_proj_weight_palettized, x = input_763)[name = string("gate_113")]; string up_57_pad_type_0 = const()[name = string("up_57_pad_type_0"), val = string("valid")]; tensor up_57_strides_0 = const()[name = string("up_57_strides_0"), val = tensor([1, 1])]; tensor up_57_pad_0 = const()[name = string("up_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_57_dilations_0 = const()[name = string("up_57_dilations_0"), val = tensor([1, 1])]; int32 up_57_groups_0 = const()[name = string("up_57_groups_0"), val = int32(1)]; tensor up_57 = conv(dilations = up_57_dilations_0, groups = up_57_groups_0, pad = up_57_pad_0, pad_type = up_57_pad_type_0, strides = up_57_strides_0, weight = layers_28_mlp_up_proj_weight_palettized, x = input_763)[name = string("up_57")]; string gate_115_mode_0 = const()[name = string("gate_115_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_115 = gelu(mode = gate_115_mode_0, x = gate_113)[name = string("gate_115")]; tensor input_765 = mul(x = gate_115, y = up_57)[name = string("input_765")]; string mlp_out_57_pad_type_0 = const()[name = string("mlp_out_57_pad_type_0"), val = string("valid")]; tensor mlp_out_57_strides_0 = const()[name = string("mlp_out_57_strides_0"), val = tensor([1, 1])]; tensor mlp_out_57_pad_0 = const()[name = string("mlp_out_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_57_dilations_0 = const()[name = string("mlp_out_57_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_57_groups_0 = const()[name = string("mlp_out_57_groups_0"), val = int32(1)]; tensor mlp_out_57 = conv(dilations = mlp_out_57_dilations_0, groups = mlp_out_57_groups_0, pad = mlp_out_57_pad_0, pad_type = mlp_out_57_pad_type_0, strides = mlp_out_57_strides_0, weight = layers_28_mlp_down_proj_weight_palettized, x = input_765)[name = string("mlp_out_57")]; tensor var_14328_axes_0 = const()[name = string("op_14328_axes_0"), val = tensor([2])]; tensor var_14328 = squeeze(axes = var_14328_axes_0, x = mlp_out_57)[name = string("op_14328")]; tensor var_14332 = const()[name = string("op_14332"), val = tensor([0, 2, 1])]; int32 var_14338 = const()[name = string("op_14338"), val = int32(-1)]; fp16 const_443_promoted_to_fp16 = const()[name = string("const_443_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_779 = transpose(perm = var_14332, x = var_14328)[name = string("transpose_44")]; tensor var_14344_cast_fp16 = mul(x = x_779, y = const_443_promoted_to_fp16)[name = string("op_14344_cast_fp16")]; bool input_767_interleave_0 = const()[name = string("input_767_interleave_0"), val = bool(false)]; tensor input_767_cast_fp16 = concat(axis = var_14338, interleave = input_767_interleave_0, values = (x_779, var_14344_cast_fp16))[name = string("input_767_cast_fp16")]; tensor normed_749_axes_0 = const()[name = string("normed_749_axes_0"), val = tensor([-1])]; fp16 var_14336_to_fp16 = const()[name = string("op_14336_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_749_cast_fp16 = layer_norm(axes = normed_749_axes_0, epsilon = var_14336_to_fp16, x = input_767_cast_fp16)[name = string("normed_749_cast_fp16")]; tensor var_14349_split_sizes_0 = const()[name = string("op_14349_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14349_axis_0 = const()[name = string("op_14349_axis_0"), val = int32(-1)]; tensor var_14349_cast_fp16_0, tensor var_14349_cast_fp16_1 = split(axis = var_14349_axis_0, split_sizes = var_14349_split_sizes_0, x = normed_749_cast_fp16)[name = string("op_14349_cast_fp16")]; tensor const_444_to_fp16 = const()[name = string("const_444_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121373440)))]; tensor var_14352_cast_fp16 = mul(x = var_14349_cast_fp16_0, y = const_444_to_fp16)[name = string("op_14352_cast_fp16")]; tensor hidden_states_345_cast_fp16 = add(x = x_775_cast_fp16, y = var_14352_cast_fp16)[name = string("hidden_states_345_cast_fp16")]; tensor per_layer_slice_57_begin_0 = const()[name = string("per_layer_slice_57_begin_0"), val = tensor([0, 0, 7168])]; tensor per_layer_slice_57_end_0 = const()[name = string("per_layer_slice_57_end_0"), val = tensor([1, 1, 7424])]; tensor per_layer_slice_57_end_mask_0 = const()[name = string("per_layer_slice_57_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_57_cast_fp16 = slice_by_index(begin = per_layer_slice_57_begin_0, end = per_layer_slice_57_end_0, end_mask = per_layer_slice_57_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_57_cast_fp16")]; tensor gated_113 = linear(bias = linear_0_bias_0, weight = layers_28_per_layer_input_gate_weight_palettized, x = hidden_states_345_cast_fp16)[name = string("linear_56")]; string gated_115_mode_0 = const()[name = string("gated_115_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_115 = gelu(mode = gated_115_mode_0, x = gated_113)[name = string("gated_115")]; tensor input_771_cast_fp16 = mul(x = gated_115, y = per_layer_slice_57_cast_fp16)[name = string("input_771_cast_fp16")]; tensor layers_28_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121376576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121573248))))[name = string("layers_28_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_57_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_28_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_771_cast_fp16)[name = string("linear_57_cast_fp16")]; int32 var_14389 = const()[name = string("op_14389"), val = int32(-1)]; fp16 const_445_promoted_to_fp16 = const()[name = string("const_445_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14395_cast_fp16 = mul(x = linear_57_cast_fp16, y = const_445_promoted_to_fp16)[name = string("op_14395_cast_fp16")]; bool input_773_interleave_0 = const()[name = string("input_773_interleave_0"), val = bool(false)]; tensor input_773_cast_fp16 = concat(axis = var_14389, interleave = input_773_interleave_0, values = (linear_57_cast_fp16, var_14395_cast_fp16))[name = string("input_773_cast_fp16")]; tensor normed_753_axes_0 = const()[name = string("normed_753_axes_0"), val = tensor([-1])]; fp16 var_14387_to_fp16 = const()[name = string("op_14387_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_753_cast_fp16 = layer_norm(axes = normed_753_axes_0, epsilon = var_14387_to_fp16, x = input_773_cast_fp16)[name = string("normed_753_cast_fp16")]; tensor var_14400_split_sizes_0 = const()[name = string("op_14400_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14400_axis_0 = const()[name = string("op_14400_axis_0"), val = int32(-1)]; tensor var_14400_cast_fp16_0, tensor var_14400_cast_fp16_1 = split(axis = var_14400_axis_0, split_sizes = var_14400_split_sizes_0, x = normed_753_cast_fp16)[name = string("op_14400_cast_fp16")]; tensor const_446_to_fp16 = const()[name = string("const_446_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121574848)))]; tensor var_14403_cast_fp16 = mul(x = var_14400_cast_fp16_0, y = const_446_to_fp16)[name = string("op_14403_cast_fp16")]; tensor hidden_states_349_cast_fp16 = add(x = hidden_states_345_cast_fp16, y = var_14403_cast_fp16)[name = string("hidden_states_349_cast_fp16")]; tensor layers_28_layer_scalar_to_fp16 = const()[name = string("layers_28_layer_scalar_to_fp16"), val = tensor([0x1.a4p-1])]; tensor x_787_cast_fp16 = mul(x = hidden_states_349_cast_fp16, y = layers_28_layer_scalar_to_fp16)[name = string("x_787_cast_fp16")]; int32 var_14411 = const()[name = string("op_14411"), val = int32(-1)]; fp16 const_447_promoted_to_fp16 = const()[name = string("const_447_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14417_cast_fp16 = mul(x = x_787_cast_fp16, y = const_447_promoted_to_fp16)[name = string("op_14417_cast_fp16")]; bool input_775_interleave_0 = const()[name = string("input_775_interleave_0"), val = bool(false)]; tensor input_775_cast_fp16 = concat(axis = var_14411, interleave = input_775_interleave_0, values = (x_787_cast_fp16, var_14417_cast_fp16))[name = string("input_775_cast_fp16")]; tensor normed_757_axes_0 = const()[name = string("normed_757_axes_0"), val = tensor([-1])]; fp16 var_14409_to_fp16 = const()[name = string("op_14409_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_757_cast_fp16 = layer_norm(axes = normed_757_axes_0, epsilon = var_14409_to_fp16, x = input_775_cast_fp16)[name = string("normed_757_cast_fp16")]; tensor var_14422_split_sizes_0 = const()[name = string("op_14422_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14422_axis_0 = const()[name = string("op_14422_axis_0"), val = int32(-1)]; tensor var_14422_cast_fp16_0, tensor var_14422_cast_fp16_1 = split(axis = var_14422_axis_0, split_sizes = var_14422_split_sizes_0, x = normed_757_cast_fp16)[name = string("op_14422_cast_fp16")]; tensor const_448_to_fp16 = const()[name = string("const_448_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121577984)))]; tensor var_14425_cast_fp16 = mul(x = var_14422_cast_fp16_0, y = const_448_to_fp16)[name = string("op_14425_cast_fp16")]; tensor var_14433 = const()[name = string("op_14433"), val = tensor([0, 2, 1])]; tensor var_14436_axes_0 = const()[name = string("op_14436_axes_0"), val = tensor([2])]; tensor var_14434_cast_fp16 = transpose(perm = var_14433, x = var_14425_cast_fp16)[name = string("transpose_43")]; tensor var_14436_cast_fp16 = expand_dims(axes = var_14436_axes_0, x = var_14434_cast_fp16)[name = string("op_14436_cast_fp16")]; string var_14452_pad_type_0 = const()[name = string("op_14452_pad_type_0"), val = string("valid")]; tensor var_14452_strides_0 = const()[name = string("op_14452_strides_0"), val = tensor([1, 1])]; tensor var_14452_pad_0 = const()[name = string("op_14452_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_14452_dilations_0 = const()[name = string("op_14452_dilations_0"), val = tensor([1, 1])]; int32 var_14452_groups_0 = const()[name = string("op_14452_groups_0"), val = int32(1)]; tensor var_14452 = conv(dilations = var_14452_dilations_0, groups = var_14452_groups_0, pad = var_14452_pad_0, pad_type = var_14452_pad_type_0, strides = var_14452_strides_0, weight = layers_29_self_attn_q_proj_weight_palettized, x = var_14436_cast_fp16)[name = string("op_14452")]; tensor var_14457 = const()[name = string("op_14457"), val = tensor([1, 8, 512, 1])]; tensor var_14458 = reshape(shape = var_14457, x = var_14452)[name = string("op_14458")]; tensor var_14463 = const()[name = string("op_14463"), val = tensor([0, 1, 3, 2])]; tensor var_14473 = const()[name = string("op_14473"), val = tensor([1, 8, 512])]; tensor var_14464 = transpose(perm = var_14463, x = var_14458)[name = string("transpose_42")]; tensor x_791 = reshape(shape = var_14473, x = var_14464)[name = string("x_791")]; int32 var_14479 = const()[name = string("op_14479"), val = int32(-1)]; fp16 const_449_promoted_to_fp16 = const()[name = string("const_449_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14485_cast_fp16 = mul(x = x_791, y = const_449_promoted_to_fp16)[name = string("op_14485_cast_fp16")]; bool input_779_interleave_0 = const()[name = string("input_779_interleave_0"), val = bool(false)]; tensor input_779_cast_fp16 = concat(axis = var_14479, interleave = input_779_interleave_0, values = (x_791, var_14485_cast_fp16))[name = string("input_779_cast_fp16")]; tensor normed_761_axes_0 = const()[name = string("normed_761_axes_0"), val = tensor([-1])]; fp16 var_14477_to_fp16 = const()[name = string("op_14477_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_761_cast_fp16 = layer_norm(axes = normed_761_axes_0, epsilon = var_14477_to_fp16, x = input_779_cast_fp16)[name = string("normed_761_cast_fp16")]; tensor var_14490_split_sizes_0 = const()[name = string("op_14490_split_sizes_0"), val = tensor([512, 512])]; int32 var_14490_axis_0 = const()[name = string("op_14490_axis_0"), val = int32(-1)]; tensor var_14490_cast_fp16_0, tensor var_14490_cast_fp16_1 = split(axis = var_14490_axis_0, split_sizes = var_14490_split_sizes_0, x = normed_761_cast_fp16)[name = string("op_14490_cast_fp16")]; tensor var_14493_cast_fp16 = mul(x = var_14490_cast_fp16_0, y = const_252_to_fp16)[name = string("op_14493_cast_fp16")]; tensor var_14499 = const()[name = string("op_14499"), val = tensor([1, 8, 1, 512])]; tensor q_207 = reshape(shape = var_14499, x = var_14493_cast_fp16)[name = string("q_207")]; tensor var_14501 = mul(x = q_207, y = cos)[name = string("op_14501")]; tensor var_14502_split_sizes_0 = const()[name = string("op_14502_split_sizes_0"), val = tensor([256, 256])]; int32 var_14502_axis_0 = const()[name = string("op_14502_axis_0"), val = int32(-1)]; tensor var_14502_0, tensor var_14502_1 = split(axis = var_14502_axis_0, split_sizes = var_14502_split_sizes_0, x = q_207)[name = string("op_14502")]; fp16 const_451_promoted = const()[name = string("const_451_promoted"), val = fp16(-0x1p+0)]; tensor var_14504 = mul(x = var_14502_1, y = const_451_promoted)[name = string("op_14504")]; int32 var_14506 = const()[name = string("op_14506"), val = int32(-1)]; bool var_14507_interleave_0 = const()[name = string("op_14507_interleave_0"), val = bool(false)]; tensor var_14507 = concat(axis = var_14506, interleave = var_14507_interleave_0, values = (var_14504, var_14502_0))[name = string("op_14507")]; tensor var_14508 = mul(x = var_14507, y = sin)[name = string("op_14508")]; tensor q_209 = add(x = var_14501, y = var_14508)[name = string("q_209")]; bool var_14522_transpose_x_0 = const()[name = string("op_14522_transpose_x_0"), val = bool(false)]; bool var_14522_transpose_y_0 = const()[name = string("op_14522_transpose_y_0"), val = bool(false)]; tensor var_14522_cast_fp16 = matmul(transpose_x = var_14522_transpose_x_0, transpose_y = var_14522_transpose_y_0, x = q_209, y = transpose_154_cast_fp16)[name = string("op_14522_cast_fp16")]; tensor attn_weights_177_cast_fp16 = add(x = var_14522_cast_fp16, y = causal_mask)[name = string("attn_weights_177_cast_fp16")]; int32 var_14527 = const()[name = string("op_14527"), val = int32(-1)]; tensor attn_weights_179_cast_fp16 = softmax(axis = var_14527, x = attn_weights_177_cast_fp16)[name = string("attn_weights_179_cast_fp16")]; bool attn_output_175_transpose_x_0 = const()[name = string("attn_output_175_transpose_x_0"), val = bool(false)]; bool attn_output_175_transpose_y_0 = const()[name = string("attn_output_175_transpose_y_0"), val = bool(false)]; tensor attn_output_175_cast_fp16 = matmul(transpose_x = attn_output_175_transpose_x_0, transpose_y = attn_output_175_transpose_y_0, x = attn_weights_179_cast_fp16, y = V_expanded_29_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor var_14535 = const()[name = string("op_14535"), val = tensor([0, 2, 1, 3])]; tensor var_14542 = const()[name = string("op_14542"), val = tensor([1, 1, -1])]; tensor var_14536_cast_fp16 = transpose(perm = var_14535, x = attn_output_175_cast_fp16)[name = string("transpose_41")]; tensor attn_output_177_cast_fp16 = reshape(shape = var_14542, x = var_14536_cast_fp16)[name = string("attn_output_177_cast_fp16")]; tensor var_14547 = const()[name = string("op_14547"), val = tensor([0, 2, 1])]; string var_14563_pad_type_0 = const()[name = string("op_14563_pad_type_0"), val = string("valid")]; int32 var_14563_groups_0 = const()[name = string("op_14563_groups_0"), val = int32(1)]; tensor var_14563_strides_0 = const()[name = string("op_14563_strides_0"), val = tensor([1])]; tensor var_14563_pad_0 = const()[name = string("op_14563_pad_0"), val = tensor([0, 0])]; tensor var_14563_dilations_0 = const()[name = string("op_14563_dilations_0"), val = tensor([1])]; tensor squeeze_29_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1121581120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124726912))))[name = string("squeeze_29_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_14548_cast_fp16 = transpose(perm = var_14547, x = attn_output_177_cast_fp16)[name = string("transpose_40")]; tensor var_14563_cast_fp16 = conv(dilations = var_14563_dilations_0, groups = var_14563_groups_0, pad = var_14563_pad_0, pad_type = var_14563_pad_type_0, strides = var_14563_strides_0, weight = squeeze_29_cast_fp16_to_fp32_to_fp16_palettized, x = var_14548_cast_fp16)[name = string("op_14563_cast_fp16")]; tensor var_14567 = const()[name = string("op_14567"), val = tensor([0, 2, 1])]; int32 var_14573 = const()[name = string("op_14573"), val = int32(-1)]; fp16 const_452_promoted_to_fp16 = const()[name = string("const_452_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_795_cast_fp16 = transpose(perm = var_14567, x = var_14563_cast_fp16)[name = string("transpose_39")]; tensor var_14579_cast_fp16 = mul(x = x_795_cast_fp16, y = const_452_promoted_to_fp16)[name = string("op_14579_cast_fp16")]; bool input_783_interleave_0 = const()[name = string("input_783_interleave_0"), val = bool(false)]; tensor input_783_cast_fp16 = concat(axis = var_14573, interleave = input_783_interleave_0, values = (x_795_cast_fp16, var_14579_cast_fp16))[name = string("input_783_cast_fp16")]; tensor normed_765_axes_0 = const()[name = string("normed_765_axes_0"), val = tensor([-1])]; fp16 var_14571_to_fp16 = const()[name = string("op_14571_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_765_cast_fp16 = layer_norm(axes = normed_765_axes_0, epsilon = var_14571_to_fp16, x = input_783_cast_fp16)[name = string("normed_765_cast_fp16")]; tensor var_14584_split_sizes_0 = const()[name = string("op_14584_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14584_axis_0 = const()[name = string("op_14584_axis_0"), val = int32(-1)]; tensor var_14584_cast_fp16_0, tensor var_14584_cast_fp16_1 = split(axis = var_14584_axis_0, split_sizes = var_14584_split_sizes_0, x = normed_765_cast_fp16)[name = string("op_14584_cast_fp16")]; tensor const_453_to_fp16 = const()[name = string("const_453_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124728512)))]; tensor var_14587_cast_fp16 = mul(x = var_14584_cast_fp16_0, y = const_453_to_fp16)[name = string("op_14587_cast_fp16")]; tensor x_799_cast_fp16 = add(x = x_787_cast_fp16, y = var_14587_cast_fp16)[name = string("x_799_cast_fp16")]; int32 var_14594 = const()[name = string("op_14594"), val = int32(-1)]; fp16 const_454_promoted_to_fp16 = const()[name = string("const_454_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14600_cast_fp16 = mul(x = x_799_cast_fp16, y = const_454_promoted_to_fp16)[name = string("op_14600_cast_fp16")]; bool input_785_interleave_0 = const()[name = string("input_785_interleave_0"), val = bool(false)]; tensor input_785_cast_fp16 = concat(axis = var_14594, interleave = input_785_interleave_0, values = (x_799_cast_fp16, var_14600_cast_fp16))[name = string("input_785_cast_fp16")]; tensor normed_769_axes_0 = const()[name = string("normed_769_axes_0"), val = tensor([-1])]; fp16 var_14592_to_fp16 = const()[name = string("op_14592_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_769_cast_fp16 = layer_norm(axes = normed_769_axes_0, epsilon = var_14592_to_fp16, x = input_785_cast_fp16)[name = string("normed_769_cast_fp16")]; tensor var_14605_split_sizes_0 = const()[name = string("op_14605_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14605_axis_0 = const()[name = string("op_14605_axis_0"), val = int32(-1)]; tensor var_14605_cast_fp16_0, tensor var_14605_cast_fp16_1 = split(axis = var_14605_axis_0, split_sizes = var_14605_split_sizes_0, x = normed_769_cast_fp16)[name = string("op_14605_cast_fp16")]; tensor const_455_to_fp16 = const()[name = string("const_455_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124731648)))]; tensor var_14608_cast_fp16 = mul(x = var_14605_cast_fp16_0, y = const_455_to_fp16)[name = string("op_14608_cast_fp16")]; tensor var_14621 = const()[name = string("op_14621"), val = tensor([0, 2, 1])]; tensor input_787_axes_0 = const()[name = string("input_787_axes_0"), val = tensor([2])]; tensor var_14622 = transpose(perm = var_14621, x = var_14608_cast_fp16)[name = string("transpose_38")]; tensor input_787 = expand_dims(axes = input_787_axes_0, x = var_14622)[name = string("input_787")]; string gate_117_pad_type_0 = const()[name = string("gate_117_pad_type_0"), val = string("valid")]; tensor gate_117_strides_0 = const()[name = string("gate_117_strides_0"), val = tensor([1, 1])]; tensor gate_117_pad_0 = const()[name = string("gate_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_117_dilations_0 = const()[name = string("gate_117_dilations_0"), val = tensor([1, 1])]; int32 gate_117_groups_0 = const()[name = string("gate_117_groups_0"), val = int32(1)]; tensor gate_117 = conv(dilations = gate_117_dilations_0, groups = gate_117_groups_0, pad = gate_117_pad_0, pad_type = gate_117_pad_type_0, strides = gate_117_strides_0, weight = layers_29_mlp_gate_proj_weight_palettized, x = input_787)[name = string("gate_117")]; string up_59_pad_type_0 = const()[name = string("up_59_pad_type_0"), val = string("valid")]; tensor up_59_strides_0 = const()[name = string("up_59_strides_0"), val = tensor([1, 1])]; tensor up_59_pad_0 = const()[name = string("up_59_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_59_dilations_0 = const()[name = string("up_59_dilations_0"), val = tensor([1, 1])]; int32 up_59_groups_0 = const()[name = string("up_59_groups_0"), val = int32(1)]; tensor up_59 = conv(dilations = up_59_dilations_0, groups = up_59_groups_0, pad = up_59_pad_0, pad_type = up_59_pad_type_0, strides = up_59_strides_0, weight = layers_29_mlp_up_proj_weight_palettized, x = input_787)[name = string("up_59")]; string gate_119_mode_0 = const()[name = string("gate_119_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_119 = gelu(mode = gate_119_mode_0, x = gate_117)[name = string("gate_119")]; tensor input_789 = mul(x = gate_119, y = up_59)[name = string("input_789")]; string mlp_out_59_pad_type_0 = const()[name = string("mlp_out_59_pad_type_0"), val = string("valid")]; tensor mlp_out_59_strides_0 = const()[name = string("mlp_out_59_strides_0"), val = tensor([1, 1])]; tensor mlp_out_59_pad_0 = const()[name = string("mlp_out_59_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_59_dilations_0 = const()[name = string("mlp_out_59_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_59_groups_0 = const()[name = string("mlp_out_59_groups_0"), val = int32(1)]; tensor mlp_out_59 = conv(dilations = mlp_out_59_dilations_0, groups = mlp_out_59_groups_0, pad = mlp_out_59_pad_0, pad_type = mlp_out_59_pad_type_0, strides = mlp_out_59_strides_0, weight = layers_29_mlp_down_proj_weight_palettized, x = input_789)[name = string("mlp_out_59")]; tensor var_14662_axes_0 = const()[name = string("op_14662_axes_0"), val = tensor([2])]; tensor var_14662 = squeeze(axes = var_14662_axes_0, x = mlp_out_59)[name = string("op_14662")]; tensor var_14666 = const()[name = string("op_14666"), val = tensor([0, 2, 1])]; int32 var_14672 = const()[name = string("op_14672"), val = int32(-1)]; fp16 const_456_promoted_to_fp16 = const()[name = string("const_456_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_803 = transpose(perm = var_14666, x = var_14662)[name = string("transpose_37")]; tensor var_14678_cast_fp16 = mul(x = x_803, y = const_456_promoted_to_fp16)[name = string("op_14678_cast_fp16")]; bool input_791_interleave_0 = const()[name = string("input_791_interleave_0"), val = bool(false)]; tensor input_791_cast_fp16 = concat(axis = var_14672, interleave = input_791_interleave_0, values = (x_803, var_14678_cast_fp16))[name = string("input_791_cast_fp16")]; tensor normed_773_axes_0 = const()[name = string("normed_773_axes_0"), val = tensor([-1])]; fp16 var_14670_to_fp16 = const()[name = string("op_14670_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_773_cast_fp16 = layer_norm(axes = normed_773_axes_0, epsilon = var_14670_to_fp16, x = input_791_cast_fp16)[name = string("normed_773_cast_fp16")]; tensor var_14683_split_sizes_0 = const()[name = string("op_14683_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14683_axis_0 = const()[name = string("op_14683_axis_0"), val = int32(-1)]; tensor var_14683_cast_fp16_0, tensor var_14683_cast_fp16_1 = split(axis = var_14683_axis_0, split_sizes = var_14683_split_sizes_0, x = normed_773_cast_fp16)[name = string("op_14683_cast_fp16")]; tensor const_457_to_fp16 = const()[name = string("const_457_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124734784)))]; tensor var_14686_cast_fp16 = mul(x = var_14683_cast_fp16_0, y = const_457_to_fp16)[name = string("op_14686_cast_fp16")]; tensor hidden_states_357_cast_fp16 = add(x = x_799_cast_fp16, y = var_14686_cast_fp16)[name = string("hidden_states_357_cast_fp16")]; tensor per_layer_slice_59_begin_0 = const()[name = string("per_layer_slice_59_begin_0"), val = tensor([0, 0, 7424])]; tensor per_layer_slice_59_end_0 = const()[name = string("per_layer_slice_59_end_0"), val = tensor([1, 1, 7680])]; tensor per_layer_slice_59_end_mask_0 = const()[name = string("per_layer_slice_59_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_59_cast_fp16 = slice_by_index(begin = per_layer_slice_59_begin_0, end = per_layer_slice_59_end_0, end_mask = per_layer_slice_59_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_59_cast_fp16")]; tensor gated_117 = linear(bias = linear_0_bias_0, weight = layers_29_per_layer_input_gate_weight_palettized, x = hidden_states_357_cast_fp16)[name = string("linear_58")]; string gated_119_mode_0 = const()[name = string("gated_119_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_119 = gelu(mode = gated_119_mode_0, x = gated_117)[name = string("gated_119")]; tensor input_795_cast_fp16 = mul(x = gated_119, y = per_layer_slice_59_cast_fp16)[name = string("input_795_cast_fp16")]; tensor layers_29_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124737920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124934592))))[name = string("layers_29_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_59_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_29_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_795_cast_fp16)[name = string("linear_59_cast_fp16")]; int32 var_14723 = const()[name = string("op_14723"), val = int32(-1)]; fp16 const_458_promoted_to_fp16 = const()[name = string("const_458_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14729_cast_fp16 = mul(x = linear_59_cast_fp16, y = const_458_promoted_to_fp16)[name = string("op_14729_cast_fp16")]; bool input_797_interleave_0 = const()[name = string("input_797_interleave_0"), val = bool(false)]; tensor input_797_cast_fp16 = concat(axis = var_14723, interleave = input_797_interleave_0, values = (linear_59_cast_fp16, var_14729_cast_fp16))[name = string("input_797_cast_fp16")]; tensor normed_777_axes_0 = const()[name = string("normed_777_axes_0"), val = tensor([-1])]; fp16 var_14721_to_fp16 = const()[name = string("op_14721_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_777_cast_fp16 = layer_norm(axes = normed_777_axes_0, epsilon = var_14721_to_fp16, x = input_797_cast_fp16)[name = string("normed_777_cast_fp16")]; tensor var_14734_split_sizes_0 = const()[name = string("op_14734_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14734_axis_0 = const()[name = string("op_14734_axis_0"), val = int32(-1)]; tensor var_14734_cast_fp16_0, tensor var_14734_cast_fp16_1 = split(axis = var_14734_axis_0, split_sizes = var_14734_split_sizes_0, x = normed_777_cast_fp16)[name = string("op_14734_cast_fp16")]; tensor const_459_to_fp16 = const()[name = string("const_459_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124936192)))]; tensor var_14737_cast_fp16 = mul(x = var_14734_cast_fp16_0, y = const_459_to_fp16)[name = string("op_14737_cast_fp16")]; tensor hidden_states_361_cast_fp16 = add(x = hidden_states_357_cast_fp16, y = var_14737_cast_fp16)[name = string("hidden_states_361_cast_fp16")]; tensor layers_29_layer_scalar_to_fp16 = const()[name = string("layers_29_layer_scalar_to_fp16"), val = tensor([0x1.ap-1])]; tensor x_811_cast_fp16 = mul(x = hidden_states_361_cast_fp16, y = layers_29_layer_scalar_to_fp16)[name = string("x_811_cast_fp16")]; int32 var_14745 = const()[name = string("op_14745"), val = int32(-1)]; fp16 const_460_promoted_to_fp16 = const()[name = string("const_460_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14751_cast_fp16 = mul(x = x_811_cast_fp16, y = const_460_promoted_to_fp16)[name = string("op_14751_cast_fp16")]; bool input_799_interleave_0 = const()[name = string("input_799_interleave_0"), val = bool(false)]; tensor input_799_cast_fp16 = concat(axis = var_14745, interleave = input_799_interleave_0, values = (x_811_cast_fp16, var_14751_cast_fp16))[name = string("input_799_cast_fp16")]; tensor normed_781_axes_0 = const()[name = string("normed_781_axes_0"), val = tensor([-1])]; fp16 var_14743_to_fp16 = const()[name = string("op_14743_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_781_cast_fp16 = layer_norm(axes = normed_781_axes_0, epsilon = var_14743_to_fp16, x = input_799_cast_fp16)[name = string("normed_781_cast_fp16")]; tensor var_14756_split_sizes_0 = const()[name = string("op_14756_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14756_axis_0 = const()[name = string("op_14756_axis_0"), val = int32(-1)]; tensor var_14756_cast_fp16_0, tensor var_14756_cast_fp16_1 = split(axis = var_14756_axis_0, split_sizes = var_14756_split_sizes_0, x = normed_781_cast_fp16)[name = string("op_14756_cast_fp16")]; tensor const_461_to_fp16 = const()[name = string("const_461_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124939328)))]; tensor var_14759_cast_fp16 = mul(x = var_14756_cast_fp16_0, y = const_461_to_fp16)[name = string("op_14759_cast_fp16")]; tensor var_14767 = const()[name = string("op_14767"), val = tensor([0, 2, 1])]; tensor var_14770_axes_0 = const()[name = string("op_14770_axes_0"), val = tensor([2])]; tensor var_14768_cast_fp16 = transpose(perm = var_14767, x = var_14759_cast_fp16)[name = string("transpose_36")]; tensor var_14770_cast_fp16 = expand_dims(axes = var_14770_axes_0, x = var_14768_cast_fp16)[name = string("op_14770_cast_fp16")]; string var_14786_pad_type_0 = const()[name = string("op_14786_pad_type_0"), val = string("valid")]; tensor var_14786_strides_0 = const()[name = string("op_14786_strides_0"), val = tensor([1, 1])]; tensor var_14786_pad_0 = const()[name = string("op_14786_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_14786_dilations_0 = const()[name = string("op_14786_dilations_0"), val = tensor([1, 1])]; int32 var_14786_groups_0 = const()[name = string("op_14786_groups_0"), val = int32(1)]; tensor var_14786 = conv(dilations = var_14786_dilations_0, groups = var_14786_groups_0, pad = var_14786_pad_0, pad_type = var_14786_pad_type_0, strides = var_14786_strides_0, weight = layers_30_self_attn_q_proj_weight_palettized, x = var_14770_cast_fp16)[name = string("op_14786")]; tensor var_14791 = const()[name = string("op_14791"), val = tensor([1, 8, 256, 1])]; tensor var_14792 = reshape(shape = var_14791, x = var_14786)[name = string("op_14792")]; tensor var_14797 = const()[name = string("op_14797"), val = tensor([0, 1, 3, 2])]; tensor var_14807 = const()[name = string("op_14807"), val = tensor([1, 8, 256])]; tensor var_14798 = transpose(perm = var_14797, x = var_14792)[name = string("transpose_35")]; tensor x_815 = reshape(shape = var_14807, x = var_14798)[name = string("x_815")]; int32 var_14813 = const()[name = string("op_14813"), val = int32(-1)]; fp16 const_462_promoted_to_fp16 = const()[name = string("const_462_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14819_cast_fp16 = mul(x = x_815, y = const_462_promoted_to_fp16)[name = string("op_14819_cast_fp16")]; bool input_803_interleave_0 = const()[name = string("input_803_interleave_0"), val = bool(false)]; tensor input_803_cast_fp16 = concat(axis = var_14813, interleave = input_803_interleave_0, values = (x_815, var_14819_cast_fp16))[name = string("input_803_cast_fp16")]; tensor normed_785_axes_0 = const()[name = string("normed_785_axes_0"), val = tensor([-1])]; fp16 var_14811_to_fp16 = const()[name = string("op_14811_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_785_cast_fp16 = layer_norm(axes = normed_785_axes_0, epsilon = var_14811_to_fp16, x = input_803_cast_fp16)[name = string("normed_785_cast_fp16")]; tensor var_14824_split_sizes_0 = const()[name = string("op_14824_split_sizes_0"), val = tensor([256, 256])]; int32 var_14824_axis_0 = const()[name = string("op_14824_axis_0"), val = int32(-1)]; tensor var_14824_cast_fp16_0, tensor var_14824_cast_fp16_1 = split(axis = var_14824_axis_0, split_sizes = var_14824_split_sizes_0, x = normed_785_cast_fp16)[name = string("op_14824_cast_fp16")]; tensor var_14827_cast_fp16 = mul(x = var_14824_cast_fp16_0, y = const_234_to_fp16)[name = string("op_14827_cast_fp16")]; tensor var_14833 = const()[name = string("op_14833"), val = tensor([1, 8, 1, 256])]; tensor q_213 = reshape(shape = var_14833, x = var_14827_cast_fp16)[name = string("q_213")]; tensor var_14835 = mul(x = q_213, y = cos_1)[name = string("op_14835")]; tensor var_14836_split_sizes_0 = const()[name = string("op_14836_split_sizes_0"), val = tensor([128, 128])]; int32 var_14836_axis_0 = const()[name = string("op_14836_axis_0"), val = int32(-1)]; tensor var_14836_0, tensor var_14836_1 = split(axis = var_14836_axis_0, split_sizes = var_14836_split_sizes_0, x = q_213)[name = string("op_14836")]; fp16 const_464_promoted = const()[name = string("const_464_promoted"), val = fp16(-0x1p+0)]; tensor var_14838 = mul(x = var_14836_1, y = const_464_promoted)[name = string("op_14838")]; int32 var_14840 = const()[name = string("op_14840"), val = int32(-1)]; bool var_14841_interleave_0 = const()[name = string("op_14841_interleave_0"), val = bool(false)]; tensor var_14841 = concat(axis = var_14840, interleave = var_14841_interleave_0, values = (var_14838, var_14836_0))[name = string("op_14841")]; tensor var_14842 = mul(x = var_14841, y = sin_1)[name = string("op_14842")]; tensor q_215 = add(x = var_14835, y = var_14842)[name = string("q_215")]; bool var_14856_transpose_x_0 = const()[name = string("op_14856_transpose_x_0"), val = bool(false)]; bool var_14856_transpose_y_0 = const()[name = string("op_14856_transpose_y_0"), val = bool(false)]; tensor var_14856_cast_fp16 = matmul(transpose_x = var_14856_transpose_x_0, transpose_y = var_14856_transpose_y_0, x = q_215, y = transpose_153_cast_fp16)[name = string("op_14856_cast_fp16")]; tensor attn_weights_183_cast_fp16 = add(x = var_14856_cast_fp16, y = causal_mask)[name = string("attn_weights_183_cast_fp16")]; int32 var_14861 = const()[name = string("op_14861"), val = int32(-1)]; tensor attn_weights_185_cast_fp16 = softmax(axis = var_14861, x = attn_weights_183_cast_fp16)[name = string("attn_weights_185_cast_fp16")]; bool attn_output_181_transpose_x_0 = const()[name = string("attn_output_181_transpose_x_0"), val = bool(false)]; bool attn_output_181_transpose_y_0 = const()[name = string("attn_output_181_transpose_y_0"), val = bool(false)]; tensor attn_output_181_cast_fp16 = matmul(transpose_x = attn_output_181_transpose_x_0, transpose_y = attn_output_181_transpose_y_0, x = attn_weights_185_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_181_cast_fp16")]; tensor var_14869 = const()[name = string("op_14869"), val = tensor([0, 2, 1, 3])]; tensor var_14876 = const()[name = string("op_14876"), val = tensor([1, 1, -1])]; tensor var_14870_cast_fp16 = transpose(perm = var_14869, x = attn_output_181_cast_fp16)[name = string("transpose_34")]; tensor attn_output_183_cast_fp16 = reshape(shape = var_14876, x = var_14870_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor var_14881 = const()[name = string("op_14881"), val = tensor([0, 2, 1])]; string var_14897_pad_type_0 = const()[name = string("op_14897_pad_type_0"), val = string("valid")]; int32 var_14897_groups_0 = const()[name = string("op_14897_groups_0"), val = int32(1)]; tensor var_14897_strides_0 = const()[name = string("op_14897_strides_0"), val = tensor([1])]; tensor var_14897_pad_0 = const()[name = string("op_14897_pad_0"), val = tensor([0, 0])]; tensor var_14897_dilations_0 = const()[name = string("op_14897_dilations_0"), val = tensor([1])]; tensor squeeze_30_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1124942464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126515392))))[name = string("squeeze_30_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_14882_cast_fp16 = transpose(perm = var_14881, x = attn_output_183_cast_fp16)[name = string("transpose_33")]; tensor var_14897_cast_fp16 = conv(dilations = var_14897_dilations_0, groups = var_14897_groups_0, pad = var_14897_pad_0, pad_type = var_14897_pad_type_0, strides = var_14897_strides_0, weight = squeeze_30_cast_fp16_to_fp32_to_fp16_palettized, x = var_14882_cast_fp16)[name = string("op_14897_cast_fp16")]; tensor var_14901 = const()[name = string("op_14901"), val = tensor([0, 2, 1])]; int32 var_14907 = const()[name = string("op_14907"), val = int32(-1)]; fp16 const_465_promoted_to_fp16 = const()[name = string("const_465_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_819_cast_fp16 = transpose(perm = var_14901, x = var_14897_cast_fp16)[name = string("transpose_32")]; tensor var_14913_cast_fp16 = mul(x = x_819_cast_fp16, y = const_465_promoted_to_fp16)[name = string("op_14913_cast_fp16")]; bool input_807_interleave_0 = const()[name = string("input_807_interleave_0"), val = bool(false)]; tensor input_807_cast_fp16 = concat(axis = var_14907, interleave = input_807_interleave_0, values = (x_819_cast_fp16, var_14913_cast_fp16))[name = string("input_807_cast_fp16")]; tensor normed_789_axes_0 = const()[name = string("normed_789_axes_0"), val = tensor([-1])]; fp16 var_14905_to_fp16 = const()[name = string("op_14905_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_789_cast_fp16 = layer_norm(axes = normed_789_axes_0, epsilon = var_14905_to_fp16, x = input_807_cast_fp16)[name = string("normed_789_cast_fp16")]; tensor var_14918_split_sizes_0 = const()[name = string("op_14918_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14918_axis_0 = const()[name = string("op_14918_axis_0"), val = int32(-1)]; tensor var_14918_cast_fp16_0, tensor var_14918_cast_fp16_1 = split(axis = var_14918_axis_0, split_sizes = var_14918_split_sizes_0, x = normed_789_cast_fp16)[name = string("op_14918_cast_fp16")]; tensor const_466_to_fp16 = const()[name = string("const_466_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126516992)))]; tensor var_14921_cast_fp16 = mul(x = var_14918_cast_fp16_0, y = const_466_to_fp16)[name = string("op_14921_cast_fp16")]; tensor x_823_cast_fp16 = add(x = x_811_cast_fp16, y = var_14921_cast_fp16)[name = string("x_823_cast_fp16")]; int32 var_14928 = const()[name = string("op_14928"), val = int32(-1)]; fp16 const_467_promoted_to_fp16 = const()[name = string("const_467_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14934_cast_fp16 = mul(x = x_823_cast_fp16, y = const_467_promoted_to_fp16)[name = string("op_14934_cast_fp16")]; bool input_809_interleave_0 = const()[name = string("input_809_interleave_0"), val = bool(false)]; tensor input_809_cast_fp16 = concat(axis = var_14928, interleave = input_809_interleave_0, values = (x_823_cast_fp16, var_14934_cast_fp16))[name = string("input_809_cast_fp16")]; tensor normed_793_axes_0 = const()[name = string("normed_793_axes_0"), val = tensor([-1])]; fp16 var_14926_to_fp16 = const()[name = string("op_14926_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_793_cast_fp16 = layer_norm(axes = normed_793_axes_0, epsilon = var_14926_to_fp16, x = input_809_cast_fp16)[name = string("normed_793_cast_fp16")]; tensor var_14939_split_sizes_0 = const()[name = string("op_14939_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_14939_axis_0 = const()[name = string("op_14939_axis_0"), val = int32(-1)]; tensor var_14939_cast_fp16_0, tensor var_14939_cast_fp16_1 = split(axis = var_14939_axis_0, split_sizes = var_14939_split_sizes_0, x = normed_793_cast_fp16)[name = string("op_14939_cast_fp16")]; tensor const_468_to_fp16 = const()[name = string("const_468_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126520128)))]; tensor var_14942_cast_fp16 = mul(x = var_14939_cast_fp16_0, y = const_468_to_fp16)[name = string("op_14942_cast_fp16")]; tensor var_14955 = const()[name = string("op_14955"), val = tensor([0, 2, 1])]; tensor input_811_axes_0 = const()[name = string("input_811_axes_0"), val = tensor([2])]; tensor var_14956 = transpose(perm = var_14955, x = var_14942_cast_fp16)[name = string("transpose_31")]; tensor input_811 = expand_dims(axes = input_811_axes_0, x = var_14956)[name = string("input_811")]; string gate_121_pad_type_0 = const()[name = string("gate_121_pad_type_0"), val = string("valid")]; tensor gate_121_strides_0 = const()[name = string("gate_121_strides_0"), val = tensor([1, 1])]; tensor gate_121_pad_0 = const()[name = string("gate_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_121_dilations_0 = const()[name = string("gate_121_dilations_0"), val = tensor([1, 1])]; int32 gate_121_groups_0 = const()[name = string("gate_121_groups_0"), val = int32(1)]; tensor gate_121 = conv(dilations = gate_121_dilations_0, groups = gate_121_groups_0, pad = gate_121_pad_0, pad_type = gate_121_pad_type_0, strides = gate_121_strides_0, weight = layers_30_mlp_gate_proj_weight_palettized, x = input_811)[name = string("gate_121")]; string up_61_pad_type_0 = const()[name = string("up_61_pad_type_0"), val = string("valid")]; tensor up_61_strides_0 = const()[name = string("up_61_strides_0"), val = tensor([1, 1])]; tensor up_61_pad_0 = const()[name = string("up_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_61_dilations_0 = const()[name = string("up_61_dilations_0"), val = tensor([1, 1])]; int32 up_61_groups_0 = const()[name = string("up_61_groups_0"), val = int32(1)]; tensor up_61 = conv(dilations = up_61_dilations_0, groups = up_61_groups_0, pad = up_61_pad_0, pad_type = up_61_pad_type_0, strides = up_61_strides_0, weight = layers_30_mlp_up_proj_weight_palettized, x = input_811)[name = string("up_61")]; string gate_123_mode_0 = const()[name = string("gate_123_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_123 = gelu(mode = gate_123_mode_0, x = gate_121)[name = string("gate_123")]; tensor input_813 = mul(x = gate_123, y = up_61)[name = string("input_813")]; string mlp_out_61_pad_type_0 = const()[name = string("mlp_out_61_pad_type_0"), val = string("valid")]; tensor mlp_out_61_strides_0 = const()[name = string("mlp_out_61_strides_0"), val = tensor([1, 1])]; tensor mlp_out_61_pad_0 = const()[name = string("mlp_out_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_61_dilations_0 = const()[name = string("mlp_out_61_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_61_groups_0 = const()[name = string("mlp_out_61_groups_0"), val = int32(1)]; tensor mlp_out_61 = conv(dilations = mlp_out_61_dilations_0, groups = mlp_out_61_groups_0, pad = mlp_out_61_pad_0, pad_type = mlp_out_61_pad_type_0, strides = mlp_out_61_strides_0, weight = layers_30_mlp_down_proj_weight_palettized, x = input_813)[name = string("mlp_out_61")]; tensor var_14996_axes_0 = const()[name = string("op_14996_axes_0"), val = tensor([2])]; tensor var_14996 = squeeze(axes = var_14996_axes_0, x = mlp_out_61)[name = string("op_14996")]; tensor var_15000 = const()[name = string("op_15000"), val = tensor([0, 2, 1])]; int32 var_15006 = const()[name = string("op_15006"), val = int32(-1)]; fp16 const_469_promoted_to_fp16 = const()[name = string("const_469_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_827 = transpose(perm = var_15000, x = var_14996)[name = string("transpose_30")]; tensor var_15012_cast_fp16 = mul(x = x_827, y = const_469_promoted_to_fp16)[name = string("op_15012_cast_fp16")]; bool input_815_interleave_0 = const()[name = string("input_815_interleave_0"), val = bool(false)]; tensor input_815_cast_fp16 = concat(axis = var_15006, interleave = input_815_interleave_0, values = (x_827, var_15012_cast_fp16))[name = string("input_815_cast_fp16")]; tensor normed_797_axes_0 = const()[name = string("normed_797_axes_0"), val = tensor([-1])]; fp16 var_15004_to_fp16 = const()[name = string("op_15004_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_797_cast_fp16 = layer_norm(axes = normed_797_axes_0, epsilon = var_15004_to_fp16, x = input_815_cast_fp16)[name = string("normed_797_cast_fp16")]; tensor var_15017_split_sizes_0 = const()[name = string("op_15017_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15017_axis_0 = const()[name = string("op_15017_axis_0"), val = int32(-1)]; tensor var_15017_cast_fp16_0, tensor var_15017_cast_fp16_1 = split(axis = var_15017_axis_0, split_sizes = var_15017_split_sizes_0, x = normed_797_cast_fp16)[name = string("op_15017_cast_fp16")]; tensor const_470_to_fp16 = const()[name = string("const_470_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126523264)))]; tensor var_15020_cast_fp16 = mul(x = var_15017_cast_fp16_0, y = const_470_to_fp16)[name = string("op_15020_cast_fp16")]; tensor hidden_states_369_cast_fp16 = add(x = x_823_cast_fp16, y = var_15020_cast_fp16)[name = string("hidden_states_369_cast_fp16")]; tensor per_layer_slice_61_begin_0 = const()[name = string("per_layer_slice_61_begin_0"), val = tensor([0, 0, 7680])]; tensor per_layer_slice_61_end_0 = const()[name = string("per_layer_slice_61_end_0"), val = tensor([1, 1, 7936])]; tensor per_layer_slice_61_end_mask_0 = const()[name = string("per_layer_slice_61_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_61_cast_fp16 = slice_by_index(begin = per_layer_slice_61_begin_0, end = per_layer_slice_61_end_0, end_mask = per_layer_slice_61_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_61_cast_fp16")]; tensor gated_121 = linear(bias = linear_0_bias_0, weight = layers_30_per_layer_input_gate_weight_palettized, x = hidden_states_369_cast_fp16)[name = string("linear_60")]; string gated_123_mode_0 = const()[name = string("gated_123_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_123 = gelu(mode = gated_123_mode_0, x = gated_121)[name = string("gated_123")]; tensor input_819_cast_fp16 = mul(x = gated_123, y = per_layer_slice_61_cast_fp16)[name = string("input_819_cast_fp16")]; tensor layers_30_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126526400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126723072))))[name = string("layers_30_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_61_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_30_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_819_cast_fp16)[name = string("linear_61_cast_fp16")]; int32 var_15057 = const()[name = string("op_15057"), val = int32(-1)]; fp16 const_471_promoted_to_fp16 = const()[name = string("const_471_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15063_cast_fp16 = mul(x = linear_61_cast_fp16, y = const_471_promoted_to_fp16)[name = string("op_15063_cast_fp16")]; bool input_821_interleave_0 = const()[name = string("input_821_interleave_0"), val = bool(false)]; tensor input_821_cast_fp16 = concat(axis = var_15057, interleave = input_821_interleave_0, values = (linear_61_cast_fp16, var_15063_cast_fp16))[name = string("input_821_cast_fp16")]; tensor normed_801_axes_0 = const()[name = string("normed_801_axes_0"), val = tensor([-1])]; fp16 var_15055_to_fp16 = const()[name = string("op_15055_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_801_cast_fp16 = layer_norm(axes = normed_801_axes_0, epsilon = var_15055_to_fp16, x = input_821_cast_fp16)[name = string("normed_801_cast_fp16")]; tensor var_15068_split_sizes_0 = const()[name = string("op_15068_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15068_axis_0 = const()[name = string("op_15068_axis_0"), val = int32(-1)]; tensor var_15068_cast_fp16_0, tensor var_15068_cast_fp16_1 = split(axis = var_15068_axis_0, split_sizes = var_15068_split_sizes_0, x = normed_801_cast_fp16)[name = string("op_15068_cast_fp16")]; tensor const_472_to_fp16 = const()[name = string("const_472_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126724672)))]; tensor var_15071_cast_fp16 = mul(x = var_15068_cast_fp16_0, y = const_472_to_fp16)[name = string("op_15071_cast_fp16")]; tensor hidden_states_373_cast_fp16 = add(x = hidden_states_369_cast_fp16, y = var_15071_cast_fp16)[name = string("hidden_states_373_cast_fp16")]; tensor layers_30_layer_scalar_to_fp16 = const()[name = string("layers_30_layer_scalar_to_fp16"), val = tensor([0x1.bep-1])]; tensor x_835_cast_fp16 = mul(x = hidden_states_373_cast_fp16, y = layers_30_layer_scalar_to_fp16)[name = string("x_835_cast_fp16")]; int32 var_15079 = const()[name = string("op_15079"), val = int32(-1)]; fp16 const_473_promoted_to_fp16 = const()[name = string("const_473_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15085_cast_fp16 = mul(x = x_835_cast_fp16, y = const_473_promoted_to_fp16)[name = string("op_15085_cast_fp16")]; bool input_823_interleave_0 = const()[name = string("input_823_interleave_0"), val = bool(false)]; tensor input_823_cast_fp16 = concat(axis = var_15079, interleave = input_823_interleave_0, values = (x_835_cast_fp16, var_15085_cast_fp16))[name = string("input_823_cast_fp16")]; tensor normed_805_axes_0 = const()[name = string("normed_805_axes_0"), val = tensor([-1])]; fp16 var_15077_to_fp16 = const()[name = string("op_15077_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_805_cast_fp16 = layer_norm(axes = normed_805_axes_0, epsilon = var_15077_to_fp16, x = input_823_cast_fp16)[name = string("normed_805_cast_fp16")]; tensor var_15090_split_sizes_0 = const()[name = string("op_15090_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15090_axis_0 = const()[name = string("op_15090_axis_0"), val = int32(-1)]; tensor var_15090_cast_fp16_0, tensor var_15090_cast_fp16_1 = split(axis = var_15090_axis_0, split_sizes = var_15090_split_sizes_0, x = normed_805_cast_fp16)[name = string("op_15090_cast_fp16")]; tensor const_474_to_fp16 = const()[name = string("const_474_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126727808)))]; tensor var_15093_cast_fp16 = mul(x = var_15090_cast_fp16_0, y = const_474_to_fp16)[name = string("op_15093_cast_fp16")]; tensor var_15101 = const()[name = string("op_15101"), val = tensor([0, 2, 1])]; tensor var_15104_axes_0 = const()[name = string("op_15104_axes_0"), val = tensor([2])]; tensor var_15102_cast_fp16 = transpose(perm = var_15101, x = var_15093_cast_fp16)[name = string("transpose_29")]; tensor var_15104_cast_fp16 = expand_dims(axes = var_15104_axes_0, x = var_15102_cast_fp16)[name = string("op_15104_cast_fp16")]; string var_15120_pad_type_0 = const()[name = string("op_15120_pad_type_0"), val = string("valid")]; tensor var_15120_strides_0 = const()[name = string("op_15120_strides_0"), val = tensor([1, 1])]; tensor var_15120_pad_0 = const()[name = string("op_15120_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_15120_dilations_0 = const()[name = string("op_15120_dilations_0"), val = tensor([1, 1])]; int32 var_15120_groups_0 = const()[name = string("op_15120_groups_0"), val = int32(1)]; tensor var_15120 = conv(dilations = var_15120_dilations_0, groups = var_15120_groups_0, pad = var_15120_pad_0, pad_type = var_15120_pad_type_0, strides = var_15120_strides_0, weight = layers_31_self_attn_q_proj_weight_palettized, x = var_15104_cast_fp16)[name = string("op_15120")]; tensor var_15125 = const()[name = string("op_15125"), val = tensor([1, 8, 256, 1])]; tensor var_15126 = reshape(shape = var_15125, x = var_15120)[name = string("op_15126")]; tensor var_15131 = const()[name = string("op_15131"), val = tensor([0, 1, 3, 2])]; tensor var_15141 = const()[name = string("op_15141"), val = tensor([1, 8, 256])]; tensor var_15132 = transpose(perm = var_15131, x = var_15126)[name = string("transpose_28")]; tensor x_839 = reshape(shape = var_15141, x = var_15132)[name = string("x_839")]; int32 var_15147 = const()[name = string("op_15147"), val = int32(-1)]; fp16 const_475_promoted_to_fp16 = const()[name = string("const_475_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15153_cast_fp16 = mul(x = x_839, y = const_475_promoted_to_fp16)[name = string("op_15153_cast_fp16")]; bool input_827_interleave_0 = const()[name = string("input_827_interleave_0"), val = bool(false)]; tensor input_827_cast_fp16 = concat(axis = var_15147, interleave = input_827_interleave_0, values = (x_839, var_15153_cast_fp16))[name = string("input_827_cast_fp16")]; tensor normed_809_axes_0 = const()[name = string("normed_809_axes_0"), val = tensor([-1])]; fp16 var_15145_to_fp16 = const()[name = string("op_15145_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_809_cast_fp16 = layer_norm(axes = normed_809_axes_0, epsilon = var_15145_to_fp16, x = input_827_cast_fp16)[name = string("normed_809_cast_fp16")]; tensor var_15158_split_sizes_0 = const()[name = string("op_15158_split_sizes_0"), val = tensor([256, 256])]; int32 var_15158_axis_0 = const()[name = string("op_15158_axis_0"), val = int32(-1)]; tensor var_15158_cast_fp16_0, tensor var_15158_cast_fp16_1 = split(axis = var_15158_axis_0, split_sizes = var_15158_split_sizes_0, x = normed_809_cast_fp16)[name = string("op_15158_cast_fp16")]; tensor var_15161_cast_fp16 = mul(x = var_15158_cast_fp16_0, y = const_234_to_fp16)[name = string("op_15161_cast_fp16")]; tensor var_15167 = const()[name = string("op_15167"), val = tensor([1, 8, 1, 256])]; tensor q_219 = reshape(shape = var_15167, x = var_15161_cast_fp16)[name = string("q_219")]; tensor var_15169 = mul(x = q_219, y = cos_1)[name = string("op_15169")]; tensor var_15170_split_sizes_0 = const()[name = string("op_15170_split_sizes_0"), val = tensor([128, 128])]; int32 var_15170_axis_0 = const()[name = string("op_15170_axis_0"), val = int32(-1)]; tensor var_15170_0, tensor var_15170_1 = split(axis = var_15170_axis_0, split_sizes = var_15170_split_sizes_0, x = q_219)[name = string("op_15170")]; fp16 const_477_promoted = const()[name = string("const_477_promoted"), val = fp16(-0x1p+0)]; tensor var_15172 = mul(x = var_15170_1, y = const_477_promoted)[name = string("op_15172")]; int32 var_15174 = const()[name = string("op_15174"), val = int32(-1)]; bool var_15175_interleave_0 = const()[name = string("op_15175_interleave_0"), val = bool(false)]; tensor var_15175 = concat(axis = var_15174, interleave = var_15175_interleave_0, values = (var_15172, var_15170_0))[name = string("op_15175")]; tensor var_15176 = mul(x = var_15175, y = sin_1)[name = string("op_15176")]; tensor q_221 = add(x = var_15169, y = var_15176)[name = string("q_221")]; bool var_15190_transpose_x_0 = const()[name = string("op_15190_transpose_x_0"), val = bool(false)]; bool var_15190_transpose_y_0 = const()[name = string("op_15190_transpose_y_0"), val = bool(false)]; tensor var_15190_cast_fp16 = matmul(transpose_x = var_15190_transpose_x_0, transpose_y = var_15190_transpose_y_0, x = q_221, y = transpose_153_cast_fp16)[name = string("op_15190_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = var_15190_cast_fp16, y = causal_mask)[name = string("attn_weights_189_cast_fp16")]; int32 var_15195 = const()[name = string("op_15195"), val = int32(-1)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_15195, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_187_transpose_x_0 = const()[name = string("attn_output_187_transpose_x_0"), val = bool(false)]; bool attn_output_187_transpose_y_0 = const()[name = string("attn_output_187_transpose_y_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = matmul(transpose_x = attn_output_187_transpose_x_0, transpose_y = attn_output_187_transpose_y_0, x = attn_weights_191_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_187_cast_fp16")]; tensor var_15203 = const()[name = string("op_15203"), val = tensor([0, 2, 1, 3])]; tensor var_15210 = const()[name = string("op_15210"), val = tensor([1, 1, -1])]; tensor var_15204_cast_fp16 = transpose(perm = var_15203, x = attn_output_187_cast_fp16)[name = string("transpose_27")]; tensor attn_output_189_cast_fp16 = reshape(shape = var_15210, x = var_15204_cast_fp16)[name = string("attn_output_189_cast_fp16")]; tensor var_15215 = const()[name = string("op_15215"), val = tensor([0, 2, 1])]; string var_15231_pad_type_0 = const()[name = string("op_15231_pad_type_0"), val = string("valid")]; int32 var_15231_groups_0 = const()[name = string("op_15231_groups_0"), val = int32(1)]; tensor var_15231_strides_0 = const()[name = string("op_15231_strides_0"), val = tensor([1])]; tensor var_15231_pad_0 = const()[name = string("op_15231_pad_0"), val = tensor([0, 0])]; tensor var_15231_dilations_0 = const()[name = string("op_15231_dilations_0"), val = tensor([1])]; tensor squeeze_31_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1126730944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128303872))))[name = string("squeeze_31_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_15216_cast_fp16 = transpose(perm = var_15215, x = attn_output_189_cast_fp16)[name = string("transpose_26")]; tensor var_15231_cast_fp16 = conv(dilations = var_15231_dilations_0, groups = var_15231_groups_0, pad = var_15231_pad_0, pad_type = var_15231_pad_type_0, strides = var_15231_strides_0, weight = squeeze_31_cast_fp16_to_fp32_to_fp16_palettized, x = var_15216_cast_fp16)[name = string("op_15231_cast_fp16")]; tensor var_15235 = const()[name = string("op_15235"), val = tensor([0, 2, 1])]; int32 var_15241 = const()[name = string("op_15241"), val = int32(-1)]; fp16 const_478_promoted_to_fp16 = const()[name = string("const_478_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_843_cast_fp16 = transpose(perm = var_15235, x = var_15231_cast_fp16)[name = string("transpose_25")]; tensor var_15247_cast_fp16 = mul(x = x_843_cast_fp16, y = const_478_promoted_to_fp16)[name = string("op_15247_cast_fp16")]; bool input_831_interleave_0 = const()[name = string("input_831_interleave_0"), val = bool(false)]; tensor input_831_cast_fp16 = concat(axis = var_15241, interleave = input_831_interleave_0, values = (x_843_cast_fp16, var_15247_cast_fp16))[name = string("input_831_cast_fp16")]; tensor normed_813_axes_0 = const()[name = string("normed_813_axes_0"), val = tensor([-1])]; fp16 var_15239_to_fp16 = const()[name = string("op_15239_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_813_cast_fp16 = layer_norm(axes = normed_813_axes_0, epsilon = var_15239_to_fp16, x = input_831_cast_fp16)[name = string("normed_813_cast_fp16")]; tensor var_15252_split_sizes_0 = const()[name = string("op_15252_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15252_axis_0 = const()[name = string("op_15252_axis_0"), val = int32(-1)]; tensor var_15252_cast_fp16_0, tensor var_15252_cast_fp16_1 = split(axis = var_15252_axis_0, split_sizes = var_15252_split_sizes_0, x = normed_813_cast_fp16)[name = string("op_15252_cast_fp16")]; tensor const_479_to_fp16 = const()[name = string("const_479_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128305472)))]; tensor var_15255_cast_fp16 = mul(x = var_15252_cast_fp16_0, y = const_479_to_fp16)[name = string("op_15255_cast_fp16")]; tensor x_847_cast_fp16 = add(x = x_835_cast_fp16, y = var_15255_cast_fp16)[name = string("x_847_cast_fp16")]; int32 var_15262 = const()[name = string("op_15262"), val = int32(-1)]; fp16 const_480_promoted_to_fp16 = const()[name = string("const_480_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15268_cast_fp16 = mul(x = x_847_cast_fp16, y = const_480_promoted_to_fp16)[name = string("op_15268_cast_fp16")]; bool input_833_interleave_0 = const()[name = string("input_833_interleave_0"), val = bool(false)]; tensor input_833_cast_fp16 = concat(axis = var_15262, interleave = input_833_interleave_0, values = (x_847_cast_fp16, var_15268_cast_fp16))[name = string("input_833_cast_fp16")]; tensor normed_817_axes_0 = const()[name = string("normed_817_axes_0"), val = tensor([-1])]; fp16 var_15260_to_fp16 = const()[name = string("op_15260_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_817_cast_fp16 = layer_norm(axes = normed_817_axes_0, epsilon = var_15260_to_fp16, x = input_833_cast_fp16)[name = string("normed_817_cast_fp16")]; tensor var_15273_split_sizes_0 = const()[name = string("op_15273_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15273_axis_0 = const()[name = string("op_15273_axis_0"), val = int32(-1)]; tensor var_15273_cast_fp16_0, tensor var_15273_cast_fp16_1 = split(axis = var_15273_axis_0, split_sizes = var_15273_split_sizes_0, x = normed_817_cast_fp16)[name = string("op_15273_cast_fp16")]; tensor const_481_to_fp16 = const()[name = string("const_481_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128308608)))]; tensor var_15276_cast_fp16 = mul(x = var_15273_cast_fp16_0, y = const_481_to_fp16)[name = string("op_15276_cast_fp16")]; tensor var_15289 = const()[name = string("op_15289"), val = tensor([0, 2, 1])]; tensor input_835_axes_0 = const()[name = string("input_835_axes_0"), val = tensor([2])]; tensor var_15290 = transpose(perm = var_15289, x = var_15276_cast_fp16)[name = string("transpose_24")]; tensor input_835 = expand_dims(axes = input_835_axes_0, x = var_15290)[name = string("input_835")]; string gate_125_pad_type_0 = const()[name = string("gate_125_pad_type_0"), val = string("valid")]; tensor gate_125_strides_0 = const()[name = string("gate_125_strides_0"), val = tensor([1, 1])]; tensor gate_125_pad_0 = const()[name = string("gate_125_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_125_dilations_0 = const()[name = string("gate_125_dilations_0"), val = tensor([1, 1])]; int32 gate_125_groups_0 = const()[name = string("gate_125_groups_0"), val = int32(1)]; tensor gate_125 = conv(dilations = gate_125_dilations_0, groups = gate_125_groups_0, pad = gate_125_pad_0, pad_type = gate_125_pad_type_0, strides = gate_125_strides_0, weight = layers_31_mlp_gate_proj_weight_palettized, x = input_835)[name = string("gate_125")]; string up_63_pad_type_0 = const()[name = string("up_63_pad_type_0"), val = string("valid")]; tensor up_63_strides_0 = const()[name = string("up_63_strides_0"), val = tensor([1, 1])]; tensor up_63_pad_0 = const()[name = string("up_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_63_dilations_0 = const()[name = string("up_63_dilations_0"), val = tensor([1, 1])]; int32 up_63_groups_0 = const()[name = string("up_63_groups_0"), val = int32(1)]; tensor up_63 = conv(dilations = up_63_dilations_0, groups = up_63_groups_0, pad = up_63_pad_0, pad_type = up_63_pad_type_0, strides = up_63_strides_0, weight = layers_31_mlp_up_proj_weight_palettized, x = input_835)[name = string("up_63")]; string gate_127_mode_0 = const()[name = string("gate_127_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_127 = gelu(mode = gate_127_mode_0, x = gate_125)[name = string("gate_127")]; tensor input_837 = mul(x = gate_127, y = up_63)[name = string("input_837")]; string mlp_out_63_pad_type_0 = const()[name = string("mlp_out_63_pad_type_0"), val = string("valid")]; tensor mlp_out_63_strides_0 = const()[name = string("mlp_out_63_strides_0"), val = tensor([1, 1])]; tensor mlp_out_63_pad_0 = const()[name = string("mlp_out_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_63_dilations_0 = const()[name = string("mlp_out_63_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_63_groups_0 = const()[name = string("mlp_out_63_groups_0"), val = int32(1)]; tensor mlp_out_63 = conv(dilations = mlp_out_63_dilations_0, groups = mlp_out_63_groups_0, pad = mlp_out_63_pad_0, pad_type = mlp_out_63_pad_type_0, strides = mlp_out_63_strides_0, weight = layers_31_mlp_down_proj_weight_palettized, x = input_837)[name = string("mlp_out_63")]; tensor var_15330_axes_0 = const()[name = string("op_15330_axes_0"), val = tensor([2])]; tensor var_15330 = squeeze(axes = var_15330_axes_0, x = mlp_out_63)[name = string("op_15330")]; tensor var_15334 = const()[name = string("op_15334"), val = tensor([0, 2, 1])]; int32 var_15340 = const()[name = string("op_15340"), val = int32(-1)]; fp16 const_482_promoted_to_fp16 = const()[name = string("const_482_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_851 = transpose(perm = var_15334, x = var_15330)[name = string("transpose_23")]; tensor var_15346_cast_fp16 = mul(x = x_851, y = const_482_promoted_to_fp16)[name = string("op_15346_cast_fp16")]; bool input_839_interleave_0 = const()[name = string("input_839_interleave_0"), val = bool(false)]; tensor input_839_cast_fp16 = concat(axis = var_15340, interleave = input_839_interleave_0, values = (x_851, var_15346_cast_fp16))[name = string("input_839_cast_fp16")]; tensor normed_821_axes_0 = const()[name = string("normed_821_axes_0"), val = tensor([-1])]; fp16 var_15338_to_fp16 = const()[name = string("op_15338_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_821_cast_fp16 = layer_norm(axes = normed_821_axes_0, epsilon = var_15338_to_fp16, x = input_839_cast_fp16)[name = string("normed_821_cast_fp16")]; tensor var_15351_split_sizes_0 = const()[name = string("op_15351_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15351_axis_0 = const()[name = string("op_15351_axis_0"), val = int32(-1)]; tensor var_15351_cast_fp16_0, tensor var_15351_cast_fp16_1 = split(axis = var_15351_axis_0, split_sizes = var_15351_split_sizes_0, x = normed_821_cast_fp16)[name = string("op_15351_cast_fp16")]; tensor const_483_to_fp16 = const()[name = string("const_483_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128311744)))]; tensor var_15354_cast_fp16 = mul(x = var_15351_cast_fp16_0, y = const_483_to_fp16)[name = string("op_15354_cast_fp16")]; tensor hidden_states_381_cast_fp16 = add(x = x_847_cast_fp16, y = var_15354_cast_fp16)[name = string("hidden_states_381_cast_fp16")]; tensor per_layer_slice_63_begin_0 = const()[name = string("per_layer_slice_63_begin_0"), val = tensor([0, 0, 7936])]; tensor per_layer_slice_63_end_0 = const()[name = string("per_layer_slice_63_end_0"), val = tensor([1, 1, 8192])]; tensor per_layer_slice_63_end_mask_0 = const()[name = string("per_layer_slice_63_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_63_cast_fp16 = slice_by_index(begin = per_layer_slice_63_begin_0, end = per_layer_slice_63_end_0, end_mask = per_layer_slice_63_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_63_cast_fp16")]; tensor gated_125 = linear(bias = linear_0_bias_0, weight = layers_31_per_layer_input_gate_weight_palettized, x = hidden_states_381_cast_fp16)[name = string("linear_62")]; string gated_127_mode_0 = const()[name = string("gated_127_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_127 = gelu(mode = gated_127_mode_0, x = gated_125)[name = string("gated_127")]; tensor input_843_cast_fp16 = mul(x = gated_127, y = per_layer_slice_63_cast_fp16)[name = string("input_843_cast_fp16")]; tensor layers_31_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128314880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128511552))))[name = string("layers_31_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_63_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_31_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_843_cast_fp16)[name = string("linear_63_cast_fp16")]; int32 var_15391 = const()[name = string("op_15391"), val = int32(-1)]; fp16 const_484_promoted_to_fp16 = const()[name = string("const_484_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15397_cast_fp16 = mul(x = linear_63_cast_fp16, y = const_484_promoted_to_fp16)[name = string("op_15397_cast_fp16")]; bool input_845_interleave_0 = const()[name = string("input_845_interleave_0"), val = bool(false)]; tensor input_845_cast_fp16 = concat(axis = var_15391, interleave = input_845_interleave_0, values = (linear_63_cast_fp16, var_15397_cast_fp16))[name = string("input_845_cast_fp16")]; tensor normed_825_axes_0 = const()[name = string("normed_825_axes_0"), val = tensor([-1])]; fp16 var_15389_to_fp16 = const()[name = string("op_15389_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_825_cast_fp16 = layer_norm(axes = normed_825_axes_0, epsilon = var_15389_to_fp16, x = input_845_cast_fp16)[name = string("normed_825_cast_fp16")]; tensor var_15402_split_sizes_0 = const()[name = string("op_15402_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15402_axis_0 = const()[name = string("op_15402_axis_0"), val = int32(-1)]; tensor var_15402_cast_fp16_0, tensor var_15402_cast_fp16_1 = split(axis = var_15402_axis_0, split_sizes = var_15402_split_sizes_0, x = normed_825_cast_fp16)[name = string("op_15402_cast_fp16")]; tensor const_485_to_fp16 = const()[name = string("const_485_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128513152)))]; tensor var_15405_cast_fp16 = mul(x = var_15402_cast_fp16_0, y = const_485_to_fp16)[name = string("op_15405_cast_fp16")]; tensor hidden_states_385_cast_fp16 = add(x = hidden_states_381_cast_fp16, y = var_15405_cast_fp16)[name = string("hidden_states_385_cast_fp16")]; tensor layers_31_layer_scalar_to_fp16 = const()[name = string("layers_31_layer_scalar_to_fp16"), val = tensor([0x1.a8p-1])]; tensor x_859_cast_fp16 = mul(x = hidden_states_385_cast_fp16, y = layers_31_layer_scalar_to_fp16)[name = string("x_859_cast_fp16")]; int32 var_15413 = const()[name = string("op_15413"), val = int32(-1)]; fp16 const_486_promoted_to_fp16 = const()[name = string("const_486_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15419_cast_fp16 = mul(x = x_859_cast_fp16, y = const_486_promoted_to_fp16)[name = string("op_15419_cast_fp16")]; bool input_847_interleave_0 = const()[name = string("input_847_interleave_0"), val = bool(false)]; tensor input_847_cast_fp16 = concat(axis = var_15413, interleave = input_847_interleave_0, values = (x_859_cast_fp16, var_15419_cast_fp16))[name = string("input_847_cast_fp16")]; tensor normed_829_axes_0 = const()[name = string("normed_829_axes_0"), val = tensor([-1])]; fp16 var_15411_to_fp16 = const()[name = string("op_15411_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_829_cast_fp16 = layer_norm(axes = normed_829_axes_0, epsilon = var_15411_to_fp16, x = input_847_cast_fp16)[name = string("normed_829_cast_fp16")]; tensor var_15424_split_sizes_0 = const()[name = string("op_15424_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15424_axis_0 = const()[name = string("op_15424_axis_0"), val = int32(-1)]; tensor var_15424_cast_fp16_0, tensor var_15424_cast_fp16_1 = split(axis = var_15424_axis_0, split_sizes = var_15424_split_sizes_0, x = normed_829_cast_fp16)[name = string("op_15424_cast_fp16")]; tensor const_487_to_fp16 = const()[name = string("const_487_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128516288)))]; tensor var_15427_cast_fp16 = mul(x = var_15424_cast_fp16_0, y = const_487_to_fp16)[name = string("op_15427_cast_fp16")]; tensor var_15435 = const()[name = string("op_15435"), val = tensor([0, 2, 1])]; tensor var_15438_axes_0 = const()[name = string("op_15438_axes_0"), val = tensor([2])]; tensor var_15436_cast_fp16 = transpose(perm = var_15435, x = var_15427_cast_fp16)[name = string("transpose_22")]; tensor var_15438_cast_fp16 = expand_dims(axes = var_15438_axes_0, x = var_15436_cast_fp16)[name = string("op_15438_cast_fp16")]; string var_15454_pad_type_0 = const()[name = string("op_15454_pad_type_0"), val = string("valid")]; tensor var_15454_strides_0 = const()[name = string("op_15454_strides_0"), val = tensor([1, 1])]; tensor var_15454_pad_0 = const()[name = string("op_15454_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_15454_dilations_0 = const()[name = string("op_15454_dilations_0"), val = tensor([1, 1])]; int32 var_15454_groups_0 = const()[name = string("op_15454_groups_0"), val = int32(1)]; tensor var_15454 = conv(dilations = var_15454_dilations_0, groups = var_15454_groups_0, pad = var_15454_pad_0, pad_type = var_15454_pad_type_0, strides = var_15454_strides_0, weight = layers_32_self_attn_q_proj_weight_palettized, x = var_15438_cast_fp16)[name = string("op_15454")]; tensor var_15459 = const()[name = string("op_15459"), val = tensor([1, 8, 256, 1])]; tensor var_15460 = reshape(shape = var_15459, x = var_15454)[name = string("op_15460")]; tensor var_15465 = const()[name = string("op_15465"), val = tensor([0, 1, 3, 2])]; tensor var_15475 = const()[name = string("op_15475"), val = tensor([1, 8, 256])]; tensor var_15466 = transpose(perm = var_15465, x = var_15460)[name = string("transpose_21")]; tensor x_863 = reshape(shape = var_15475, x = var_15466)[name = string("x_863")]; int32 var_15481 = const()[name = string("op_15481"), val = int32(-1)]; fp16 const_488_promoted_to_fp16 = const()[name = string("const_488_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15487_cast_fp16 = mul(x = x_863, y = const_488_promoted_to_fp16)[name = string("op_15487_cast_fp16")]; bool input_851_interleave_0 = const()[name = string("input_851_interleave_0"), val = bool(false)]; tensor input_851_cast_fp16 = concat(axis = var_15481, interleave = input_851_interleave_0, values = (x_863, var_15487_cast_fp16))[name = string("input_851_cast_fp16")]; tensor normed_833_axes_0 = const()[name = string("normed_833_axes_0"), val = tensor([-1])]; fp16 var_15479_to_fp16 = const()[name = string("op_15479_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_833_cast_fp16 = layer_norm(axes = normed_833_axes_0, epsilon = var_15479_to_fp16, x = input_851_cast_fp16)[name = string("normed_833_cast_fp16")]; tensor var_15492_split_sizes_0 = const()[name = string("op_15492_split_sizes_0"), val = tensor([256, 256])]; int32 var_15492_axis_0 = const()[name = string("op_15492_axis_0"), val = int32(-1)]; tensor var_15492_cast_fp16_0, tensor var_15492_cast_fp16_1 = split(axis = var_15492_axis_0, split_sizes = var_15492_split_sizes_0, x = normed_833_cast_fp16)[name = string("op_15492_cast_fp16")]; tensor var_15495_cast_fp16 = mul(x = var_15492_cast_fp16_0, y = const_234_to_fp16)[name = string("op_15495_cast_fp16")]; tensor var_15501 = const()[name = string("op_15501"), val = tensor([1, 8, 1, 256])]; tensor q_225 = reshape(shape = var_15501, x = var_15495_cast_fp16)[name = string("q_225")]; tensor var_15503 = mul(x = q_225, y = cos_1)[name = string("op_15503")]; tensor var_15504_split_sizes_0 = const()[name = string("op_15504_split_sizes_0"), val = tensor([128, 128])]; int32 var_15504_axis_0 = const()[name = string("op_15504_axis_0"), val = int32(-1)]; tensor var_15504_0, tensor var_15504_1 = split(axis = var_15504_axis_0, split_sizes = var_15504_split_sizes_0, x = q_225)[name = string("op_15504")]; fp16 const_490_promoted = const()[name = string("const_490_promoted"), val = fp16(-0x1p+0)]; tensor var_15506 = mul(x = var_15504_1, y = const_490_promoted)[name = string("op_15506")]; int32 var_15508 = const()[name = string("op_15508"), val = int32(-1)]; bool var_15509_interleave_0 = const()[name = string("op_15509_interleave_0"), val = bool(false)]; tensor var_15509 = concat(axis = var_15508, interleave = var_15509_interleave_0, values = (var_15506, var_15504_0))[name = string("op_15509")]; tensor var_15510 = mul(x = var_15509, y = sin_1)[name = string("op_15510")]; tensor q_227 = add(x = var_15503, y = var_15510)[name = string("q_227")]; bool var_15524_transpose_x_0 = const()[name = string("op_15524_transpose_x_0"), val = bool(false)]; bool var_15524_transpose_y_0 = const()[name = string("op_15524_transpose_y_0"), val = bool(false)]; tensor var_15524_cast_fp16 = matmul(transpose_x = var_15524_transpose_x_0, transpose_y = var_15524_transpose_y_0, x = q_227, y = transpose_153_cast_fp16)[name = string("op_15524_cast_fp16")]; tensor attn_weights_195_cast_fp16 = add(x = var_15524_cast_fp16, y = causal_mask)[name = string("attn_weights_195_cast_fp16")]; int32 var_15529 = const()[name = string("op_15529"), val = int32(-1)]; tensor attn_weights_197_cast_fp16 = softmax(axis = var_15529, x = attn_weights_195_cast_fp16)[name = string("attn_weights_197_cast_fp16")]; bool attn_output_193_transpose_x_0 = const()[name = string("attn_output_193_transpose_x_0"), val = bool(false)]; bool attn_output_193_transpose_y_0 = const()[name = string("attn_output_193_transpose_y_0"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_0, transpose_y = attn_output_193_transpose_y_0, x = attn_weights_197_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_193_cast_fp16")]; tensor var_15537 = const()[name = string("op_15537"), val = tensor([0, 2, 1, 3])]; tensor var_15544 = const()[name = string("op_15544"), val = tensor([1, 1, -1])]; tensor var_15538_cast_fp16 = transpose(perm = var_15537, x = attn_output_193_cast_fp16)[name = string("transpose_20")]; tensor attn_output_195_cast_fp16 = reshape(shape = var_15544, x = var_15538_cast_fp16)[name = string("attn_output_195_cast_fp16")]; tensor var_15549 = const()[name = string("op_15549"), val = tensor([0, 2, 1])]; string var_15565_pad_type_0 = const()[name = string("op_15565_pad_type_0"), val = string("valid")]; int32 var_15565_groups_0 = const()[name = string("op_15565_groups_0"), val = int32(1)]; tensor var_15565_strides_0 = const()[name = string("op_15565_strides_0"), val = tensor([1])]; tensor var_15565_pad_0 = const()[name = string("op_15565_pad_0"), val = tensor([0, 0])]; tensor var_15565_dilations_0 = const()[name = string("op_15565_dilations_0"), val = tensor([1])]; tensor squeeze_32_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1128519424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130092352))))[name = string("squeeze_32_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_15550_cast_fp16 = transpose(perm = var_15549, x = attn_output_195_cast_fp16)[name = string("transpose_19")]; tensor var_15565_cast_fp16 = conv(dilations = var_15565_dilations_0, groups = var_15565_groups_0, pad = var_15565_pad_0, pad_type = var_15565_pad_type_0, strides = var_15565_strides_0, weight = squeeze_32_cast_fp16_to_fp32_to_fp16_palettized, x = var_15550_cast_fp16)[name = string("op_15565_cast_fp16")]; tensor var_15569 = const()[name = string("op_15569"), val = tensor([0, 2, 1])]; int32 var_15575 = const()[name = string("op_15575"), val = int32(-1)]; fp16 const_491_promoted_to_fp16 = const()[name = string("const_491_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_867_cast_fp16 = transpose(perm = var_15569, x = var_15565_cast_fp16)[name = string("transpose_18")]; tensor var_15581_cast_fp16 = mul(x = x_867_cast_fp16, y = const_491_promoted_to_fp16)[name = string("op_15581_cast_fp16")]; bool input_855_interleave_0 = const()[name = string("input_855_interleave_0"), val = bool(false)]; tensor input_855_cast_fp16 = concat(axis = var_15575, interleave = input_855_interleave_0, values = (x_867_cast_fp16, var_15581_cast_fp16))[name = string("input_855_cast_fp16")]; tensor normed_837_axes_0 = const()[name = string("normed_837_axes_0"), val = tensor([-1])]; fp16 var_15573_to_fp16 = const()[name = string("op_15573_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_837_cast_fp16 = layer_norm(axes = normed_837_axes_0, epsilon = var_15573_to_fp16, x = input_855_cast_fp16)[name = string("normed_837_cast_fp16")]; tensor var_15586_split_sizes_0 = const()[name = string("op_15586_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15586_axis_0 = const()[name = string("op_15586_axis_0"), val = int32(-1)]; tensor var_15586_cast_fp16_0, tensor var_15586_cast_fp16_1 = split(axis = var_15586_axis_0, split_sizes = var_15586_split_sizes_0, x = normed_837_cast_fp16)[name = string("op_15586_cast_fp16")]; tensor const_492_to_fp16 = const()[name = string("const_492_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130093952)))]; tensor var_15589_cast_fp16 = mul(x = var_15586_cast_fp16_0, y = const_492_to_fp16)[name = string("op_15589_cast_fp16")]; tensor x_871_cast_fp16 = add(x = x_859_cast_fp16, y = var_15589_cast_fp16)[name = string("x_871_cast_fp16")]; int32 var_15596 = const()[name = string("op_15596"), val = int32(-1)]; fp16 const_493_promoted_to_fp16 = const()[name = string("const_493_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15602_cast_fp16 = mul(x = x_871_cast_fp16, y = const_493_promoted_to_fp16)[name = string("op_15602_cast_fp16")]; bool input_857_interleave_0 = const()[name = string("input_857_interleave_0"), val = bool(false)]; tensor input_857_cast_fp16 = concat(axis = var_15596, interleave = input_857_interleave_0, values = (x_871_cast_fp16, var_15602_cast_fp16))[name = string("input_857_cast_fp16")]; tensor normed_841_axes_0 = const()[name = string("normed_841_axes_0"), val = tensor([-1])]; fp16 var_15594_to_fp16 = const()[name = string("op_15594_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_841_cast_fp16 = layer_norm(axes = normed_841_axes_0, epsilon = var_15594_to_fp16, x = input_857_cast_fp16)[name = string("normed_841_cast_fp16")]; tensor var_15607_split_sizes_0 = const()[name = string("op_15607_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15607_axis_0 = const()[name = string("op_15607_axis_0"), val = int32(-1)]; tensor var_15607_cast_fp16_0, tensor var_15607_cast_fp16_1 = split(axis = var_15607_axis_0, split_sizes = var_15607_split_sizes_0, x = normed_841_cast_fp16)[name = string("op_15607_cast_fp16")]; tensor const_494_to_fp16 = const()[name = string("const_494_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130097088)))]; tensor var_15610_cast_fp16 = mul(x = var_15607_cast_fp16_0, y = const_494_to_fp16)[name = string("op_15610_cast_fp16")]; tensor var_15623 = const()[name = string("op_15623"), val = tensor([0, 2, 1])]; tensor input_859_axes_0 = const()[name = string("input_859_axes_0"), val = tensor([2])]; tensor var_15624 = transpose(perm = var_15623, x = var_15610_cast_fp16)[name = string("transpose_17")]; tensor input_859 = expand_dims(axes = input_859_axes_0, x = var_15624)[name = string("input_859")]; string gate_129_pad_type_0 = const()[name = string("gate_129_pad_type_0"), val = string("valid")]; tensor gate_129_strides_0 = const()[name = string("gate_129_strides_0"), val = tensor([1, 1])]; tensor gate_129_pad_0 = const()[name = string("gate_129_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_129_dilations_0 = const()[name = string("gate_129_dilations_0"), val = tensor([1, 1])]; int32 gate_129_groups_0 = const()[name = string("gate_129_groups_0"), val = int32(1)]; tensor gate_129 = conv(dilations = gate_129_dilations_0, groups = gate_129_groups_0, pad = gate_129_pad_0, pad_type = gate_129_pad_type_0, strides = gate_129_strides_0, weight = layers_32_mlp_gate_proj_weight_palettized, x = input_859)[name = string("gate_129")]; string up_65_pad_type_0 = const()[name = string("up_65_pad_type_0"), val = string("valid")]; tensor up_65_strides_0 = const()[name = string("up_65_strides_0"), val = tensor([1, 1])]; tensor up_65_pad_0 = const()[name = string("up_65_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_65_dilations_0 = const()[name = string("up_65_dilations_0"), val = tensor([1, 1])]; int32 up_65_groups_0 = const()[name = string("up_65_groups_0"), val = int32(1)]; tensor up_65 = conv(dilations = up_65_dilations_0, groups = up_65_groups_0, pad = up_65_pad_0, pad_type = up_65_pad_type_0, strides = up_65_strides_0, weight = layers_32_mlp_up_proj_weight_palettized, x = input_859)[name = string("up_65")]; string gate_131_mode_0 = const()[name = string("gate_131_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_131 = gelu(mode = gate_131_mode_0, x = gate_129)[name = string("gate_131")]; tensor input_861 = mul(x = gate_131, y = up_65)[name = string("input_861")]; string mlp_out_65_pad_type_0 = const()[name = string("mlp_out_65_pad_type_0"), val = string("valid")]; tensor mlp_out_65_strides_0 = const()[name = string("mlp_out_65_strides_0"), val = tensor([1, 1])]; tensor mlp_out_65_pad_0 = const()[name = string("mlp_out_65_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_65_dilations_0 = const()[name = string("mlp_out_65_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_65_groups_0 = const()[name = string("mlp_out_65_groups_0"), val = int32(1)]; tensor mlp_out_65 = conv(dilations = mlp_out_65_dilations_0, groups = mlp_out_65_groups_0, pad = mlp_out_65_pad_0, pad_type = mlp_out_65_pad_type_0, strides = mlp_out_65_strides_0, weight = layers_32_mlp_down_proj_weight_palettized, x = input_861)[name = string("mlp_out_65")]; tensor var_15664_axes_0 = const()[name = string("op_15664_axes_0"), val = tensor([2])]; tensor var_15664 = squeeze(axes = var_15664_axes_0, x = mlp_out_65)[name = string("op_15664")]; tensor var_15668 = const()[name = string("op_15668"), val = tensor([0, 2, 1])]; int32 var_15674 = const()[name = string("op_15674"), val = int32(-1)]; fp16 const_495_promoted_to_fp16 = const()[name = string("const_495_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_875 = transpose(perm = var_15668, x = var_15664)[name = string("transpose_16")]; tensor var_15680_cast_fp16 = mul(x = x_875, y = const_495_promoted_to_fp16)[name = string("op_15680_cast_fp16")]; bool input_863_interleave_0 = const()[name = string("input_863_interleave_0"), val = bool(false)]; tensor input_863_cast_fp16 = concat(axis = var_15674, interleave = input_863_interleave_0, values = (x_875, var_15680_cast_fp16))[name = string("input_863_cast_fp16")]; tensor normed_845_axes_0 = const()[name = string("normed_845_axes_0"), val = tensor([-1])]; fp16 var_15672_to_fp16 = const()[name = string("op_15672_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_845_cast_fp16 = layer_norm(axes = normed_845_axes_0, epsilon = var_15672_to_fp16, x = input_863_cast_fp16)[name = string("normed_845_cast_fp16")]; tensor var_15685_split_sizes_0 = const()[name = string("op_15685_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15685_axis_0 = const()[name = string("op_15685_axis_0"), val = int32(-1)]; tensor var_15685_cast_fp16_0, tensor var_15685_cast_fp16_1 = split(axis = var_15685_axis_0, split_sizes = var_15685_split_sizes_0, x = normed_845_cast_fp16)[name = string("op_15685_cast_fp16")]; tensor const_496_to_fp16 = const()[name = string("const_496_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130100224)))]; tensor var_15688_cast_fp16 = mul(x = var_15685_cast_fp16_0, y = const_496_to_fp16)[name = string("op_15688_cast_fp16")]; tensor hidden_states_393_cast_fp16 = add(x = x_871_cast_fp16, y = var_15688_cast_fp16)[name = string("hidden_states_393_cast_fp16")]; tensor per_layer_slice_65_begin_0 = const()[name = string("per_layer_slice_65_begin_0"), val = tensor([0, 0, 8192])]; tensor per_layer_slice_65_end_0 = const()[name = string("per_layer_slice_65_end_0"), val = tensor([1, 1, 8448])]; tensor per_layer_slice_65_end_mask_0 = const()[name = string("per_layer_slice_65_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_65_cast_fp16 = slice_by_index(begin = per_layer_slice_65_begin_0, end = per_layer_slice_65_end_0, end_mask = per_layer_slice_65_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_65_cast_fp16")]; tensor gated_129 = linear(bias = linear_0_bias_0, weight = layers_32_per_layer_input_gate_weight_palettized, x = hidden_states_393_cast_fp16)[name = string("linear_64")]; string gated_131_mode_0 = const()[name = string("gated_131_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_131 = gelu(mode = gated_131_mode_0, x = gated_129)[name = string("gated_131")]; tensor input_867_cast_fp16 = mul(x = gated_131, y = per_layer_slice_65_cast_fp16)[name = string("input_867_cast_fp16")]; tensor layers_32_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130103360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130300032))))[name = string("layers_32_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_65_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_32_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_867_cast_fp16)[name = string("linear_65_cast_fp16")]; int32 var_15725 = const()[name = string("op_15725"), val = int32(-1)]; fp16 const_497_promoted_to_fp16 = const()[name = string("const_497_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15731_cast_fp16 = mul(x = linear_65_cast_fp16, y = const_497_promoted_to_fp16)[name = string("op_15731_cast_fp16")]; bool input_869_interleave_0 = const()[name = string("input_869_interleave_0"), val = bool(false)]; tensor input_869_cast_fp16 = concat(axis = var_15725, interleave = input_869_interleave_0, values = (linear_65_cast_fp16, var_15731_cast_fp16))[name = string("input_869_cast_fp16")]; tensor normed_849_axes_0 = const()[name = string("normed_849_axes_0"), val = tensor([-1])]; fp16 var_15723_to_fp16 = const()[name = string("op_15723_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_849_cast_fp16 = layer_norm(axes = normed_849_axes_0, epsilon = var_15723_to_fp16, x = input_869_cast_fp16)[name = string("normed_849_cast_fp16")]; tensor var_15736_split_sizes_0 = const()[name = string("op_15736_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15736_axis_0 = const()[name = string("op_15736_axis_0"), val = int32(-1)]; tensor var_15736_cast_fp16_0, tensor var_15736_cast_fp16_1 = split(axis = var_15736_axis_0, split_sizes = var_15736_split_sizes_0, x = normed_849_cast_fp16)[name = string("op_15736_cast_fp16")]; tensor const_498_to_fp16 = const()[name = string("const_498_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130301632)))]; tensor var_15739_cast_fp16 = mul(x = var_15736_cast_fp16_0, y = const_498_to_fp16)[name = string("op_15739_cast_fp16")]; tensor hidden_states_397_cast_fp16 = add(x = hidden_states_393_cast_fp16, y = var_15739_cast_fp16)[name = string("hidden_states_397_cast_fp16")]; tensor layers_32_layer_scalar_to_fp16 = const()[name = string("layers_32_layer_scalar_to_fp16"), val = tensor([0x1.bep-1])]; tensor x_883_cast_fp16 = mul(x = hidden_states_397_cast_fp16, y = layers_32_layer_scalar_to_fp16)[name = string("x_883_cast_fp16")]; int32 var_15747 = const()[name = string("op_15747"), val = int32(-1)]; fp16 const_499_promoted_to_fp16 = const()[name = string("const_499_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15753_cast_fp16 = mul(x = x_883_cast_fp16, y = const_499_promoted_to_fp16)[name = string("op_15753_cast_fp16")]; bool input_871_interleave_0 = const()[name = string("input_871_interleave_0"), val = bool(false)]; tensor input_871_cast_fp16 = concat(axis = var_15747, interleave = input_871_interleave_0, values = (x_883_cast_fp16, var_15753_cast_fp16))[name = string("input_871_cast_fp16")]; tensor normed_853_axes_0 = const()[name = string("normed_853_axes_0"), val = tensor([-1])]; fp16 var_15745_to_fp16 = const()[name = string("op_15745_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_853_cast_fp16 = layer_norm(axes = normed_853_axes_0, epsilon = var_15745_to_fp16, x = input_871_cast_fp16)[name = string("normed_853_cast_fp16")]; tensor var_15758_split_sizes_0 = const()[name = string("op_15758_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15758_axis_0 = const()[name = string("op_15758_axis_0"), val = int32(-1)]; tensor var_15758_cast_fp16_0, tensor var_15758_cast_fp16_1 = split(axis = var_15758_axis_0, split_sizes = var_15758_split_sizes_0, x = normed_853_cast_fp16)[name = string("op_15758_cast_fp16")]; tensor const_500_to_fp16 = const()[name = string("const_500_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130304768)))]; tensor var_15761_cast_fp16 = mul(x = var_15758_cast_fp16_0, y = const_500_to_fp16)[name = string("op_15761_cast_fp16")]; tensor var_15769 = const()[name = string("op_15769"), val = tensor([0, 2, 1])]; tensor var_15772_axes_0 = const()[name = string("op_15772_axes_0"), val = tensor([2])]; tensor var_15770_cast_fp16 = transpose(perm = var_15769, x = var_15761_cast_fp16)[name = string("transpose_15")]; tensor var_15772_cast_fp16 = expand_dims(axes = var_15772_axes_0, x = var_15770_cast_fp16)[name = string("op_15772_cast_fp16")]; string var_15788_pad_type_0 = const()[name = string("op_15788_pad_type_0"), val = string("valid")]; tensor var_15788_strides_0 = const()[name = string("op_15788_strides_0"), val = tensor([1, 1])]; tensor var_15788_pad_0 = const()[name = string("op_15788_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_15788_dilations_0 = const()[name = string("op_15788_dilations_0"), val = tensor([1, 1])]; int32 var_15788_groups_0 = const()[name = string("op_15788_groups_0"), val = int32(1)]; tensor var_15788 = conv(dilations = var_15788_dilations_0, groups = var_15788_groups_0, pad = var_15788_pad_0, pad_type = var_15788_pad_type_0, strides = var_15788_strides_0, weight = layers_33_self_attn_q_proj_weight_palettized, x = var_15772_cast_fp16)[name = string("op_15788")]; tensor var_15793 = const()[name = string("op_15793"), val = tensor([1, 8, 256, 1])]; tensor var_15794 = reshape(shape = var_15793, x = var_15788)[name = string("op_15794")]; tensor var_15799 = const()[name = string("op_15799"), val = tensor([0, 1, 3, 2])]; tensor var_15809 = const()[name = string("op_15809"), val = tensor([1, 8, 256])]; tensor var_15800 = transpose(perm = var_15799, x = var_15794)[name = string("transpose_14")]; tensor x_887 = reshape(shape = var_15809, x = var_15800)[name = string("x_887")]; int32 var_15815 = const()[name = string("op_15815"), val = int32(-1)]; fp16 const_501_promoted_to_fp16 = const()[name = string("const_501_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15821_cast_fp16 = mul(x = x_887, y = const_501_promoted_to_fp16)[name = string("op_15821_cast_fp16")]; bool input_875_interleave_0 = const()[name = string("input_875_interleave_0"), val = bool(false)]; tensor input_875_cast_fp16 = concat(axis = var_15815, interleave = input_875_interleave_0, values = (x_887, var_15821_cast_fp16))[name = string("input_875_cast_fp16")]; tensor normed_857_axes_0 = const()[name = string("normed_857_axes_0"), val = tensor([-1])]; fp16 var_15813_to_fp16 = const()[name = string("op_15813_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_857_cast_fp16 = layer_norm(axes = normed_857_axes_0, epsilon = var_15813_to_fp16, x = input_875_cast_fp16)[name = string("normed_857_cast_fp16")]; tensor var_15826_split_sizes_0 = const()[name = string("op_15826_split_sizes_0"), val = tensor([256, 256])]; int32 var_15826_axis_0 = const()[name = string("op_15826_axis_0"), val = int32(-1)]; tensor var_15826_cast_fp16_0, tensor var_15826_cast_fp16_1 = split(axis = var_15826_axis_0, split_sizes = var_15826_split_sizes_0, x = normed_857_cast_fp16)[name = string("op_15826_cast_fp16")]; tensor var_15829_cast_fp16 = mul(x = var_15826_cast_fp16_0, y = const_234_to_fp16)[name = string("op_15829_cast_fp16")]; tensor var_15835 = const()[name = string("op_15835"), val = tensor([1, 8, 1, 256])]; tensor q_231 = reshape(shape = var_15835, x = var_15829_cast_fp16)[name = string("q_231")]; tensor var_15837 = mul(x = q_231, y = cos_1)[name = string("op_15837")]; tensor var_15838_split_sizes_0 = const()[name = string("op_15838_split_sizes_0"), val = tensor([128, 128])]; int32 var_15838_axis_0 = const()[name = string("op_15838_axis_0"), val = int32(-1)]; tensor var_15838_0, tensor var_15838_1 = split(axis = var_15838_axis_0, split_sizes = var_15838_split_sizes_0, x = q_231)[name = string("op_15838")]; fp16 const_503_promoted = const()[name = string("const_503_promoted"), val = fp16(-0x1p+0)]; tensor var_15840 = mul(x = var_15838_1, y = const_503_promoted)[name = string("op_15840")]; int32 var_15842 = const()[name = string("op_15842"), val = int32(-1)]; bool var_15843_interleave_0 = const()[name = string("op_15843_interleave_0"), val = bool(false)]; tensor var_15843 = concat(axis = var_15842, interleave = var_15843_interleave_0, values = (var_15840, var_15838_0))[name = string("op_15843")]; tensor var_15844 = mul(x = var_15843, y = sin_1)[name = string("op_15844")]; tensor q_233 = add(x = var_15837, y = var_15844)[name = string("q_233")]; bool var_15858_transpose_x_0 = const()[name = string("op_15858_transpose_x_0"), val = bool(false)]; bool var_15858_transpose_y_0 = const()[name = string("op_15858_transpose_y_0"), val = bool(false)]; tensor var_15858_cast_fp16 = matmul(transpose_x = var_15858_transpose_x_0, transpose_y = var_15858_transpose_y_0, x = q_233, y = transpose_153_cast_fp16)[name = string("op_15858_cast_fp16")]; tensor attn_weights_201_cast_fp16 = add(x = var_15858_cast_fp16, y = causal_mask)[name = string("attn_weights_201_cast_fp16")]; int32 var_15863 = const()[name = string("op_15863"), val = int32(-1)]; tensor attn_weights_203_cast_fp16 = softmax(axis = var_15863, x = attn_weights_201_cast_fp16)[name = string("attn_weights_203_cast_fp16")]; bool attn_output_199_transpose_x_0 = const()[name = string("attn_output_199_transpose_x_0"), val = bool(false)]; bool attn_output_199_transpose_y_0 = const()[name = string("attn_output_199_transpose_y_0"), val = bool(false)]; tensor attn_output_199_cast_fp16 = matmul(transpose_x = attn_output_199_transpose_x_0, transpose_y = attn_output_199_transpose_y_0, x = attn_weights_203_cast_fp16, y = V_expanded_27_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor var_15871 = const()[name = string("op_15871"), val = tensor([0, 2, 1, 3])]; tensor var_15878 = const()[name = string("op_15878"), val = tensor([1, 1, -1])]; tensor var_15872_cast_fp16 = transpose(perm = var_15871, x = attn_output_199_cast_fp16)[name = string("transpose_13")]; tensor attn_output_201_cast_fp16 = reshape(shape = var_15878, x = var_15872_cast_fp16)[name = string("attn_output_201_cast_fp16")]; tensor var_15883 = const()[name = string("op_15883"), val = tensor([0, 2, 1])]; string var_15899_pad_type_0 = const()[name = string("op_15899_pad_type_0"), val = string("valid")]; int32 var_15899_groups_0 = const()[name = string("op_15899_groups_0"), val = int32(1)]; tensor var_15899_strides_0 = const()[name = string("op_15899_strides_0"), val = tensor([1])]; tensor var_15899_pad_0 = const()[name = string("op_15899_pad_0"), val = tensor([0, 0])]; tensor var_15899_dilations_0 = const()[name = string("op_15899_dilations_0"), val = tensor([1])]; tensor squeeze_33_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1130307904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1131880832))))[name = string("squeeze_33_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_15884_cast_fp16 = transpose(perm = var_15883, x = attn_output_201_cast_fp16)[name = string("transpose_12")]; tensor var_15899_cast_fp16 = conv(dilations = var_15899_dilations_0, groups = var_15899_groups_0, pad = var_15899_pad_0, pad_type = var_15899_pad_type_0, strides = var_15899_strides_0, weight = squeeze_33_cast_fp16_to_fp32_to_fp16_palettized, x = var_15884_cast_fp16)[name = string("op_15899_cast_fp16")]; tensor var_15903 = const()[name = string("op_15903"), val = tensor([0, 2, 1])]; int32 var_15909 = const()[name = string("op_15909"), val = int32(-1)]; fp16 const_504_promoted_to_fp16 = const()[name = string("const_504_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_891_cast_fp16 = transpose(perm = var_15903, x = var_15899_cast_fp16)[name = string("transpose_11")]; tensor var_15915_cast_fp16 = mul(x = x_891_cast_fp16, y = const_504_promoted_to_fp16)[name = string("op_15915_cast_fp16")]; bool input_879_interleave_0 = const()[name = string("input_879_interleave_0"), val = bool(false)]; tensor input_879_cast_fp16 = concat(axis = var_15909, interleave = input_879_interleave_0, values = (x_891_cast_fp16, var_15915_cast_fp16))[name = string("input_879_cast_fp16")]; tensor normed_861_axes_0 = const()[name = string("normed_861_axes_0"), val = tensor([-1])]; fp16 var_15907_to_fp16 = const()[name = string("op_15907_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_861_cast_fp16 = layer_norm(axes = normed_861_axes_0, epsilon = var_15907_to_fp16, x = input_879_cast_fp16)[name = string("normed_861_cast_fp16")]; tensor var_15920_split_sizes_0 = const()[name = string("op_15920_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15920_axis_0 = const()[name = string("op_15920_axis_0"), val = int32(-1)]; tensor var_15920_cast_fp16_0, tensor var_15920_cast_fp16_1 = split(axis = var_15920_axis_0, split_sizes = var_15920_split_sizes_0, x = normed_861_cast_fp16)[name = string("op_15920_cast_fp16")]; tensor const_505_to_fp16 = const()[name = string("const_505_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1131882432)))]; tensor var_15923_cast_fp16 = mul(x = var_15920_cast_fp16_0, y = const_505_to_fp16)[name = string("op_15923_cast_fp16")]; tensor x_895_cast_fp16 = add(x = x_883_cast_fp16, y = var_15923_cast_fp16)[name = string("x_895_cast_fp16")]; int32 var_15930 = const()[name = string("op_15930"), val = int32(-1)]; fp16 const_506_promoted_to_fp16 = const()[name = string("const_506_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_15936_cast_fp16 = mul(x = x_895_cast_fp16, y = const_506_promoted_to_fp16)[name = string("op_15936_cast_fp16")]; bool input_881_interleave_0 = const()[name = string("input_881_interleave_0"), val = bool(false)]; tensor input_881_cast_fp16 = concat(axis = var_15930, interleave = input_881_interleave_0, values = (x_895_cast_fp16, var_15936_cast_fp16))[name = string("input_881_cast_fp16")]; tensor normed_865_axes_0 = const()[name = string("normed_865_axes_0"), val = tensor([-1])]; fp16 var_15928_to_fp16 = const()[name = string("op_15928_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_865_cast_fp16 = layer_norm(axes = normed_865_axes_0, epsilon = var_15928_to_fp16, x = input_881_cast_fp16)[name = string("normed_865_cast_fp16")]; tensor var_15941_split_sizes_0 = const()[name = string("op_15941_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_15941_axis_0 = const()[name = string("op_15941_axis_0"), val = int32(-1)]; tensor var_15941_cast_fp16_0, tensor var_15941_cast_fp16_1 = split(axis = var_15941_axis_0, split_sizes = var_15941_split_sizes_0, x = normed_865_cast_fp16)[name = string("op_15941_cast_fp16")]; tensor const_507_to_fp16 = const()[name = string("const_507_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1131885568)))]; tensor var_15944_cast_fp16 = mul(x = var_15941_cast_fp16_0, y = const_507_to_fp16)[name = string("op_15944_cast_fp16")]; tensor var_15957 = const()[name = string("op_15957"), val = tensor([0, 2, 1])]; tensor input_883_axes_0 = const()[name = string("input_883_axes_0"), val = tensor([2])]; tensor var_15958 = transpose(perm = var_15957, x = var_15944_cast_fp16)[name = string("transpose_10")]; tensor input_883 = expand_dims(axes = input_883_axes_0, x = var_15958)[name = string("input_883")]; string gate_133_pad_type_0 = const()[name = string("gate_133_pad_type_0"), val = string("valid")]; tensor gate_133_strides_0 = const()[name = string("gate_133_strides_0"), val = tensor([1, 1])]; tensor gate_133_pad_0 = const()[name = string("gate_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_133_dilations_0 = const()[name = string("gate_133_dilations_0"), val = tensor([1, 1])]; int32 gate_133_groups_0 = const()[name = string("gate_133_groups_0"), val = int32(1)]; tensor gate_133 = conv(dilations = gate_133_dilations_0, groups = gate_133_groups_0, pad = gate_133_pad_0, pad_type = gate_133_pad_type_0, strides = gate_133_strides_0, weight = layers_33_mlp_gate_proj_weight_palettized, x = input_883)[name = string("gate_133")]; string up_67_pad_type_0 = const()[name = string("up_67_pad_type_0"), val = string("valid")]; tensor up_67_strides_0 = const()[name = string("up_67_strides_0"), val = tensor([1, 1])]; tensor up_67_pad_0 = const()[name = string("up_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_67_dilations_0 = const()[name = string("up_67_dilations_0"), val = tensor([1, 1])]; int32 up_67_groups_0 = const()[name = string("up_67_groups_0"), val = int32(1)]; tensor up_67 = conv(dilations = up_67_dilations_0, groups = up_67_groups_0, pad = up_67_pad_0, pad_type = up_67_pad_type_0, strides = up_67_strides_0, weight = layers_33_mlp_up_proj_weight_palettized, x = input_883)[name = string("up_67")]; string gate_135_mode_0 = const()[name = string("gate_135_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_135 = gelu(mode = gate_135_mode_0, x = gate_133)[name = string("gate_135")]; tensor input_885 = mul(x = gate_135, y = up_67)[name = string("input_885")]; string mlp_out_67_pad_type_0 = const()[name = string("mlp_out_67_pad_type_0"), val = string("valid")]; tensor mlp_out_67_strides_0 = const()[name = string("mlp_out_67_strides_0"), val = tensor([1, 1])]; tensor mlp_out_67_pad_0 = const()[name = string("mlp_out_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_67_dilations_0 = const()[name = string("mlp_out_67_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_67_groups_0 = const()[name = string("mlp_out_67_groups_0"), val = int32(1)]; tensor mlp_out_67 = conv(dilations = mlp_out_67_dilations_0, groups = mlp_out_67_groups_0, pad = mlp_out_67_pad_0, pad_type = mlp_out_67_pad_type_0, strides = mlp_out_67_strides_0, weight = layers_33_mlp_down_proj_weight_palettized, x = input_885)[name = string("mlp_out_67")]; tensor var_15998_axes_0 = const()[name = string("op_15998_axes_0"), val = tensor([2])]; tensor var_15998 = squeeze(axes = var_15998_axes_0, x = mlp_out_67)[name = string("op_15998")]; tensor var_16002 = const()[name = string("op_16002"), val = tensor([0, 2, 1])]; int32 var_16008 = const()[name = string("op_16008"), val = int32(-1)]; fp16 const_508_promoted_to_fp16 = const()[name = string("const_508_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_899 = transpose(perm = var_16002, x = var_15998)[name = string("transpose_9")]; tensor var_16014_cast_fp16 = mul(x = x_899, y = const_508_promoted_to_fp16)[name = string("op_16014_cast_fp16")]; bool input_887_interleave_0 = const()[name = string("input_887_interleave_0"), val = bool(false)]; tensor input_887_cast_fp16 = concat(axis = var_16008, interleave = input_887_interleave_0, values = (x_899, var_16014_cast_fp16))[name = string("input_887_cast_fp16")]; tensor normed_869_axes_0 = const()[name = string("normed_869_axes_0"), val = tensor([-1])]; fp16 var_16006_to_fp16 = const()[name = string("op_16006_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_869_cast_fp16 = layer_norm(axes = normed_869_axes_0, epsilon = var_16006_to_fp16, x = input_887_cast_fp16)[name = string("normed_869_cast_fp16")]; tensor var_16019_split_sizes_0 = const()[name = string("op_16019_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16019_axis_0 = const()[name = string("op_16019_axis_0"), val = int32(-1)]; tensor var_16019_cast_fp16_0, tensor var_16019_cast_fp16_1 = split(axis = var_16019_axis_0, split_sizes = var_16019_split_sizes_0, x = normed_869_cast_fp16)[name = string("op_16019_cast_fp16")]; tensor const_509_to_fp16 = const()[name = string("const_509_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1131888704)))]; tensor var_16022_cast_fp16 = mul(x = var_16019_cast_fp16_0, y = const_509_to_fp16)[name = string("op_16022_cast_fp16")]; tensor hidden_states_405_cast_fp16 = add(x = x_895_cast_fp16, y = var_16022_cast_fp16)[name = string("hidden_states_405_cast_fp16")]; tensor per_layer_slice_67_begin_0 = const()[name = string("per_layer_slice_67_begin_0"), val = tensor([0, 0, 8448])]; tensor per_layer_slice_67_end_0 = const()[name = string("per_layer_slice_67_end_0"), val = tensor([1, 1, 8704])]; tensor per_layer_slice_67_end_mask_0 = const()[name = string("per_layer_slice_67_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_67_cast_fp16 = slice_by_index(begin = per_layer_slice_67_begin_0, end = per_layer_slice_67_end_0, end_mask = per_layer_slice_67_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_67_cast_fp16")]; tensor gated_133 = linear(bias = linear_0_bias_0, weight = layers_33_per_layer_input_gate_weight_palettized, x = hidden_states_405_cast_fp16)[name = string("linear_66")]; string gated_135_mode_0 = const()[name = string("gated_135_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_135 = gelu(mode = gated_135_mode_0, x = gated_133)[name = string("gated_135")]; tensor input_891_cast_fp16 = mul(x = gated_135, y = per_layer_slice_67_cast_fp16)[name = string("input_891_cast_fp16")]; tensor layers_33_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1131891840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1132088512))))[name = string("layers_33_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_67_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_33_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_891_cast_fp16)[name = string("linear_67_cast_fp16")]; int32 var_16059 = const()[name = string("op_16059"), val = int32(-1)]; fp16 const_510_promoted_to_fp16 = const()[name = string("const_510_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_16065_cast_fp16 = mul(x = linear_67_cast_fp16, y = const_510_promoted_to_fp16)[name = string("op_16065_cast_fp16")]; bool input_893_interleave_0 = const()[name = string("input_893_interleave_0"), val = bool(false)]; tensor input_893_cast_fp16 = concat(axis = var_16059, interleave = input_893_interleave_0, values = (linear_67_cast_fp16, var_16065_cast_fp16))[name = string("input_893_cast_fp16")]; tensor normed_873_axes_0 = const()[name = string("normed_873_axes_0"), val = tensor([-1])]; fp16 var_16057_to_fp16 = const()[name = string("op_16057_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_873_cast_fp16 = layer_norm(axes = normed_873_axes_0, epsilon = var_16057_to_fp16, x = input_893_cast_fp16)[name = string("normed_873_cast_fp16")]; tensor var_16070_split_sizes_0 = const()[name = string("op_16070_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16070_axis_0 = const()[name = string("op_16070_axis_0"), val = int32(-1)]; tensor var_16070_cast_fp16_0, tensor var_16070_cast_fp16_1 = split(axis = var_16070_axis_0, split_sizes = var_16070_split_sizes_0, x = normed_873_cast_fp16)[name = string("op_16070_cast_fp16")]; tensor const_511_to_fp16 = const()[name = string("const_511_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1132090112)))]; tensor var_16073_cast_fp16 = mul(x = var_16070_cast_fp16_0, y = const_511_to_fp16)[name = string("op_16073_cast_fp16")]; tensor hidden_states_409_cast_fp16 = add(x = hidden_states_405_cast_fp16, y = var_16073_cast_fp16)[name = string("hidden_states_409_cast_fp16")]; tensor layers_33_layer_scalar_to_fp16 = const()[name = string("layers_33_layer_scalar_to_fp16"), val = tensor([0x1.64p-1])]; tensor x_907_cast_fp16 = mul(x = hidden_states_409_cast_fp16, y = layers_33_layer_scalar_to_fp16)[name = string("x_907_cast_fp16")]; int32 var_16081 = const()[name = string("op_16081"), val = int32(-1)]; fp16 const_512_promoted_to_fp16 = const()[name = string("const_512_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_16087_cast_fp16 = mul(x = x_907_cast_fp16, y = const_512_promoted_to_fp16)[name = string("op_16087_cast_fp16")]; bool input_895_interleave_0 = const()[name = string("input_895_interleave_0"), val = bool(false)]; tensor input_895_cast_fp16 = concat(axis = var_16081, interleave = input_895_interleave_0, values = (x_907_cast_fp16, var_16087_cast_fp16))[name = string("input_895_cast_fp16")]; tensor normed_877_axes_0 = const()[name = string("normed_877_axes_0"), val = tensor([-1])]; fp16 var_16079_to_fp16 = const()[name = string("op_16079_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_877_cast_fp16 = layer_norm(axes = normed_877_axes_0, epsilon = var_16079_to_fp16, x = input_895_cast_fp16)[name = string("normed_877_cast_fp16")]; tensor var_16092_split_sizes_0 = const()[name = string("op_16092_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16092_axis_0 = const()[name = string("op_16092_axis_0"), val = int32(-1)]; tensor var_16092_cast_fp16_0, tensor var_16092_cast_fp16_1 = split(axis = var_16092_axis_0, split_sizes = var_16092_split_sizes_0, x = normed_877_cast_fp16)[name = string("op_16092_cast_fp16")]; tensor const_513_to_fp16 = const()[name = string("const_513_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1132093248)))]; tensor var_16095_cast_fp16 = mul(x = var_16092_cast_fp16_0, y = const_513_to_fp16)[name = string("op_16095_cast_fp16")]; tensor var_16103 = const()[name = string("op_16103"), val = tensor([0, 2, 1])]; tensor var_16106_axes_0 = const()[name = string("op_16106_axes_0"), val = tensor([2])]; tensor var_16104_cast_fp16 = transpose(perm = var_16103, x = var_16095_cast_fp16)[name = string("transpose_8")]; tensor var_16106_cast_fp16 = expand_dims(axes = var_16106_axes_0, x = var_16104_cast_fp16)[name = string("op_16106_cast_fp16")]; string var_16122_pad_type_0 = const()[name = string("op_16122_pad_type_0"), val = string("valid")]; tensor var_16122_strides_0 = const()[name = string("op_16122_strides_0"), val = tensor([1, 1])]; tensor var_16122_pad_0 = const()[name = string("op_16122_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_16122_dilations_0 = const()[name = string("op_16122_dilations_0"), val = tensor([1, 1])]; int32 var_16122_groups_0 = const()[name = string("op_16122_groups_0"), val = int32(1)]; tensor var_16122 = conv(dilations = var_16122_dilations_0, groups = var_16122_groups_0, pad = var_16122_pad_0, pad_type = var_16122_pad_type_0, strides = var_16122_strides_0, weight = layers_34_self_attn_q_proj_weight_palettized, x = var_16106_cast_fp16)[name = string("op_16122")]; tensor var_16127 = const()[name = string("op_16127"), val = tensor([1, 8, 512, 1])]; tensor var_16128 = reshape(shape = var_16127, x = var_16122)[name = string("op_16128")]; tensor var_16133 = const()[name = string("op_16133"), val = tensor([0, 1, 3, 2])]; tensor var_16143 = const()[name = string("op_16143"), val = tensor([1, 8, 512])]; tensor var_16134 = transpose(perm = var_16133, x = var_16128)[name = string("transpose_7")]; tensor x_911 = reshape(shape = var_16143, x = var_16134)[name = string("x_911")]; int32 var_16149 = const()[name = string("op_16149"), val = int32(-1)]; fp16 const_514_promoted_to_fp16 = const()[name = string("const_514_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_16155_cast_fp16 = mul(x = x_911, y = const_514_promoted_to_fp16)[name = string("op_16155_cast_fp16")]; bool input_899_interleave_0 = const()[name = string("input_899_interleave_0"), val = bool(false)]; tensor input_899_cast_fp16 = concat(axis = var_16149, interleave = input_899_interleave_0, values = (x_911, var_16155_cast_fp16))[name = string("input_899_cast_fp16")]; tensor normed_881_axes_0 = const()[name = string("normed_881_axes_0"), val = tensor([-1])]; fp16 var_16147_to_fp16 = const()[name = string("op_16147_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_881_cast_fp16 = layer_norm(axes = normed_881_axes_0, epsilon = var_16147_to_fp16, x = input_899_cast_fp16)[name = string("normed_881_cast_fp16")]; tensor var_16160_split_sizes_0 = const()[name = string("op_16160_split_sizes_0"), val = tensor([512, 512])]; int32 var_16160_axis_0 = const()[name = string("op_16160_axis_0"), val = int32(-1)]; tensor var_16160_cast_fp16_0, tensor var_16160_cast_fp16_1 = split(axis = var_16160_axis_0, split_sizes = var_16160_split_sizes_0, x = normed_881_cast_fp16)[name = string("op_16160_cast_fp16")]; tensor var_16163_cast_fp16 = mul(x = var_16160_cast_fp16_0, y = const_252_to_fp16)[name = string("op_16163_cast_fp16")]; tensor var_16169 = const()[name = string("op_16169"), val = tensor([1, 8, 1, 512])]; tensor q_237 = reshape(shape = var_16169, x = var_16163_cast_fp16)[name = string("q_237")]; tensor var_16171 = mul(x = q_237, y = cos)[name = string("op_16171")]; tensor var_16172_split_sizes_0 = const()[name = string("op_16172_split_sizes_0"), val = tensor([256, 256])]; int32 var_16172_axis_0 = const()[name = string("op_16172_axis_0"), val = int32(-1)]; tensor var_16172_0, tensor var_16172_1 = split(axis = var_16172_axis_0, split_sizes = var_16172_split_sizes_0, x = q_237)[name = string("op_16172")]; fp16 const_516_promoted = const()[name = string("const_516_promoted"), val = fp16(-0x1p+0)]; tensor var_16174 = mul(x = var_16172_1, y = const_516_promoted)[name = string("op_16174")]; int32 var_16176 = const()[name = string("op_16176"), val = int32(-1)]; bool var_16177_interleave_0 = const()[name = string("op_16177_interleave_0"), val = bool(false)]; tensor var_16177 = concat(axis = var_16176, interleave = var_16177_interleave_0, values = (var_16174, var_16172_0))[name = string("op_16177")]; tensor var_16178 = mul(x = var_16177, y = sin)[name = string("op_16178")]; tensor q = add(x = var_16171, y = var_16178)[name = string("q")]; bool var_16192_transpose_x_0 = const()[name = string("op_16192_transpose_x_0"), val = bool(false)]; bool var_16192_transpose_y_0 = const()[name = string("op_16192_transpose_y_0"), val = bool(false)]; tensor var_16192_cast_fp16 = matmul(transpose_x = var_16192_transpose_x_0, transpose_y = var_16192_transpose_y_0, x = q, y = transpose_154_cast_fp16)[name = string("op_16192_cast_fp16")]; tensor attn_weights_207_cast_fp16 = add(x = var_16192_cast_fp16, y = causal_mask)[name = string("attn_weights_207_cast_fp16")]; int32 var_16197 = const()[name = string("op_16197"), val = int32(-1)]; tensor attn_weights_cast_fp16 = softmax(axis = var_16197, x = attn_weights_207_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_205_transpose_x_0 = const()[name = string("attn_output_205_transpose_x_0"), val = bool(false)]; bool attn_output_205_transpose_y_0 = const()[name = string("attn_output_205_transpose_y_0"), val = bool(false)]; tensor attn_output_205_cast_fp16 = matmul(transpose_x = attn_output_205_transpose_x_0, transpose_y = attn_output_205_transpose_y_0, x = attn_weights_cast_fp16, y = V_expanded_29_cast_fp16)[name = string("attn_output_205_cast_fp16")]; tensor var_16205 = const()[name = string("op_16205"), val = tensor([0, 2, 1, 3])]; tensor var_16212 = const()[name = string("op_16212"), val = tensor([1, 1, -1])]; tensor var_16206_cast_fp16 = transpose(perm = var_16205, x = attn_output_205_cast_fp16)[name = string("transpose_6")]; tensor attn_output_207_cast_fp16 = reshape(shape = var_16212, x = var_16206_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor var_16217 = const()[name = string("op_16217"), val = tensor([0, 2, 1])]; string var_16233_pad_type_0 = const()[name = string("op_16233_pad_type_0"), val = string("valid")]; int32 var_16233_groups_0 = const()[name = string("op_16233_groups_0"), val = int32(1)]; tensor var_16233_strides_0 = const()[name = string("op_16233_strides_0"), val = tensor([1])]; tensor var_16233_pad_0 = const()[name = string("op_16233_pad_0"), val = tensor([0, 0])]; tensor var_16233_dilations_0 = const()[name = string("op_16233_dilations_0"), val = tensor([1])]; tensor squeeze_34_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1132096384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135242176))))[name = string("squeeze_34_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_16218_cast_fp16 = transpose(perm = var_16217, x = attn_output_207_cast_fp16)[name = string("transpose_5")]; tensor var_16233_cast_fp16 = conv(dilations = var_16233_dilations_0, groups = var_16233_groups_0, pad = var_16233_pad_0, pad_type = var_16233_pad_type_0, strides = var_16233_strides_0, weight = squeeze_34_cast_fp16_to_fp32_to_fp16_palettized, x = var_16218_cast_fp16)[name = string("op_16233_cast_fp16")]; tensor var_16237 = const()[name = string("op_16237"), val = tensor([0, 2, 1])]; int32 var_16243 = const()[name = string("op_16243"), val = int32(-1)]; fp16 const_517_promoted_to_fp16 = const()[name = string("const_517_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_915_cast_fp16 = transpose(perm = var_16237, x = var_16233_cast_fp16)[name = string("transpose_4")]; tensor var_16249_cast_fp16 = mul(x = x_915_cast_fp16, y = const_517_promoted_to_fp16)[name = string("op_16249_cast_fp16")]; bool input_903_interleave_0 = const()[name = string("input_903_interleave_0"), val = bool(false)]; tensor input_903_cast_fp16 = concat(axis = var_16243, interleave = input_903_interleave_0, values = (x_915_cast_fp16, var_16249_cast_fp16))[name = string("input_903_cast_fp16")]; tensor normed_885_axes_0 = const()[name = string("normed_885_axes_0"), val = tensor([-1])]; fp16 var_16241_to_fp16 = const()[name = string("op_16241_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_885_cast_fp16 = layer_norm(axes = normed_885_axes_0, epsilon = var_16241_to_fp16, x = input_903_cast_fp16)[name = string("normed_885_cast_fp16")]; tensor var_16254_split_sizes_0 = const()[name = string("op_16254_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16254_axis_0 = const()[name = string("op_16254_axis_0"), val = int32(-1)]; tensor var_16254_cast_fp16_0, tensor var_16254_cast_fp16_1 = split(axis = var_16254_axis_0, split_sizes = var_16254_split_sizes_0, x = normed_885_cast_fp16)[name = string("op_16254_cast_fp16")]; tensor const_518_to_fp16 = const()[name = string("const_518_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135243776)))]; tensor var_16257_cast_fp16 = mul(x = var_16254_cast_fp16_0, y = const_518_to_fp16)[name = string("op_16257_cast_fp16")]; tensor x_919_cast_fp16 = add(x = x_907_cast_fp16, y = var_16257_cast_fp16)[name = string("x_919_cast_fp16")]; int32 var_16264 = const()[name = string("op_16264"), val = int32(-1)]; fp16 const_519_promoted_to_fp16 = const()[name = string("const_519_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_16270_cast_fp16 = mul(x = x_919_cast_fp16, y = const_519_promoted_to_fp16)[name = string("op_16270_cast_fp16")]; bool input_905_interleave_0 = const()[name = string("input_905_interleave_0"), val = bool(false)]; tensor input_905_cast_fp16 = concat(axis = var_16264, interleave = input_905_interleave_0, values = (x_919_cast_fp16, var_16270_cast_fp16))[name = string("input_905_cast_fp16")]; tensor normed_889_axes_0 = const()[name = string("normed_889_axes_0"), val = tensor([-1])]; fp16 var_16262_to_fp16 = const()[name = string("op_16262_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_889_cast_fp16 = layer_norm(axes = normed_889_axes_0, epsilon = var_16262_to_fp16, x = input_905_cast_fp16)[name = string("normed_889_cast_fp16")]; tensor var_16275_split_sizes_0 = const()[name = string("op_16275_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16275_axis_0 = const()[name = string("op_16275_axis_0"), val = int32(-1)]; tensor var_16275_cast_fp16_0, tensor var_16275_cast_fp16_1 = split(axis = var_16275_axis_0, split_sizes = var_16275_split_sizes_0, x = normed_889_cast_fp16)[name = string("op_16275_cast_fp16")]; tensor const_520_to_fp16 = const()[name = string("const_520_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135246912)))]; tensor var_16278_cast_fp16 = mul(x = var_16275_cast_fp16_0, y = const_520_to_fp16)[name = string("op_16278_cast_fp16")]; tensor var_16291 = const()[name = string("op_16291"), val = tensor([0, 2, 1])]; tensor input_907_axes_0 = const()[name = string("input_907_axes_0"), val = tensor([2])]; tensor var_16292 = transpose(perm = var_16291, x = var_16278_cast_fp16)[name = string("transpose_3")]; tensor input_907 = expand_dims(axes = input_907_axes_0, x = var_16292)[name = string("input_907")]; string gate_137_pad_type_0 = const()[name = string("gate_137_pad_type_0"), val = string("valid")]; tensor gate_137_strides_0 = const()[name = string("gate_137_strides_0"), val = tensor([1, 1])]; tensor gate_137_pad_0 = const()[name = string("gate_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_137_dilations_0 = const()[name = string("gate_137_dilations_0"), val = tensor([1, 1])]; int32 gate_137_groups_0 = const()[name = string("gate_137_groups_0"), val = int32(1)]; tensor gate_137 = conv(dilations = gate_137_dilations_0, groups = gate_137_groups_0, pad = gate_137_pad_0, pad_type = gate_137_pad_type_0, strides = gate_137_strides_0, weight = layers_34_mlp_gate_proj_weight_palettized, x = input_907)[name = string("gate_137")]; string up_pad_type_0 = const()[name = string("up_pad_type_0"), val = string("valid")]; tensor up_strides_0 = const()[name = string("up_strides_0"), val = tensor([1, 1])]; tensor up_pad_0 = const()[name = string("up_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_dilations_0 = const()[name = string("up_dilations_0"), val = tensor([1, 1])]; int32 up_groups_0 = const()[name = string("up_groups_0"), val = int32(1)]; tensor up = conv(dilations = up_dilations_0, groups = up_groups_0, pad = up_pad_0, pad_type = up_pad_type_0, strides = up_strides_0, weight = layers_34_mlp_up_proj_weight_palettized, x = input_907)[name = string("up")]; string gate_mode_0 = const()[name = string("gate_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate = gelu(mode = gate_mode_0, x = gate_137)[name = string("gate")]; tensor input_909 = mul(x = gate, y = up)[name = string("input_909")]; string mlp_out_pad_type_0 = const()[name = string("mlp_out_pad_type_0"), val = string("valid")]; tensor mlp_out_strides_0 = const()[name = string("mlp_out_strides_0"), val = tensor([1, 1])]; tensor mlp_out_pad_0 = const()[name = string("mlp_out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_dilations_0 = const()[name = string("mlp_out_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_groups_0 = const()[name = string("mlp_out_groups_0"), val = int32(1)]; tensor mlp_out = conv(dilations = mlp_out_dilations_0, groups = mlp_out_groups_0, pad = mlp_out_pad_0, pad_type = mlp_out_pad_type_0, strides = mlp_out_strides_0, weight = layers_34_mlp_down_proj_weight_palettized, x = input_909)[name = string("mlp_out")]; tensor var_16332_axes_0 = const()[name = string("op_16332_axes_0"), val = tensor([2])]; tensor var_16332 = squeeze(axes = var_16332_axes_0, x = mlp_out)[name = string("op_16332")]; tensor var_16336 = const()[name = string("op_16336"), val = tensor([0, 2, 1])]; int32 var_16342 = const()[name = string("op_16342"), val = int32(-1)]; fp16 const_521_promoted_to_fp16 = const()[name = string("const_521_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_923 = transpose(perm = var_16336, x = var_16332)[name = string("transpose_2")]; tensor var_16348_cast_fp16 = mul(x = x_923, y = const_521_promoted_to_fp16)[name = string("op_16348_cast_fp16")]; bool input_911_interleave_0 = const()[name = string("input_911_interleave_0"), val = bool(false)]; tensor input_911_cast_fp16 = concat(axis = var_16342, interleave = input_911_interleave_0, values = (x_923, var_16348_cast_fp16))[name = string("input_911_cast_fp16")]; tensor normed_893_axes_0 = const()[name = string("normed_893_axes_0"), val = tensor([-1])]; fp16 var_16340_to_fp16 = const()[name = string("op_16340_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_893_cast_fp16 = layer_norm(axes = normed_893_axes_0, epsilon = var_16340_to_fp16, x = input_911_cast_fp16)[name = string("normed_893_cast_fp16")]; tensor var_16353_split_sizes_0 = const()[name = string("op_16353_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16353_axis_0 = const()[name = string("op_16353_axis_0"), val = int32(-1)]; tensor var_16353_cast_fp16_0, tensor var_16353_cast_fp16_1 = split(axis = var_16353_axis_0, split_sizes = var_16353_split_sizes_0, x = normed_893_cast_fp16)[name = string("op_16353_cast_fp16")]; tensor const_522_to_fp16 = const()[name = string("const_522_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135250048)))]; tensor var_16356_cast_fp16 = mul(x = var_16353_cast_fp16_0, y = const_522_to_fp16)[name = string("op_16356_cast_fp16")]; tensor hidden_states_417_cast_fp16 = add(x = x_919_cast_fp16, y = var_16356_cast_fp16)[name = string("hidden_states_417_cast_fp16")]; tensor per_layer_slice_begin_0 = const()[name = string("per_layer_slice_begin_0"), val = tensor([0, 0, 8704])]; tensor per_layer_slice_end_0 = const()[name = string("per_layer_slice_end_0"), val = tensor([1, 1, 1])]; tensor per_layer_slice_end_mask_0 = const()[name = string("per_layer_slice_end_mask_0"), val = tensor([true, true, true])]; tensor per_layer_slice_cast_fp16 = slice_by_index(begin = per_layer_slice_begin_0, end = per_layer_slice_end_0, end_mask = per_layer_slice_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_cast_fp16")]; tensor gated_137 = linear(bias = linear_0_bias_0, weight = layers_34_per_layer_input_gate_weight_palettized, x = hidden_states_417_cast_fp16)[name = string("linear_68")]; string gated_mode_0 = const()[name = string("gated_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated = gelu(mode = gated_mode_0, x = gated_137)[name = string("gated")]; tensor input_915_cast_fp16 = mul(x = gated, y = per_layer_slice_cast_fp16)[name = string("input_915_cast_fp16")]; tensor layers_34_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135253184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135449856))))[name = string("layers_34_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_69_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_34_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_915_cast_fp16)[name = string("linear_69_cast_fp16")]; int32 var_16393 = const()[name = string("op_16393"), val = int32(-1)]; fp16 const_523_promoted_to_fp16 = const()[name = string("const_523_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_16399_cast_fp16 = mul(x = linear_69_cast_fp16, y = const_523_promoted_to_fp16)[name = string("op_16399_cast_fp16")]; bool input_917_interleave_0 = const()[name = string("input_917_interleave_0"), val = bool(false)]; tensor input_917_cast_fp16 = concat(axis = var_16393, interleave = input_917_interleave_0, values = (linear_69_cast_fp16, var_16399_cast_fp16))[name = string("input_917_cast_fp16")]; tensor normed_897_axes_0 = const()[name = string("normed_897_axes_0"), val = tensor([-1])]; fp16 var_16391_to_fp16 = const()[name = string("op_16391_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_897_cast_fp16 = layer_norm(axes = normed_897_axes_0, epsilon = var_16391_to_fp16, x = input_917_cast_fp16)[name = string("normed_897_cast_fp16")]; tensor var_16404_split_sizes_0 = const()[name = string("op_16404_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16404_axis_0 = const()[name = string("op_16404_axis_0"), val = int32(-1)]; tensor var_16404_cast_fp16_0, tensor var_16404_cast_fp16_1 = split(axis = var_16404_axis_0, split_sizes = var_16404_split_sizes_0, x = normed_897_cast_fp16)[name = string("op_16404_cast_fp16")]; tensor const_524_to_fp16 = const()[name = string("const_524_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135451456)))]; tensor var_16407_cast_fp16 = mul(x = var_16404_cast_fp16_0, y = const_524_to_fp16)[name = string("op_16407_cast_fp16")]; tensor hidden_states_421_cast_fp16 = add(x = hidden_states_417_cast_fp16, y = var_16407_cast_fp16)[name = string("hidden_states_421_cast_fp16")]; tensor layers_34_layer_scalar_to_fp16 = const()[name = string("layers_34_layer_scalar_to_fp16"), val = tensor([0x1.56p-3])]; tensor x_931_cast_fp16 = mul(x = hidden_states_421_cast_fp16, y = layers_34_layer_scalar_to_fp16)[name = string("x_931_cast_fp16")]; int32 var_16415 = const()[name = string("op_16415"), val = int32(-1)]; fp16 const_525_promoted_to_fp16 = const()[name = string("const_525_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_16421_cast_fp16 = mul(x = x_931_cast_fp16, y = const_525_promoted_to_fp16)[name = string("op_16421_cast_fp16")]; bool input_919_interleave_0 = const()[name = string("input_919_interleave_0"), val = bool(false)]; tensor input_919_cast_fp16 = concat(axis = var_16415, interleave = input_919_interleave_0, values = (x_931_cast_fp16, var_16421_cast_fp16))[name = string("input_919_cast_fp16")]; tensor normed_901_axes_0 = const()[name = string("normed_901_axes_0"), val = tensor([-1])]; fp16 var_16413_to_fp16 = const()[name = string("op_16413_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_901_cast_fp16 = layer_norm(axes = normed_901_axes_0, epsilon = var_16413_to_fp16, x = input_919_cast_fp16)[name = string("normed_901_cast_fp16")]; tensor var_16426_split_sizes_0 = const()[name = string("op_16426_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_16426_axis_0 = const()[name = string("op_16426_axis_0"), val = int32(-1)]; tensor var_16426_cast_fp16_0, tensor var_16426_cast_fp16_1 = split(axis = var_16426_axis_0, split_sizes = var_16426_split_sizes_0, x = normed_901_cast_fp16)[name = string("op_16426_cast_fp16")]; tensor const_526_to_fp16 = const()[name = string("const_526_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135454592)))]; tensor var_16429_cast_fp16 = mul(x = var_16426_cast_fp16_0, y = const_526_to_fp16)[name = string("op_16429_cast_fp16")]; tensor var_16439 = const()[name = string("op_16439"), val = tensor([0, 2, 1])]; tensor squeeze_35_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1135457728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336784384))))[name = string("squeeze_35_palettized")]; string var_16455_pad_type_0 = const()[name = string("op_16455_pad_type_0"), val = string("valid")]; int32 var_16455_groups_0 = const()[name = string("op_16455_groups_0"), val = int32(1)]; tensor var_16455_strides_0 = const()[name = string("op_16455_strides_0"), val = tensor([1])]; tensor var_16455_pad_0 = const()[name = string("op_16455_pad_0"), val = tensor([0, 0])]; tensor var_16455_dilations_0 = const()[name = string("op_16455_dilations_0"), val = tensor([1])]; tensor var_16440 = transpose(perm = var_16439, x = var_16429_cast_fp16)[name = string("transpose_1")]; tensor var_16455 = conv(dilations = var_16455_dilations_0, groups = var_16455_groups_0, pad = var_16455_pad_0, pad_type = var_16455_pad_type_0, strides = var_16455_strides_0, weight = squeeze_35_palettized, x = var_16440)[name = string("op_16455")]; tensor var_16459 = const()[name = string("op_16459"), val = tensor([0, 2, 1])]; fp16 _inversed_16462_y_0_to_fp16 = const()[name = string("_inversed_16462_y_0_to_fp16"), val = fp16(0x1.11p-5)]; tensor logits_1 = transpose(perm = var_16459, x = var_16455)[name = string("transpose_0")]; tensor _inversed_16462_cast_fp16 = mul(x = logits_1, y = _inversed_16462_y_0_to_fp16)[name = string("_inversed_16462_cast_fp16")]; tensor var_16463_cast_fp16 = tanh(x = _inversed_16462_cast_fp16)[name = string("op_16463_cast_fp16")]; fp16 var_16464_to_fp16 = const()[name = string("op_16464_to_fp16"), val = fp16(0x1.ep+4)]; tensor logits_3_cast_fp16 = mul(x = var_16463_cast_fp16, y = var_16464_to_fp16)[name = string("logits_3_cast_fp16")]; tensor logits_axes_0 = const()[name = string("logits_axes_0"), val = tensor([0])]; tensor logits_cast_fp16 = squeeze(axes = logits_axes_0, x = logits_3_cast_fp16)[name = string("logits_cast_fp16")]; int32 var_16469 = const()[name = string("op_16469"), val = int32(-1)]; int32 token_id_axis_0 = const()[name = string("token_id_axis_0"), val = int32(-1)]; bool token_id_keep_dims_0 = const()[name = string("token_id_keep_dims_0"), val = bool(false)]; string token_id_output_dtype_0 = const()[name = string("token_id_output_dtype_0"), val = string("int32")]; tensor token_id = reduce_argmax(axis = token_id_axis_0, keep_dims = token_id_keep_dims_0, output_dtype = token_id_output_dtype_0, x = logits_cast_fp16)[name = string("token_id_cast_fp16")]; tensor var_16471_axes_0 = const()[name = string("op_16471_axes_0"), val = tensor([-1])]; tensor var_16471 = expand_dims(axes = var_16471_axes_0, x = token_id)[name = string("op_16471")]; bool var_16472_validate_indices_0 = const()[name = string("op_16472_validate_indices_0"), val = bool(false)]; tensor var_16472_cast_fp16 = gather_along_axis(axis = var_16469, indices = var_16471, validate_indices = var_16472_validate_indices_0, x = logits_cast_fp16)[name = string("op_16472_cast_fp16")]; tensor var_16473_axes_0 = const()[name = string("op_16473_axes_0"), val = tensor([-1])]; tensor token_logit = squeeze(axes = var_16473_axes_0, x = var_16472_cast_fp16)[name = string("op_16473_cast_fp16")]; } -> (token_id, token_logit); }