program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}})] { func decode_q1(tensor K_full_in, tensor K_sliding_in, tensor V_full_in, tensor V_sliding_in, tensor causal_mask_full, tensor causal_mask_sliding, tensor cos_f, tensor cos_s, tensor hidden_states, tensor per_layer_combined, tensor sin_f, tensor sin_s, tensor update_mask) { tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2621568))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_q_norm_weight = const()[name = string("layers_0_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2623680)))]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2624256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3279680))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3280256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3935680))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_self_attn_k_norm_weight = const()[name = string("layers_0_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3936256)))]; tensor layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3936832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17044096))))[name = string("layers_0_mlp_gate_proj_weight_palettized")]; tensor layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17054400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30161664))))[name = string("layers_0_mlp_up_proj_weight_palettized")]; tensor layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30171968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43279232))))[name = string("layers_0_mlp_down_proj_weight_palettized")]; tensor layers_0_post_feedforward_layernorm_weight = const()[name = string("layers_0_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43281856)))]; tensor layers_0_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43287040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43614784))))[name = string("layers_0_per_layer_input_gate_weight_palettized")]; tensor layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43615104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46236608))))[name = string("layers_1_self_attn_q_proj_weight_palettized")]; tensor layers_1_self_attn_q_norm_weight = const()[name = string("layers_1_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46238720)))]; tensor layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46239296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46894720))))[name = string("layers_1_self_attn_k_proj_weight_palettized")]; tensor layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46895296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47550720))))[name = string("layers_1_self_attn_v_proj_weight_palettized")]; tensor layers_1_self_attn_k_norm_weight = const()[name = string("layers_1_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47551296)))]; tensor layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47551872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60659136))))[name = string("layers_1_mlp_gate_proj_weight_palettized")]; tensor layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60669440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73776704))))[name = string("layers_1_mlp_up_proj_weight_palettized")]; tensor layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73787008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86894272))))[name = string("layers_1_mlp_down_proj_weight_palettized")]; tensor layers_1_post_feedforward_layernorm_weight = const()[name = string("layers_1_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86896896)))]; tensor layers_1_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86902080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87229824))))[name = string("layers_1_per_layer_input_gate_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87230144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89851648))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_q_norm_weight = const()[name = string("layers_2_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89853760)))]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89854336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90509760))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90510336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91165760))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_self_attn_k_norm_weight = const()[name = string("layers_2_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91166336)))]; tensor layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91166912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104274176))))[name = string("layers_2_mlp_gate_proj_weight_palettized")]; tensor layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104284480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117391744))))[name = string("layers_2_mlp_up_proj_weight_palettized")]; tensor layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117402048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130509312))))[name = string("layers_2_mlp_down_proj_weight_palettized")]; tensor layers_2_post_feedforward_layernorm_weight = const()[name = string("layers_2_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130511936)))]; tensor layers_2_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130517120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130844864))))[name = string("layers_2_per_layer_input_gate_weight_palettized")]; tensor layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130845184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133466688))))[name = string("layers_3_self_attn_q_proj_weight_palettized")]; tensor layers_3_self_attn_q_norm_weight = const()[name = string("layers_3_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133468800)))]; tensor layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133469376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134124800))))[name = string("layers_3_self_attn_k_proj_weight_palettized")]; tensor layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134125376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134780800))))[name = string("layers_3_self_attn_v_proj_weight_palettized")]; tensor layers_3_self_attn_k_norm_weight = const()[name = string("layers_3_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134781376)))]; tensor layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134781952))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147889216))))[name = string("layers_3_mlp_gate_proj_weight_palettized")]; tensor layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147899520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161006784))))[name = string("layers_3_mlp_up_proj_weight_palettized")]; tensor layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161017088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174124352))))[name = string("layers_3_mlp_down_proj_weight_palettized")]; tensor layers_3_post_feedforward_layernorm_weight = const()[name = string("layers_3_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174126976)))]; tensor layers_3_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174132160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174459904))))[name = string("layers_3_per_layer_input_gate_weight_palettized")]; tensor layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174460224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177081728))))[name = string("layers_4_self_attn_q_proj_weight_palettized")]; tensor layers_4_self_attn_q_norm_weight = const()[name = string("layers_4_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177083840)))]; tensor layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177084416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177739840))))[name = string("layers_4_self_attn_k_proj_weight_palettized")]; tensor layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177740416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178395840))))[name = string("layers_4_self_attn_v_proj_weight_palettized")]; tensor layers_4_self_attn_k_norm_weight = const()[name = string("layers_4_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178396416)))]; tensor layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178396992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(191504256))))[name = string("layers_4_mlp_gate_proj_weight_palettized")]; tensor layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(191514560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204621824))))[name = string("layers_4_mlp_up_proj_weight_palettized")]; tensor layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204632128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217739392))))[name = string("layers_4_mlp_down_proj_weight_palettized")]; tensor layers_4_post_feedforward_layernorm_weight = const()[name = string("layers_4_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217742016)))]; tensor layers_4_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217747200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218074944))))[name = string("layers_4_per_layer_input_gate_weight_palettized")]; tensor layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218075264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223318208))))[name = string("layers_5_self_attn_q_proj_weight_palettized")]; tensor layers_5_self_attn_q_norm_weight = const()[name = string("layers_5_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223322368)))]; tensor layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223323456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224634240))))[name = string("layers_5_self_attn_k_proj_weight_palettized")]; tensor layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224635328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225946112))))[name = string("layers_5_self_attn_v_proj_weight_palettized")]; tensor layers_5_self_attn_k_norm_weight = const()[name = string("layers_5_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225947200)))]; tensor layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225948288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239055552))))[name = string("layers_5_mlp_gate_proj_weight_palettized")]; tensor layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239065856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252173120))))[name = string("layers_5_mlp_up_proj_weight_palettized")]; tensor layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252183424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265290688))))[name = string("layers_5_mlp_down_proj_weight_palettized")]; tensor layers_5_post_feedforward_layernorm_weight = const()[name = string("layers_5_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265293312)))]; tensor layers_5_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265298496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265626240))))[name = string("layers_5_per_layer_input_gate_weight_palettized")]; tensor layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265626560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268248064))))[name = string("layers_6_self_attn_q_proj_weight_palettized")]; tensor layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268250176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268905600))))[name = string("layers_6_self_attn_k_proj_weight_palettized")]; tensor layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268906176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269561600))))[name = string("layers_6_self_attn_v_proj_weight_palettized")]; tensor layers_6_self_attn_k_norm_weight = const()[name = string("layers_6_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269562176)))]; tensor layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269562752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282670016))))[name = string("layers_6_mlp_gate_proj_weight_palettized")]; tensor layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282680320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295787584))))[name = string("layers_6_mlp_up_proj_weight_palettized")]; tensor layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295797888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308905152))))[name = string("layers_6_mlp_down_proj_weight_palettized")]; tensor layers_6_post_feedforward_layernorm_weight = const()[name = string("layers_6_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308907776)))]; tensor layers_6_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308912960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(309240704))))[name = string("layers_6_per_layer_input_gate_weight_palettized")]; tensor layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(309241024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311862528))))[name = string("layers_7_self_attn_q_proj_weight_palettized")]; tensor layers_7_self_attn_q_norm_weight = const()[name = string("layers_7_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311864640)))]; tensor layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311865216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312520640))))[name = string("layers_7_self_attn_k_proj_weight_palettized")]; tensor layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312521216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313176640))))[name = string("layers_7_self_attn_v_proj_weight_palettized")]; tensor layers_7_self_attn_k_norm_weight = const()[name = string("layers_7_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313177216)))]; tensor layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313177792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326285056))))[name = string("layers_7_mlp_gate_proj_weight_palettized")]; tensor layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326295360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339402624))))[name = string("layers_7_mlp_up_proj_weight_palettized")]; tensor layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339412928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352520192))))[name = string("layers_7_mlp_down_proj_weight_palettized")]; tensor layers_7_post_feedforward_layernorm_weight = const()[name = string("layers_7_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352522816)))]; tensor layers_7_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352528000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352855744))))[name = string("layers_7_per_layer_input_gate_weight_palettized")]; tensor layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352856064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355477568))))[name = string("layers_8_self_attn_q_proj_weight_palettized")]; tensor layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355479680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356135104))))[name = string("layers_8_self_attn_k_proj_weight_palettized")]; tensor layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356135680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356791104))))[name = string("layers_8_self_attn_v_proj_weight_palettized")]; tensor layers_8_self_attn_k_norm_weight = const()[name = string("layers_8_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356791680)))]; tensor layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356792256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369899520))))[name = string("layers_8_mlp_gate_proj_weight_palettized")]; tensor layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369909824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(383017088))))[name = string("layers_8_mlp_up_proj_weight_palettized")]; tensor layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(383027392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396134656))))[name = string("layers_8_mlp_down_proj_weight_palettized")]; tensor layers_8_post_feedforward_layernorm_weight = const()[name = string("layers_8_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396137280)))]; tensor layers_8_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396142464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396470208))))[name = string("layers_8_per_layer_input_gate_weight_palettized")]; tensor layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396470528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399092032))))[name = string("layers_9_self_attn_q_proj_weight_palettized")]; tensor layers_9_self_attn_q_norm_weight = const()[name = string("layers_9_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399094144)))]; tensor layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399094720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399750144))))[name = string("layers_9_self_attn_k_proj_weight_palettized")]; tensor layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399750720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400406144))))[name = string("layers_9_self_attn_v_proj_weight_palettized")]; tensor layers_9_self_attn_k_norm_weight = const()[name = string("layers_9_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400406720)))]; tensor layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400407296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413514560))))[name = string("layers_9_mlp_gate_proj_weight_palettized")]; tensor layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413524864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(426632128))))[name = string("layers_9_mlp_up_proj_weight_palettized")]; tensor layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(426642432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(439749696))))[name = string("layers_9_mlp_down_proj_weight_palettized")]; tensor layers_9_post_feedforward_layernorm_weight = const()[name = string("layers_9_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(439752320)))]; tensor layers_9_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(439757504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(440085248))))[name = string("layers_9_per_layer_input_gate_weight_palettized")]; tensor layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(440085568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(442707072))))[name = string("layers_10_self_attn_q_proj_weight_palettized")]; tensor layers_10_self_attn_q_norm_weight = const()[name = string("layers_10_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(442709184)))]; tensor layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(442709760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(443365184))))[name = string("layers_10_self_attn_k_proj_weight_palettized")]; tensor layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(443365760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(444021184))))[name = string("layers_10_self_attn_v_proj_weight_palettized")]; tensor layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(444021760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(457129024))))[name = string("layers_10_mlp_gate_proj_weight_palettized")]; tensor layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(457139328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(470246592))))[name = string("layers_10_mlp_up_proj_weight_palettized")]; tensor layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(470256896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483364160))))[name = string("layers_10_mlp_down_proj_weight_palettized")]; tensor layers_10_post_feedforward_layernorm_weight = const()[name = string("layers_10_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483366784)))]; tensor layers_10_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483371968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483699712))))[name = string("layers_10_per_layer_input_gate_weight_palettized")]; tensor layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483700032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488942976))))[name = string("layers_11_self_attn_q_proj_weight_palettized")]; tensor layers_11_self_attn_q_norm_weight = const()[name = string("layers_11_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488947136)))]; tensor layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488948224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490259008))))[name = string("layers_11_self_attn_k_proj_weight_palettized")]; tensor layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490260096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(491570880))))[name = string("layers_11_self_attn_v_proj_weight_palettized")]; tensor layers_11_self_attn_k_norm_weight = const()[name = string("layers_11_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(491571968)))]; tensor layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(491573056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(504680320))))[name = string("layers_11_mlp_gate_proj_weight_palettized")]; tensor layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(504690624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(517797888))))[name = string("layers_11_mlp_up_proj_weight_palettized")]; tensor layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(517808192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530915456))))[name = string("layers_11_mlp_down_proj_weight_palettized")]; tensor layers_11_post_feedforward_layernorm_weight = const()[name = string("layers_11_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530918080)))]; tensor layers_11_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530923264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531251008))))[name = string("layers_11_per_layer_input_gate_weight_palettized")]; tensor var_736_begin_0 = const()[name = string("op_736_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_736_end_0 = const()[name = string("op_736_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_736_end_mask_0 = const()[name = string("op_736_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_736_squeeze_mask_0 = const()[name = string("op_736_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_736_cast_fp16 = slice_by_index(begin = var_736_begin_0, end = var_736_end_0, end_mask = var_736_end_mask_0, squeeze_mask = var_736_squeeze_mask_0, x = K_sliding_in)[name = string("op_736_cast_fp16")]; tensor K_sliding_slot_1_axes_0 = const()[name = string("K_sliding_slot_1_axes_0"), val = tensor([0])]; tensor K_sliding_slot_1_cast_fp16 = expand_dims(axes = K_sliding_slot_1_axes_0, x = var_736_cast_fp16)[name = string("K_sliding_slot_1_cast_fp16")]; tensor var_741_begin_0 = const()[name = string("op_741_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_741_end_0 = const()[name = string("op_741_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_741_end_mask_0 = const()[name = string("op_741_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_741_squeeze_mask_0 = const()[name = string("op_741_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_741_cast_fp16 = slice_by_index(begin = var_741_begin_0, end = var_741_end_0, end_mask = var_741_end_mask_0, squeeze_mask = var_741_squeeze_mask_0, x = V_sliding_in)[name = string("op_741_cast_fp16")]; tensor V_sliding_slot_1_axes_0 = const()[name = string("V_sliding_slot_1_axes_0"), val = tensor([0])]; tensor V_sliding_slot_1_cast_fp16 = expand_dims(axes = V_sliding_slot_1_axes_0, x = var_741_cast_fp16)[name = string("V_sliding_slot_1_cast_fp16")]; int32 var_748 = const()[name = string("op_748"), val = int32(-1)]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_750_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_750_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_748, interleave = input_1_interleave_0, values = (hidden_states, var_750_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_745_to_fp16 = const()[name = string("op_745_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_745_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_755_split_sizes_0 = const()[name = string("op_755_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_755_axis_0 = const()[name = string("op_755_axis_0"), val = int32(-1)]; tensor var_755_cast_fp16_0, tensor var_755_cast_fp16_1 = split(axis = var_755_axis_0, split_sizes = var_755_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_755_cast_fp16")]; tensor layers_0_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531251328)))]; tensor h_1_cast_fp16 = mul(x = var_755_cast_fp16_0, y = layers_0_input_layernorm_weight_promoted_to_fp16)[name = string("h_1_cast_fp16")]; tensor var_761 = const()[name = string("op_761"), val = tensor([0, 2, 1])]; tensor var_764_axes_0 = const()[name = string("op_764_axes_0"), val = tensor([2])]; tensor var_762_cast_fp16 = transpose(perm = var_761, x = h_1_cast_fp16)[name = string("transpose_215")]; tensor var_764_cast_fp16 = expand_dims(axes = var_764_axes_0, x = var_762_cast_fp16)[name = string("op_764_cast_fp16")]; string var_780_pad_type_0 = const()[name = string("op_780_pad_type_0"), val = string("valid")]; tensor var_780_strides_0 = const()[name = string("op_780_strides_0"), val = tensor([1, 1])]; tensor var_780_pad_0 = const()[name = string("op_780_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_780_dilations_0 = const()[name = string("op_780_dilations_0"), val = tensor([1, 1])]; int32 var_780_groups_0 = const()[name = string("op_780_groups_0"), val = int32(1)]; tensor var_780 = conv(dilations = var_780_dilations_0, groups = var_780_groups_0, pad = var_780_pad_0, pad_type = var_780_pad_type_0, strides = var_780_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_764_cast_fp16)[name = string("op_780")]; tensor var_785 = const()[name = string("op_785"), val = tensor([1, 8, 256, 1])]; tensor var_786 = reshape(shape = var_785, x = var_780)[name = string("op_786")]; tensor var_791 = const()[name = string("op_791"), val = tensor([0, 1, 3, 2])]; tensor var_801 = const()[name = string("op_801"), val = tensor([1, 8, 256])]; tensor var_792 = transpose(perm = var_791, x = var_786)[name = string("transpose_214")]; tensor x_1 = reshape(shape = var_801, x = var_792)[name = string("x_1")]; int32 var_807 = const()[name = string("op_807"), val = int32(-1)]; fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; tensor var_809 = mul(x = x_1, y = const_1_promoted)[name = string("op_809")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_807, interleave = input_5_interleave_0, values = (x_1, var_809))[name = string("input_5")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_804_to_fp16 = const()[name = string("op_804_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_804_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; tensor var_814_split_sizes_0 = const()[name = string("op_814_split_sizes_0"), val = tensor([256, 256])]; int32 var_814_axis_0 = const()[name = string("op_814_axis_0"), val = int32(-1)]; tensor var_814_0, tensor var_814_1 = split(axis = var_814_axis_0, split_sizes = var_814_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_814")]; tensor var_816 = mul(x = var_814_0, y = layers_0_self_attn_q_norm_weight)[name = string("op_816")]; tensor var_821 = const()[name = string("op_821"), val = tensor([1, 8, 1, 256])]; tensor q_3 = reshape(shape = var_821, x = var_816)[name = string("q_3")]; tensor var_823_cast_fp16 = mul(x = q_3, y = cos_s)[name = string("op_823_cast_fp16")]; tensor var_824_split_sizes_0 = const()[name = string("op_824_split_sizes_0"), val = tensor([128, 128])]; int32 var_824_axis_0 = const()[name = string("op_824_axis_0"), val = int32(-1)]; tensor var_824_0, tensor var_824_1 = split(axis = var_824_axis_0, split_sizes = var_824_split_sizes_0, x = q_3)[name = string("op_824")]; fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; tensor var_826 = mul(x = var_824_1, y = const_2_promoted)[name = string("op_826")]; int32 var_828 = const()[name = string("op_828"), val = int32(-1)]; bool var_829_interleave_0 = const()[name = string("op_829_interleave_0"), val = bool(false)]; tensor var_829 = concat(axis = var_828, interleave = var_829_interleave_0, values = (var_826, var_824_0))[name = string("op_829")]; tensor var_830_cast_fp16 = mul(x = var_829, y = sin_s)[name = string("op_830_cast_fp16")]; tensor q_7_cast_fp16 = add(x = var_823_cast_fp16, y = var_830_cast_fp16)[name = string("q_7_cast_fp16")]; string var_843_pad_type_0 = const()[name = string("op_843_pad_type_0"), val = string("valid")]; tensor var_843_strides_0 = const()[name = string("op_843_strides_0"), val = tensor([1, 1])]; tensor var_843_pad_0 = const()[name = string("op_843_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_843_dilations_0 = const()[name = string("op_843_dilations_0"), val = tensor([1, 1])]; int32 var_843_groups_0 = const()[name = string("op_843_groups_0"), val = int32(1)]; tensor var_843 = conv(dilations = var_843_dilations_0, groups = var_843_groups_0, pad = var_843_pad_0, pad_type = var_843_pad_type_0, strides = var_843_strides_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_764_cast_fp16)[name = string("op_843")]; tensor var_848 = const()[name = string("op_848"), val = tensor([1, 2, 256, 1])]; tensor var_849 = reshape(shape = var_848, x = var_843)[name = string("op_849")]; tensor var_854 = const()[name = string("op_854"), val = tensor([0, 1, 3, 2])]; string var_871_pad_type_0 = const()[name = string("op_871_pad_type_0"), val = string("valid")]; tensor var_871_strides_0 = const()[name = string("op_871_strides_0"), val = tensor([1, 1])]; tensor var_871_pad_0 = const()[name = string("op_871_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_871_dilations_0 = const()[name = string("op_871_dilations_0"), val = tensor([1, 1])]; int32 var_871_groups_0 = const()[name = string("op_871_groups_0"), val = int32(1)]; tensor var_871 = conv(dilations = var_871_dilations_0, groups = var_871_groups_0, pad = var_871_pad_0, pad_type = var_871_pad_type_0, strides = var_871_strides_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_764_cast_fp16)[name = string("op_871")]; tensor var_876 = const()[name = string("op_876"), val = tensor([1, 2, 256, 1])]; tensor var_877 = reshape(shape = var_876, x = var_871)[name = string("op_877")]; tensor var_882 = const()[name = string("op_882"), val = tensor([0, 1, 3, 2])]; tensor var_892 = const()[name = string("op_892"), val = tensor([1, 2, 256])]; tensor var_855 = transpose(perm = var_854, x = var_849)[name = string("transpose_213")]; tensor x_3 = reshape(shape = var_892, x = var_855)[name = string("x_3")]; int32 var_898 = const()[name = string("op_898"), val = int32(-1)]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_900 = mul(x = x_3, y = const_3_promoted)[name = string("op_900")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7 = concat(axis = var_898, interleave = input_7_interleave_0, values = (x_3, var_900))[name = string("input_7")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_895_to_fp16 = const()[name = string("op_895_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_895_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; tensor var_905_split_sizes_0 = const()[name = string("op_905_split_sizes_0"), val = tensor([256, 256])]; int32 var_905_axis_0 = const()[name = string("op_905_axis_0"), val = int32(-1)]; tensor var_905_0, tensor var_905_1 = split(axis = var_905_axis_0, split_sizes = var_905_split_sizes_0, x = normed_9_cast_fp16)[name = string("op_905")]; tensor var_907 = mul(x = var_905_0, y = layers_0_self_attn_k_norm_weight)[name = string("op_907")]; tensor var_912 = const()[name = string("op_912"), val = tensor([1, 2, 1, 256])]; tensor q_5 = reshape(shape = var_912, x = var_907)[name = string("q_5")]; fp16 var_914_promoted = const()[name = string("op_914_promoted"), val = fp16(0x1p+1)]; tensor var_883 = transpose(perm = var_882, x = var_877)[name = string("transpose_212")]; tensor var_915 = pow(x = var_883, y = var_914_promoted)[name = string("op_915")]; tensor var_920_axes_0 = const()[name = string("op_920_axes_0"), val = tensor([-1])]; bool var_920_keep_dims_0 = const()[name = string("op_920_keep_dims_0"), val = bool(true)]; tensor var_920 = reduce_mean(axes = var_920_axes_0, keep_dims = var_920_keep_dims_0, x = var_915)[name = string("op_920")]; fp16 var_922_to_fp16 = const()[name = string("op_922_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_1_cast_fp16 = add(x = var_920, y = var_922_to_fp16)[name = string("mean_sq_1_cast_fp16")]; fp32 var_924_epsilon_0 = const()[name = string("op_924_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_924_cast_fp16 = rsqrt(epsilon = var_924_epsilon_0, x = mean_sq_1_cast_fp16)[name = string("op_924_cast_fp16")]; tensor input_11_cast_fp16 = mul(x = var_883, y = var_924_cast_fp16)[name = string("input_11_cast_fp16")]; tensor var_926_cast_fp16 = mul(x = q_5, y = cos_s)[name = string("op_926_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([128, 128])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(-1)]; tensor var_927_0, tensor var_927_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = q_5)[name = string("op_927")]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_929 = mul(x = var_927_1, y = const_4_promoted)[name = string("op_929")]; int32 var_931 = const()[name = string("op_931"), val = int32(-1)]; bool var_932_interleave_0 = const()[name = string("op_932_interleave_0"), val = bool(false)]; tensor var_932 = concat(axis = var_931, interleave = var_932_interleave_0, values = (var_929, var_927_0))[name = string("op_932")]; tensor var_933_cast_fp16 = mul(x = var_932, y = sin_s)[name = string("op_933_cast_fp16")]; tensor input_9_cast_fp16 = add(x = var_926_cast_fp16, y = var_933_cast_fp16)[name = string("input_9_cast_fp16")]; tensor k_padded_1_pad_0 = const()[name = string("k_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_1_mode_0 = const()[name = string("k_padded_1_mode_0"), val = string("constant")]; fp16 const_5_to_fp16 = const()[name = string("const_5_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_1_cast_fp16 = pad(constant_val = const_5_to_fp16, mode = k_padded_1_mode_0, pad = k_padded_1_pad_0, x = input_9_cast_fp16)[name = string("k_padded_1_cast_fp16")]; tensor v_padded_1_pad_0 = const()[name = string("v_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_1_mode_0 = const()[name = string("v_padded_1_mode_0"), val = string("constant")]; fp16 const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_1_cast_fp16 = pad(constant_val = const_6_to_fp16, mode = v_padded_1_mode_0, pad = v_padded_1_pad_0, x = input_11_cast_fp16)[name = string("v_padded_1_cast_fp16")]; tensor var_962_begin_0 = const()[name = string("op_962_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_962_end_0 = const()[name = string("op_962_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_962_end_mask_0 = const()[name = string("op_962_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_962_cast_fp16 = slice_by_index(begin = var_962_begin_0, end = var_962_end_0, end_mask = var_962_end_mask_0, x = K_sliding_slot_1_cast_fp16)[name = string("op_962_cast_fp16")]; int32 var_969 = const()[name = string("op_969"), val = int32(2)]; bool K_sliding_out_1_interleave_0 = const()[name = string("K_sliding_out_1_interleave_0"), val = bool(false)]; tensor K_sliding_out_1_cast_fp16 = concat(axis = var_969, interleave = K_sliding_out_1_interleave_0, values = (var_962_cast_fp16, k_padded_1_cast_fp16))[name = string("K_sliding_out_1_cast_fp16")]; tensor var_985_begin_0 = const()[name = string("op_985_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_985_end_0 = const()[name = string("op_985_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_985_end_mask_0 = const()[name = string("op_985_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_985_cast_fp16 = slice_by_index(begin = var_985_begin_0, end = var_985_end_0, end_mask = var_985_end_mask_0, x = V_sliding_slot_1_cast_fp16)[name = string("op_985_cast_fp16")]; int32 var_992 = const()[name = string("op_992"), val = int32(2)]; bool V_sliding_out_1_interleave_0 = const()[name = string("V_sliding_out_1_interleave_0"), val = bool(false)]; tensor V_sliding_out_1_cast_fp16 = concat(axis = var_992, interleave = V_sliding_out_1_interleave_0, values = (var_985_cast_fp16, v_padded_1_cast_fp16))[name = string("V_sliding_out_1_cast_fp16")]; tensor K_for_attn_1_begin_0 = const()[name = string("K_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_1_end_0 = const()[name = string("K_for_attn_1_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_1_end_mask_0 = const()[name = string("K_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_1_cast_fp16 = slice_by_index(begin = K_for_attn_1_begin_0, end = K_for_attn_1_end_0, end_mask = K_for_attn_1_end_mask_0, x = K_sliding_out_1_cast_fp16)[name = string("K_for_attn_1_cast_fp16")]; tensor V_for_attn_1_begin_0 = const()[name = string("V_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_1_end_0 = const()[name = string("V_for_attn_1_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_1_end_mask_0 = const()[name = string("V_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_1_cast_fp16 = slice_by_index(begin = V_for_attn_1_begin_0, end = V_for_attn_1_end_0, end_mask = V_for_attn_1_end_mask_0, x = V_sliding_out_1_cast_fp16)[name = string("V_for_attn_1_cast_fp16")]; tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = K_for_attn_1_cast_fp16)[name = string("transpose_211")]; tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; tensor concat_0 = const()[name = string("concat_0"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_0, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_1 = const()[name = string("concat_1"), val = tensor([-1, 1, 512, 256])]; tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_210")]; tensor reshape_1_cast_fp16 = reshape(shape = concat_1, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; tensor transpose_48_perm_0 = const()[name = string("transpose_48_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = V_for_attn_1_cast_fp16)[name = string("transpose_209")]; tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; tensor concat_2 = const()[name = string("concat_2"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_2, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_3 = const()[name = string("concat_3"), val = tensor([-1, 1, 512, 256])]; tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_208")]; tensor reshape_3_cast_fp16 = reshape(shape = concat_3, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor V_expanded_1_perm_0 = const()[name = string("V_expanded_1_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor transpose_48_cast_fp16 = transpose(perm = transpose_48_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_207")]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = q_7_cast_fp16, y = transpose_48_cast_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor x_7_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask_sliding)[name = string("x_7_cast_fp16")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = x_7_cast_fp16)[name = string("reduce_max_0")]; tensor var_1033 = sub(x = x_7_cast_fp16, y = reduce_max_0)[name = string("op_1033")]; tensor var_1039 = exp(x = var_1033)[name = string("op_1039")]; tensor var_1049_axes_0 = const()[name = string("op_1049_axes_0"), val = tensor([-1])]; bool var_1049_keep_dims_0 = const()[name = string("op_1049_keep_dims_0"), val = bool(true)]; tensor var_1049 = reduce_sum(axes = var_1049_axes_0, keep_dims = var_1049_keep_dims_0, x = var_1039)[name = string("op_1049")]; tensor var_1055_cast_fp16 = real_div(x = var_1039, y = var_1049)[name = string("op_1055_cast_fp16")]; bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; tensor V_expanded_1_cast_fp16 = transpose(perm = V_expanded_1_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_206")]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = var_1055_cast_fp16, y = V_expanded_1_cast_fp16)[name = string("attn_output_1_cast_fp16")]; tensor var_1066 = const()[name = string("op_1066"), val = tensor([0, 2, 1, 3])]; tensor var_1073 = const()[name = string("op_1073"), val = tensor([1, 1, -1])]; tensor var_1067_cast_fp16 = transpose(perm = var_1066, x = attn_output_1_cast_fp16)[name = string("transpose_205")]; tensor attn_output_3_cast_fp16 = reshape(shape = var_1073, x = var_1067_cast_fp16)[name = string("attn_output_3_cast_fp16")]; tensor var_1078 = const()[name = string("op_1078"), val = tensor([0, 2, 1])]; string var_1094_pad_type_0 = const()[name = string("op_1094_pad_type_0"), val = string("valid")]; int32 var_1094_groups_0 = const()[name = string("op_1094_groups_0"), val = int32(1)]; tensor var_1094_strides_0 = const()[name = string("op_1094_strides_0"), val = tensor([1])]; tensor var_1094_pad_0 = const()[name = string("op_1094_pad_0"), val = tensor([0, 0])]; tensor var_1094_dilations_0 = const()[name = string("op_1094_dilations_0"), val = tensor([1])]; tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531256512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533878016))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_1079_cast_fp16 = transpose(perm = var_1078, x = attn_output_3_cast_fp16)[name = string("transpose_204")]; tensor var_1094_cast_fp16 = conv(dilations = var_1094_dilations_0, groups = var_1094_groups_0, pad = var_1094_pad_0, pad_type = var_1094_pad_type_0, strides = var_1094_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1079_cast_fp16)[name = string("op_1094_cast_fp16")]; tensor var_1098 = const()[name = string("op_1098"), val = tensor([0, 2, 1])]; int32 var_1104 = const()[name = string("op_1104"), val = int32(-1)]; fp16 const_7_promoted_to_fp16 = const()[name = string("const_7_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_11_cast_fp16 = transpose(perm = var_1098, x = var_1094_cast_fp16)[name = string("transpose_203")]; tensor var_1106_cast_fp16 = mul(x = x_11_cast_fp16, y = const_7_promoted_to_fp16)[name = string("op_1106_cast_fp16")]; bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; tensor input_15_cast_fp16 = concat(axis = var_1104, interleave = input_15_interleave_0, values = (x_11_cast_fp16, var_1106_cast_fp16))[name = string("input_15_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_1101_to_fp16 = const()[name = string("op_1101_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1101_to_fp16, x = input_15_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor var_1111_split_sizes_0 = const()[name = string("op_1111_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1111_axis_0 = const()[name = string("op_1111_axis_0"), val = int32(-1)]; tensor var_1111_cast_fp16_0, tensor var_1111_cast_fp16_1 = split(axis = var_1111_axis_0, split_sizes = var_1111_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_1111_cast_fp16")]; tensor layers_0_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533880640)))]; tensor attn_output_5_cast_fp16 = mul(x = var_1111_cast_fp16_0, y = layers_0_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_5_cast_fp16")]; tensor x_13_cast_fp16 = add(x = hidden_states, y = attn_output_5_cast_fp16)[name = string("x_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-1)]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1122_cast_fp16 = mul(x = x_13_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1122_cast_fp16")]; bool input_17_interleave_0 = const()[name = string("input_17_interleave_0"), val = bool(false)]; tensor input_17_cast_fp16 = concat(axis = var_1120, interleave = input_17_interleave_0, values = (x_13_cast_fp16, var_1122_cast_fp16))[name = string("input_17_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_1117_to_fp16 = const()[name = string("op_1117_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1117_to_fp16, x = input_17_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_1127_split_sizes_0 = const()[name = string("op_1127_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1127_axis_0 = const()[name = string("op_1127_axis_0"), val = int32(-1)]; tensor var_1127_cast_fp16_0, tensor var_1127_cast_fp16_1 = split(axis = var_1127_axis_0, split_sizes = var_1127_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_1127_cast_fp16")]; tensor layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533885824)))]; tensor h_3_cast_fp16 = mul(x = var_1127_cast_fp16_0, y = layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_3_cast_fp16")]; tensor var_1138 = const()[name = string("op_1138"), val = tensor([0, 2, 1])]; tensor input_19_axes_0 = const()[name = string("input_19_axes_0"), val = tensor([2])]; tensor var_1139 = transpose(perm = var_1138, x = h_3_cast_fp16)[name = string("transpose_202")]; tensor input_19 = expand_dims(axes = input_19_axes_0, x = var_1139)[name = string("input_19")]; string gate_1_pad_type_0 = const()[name = string("gate_1_pad_type_0"), val = string("valid")]; tensor gate_1_strides_0 = const()[name = string("gate_1_strides_0"), val = tensor([1, 1])]; tensor gate_1_pad_0 = const()[name = string("gate_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_1_dilations_0 = const()[name = string("gate_1_dilations_0"), val = tensor([1, 1])]; int32 gate_1_groups_0 = const()[name = string("gate_1_groups_0"), val = int32(1)]; tensor gate_1 = conv(dilations = gate_1_dilations_0, groups = gate_1_groups_0, pad = gate_1_pad_0, pad_type = gate_1_pad_type_0, strides = gate_1_strides_0, weight = layers_0_mlp_gate_proj_weight_palettized, x = input_19)[name = string("gate_1")]; string up_1_pad_type_0 = const()[name = string("up_1_pad_type_0"), val = string("valid")]; tensor up_1_strides_0 = const()[name = string("up_1_strides_0"), val = tensor([1, 1])]; tensor up_1_pad_0 = const()[name = string("up_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_1_dilations_0 = const()[name = string("up_1_dilations_0"), val = tensor([1, 1])]; int32 up_1_groups_0 = const()[name = string("up_1_groups_0"), val = int32(1)]; tensor up_1 = conv(dilations = up_1_dilations_0, groups = up_1_groups_0, pad = up_1_pad_0, pad_type = up_1_pad_type_0, strides = up_1_strides_0, weight = layers_0_mlp_up_proj_weight_palettized, x = input_19)[name = string("up_1")]; string gate_3_mode_0 = const()[name = string("gate_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_3 = gelu(mode = gate_3_mode_0, x = gate_1)[name = string("gate_3")]; tensor input_21 = mul(x = gate_3, y = up_1)[name = string("input_21")]; string mlp_out_1_pad_type_0 = const()[name = string("mlp_out_1_pad_type_0"), val = string("valid")]; tensor mlp_out_1_strides_0 = const()[name = string("mlp_out_1_strides_0"), val = tensor([1, 1])]; tensor mlp_out_1_pad_0 = const()[name = string("mlp_out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_1_dilations_0 = const()[name = string("mlp_out_1_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_1_groups_0 = const()[name = string("mlp_out_1_groups_0"), val = int32(1)]; tensor mlp_out_1 = conv(dilations = mlp_out_1_dilations_0, groups = mlp_out_1_groups_0, pad = mlp_out_1_pad_0, pad_type = mlp_out_1_pad_type_0, strides = mlp_out_1_strides_0, weight = layers_0_mlp_down_proj_weight_palettized, x = input_21)[name = string("mlp_out_1")]; tensor var_1179_axes_0 = const()[name = string("op_1179_axes_0"), val = tensor([2])]; tensor var_1179 = squeeze(axes = var_1179_axes_0, x = mlp_out_1)[name = string("op_1179")]; tensor var_1183 = const()[name = string("op_1183"), val = tensor([0, 2, 1])]; int32 var_1189 = const()[name = string("op_1189"), val = int32(-1)]; fp16 const_9_promoted = const()[name = string("const_9_promoted"), val = fp16(-0x1p+0)]; tensor x_15 = transpose(perm = var_1183, x = var_1179)[name = string("transpose_201")]; tensor var_1191 = mul(x = x_15, y = const_9_promoted)[name = string("op_1191")]; bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; tensor input_23 = concat(axis = var_1189, interleave = input_23_interleave_0, values = (x_15, var_1191))[name = string("input_23")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_1186_to_fp16 = const()[name = string("op_1186_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1186_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; tensor var_1196_split_sizes_0 = const()[name = string("op_1196_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1196_axis_0 = const()[name = string("op_1196_axis_0"), val = int32(-1)]; tensor var_1196_0, tensor var_1196_1 = split(axis = var_1196_axis_0, split_sizes = var_1196_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_1196")]; tensor hidden_states_3 = mul(x = var_1196_0, y = layers_0_post_feedforward_layernorm_weight)[name = string("hidden_states_3")]; tensor hidden_states_5_cast_fp16 = add(x = x_13_cast_fp16, y = hidden_states_3)[name = string("hidden_states_5_cast_fp16")]; tensor per_layer_slice_1_begin_0 = const()[name = string("per_layer_slice_1_begin_0"), val = tensor([0, 0, 3072])]; tensor per_layer_slice_1_end_0 = const()[name = string("per_layer_slice_1_end_0"), val = tensor([1, 1, 3328])]; tensor per_layer_slice_1_end_mask_0 = const()[name = string("per_layer_slice_1_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_1_cast_fp16 = slice_by_index(begin = per_layer_slice_1_begin_0, end = per_layer_slice_1_end_0, end_mask = per_layer_slice_1_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_1_cast_fp16")]; tensor var_1224 = const()[name = string("op_1224"), val = tensor([0, 2, 1])]; tensor input_25_axes_0 = const()[name = string("input_25_axes_0"), val = tensor([2])]; tensor var_1225 = transpose(perm = var_1224, x = hidden_states_5_cast_fp16)[name = string("transpose_200")]; tensor input_25 = expand_dims(axes = input_25_axes_0, x = var_1225)[name = string("input_25")]; string gated_1_pad_type_0 = const()[name = string("gated_1_pad_type_0"), val = string("valid")]; tensor gated_1_strides_0 = const()[name = string("gated_1_strides_0"), val = tensor([1, 1])]; tensor gated_1_pad_0 = const()[name = string("gated_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_1_dilations_0 = const()[name = string("gated_1_dilations_0"), val = tensor([1, 1])]; int32 gated_1_groups_0 = const()[name = string("gated_1_groups_0"), val = int32(1)]; tensor gated_1 = conv(dilations = gated_1_dilations_0, groups = gated_1_groups_0, pad = gated_1_pad_0, pad_type = gated_1_pad_type_0, strides = gated_1_strides_0, weight = layers_0_per_layer_input_gate_weight_palettized, x = input_25)[name = string("gated_1")]; string gated_3_mode_0 = const()[name = string("gated_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_3 = gelu(mode = gated_3_mode_0, x = gated_1)[name = string("gated_3")]; tensor var_1244 = const()[name = string("op_1244"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_1_axes_0 = const()[name = string("per_layer_slice_conv_1_axes_0"), val = tensor([2])]; tensor var_1245_cast_fp16 = transpose(perm = var_1244, x = per_layer_slice_1_cast_fp16)[name = string("transpose_199")]; tensor per_layer_slice_conv_1_cast_fp16 = expand_dims(axes = per_layer_slice_conv_1_axes_0, x = var_1245_cast_fp16)[name = string("per_layer_slice_conv_1_cast_fp16")]; tensor input_27_cast_fp16 = mul(x = gated_3, y = per_layer_slice_conv_1_cast_fp16)[name = string("input_27_cast_fp16")]; string gated_5_pad_type_0 = const()[name = string("gated_5_pad_type_0"), val = string("valid")]; tensor gated_5_strides_0 = const()[name = string("gated_5_strides_0"), val = tensor([1, 1])]; tensor gated_5_pad_0 = const()[name = string("gated_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_5_dilations_0 = const()[name = string("gated_5_dilations_0"), val = tensor([1, 1])]; int32 gated_5_groups_0 = const()[name = string("gated_5_groups_0"), val = int32(1)]; tensor layers_0_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533891008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534218752))))[name = string("layers_0_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_5_cast_fp16 = conv(dilations = gated_5_dilations_0, groups = gated_5_groups_0, pad = gated_5_pad_0, pad_type = gated_5_pad_type_0, strides = gated_5_strides_0, weight = layers_0_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_27_cast_fp16)[name = string("gated_5_cast_fp16")]; tensor var_1261_axes_0 = const()[name = string("op_1261_axes_0"), val = tensor([2])]; tensor var_1261_cast_fp16 = squeeze(axes = var_1261_axes_0, x = gated_5_cast_fp16)[name = string("op_1261_cast_fp16")]; tensor var_1265 = const()[name = string("op_1265"), val = tensor([0, 2, 1])]; int32 var_1271 = const()[name = string("op_1271"), val = int32(-1)]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_17_cast_fp16 = transpose(perm = var_1265, x = var_1261_cast_fp16)[name = string("transpose_198")]; tensor var_1273_cast_fp16 = mul(x = x_17_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1273_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_1271, interleave = input_29_interleave_0, values = (x_17_cast_fp16, var_1273_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_1268_to_fp16 = const()[name = string("op_1268_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1268_to_fp16, x = input_29_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor var_1278_split_sizes_0 = const()[name = string("op_1278_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1278_axis_0 = const()[name = string("op_1278_axis_0"), val = int32(-1)]; tensor var_1278_cast_fp16_0, tensor var_1278_cast_fp16_1 = split(axis = var_1278_axis_0, split_sizes = var_1278_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_1278_cast_fp16")]; tensor layers_0_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534221376)))]; tensor hidden_states_9_cast_fp16 = mul(x = var_1278_cast_fp16_0, y = layers_0_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_9_cast_fp16")]; tensor hidden_states_11_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_9_cast_fp16)[name = string("hidden_states_11_cast_fp16")]; tensor const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = tensor([0x1.7ep-1])]; tensor x_19_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_11_promoted_to_fp16)[name = string("x_19_cast_fp16")]; tensor var_1290_axes_0 = const()[name = string("op_1290_axes_0"), val = tensor([0])]; tensor var_1290_cast_fp16 = squeeze(axes = var_1290_axes_0, x = K_sliding_out_1_cast_fp16)[name = string("op_1290_cast_fp16")]; tensor var_1292_axes_0 = const()[name = string("op_1292_axes_0"), val = tensor([0])]; tensor var_1292_cast_fp16 = squeeze(axes = var_1292_axes_0, x = V_sliding_out_1_cast_fp16)[name = string("op_1292_cast_fp16")]; tensor var_1295_begin_0 = const()[name = string("op_1295_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1295_end_0 = const()[name = string("op_1295_end_0"), val = tensor([2, 2, 512, 512])]; tensor var_1295_end_mask_0 = const()[name = string("op_1295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1295_squeeze_mask_0 = const()[name = string("op_1295_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_1295_cast_fp16 = slice_by_index(begin = var_1295_begin_0, end = var_1295_end_0, end_mask = var_1295_end_mask_0, squeeze_mask = var_1295_squeeze_mask_0, x = K_sliding_in)[name = string("op_1295_cast_fp16")]; tensor K_sliding_slot_3_axes_0 = const()[name = string("K_sliding_slot_3_axes_0"), val = tensor([0])]; tensor K_sliding_slot_3_cast_fp16 = expand_dims(axes = K_sliding_slot_3_axes_0, x = var_1295_cast_fp16)[name = string("K_sliding_slot_3_cast_fp16")]; tensor var_1300_begin_0 = const()[name = string("op_1300_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1300_end_0 = const()[name = string("op_1300_end_0"), val = tensor([2, 2, 512, 512])]; tensor var_1300_end_mask_0 = const()[name = string("op_1300_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1300_squeeze_mask_0 = const()[name = string("op_1300_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_1300_cast_fp16 = slice_by_index(begin = var_1300_begin_0, end = var_1300_end_0, end_mask = var_1300_end_mask_0, squeeze_mask = var_1300_squeeze_mask_0, x = V_sliding_in)[name = string("op_1300_cast_fp16")]; tensor V_sliding_slot_3_axes_0 = const()[name = string("V_sliding_slot_3_axes_0"), val = tensor([0])]; tensor V_sliding_slot_3_cast_fp16 = expand_dims(axes = V_sliding_slot_3_axes_0, x = var_1300_cast_fp16)[name = string("V_sliding_slot_3_cast_fp16")]; int32 var_1307 = const()[name = string("op_1307"), val = int32(-1)]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1309_cast_fp16 = mul(x = x_19_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_1309_cast_fp16")]; bool input_31_interleave_0 = const()[name = string("input_31_interleave_0"), val = bool(false)]; tensor input_31_cast_fp16 = concat(axis = var_1307, interleave = input_31_interleave_0, values = (x_19_cast_fp16, var_1309_cast_fp16))[name = string("input_31_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_1304_to_fp16 = const()[name = string("op_1304_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1304_to_fp16, x = input_31_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_1314_split_sizes_0 = const()[name = string("op_1314_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1314_axis_0 = const()[name = string("op_1314_axis_0"), val = int32(-1)]; tensor var_1314_cast_fp16_0, tensor var_1314_cast_fp16_1 = split(axis = var_1314_axis_0, split_sizes = var_1314_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_1314_cast_fp16")]; tensor layers_1_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534226560)))]; tensor h_7_cast_fp16 = mul(x = var_1314_cast_fp16_0, y = layers_1_input_layernorm_weight_promoted_to_fp16)[name = string("h_7_cast_fp16")]; tensor var_1320 = const()[name = string("op_1320"), val = tensor([0, 2, 1])]; tensor var_1323_axes_0 = const()[name = string("op_1323_axes_0"), val = tensor([2])]; tensor var_1321_cast_fp16 = transpose(perm = var_1320, x = h_7_cast_fp16)[name = string("transpose_197")]; tensor var_1323_cast_fp16 = expand_dims(axes = var_1323_axes_0, x = var_1321_cast_fp16)[name = string("op_1323_cast_fp16")]; string var_1339_pad_type_0 = const()[name = string("op_1339_pad_type_0"), val = string("valid")]; tensor var_1339_strides_0 = const()[name = string("op_1339_strides_0"), val = tensor([1, 1])]; tensor var_1339_pad_0 = const()[name = string("op_1339_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1339_dilations_0 = const()[name = string("op_1339_dilations_0"), val = tensor([1, 1])]; int32 var_1339_groups_0 = const()[name = string("op_1339_groups_0"), val = int32(1)]; tensor var_1339 = conv(dilations = var_1339_dilations_0, groups = var_1339_groups_0, pad = var_1339_pad_0, pad_type = var_1339_pad_type_0, strides = var_1339_strides_0, weight = layers_1_self_attn_q_proj_weight_palettized, x = var_1323_cast_fp16)[name = string("op_1339")]; tensor var_1344 = const()[name = string("op_1344"), val = tensor([1, 8, 256, 1])]; tensor var_1345 = reshape(shape = var_1344, x = var_1339)[name = string("op_1345")]; tensor var_1350 = const()[name = string("op_1350"), val = tensor([0, 1, 3, 2])]; tensor var_1360 = const()[name = string("op_1360"), val = tensor([1, 8, 256])]; tensor var_1351 = transpose(perm = var_1350, x = var_1345)[name = string("transpose_196")]; tensor x_21 = reshape(shape = var_1360, x = var_1351)[name = string("x_21")]; int32 var_1366 = const()[name = string("op_1366"), val = int32(-1)]; fp16 const_13_promoted = const()[name = string("const_13_promoted"), val = fp16(-0x1p+0)]; tensor var_1368 = mul(x = x_21, y = const_13_promoted)[name = string("op_1368")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35 = concat(axis = var_1366, interleave = input_35_interleave_0, values = (x_21, var_1368))[name = string("input_35")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_1363_to_fp16 = const()[name = string("op_1363_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1363_to_fp16, x = input_35)[name = string("normed_33_cast_fp16")]; tensor var_1373_split_sizes_0 = const()[name = string("op_1373_split_sizes_0"), val = tensor([256, 256])]; int32 var_1373_axis_0 = const()[name = string("op_1373_axis_0"), val = int32(-1)]; tensor var_1373_0, tensor var_1373_1 = split(axis = var_1373_axis_0, split_sizes = var_1373_split_sizes_0, x = normed_33_cast_fp16)[name = string("op_1373")]; tensor var_1375 = mul(x = var_1373_0, y = layers_1_self_attn_q_norm_weight)[name = string("op_1375")]; tensor var_1380 = const()[name = string("op_1380"), val = tensor([1, 8, 1, 256])]; tensor q_11 = reshape(shape = var_1380, x = var_1375)[name = string("q_11")]; tensor var_1382_cast_fp16 = mul(x = q_11, y = cos_s)[name = string("op_1382_cast_fp16")]; tensor var_1383_split_sizes_0 = const()[name = string("op_1383_split_sizes_0"), val = tensor([128, 128])]; int32 var_1383_axis_0 = const()[name = string("op_1383_axis_0"), val = int32(-1)]; tensor var_1383_0, tensor var_1383_1 = split(axis = var_1383_axis_0, split_sizes = var_1383_split_sizes_0, x = q_11)[name = string("op_1383")]; fp16 const_14_promoted = const()[name = string("const_14_promoted"), val = fp16(-0x1p+0)]; tensor var_1385 = mul(x = var_1383_1, y = const_14_promoted)[name = string("op_1385")]; int32 var_1387 = const()[name = string("op_1387"), val = int32(-1)]; bool var_1388_interleave_0 = const()[name = string("op_1388_interleave_0"), val = bool(false)]; tensor var_1388 = concat(axis = var_1387, interleave = var_1388_interleave_0, values = (var_1385, var_1383_0))[name = string("op_1388")]; tensor var_1389_cast_fp16 = mul(x = var_1388, y = sin_s)[name = string("op_1389_cast_fp16")]; tensor q_15_cast_fp16 = add(x = var_1382_cast_fp16, y = var_1389_cast_fp16)[name = string("q_15_cast_fp16")]; string var_1402_pad_type_0 = const()[name = string("op_1402_pad_type_0"), val = string("valid")]; tensor var_1402_strides_0 = const()[name = string("op_1402_strides_0"), val = tensor([1, 1])]; tensor var_1402_pad_0 = const()[name = string("op_1402_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1402_dilations_0 = const()[name = string("op_1402_dilations_0"), val = tensor([1, 1])]; int32 var_1402_groups_0 = const()[name = string("op_1402_groups_0"), val = int32(1)]; tensor var_1402 = conv(dilations = var_1402_dilations_0, groups = var_1402_groups_0, pad = var_1402_pad_0, pad_type = var_1402_pad_type_0, strides = var_1402_strides_0, weight = layers_1_self_attn_k_proj_weight_palettized, x = var_1323_cast_fp16)[name = string("op_1402")]; tensor var_1407 = const()[name = string("op_1407"), val = tensor([1, 2, 256, 1])]; tensor var_1408 = reshape(shape = var_1407, x = var_1402)[name = string("op_1408")]; tensor var_1413 = const()[name = string("op_1413"), val = tensor([0, 1, 3, 2])]; string var_1430_pad_type_0 = const()[name = string("op_1430_pad_type_0"), val = string("valid")]; tensor var_1430_strides_0 = const()[name = string("op_1430_strides_0"), val = tensor([1, 1])]; tensor var_1430_pad_0 = const()[name = string("op_1430_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1430_dilations_0 = const()[name = string("op_1430_dilations_0"), val = tensor([1, 1])]; int32 var_1430_groups_0 = const()[name = string("op_1430_groups_0"), val = int32(1)]; tensor var_1430 = conv(dilations = var_1430_dilations_0, groups = var_1430_groups_0, pad = var_1430_pad_0, pad_type = var_1430_pad_type_0, strides = var_1430_strides_0, weight = layers_1_self_attn_v_proj_weight_palettized, x = var_1323_cast_fp16)[name = string("op_1430")]; tensor var_1435 = const()[name = string("op_1435"), val = tensor([1, 2, 256, 1])]; tensor var_1436 = reshape(shape = var_1435, x = var_1430)[name = string("op_1436")]; tensor var_1441 = const()[name = string("op_1441"), val = tensor([0, 1, 3, 2])]; tensor var_1451 = const()[name = string("op_1451"), val = tensor([1, 2, 256])]; tensor var_1414 = transpose(perm = var_1413, x = var_1408)[name = string("transpose_195")]; tensor x_23 = reshape(shape = var_1451, x = var_1414)[name = string("x_23")]; int32 var_1457 = const()[name = string("op_1457"), val = int32(-1)]; fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; tensor var_1459 = mul(x = x_23, y = const_15_promoted)[name = string("op_1459")]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37 = concat(axis = var_1457, interleave = input_37_interleave_0, values = (x_23, var_1459))[name = string("input_37")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_1454_to_fp16 = const()[name = string("op_1454_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1454_to_fp16, x = input_37)[name = string("normed_37_cast_fp16")]; tensor var_1464_split_sizes_0 = const()[name = string("op_1464_split_sizes_0"), val = tensor([256, 256])]; int32 var_1464_axis_0 = const()[name = string("op_1464_axis_0"), val = int32(-1)]; tensor var_1464_0, tensor var_1464_1 = split(axis = var_1464_axis_0, split_sizes = var_1464_split_sizes_0, x = normed_37_cast_fp16)[name = string("op_1464")]; tensor var_1466 = mul(x = var_1464_0, y = layers_1_self_attn_k_norm_weight)[name = string("op_1466")]; tensor var_1471 = const()[name = string("op_1471"), val = tensor([1, 2, 1, 256])]; tensor q_13 = reshape(shape = var_1471, x = var_1466)[name = string("q_13")]; fp16 var_1473_promoted = const()[name = string("op_1473_promoted"), val = fp16(0x1p+1)]; tensor var_1442 = transpose(perm = var_1441, x = var_1436)[name = string("transpose_194")]; tensor var_1474 = pow(x = var_1442, y = var_1473_promoted)[name = string("op_1474")]; tensor var_1479_axes_0 = const()[name = string("op_1479_axes_0"), val = tensor([-1])]; bool var_1479_keep_dims_0 = const()[name = string("op_1479_keep_dims_0"), val = bool(true)]; tensor var_1479 = reduce_mean(axes = var_1479_axes_0, keep_dims = var_1479_keep_dims_0, x = var_1474)[name = string("op_1479")]; fp16 var_1481_to_fp16 = const()[name = string("op_1481_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_3_cast_fp16 = add(x = var_1479, y = var_1481_to_fp16)[name = string("mean_sq_3_cast_fp16")]; fp32 var_1483_epsilon_0 = const()[name = string("op_1483_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_1483_cast_fp16 = rsqrt(epsilon = var_1483_epsilon_0, x = mean_sq_3_cast_fp16)[name = string("op_1483_cast_fp16")]; tensor input_41_cast_fp16 = mul(x = var_1442, y = var_1483_cast_fp16)[name = string("input_41_cast_fp16")]; tensor var_1485_cast_fp16 = mul(x = q_13, y = cos_s)[name = string("op_1485_cast_fp16")]; tensor var_1486_split_sizes_0 = const()[name = string("op_1486_split_sizes_0"), val = tensor([128, 128])]; int32 var_1486_axis_0 = const()[name = string("op_1486_axis_0"), val = int32(-1)]; tensor var_1486_0, tensor var_1486_1 = split(axis = var_1486_axis_0, split_sizes = var_1486_split_sizes_0, x = q_13)[name = string("op_1486")]; fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; tensor var_1488 = mul(x = var_1486_1, y = const_16_promoted)[name = string("op_1488")]; int32 var_1490 = const()[name = string("op_1490"), val = int32(-1)]; bool var_1491_interleave_0 = const()[name = string("op_1491_interleave_0"), val = bool(false)]; tensor var_1491 = concat(axis = var_1490, interleave = var_1491_interleave_0, values = (var_1488, var_1486_0))[name = string("op_1491")]; tensor var_1492_cast_fp16 = mul(x = var_1491, y = sin_s)[name = string("op_1492_cast_fp16")]; tensor input_39_cast_fp16 = add(x = var_1485_cast_fp16, y = var_1492_cast_fp16)[name = string("input_39_cast_fp16")]; tensor k_padded_3_pad_0 = const()[name = string("k_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_3_mode_0 = const()[name = string("k_padded_3_mode_0"), val = string("constant")]; fp16 const_17_to_fp16 = const()[name = string("const_17_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_3_cast_fp16 = pad(constant_val = const_17_to_fp16, mode = k_padded_3_mode_0, pad = k_padded_3_pad_0, x = input_39_cast_fp16)[name = string("k_padded_3_cast_fp16")]; tensor v_padded_3_pad_0 = const()[name = string("v_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_3_mode_0 = const()[name = string("v_padded_3_mode_0"), val = string("constant")]; fp16 const_18_to_fp16 = const()[name = string("const_18_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_3_cast_fp16 = pad(constant_val = const_18_to_fp16, mode = v_padded_3_mode_0, pad = v_padded_3_pad_0, x = input_41_cast_fp16)[name = string("v_padded_3_cast_fp16")]; tensor var_1521_begin_0 = const()[name = string("op_1521_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_1521_end_0 = const()[name = string("op_1521_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1521_end_mask_0 = const()[name = string("op_1521_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1521_cast_fp16 = slice_by_index(begin = var_1521_begin_0, end = var_1521_end_0, end_mask = var_1521_end_mask_0, x = K_sliding_slot_3_cast_fp16)[name = string("op_1521_cast_fp16")]; int32 var_1528 = const()[name = string("op_1528"), val = int32(2)]; bool K_sliding_out_3_interleave_0 = const()[name = string("K_sliding_out_3_interleave_0"), val = bool(false)]; tensor K_sliding_out_3_cast_fp16 = concat(axis = var_1528, interleave = K_sliding_out_3_interleave_0, values = (var_1521_cast_fp16, k_padded_3_cast_fp16))[name = string("K_sliding_out_3_cast_fp16")]; tensor var_1544_begin_0 = const()[name = string("op_1544_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_1544_end_0 = const()[name = string("op_1544_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1544_end_mask_0 = const()[name = string("op_1544_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1544_cast_fp16 = slice_by_index(begin = var_1544_begin_0, end = var_1544_end_0, end_mask = var_1544_end_mask_0, x = V_sliding_slot_3_cast_fp16)[name = string("op_1544_cast_fp16")]; int32 var_1551 = const()[name = string("op_1551"), val = int32(2)]; bool V_sliding_out_3_interleave_0 = const()[name = string("V_sliding_out_3_interleave_0"), val = bool(false)]; tensor V_sliding_out_3_cast_fp16 = concat(axis = var_1551, interleave = V_sliding_out_3_interleave_0, values = (var_1544_cast_fp16, v_padded_3_cast_fp16))[name = string("V_sliding_out_3_cast_fp16")]; tensor K_for_attn_3_begin_0 = const()[name = string("K_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_3_end_0 = const()[name = string("K_for_attn_3_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_3_end_mask_0 = const()[name = string("K_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_3_cast_fp16 = slice_by_index(begin = K_for_attn_3_begin_0, end = K_for_attn_3_end_0, end_mask = K_for_attn_3_end_mask_0, x = K_sliding_out_3_cast_fp16)[name = string("K_for_attn_3_cast_fp16")]; tensor V_for_attn_3_begin_0 = const()[name = string("V_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_3_end_0 = const()[name = string("V_for_attn_3_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_3_end_mask_0 = const()[name = string("V_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_3_cast_fp16 = slice_by_index(begin = V_for_attn_3_begin_0, end = V_for_attn_3_end_0, end_mask = V_for_attn_3_end_mask_0, x = V_sliding_out_3_cast_fp16)[name = string("V_for_attn_3_cast_fp16")]; tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = K_for_attn_3_cast_fp16)[name = string("transpose_193")]; tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; tensor concat_4 = const()[name = string("concat_4"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_4, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_5 = const()[name = string("concat_5"), val = tensor([-1, 1, 512, 256])]; tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_192")]; tensor reshape_5_cast_fp16 = reshape(shape = concat_5, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor transpose_49_perm_0 = const()[name = string("transpose_49_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = V_for_attn_3_cast_fp16)[name = string("transpose_191")]; tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; tensor concat_6 = const()[name = string("concat_6"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_6, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_7 = const()[name = string("concat_7"), val = tensor([-1, 1, 512, 256])]; tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_190")]; tensor reshape_7_cast_fp16 = reshape(shape = concat_7, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; tensor V_expanded_3_perm_0 = const()[name = string("V_expanded_3_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_5_transpose_x_0 = const()[name = string("attn_weights_5_transpose_x_0"), val = bool(false)]; bool attn_weights_5_transpose_y_0 = const()[name = string("attn_weights_5_transpose_y_0"), val = bool(false)]; tensor transpose_49_cast_fp16 = transpose(perm = transpose_49_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_189")]; tensor attn_weights_5_cast_fp16 = matmul(transpose_x = attn_weights_5_transpose_x_0, transpose_y = attn_weights_5_transpose_y_0, x = q_15_cast_fp16, y = transpose_49_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; tensor x_27_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask_sliding)[name = string("x_27_cast_fp16")]; tensor reduce_max_1_axes_0 = const()[name = string("reduce_max_1_axes_0"), val = tensor([-1])]; bool reduce_max_1_keep_dims_0 = const()[name = string("reduce_max_1_keep_dims_0"), val = bool(true)]; tensor reduce_max_1 = reduce_max(axes = reduce_max_1_axes_0, keep_dims = reduce_max_1_keep_dims_0, x = x_27_cast_fp16)[name = string("reduce_max_1")]; tensor var_1592 = sub(x = x_27_cast_fp16, y = reduce_max_1)[name = string("op_1592")]; tensor var_1598 = exp(x = var_1592)[name = string("op_1598")]; tensor var_1608_axes_0 = const()[name = string("op_1608_axes_0"), val = tensor([-1])]; bool var_1608_keep_dims_0 = const()[name = string("op_1608_keep_dims_0"), val = bool(true)]; tensor var_1608 = reduce_sum(axes = var_1608_axes_0, keep_dims = var_1608_keep_dims_0, x = var_1598)[name = string("op_1608")]; tensor var_1614_cast_fp16 = real_div(x = var_1598, y = var_1608)[name = string("op_1614_cast_fp16")]; bool attn_output_7_transpose_x_0 = const()[name = string("attn_output_7_transpose_x_0"), val = bool(false)]; bool attn_output_7_transpose_y_0 = const()[name = string("attn_output_7_transpose_y_0"), val = bool(false)]; tensor V_expanded_3_cast_fp16 = transpose(perm = V_expanded_3_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_188")]; tensor attn_output_7_cast_fp16 = matmul(transpose_x = attn_output_7_transpose_x_0, transpose_y = attn_output_7_transpose_y_0, x = var_1614_cast_fp16, y = V_expanded_3_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor var_1625 = const()[name = string("op_1625"), val = tensor([0, 2, 1, 3])]; tensor var_1632 = const()[name = string("op_1632"), val = tensor([1, 1, -1])]; tensor var_1626_cast_fp16 = transpose(perm = var_1625, x = attn_output_7_cast_fp16)[name = string("transpose_187")]; tensor attn_output_9_cast_fp16 = reshape(shape = var_1632, x = var_1626_cast_fp16)[name = string("attn_output_9_cast_fp16")]; tensor var_1637 = const()[name = string("op_1637"), val = tensor([0, 2, 1])]; string var_1653_pad_type_0 = const()[name = string("op_1653_pad_type_0"), val = string("valid")]; int32 var_1653_groups_0 = const()[name = string("op_1653_groups_0"), val = int32(1)]; tensor var_1653_strides_0 = const()[name = string("op_1653_strides_0"), val = tensor([1])]; tensor var_1653_pad_0 = const()[name = string("op_1653_pad_0"), val = tensor([0, 0])]; tensor var_1653_dilations_0 = const()[name = string("op_1653_dilations_0"), val = tensor([1])]; tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534231744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536853248))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_1638_cast_fp16 = transpose(perm = var_1637, x = attn_output_9_cast_fp16)[name = string("transpose_186")]; tensor var_1653_cast_fp16 = conv(dilations = var_1653_dilations_0, groups = var_1653_groups_0, pad = var_1653_pad_0, pad_type = var_1653_pad_type_0, strides = var_1653_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_1638_cast_fp16)[name = string("op_1653_cast_fp16")]; tensor var_1657 = const()[name = string("op_1657"), val = tensor([0, 2, 1])]; int32 var_1663 = const()[name = string("op_1663"), val = int32(-1)]; fp16 const_19_promoted_to_fp16 = const()[name = string("const_19_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_31_cast_fp16 = transpose(perm = var_1657, x = var_1653_cast_fp16)[name = string("transpose_185")]; tensor var_1665_cast_fp16 = mul(x = x_31_cast_fp16, y = const_19_promoted_to_fp16)[name = string("op_1665_cast_fp16")]; bool input_45_interleave_0 = const()[name = string("input_45_interleave_0"), val = bool(false)]; tensor input_45_cast_fp16 = concat(axis = var_1663, interleave = input_45_interleave_0, values = (x_31_cast_fp16, var_1665_cast_fp16))[name = string("input_45_cast_fp16")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_1660_to_fp16 = const()[name = string("op_1660_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1660_to_fp16, x = input_45_cast_fp16)[name = string("normed_41_cast_fp16")]; tensor var_1670_split_sizes_0 = const()[name = string("op_1670_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1670_axis_0 = const()[name = string("op_1670_axis_0"), val = int32(-1)]; tensor var_1670_cast_fp16_0, tensor var_1670_cast_fp16_1 = split(axis = var_1670_axis_0, split_sizes = var_1670_split_sizes_0, x = normed_41_cast_fp16)[name = string("op_1670_cast_fp16")]; tensor layers_1_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536855872)))]; tensor attn_output_11_cast_fp16 = mul(x = var_1670_cast_fp16_0, y = layers_1_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_11_cast_fp16")]; tensor x_33_cast_fp16 = add(x = x_19_cast_fp16, y = attn_output_11_cast_fp16)[name = string("x_33_cast_fp16")]; int32 var_1679 = const()[name = string("op_1679"), val = int32(-1)]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1681_cast_fp16 = mul(x = x_33_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1681_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_1679, interleave = input_47_interleave_0, values = (x_33_cast_fp16, var_1681_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_1676_to_fp16 = const()[name = string("op_1676_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_1676_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor var_1686_split_sizes_0 = const()[name = string("op_1686_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1686_axis_0 = const()[name = string("op_1686_axis_0"), val = int32(-1)]; tensor var_1686_cast_fp16_0, tensor var_1686_cast_fp16_1 = split(axis = var_1686_axis_0, split_sizes = var_1686_split_sizes_0, x = normed_45_cast_fp16)[name = string("op_1686_cast_fp16")]; tensor layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536861056)))]; tensor h_9_cast_fp16 = mul(x = var_1686_cast_fp16_0, y = layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_9_cast_fp16")]; tensor var_1697 = const()[name = string("op_1697"), val = tensor([0, 2, 1])]; tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; tensor var_1698 = transpose(perm = var_1697, x = h_9_cast_fp16)[name = string("transpose_184")]; tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_1698)[name = string("input_49")]; string gate_5_pad_type_0 = const()[name = string("gate_5_pad_type_0"), val = string("valid")]; tensor gate_5_strides_0 = const()[name = string("gate_5_strides_0"), val = tensor([1, 1])]; tensor gate_5_pad_0 = const()[name = string("gate_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_5_dilations_0 = const()[name = string("gate_5_dilations_0"), val = tensor([1, 1])]; int32 gate_5_groups_0 = const()[name = string("gate_5_groups_0"), val = int32(1)]; tensor gate_5 = conv(dilations = gate_5_dilations_0, groups = gate_5_groups_0, pad = gate_5_pad_0, pad_type = gate_5_pad_type_0, strides = gate_5_strides_0, weight = layers_1_mlp_gate_proj_weight_palettized, x = input_49)[name = string("gate_5")]; string up_3_pad_type_0 = const()[name = string("up_3_pad_type_0"), val = string("valid")]; tensor up_3_strides_0 = const()[name = string("up_3_strides_0"), val = tensor([1, 1])]; tensor up_3_pad_0 = const()[name = string("up_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_3_dilations_0 = const()[name = string("up_3_dilations_0"), val = tensor([1, 1])]; int32 up_3_groups_0 = const()[name = string("up_3_groups_0"), val = int32(1)]; tensor up_3 = conv(dilations = up_3_dilations_0, groups = up_3_groups_0, pad = up_3_pad_0, pad_type = up_3_pad_type_0, strides = up_3_strides_0, weight = layers_1_mlp_up_proj_weight_palettized, x = input_49)[name = string("up_3")]; string gate_7_mode_0 = const()[name = string("gate_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_7 = gelu(mode = gate_7_mode_0, x = gate_5)[name = string("gate_7")]; tensor input_51 = mul(x = gate_7, y = up_3)[name = string("input_51")]; string mlp_out_3_pad_type_0 = const()[name = string("mlp_out_3_pad_type_0"), val = string("valid")]; tensor mlp_out_3_strides_0 = const()[name = string("mlp_out_3_strides_0"), val = tensor([1, 1])]; tensor mlp_out_3_pad_0 = const()[name = string("mlp_out_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_3_dilations_0 = const()[name = string("mlp_out_3_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_3_groups_0 = const()[name = string("mlp_out_3_groups_0"), val = int32(1)]; tensor mlp_out_3 = conv(dilations = mlp_out_3_dilations_0, groups = mlp_out_3_groups_0, pad = mlp_out_3_pad_0, pad_type = mlp_out_3_pad_type_0, strides = mlp_out_3_strides_0, weight = layers_1_mlp_down_proj_weight_palettized, x = input_51)[name = string("mlp_out_3")]; tensor var_1738_axes_0 = const()[name = string("op_1738_axes_0"), val = tensor([2])]; tensor var_1738 = squeeze(axes = var_1738_axes_0, x = mlp_out_3)[name = string("op_1738")]; tensor var_1742 = const()[name = string("op_1742"), val = tensor([0, 2, 1])]; int32 var_1748 = const()[name = string("op_1748"), val = int32(-1)]; fp16 const_21_promoted = const()[name = string("const_21_promoted"), val = fp16(-0x1p+0)]; tensor x_35 = transpose(perm = var_1742, x = var_1738)[name = string("transpose_183")]; tensor var_1750 = mul(x = x_35, y = const_21_promoted)[name = string("op_1750")]; bool input_53_interleave_0 = const()[name = string("input_53_interleave_0"), val = bool(false)]; tensor input_53 = concat(axis = var_1748, interleave = input_53_interleave_0, values = (x_35, var_1750))[name = string("input_53")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_1745_to_fp16 = const()[name = string("op_1745_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_1745_to_fp16, x = input_53)[name = string("normed_49_cast_fp16")]; tensor var_1755_split_sizes_0 = const()[name = string("op_1755_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1755_axis_0 = const()[name = string("op_1755_axis_0"), val = int32(-1)]; tensor var_1755_0, tensor var_1755_1 = split(axis = var_1755_axis_0, split_sizes = var_1755_split_sizes_0, x = normed_49_cast_fp16)[name = string("op_1755")]; tensor hidden_states_13 = mul(x = var_1755_0, y = layers_1_post_feedforward_layernorm_weight)[name = string("hidden_states_13")]; tensor hidden_states_15_cast_fp16 = add(x = x_33_cast_fp16, y = hidden_states_13)[name = string("hidden_states_15_cast_fp16")]; tensor per_layer_slice_3_begin_0 = const()[name = string("per_layer_slice_3_begin_0"), val = tensor([0, 0, 3328])]; tensor per_layer_slice_3_end_0 = const()[name = string("per_layer_slice_3_end_0"), val = tensor([1, 1, 3584])]; tensor per_layer_slice_3_end_mask_0 = const()[name = string("per_layer_slice_3_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_3_cast_fp16 = slice_by_index(begin = per_layer_slice_3_begin_0, end = per_layer_slice_3_end_0, end_mask = per_layer_slice_3_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_3_cast_fp16")]; tensor var_1783 = const()[name = string("op_1783"), val = tensor([0, 2, 1])]; tensor input_55_axes_0 = const()[name = string("input_55_axes_0"), val = tensor([2])]; tensor var_1784 = transpose(perm = var_1783, x = hidden_states_15_cast_fp16)[name = string("transpose_182")]; tensor input_55 = expand_dims(axes = input_55_axes_0, x = var_1784)[name = string("input_55")]; string gated_7_pad_type_0 = const()[name = string("gated_7_pad_type_0"), val = string("valid")]; tensor gated_7_strides_0 = const()[name = string("gated_7_strides_0"), val = tensor([1, 1])]; tensor gated_7_pad_0 = const()[name = string("gated_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_7_dilations_0 = const()[name = string("gated_7_dilations_0"), val = tensor([1, 1])]; int32 gated_7_groups_0 = const()[name = string("gated_7_groups_0"), val = int32(1)]; tensor gated_7 = conv(dilations = gated_7_dilations_0, groups = gated_7_groups_0, pad = gated_7_pad_0, pad_type = gated_7_pad_type_0, strides = gated_7_strides_0, weight = layers_1_per_layer_input_gate_weight_palettized, x = input_55)[name = string("gated_7")]; string gated_9_mode_0 = const()[name = string("gated_9_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_9 = gelu(mode = gated_9_mode_0, x = gated_7)[name = string("gated_9")]; tensor var_1803 = const()[name = string("op_1803"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_3_axes_0 = const()[name = string("per_layer_slice_conv_3_axes_0"), val = tensor([2])]; tensor var_1804_cast_fp16 = transpose(perm = var_1803, x = per_layer_slice_3_cast_fp16)[name = string("transpose_181")]; tensor per_layer_slice_conv_3_cast_fp16 = expand_dims(axes = per_layer_slice_conv_3_axes_0, x = var_1804_cast_fp16)[name = string("per_layer_slice_conv_3_cast_fp16")]; tensor input_57_cast_fp16 = mul(x = gated_9, y = per_layer_slice_conv_3_cast_fp16)[name = string("input_57_cast_fp16")]; string gated_11_pad_type_0 = const()[name = string("gated_11_pad_type_0"), val = string("valid")]; tensor gated_11_strides_0 = const()[name = string("gated_11_strides_0"), val = tensor([1, 1])]; tensor gated_11_pad_0 = const()[name = string("gated_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_11_dilations_0 = const()[name = string("gated_11_dilations_0"), val = tensor([1, 1])]; int32 gated_11_groups_0 = const()[name = string("gated_11_groups_0"), val = int32(1)]; tensor layers_1_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536866240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537193984))))[name = string("layers_1_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_11_cast_fp16 = conv(dilations = gated_11_dilations_0, groups = gated_11_groups_0, pad = gated_11_pad_0, pad_type = gated_11_pad_type_0, strides = gated_11_strides_0, weight = layers_1_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_57_cast_fp16)[name = string("gated_11_cast_fp16")]; tensor var_1820_axes_0 = const()[name = string("op_1820_axes_0"), val = tensor([2])]; tensor var_1820_cast_fp16 = squeeze(axes = var_1820_axes_0, x = gated_11_cast_fp16)[name = string("op_1820_cast_fp16")]; tensor var_1824 = const()[name = string("op_1824"), val = tensor([0, 2, 1])]; int32 var_1830 = const()[name = string("op_1830"), val = int32(-1)]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_37_cast_fp16 = transpose(perm = var_1824, x = var_1820_cast_fp16)[name = string("transpose_180")]; tensor var_1832_cast_fp16 = mul(x = x_37_cast_fp16, y = const_22_promoted_to_fp16)[name = string("op_1832_cast_fp16")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59_cast_fp16 = concat(axis = var_1830, interleave = input_59_interleave_0, values = (x_37_cast_fp16, var_1832_cast_fp16))[name = string("input_59_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_1827_to_fp16 = const()[name = string("op_1827_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_1827_to_fp16, x = input_59_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor var_1837_split_sizes_0 = const()[name = string("op_1837_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1837_axis_0 = const()[name = string("op_1837_axis_0"), val = int32(-1)]; tensor var_1837_cast_fp16_0, tensor var_1837_cast_fp16_1 = split(axis = var_1837_axis_0, split_sizes = var_1837_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_1837_cast_fp16")]; tensor layers_1_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537196608)))]; tensor hidden_states_19_cast_fp16 = mul(x = var_1837_cast_fp16_0, y = layers_1_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_19_cast_fp16")]; tensor hidden_states_21_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_19_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; tensor const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = tensor([0x1.6cp-1])]; tensor x_39_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_23_promoted_to_fp16)[name = string("x_39_cast_fp16")]; tensor var_1849_axes_0 = const()[name = string("op_1849_axes_0"), val = tensor([0])]; tensor var_1849_cast_fp16 = squeeze(axes = var_1849_axes_0, x = K_sliding_out_3_cast_fp16)[name = string("op_1849_cast_fp16")]; tensor var_1851_axes_0 = const()[name = string("op_1851_axes_0"), val = tensor([0])]; tensor var_1851_cast_fp16 = squeeze(axes = var_1851_axes_0, x = V_sliding_out_3_cast_fp16)[name = string("op_1851_cast_fp16")]; tensor var_1854_begin_0 = const()[name = string("op_1854_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1854_end_0 = const()[name = string("op_1854_end_0"), val = tensor([3, 2, 512, 512])]; tensor var_1854_end_mask_0 = const()[name = string("op_1854_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1854_squeeze_mask_0 = const()[name = string("op_1854_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_1854_cast_fp16 = slice_by_index(begin = var_1854_begin_0, end = var_1854_end_0, end_mask = var_1854_end_mask_0, squeeze_mask = var_1854_squeeze_mask_0, x = K_sliding_in)[name = string("op_1854_cast_fp16")]; tensor K_sliding_slot_5_axes_0 = const()[name = string("K_sliding_slot_5_axes_0"), val = tensor([0])]; tensor K_sliding_slot_5_cast_fp16 = expand_dims(axes = K_sliding_slot_5_axes_0, x = var_1854_cast_fp16)[name = string("K_sliding_slot_5_cast_fp16")]; tensor var_1859_begin_0 = const()[name = string("op_1859_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1859_end_0 = const()[name = string("op_1859_end_0"), val = tensor([3, 2, 512, 512])]; tensor var_1859_end_mask_0 = const()[name = string("op_1859_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1859_squeeze_mask_0 = const()[name = string("op_1859_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_1859_cast_fp16 = slice_by_index(begin = var_1859_begin_0, end = var_1859_end_0, end_mask = var_1859_end_mask_0, squeeze_mask = var_1859_squeeze_mask_0, x = V_sliding_in)[name = string("op_1859_cast_fp16")]; tensor V_sliding_slot_5_axes_0 = const()[name = string("V_sliding_slot_5_axes_0"), val = tensor([0])]; tensor V_sliding_slot_5_cast_fp16 = expand_dims(axes = V_sliding_slot_5_axes_0, x = var_1859_cast_fp16)[name = string("V_sliding_slot_5_cast_fp16")]; int32 var_1866 = const()[name = string("op_1866"), val = int32(-1)]; fp16 const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1868_cast_fp16 = mul(x = x_39_cast_fp16, y = const_24_promoted_to_fp16)[name = string("op_1868_cast_fp16")]; bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; tensor input_61_cast_fp16 = concat(axis = var_1866, interleave = input_61_interleave_0, values = (x_39_cast_fp16, var_1868_cast_fp16))[name = string("input_61_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_1863_to_fp16 = const()[name = string("op_1863_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_1863_to_fp16, x = input_61_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_1873_split_sizes_0 = const()[name = string("op_1873_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1873_axis_0 = const()[name = string("op_1873_axis_0"), val = int32(-1)]; tensor var_1873_cast_fp16_0, tensor var_1873_cast_fp16_1 = split(axis = var_1873_axis_0, split_sizes = var_1873_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_1873_cast_fp16")]; tensor layers_2_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537201792)))]; tensor h_13_cast_fp16 = mul(x = var_1873_cast_fp16_0, y = layers_2_input_layernorm_weight_promoted_to_fp16)[name = string("h_13_cast_fp16")]; tensor var_1879 = const()[name = string("op_1879"), val = tensor([0, 2, 1])]; tensor var_1882_axes_0 = const()[name = string("op_1882_axes_0"), val = tensor([2])]; tensor var_1880_cast_fp16 = transpose(perm = var_1879, x = h_13_cast_fp16)[name = string("transpose_179")]; tensor var_1882_cast_fp16 = expand_dims(axes = var_1882_axes_0, x = var_1880_cast_fp16)[name = string("op_1882_cast_fp16")]; string var_1898_pad_type_0 = const()[name = string("op_1898_pad_type_0"), val = string("valid")]; tensor var_1898_strides_0 = const()[name = string("op_1898_strides_0"), val = tensor([1, 1])]; tensor var_1898_pad_0 = const()[name = string("op_1898_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1898_dilations_0 = const()[name = string("op_1898_dilations_0"), val = tensor([1, 1])]; int32 var_1898_groups_0 = const()[name = string("op_1898_groups_0"), val = int32(1)]; tensor var_1898 = conv(dilations = var_1898_dilations_0, groups = var_1898_groups_0, pad = var_1898_pad_0, pad_type = var_1898_pad_type_0, strides = var_1898_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_1882_cast_fp16)[name = string("op_1898")]; tensor var_1903 = const()[name = string("op_1903"), val = tensor([1, 8, 256, 1])]; tensor var_1904 = reshape(shape = var_1903, x = var_1898)[name = string("op_1904")]; tensor var_1909 = const()[name = string("op_1909"), val = tensor([0, 1, 3, 2])]; tensor var_1919 = const()[name = string("op_1919"), val = tensor([1, 8, 256])]; tensor var_1910 = transpose(perm = var_1909, x = var_1904)[name = string("transpose_178")]; tensor x_41 = reshape(shape = var_1919, x = var_1910)[name = string("x_41")]; int32 var_1925 = const()[name = string("op_1925"), val = int32(-1)]; fp16 const_25_promoted = const()[name = string("const_25_promoted"), val = fp16(-0x1p+0)]; tensor var_1927 = mul(x = x_41, y = const_25_promoted)[name = string("op_1927")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65 = concat(axis = var_1925, interleave = input_65_interleave_0, values = (x_41, var_1927))[name = string("input_65")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_1922_to_fp16 = const()[name = string("op_1922_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_1922_to_fp16, x = input_65)[name = string("normed_61_cast_fp16")]; tensor var_1932_split_sizes_0 = const()[name = string("op_1932_split_sizes_0"), val = tensor([256, 256])]; int32 var_1932_axis_0 = const()[name = string("op_1932_axis_0"), val = int32(-1)]; tensor var_1932_0, tensor var_1932_1 = split(axis = var_1932_axis_0, split_sizes = var_1932_split_sizes_0, x = normed_61_cast_fp16)[name = string("op_1932")]; tensor var_1934 = mul(x = var_1932_0, y = layers_2_self_attn_q_norm_weight)[name = string("op_1934")]; tensor var_1939 = const()[name = string("op_1939"), val = tensor([1, 8, 1, 256])]; tensor q_19 = reshape(shape = var_1939, x = var_1934)[name = string("q_19")]; tensor var_1941_cast_fp16 = mul(x = q_19, y = cos_s)[name = string("op_1941_cast_fp16")]; tensor var_1942_split_sizes_0 = const()[name = string("op_1942_split_sizes_0"), val = tensor([128, 128])]; int32 var_1942_axis_0 = const()[name = string("op_1942_axis_0"), val = int32(-1)]; tensor var_1942_0, tensor var_1942_1 = split(axis = var_1942_axis_0, split_sizes = var_1942_split_sizes_0, x = q_19)[name = string("op_1942")]; fp16 const_26_promoted = const()[name = string("const_26_promoted"), val = fp16(-0x1p+0)]; tensor var_1944 = mul(x = var_1942_1, y = const_26_promoted)[name = string("op_1944")]; int32 var_1946 = const()[name = string("op_1946"), val = int32(-1)]; bool var_1947_interleave_0 = const()[name = string("op_1947_interleave_0"), val = bool(false)]; tensor var_1947 = concat(axis = var_1946, interleave = var_1947_interleave_0, values = (var_1944, var_1942_0))[name = string("op_1947")]; tensor var_1948_cast_fp16 = mul(x = var_1947, y = sin_s)[name = string("op_1948_cast_fp16")]; tensor q_23_cast_fp16 = add(x = var_1941_cast_fp16, y = var_1948_cast_fp16)[name = string("q_23_cast_fp16")]; string var_1961_pad_type_0 = const()[name = string("op_1961_pad_type_0"), val = string("valid")]; tensor var_1961_strides_0 = const()[name = string("op_1961_strides_0"), val = tensor([1, 1])]; tensor var_1961_pad_0 = const()[name = string("op_1961_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1961_dilations_0 = const()[name = string("op_1961_dilations_0"), val = tensor([1, 1])]; int32 var_1961_groups_0 = const()[name = string("op_1961_groups_0"), val = int32(1)]; tensor var_1961 = conv(dilations = var_1961_dilations_0, groups = var_1961_groups_0, pad = var_1961_pad_0, pad_type = var_1961_pad_type_0, strides = var_1961_strides_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_1882_cast_fp16)[name = string("op_1961")]; tensor var_1966 = const()[name = string("op_1966"), val = tensor([1, 2, 256, 1])]; tensor var_1967 = reshape(shape = var_1966, x = var_1961)[name = string("op_1967")]; tensor var_1972 = const()[name = string("op_1972"), val = tensor([0, 1, 3, 2])]; string var_1989_pad_type_0 = const()[name = string("op_1989_pad_type_0"), val = string("valid")]; tensor var_1989_strides_0 = const()[name = string("op_1989_strides_0"), val = tensor([1, 1])]; tensor var_1989_pad_0 = const()[name = string("op_1989_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1989_dilations_0 = const()[name = string("op_1989_dilations_0"), val = tensor([1, 1])]; int32 var_1989_groups_0 = const()[name = string("op_1989_groups_0"), val = int32(1)]; tensor var_1989 = conv(dilations = var_1989_dilations_0, groups = var_1989_groups_0, pad = var_1989_pad_0, pad_type = var_1989_pad_type_0, strides = var_1989_strides_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_1882_cast_fp16)[name = string("op_1989")]; tensor var_1994 = const()[name = string("op_1994"), val = tensor([1, 2, 256, 1])]; tensor var_1995 = reshape(shape = var_1994, x = var_1989)[name = string("op_1995")]; tensor var_2000 = const()[name = string("op_2000"), val = tensor([0, 1, 3, 2])]; tensor var_2010 = const()[name = string("op_2010"), val = tensor([1, 2, 256])]; tensor var_1973 = transpose(perm = var_1972, x = var_1967)[name = string("transpose_177")]; tensor x_43 = reshape(shape = var_2010, x = var_1973)[name = string("x_43")]; int32 var_2016 = const()[name = string("op_2016"), val = int32(-1)]; fp16 const_27_promoted = const()[name = string("const_27_promoted"), val = fp16(-0x1p+0)]; tensor var_2018 = mul(x = x_43, y = const_27_promoted)[name = string("op_2018")]; bool input_67_interleave_0 = const()[name = string("input_67_interleave_0"), val = bool(false)]; tensor input_67 = concat(axis = var_2016, interleave = input_67_interleave_0, values = (x_43, var_2018))[name = string("input_67")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_2013_to_fp16 = const()[name = string("op_2013_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_2013_to_fp16, x = input_67)[name = string("normed_65_cast_fp16")]; tensor var_2023_split_sizes_0 = const()[name = string("op_2023_split_sizes_0"), val = tensor([256, 256])]; int32 var_2023_axis_0 = const()[name = string("op_2023_axis_0"), val = int32(-1)]; tensor var_2023_0, tensor var_2023_1 = split(axis = var_2023_axis_0, split_sizes = var_2023_split_sizes_0, x = normed_65_cast_fp16)[name = string("op_2023")]; tensor var_2025 = mul(x = var_2023_0, y = layers_2_self_attn_k_norm_weight)[name = string("op_2025")]; tensor var_2030 = const()[name = string("op_2030"), val = tensor([1, 2, 1, 256])]; tensor q_21 = reshape(shape = var_2030, x = var_2025)[name = string("q_21")]; fp16 var_2032_promoted = const()[name = string("op_2032_promoted"), val = fp16(0x1p+1)]; tensor var_2001 = transpose(perm = var_2000, x = var_1995)[name = string("transpose_176")]; tensor var_2033 = pow(x = var_2001, y = var_2032_promoted)[name = string("op_2033")]; tensor var_2038_axes_0 = const()[name = string("op_2038_axes_0"), val = tensor([-1])]; bool var_2038_keep_dims_0 = const()[name = string("op_2038_keep_dims_0"), val = bool(true)]; tensor var_2038 = reduce_mean(axes = var_2038_axes_0, keep_dims = var_2038_keep_dims_0, x = var_2033)[name = string("op_2038")]; fp16 var_2040_to_fp16 = const()[name = string("op_2040_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_5_cast_fp16 = add(x = var_2038, y = var_2040_to_fp16)[name = string("mean_sq_5_cast_fp16")]; fp32 var_2042_epsilon_0 = const()[name = string("op_2042_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2042_cast_fp16 = rsqrt(epsilon = var_2042_epsilon_0, x = mean_sq_5_cast_fp16)[name = string("op_2042_cast_fp16")]; tensor input_71_cast_fp16 = mul(x = var_2001, y = var_2042_cast_fp16)[name = string("input_71_cast_fp16")]; tensor var_2044_cast_fp16 = mul(x = q_21, y = cos_s)[name = string("op_2044_cast_fp16")]; tensor var_2045_split_sizes_0 = const()[name = string("op_2045_split_sizes_0"), val = tensor([128, 128])]; int32 var_2045_axis_0 = const()[name = string("op_2045_axis_0"), val = int32(-1)]; tensor var_2045_0, tensor var_2045_1 = split(axis = var_2045_axis_0, split_sizes = var_2045_split_sizes_0, x = q_21)[name = string("op_2045")]; fp16 const_28_promoted = const()[name = string("const_28_promoted"), val = fp16(-0x1p+0)]; tensor var_2047 = mul(x = var_2045_1, y = const_28_promoted)[name = string("op_2047")]; int32 var_2049 = const()[name = string("op_2049"), val = int32(-1)]; bool var_2050_interleave_0 = const()[name = string("op_2050_interleave_0"), val = bool(false)]; tensor var_2050 = concat(axis = var_2049, interleave = var_2050_interleave_0, values = (var_2047, var_2045_0))[name = string("op_2050")]; tensor var_2051_cast_fp16 = mul(x = var_2050, y = sin_s)[name = string("op_2051_cast_fp16")]; tensor input_69_cast_fp16 = add(x = var_2044_cast_fp16, y = var_2051_cast_fp16)[name = string("input_69_cast_fp16")]; tensor k_padded_5_pad_0 = const()[name = string("k_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_5_mode_0 = const()[name = string("k_padded_5_mode_0"), val = string("constant")]; fp16 const_29_to_fp16 = const()[name = string("const_29_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_5_cast_fp16 = pad(constant_val = const_29_to_fp16, mode = k_padded_5_mode_0, pad = k_padded_5_pad_0, x = input_69_cast_fp16)[name = string("k_padded_5_cast_fp16")]; tensor v_padded_5_pad_0 = const()[name = string("v_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_5_mode_0 = const()[name = string("v_padded_5_mode_0"), val = string("constant")]; fp16 const_30_to_fp16 = const()[name = string("const_30_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_5_cast_fp16 = pad(constant_val = const_30_to_fp16, mode = v_padded_5_mode_0, pad = v_padded_5_pad_0, x = input_71_cast_fp16)[name = string("v_padded_5_cast_fp16")]; tensor var_2080_begin_0 = const()[name = string("op_2080_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_2080_end_0 = const()[name = string("op_2080_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2080_end_mask_0 = const()[name = string("op_2080_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2080_cast_fp16 = slice_by_index(begin = var_2080_begin_0, end = var_2080_end_0, end_mask = var_2080_end_mask_0, x = K_sliding_slot_5_cast_fp16)[name = string("op_2080_cast_fp16")]; int32 var_2087 = const()[name = string("op_2087"), val = int32(2)]; bool K_sliding_out_5_interleave_0 = const()[name = string("K_sliding_out_5_interleave_0"), val = bool(false)]; tensor K_sliding_out_5_cast_fp16 = concat(axis = var_2087, interleave = K_sliding_out_5_interleave_0, values = (var_2080_cast_fp16, k_padded_5_cast_fp16))[name = string("K_sliding_out_5_cast_fp16")]; tensor var_2103_begin_0 = const()[name = string("op_2103_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_2103_end_0 = const()[name = string("op_2103_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2103_end_mask_0 = const()[name = string("op_2103_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2103_cast_fp16 = slice_by_index(begin = var_2103_begin_0, end = var_2103_end_0, end_mask = var_2103_end_mask_0, x = V_sliding_slot_5_cast_fp16)[name = string("op_2103_cast_fp16")]; int32 var_2110 = const()[name = string("op_2110"), val = int32(2)]; bool V_sliding_out_5_interleave_0 = const()[name = string("V_sliding_out_5_interleave_0"), val = bool(false)]; tensor V_sliding_out_5_cast_fp16 = concat(axis = var_2110, interleave = V_sliding_out_5_interleave_0, values = (var_2103_cast_fp16, v_padded_5_cast_fp16))[name = string("V_sliding_out_5_cast_fp16")]; tensor K_for_attn_5_begin_0 = const()[name = string("K_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_5_end_0 = const()[name = string("K_for_attn_5_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_5_end_mask_0 = const()[name = string("K_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_5_cast_fp16 = slice_by_index(begin = K_for_attn_5_begin_0, end = K_for_attn_5_end_0, end_mask = K_for_attn_5_end_mask_0, x = K_sliding_out_5_cast_fp16)[name = string("K_for_attn_5_cast_fp16")]; tensor V_for_attn_5_begin_0 = const()[name = string("V_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_5_end_0 = const()[name = string("V_for_attn_5_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_5_end_mask_0 = const()[name = string("V_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_5_cast_fp16 = slice_by_index(begin = V_for_attn_5_begin_0, end = V_for_attn_5_end_0, end_mask = V_for_attn_5_end_mask_0, x = V_sliding_out_5_cast_fp16)[name = string("V_for_attn_5_cast_fp16")]; tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = K_for_attn_5_cast_fp16)[name = string("transpose_175")]; tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; tensor concat_8 = const()[name = string("concat_8"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_8, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_9 = const()[name = string("concat_9"), val = tensor([-1, 1, 512, 256])]; tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_174")]; tensor reshape_9_cast_fp16 = reshape(shape = concat_9, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor transpose_50_perm_0 = const()[name = string("transpose_50_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = V_for_attn_5_cast_fp16)[name = string("transpose_173")]; tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; tensor concat_10 = const()[name = string("concat_10"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_10, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_11 = const()[name = string("concat_11"), val = tensor([-1, 1, 512, 256])]; tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_172")]; tensor reshape_11_cast_fp16 = reshape(shape = concat_11, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor V_expanded_5_perm_0 = const()[name = string("V_expanded_5_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor transpose_50_cast_fp16 = transpose(perm = transpose_50_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_171")]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = q_23_cast_fp16, y = transpose_50_cast_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor x_47_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask_sliding)[name = string("x_47_cast_fp16")]; tensor reduce_max_2_axes_0 = const()[name = string("reduce_max_2_axes_0"), val = tensor([-1])]; bool reduce_max_2_keep_dims_0 = const()[name = string("reduce_max_2_keep_dims_0"), val = bool(true)]; tensor reduce_max_2 = reduce_max(axes = reduce_max_2_axes_0, keep_dims = reduce_max_2_keep_dims_0, x = x_47_cast_fp16)[name = string("reduce_max_2")]; tensor var_2151 = sub(x = x_47_cast_fp16, y = reduce_max_2)[name = string("op_2151")]; tensor var_2157 = exp(x = var_2151)[name = string("op_2157")]; tensor var_2167_axes_0 = const()[name = string("op_2167_axes_0"), val = tensor([-1])]; bool var_2167_keep_dims_0 = const()[name = string("op_2167_keep_dims_0"), val = bool(true)]; tensor var_2167 = reduce_sum(axes = var_2167_axes_0, keep_dims = var_2167_keep_dims_0, x = var_2157)[name = string("op_2167")]; tensor var_2173_cast_fp16 = real_div(x = var_2157, y = var_2167)[name = string("op_2173_cast_fp16")]; bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; tensor V_expanded_5_cast_fp16 = transpose(perm = V_expanded_5_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_170")]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = var_2173_cast_fp16, y = V_expanded_5_cast_fp16)[name = string("attn_output_13_cast_fp16")]; tensor var_2184 = const()[name = string("op_2184"), val = tensor([0, 2, 1, 3])]; tensor var_2191 = const()[name = string("op_2191"), val = tensor([1, 1, -1])]; tensor var_2185_cast_fp16 = transpose(perm = var_2184, x = attn_output_13_cast_fp16)[name = string("transpose_169")]; tensor attn_output_15_cast_fp16 = reshape(shape = var_2191, x = var_2185_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor var_2196 = const()[name = string("op_2196"), val = tensor([0, 2, 1])]; string var_2212_pad_type_0 = const()[name = string("op_2212_pad_type_0"), val = string("valid")]; int32 var_2212_groups_0 = const()[name = string("op_2212_groups_0"), val = int32(1)]; tensor var_2212_strides_0 = const()[name = string("op_2212_strides_0"), val = tensor([1])]; tensor var_2212_pad_0 = const()[name = string("op_2212_pad_0"), val = tensor([0, 0])]; tensor var_2212_dilations_0 = const()[name = string("op_2212_dilations_0"), val = tensor([1])]; tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537206976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539828480))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2197_cast_fp16 = transpose(perm = var_2196, x = attn_output_15_cast_fp16)[name = string("transpose_168")]; tensor var_2212_cast_fp16 = conv(dilations = var_2212_dilations_0, groups = var_2212_groups_0, pad = var_2212_pad_0, pad_type = var_2212_pad_type_0, strides = var_2212_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2197_cast_fp16)[name = string("op_2212_cast_fp16")]; tensor var_2216 = const()[name = string("op_2216"), val = tensor([0, 2, 1])]; int32 var_2222 = const()[name = string("op_2222"), val = int32(-1)]; fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_51_cast_fp16 = transpose(perm = var_2216, x = var_2212_cast_fp16)[name = string("transpose_167")]; tensor var_2224_cast_fp16 = mul(x = x_51_cast_fp16, y = const_31_promoted_to_fp16)[name = string("op_2224_cast_fp16")]; bool input_75_interleave_0 = const()[name = string("input_75_interleave_0"), val = bool(false)]; tensor input_75_cast_fp16 = concat(axis = var_2222, interleave = input_75_interleave_0, values = (x_51_cast_fp16, var_2224_cast_fp16))[name = string("input_75_cast_fp16")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_2219_to_fp16 = const()[name = string("op_2219_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_2219_to_fp16, x = input_75_cast_fp16)[name = string("normed_69_cast_fp16")]; tensor var_2229_split_sizes_0 = const()[name = string("op_2229_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2229_axis_0 = const()[name = string("op_2229_axis_0"), val = int32(-1)]; tensor var_2229_cast_fp16_0, tensor var_2229_cast_fp16_1 = split(axis = var_2229_axis_0, split_sizes = var_2229_split_sizes_0, x = normed_69_cast_fp16)[name = string("op_2229_cast_fp16")]; tensor layers_2_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539831104)))]; tensor attn_output_17_cast_fp16 = mul(x = var_2229_cast_fp16_0, y = layers_2_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_17_cast_fp16")]; tensor x_53_cast_fp16 = add(x = x_39_cast_fp16, y = attn_output_17_cast_fp16)[name = string("x_53_cast_fp16")]; int32 var_2238 = const()[name = string("op_2238"), val = int32(-1)]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2240_cast_fp16 = mul(x = x_53_cast_fp16, y = const_32_promoted_to_fp16)[name = string("op_2240_cast_fp16")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77_cast_fp16 = concat(axis = var_2238, interleave = input_77_interleave_0, values = (x_53_cast_fp16, var_2240_cast_fp16))[name = string("input_77_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_2235_to_fp16 = const()[name = string("op_2235_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_2235_to_fp16, x = input_77_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor var_2245_split_sizes_0 = const()[name = string("op_2245_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2245_axis_0 = const()[name = string("op_2245_axis_0"), val = int32(-1)]; tensor var_2245_cast_fp16_0, tensor var_2245_cast_fp16_1 = split(axis = var_2245_axis_0, split_sizes = var_2245_split_sizes_0, x = normed_73_cast_fp16)[name = string("op_2245_cast_fp16")]; tensor layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539836288)))]; tensor h_15_cast_fp16 = mul(x = var_2245_cast_fp16_0, y = layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_15_cast_fp16")]; tensor var_2256 = const()[name = string("op_2256"), val = tensor([0, 2, 1])]; tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; tensor var_2257 = transpose(perm = var_2256, x = h_15_cast_fp16)[name = string("transpose_166")]; tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_2257)[name = string("input_79")]; string gate_9_pad_type_0 = const()[name = string("gate_9_pad_type_0"), val = string("valid")]; tensor gate_9_strides_0 = const()[name = string("gate_9_strides_0"), val = tensor([1, 1])]; tensor gate_9_pad_0 = const()[name = string("gate_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_9_dilations_0 = const()[name = string("gate_9_dilations_0"), val = tensor([1, 1])]; int32 gate_9_groups_0 = const()[name = string("gate_9_groups_0"), val = int32(1)]; tensor gate_9 = conv(dilations = gate_9_dilations_0, groups = gate_9_groups_0, pad = gate_9_pad_0, pad_type = gate_9_pad_type_0, strides = gate_9_strides_0, weight = layers_2_mlp_gate_proj_weight_palettized, x = input_79)[name = string("gate_9")]; string up_5_pad_type_0 = const()[name = string("up_5_pad_type_0"), val = string("valid")]; tensor up_5_strides_0 = const()[name = string("up_5_strides_0"), val = tensor([1, 1])]; tensor up_5_pad_0 = const()[name = string("up_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_5_dilations_0 = const()[name = string("up_5_dilations_0"), val = tensor([1, 1])]; int32 up_5_groups_0 = const()[name = string("up_5_groups_0"), val = int32(1)]; tensor up_5 = conv(dilations = up_5_dilations_0, groups = up_5_groups_0, pad = up_5_pad_0, pad_type = up_5_pad_type_0, strides = up_5_strides_0, weight = layers_2_mlp_up_proj_weight_palettized, x = input_79)[name = string("up_5")]; string gate_11_mode_0 = const()[name = string("gate_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_11 = gelu(mode = gate_11_mode_0, x = gate_9)[name = string("gate_11")]; tensor input_81 = mul(x = gate_11, y = up_5)[name = string("input_81")]; string mlp_out_5_pad_type_0 = const()[name = string("mlp_out_5_pad_type_0"), val = string("valid")]; tensor mlp_out_5_strides_0 = const()[name = string("mlp_out_5_strides_0"), val = tensor([1, 1])]; tensor mlp_out_5_pad_0 = const()[name = string("mlp_out_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_5_dilations_0 = const()[name = string("mlp_out_5_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_5_groups_0 = const()[name = string("mlp_out_5_groups_0"), val = int32(1)]; tensor mlp_out_5 = conv(dilations = mlp_out_5_dilations_0, groups = mlp_out_5_groups_0, pad = mlp_out_5_pad_0, pad_type = mlp_out_5_pad_type_0, strides = mlp_out_5_strides_0, weight = layers_2_mlp_down_proj_weight_palettized, x = input_81)[name = string("mlp_out_5")]; tensor var_2297_axes_0 = const()[name = string("op_2297_axes_0"), val = tensor([2])]; tensor var_2297 = squeeze(axes = var_2297_axes_0, x = mlp_out_5)[name = string("op_2297")]; tensor var_2301 = const()[name = string("op_2301"), val = tensor([0, 2, 1])]; int32 var_2307 = const()[name = string("op_2307"), val = int32(-1)]; fp16 const_33_promoted = const()[name = string("const_33_promoted"), val = fp16(-0x1p+0)]; tensor x_55 = transpose(perm = var_2301, x = var_2297)[name = string("transpose_165")]; tensor var_2309 = mul(x = x_55, y = const_33_promoted)[name = string("op_2309")]; bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; tensor input_83 = concat(axis = var_2307, interleave = input_83_interleave_0, values = (x_55, var_2309))[name = string("input_83")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_2304_to_fp16 = const()[name = string("op_2304_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_2304_to_fp16, x = input_83)[name = string("normed_77_cast_fp16")]; tensor var_2314_split_sizes_0 = const()[name = string("op_2314_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2314_axis_0 = const()[name = string("op_2314_axis_0"), val = int32(-1)]; tensor var_2314_0, tensor var_2314_1 = split(axis = var_2314_axis_0, split_sizes = var_2314_split_sizes_0, x = normed_77_cast_fp16)[name = string("op_2314")]; tensor hidden_states_23 = mul(x = var_2314_0, y = layers_2_post_feedforward_layernorm_weight)[name = string("hidden_states_23")]; tensor hidden_states_25_cast_fp16 = add(x = x_53_cast_fp16, y = hidden_states_23)[name = string("hidden_states_25_cast_fp16")]; tensor per_layer_slice_5_begin_0 = const()[name = string("per_layer_slice_5_begin_0"), val = tensor([0, 0, 3584])]; tensor per_layer_slice_5_end_0 = const()[name = string("per_layer_slice_5_end_0"), val = tensor([1, 1, 3840])]; tensor per_layer_slice_5_end_mask_0 = const()[name = string("per_layer_slice_5_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_5_cast_fp16 = slice_by_index(begin = per_layer_slice_5_begin_0, end = per_layer_slice_5_end_0, end_mask = per_layer_slice_5_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_5_cast_fp16")]; tensor var_2342 = const()[name = string("op_2342"), val = tensor([0, 2, 1])]; tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; tensor var_2343 = transpose(perm = var_2342, x = hidden_states_25_cast_fp16)[name = string("transpose_164")]; tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_2343)[name = string("input_85")]; string gated_13_pad_type_0 = const()[name = string("gated_13_pad_type_0"), val = string("valid")]; tensor gated_13_strides_0 = const()[name = string("gated_13_strides_0"), val = tensor([1, 1])]; tensor gated_13_pad_0 = const()[name = string("gated_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_13_dilations_0 = const()[name = string("gated_13_dilations_0"), val = tensor([1, 1])]; int32 gated_13_groups_0 = const()[name = string("gated_13_groups_0"), val = int32(1)]; tensor gated_13 = conv(dilations = gated_13_dilations_0, groups = gated_13_groups_0, pad = gated_13_pad_0, pad_type = gated_13_pad_type_0, strides = gated_13_strides_0, weight = layers_2_per_layer_input_gate_weight_palettized, x = input_85)[name = string("gated_13")]; string gated_15_mode_0 = const()[name = string("gated_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_15 = gelu(mode = gated_15_mode_0, x = gated_13)[name = string("gated_15")]; tensor var_2362 = const()[name = string("op_2362"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_5_axes_0 = const()[name = string("per_layer_slice_conv_5_axes_0"), val = tensor([2])]; tensor var_2363_cast_fp16 = transpose(perm = var_2362, x = per_layer_slice_5_cast_fp16)[name = string("transpose_163")]; tensor per_layer_slice_conv_5_cast_fp16 = expand_dims(axes = per_layer_slice_conv_5_axes_0, x = var_2363_cast_fp16)[name = string("per_layer_slice_conv_5_cast_fp16")]; tensor input_87_cast_fp16 = mul(x = gated_15, y = per_layer_slice_conv_5_cast_fp16)[name = string("input_87_cast_fp16")]; string gated_17_pad_type_0 = const()[name = string("gated_17_pad_type_0"), val = string("valid")]; tensor gated_17_strides_0 = const()[name = string("gated_17_strides_0"), val = tensor([1, 1])]; tensor gated_17_pad_0 = const()[name = string("gated_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_17_dilations_0 = const()[name = string("gated_17_dilations_0"), val = tensor([1, 1])]; int32 gated_17_groups_0 = const()[name = string("gated_17_groups_0"), val = int32(1)]; tensor layers_2_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539841472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540169216))))[name = string("layers_2_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_17_cast_fp16 = conv(dilations = gated_17_dilations_0, groups = gated_17_groups_0, pad = gated_17_pad_0, pad_type = gated_17_pad_type_0, strides = gated_17_strides_0, weight = layers_2_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_87_cast_fp16)[name = string("gated_17_cast_fp16")]; tensor var_2379_axes_0 = const()[name = string("op_2379_axes_0"), val = tensor([2])]; tensor var_2379_cast_fp16 = squeeze(axes = var_2379_axes_0, x = gated_17_cast_fp16)[name = string("op_2379_cast_fp16")]; tensor var_2383 = const()[name = string("op_2383"), val = tensor([0, 2, 1])]; int32 var_2389 = const()[name = string("op_2389"), val = int32(-1)]; fp16 const_34_promoted_to_fp16 = const()[name = string("const_34_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_57_cast_fp16 = transpose(perm = var_2383, x = var_2379_cast_fp16)[name = string("transpose_162")]; tensor var_2391_cast_fp16 = mul(x = x_57_cast_fp16, y = const_34_promoted_to_fp16)[name = string("op_2391_cast_fp16")]; bool input_89_interleave_0 = const()[name = string("input_89_interleave_0"), val = bool(false)]; tensor input_89_cast_fp16 = concat(axis = var_2389, interleave = input_89_interleave_0, values = (x_57_cast_fp16, var_2391_cast_fp16))[name = string("input_89_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_2386_to_fp16 = const()[name = string("op_2386_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_2386_to_fp16, x = input_89_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor var_2396_split_sizes_0 = const()[name = string("op_2396_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2396_axis_0 = const()[name = string("op_2396_axis_0"), val = int32(-1)]; tensor var_2396_cast_fp16_0, tensor var_2396_cast_fp16_1 = split(axis = var_2396_axis_0, split_sizes = var_2396_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_2396_cast_fp16")]; tensor layers_2_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540171840)))]; tensor hidden_states_29_cast_fp16 = mul(x = var_2396_cast_fp16_0, y = layers_2_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_29_cast_fp16")]; tensor hidden_states_31_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_29_cast_fp16)[name = string("hidden_states_31_cast_fp16")]; tensor const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = tensor([0x1.58p-1])]; tensor x_59_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_35_promoted_to_fp16)[name = string("x_59_cast_fp16")]; tensor var_2408_axes_0 = const()[name = string("op_2408_axes_0"), val = tensor([0])]; tensor var_2408_cast_fp16 = squeeze(axes = var_2408_axes_0, x = K_sliding_out_5_cast_fp16)[name = string("op_2408_cast_fp16")]; tensor var_2410_axes_0 = const()[name = string("op_2410_axes_0"), val = tensor([0])]; tensor var_2410_cast_fp16 = squeeze(axes = var_2410_axes_0, x = V_sliding_out_5_cast_fp16)[name = string("op_2410_cast_fp16")]; tensor var_2413_begin_0 = const()[name = string("op_2413_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2413_end_0 = const()[name = string("op_2413_end_0"), val = tensor([4, 2, 512, 512])]; tensor var_2413_end_mask_0 = const()[name = string("op_2413_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2413_squeeze_mask_0 = const()[name = string("op_2413_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2413_cast_fp16 = slice_by_index(begin = var_2413_begin_0, end = var_2413_end_0, end_mask = var_2413_end_mask_0, squeeze_mask = var_2413_squeeze_mask_0, x = K_sliding_in)[name = string("op_2413_cast_fp16")]; tensor K_sliding_slot_7_axes_0 = const()[name = string("K_sliding_slot_7_axes_0"), val = tensor([0])]; tensor K_sliding_slot_7_cast_fp16 = expand_dims(axes = K_sliding_slot_7_axes_0, x = var_2413_cast_fp16)[name = string("K_sliding_slot_7_cast_fp16")]; tensor var_2418_begin_0 = const()[name = string("op_2418_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2418_end_0 = const()[name = string("op_2418_end_0"), val = tensor([4, 2, 512, 512])]; tensor var_2418_end_mask_0 = const()[name = string("op_2418_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2418_squeeze_mask_0 = const()[name = string("op_2418_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2418_cast_fp16 = slice_by_index(begin = var_2418_begin_0, end = var_2418_end_0, end_mask = var_2418_end_mask_0, squeeze_mask = var_2418_squeeze_mask_0, x = V_sliding_in)[name = string("op_2418_cast_fp16")]; tensor V_sliding_slot_7_axes_0 = const()[name = string("V_sliding_slot_7_axes_0"), val = tensor([0])]; tensor V_sliding_slot_7_cast_fp16 = expand_dims(axes = V_sliding_slot_7_axes_0, x = var_2418_cast_fp16)[name = string("V_sliding_slot_7_cast_fp16")]; int32 var_2425 = const()[name = string("op_2425"), val = int32(-1)]; fp16 const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2427_cast_fp16 = mul(x = x_59_cast_fp16, y = const_36_promoted_to_fp16)[name = string("op_2427_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_2425, interleave = input_91_interleave_0, values = (x_59_cast_fp16, var_2427_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_2422_to_fp16 = const()[name = string("op_2422_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_2422_to_fp16, x = input_91_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_2432_split_sizes_0 = const()[name = string("op_2432_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2432_axis_0 = const()[name = string("op_2432_axis_0"), val = int32(-1)]; tensor var_2432_cast_fp16_0, tensor var_2432_cast_fp16_1 = split(axis = var_2432_axis_0, split_sizes = var_2432_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_2432_cast_fp16")]; tensor layers_3_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540177024)))]; tensor h_19_cast_fp16 = mul(x = var_2432_cast_fp16_0, y = layers_3_input_layernorm_weight_promoted_to_fp16)[name = string("h_19_cast_fp16")]; tensor var_2438 = const()[name = string("op_2438"), val = tensor([0, 2, 1])]; tensor var_2441_axes_0 = const()[name = string("op_2441_axes_0"), val = tensor([2])]; tensor var_2439_cast_fp16 = transpose(perm = var_2438, x = h_19_cast_fp16)[name = string("transpose_161")]; tensor var_2441_cast_fp16 = expand_dims(axes = var_2441_axes_0, x = var_2439_cast_fp16)[name = string("op_2441_cast_fp16")]; string var_2457_pad_type_0 = const()[name = string("op_2457_pad_type_0"), val = string("valid")]; tensor var_2457_strides_0 = const()[name = string("op_2457_strides_0"), val = tensor([1, 1])]; tensor var_2457_pad_0 = const()[name = string("op_2457_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2457_dilations_0 = const()[name = string("op_2457_dilations_0"), val = tensor([1, 1])]; int32 var_2457_groups_0 = const()[name = string("op_2457_groups_0"), val = int32(1)]; tensor var_2457 = conv(dilations = var_2457_dilations_0, groups = var_2457_groups_0, pad = var_2457_pad_0, pad_type = var_2457_pad_type_0, strides = var_2457_strides_0, weight = layers_3_self_attn_q_proj_weight_palettized, x = var_2441_cast_fp16)[name = string("op_2457")]; tensor var_2462 = const()[name = string("op_2462"), val = tensor([1, 8, 256, 1])]; tensor var_2463 = reshape(shape = var_2462, x = var_2457)[name = string("op_2463")]; tensor var_2468 = const()[name = string("op_2468"), val = tensor([0, 1, 3, 2])]; tensor var_2478 = const()[name = string("op_2478"), val = tensor([1, 8, 256])]; tensor var_2469 = transpose(perm = var_2468, x = var_2463)[name = string("transpose_160")]; tensor x_61 = reshape(shape = var_2478, x = var_2469)[name = string("x_61")]; int32 var_2484 = const()[name = string("op_2484"), val = int32(-1)]; fp16 const_37_promoted = const()[name = string("const_37_promoted"), val = fp16(-0x1p+0)]; tensor var_2486 = mul(x = x_61, y = const_37_promoted)[name = string("op_2486")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95 = concat(axis = var_2484, interleave = input_95_interleave_0, values = (x_61, var_2486))[name = string("input_95")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_2481_to_fp16 = const()[name = string("op_2481_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_2481_to_fp16, x = input_95)[name = string("normed_89_cast_fp16")]; tensor var_2491_split_sizes_0 = const()[name = string("op_2491_split_sizes_0"), val = tensor([256, 256])]; int32 var_2491_axis_0 = const()[name = string("op_2491_axis_0"), val = int32(-1)]; tensor var_2491_0, tensor var_2491_1 = split(axis = var_2491_axis_0, split_sizes = var_2491_split_sizes_0, x = normed_89_cast_fp16)[name = string("op_2491")]; tensor var_2493 = mul(x = var_2491_0, y = layers_3_self_attn_q_norm_weight)[name = string("op_2493")]; tensor var_2498 = const()[name = string("op_2498"), val = tensor([1, 8, 1, 256])]; tensor q_27 = reshape(shape = var_2498, x = var_2493)[name = string("q_27")]; tensor var_2500_cast_fp16 = mul(x = q_27, y = cos_s)[name = string("op_2500_cast_fp16")]; tensor var_2501_split_sizes_0 = const()[name = string("op_2501_split_sizes_0"), val = tensor([128, 128])]; int32 var_2501_axis_0 = const()[name = string("op_2501_axis_0"), val = int32(-1)]; tensor var_2501_0, tensor var_2501_1 = split(axis = var_2501_axis_0, split_sizes = var_2501_split_sizes_0, x = q_27)[name = string("op_2501")]; fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; tensor var_2503 = mul(x = var_2501_1, y = const_38_promoted)[name = string("op_2503")]; int32 var_2505 = const()[name = string("op_2505"), val = int32(-1)]; bool var_2506_interleave_0 = const()[name = string("op_2506_interleave_0"), val = bool(false)]; tensor var_2506 = concat(axis = var_2505, interleave = var_2506_interleave_0, values = (var_2503, var_2501_0))[name = string("op_2506")]; tensor var_2507_cast_fp16 = mul(x = var_2506, y = sin_s)[name = string("op_2507_cast_fp16")]; tensor q_31_cast_fp16 = add(x = var_2500_cast_fp16, y = var_2507_cast_fp16)[name = string("q_31_cast_fp16")]; string var_2520_pad_type_0 = const()[name = string("op_2520_pad_type_0"), val = string("valid")]; tensor var_2520_strides_0 = const()[name = string("op_2520_strides_0"), val = tensor([1, 1])]; tensor var_2520_pad_0 = const()[name = string("op_2520_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2520_dilations_0 = const()[name = string("op_2520_dilations_0"), val = tensor([1, 1])]; int32 var_2520_groups_0 = const()[name = string("op_2520_groups_0"), val = int32(1)]; tensor var_2520 = conv(dilations = var_2520_dilations_0, groups = var_2520_groups_0, pad = var_2520_pad_0, pad_type = var_2520_pad_type_0, strides = var_2520_strides_0, weight = layers_3_self_attn_k_proj_weight_palettized, x = var_2441_cast_fp16)[name = string("op_2520")]; tensor var_2525 = const()[name = string("op_2525"), val = tensor([1, 2, 256, 1])]; tensor var_2526 = reshape(shape = var_2525, x = var_2520)[name = string("op_2526")]; tensor var_2531 = const()[name = string("op_2531"), val = tensor([0, 1, 3, 2])]; string var_2548_pad_type_0 = const()[name = string("op_2548_pad_type_0"), val = string("valid")]; tensor var_2548_strides_0 = const()[name = string("op_2548_strides_0"), val = tensor([1, 1])]; tensor var_2548_pad_0 = const()[name = string("op_2548_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2548_dilations_0 = const()[name = string("op_2548_dilations_0"), val = tensor([1, 1])]; int32 var_2548_groups_0 = const()[name = string("op_2548_groups_0"), val = int32(1)]; tensor var_2548 = conv(dilations = var_2548_dilations_0, groups = var_2548_groups_0, pad = var_2548_pad_0, pad_type = var_2548_pad_type_0, strides = var_2548_strides_0, weight = layers_3_self_attn_v_proj_weight_palettized, x = var_2441_cast_fp16)[name = string("op_2548")]; tensor var_2553 = const()[name = string("op_2553"), val = tensor([1, 2, 256, 1])]; tensor var_2554 = reshape(shape = var_2553, x = var_2548)[name = string("op_2554")]; tensor var_2559 = const()[name = string("op_2559"), val = tensor([0, 1, 3, 2])]; tensor var_2569 = const()[name = string("op_2569"), val = tensor([1, 2, 256])]; tensor var_2532 = transpose(perm = var_2531, x = var_2526)[name = string("transpose_159")]; tensor x_63 = reshape(shape = var_2569, x = var_2532)[name = string("x_63")]; int32 var_2575 = const()[name = string("op_2575"), val = int32(-1)]; fp16 const_39_promoted = const()[name = string("const_39_promoted"), val = fp16(-0x1p+0)]; tensor var_2577 = mul(x = x_63, y = const_39_promoted)[name = string("op_2577")]; bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; tensor input_97 = concat(axis = var_2575, interleave = input_97_interleave_0, values = (x_63, var_2577))[name = string("input_97")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_2572_to_fp16 = const()[name = string("op_2572_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_2572_to_fp16, x = input_97)[name = string("normed_93_cast_fp16")]; tensor var_2582_split_sizes_0 = const()[name = string("op_2582_split_sizes_0"), val = tensor([256, 256])]; int32 var_2582_axis_0 = const()[name = string("op_2582_axis_0"), val = int32(-1)]; tensor var_2582_0, tensor var_2582_1 = split(axis = var_2582_axis_0, split_sizes = var_2582_split_sizes_0, x = normed_93_cast_fp16)[name = string("op_2582")]; tensor var_2584 = mul(x = var_2582_0, y = layers_3_self_attn_k_norm_weight)[name = string("op_2584")]; tensor var_2589 = const()[name = string("op_2589"), val = tensor([1, 2, 1, 256])]; tensor q_29 = reshape(shape = var_2589, x = var_2584)[name = string("q_29")]; fp16 var_2591_promoted = const()[name = string("op_2591_promoted"), val = fp16(0x1p+1)]; tensor var_2560 = transpose(perm = var_2559, x = var_2554)[name = string("transpose_158")]; tensor var_2592 = pow(x = var_2560, y = var_2591_promoted)[name = string("op_2592")]; tensor var_2597_axes_0 = const()[name = string("op_2597_axes_0"), val = tensor([-1])]; bool var_2597_keep_dims_0 = const()[name = string("op_2597_keep_dims_0"), val = bool(true)]; tensor var_2597 = reduce_mean(axes = var_2597_axes_0, keep_dims = var_2597_keep_dims_0, x = var_2592)[name = string("op_2597")]; fp16 var_2599_to_fp16 = const()[name = string("op_2599_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_7_cast_fp16 = add(x = var_2597, y = var_2599_to_fp16)[name = string("mean_sq_7_cast_fp16")]; fp32 var_2601_epsilon_0 = const()[name = string("op_2601_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2601_cast_fp16 = rsqrt(epsilon = var_2601_epsilon_0, x = mean_sq_7_cast_fp16)[name = string("op_2601_cast_fp16")]; tensor input_101_cast_fp16 = mul(x = var_2560, y = var_2601_cast_fp16)[name = string("input_101_cast_fp16")]; tensor var_2603_cast_fp16 = mul(x = q_29, y = cos_s)[name = string("op_2603_cast_fp16")]; tensor var_2604_split_sizes_0 = const()[name = string("op_2604_split_sizes_0"), val = tensor([128, 128])]; int32 var_2604_axis_0 = const()[name = string("op_2604_axis_0"), val = int32(-1)]; tensor var_2604_0, tensor var_2604_1 = split(axis = var_2604_axis_0, split_sizes = var_2604_split_sizes_0, x = q_29)[name = string("op_2604")]; fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; tensor var_2606 = mul(x = var_2604_1, y = const_40_promoted)[name = string("op_2606")]; int32 var_2608 = const()[name = string("op_2608"), val = int32(-1)]; bool var_2609_interleave_0 = const()[name = string("op_2609_interleave_0"), val = bool(false)]; tensor var_2609 = concat(axis = var_2608, interleave = var_2609_interleave_0, values = (var_2606, var_2604_0))[name = string("op_2609")]; tensor var_2610_cast_fp16 = mul(x = var_2609, y = sin_s)[name = string("op_2610_cast_fp16")]; tensor input_99_cast_fp16 = add(x = var_2603_cast_fp16, y = var_2610_cast_fp16)[name = string("input_99_cast_fp16")]; tensor k_padded_7_pad_0 = const()[name = string("k_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_7_mode_0 = const()[name = string("k_padded_7_mode_0"), val = string("constant")]; fp16 const_41_to_fp16 = const()[name = string("const_41_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_7_cast_fp16 = pad(constant_val = const_41_to_fp16, mode = k_padded_7_mode_0, pad = k_padded_7_pad_0, x = input_99_cast_fp16)[name = string("k_padded_7_cast_fp16")]; tensor v_padded_7_pad_0 = const()[name = string("v_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_7_mode_0 = const()[name = string("v_padded_7_mode_0"), val = string("constant")]; fp16 const_42_to_fp16 = const()[name = string("const_42_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_7_cast_fp16 = pad(constant_val = const_42_to_fp16, mode = v_padded_7_mode_0, pad = v_padded_7_pad_0, x = input_101_cast_fp16)[name = string("v_padded_7_cast_fp16")]; tensor var_2639_begin_0 = const()[name = string("op_2639_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_2639_end_0 = const()[name = string("op_2639_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2639_end_mask_0 = const()[name = string("op_2639_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2639_cast_fp16 = slice_by_index(begin = var_2639_begin_0, end = var_2639_end_0, end_mask = var_2639_end_mask_0, x = K_sliding_slot_7_cast_fp16)[name = string("op_2639_cast_fp16")]; int32 var_2646 = const()[name = string("op_2646"), val = int32(2)]; bool K_sliding_out_7_interleave_0 = const()[name = string("K_sliding_out_7_interleave_0"), val = bool(false)]; tensor K_sliding_out_7_cast_fp16 = concat(axis = var_2646, interleave = K_sliding_out_7_interleave_0, values = (var_2639_cast_fp16, k_padded_7_cast_fp16))[name = string("K_sliding_out_7_cast_fp16")]; tensor var_2662_begin_0 = const()[name = string("op_2662_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_2662_end_0 = const()[name = string("op_2662_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2662_end_mask_0 = const()[name = string("op_2662_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2662_cast_fp16 = slice_by_index(begin = var_2662_begin_0, end = var_2662_end_0, end_mask = var_2662_end_mask_0, x = V_sliding_slot_7_cast_fp16)[name = string("op_2662_cast_fp16")]; int32 var_2669 = const()[name = string("op_2669"), val = int32(2)]; bool V_sliding_out_7_interleave_0 = const()[name = string("V_sliding_out_7_interleave_0"), val = bool(false)]; tensor V_sliding_out_7_cast_fp16 = concat(axis = var_2669, interleave = V_sliding_out_7_interleave_0, values = (var_2662_cast_fp16, v_padded_7_cast_fp16))[name = string("V_sliding_out_7_cast_fp16")]; tensor K_for_attn_7_begin_0 = const()[name = string("K_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_7_end_0 = const()[name = string("K_for_attn_7_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_7_end_mask_0 = const()[name = string("K_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_7_cast_fp16 = slice_by_index(begin = K_for_attn_7_begin_0, end = K_for_attn_7_end_0, end_mask = K_for_attn_7_end_mask_0, x = K_sliding_out_7_cast_fp16)[name = string("K_for_attn_7_cast_fp16")]; tensor V_for_attn_7_begin_0 = const()[name = string("V_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_7_end_0 = const()[name = string("V_for_attn_7_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_7_end_mask_0 = const()[name = string("V_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_7_cast_fp16 = slice_by_index(begin = V_for_attn_7_begin_0, end = V_for_attn_7_end_0, end_mask = V_for_attn_7_end_mask_0, x = V_sliding_out_7_cast_fp16)[name = string("V_for_attn_7_cast_fp16")]; tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = K_for_attn_7_cast_fp16)[name = string("transpose_157")]; tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; tensor concat_12 = const()[name = string("concat_12"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_12, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_13 = const()[name = string("concat_13"), val = tensor([-1, 1, 512, 256])]; tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_156")]; tensor reshape_13_cast_fp16 = reshape(shape = concat_13, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; tensor transpose_51_perm_0 = const()[name = string("transpose_51_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = V_for_attn_7_cast_fp16)[name = string("transpose_155")]; tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; tensor concat_14 = const()[name = string("concat_14"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_14, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_15 = const()[name = string("concat_15"), val = tensor([-1, 1, 512, 256])]; tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_154")]; tensor reshape_15_cast_fp16 = reshape(shape = concat_15, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor V_expanded_7_perm_0 = const()[name = string("V_expanded_7_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_13_transpose_x_0 = const()[name = string("attn_weights_13_transpose_x_0"), val = bool(false)]; bool attn_weights_13_transpose_y_0 = const()[name = string("attn_weights_13_transpose_y_0"), val = bool(false)]; tensor transpose_51_cast_fp16 = transpose(perm = transpose_51_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_153")]; tensor attn_weights_13_cast_fp16 = matmul(transpose_x = attn_weights_13_transpose_x_0, transpose_y = attn_weights_13_transpose_y_0, x = q_31_cast_fp16, y = transpose_51_cast_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor x_67_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask_sliding)[name = string("x_67_cast_fp16")]; tensor reduce_max_3_axes_0 = const()[name = string("reduce_max_3_axes_0"), val = tensor([-1])]; bool reduce_max_3_keep_dims_0 = const()[name = string("reduce_max_3_keep_dims_0"), val = bool(true)]; tensor reduce_max_3 = reduce_max(axes = reduce_max_3_axes_0, keep_dims = reduce_max_3_keep_dims_0, x = x_67_cast_fp16)[name = string("reduce_max_3")]; tensor var_2710 = sub(x = x_67_cast_fp16, y = reduce_max_3)[name = string("op_2710")]; tensor var_2716 = exp(x = var_2710)[name = string("op_2716")]; tensor var_2726_axes_0 = const()[name = string("op_2726_axes_0"), val = tensor([-1])]; bool var_2726_keep_dims_0 = const()[name = string("op_2726_keep_dims_0"), val = bool(true)]; tensor var_2726 = reduce_sum(axes = var_2726_axes_0, keep_dims = var_2726_keep_dims_0, x = var_2716)[name = string("op_2726")]; tensor var_2732_cast_fp16 = real_div(x = var_2716, y = var_2726)[name = string("op_2732_cast_fp16")]; bool attn_output_19_transpose_x_0 = const()[name = string("attn_output_19_transpose_x_0"), val = bool(false)]; bool attn_output_19_transpose_y_0 = const()[name = string("attn_output_19_transpose_y_0"), val = bool(false)]; tensor V_expanded_7_cast_fp16 = transpose(perm = V_expanded_7_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_152")]; tensor attn_output_19_cast_fp16 = matmul(transpose_x = attn_output_19_transpose_x_0, transpose_y = attn_output_19_transpose_y_0, x = var_2732_cast_fp16, y = V_expanded_7_cast_fp16)[name = string("attn_output_19_cast_fp16")]; tensor var_2743 = const()[name = string("op_2743"), val = tensor([0, 2, 1, 3])]; tensor var_2750 = const()[name = string("op_2750"), val = tensor([1, 1, -1])]; tensor var_2744_cast_fp16 = transpose(perm = var_2743, x = attn_output_19_cast_fp16)[name = string("transpose_151")]; tensor attn_output_21_cast_fp16 = reshape(shape = var_2750, x = var_2744_cast_fp16)[name = string("attn_output_21_cast_fp16")]; tensor var_2755 = const()[name = string("op_2755"), val = tensor([0, 2, 1])]; string var_2771_pad_type_0 = const()[name = string("op_2771_pad_type_0"), val = string("valid")]; int32 var_2771_groups_0 = const()[name = string("op_2771_groups_0"), val = int32(1)]; tensor var_2771_strides_0 = const()[name = string("op_2771_strides_0"), val = tensor([1])]; tensor var_2771_pad_0 = const()[name = string("op_2771_pad_0"), val = tensor([0, 0])]; tensor var_2771_dilations_0 = const()[name = string("op_2771_dilations_0"), val = tensor([1])]; tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540182208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542803712))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2756_cast_fp16 = transpose(perm = var_2755, x = attn_output_21_cast_fp16)[name = string("transpose_150")]; tensor var_2771_cast_fp16 = conv(dilations = var_2771_dilations_0, groups = var_2771_groups_0, pad = var_2771_pad_0, pad_type = var_2771_pad_type_0, strides = var_2771_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_2756_cast_fp16)[name = string("op_2771_cast_fp16")]; tensor var_2775 = const()[name = string("op_2775"), val = tensor([0, 2, 1])]; int32 var_2781 = const()[name = string("op_2781"), val = int32(-1)]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_71_cast_fp16 = transpose(perm = var_2775, x = var_2771_cast_fp16)[name = string("transpose_149")]; tensor var_2783_cast_fp16 = mul(x = x_71_cast_fp16, y = const_43_promoted_to_fp16)[name = string("op_2783_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_2781, interleave = input_105_interleave_0, values = (x_71_cast_fp16, var_2783_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_2778_to_fp16 = const()[name = string("op_2778_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_2778_to_fp16, x = input_105_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor var_2788_split_sizes_0 = const()[name = string("op_2788_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2788_axis_0 = const()[name = string("op_2788_axis_0"), val = int32(-1)]; tensor var_2788_cast_fp16_0, tensor var_2788_cast_fp16_1 = split(axis = var_2788_axis_0, split_sizes = var_2788_split_sizes_0, x = normed_97_cast_fp16)[name = string("op_2788_cast_fp16")]; tensor layers_3_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542806336)))]; tensor attn_output_23_cast_fp16 = mul(x = var_2788_cast_fp16_0, y = layers_3_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_23_cast_fp16")]; tensor x_73_cast_fp16 = add(x = x_59_cast_fp16, y = attn_output_23_cast_fp16)[name = string("x_73_cast_fp16")]; int32 var_2797 = const()[name = string("op_2797"), val = int32(-1)]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2799_cast_fp16 = mul(x = x_73_cast_fp16, y = const_44_promoted_to_fp16)[name = string("op_2799_cast_fp16")]; bool input_107_interleave_0 = const()[name = string("input_107_interleave_0"), val = bool(false)]; tensor input_107_cast_fp16 = concat(axis = var_2797, interleave = input_107_interleave_0, values = (x_73_cast_fp16, var_2799_cast_fp16))[name = string("input_107_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_2794_to_fp16 = const()[name = string("op_2794_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_2794_to_fp16, x = input_107_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor var_2804_split_sizes_0 = const()[name = string("op_2804_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2804_axis_0 = const()[name = string("op_2804_axis_0"), val = int32(-1)]; tensor var_2804_cast_fp16_0, tensor var_2804_cast_fp16_1 = split(axis = var_2804_axis_0, split_sizes = var_2804_split_sizes_0, x = normed_101_cast_fp16)[name = string("op_2804_cast_fp16")]; tensor layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542811520)))]; tensor h_21_cast_fp16 = mul(x = var_2804_cast_fp16_0, y = layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_21_cast_fp16")]; tensor var_2815 = const()[name = string("op_2815"), val = tensor([0, 2, 1])]; tensor input_109_axes_0 = const()[name = string("input_109_axes_0"), val = tensor([2])]; tensor var_2816 = transpose(perm = var_2815, x = h_21_cast_fp16)[name = string("transpose_148")]; tensor input_109 = expand_dims(axes = input_109_axes_0, x = var_2816)[name = string("input_109")]; string gate_13_pad_type_0 = const()[name = string("gate_13_pad_type_0"), val = string("valid")]; tensor gate_13_strides_0 = const()[name = string("gate_13_strides_0"), val = tensor([1, 1])]; tensor gate_13_pad_0 = const()[name = string("gate_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_13_dilations_0 = const()[name = string("gate_13_dilations_0"), val = tensor([1, 1])]; int32 gate_13_groups_0 = const()[name = string("gate_13_groups_0"), val = int32(1)]; tensor gate_13 = conv(dilations = gate_13_dilations_0, groups = gate_13_groups_0, pad = gate_13_pad_0, pad_type = gate_13_pad_type_0, strides = gate_13_strides_0, weight = layers_3_mlp_gate_proj_weight_palettized, x = input_109)[name = string("gate_13")]; string up_7_pad_type_0 = const()[name = string("up_7_pad_type_0"), val = string("valid")]; tensor up_7_strides_0 = const()[name = string("up_7_strides_0"), val = tensor([1, 1])]; tensor up_7_pad_0 = const()[name = string("up_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_7_dilations_0 = const()[name = string("up_7_dilations_0"), val = tensor([1, 1])]; int32 up_7_groups_0 = const()[name = string("up_7_groups_0"), val = int32(1)]; tensor up_7 = conv(dilations = up_7_dilations_0, groups = up_7_groups_0, pad = up_7_pad_0, pad_type = up_7_pad_type_0, strides = up_7_strides_0, weight = layers_3_mlp_up_proj_weight_palettized, x = input_109)[name = string("up_7")]; string gate_15_mode_0 = const()[name = string("gate_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_15 = gelu(mode = gate_15_mode_0, x = gate_13)[name = string("gate_15")]; tensor input_111 = mul(x = gate_15, y = up_7)[name = string("input_111")]; string mlp_out_7_pad_type_0 = const()[name = string("mlp_out_7_pad_type_0"), val = string("valid")]; tensor mlp_out_7_strides_0 = const()[name = string("mlp_out_7_strides_0"), val = tensor([1, 1])]; tensor mlp_out_7_pad_0 = const()[name = string("mlp_out_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_7_dilations_0 = const()[name = string("mlp_out_7_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_7_groups_0 = const()[name = string("mlp_out_7_groups_0"), val = int32(1)]; tensor mlp_out_7 = conv(dilations = mlp_out_7_dilations_0, groups = mlp_out_7_groups_0, pad = mlp_out_7_pad_0, pad_type = mlp_out_7_pad_type_0, strides = mlp_out_7_strides_0, weight = layers_3_mlp_down_proj_weight_palettized, x = input_111)[name = string("mlp_out_7")]; tensor var_2856_axes_0 = const()[name = string("op_2856_axes_0"), val = tensor([2])]; tensor var_2856 = squeeze(axes = var_2856_axes_0, x = mlp_out_7)[name = string("op_2856")]; tensor var_2860 = const()[name = string("op_2860"), val = tensor([0, 2, 1])]; int32 var_2866 = const()[name = string("op_2866"), val = int32(-1)]; fp16 const_45_promoted = const()[name = string("const_45_promoted"), val = fp16(-0x1p+0)]; tensor x_75 = transpose(perm = var_2860, x = var_2856)[name = string("transpose_147")]; tensor var_2868 = mul(x = x_75, y = const_45_promoted)[name = string("op_2868")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113 = concat(axis = var_2866, interleave = input_113_interleave_0, values = (x_75, var_2868))[name = string("input_113")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_2863_to_fp16 = const()[name = string("op_2863_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_2863_to_fp16, x = input_113)[name = string("normed_105_cast_fp16")]; tensor var_2873_split_sizes_0 = const()[name = string("op_2873_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2873_axis_0 = const()[name = string("op_2873_axis_0"), val = int32(-1)]; tensor var_2873_0, tensor var_2873_1 = split(axis = var_2873_axis_0, split_sizes = var_2873_split_sizes_0, x = normed_105_cast_fp16)[name = string("op_2873")]; tensor hidden_states_33 = mul(x = var_2873_0, y = layers_3_post_feedforward_layernorm_weight)[name = string("hidden_states_33")]; tensor hidden_states_35_cast_fp16 = add(x = x_73_cast_fp16, y = hidden_states_33)[name = string("hidden_states_35_cast_fp16")]; tensor per_layer_slice_7_begin_0 = const()[name = string("per_layer_slice_7_begin_0"), val = tensor([0, 0, 3840])]; tensor per_layer_slice_7_end_0 = const()[name = string("per_layer_slice_7_end_0"), val = tensor([1, 1, 4096])]; tensor per_layer_slice_7_end_mask_0 = const()[name = string("per_layer_slice_7_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_7_cast_fp16 = slice_by_index(begin = per_layer_slice_7_begin_0, end = per_layer_slice_7_end_0, end_mask = per_layer_slice_7_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_7_cast_fp16")]; tensor var_2901 = const()[name = string("op_2901"), val = tensor([0, 2, 1])]; tensor input_115_axes_0 = const()[name = string("input_115_axes_0"), val = tensor([2])]; tensor var_2902 = transpose(perm = var_2901, x = hidden_states_35_cast_fp16)[name = string("transpose_146")]; tensor input_115 = expand_dims(axes = input_115_axes_0, x = var_2902)[name = string("input_115")]; string gated_19_pad_type_0 = const()[name = string("gated_19_pad_type_0"), val = string("valid")]; tensor gated_19_strides_0 = const()[name = string("gated_19_strides_0"), val = tensor([1, 1])]; tensor gated_19_pad_0 = const()[name = string("gated_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_19_dilations_0 = const()[name = string("gated_19_dilations_0"), val = tensor([1, 1])]; int32 gated_19_groups_0 = const()[name = string("gated_19_groups_0"), val = int32(1)]; tensor gated_19 = conv(dilations = gated_19_dilations_0, groups = gated_19_groups_0, pad = gated_19_pad_0, pad_type = gated_19_pad_type_0, strides = gated_19_strides_0, weight = layers_3_per_layer_input_gate_weight_palettized, x = input_115)[name = string("gated_19")]; string gated_21_mode_0 = const()[name = string("gated_21_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_21 = gelu(mode = gated_21_mode_0, x = gated_19)[name = string("gated_21")]; tensor var_2921 = const()[name = string("op_2921"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_7_axes_0 = const()[name = string("per_layer_slice_conv_7_axes_0"), val = tensor([2])]; tensor var_2922_cast_fp16 = transpose(perm = var_2921, x = per_layer_slice_7_cast_fp16)[name = string("transpose_145")]; tensor per_layer_slice_conv_7_cast_fp16 = expand_dims(axes = per_layer_slice_conv_7_axes_0, x = var_2922_cast_fp16)[name = string("per_layer_slice_conv_7_cast_fp16")]; tensor input_117_cast_fp16 = mul(x = gated_21, y = per_layer_slice_conv_7_cast_fp16)[name = string("input_117_cast_fp16")]; string gated_23_pad_type_0 = const()[name = string("gated_23_pad_type_0"), val = string("valid")]; tensor gated_23_strides_0 = const()[name = string("gated_23_strides_0"), val = tensor([1, 1])]; tensor gated_23_pad_0 = const()[name = string("gated_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_23_dilations_0 = const()[name = string("gated_23_dilations_0"), val = tensor([1, 1])]; int32 gated_23_groups_0 = const()[name = string("gated_23_groups_0"), val = int32(1)]; tensor layers_3_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542816704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543144448))))[name = string("layers_3_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_23_cast_fp16 = conv(dilations = gated_23_dilations_0, groups = gated_23_groups_0, pad = gated_23_pad_0, pad_type = gated_23_pad_type_0, strides = gated_23_strides_0, weight = layers_3_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_117_cast_fp16)[name = string("gated_23_cast_fp16")]; tensor var_2938_axes_0 = const()[name = string("op_2938_axes_0"), val = tensor([2])]; tensor var_2938_cast_fp16 = squeeze(axes = var_2938_axes_0, x = gated_23_cast_fp16)[name = string("op_2938_cast_fp16")]; tensor var_2942 = const()[name = string("op_2942"), val = tensor([0, 2, 1])]; int32 var_2948 = const()[name = string("op_2948"), val = int32(-1)]; fp16 const_46_promoted_to_fp16 = const()[name = string("const_46_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_77_cast_fp16 = transpose(perm = var_2942, x = var_2938_cast_fp16)[name = string("transpose_144")]; tensor var_2950_cast_fp16 = mul(x = x_77_cast_fp16, y = const_46_promoted_to_fp16)[name = string("op_2950_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_2948, interleave = input_119_interleave_0, values = (x_77_cast_fp16, var_2950_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_2945_to_fp16 = const()[name = string("op_2945_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_2945_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor var_2955_split_sizes_0 = const()[name = string("op_2955_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2955_axis_0 = const()[name = string("op_2955_axis_0"), val = int32(-1)]; tensor var_2955_cast_fp16_0, tensor var_2955_cast_fp16_1 = split(axis = var_2955_axis_0, split_sizes = var_2955_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_2955_cast_fp16")]; tensor layers_3_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543147072)))]; tensor hidden_states_39_cast_fp16 = mul(x = var_2955_cast_fp16_0, y = layers_3_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_39_cast_fp16")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; tensor const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = tensor([0x1.14p-1])]; tensor x_79_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_47_promoted_to_fp16)[name = string("x_79_cast_fp16")]; tensor var_2967_axes_0 = const()[name = string("op_2967_axes_0"), val = tensor([0])]; tensor var_2967_cast_fp16 = squeeze(axes = var_2967_axes_0, x = K_sliding_out_7_cast_fp16)[name = string("op_2967_cast_fp16")]; tensor var_2969_axes_0 = const()[name = string("op_2969_axes_0"), val = tensor([0])]; tensor var_2969_cast_fp16 = squeeze(axes = var_2969_axes_0, x = V_sliding_out_7_cast_fp16)[name = string("op_2969_cast_fp16")]; tensor var_2972_begin_0 = const()[name = string("op_2972_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2972_end_0 = const()[name = string("op_2972_end_0"), val = tensor([5, 2, 512, 512])]; tensor var_2972_end_mask_0 = const()[name = string("op_2972_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2972_squeeze_mask_0 = const()[name = string("op_2972_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2972_cast_fp16 = slice_by_index(begin = var_2972_begin_0, end = var_2972_end_0, end_mask = var_2972_end_mask_0, squeeze_mask = var_2972_squeeze_mask_0, x = K_sliding_in)[name = string("op_2972_cast_fp16")]; tensor K_sliding_slot_9_axes_0 = const()[name = string("K_sliding_slot_9_axes_0"), val = tensor([0])]; tensor K_sliding_slot_9_cast_fp16 = expand_dims(axes = K_sliding_slot_9_axes_0, x = var_2972_cast_fp16)[name = string("K_sliding_slot_9_cast_fp16")]; tensor var_2977_begin_0 = const()[name = string("op_2977_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2977_end_0 = const()[name = string("op_2977_end_0"), val = tensor([5, 2, 512, 512])]; tensor var_2977_end_mask_0 = const()[name = string("op_2977_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2977_squeeze_mask_0 = const()[name = string("op_2977_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_2977_cast_fp16 = slice_by_index(begin = var_2977_begin_0, end = var_2977_end_0, end_mask = var_2977_end_mask_0, squeeze_mask = var_2977_squeeze_mask_0, x = V_sliding_in)[name = string("op_2977_cast_fp16")]; tensor V_sliding_slot_9_axes_0 = const()[name = string("V_sliding_slot_9_axes_0"), val = tensor([0])]; tensor V_sliding_slot_9_cast_fp16 = expand_dims(axes = V_sliding_slot_9_axes_0, x = var_2977_cast_fp16)[name = string("V_sliding_slot_9_cast_fp16")]; int32 var_2984 = const()[name = string("op_2984"), val = int32(-1)]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2986_cast_fp16 = mul(x = x_79_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2986_cast_fp16")]; bool input_121_interleave_0 = const()[name = string("input_121_interleave_0"), val = bool(false)]; tensor input_121_cast_fp16 = concat(axis = var_2984, interleave = input_121_interleave_0, values = (x_79_cast_fp16, var_2986_cast_fp16))[name = string("input_121_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_2981_to_fp16 = const()[name = string("op_2981_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_2981_to_fp16, x = input_121_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_2991_split_sizes_0 = const()[name = string("op_2991_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2991_axis_0 = const()[name = string("op_2991_axis_0"), val = int32(-1)]; tensor var_2991_cast_fp16_0, tensor var_2991_cast_fp16_1 = split(axis = var_2991_axis_0, split_sizes = var_2991_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_2991_cast_fp16")]; tensor layers_4_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543152256)))]; tensor h_25_cast_fp16 = mul(x = var_2991_cast_fp16_0, y = layers_4_input_layernorm_weight_promoted_to_fp16)[name = string("h_25_cast_fp16")]; tensor var_2997 = const()[name = string("op_2997"), val = tensor([0, 2, 1])]; tensor var_3000_axes_0 = const()[name = string("op_3000_axes_0"), val = tensor([2])]; tensor var_2998_cast_fp16 = transpose(perm = var_2997, x = h_25_cast_fp16)[name = string("transpose_143")]; tensor var_3000_cast_fp16 = expand_dims(axes = var_3000_axes_0, x = var_2998_cast_fp16)[name = string("op_3000_cast_fp16")]; string var_3016_pad_type_0 = const()[name = string("op_3016_pad_type_0"), val = string("valid")]; tensor var_3016_strides_0 = const()[name = string("op_3016_strides_0"), val = tensor([1, 1])]; tensor var_3016_pad_0 = const()[name = string("op_3016_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3016_dilations_0 = const()[name = string("op_3016_dilations_0"), val = tensor([1, 1])]; int32 var_3016_groups_0 = const()[name = string("op_3016_groups_0"), val = int32(1)]; tensor var_3016 = conv(dilations = var_3016_dilations_0, groups = var_3016_groups_0, pad = var_3016_pad_0, pad_type = var_3016_pad_type_0, strides = var_3016_strides_0, weight = layers_4_self_attn_q_proj_weight_palettized, x = var_3000_cast_fp16)[name = string("op_3016")]; tensor var_3021 = const()[name = string("op_3021"), val = tensor([1, 8, 256, 1])]; tensor var_3022 = reshape(shape = var_3021, x = var_3016)[name = string("op_3022")]; tensor var_3027 = const()[name = string("op_3027"), val = tensor([0, 1, 3, 2])]; tensor var_3037 = const()[name = string("op_3037"), val = tensor([1, 8, 256])]; tensor var_3028 = transpose(perm = var_3027, x = var_3022)[name = string("transpose_142")]; tensor x_81 = reshape(shape = var_3037, x = var_3028)[name = string("x_81")]; int32 var_3043 = const()[name = string("op_3043"), val = int32(-1)]; fp16 const_49_promoted = const()[name = string("const_49_promoted"), val = fp16(-0x1p+0)]; tensor var_3045 = mul(x = x_81, y = const_49_promoted)[name = string("op_3045")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125 = concat(axis = var_3043, interleave = input_125_interleave_0, values = (x_81, var_3045))[name = string("input_125")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_3040_to_fp16 = const()[name = string("op_3040_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_3040_to_fp16, x = input_125)[name = string("normed_117_cast_fp16")]; tensor var_3050_split_sizes_0 = const()[name = string("op_3050_split_sizes_0"), val = tensor([256, 256])]; int32 var_3050_axis_0 = const()[name = string("op_3050_axis_0"), val = int32(-1)]; tensor var_3050_0, tensor var_3050_1 = split(axis = var_3050_axis_0, split_sizes = var_3050_split_sizes_0, x = normed_117_cast_fp16)[name = string("op_3050")]; tensor var_3052 = mul(x = var_3050_0, y = layers_4_self_attn_q_norm_weight)[name = string("op_3052")]; tensor var_3057 = const()[name = string("op_3057"), val = tensor([1, 8, 1, 256])]; tensor q_35 = reshape(shape = var_3057, x = var_3052)[name = string("q_35")]; tensor var_3059_cast_fp16 = mul(x = q_35, y = cos_s)[name = string("op_3059_cast_fp16")]; tensor var_3060_split_sizes_0 = const()[name = string("op_3060_split_sizes_0"), val = tensor([128, 128])]; int32 var_3060_axis_0 = const()[name = string("op_3060_axis_0"), val = int32(-1)]; tensor var_3060_0, tensor var_3060_1 = split(axis = var_3060_axis_0, split_sizes = var_3060_split_sizes_0, x = q_35)[name = string("op_3060")]; fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; tensor var_3062 = mul(x = var_3060_1, y = const_50_promoted)[name = string("op_3062")]; int32 var_3064 = const()[name = string("op_3064"), val = int32(-1)]; bool var_3065_interleave_0 = const()[name = string("op_3065_interleave_0"), val = bool(false)]; tensor var_3065 = concat(axis = var_3064, interleave = var_3065_interleave_0, values = (var_3062, var_3060_0))[name = string("op_3065")]; tensor var_3066_cast_fp16 = mul(x = var_3065, y = sin_s)[name = string("op_3066_cast_fp16")]; tensor q_39_cast_fp16 = add(x = var_3059_cast_fp16, y = var_3066_cast_fp16)[name = string("q_39_cast_fp16")]; string var_3079_pad_type_0 = const()[name = string("op_3079_pad_type_0"), val = string("valid")]; tensor var_3079_strides_0 = const()[name = string("op_3079_strides_0"), val = tensor([1, 1])]; tensor var_3079_pad_0 = const()[name = string("op_3079_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3079_dilations_0 = const()[name = string("op_3079_dilations_0"), val = tensor([1, 1])]; int32 var_3079_groups_0 = const()[name = string("op_3079_groups_0"), val = int32(1)]; tensor var_3079 = conv(dilations = var_3079_dilations_0, groups = var_3079_groups_0, pad = var_3079_pad_0, pad_type = var_3079_pad_type_0, strides = var_3079_strides_0, weight = layers_4_self_attn_k_proj_weight_palettized, x = var_3000_cast_fp16)[name = string("op_3079")]; tensor var_3084 = const()[name = string("op_3084"), val = tensor([1, 2, 256, 1])]; tensor var_3085 = reshape(shape = var_3084, x = var_3079)[name = string("op_3085")]; tensor var_3090 = const()[name = string("op_3090"), val = tensor([0, 1, 3, 2])]; string var_3107_pad_type_0 = const()[name = string("op_3107_pad_type_0"), val = string("valid")]; tensor var_3107_strides_0 = const()[name = string("op_3107_strides_0"), val = tensor([1, 1])]; tensor var_3107_pad_0 = const()[name = string("op_3107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3107_dilations_0 = const()[name = string("op_3107_dilations_0"), val = tensor([1, 1])]; int32 var_3107_groups_0 = const()[name = string("op_3107_groups_0"), val = int32(1)]; tensor var_3107 = conv(dilations = var_3107_dilations_0, groups = var_3107_groups_0, pad = var_3107_pad_0, pad_type = var_3107_pad_type_0, strides = var_3107_strides_0, weight = layers_4_self_attn_v_proj_weight_palettized, x = var_3000_cast_fp16)[name = string("op_3107")]; tensor var_3112 = const()[name = string("op_3112"), val = tensor([1, 2, 256, 1])]; tensor var_3113 = reshape(shape = var_3112, x = var_3107)[name = string("op_3113")]; tensor var_3118 = const()[name = string("op_3118"), val = tensor([0, 1, 3, 2])]; tensor var_3128 = const()[name = string("op_3128"), val = tensor([1, 2, 256])]; tensor var_3091 = transpose(perm = var_3090, x = var_3085)[name = string("transpose_141")]; tensor x_83 = reshape(shape = var_3128, x = var_3091)[name = string("x_83")]; int32 var_3134 = const()[name = string("op_3134"), val = int32(-1)]; fp16 const_51_promoted = const()[name = string("const_51_promoted"), val = fp16(-0x1p+0)]; tensor var_3136 = mul(x = x_83, y = const_51_promoted)[name = string("op_3136")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127 = concat(axis = var_3134, interleave = input_127_interleave_0, values = (x_83, var_3136))[name = string("input_127")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_3131_to_fp16 = const()[name = string("op_3131_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_3131_to_fp16, x = input_127)[name = string("normed_121_cast_fp16")]; tensor var_3141_split_sizes_0 = const()[name = string("op_3141_split_sizes_0"), val = tensor([256, 256])]; int32 var_3141_axis_0 = const()[name = string("op_3141_axis_0"), val = int32(-1)]; tensor var_3141_0, tensor var_3141_1 = split(axis = var_3141_axis_0, split_sizes = var_3141_split_sizes_0, x = normed_121_cast_fp16)[name = string("op_3141")]; tensor var_3143 = mul(x = var_3141_0, y = layers_4_self_attn_k_norm_weight)[name = string("op_3143")]; tensor var_3148 = const()[name = string("op_3148"), val = tensor([1, 2, 1, 256])]; tensor q_37 = reshape(shape = var_3148, x = var_3143)[name = string("q_37")]; fp16 var_3150_promoted = const()[name = string("op_3150_promoted"), val = fp16(0x1p+1)]; tensor var_3119 = transpose(perm = var_3118, x = var_3113)[name = string("transpose_140")]; tensor var_3151 = pow(x = var_3119, y = var_3150_promoted)[name = string("op_3151")]; tensor var_3156_axes_0 = const()[name = string("op_3156_axes_0"), val = tensor([-1])]; bool var_3156_keep_dims_0 = const()[name = string("op_3156_keep_dims_0"), val = bool(true)]; tensor var_3156 = reduce_mean(axes = var_3156_axes_0, keep_dims = var_3156_keep_dims_0, x = var_3151)[name = string("op_3156")]; fp16 var_3158_to_fp16 = const()[name = string("op_3158_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_9_cast_fp16 = add(x = var_3156, y = var_3158_to_fp16)[name = string("mean_sq_9_cast_fp16")]; fp32 var_3160_epsilon_0 = const()[name = string("op_3160_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3160_cast_fp16 = rsqrt(epsilon = var_3160_epsilon_0, x = mean_sq_9_cast_fp16)[name = string("op_3160_cast_fp16")]; tensor input_131_cast_fp16 = mul(x = var_3119, y = var_3160_cast_fp16)[name = string("input_131_cast_fp16")]; tensor var_3162_cast_fp16 = mul(x = q_37, y = cos_s)[name = string("op_3162_cast_fp16")]; tensor var_3163_split_sizes_0 = const()[name = string("op_3163_split_sizes_0"), val = tensor([128, 128])]; int32 var_3163_axis_0 = const()[name = string("op_3163_axis_0"), val = int32(-1)]; tensor var_3163_0, tensor var_3163_1 = split(axis = var_3163_axis_0, split_sizes = var_3163_split_sizes_0, x = q_37)[name = string("op_3163")]; fp16 const_52_promoted = const()[name = string("const_52_promoted"), val = fp16(-0x1p+0)]; tensor var_3165 = mul(x = var_3163_1, y = const_52_promoted)[name = string("op_3165")]; int32 var_3167 = const()[name = string("op_3167"), val = int32(-1)]; bool var_3168_interleave_0 = const()[name = string("op_3168_interleave_0"), val = bool(false)]; tensor var_3168 = concat(axis = var_3167, interleave = var_3168_interleave_0, values = (var_3165, var_3163_0))[name = string("op_3168")]; tensor var_3169_cast_fp16 = mul(x = var_3168, y = sin_s)[name = string("op_3169_cast_fp16")]; tensor input_129_cast_fp16 = add(x = var_3162_cast_fp16, y = var_3169_cast_fp16)[name = string("input_129_cast_fp16")]; tensor k_padded_9_pad_0 = const()[name = string("k_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_9_mode_0 = const()[name = string("k_padded_9_mode_0"), val = string("constant")]; fp16 const_53_to_fp16 = const()[name = string("const_53_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_9_cast_fp16 = pad(constant_val = const_53_to_fp16, mode = k_padded_9_mode_0, pad = k_padded_9_pad_0, x = input_129_cast_fp16)[name = string("k_padded_9_cast_fp16")]; tensor v_padded_9_pad_0 = const()[name = string("v_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_9_mode_0 = const()[name = string("v_padded_9_mode_0"), val = string("constant")]; fp16 const_54_to_fp16 = const()[name = string("const_54_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_9_cast_fp16 = pad(constant_val = const_54_to_fp16, mode = v_padded_9_mode_0, pad = v_padded_9_pad_0, x = input_131_cast_fp16)[name = string("v_padded_9_cast_fp16")]; tensor var_3198_begin_0 = const()[name = string("op_3198_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_3198_end_0 = const()[name = string("op_3198_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_3198_end_mask_0 = const()[name = string("op_3198_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3198_cast_fp16 = slice_by_index(begin = var_3198_begin_0, end = var_3198_end_0, end_mask = var_3198_end_mask_0, x = K_sliding_slot_9_cast_fp16)[name = string("op_3198_cast_fp16")]; int32 var_3205 = const()[name = string("op_3205"), val = int32(2)]; bool K_sliding_out_9_interleave_0 = const()[name = string("K_sliding_out_9_interleave_0"), val = bool(false)]; tensor K_sliding_out_9_cast_fp16 = concat(axis = var_3205, interleave = K_sliding_out_9_interleave_0, values = (var_3198_cast_fp16, k_padded_9_cast_fp16))[name = string("K_sliding_out_9_cast_fp16")]; tensor var_3221_begin_0 = const()[name = string("op_3221_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_3221_end_0 = const()[name = string("op_3221_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_3221_end_mask_0 = const()[name = string("op_3221_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3221_cast_fp16 = slice_by_index(begin = var_3221_begin_0, end = var_3221_end_0, end_mask = var_3221_end_mask_0, x = V_sliding_slot_9_cast_fp16)[name = string("op_3221_cast_fp16")]; int32 var_3228 = const()[name = string("op_3228"), val = int32(2)]; bool V_sliding_out_9_interleave_0 = const()[name = string("V_sliding_out_9_interleave_0"), val = bool(false)]; tensor V_sliding_out_9_cast_fp16 = concat(axis = var_3228, interleave = V_sliding_out_9_interleave_0, values = (var_3221_cast_fp16, v_padded_9_cast_fp16))[name = string("V_sliding_out_9_cast_fp16")]; tensor K_for_attn_9_begin_0 = const()[name = string("K_for_attn_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_9_end_0 = const()[name = string("K_for_attn_9_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_9_end_mask_0 = const()[name = string("K_for_attn_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_9_cast_fp16 = slice_by_index(begin = K_for_attn_9_begin_0, end = K_for_attn_9_end_0, end_mask = K_for_attn_9_end_mask_0, x = K_sliding_out_9_cast_fp16)[name = string("K_for_attn_9_cast_fp16")]; tensor V_for_attn_9_begin_0 = const()[name = string("V_for_attn_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_9_end_0 = const()[name = string("V_for_attn_9_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_9_end_mask_0 = const()[name = string("V_for_attn_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_9_cast_fp16 = slice_by_index(begin = V_for_attn_9_begin_0, end = V_for_attn_9_end_0, end_mask = V_for_attn_9_end_mask_0, x = V_sliding_out_9_cast_fp16)[name = string("V_for_attn_9_cast_fp16")]; tensor transpose_16_perm_0 = const()[name = string("transpose_16_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_8_reps_0 = const()[name = string("tile_8_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_16_cast_fp16 = transpose(perm = transpose_16_perm_0, x = K_for_attn_9_cast_fp16)[name = string("transpose_139")]; tensor tile_8_cast_fp16 = tile(reps = tile_8_reps_0, x = transpose_16_cast_fp16)[name = string("tile_8_cast_fp16")]; tensor concat_16 = const()[name = string("concat_16"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_16, x = tile_8_cast_fp16)[name = string("reshape_16_cast_fp16")]; tensor transpose_17_perm_0 = const()[name = string("transpose_17_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_17 = const()[name = string("concat_17"), val = tensor([-1, 1, 512, 256])]; tensor transpose_17_cast_fp16 = transpose(perm = transpose_17_perm_0, x = reshape_16_cast_fp16)[name = string("transpose_138")]; tensor reshape_17_cast_fp16 = reshape(shape = concat_17, x = transpose_17_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor transpose_52_perm_0 = const()[name = string("transpose_52_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_18_perm_0 = const()[name = string("transpose_18_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_9_reps_0 = const()[name = string("tile_9_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_18_cast_fp16 = transpose(perm = transpose_18_perm_0, x = V_for_attn_9_cast_fp16)[name = string("transpose_137")]; tensor tile_9_cast_fp16 = tile(reps = tile_9_reps_0, x = transpose_18_cast_fp16)[name = string("tile_9_cast_fp16")]; tensor concat_18 = const()[name = string("concat_18"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_18, x = tile_9_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor transpose_19_perm_0 = const()[name = string("transpose_19_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_19 = const()[name = string("concat_19"), val = tensor([-1, 1, 512, 256])]; tensor transpose_19_cast_fp16 = transpose(perm = transpose_19_perm_0, x = reshape_18_cast_fp16)[name = string("transpose_136")]; tensor reshape_19_cast_fp16 = reshape(shape = concat_19, x = transpose_19_cast_fp16)[name = string("reshape_19_cast_fp16")]; tensor V_expanded_9_perm_0 = const()[name = string("V_expanded_9_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor transpose_52_cast_fp16 = transpose(perm = transpose_52_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_135")]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = q_39_cast_fp16, y = transpose_52_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; tensor x_87_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask_sliding)[name = string("x_87_cast_fp16")]; tensor reduce_max_4_axes_0 = const()[name = string("reduce_max_4_axes_0"), val = tensor([-1])]; bool reduce_max_4_keep_dims_0 = const()[name = string("reduce_max_4_keep_dims_0"), val = bool(true)]; tensor reduce_max_4 = reduce_max(axes = reduce_max_4_axes_0, keep_dims = reduce_max_4_keep_dims_0, x = x_87_cast_fp16)[name = string("reduce_max_4")]; tensor var_3269 = sub(x = x_87_cast_fp16, y = reduce_max_4)[name = string("op_3269")]; tensor var_3275 = exp(x = var_3269)[name = string("op_3275")]; tensor var_3285_axes_0 = const()[name = string("op_3285_axes_0"), val = tensor([-1])]; bool var_3285_keep_dims_0 = const()[name = string("op_3285_keep_dims_0"), val = bool(true)]; tensor var_3285 = reduce_sum(axes = var_3285_axes_0, keep_dims = var_3285_keep_dims_0, x = var_3275)[name = string("op_3285")]; tensor var_3291_cast_fp16 = real_div(x = var_3275, y = var_3285)[name = string("op_3291_cast_fp16")]; bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; tensor V_expanded_9_cast_fp16 = transpose(perm = V_expanded_9_perm_0, x = reshape_19_cast_fp16)[name = string("transpose_134")]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = var_3291_cast_fp16, y = V_expanded_9_cast_fp16)[name = string("attn_output_25_cast_fp16")]; tensor var_3302 = const()[name = string("op_3302"), val = tensor([0, 2, 1, 3])]; tensor var_3309 = const()[name = string("op_3309"), val = tensor([1, 1, -1])]; tensor var_3303_cast_fp16 = transpose(perm = var_3302, x = attn_output_25_cast_fp16)[name = string("transpose_133")]; tensor attn_output_27_cast_fp16 = reshape(shape = var_3309, x = var_3303_cast_fp16)[name = string("attn_output_27_cast_fp16")]; tensor var_3314 = const()[name = string("op_3314"), val = tensor([0, 2, 1])]; string var_3330_pad_type_0 = const()[name = string("op_3330_pad_type_0"), val = string("valid")]; int32 var_3330_groups_0 = const()[name = string("op_3330_groups_0"), val = int32(1)]; tensor var_3330_strides_0 = const()[name = string("op_3330_strides_0"), val = tensor([1])]; tensor var_3330_pad_0 = const()[name = string("op_3330_pad_0"), val = tensor([0, 0])]; tensor var_3330_dilations_0 = const()[name = string("op_3330_dilations_0"), val = tensor([1])]; tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543157440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545778944))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3315_cast_fp16 = transpose(perm = var_3314, x = attn_output_27_cast_fp16)[name = string("transpose_132")]; tensor var_3330_cast_fp16 = conv(dilations = var_3330_dilations_0, groups = var_3330_groups_0, pad = var_3330_pad_0, pad_type = var_3330_pad_type_0, strides = var_3330_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3315_cast_fp16)[name = string("op_3330_cast_fp16")]; tensor var_3334 = const()[name = string("op_3334"), val = tensor([0, 2, 1])]; int32 var_3340 = const()[name = string("op_3340"), val = int32(-1)]; fp16 const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_91_cast_fp16 = transpose(perm = var_3334, x = var_3330_cast_fp16)[name = string("transpose_131")]; tensor var_3342_cast_fp16 = mul(x = x_91_cast_fp16, y = const_55_promoted_to_fp16)[name = string("op_3342_cast_fp16")]; bool input_135_interleave_0 = const()[name = string("input_135_interleave_0"), val = bool(false)]; tensor input_135_cast_fp16 = concat(axis = var_3340, interleave = input_135_interleave_0, values = (x_91_cast_fp16, var_3342_cast_fp16))[name = string("input_135_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_3337_to_fp16 = const()[name = string("op_3337_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_3337_to_fp16, x = input_135_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor var_3347_split_sizes_0 = const()[name = string("op_3347_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3347_axis_0 = const()[name = string("op_3347_axis_0"), val = int32(-1)]; tensor var_3347_cast_fp16_0, tensor var_3347_cast_fp16_1 = split(axis = var_3347_axis_0, split_sizes = var_3347_split_sizes_0, x = normed_125_cast_fp16)[name = string("op_3347_cast_fp16")]; tensor layers_4_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545781568)))]; tensor attn_output_29_cast_fp16 = mul(x = var_3347_cast_fp16_0, y = layers_4_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_29_cast_fp16")]; tensor x_93_cast_fp16 = add(x = x_79_cast_fp16, y = attn_output_29_cast_fp16)[name = string("x_93_cast_fp16")]; int32 var_3356 = const()[name = string("op_3356"), val = int32(-1)]; fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3358_cast_fp16 = mul(x = x_93_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_3358_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_3356, interleave = input_137_interleave_0, values = (x_93_cast_fp16, var_3358_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_3353_to_fp16 = const()[name = string("op_3353_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_3353_to_fp16, x = input_137_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor var_3363_split_sizes_0 = const()[name = string("op_3363_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3363_axis_0 = const()[name = string("op_3363_axis_0"), val = int32(-1)]; tensor var_3363_cast_fp16_0, tensor var_3363_cast_fp16_1 = split(axis = var_3363_axis_0, split_sizes = var_3363_split_sizes_0, x = normed_129_cast_fp16)[name = string("op_3363_cast_fp16")]; tensor layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545786752)))]; tensor h_27_cast_fp16 = mul(x = var_3363_cast_fp16_0, y = layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_27_cast_fp16")]; tensor var_3374 = const()[name = string("op_3374"), val = tensor([0, 2, 1])]; tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; tensor var_3375 = transpose(perm = var_3374, x = h_27_cast_fp16)[name = string("transpose_130")]; tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_3375)[name = string("input_139")]; string gate_17_pad_type_0 = const()[name = string("gate_17_pad_type_0"), val = string("valid")]; tensor gate_17_strides_0 = const()[name = string("gate_17_strides_0"), val = tensor([1, 1])]; tensor gate_17_pad_0 = const()[name = string("gate_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_17_dilations_0 = const()[name = string("gate_17_dilations_0"), val = tensor([1, 1])]; int32 gate_17_groups_0 = const()[name = string("gate_17_groups_0"), val = int32(1)]; tensor gate_17 = conv(dilations = gate_17_dilations_0, groups = gate_17_groups_0, pad = gate_17_pad_0, pad_type = gate_17_pad_type_0, strides = gate_17_strides_0, weight = layers_4_mlp_gate_proj_weight_palettized, x = input_139)[name = string("gate_17")]; string up_9_pad_type_0 = const()[name = string("up_9_pad_type_0"), val = string("valid")]; tensor up_9_strides_0 = const()[name = string("up_9_strides_0"), val = tensor([1, 1])]; tensor up_9_pad_0 = const()[name = string("up_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_9_dilations_0 = const()[name = string("up_9_dilations_0"), val = tensor([1, 1])]; int32 up_9_groups_0 = const()[name = string("up_9_groups_0"), val = int32(1)]; tensor up_9 = conv(dilations = up_9_dilations_0, groups = up_9_groups_0, pad = up_9_pad_0, pad_type = up_9_pad_type_0, strides = up_9_strides_0, weight = layers_4_mlp_up_proj_weight_palettized, x = input_139)[name = string("up_9")]; string gate_19_mode_0 = const()[name = string("gate_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_19 = gelu(mode = gate_19_mode_0, x = gate_17)[name = string("gate_19")]; tensor input_141 = mul(x = gate_19, y = up_9)[name = string("input_141")]; string mlp_out_9_pad_type_0 = const()[name = string("mlp_out_9_pad_type_0"), val = string("valid")]; tensor mlp_out_9_strides_0 = const()[name = string("mlp_out_9_strides_0"), val = tensor([1, 1])]; tensor mlp_out_9_pad_0 = const()[name = string("mlp_out_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_9_dilations_0 = const()[name = string("mlp_out_9_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_9_groups_0 = const()[name = string("mlp_out_9_groups_0"), val = int32(1)]; tensor mlp_out_9 = conv(dilations = mlp_out_9_dilations_0, groups = mlp_out_9_groups_0, pad = mlp_out_9_pad_0, pad_type = mlp_out_9_pad_type_0, strides = mlp_out_9_strides_0, weight = layers_4_mlp_down_proj_weight_palettized, x = input_141)[name = string("mlp_out_9")]; tensor var_3415_axes_0 = const()[name = string("op_3415_axes_0"), val = tensor([2])]; tensor var_3415 = squeeze(axes = var_3415_axes_0, x = mlp_out_9)[name = string("op_3415")]; tensor var_3419 = const()[name = string("op_3419"), val = tensor([0, 2, 1])]; int32 var_3425 = const()[name = string("op_3425"), val = int32(-1)]; fp16 const_57_promoted = const()[name = string("const_57_promoted"), val = fp16(-0x1p+0)]; tensor x_95 = transpose(perm = var_3419, x = var_3415)[name = string("transpose_129")]; tensor var_3427 = mul(x = x_95, y = const_57_promoted)[name = string("op_3427")]; bool input_143_interleave_0 = const()[name = string("input_143_interleave_0"), val = bool(false)]; tensor input_143 = concat(axis = var_3425, interleave = input_143_interleave_0, values = (x_95, var_3427))[name = string("input_143")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_3422_to_fp16, x = input_143)[name = string("normed_133_cast_fp16")]; tensor var_3432_split_sizes_0 = const()[name = string("op_3432_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3432_axis_0 = const()[name = string("op_3432_axis_0"), val = int32(-1)]; tensor var_3432_0, tensor var_3432_1 = split(axis = var_3432_axis_0, split_sizes = var_3432_split_sizes_0, x = normed_133_cast_fp16)[name = string("op_3432")]; tensor hidden_states_43 = mul(x = var_3432_0, y = layers_4_post_feedforward_layernorm_weight)[name = string("hidden_states_43")]; tensor hidden_states_45_cast_fp16 = add(x = x_93_cast_fp16, y = hidden_states_43)[name = string("hidden_states_45_cast_fp16")]; tensor per_layer_slice_9_begin_0 = const()[name = string("per_layer_slice_9_begin_0"), val = tensor([0, 0, 4096])]; tensor per_layer_slice_9_end_0 = const()[name = string("per_layer_slice_9_end_0"), val = tensor([1, 1, 4352])]; tensor per_layer_slice_9_end_mask_0 = const()[name = string("per_layer_slice_9_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_9_cast_fp16 = slice_by_index(begin = per_layer_slice_9_begin_0, end = per_layer_slice_9_end_0, end_mask = per_layer_slice_9_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_9_cast_fp16")]; tensor var_3460 = const()[name = string("op_3460"), val = tensor([0, 2, 1])]; tensor input_145_axes_0 = const()[name = string("input_145_axes_0"), val = tensor([2])]; tensor var_3461 = transpose(perm = var_3460, x = hidden_states_45_cast_fp16)[name = string("transpose_128")]; tensor input_145 = expand_dims(axes = input_145_axes_0, x = var_3461)[name = string("input_145")]; string gated_25_pad_type_0 = const()[name = string("gated_25_pad_type_0"), val = string("valid")]; tensor gated_25_strides_0 = const()[name = string("gated_25_strides_0"), val = tensor([1, 1])]; tensor gated_25_pad_0 = const()[name = string("gated_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_25_dilations_0 = const()[name = string("gated_25_dilations_0"), val = tensor([1, 1])]; int32 gated_25_groups_0 = const()[name = string("gated_25_groups_0"), val = int32(1)]; tensor gated_25 = conv(dilations = gated_25_dilations_0, groups = gated_25_groups_0, pad = gated_25_pad_0, pad_type = gated_25_pad_type_0, strides = gated_25_strides_0, weight = layers_4_per_layer_input_gate_weight_palettized, x = input_145)[name = string("gated_25")]; string gated_27_mode_0 = const()[name = string("gated_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_27 = gelu(mode = gated_27_mode_0, x = gated_25)[name = string("gated_27")]; tensor var_3480 = const()[name = string("op_3480"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_9_axes_0 = const()[name = string("per_layer_slice_conv_9_axes_0"), val = tensor([2])]; tensor var_3481_cast_fp16 = transpose(perm = var_3480, x = per_layer_slice_9_cast_fp16)[name = string("transpose_127")]; tensor per_layer_slice_conv_9_cast_fp16 = expand_dims(axes = per_layer_slice_conv_9_axes_0, x = var_3481_cast_fp16)[name = string("per_layer_slice_conv_9_cast_fp16")]; tensor input_147_cast_fp16 = mul(x = gated_27, y = per_layer_slice_conv_9_cast_fp16)[name = string("input_147_cast_fp16")]; string gated_29_pad_type_0 = const()[name = string("gated_29_pad_type_0"), val = string("valid")]; tensor gated_29_strides_0 = const()[name = string("gated_29_strides_0"), val = tensor([1, 1])]; tensor gated_29_pad_0 = const()[name = string("gated_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_29_dilations_0 = const()[name = string("gated_29_dilations_0"), val = tensor([1, 1])]; int32 gated_29_groups_0 = const()[name = string("gated_29_groups_0"), val = int32(1)]; tensor layers_4_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545791936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546119680))))[name = string("layers_4_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_29_cast_fp16 = conv(dilations = gated_29_dilations_0, groups = gated_29_groups_0, pad = gated_29_pad_0, pad_type = gated_29_pad_type_0, strides = gated_29_strides_0, weight = layers_4_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_147_cast_fp16)[name = string("gated_29_cast_fp16")]; tensor var_3497_axes_0 = const()[name = string("op_3497_axes_0"), val = tensor([2])]; tensor var_3497_cast_fp16 = squeeze(axes = var_3497_axes_0, x = gated_29_cast_fp16)[name = string("op_3497_cast_fp16")]; tensor var_3501 = const()[name = string("op_3501"), val = tensor([0, 2, 1])]; int32 var_3507 = const()[name = string("op_3507"), val = int32(-1)]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_97_cast_fp16 = transpose(perm = var_3501, x = var_3497_cast_fp16)[name = string("transpose_126")]; tensor var_3509_cast_fp16 = mul(x = x_97_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_3509_cast_fp16")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149_cast_fp16 = concat(axis = var_3507, interleave = input_149_interleave_0, values = (x_97_cast_fp16, var_3509_cast_fp16))[name = string("input_149_cast_fp16")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_3504_to_fp16 = const()[name = string("op_3504_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_3504_to_fp16, x = input_149_cast_fp16)[name = string("normed_137_cast_fp16")]; tensor var_3514_split_sizes_0 = const()[name = string("op_3514_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3514_axis_0 = const()[name = string("op_3514_axis_0"), val = int32(-1)]; tensor var_3514_cast_fp16_0, tensor var_3514_cast_fp16_1 = split(axis = var_3514_axis_0, split_sizes = var_3514_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_3514_cast_fp16")]; tensor layers_4_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546122304)))]; tensor hidden_states_49_cast_fp16 = mul(x = var_3514_cast_fp16_0, y = layers_4_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_49_cast_fp16")]; tensor hidden_states_51_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_49_cast_fp16)[name = string("hidden_states_51_cast_fp16")]; tensor const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = tensor([0x1.46p-1])]; tensor x_99_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_59_promoted_to_fp16)[name = string("x_99_cast_fp16")]; tensor var_3526_axes_0 = const()[name = string("op_3526_axes_0"), val = tensor([0])]; tensor var_3526_cast_fp16 = squeeze(axes = var_3526_axes_0, x = K_sliding_out_9_cast_fp16)[name = string("op_3526_cast_fp16")]; tensor var_3528_axes_0 = const()[name = string("op_3528_axes_0"), val = tensor([0])]; tensor var_3528_cast_fp16 = squeeze(axes = var_3528_axes_0, x = V_sliding_out_9_cast_fp16)[name = string("op_3528_cast_fp16")]; tensor var_3531_begin_0 = const()[name = string("op_3531_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3531_end_0 = const()[name = string("op_3531_end_0"), val = tensor([1, 2, 2048, 512])]; tensor var_3531_end_mask_0 = const()[name = string("op_3531_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3531_squeeze_mask_0 = const()[name = string("op_3531_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_3531_cast_fp16 = slice_by_index(begin = var_3531_begin_0, end = var_3531_end_0, end_mask = var_3531_end_mask_0, squeeze_mask = var_3531_squeeze_mask_0, x = K_full_in)[name = string("op_3531_cast_fp16")]; tensor K_full_slot_1_axes_0 = const()[name = string("K_full_slot_1_axes_0"), val = tensor([0])]; tensor K_full_slot_1_cast_fp16 = expand_dims(axes = K_full_slot_1_axes_0, x = var_3531_cast_fp16)[name = string("K_full_slot_1_cast_fp16")]; tensor var_3536_begin_0 = const()[name = string("op_3536_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3536_end_0 = const()[name = string("op_3536_end_0"), val = tensor([1, 2, 2048, 512])]; tensor var_3536_end_mask_0 = const()[name = string("op_3536_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3536_squeeze_mask_0 = const()[name = string("op_3536_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_3536_cast_fp16 = slice_by_index(begin = var_3536_begin_0, end = var_3536_end_0, end_mask = var_3536_end_mask_0, squeeze_mask = var_3536_squeeze_mask_0, x = V_full_in)[name = string("op_3536_cast_fp16")]; tensor V_full_slot_1_axes_0 = const()[name = string("V_full_slot_1_axes_0"), val = tensor([0])]; tensor V_full_slot_1_cast_fp16 = expand_dims(axes = V_full_slot_1_axes_0, x = var_3536_cast_fp16)[name = string("V_full_slot_1_cast_fp16")]; int32 var_3543 = const()[name = string("op_3543"), val = int32(-1)]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3545_cast_fp16 = mul(x = x_99_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_3545_cast_fp16")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151_cast_fp16 = concat(axis = var_3543, interleave = input_151_interleave_0, values = (x_99_cast_fp16, var_3545_cast_fp16))[name = string("input_151_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_3540_to_fp16 = const()[name = string("op_3540_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_3540_to_fp16, x = input_151_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_3550_split_sizes_0 = const()[name = string("op_3550_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3550_axis_0 = const()[name = string("op_3550_axis_0"), val = int32(-1)]; tensor var_3550_cast_fp16_0, tensor var_3550_cast_fp16_1 = split(axis = var_3550_axis_0, split_sizes = var_3550_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_3550_cast_fp16")]; tensor layers_5_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546127488)))]; tensor h_31_cast_fp16 = mul(x = var_3550_cast_fp16_0, y = layers_5_input_layernorm_weight_promoted_to_fp16)[name = string("h_31_cast_fp16")]; tensor var_3556 = const()[name = string("op_3556"), val = tensor([0, 2, 1])]; tensor var_3559_axes_0 = const()[name = string("op_3559_axes_0"), val = tensor([2])]; tensor var_3557_cast_fp16 = transpose(perm = var_3556, x = h_31_cast_fp16)[name = string("transpose_125")]; tensor var_3559_cast_fp16 = expand_dims(axes = var_3559_axes_0, x = var_3557_cast_fp16)[name = string("op_3559_cast_fp16")]; string var_3575_pad_type_0 = const()[name = string("op_3575_pad_type_0"), val = string("valid")]; tensor var_3575_strides_0 = const()[name = string("op_3575_strides_0"), val = tensor([1, 1])]; tensor var_3575_pad_0 = const()[name = string("op_3575_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3575_dilations_0 = const()[name = string("op_3575_dilations_0"), val = tensor([1, 1])]; int32 var_3575_groups_0 = const()[name = string("op_3575_groups_0"), val = int32(1)]; tensor var_3575 = conv(dilations = var_3575_dilations_0, groups = var_3575_groups_0, pad = var_3575_pad_0, pad_type = var_3575_pad_type_0, strides = var_3575_strides_0, weight = layers_5_self_attn_q_proj_weight_palettized, x = var_3559_cast_fp16)[name = string("op_3575")]; tensor var_3580 = const()[name = string("op_3580"), val = tensor([1, 8, 512, 1])]; tensor var_3581 = reshape(shape = var_3580, x = var_3575)[name = string("op_3581")]; tensor var_3586 = const()[name = string("op_3586"), val = tensor([0, 1, 3, 2])]; tensor var_3596 = const()[name = string("op_3596"), val = tensor([1, 8, 512])]; tensor var_3587 = transpose(perm = var_3586, x = var_3581)[name = string("transpose_124")]; tensor x_101 = reshape(shape = var_3596, x = var_3587)[name = string("x_101")]; int32 var_3602 = const()[name = string("op_3602"), val = int32(-1)]; fp16 const_61_promoted = const()[name = string("const_61_promoted"), val = fp16(-0x1p+0)]; tensor var_3604 = mul(x = x_101, y = const_61_promoted)[name = string("op_3604")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155 = concat(axis = var_3602, interleave = input_155_interleave_0, values = (x_101, var_3604))[name = string("input_155")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_3599_to_fp16 = const()[name = string("op_3599_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_3599_to_fp16, x = input_155)[name = string("normed_145_cast_fp16")]; tensor var_3609_split_sizes_0 = const()[name = string("op_3609_split_sizes_0"), val = tensor([512, 512])]; int32 var_3609_axis_0 = const()[name = string("op_3609_axis_0"), val = int32(-1)]; tensor var_3609_0, tensor var_3609_1 = split(axis = var_3609_axis_0, split_sizes = var_3609_split_sizes_0, x = normed_145_cast_fp16)[name = string("op_3609")]; tensor var_3611 = mul(x = var_3609_0, y = layers_5_self_attn_q_norm_weight)[name = string("op_3611")]; tensor var_3616 = const()[name = string("op_3616"), val = tensor([1, 8, 1, 512])]; tensor q_43 = reshape(shape = var_3616, x = var_3611)[name = string("q_43")]; tensor var_3618_cast_fp16 = mul(x = q_43, y = cos_f)[name = string("op_3618_cast_fp16")]; tensor var_3619_split_sizes_0 = const()[name = string("op_3619_split_sizes_0"), val = tensor([256, 256])]; int32 var_3619_axis_0 = const()[name = string("op_3619_axis_0"), val = int32(-1)]; tensor var_3619_0, tensor var_3619_1 = split(axis = var_3619_axis_0, split_sizes = var_3619_split_sizes_0, x = q_43)[name = string("op_3619")]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_3621 = mul(x = var_3619_1, y = const_62_promoted)[name = string("op_3621")]; int32 var_3623 = const()[name = string("op_3623"), val = int32(-1)]; bool var_3624_interleave_0 = const()[name = string("op_3624_interleave_0"), val = bool(false)]; tensor var_3624 = concat(axis = var_3623, interleave = var_3624_interleave_0, values = (var_3621, var_3619_0))[name = string("op_3624")]; tensor var_3625_cast_fp16 = mul(x = var_3624, y = sin_f)[name = string("op_3625_cast_fp16")]; tensor q_47_cast_fp16 = add(x = var_3618_cast_fp16, y = var_3625_cast_fp16)[name = string("q_47_cast_fp16")]; string var_3638_pad_type_0 = const()[name = string("op_3638_pad_type_0"), val = string("valid")]; tensor var_3638_strides_0 = const()[name = string("op_3638_strides_0"), val = tensor([1, 1])]; tensor var_3638_pad_0 = const()[name = string("op_3638_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3638_dilations_0 = const()[name = string("op_3638_dilations_0"), val = tensor([1, 1])]; int32 var_3638_groups_0 = const()[name = string("op_3638_groups_0"), val = int32(1)]; tensor var_3638 = conv(dilations = var_3638_dilations_0, groups = var_3638_groups_0, pad = var_3638_pad_0, pad_type = var_3638_pad_type_0, strides = var_3638_strides_0, weight = layers_5_self_attn_k_proj_weight_palettized, x = var_3559_cast_fp16)[name = string("op_3638")]; tensor var_3643 = const()[name = string("op_3643"), val = tensor([1, 2, 512, 1])]; tensor var_3644 = reshape(shape = var_3643, x = var_3638)[name = string("op_3644")]; tensor var_3649 = const()[name = string("op_3649"), val = tensor([0, 1, 3, 2])]; string var_3666_pad_type_0 = const()[name = string("op_3666_pad_type_0"), val = string("valid")]; tensor var_3666_strides_0 = const()[name = string("op_3666_strides_0"), val = tensor([1, 1])]; tensor var_3666_pad_0 = const()[name = string("op_3666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3666_dilations_0 = const()[name = string("op_3666_dilations_0"), val = tensor([1, 1])]; int32 var_3666_groups_0 = const()[name = string("op_3666_groups_0"), val = int32(1)]; tensor var_3666 = conv(dilations = var_3666_dilations_0, groups = var_3666_groups_0, pad = var_3666_pad_0, pad_type = var_3666_pad_type_0, strides = var_3666_strides_0, weight = layers_5_self_attn_v_proj_weight_palettized, x = var_3559_cast_fp16)[name = string("op_3666")]; tensor var_3671 = const()[name = string("op_3671"), val = tensor([1, 2, 512, 1])]; tensor var_3672 = reshape(shape = var_3671, x = var_3666)[name = string("op_3672")]; tensor var_3677 = const()[name = string("op_3677"), val = tensor([0, 1, 3, 2])]; tensor var_3687 = const()[name = string("op_3687"), val = tensor([1, 2, 512])]; tensor var_3650 = transpose(perm = var_3649, x = var_3644)[name = string("transpose_123")]; tensor x_103 = reshape(shape = var_3687, x = var_3650)[name = string("x_103")]; int32 var_3693 = const()[name = string("op_3693"), val = int32(-1)]; fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; tensor var_3695 = mul(x = x_103, y = const_63_promoted)[name = string("op_3695")]; bool input_157_interleave_0 = const()[name = string("input_157_interleave_0"), val = bool(false)]; tensor input_157 = concat(axis = var_3693, interleave = input_157_interleave_0, values = (x_103, var_3695))[name = string("input_157")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_3690_to_fp16 = const()[name = string("op_3690_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_3690_to_fp16, x = input_157)[name = string("normed_149_cast_fp16")]; tensor var_3700_split_sizes_0 = const()[name = string("op_3700_split_sizes_0"), val = tensor([512, 512])]; int32 var_3700_axis_0 = const()[name = string("op_3700_axis_0"), val = int32(-1)]; tensor var_3700_0, tensor var_3700_1 = split(axis = var_3700_axis_0, split_sizes = var_3700_split_sizes_0, x = normed_149_cast_fp16)[name = string("op_3700")]; tensor var_3702 = mul(x = var_3700_0, y = layers_5_self_attn_k_norm_weight)[name = string("op_3702")]; tensor var_3707 = const()[name = string("op_3707"), val = tensor([1, 2, 1, 512])]; tensor q_45 = reshape(shape = var_3707, x = var_3702)[name = string("q_45")]; fp16 var_3709_promoted = const()[name = string("op_3709_promoted"), val = fp16(0x1p+1)]; tensor var_3678 = transpose(perm = var_3677, x = var_3672)[name = string("transpose_122")]; tensor var_3710 = pow(x = var_3678, y = var_3709_promoted)[name = string("op_3710")]; tensor var_3715_axes_0 = const()[name = string("op_3715_axes_0"), val = tensor([-1])]; bool var_3715_keep_dims_0 = const()[name = string("op_3715_keep_dims_0"), val = bool(true)]; tensor var_3715 = reduce_mean(axes = var_3715_axes_0, keep_dims = var_3715_keep_dims_0, x = var_3710)[name = string("op_3715")]; fp16 var_3717_to_fp16 = const()[name = string("op_3717_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_11_cast_fp16 = add(x = var_3715, y = var_3717_to_fp16)[name = string("mean_sq_11_cast_fp16")]; fp32 var_3719_epsilon_0 = const()[name = string("op_3719_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3719_cast_fp16 = rsqrt(epsilon = var_3719_epsilon_0, x = mean_sq_11_cast_fp16)[name = string("op_3719_cast_fp16")]; tensor v_1_cast_fp16 = mul(x = var_3678, y = var_3719_cast_fp16)[name = string("v_1_cast_fp16")]; tensor var_3721_cast_fp16 = mul(x = q_45, y = cos_f)[name = string("op_3721_cast_fp16")]; tensor var_3722_split_sizes_0 = const()[name = string("op_3722_split_sizes_0"), val = tensor([256, 256])]; int32 var_3722_axis_0 = const()[name = string("op_3722_axis_0"), val = int32(-1)]; tensor var_3722_0, tensor var_3722_1 = split(axis = var_3722_axis_0, split_sizes = var_3722_split_sizes_0, x = q_45)[name = string("op_3722")]; fp16 const_64_promoted = const()[name = string("const_64_promoted"), val = fp16(-0x1p+0)]; tensor var_3724 = mul(x = var_3722_1, y = const_64_promoted)[name = string("op_3724")]; int32 var_3726 = const()[name = string("op_3726"), val = int32(-1)]; bool var_3727_interleave_0 = const()[name = string("op_3727_interleave_0"), val = bool(false)]; tensor var_3727 = concat(axis = var_3726, interleave = var_3727_interleave_0, values = (var_3724, var_3722_0))[name = string("op_3727")]; tensor var_3728_cast_fp16 = mul(x = var_3727, y = sin_f)[name = string("op_3728_cast_fp16")]; tensor k_13_cast_fp16 = add(x = var_3721_cast_fp16, y = var_3728_cast_fp16)[name = string("k_13_cast_fp16")]; fp16 var_3731_promoted_to_fp16 = const()[name = string("op_3731_promoted_to_fp16"), val = fp16(0x1p+0)]; tensor var_3733_cast_fp16 = sub(x = var_3731_promoted_to_fp16, y = update_mask)[name = string("op_3733_cast_fp16")]; tensor var_3734_cast_fp16 = mul(x = K_full_slot_1_cast_fp16, y = var_3733_cast_fp16)[name = string("op_3734_cast_fp16")]; tensor var_3735_reps_0 = const()[name = string("op_3735_reps_0"), val = tensor([1, 1, 2048, 1])]; tensor var_3735_cast_fp16 = tile(reps = var_3735_reps_0, x = k_13_cast_fp16)[name = string("op_3735_cast_fp16")]; tensor var_3736_cast_fp16 = mul(x = var_3735_cast_fp16, y = update_mask)[name = string("op_3736_cast_fp16")]; tensor K_full_out_1_cast_fp16 = add(x = var_3734_cast_fp16, y = var_3736_cast_fp16)[name = string("K_full_out_1_cast_fp16")]; tensor var_3742_cast_fp16 = mul(x = V_full_slot_1_cast_fp16, y = var_3733_cast_fp16)[name = string("op_3742_cast_fp16")]; tensor var_3743_reps_0 = const()[name = string("op_3743_reps_0"), val = tensor([1, 1, 2048, 1])]; tensor var_3743_cast_fp16 = tile(reps = var_3743_reps_0, x = v_1_cast_fp16)[name = string("op_3743_cast_fp16")]; tensor var_3744_cast_fp16 = mul(x = var_3743_cast_fp16, y = update_mask)[name = string("op_3744_cast_fp16")]; tensor V_full_out_1_cast_fp16 = add(x = var_3742_cast_fp16, y = var_3744_cast_fp16)[name = string("V_full_out_1_cast_fp16")]; tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = K_full_out_1_cast_fp16)[name = string("transpose_121")]; tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; tensor concat_20 = const()[name = string("concat_20"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_20, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_21 = const()[name = string("concat_21"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_120")]; tensor reshape_21_cast_fp16 = reshape(shape = concat_21, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor transpose_53_perm_0 = const()[name = string("transpose_53_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = V_full_out_1_cast_fp16)[name = string("transpose_119")]; tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; tensor concat_22 = const()[name = string("concat_22"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_22, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_23 = const()[name = string("concat_23"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_118")]; tensor reshape_23_cast_fp16 = reshape(shape = concat_23, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor V_expanded_11_perm_0 = const()[name = string("V_expanded_11_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_21_transpose_x_0 = const()[name = string("attn_weights_21_transpose_x_0"), val = bool(false)]; bool attn_weights_21_transpose_y_0 = const()[name = string("attn_weights_21_transpose_y_0"), val = bool(false)]; tensor transpose_53_cast_fp16 = transpose(perm = transpose_53_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_117")]; tensor attn_weights_21_cast_fp16 = matmul(transpose_x = attn_weights_21_transpose_x_0, transpose_y = attn_weights_21_transpose_y_0, x = q_47_cast_fp16, y = transpose_53_cast_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor x_107_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask_full)[name = string("x_107_cast_fp16")]; tensor reduce_max_5_axes_0 = const()[name = string("reduce_max_5_axes_0"), val = tensor([-1])]; bool reduce_max_5_keep_dims_0 = const()[name = string("reduce_max_5_keep_dims_0"), val = bool(true)]; tensor reduce_max_5 = reduce_max(axes = reduce_max_5_axes_0, keep_dims = reduce_max_5_keep_dims_0, x = x_107_cast_fp16)[name = string("reduce_max_5")]; tensor var_3786 = sub(x = x_107_cast_fp16, y = reduce_max_5)[name = string("op_3786")]; tensor var_3792 = exp(x = var_3786)[name = string("op_3792")]; tensor var_3802_axes_0 = const()[name = string("op_3802_axes_0"), val = tensor([-1])]; bool var_3802_keep_dims_0 = const()[name = string("op_3802_keep_dims_0"), val = bool(true)]; tensor var_3802 = reduce_sum(axes = var_3802_axes_0, keep_dims = var_3802_keep_dims_0, x = var_3792)[name = string("op_3802")]; tensor var_3808_cast_fp16 = real_div(x = var_3792, y = var_3802)[name = string("op_3808_cast_fp16")]; bool attn_output_31_transpose_x_0 = const()[name = string("attn_output_31_transpose_x_0"), val = bool(false)]; bool attn_output_31_transpose_y_0 = const()[name = string("attn_output_31_transpose_y_0"), val = bool(false)]; tensor V_expanded_11_cast_fp16 = transpose(perm = V_expanded_11_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_116")]; tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = var_3808_cast_fp16, y = V_expanded_11_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor var_3819 = const()[name = string("op_3819"), val = tensor([0, 2, 1, 3])]; tensor var_3826 = const()[name = string("op_3826"), val = tensor([1, 1, -1])]; tensor var_3820_cast_fp16 = transpose(perm = var_3819, x = attn_output_31_cast_fp16)[name = string("transpose_115")]; tensor attn_output_33_cast_fp16 = reshape(shape = var_3826, x = var_3820_cast_fp16)[name = string("attn_output_33_cast_fp16")]; tensor var_3831 = const()[name = string("op_3831"), val = tensor([0, 2, 1])]; string var_3847_pad_type_0 = const()[name = string("op_3847_pad_type_0"), val = string("valid")]; int32 var_3847_groups_0 = const()[name = string("op_3847_groups_0"), val = int32(1)]; tensor var_3847_strides_0 = const()[name = string("op_3847_strides_0"), val = tensor([1])]; tensor var_3847_pad_0 = const()[name = string("op_3847_pad_0"), val = tensor([0, 0])]; tensor var_3847_dilations_0 = const()[name = string("op_3847_dilations_0"), val = tensor([1])]; tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546132672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551375616))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3832_cast_fp16 = transpose(perm = var_3831, x = attn_output_33_cast_fp16)[name = string("transpose_114")]; tensor var_3847_cast_fp16 = conv(dilations = var_3847_dilations_0, groups = var_3847_groups_0, pad = var_3847_pad_0, pad_type = var_3847_pad_type_0, strides = var_3847_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_3832_cast_fp16)[name = string("op_3847_cast_fp16")]; tensor var_3851 = const()[name = string("op_3851"), val = tensor([0, 2, 1])]; int32 var_3857 = const()[name = string("op_3857"), val = int32(-1)]; fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_111_cast_fp16 = transpose(perm = var_3851, x = var_3847_cast_fp16)[name = string("transpose_113")]; tensor var_3859_cast_fp16 = mul(x = x_111_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_3859_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_3857, interleave = input_161_interleave_0, values = (x_111_cast_fp16, var_3859_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_3854_to_fp16 = const()[name = string("op_3854_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_3854_to_fp16, x = input_161_cast_fp16)[name = string("normed_153_cast_fp16")]; tensor var_3864_split_sizes_0 = const()[name = string("op_3864_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3864_axis_0 = const()[name = string("op_3864_axis_0"), val = int32(-1)]; tensor var_3864_cast_fp16_0, tensor var_3864_cast_fp16_1 = split(axis = var_3864_axis_0, split_sizes = var_3864_split_sizes_0, x = normed_153_cast_fp16)[name = string("op_3864_cast_fp16")]; tensor layers_5_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551378240)))]; tensor attn_output_35_cast_fp16 = mul(x = var_3864_cast_fp16_0, y = layers_5_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_35_cast_fp16")]; tensor x_113_cast_fp16 = add(x = x_99_cast_fp16, y = attn_output_35_cast_fp16)[name = string("x_113_cast_fp16")]; int32 var_3873 = const()[name = string("op_3873"), val = int32(-1)]; fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3875_cast_fp16 = mul(x = x_113_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_3875_cast_fp16")]; bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; tensor input_163_cast_fp16 = concat(axis = var_3873, interleave = input_163_interleave_0, values = (x_113_cast_fp16, var_3875_cast_fp16))[name = string("input_163_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_3870_to_fp16 = const()[name = string("op_3870_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_3870_to_fp16, x = input_163_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor var_3880_split_sizes_0 = const()[name = string("op_3880_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3880_axis_0 = const()[name = string("op_3880_axis_0"), val = int32(-1)]; tensor var_3880_cast_fp16_0, tensor var_3880_cast_fp16_1 = split(axis = var_3880_axis_0, split_sizes = var_3880_split_sizes_0, x = normed_157_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551383424)))]; tensor h_33_cast_fp16 = mul(x = var_3880_cast_fp16_0, y = layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_33_cast_fp16")]; tensor var_3891 = const()[name = string("op_3891"), val = tensor([0, 2, 1])]; tensor input_165_axes_0 = const()[name = string("input_165_axes_0"), val = tensor([2])]; tensor var_3892 = transpose(perm = var_3891, x = h_33_cast_fp16)[name = string("transpose_112")]; tensor input_165 = expand_dims(axes = input_165_axes_0, x = var_3892)[name = string("input_165")]; string gate_21_pad_type_0 = const()[name = string("gate_21_pad_type_0"), val = string("valid")]; tensor gate_21_strides_0 = const()[name = string("gate_21_strides_0"), val = tensor([1, 1])]; tensor gate_21_pad_0 = const()[name = string("gate_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_21_dilations_0 = const()[name = string("gate_21_dilations_0"), val = tensor([1, 1])]; int32 gate_21_groups_0 = const()[name = string("gate_21_groups_0"), val = int32(1)]; tensor gate_21 = conv(dilations = gate_21_dilations_0, groups = gate_21_groups_0, pad = gate_21_pad_0, pad_type = gate_21_pad_type_0, strides = gate_21_strides_0, weight = layers_5_mlp_gate_proj_weight_palettized, x = input_165)[name = string("gate_21")]; string up_11_pad_type_0 = const()[name = string("up_11_pad_type_0"), val = string("valid")]; tensor up_11_strides_0 = const()[name = string("up_11_strides_0"), val = tensor([1, 1])]; tensor up_11_pad_0 = const()[name = string("up_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_11_dilations_0 = const()[name = string("up_11_dilations_0"), val = tensor([1, 1])]; int32 up_11_groups_0 = const()[name = string("up_11_groups_0"), val = int32(1)]; tensor up_11 = conv(dilations = up_11_dilations_0, groups = up_11_groups_0, pad = up_11_pad_0, pad_type = up_11_pad_type_0, strides = up_11_strides_0, weight = layers_5_mlp_up_proj_weight_palettized, x = input_165)[name = string("up_11")]; string gate_23_mode_0 = const()[name = string("gate_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_23 = gelu(mode = gate_23_mode_0, x = gate_21)[name = string("gate_23")]; tensor input_167 = mul(x = gate_23, y = up_11)[name = string("input_167")]; string mlp_out_11_pad_type_0 = const()[name = string("mlp_out_11_pad_type_0"), val = string("valid")]; tensor mlp_out_11_strides_0 = const()[name = string("mlp_out_11_strides_0"), val = tensor([1, 1])]; tensor mlp_out_11_pad_0 = const()[name = string("mlp_out_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_11_dilations_0 = const()[name = string("mlp_out_11_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_11_groups_0 = const()[name = string("mlp_out_11_groups_0"), val = int32(1)]; tensor mlp_out_11 = conv(dilations = mlp_out_11_dilations_0, groups = mlp_out_11_groups_0, pad = mlp_out_11_pad_0, pad_type = mlp_out_11_pad_type_0, strides = mlp_out_11_strides_0, weight = layers_5_mlp_down_proj_weight_palettized, x = input_167)[name = string("mlp_out_11")]; tensor var_3932_axes_0 = const()[name = string("op_3932_axes_0"), val = tensor([2])]; tensor var_3932 = squeeze(axes = var_3932_axes_0, x = mlp_out_11)[name = string("op_3932")]; tensor var_3936 = const()[name = string("op_3936"), val = tensor([0, 2, 1])]; int32 var_3942 = const()[name = string("op_3942"), val = int32(-1)]; fp16 const_67_promoted = const()[name = string("const_67_promoted"), val = fp16(-0x1p+0)]; tensor x_115 = transpose(perm = var_3936, x = var_3932)[name = string("transpose_111")]; tensor var_3944 = mul(x = x_115, y = const_67_promoted)[name = string("op_3944")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169 = concat(axis = var_3942, interleave = input_169_interleave_0, values = (x_115, var_3944))[name = string("input_169")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_3939_to_fp16 = const()[name = string("op_3939_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_3939_to_fp16, x = input_169)[name = string("normed_161_cast_fp16")]; tensor var_3949_split_sizes_0 = const()[name = string("op_3949_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3949_axis_0 = const()[name = string("op_3949_axis_0"), val = int32(-1)]; tensor var_3949_0, tensor var_3949_1 = split(axis = var_3949_axis_0, split_sizes = var_3949_split_sizes_0, x = normed_161_cast_fp16)[name = string("op_3949")]; tensor hidden_states_53 = mul(x = var_3949_0, y = layers_5_post_feedforward_layernorm_weight)[name = string("hidden_states_53")]; tensor hidden_states_55_cast_fp16 = add(x = x_113_cast_fp16, y = hidden_states_53)[name = string("hidden_states_55_cast_fp16")]; tensor per_layer_slice_11_begin_0 = const()[name = string("per_layer_slice_11_begin_0"), val = tensor([0, 0, 4352])]; tensor per_layer_slice_11_end_0 = const()[name = string("per_layer_slice_11_end_0"), val = tensor([1, 1, 4608])]; tensor per_layer_slice_11_end_mask_0 = const()[name = string("per_layer_slice_11_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_11_cast_fp16 = slice_by_index(begin = per_layer_slice_11_begin_0, end = per_layer_slice_11_end_0, end_mask = per_layer_slice_11_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_11_cast_fp16")]; tensor var_3977 = const()[name = string("op_3977"), val = tensor([0, 2, 1])]; tensor input_171_axes_0 = const()[name = string("input_171_axes_0"), val = tensor([2])]; tensor var_3978 = transpose(perm = var_3977, x = hidden_states_55_cast_fp16)[name = string("transpose_110")]; tensor input_171 = expand_dims(axes = input_171_axes_0, x = var_3978)[name = string("input_171")]; string gated_31_pad_type_0 = const()[name = string("gated_31_pad_type_0"), val = string("valid")]; tensor gated_31_strides_0 = const()[name = string("gated_31_strides_0"), val = tensor([1, 1])]; tensor gated_31_pad_0 = const()[name = string("gated_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_31_dilations_0 = const()[name = string("gated_31_dilations_0"), val = tensor([1, 1])]; int32 gated_31_groups_0 = const()[name = string("gated_31_groups_0"), val = int32(1)]; tensor gated_31 = conv(dilations = gated_31_dilations_0, groups = gated_31_groups_0, pad = gated_31_pad_0, pad_type = gated_31_pad_type_0, strides = gated_31_strides_0, weight = layers_5_per_layer_input_gate_weight_palettized, x = input_171)[name = string("gated_31")]; string gated_33_mode_0 = const()[name = string("gated_33_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_33 = gelu(mode = gated_33_mode_0, x = gated_31)[name = string("gated_33")]; tensor var_3997 = const()[name = string("op_3997"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_11_axes_0 = const()[name = string("per_layer_slice_conv_11_axes_0"), val = tensor([2])]; tensor var_3998_cast_fp16 = transpose(perm = var_3997, x = per_layer_slice_11_cast_fp16)[name = string("transpose_109")]; tensor per_layer_slice_conv_11_cast_fp16 = expand_dims(axes = per_layer_slice_conv_11_axes_0, x = var_3998_cast_fp16)[name = string("per_layer_slice_conv_11_cast_fp16")]; tensor input_173_cast_fp16 = mul(x = gated_33, y = per_layer_slice_conv_11_cast_fp16)[name = string("input_173_cast_fp16")]; string gated_35_pad_type_0 = const()[name = string("gated_35_pad_type_0"), val = string("valid")]; tensor gated_35_strides_0 = const()[name = string("gated_35_strides_0"), val = tensor([1, 1])]; tensor gated_35_pad_0 = const()[name = string("gated_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_35_dilations_0 = const()[name = string("gated_35_dilations_0"), val = tensor([1, 1])]; int32 gated_35_groups_0 = const()[name = string("gated_35_groups_0"), val = int32(1)]; tensor layers_5_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551388608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551716352))))[name = string("layers_5_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_35_cast_fp16 = conv(dilations = gated_35_dilations_0, groups = gated_35_groups_0, pad = gated_35_pad_0, pad_type = gated_35_pad_type_0, strides = gated_35_strides_0, weight = layers_5_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_173_cast_fp16)[name = string("gated_35_cast_fp16")]; tensor var_4014_axes_0 = const()[name = string("op_4014_axes_0"), val = tensor([2])]; tensor var_4014_cast_fp16 = squeeze(axes = var_4014_axes_0, x = gated_35_cast_fp16)[name = string("op_4014_cast_fp16")]; tensor var_4018 = const()[name = string("op_4018"), val = tensor([0, 2, 1])]; int32 var_4024 = const()[name = string("op_4024"), val = int32(-1)]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_117_cast_fp16 = transpose(perm = var_4018, x = var_4014_cast_fp16)[name = string("transpose_108")]; tensor var_4026_cast_fp16 = mul(x = x_117_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_4026_cast_fp16")]; bool input_175_interleave_0 = const()[name = string("input_175_interleave_0"), val = bool(false)]; tensor input_175_cast_fp16 = concat(axis = var_4024, interleave = input_175_interleave_0, values = (x_117_cast_fp16, var_4026_cast_fp16))[name = string("input_175_cast_fp16")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_4021_to_fp16 = const()[name = string("op_4021_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_4021_to_fp16, x = input_175_cast_fp16)[name = string("normed_165_cast_fp16")]; tensor var_4031_split_sizes_0 = const()[name = string("op_4031_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4031_axis_0 = const()[name = string("op_4031_axis_0"), val = int32(-1)]; tensor var_4031_cast_fp16_0, tensor var_4031_cast_fp16_1 = split(axis = var_4031_axis_0, split_sizes = var_4031_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_4031_cast_fp16")]; tensor layers_5_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551718976)))]; tensor hidden_states_59_cast_fp16 = mul(x = var_4031_cast_fp16_0, y = layers_5_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_59_cast_fp16")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_59_cast_fp16)[name = string("hidden_states_61_cast_fp16")]; tensor const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = tensor([0x1.b2p-2])]; tensor x_119_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_69_promoted_to_fp16)[name = string("x_119_cast_fp16")]; tensor var_4043_axes_0 = const()[name = string("op_4043_axes_0"), val = tensor([0])]; tensor var_4043_cast_fp16 = squeeze(axes = var_4043_axes_0, x = K_full_out_1_cast_fp16)[name = string("op_4043_cast_fp16")]; tensor var_4045_axes_0 = const()[name = string("op_4045_axes_0"), val = tensor([0])]; tensor var_4045_cast_fp16 = squeeze(axes = var_4045_axes_0, x = V_full_out_1_cast_fp16)[name = string("op_4045_cast_fp16")]; tensor var_4048_begin_0 = const()[name = string("op_4048_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_4048_end_0 = const()[name = string("op_4048_end_0"), val = tensor([6, 2, 512, 512])]; tensor var_4048_end_mask_0 = const()[name = string("op_4048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4048_squeeze_mask_0 = const()[name = string("op_4048_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4048_cast_fp16 = slice_by_index(begin = var_4048_begin_0, end = var_4048_end_0, end_mask = var_4048_end_mask_0, squeeze_mask = var_4048_squeeze_mask_0, x = K_sliding_in)[name = string("op_4048_cast_fp16")]; tensor K_sliding_slot_11_axes_0 = const()[name = string("K_sliding_slot_11_axes_0"), val = tensor([0])]; tensor K_sliding_slot_11_cast_fp16 = expand_dims(axes = K_sliding_slot_11_axes_0, x = var_4048_cast_fp16)[name = string("K_sliding_slot_11_cast_fp16")]; tensor var_4053_begin_0 = const()[name = string("op_4053_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_4053_end_0 = const()[name = string("op_4053_end_0"), val = tensor([6, 2, 512, 512])]; tensor var_4053_end_mask_0 = const()[name = string("op_4053_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4053_squeeze_mask_0 = const()[name = string("op_4053_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4053_cast_fp16 = slice_by_index(begin = var_4053_begin_0, end = var_4053_end_0, end_mask = var_4053_end_mask_0, squeeze_mask = var_4053_squeeze_mask_0, x = V_sliding_in)[name = string("op_4053_cast_fp16")]; tensor V_sliding_slot_11_axes_0 = const()[name = string("V_sliding_slot_11_axes_0"), val = tensor([0])]; tensor V_sliding_slot_11_cast_fp16 = expand_dims(axes = V_sliding_slot_11_axes_0, x = var_4053_cast_fp16)[name = string("V_sliding_slot_11_cast_fp16")]; int32 var_4060 = const()[name = string("op_4060"), val = int32(-1)]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4062_cast_fp16 = mul(x = x_119_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_4062_cast_fp16")]; bool input_177_interleave_0 = const()[name = string("input_177_interleave_0"), val = bool(false)]; tensor input_177_cast_fp16 = concat(axis = var_4060, interleave = input_177_interleave_0, values = (x_119_cast_fp16, var_4062_cast_fp16))[name = string("input_177_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_4057_to_fp16 = const()[name = string("op_4057_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_4057_to_fp16, x = input_177_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_4067_split_sizes_0 = const()[name = string("op_4067_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4067_axis_0 = const()[name = string("op_4067_axis_0"), val = int32(-1)]; tensor var_4067_cast_fp16_0, tensor var_4067_cast_fp16_1 = split(axis = var_4067_axis_0, split_sizes = var_4067_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_4067_cast_fp16")]; tensor layers_6_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551724160)))]; tensor h_37_cast_fp16 = mul(x = var_4067_cast_fp16_0, y = layers_6_input_layernorm_weight_promoted_to_fp16)[name = string("h_37_cast_fp16")]; tensor var_4073 = const()[name = string("op_4073"), val = tensor([0, 2, 1])]; tensor var_4076_axes_0 = const()[name = string("op_4076_axes_0"), val = tensor([2])]; tensor var_4074_cast_fp16 = transpose(perm = var_4073, x = h_37_cast_fp16)[name = string("transpose_107")]; tensor var_4076_cast_fp16 = expand_dims(axes = var_4076_axes_0, x = var_4074_cast_fp16)[name = string("op_4076_cast_fp16")]; string var_4092_pad_type_0 = const()[name = string("op_4092_pad_type_0"), val = string("valid")]; tensor var_4092_strides_0 = const()[name = string("op_4092_strides_0"), val = tensor([1, 1])]; tensor var_4092_pad_0 = const()[name = string("op_4092_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4092_dilations_0 = const()[name = string("op_4092_dilations_0"), val = tensor([1, 1])]; int32 var_4092_groups_0 = const()[name = string("op_4092_groups_0"), val = int32(1)]; tensor var_4092 = conv(dilations = var_4092_dilations_0, groups = var_4092_groups_0, pad = var_4092_pad_0, pad_type = var_4092_pad_type_0, strides = var_4092_strides_0, weight = layers_6_self_attn_q_proj_weight_palettized, x = var_4076_cast_fp16)[name = string("op_4092")]; tensor var_4097 = const()[name = string("op_4097"), val = tensor([1, 8, 256, 1])]; tensor var_4098 = reshape(shape = var_4097, x = var_4092)[name = string("op_4098")]; tensor var_4103 = const()[name = string("op_4103"), val = tensor([0, 1, 3, 2])]; tensor var_4113 = const()[name = string("op_4113"), val = tensor([1, 8, 256])]; tensor var_4104 = transpose(perm = var_4103, x = var_4098)[name = string("transpose_106")]; tensor x_121 = reshape(shape = var_4113, x = var_4104)[name = string("x_121")]; int32 var_4119 = const()[name = string("op_4119"), val = int32(-1)]; fp16 const_71_promoted = const()[name = string("const_71_promoted"), val = fp16(-0x1p+0)]; tensor var_4121 = mul(x = x_121, y = const_71_promoted)[name = string("op_4121")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181 = concat(axis = var_4119, interleave = input_181_interleave_0, values = (x_121, var_4121))[name = string("input_181")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_4116_to_fp16 = const()[name = string("op_4116_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_4116_to_fp16, x = input_181)[name = string("normed_173_cast_fp16")]; tensor var_4126_split_sizes_0 = const()[name = string("op_4126_split_sizes_0"), val = tensor([256, 256])]; int32 var_4126_axis_0 = const()[name = string("op_4126_axis_0"), val = int32(-1)]; tensor var_4126_0, tensor var_4126_1 = split(axis = var_4126_axis_0, split_sizes = var_4126_split_sizes_0, x = normed_173_cast_fp16)[name = string("op_4126")]; tensor var_4128 = mul(x = var_4126_0, y = layers_2_self_attn_q_norm_weight)[name = string("op_4128")]; tensor var_4133 = const()[name = string("op_4133"), val = tensor([1, 8, 1, 256])]; tensor q_51 = reshape(shape = var_4133, x = var_4128)[name = string("q_51")]; tensor var_4135_cast_fp16 = mul(x = q_51, y = cos_s)[name = string("op_4135_cast_fp16")]; tensor var_4136_split_sizes_0 = const()[name = string("op_4136_split_sizes_0"), val = tensor([128, 128])]; int32 var_4136_axis_0 = const()[name = string("op_4136_axis_0"), val = int32(-1)]; tensor var_4136_0, tensor var_4136_1 = split(axis = var_4136_axis_0, split_sizes = var_4136_split_sizes_0, x = q_51)[name = string("op_4136")]; fp16 const_72_promoted = const()[name = string("const_72_promoted"), val = fp16(-0x1p+0)]; tensor var_4138 = mul(x = var_4136_1, y = const_72_promoted)[name = string("op_4138")]; int32 var_4140 = const()[name = string("op_4140"), val = int32(-1)]; bool var_4141_interleave_0 = const()[name = string("op_4141_interleave_0"), val = bool(false)]; tensor var_4141 = concat(axis = var_4140, interleave = var_4141_interleave_0, values = (var_4138, var_4136_0))[name = string("op_4141")]; tensor var_4142_cast_fp16 = mul(x = var_4141, y = sin_s)[name = string("op_4142_cast_fp16")]; tensor q_55_cast_fp16 = add(x = var_4135_cast_fp16, y = var_4142_cast_fp16)[name = string("q_55_cast_fp16")]; string var_4155_pad_type_0 = const()[name = string("op_4155_pad_type_0"), val = string("valid")]; tensor var_4155_strides_0 = const()[name = string("op_4155_strides_0"), val = tensor([1, 1])]; tensor var_4155_pad_0 = const()[name = string("op_4155_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4155_dilations_0 = const()[name = string("op_4155_dilations_0"), val = tensor([1, 1])]; int32 var_4155_groups_0 = const()[name = string("op_4155_groups_0"), val = int32(1)]; tensor var_4155 = conv(dilations = var_4155_dilations_0, groups = var_4155_groups_0, pad = var_4155_pad_0, pad_type = var_4155_pad_type_0, strides = var_4155_strides_0, weight = layers_6_self_attn_k_proj_weight_palettized, x = var_4076_cast_fp16)[name = string("op_4155")]; tensor var_4160 = const()[name = string("op_4160"), val = tensor([1, 2, 256, 1])]; tensor var_4161 = reshape(shape = var_4160, x = var_4155)[name = string("op_4161")]; tensor var_4166 = const()[name = string("op_4166"), val = tensor([0, 1, 3, 2])]; string var_4183_pad_type_0 = const()[name = string("op_4183_pad_type_0"), val = string("valid")]; tensor var_4183_strides_0 = const()[name = string("op_4183_strides_0"), val = tensor([1, 1])]; tensor var_4183_pad_0 = const()[name = string("op_4183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4183_dilations_0 = const()[name = string("op_4183_dilations_0"), val = tensor([1, 1])]; int32 var_4183_groups_0 = const()[name = string("op_4183_groups_0"), val = int32(1)]; tensor var_4183 = conv(dilations = var_4183_dilations_0, groups = var_4183_groups_0, pad = var_4183_pad_0, pad_type = var_4183_pad_type_0, strides = var_4183_strides_0, weight = layers_6_self_attn_v_proj_weight_palettized, x = var_4076_cast_fp16)[name = string("op_4183")]; tensor var_4188 = const()[name = string("op_4188"), val = tensor([1, 2, 256, 1])]; tensor var_4189 = reshape(shape = var_4188, x = var_4183)[name = string("op_4189")]; tensor var_4194 = const()[name = string("op_4194"), val = tensor([0, 1, 3, 2])]; tensor var_4204 = const()[name = string("op_4204"), val = tensor([1, 2, 256])]; tensor var_4167 = transpose(perm = var_4166, x = var_4161)[name = string("transpose_105")]; tensor x_123 = reshape(shape = var_4204, x = var_4167)[name = string("x_123")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-1)]; fp16 const_73_promoted = const()[name = string("const_73_promoted"), val = fp16(-0x1p+0)]; tensor var_4212 = mul(x = x_123, y = const_73_promoted)[name = string("op_4212")]; bool input_183_interleave_0 = const()[name = string("input_183_interleave_0"), val = bool(false)]; tensor input_183 = concat(axis = var_4210, interleave = input_183_interleave_0, values = (x_123, var_4212))[name = string("input_183")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_4207_to_fp16 = const()[name = string("op_4207_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_4207_to_fp16, x = input_183)[name = string("normed_177_cast_fp16")]; tensor var_4217_split_sizes_0 = const()[name = string("op_4217_split_sizes_0"), val = tensor([256, 256])]; int32 var_4217_axis_0 = const()[name = string("op_4217_axis_0"), val = int32(-1)]; tensor var_4217_0, tensor var_4217_1 = split(axis = var_4217_axis_0, split_sizes = var_4217_split_sizes_0, x = normed_177_cast_fp16)[name = string("op_4217")]; tensor var_4219 = mul(x = var_4217_0, y = layers_6_self_attn_k_norm_weight)[name = string("op_4219")]; tensor var_4224 = const()[name = string("op_4224"), val = tensor([1, 2, 1, 256])]; tensor q_53 = reshape(shape = var_4224, x = var_4219)[name = string("q_53")]; fp16 var_4226_promoted = const()[name = string("op_4226_promoted"), val = fp16(0x1p+1)]; tensor var_4195 = transpose(perm = var_4194, x = var_4189)[name = string("transpose_104")]; tensor var_4227 = pow(x = var_4195, y = var_4226_promoted)[name = string("op_4227")]; tensor var_4232_axes_0 = const()[name = string("op_4232_axes_0"), val = tensor([-1])]; bool var_4232_keep_dims_0 = const()[name = string("op_4232_keep_dims_0"), val = bool(true)]; tensor var_4232 = reduce_mean(axes = var_4232_axes_0, keep_dims = var_4232_keep_dims_0, x = var_4227)[name = string("op_4232")]; fp16 var_4234_to_fp16 = const()[name = string("op_4234_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_13_cast_fp16 = add(x = var_4232, y = var_4234_to_fp16)[name = string("mean_sq_13_cast_fp16")]; fp32 var_4236_epsilon_0 = const()[name = string("op_4236_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_4236_cast_fp16 = rsqrt(epsilon = var_4236_epsilon_0, x = mean_sq_13_cast_fp16)[name = string("op_4236_cast_fp16")]; tensor input_187_cast_fp16 = mul(x = var_4195, y = var_4236_cast_fp16)[name = string("input_187_cast_fp16")]; tensor var_4238_cast_fp16 = mul(x = q_53, y = cos_s)[name = string("op_4238_cast_fp16")]; tensor var_4239_split_sizes_0 = const()[name = string("op_4239_split_sizes_0"), val = tensor([128, 128])]; int32 var_4239_axis_0 = const()[name = string("op_4239_axis_0"), val = int32(-1)]; tensor var_4239_0, tensor var_4239_1 = split(axis = var_4239_axis_0, split_sizes = var_4239_split_sizes_0, x = q_53)[name = string("op_4239")]; fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; tensor var_4241 = mul(x = var_4239_1, y = const_74_promoted)[name = string("op_4241")]; int32 var_4243 = const()[name = string("op_4243"), val = int32(-1)]; bool var_4244_interleave_0 = const()[name = string("op_4244_interleave_0"), val = bool(false)]; tensor var_4244 = concat(axis = var_4243, interleave = var_4244_interleave_0, values = (var_4241, var_4239_0))[name = string("op_4244")]; tensor var_4245_cast_fp16 = mul(x = var_4244, y = sin_s)[name = string("op_4245_cast_fp16")]; tensor input_185_cast_fp16 = add(x = var_4238_cast_fp16, y = var_4245_cast_fp16)[name = string("input_185_cast_fp16")]; tensor k_padded_11_pad_0 = const()[name = string("k_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_11_mode_0 = const()[name = string("k_padded_11_mode_0"), val = string("constant")]; fp16 const_75_to_fp16 = const()[name = string("const_75_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_11_cast_fp16 = pad(constant_val = const_75_to_fp16, mode = k_padded_11_mode_0, pad = k_padded_11_pad_0, x = input_185_cast_fp16)[name = string("k_padded_11_cast_fp16")]; tensor v_padded_11_pad_0 = const()[name = string("v_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_11_mode_0 = const()[name = string("v_padded_11_mode_0"), val = string("constant")]; fp16 const_76_to_fp16 = const()[name = string("const_76_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_11_cast_fp16 = pad(constant_val = const_76_to_fp16, mode = v_padded_11_mode_0, pad = v_padded_11_pad_0, x = input_187_cast_fp16)[name = string("v_padded_11_cast_fp16")]; tensor var_4274_begin_0 = const()[name = string("op_4274_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_4274_end_0 = const()[name = string("op_4274_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_4274_end_mask_0 = const()[name = string("op_4274_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4274_cast_fp16 = slice_by_index(begin = var_4274_begin_0, end = var_4274_end_0, end_mask = var_4274_end_mask_0, x = K_sliding_slot_11_cast_fp16)[name = string("op_4274_cast_fp16")]; int32 var_4281 = const()[name = string("op_4281"), val = int32(2)]; bool K_sliding_out_11_interleave_0 = const()[name = string("K_sliding_out_11_interleave_0"), val = bool(false)]; tensor K_sliding_out_11_cast_fp16 = concat(axis = var_4281, interleave = K_sliding_out_11_interleave_0, values = (var_4274_cast_fp16, k_padded_11_cast_fp16))[name = string("K_sliding_out_11_cast_fp16")]; tensor var_4297_begin_0 = const()[name = string("op_4297_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_4297_end_0 = const()[name = string("op_4297_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_4297_end_mask_0 = const()[name = string("op_4297_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4297_cast_fp16 = slice_by_index(begin = var_4297_begin_0, end = var_4297_end_0, end_mask = var_4297_end_mask_0, x = V_sliding_slot_11_cast_fp16)[name = string("op_4297_cast_fp16")]; int32 var_4304 = const()[name = string("op_4304"), val = int32(2)]; bool V_sliding_out_11_interleave_0 = const()[name = string("V_sliding_out_11_interleave_0"), val = bool(false)]; tensor V_sliding_out_11_cast_fp16 = concat(axis = var_4304, interleave = V_sliding_out_11_interleave_0, values = (var_4297_cast_fp16, v_padded_11_cast_fp16))[name = string("V_sliding_out_11_cast_fp16")]; tensor K_for_attn_13_begin_0 = const()[name = string("K_for_attn_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_13_end_0 = const()[name = string("K_for_attn_13_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_13_end_mask_0 = const()[name = string("K_for_attn_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_13_cast_fp16 = slice_by_index(begin = K_for_attn_13_begin_0, end = K_for_attn_13_end_0, end_mask = K_for_attn_13_end_mask_0, x = K_sliding_out_11_cast_fp16)[name = string("K_for_attn_13_cast_fp16")]; tensor V_for_attn_13_begin_0 = const()[name = string("V_for_attn_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_13_end_0 = const()[name = string("V_for_attn_13_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_13_end_mask_0 = const()[name = string("V_for_attn_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_13_cast_fp16 = slice_by_index(begin = V_for_attn_13_begin_0, end = V_for_attn_13_end_0, end_mask = V_for_attn_13_end_mask_0, x = V_sliding_out_11_cast_fp16)[name = string("V_for_attn_13_cast_fp16")]; tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = K_for_attn_13_cast_fp16)[name = string("transpose_103")]; tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; tensor concat_24 = const()[name = string("concat_24"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_24, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_25 = const()[name = string("concat_25"), val = tensor([-1, 1, 512, 256])]; tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_102")]; tensor reshape_25_cast_fp16 = reshape(shape = concat_25, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; tensor transpose_54_perm_0 = const()[name = string("transpose_54_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = V_for_attn_13_cast_fp16)[name = string("transpose_101")]; tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; tensor concat_26 = const()[name = string("concat_26"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_26, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_27 = const()[name = string("concat_27"), val = tensor([-1, 1, 512, 256])]; tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_100")]; tensor reshape_27_cast_fp16 = reshape(shape = concat_27, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor V_expanded_13_perm_0 = const()[name = string("V_expanded_13_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor transpose_54_cast_fp16 = transpose(perm = transpose_54_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_99")]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = q_55_cast_fp16, y = transpose_54_cast_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor x_127_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask_sliding)[name = string("x_127_cast_fp16")]; tensor reduce_max_6_axes_0 = const()[name = string("reduce_max_6_axes_0"), val = tensor([-1])]; bool reduce_max_6_keep_dims_0 = const()[name = string("reduce_max_6_keep_dims_0"), val = bool(true)]; tensor reduce_max_6 = reduce_max(axes = reduce_max_6_axes_0, keep_dims = reduce_max_6_keep_dims_0, x = x_127_cast_fp16)[name = string("reduce_max_6")]; tensor var_4345 = sub(x = x_127_cast_fp16, y = reduce_max_6)[name = string("op_4345")]; tensor var_4351 = exp(x = var_4345)[name = string("op_4351")]; tensor var_4361_axes_0 = const()[name = string("op_4361_axes_0"), val = tensor([-1])]; bool var_4361_keep_dims_0 = const()[name = string("op_4361_keep_dims_0"), val = bool(true)]; tensor var_4361 = reduce_sum(axes = var_4361_axes_0, keep_dims = var_4361_keep_dims_0, x = var_4351)[name = string("op_4361")]; tensor var_4367_cast_fp16 = real_div(x = var_4351, y = var_4361)[name = string("op_4367_cast_fp16")]; bool attn_output_37_transpose_x_0 = const()[name = string("attn_output_37_transpose_x_0"), val = bool(false)]; bool attn_output_37_transpose_y_0 = const()[name = string("attn_output_37_transpose_y_0"), val = bool(false)]; tensor V_expanded_13_cast_fp16 = transpose(perm = V_expanded_13_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_98")]; tensor attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = var_4367_cast_fp16, y = V_expanded_13_cast_fp16)[name = string("attn_output_37_cast_fp16")]; tensor var_4378 = const()[name = string("op_4378"), val = tensor([0, 2, 1, 3])]; tensor var_4385 = const()[name = string("op_4385"), val = tensor([1, 1, -1])]; tensor var_4379_cast_fp16 = transpose(perm = var_4378, x = attn_output_37_cast_fp16)[name = string("transpose_97")]; tensor attn_output_39_cast_fp16 = reshape(shape = var_4385, x = var_4379_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor var_4390 = const()[name = string("op_4390"), val = tensor([0, 2, 1])]; string var_4406_pad_type_0 = const()[name = string("op_4406_pad_type_0"), val = string("valid")]; int32 var_4406_groups_0 = const()[name = string("op_4406_groups_0"), val = int32(1)]; tensor var_4406_strides_0 = const()[name = string("op_4406_strides_0"), val = tensor([1])]; tensor var_4406_pad_0 = const()[name = string("op_4406_pad_0"), val = tensor([0, 0])]; tensor var_4406_dilations_0 = const()[name = string("op_4406_dilations_0"), val = tensor([1])]; tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551729344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554350848))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4391_cast_fp16 = transpose(perm = var_4390, x = attn_output_39_cast_fp16)[name = string("transpose_96")]; tensor var_4406_cast_fp16 = conv(dilations = var_4406_dilations_0, groups = var_4406_groups_0, pad = var_4406_pad_0, pad_type = var_4406_pad_type_0, strides = var_4406_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4391_cast_fp16)[name = string("op_4406_cast_fp16")]; tensor var_4410 = const()[name = string("op_4410"), val = tensor([0, 2, 1])]; int32 var_4416 = const()[name = string("op_4416"), val = int32(-1)]; fp16 const_77_promoted_to_fp16 = const()[name = string("const_77_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_131_cast_fp16 = transpose(perm = var_4410, x = var_4406_cast_fp16)[name = string("transpose_95")]; tensor var_4418_cast_fp16 = mul(x = x_131_cast_fp16, y = const_77_promoted_to_fp16)[name = string("op_4418_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_4416, interleave = input_191_interleave_0, values = (x_131_cast_fp16, var_4418_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_4413_to_fp16 = const()[name = string("op_4413_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_4413_to_fp16, x = input_191_cast_fp16)[name = string("normed_181_cast_fp16")]; tensor var_4423_split_sizes_0 = const()[name = string("op_4423_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4423_axis_0 = const()[name = string("op_4423_axis_0"), val = int32(-1)]; tensor var_4423_cast_fp16_0, tensor var_4423_cast_fp16_1 = split(axis = var_4423_axis_0, split_sizes = var_4423_split_sizes_0, x = normed_181_cast_fp16)[name = string("op_4423_cast_fp16")]; tensor layers_6_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554353472)))]; tensor attn_output_41_cast_fp16 = mul(x = var_4423_cast_fp16_0, y = layers_6_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_41_cast_fp16")]; tensor x_133_cast_fp16 = add(x = x_119_cast_fp16, y = attn_output_41_cast_fp16)[name = string("x_133_cast_fp16")]; int32 var_4432 = const()[name = string("op_4432"), val = int32(-1)]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4434_cast_fp16 = mul(x = x_133_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_4434_cast_fp16")]; bool input_193_interleave_0 = const()[name = string("input_193_interleave_0"), val = bool(false)]; tensor input_193_cast_fp16 = concat(axis = var_4432, interleave = input_193_interleave_0, values = (x_133_cast_fp16, var_4434_cast_fp16))[name = string("input_193_cast_fp16")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_4429_to_fp16 = const()[name = string("op_4429_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_4429_to_fp16, x = input_193_cast_fp16)[name = string("normed_185_cast_fp16")]; tensor var_4439_split_sizes_0 = const()[name = string("op_4439_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4439_axis_0 = const()[name = string("op_4439_axis_0"), val = int32(-1)]; tensor var_4439_cast_fp16_0, tensor var_4439_cast_fp16_1 = split(axis = var_4439_axis_0, split_sizes = var_4439_split_sizes_0, x = normed_185_cast_fp16)[name = string("op_4439_cast_fp16")]; tensor layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554358656)))]; tensor h_39_cast_fp16 = mul(x = var_4439_cast_fp16_0, y = layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_39_cast_fp16")]; tensor var_4450 = const()[name = string("op_4450"), val = tensor([0, 2, 1])]; tensor input_195_axes_0 = const()[name = string("input_195_axes_0"), val = tensor([2])]; tensor var_4451 = transpose(perm = var_4450, x = h_39_cast_fp16)[name = string("transpose_94")]; tensor input_195 = expand_dims(axes = input_195_axes_0, x = var_4451)[name = string("input_195")]; string gate_25_pad_type_0 = const()[name = string("gate_25_pad_type_0"), val = string("valid")]; tensor gate_25_strides_0 = const()[name = string("gate_25_strides_0"), val = tensor([1, 1])]; tensor gate_25_pad_0 = const()[name = string("gate_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_25_dilations_0 = const()[name = string("gate_25_dilations_0"), val = tensor([1, 1])]; int32 gate_25_groups_0 = const()[name = string("gate_25_groups_0"), val = int32(1)]; tensor gate_25 = conv(dilations = gate_25_dilations_0, groups = gate_25_groups_0, pad = gate_25_pad_0, pad_type = gate_25_pad_type_0, strides = gate_25_strides_0, weight = layers_6_mlp_gate_proj_weight_palettized, x = input_195)[name = string("gate_25")]; string up_13_pad_type_0 = const()[name = string("up_13_pad_type_0"), val = string("valid")]; tensor up_13_strides_0 = const()[name = string("up_13_strides_0"), val = tensor([1, 1])]; tensor up_13_pad_0 = const()[name = string("up_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_13_dilations_0 = const()[name = string("up_13_dilations_0"), val = tensor([1, 1])]; int32 up_13_groups_0 = const()[name = string("up_13_groups_0"), val = int32(1)]; tensor up_13 = conv(dilations = up_13_dilations_0, groups = up_13_groups_0, pad = up_13_pad_0, pad_type = up_13_pad_type_0, strides = up_13_strides_0, weight = layers_6_mlp_up_proj_weight_palettized, x = input_195)[name = string("up_13")]; string gate_27_mode_0 = const()[name = string("gate_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_27 = gelu(mode = gate_27_mode_0, x = gate_25)[name = string("gate_27")]; tensor input_197 = mul(x = gate_27, y = up_13)[name = string("input_197")]; string mlp_out_13_pad_type_0 = const()[name = string("mlp_out_13_pad_type_0"), val = string("valid")]; tensor mlp_out_13_strides_0 = const()[name = string("mlp_out_13_strides_0"), val = tensor([1, 1])]; tensor mlp_out_13_pad_0 = const()[name = string("mlp_out_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_13_dilations_0 = const()[name = string("mlp_out_13_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_13_groups_0 = const()[name = string("mlp_out_13_groups_0"), val = int32(1)]; tensor mlp_out_13 = conv(dilations = mlp_out_13_dilations_0, groups = mlp_out_13_groups_0, pad = mlp_out_13_pad_0, pad_type = mlp_out_13_pad_type_0, strides = mlp_out_13_strides_0, weight = layers_6_mlp_down_proj_weight_palettized, x = input_197)[name = string("mlp_out_13")]; tensor var_4491_axes_0 = const()[name = string("op_4491_axes_0"), val = tensor([2])]; tensor var_4491 = squeeze(axes = var_4491_axes_0, x = mlp_out_13)[name = string("op_4491")]; tensor var_4495 = const()[name = string("op_4495"), val = tensor([0, 2, 1])]; int32 var_4501 = const()[name = string("op_4501"), val = int32(-1)]; fp16 const_79_promoted = const()[name = string("const_79_promoted"), val = fp16(-0x1p+0)]; tensor x_135 = transpose(perm = var_4495, x = var_4491)[name = string("transpose_93")]; tensor var_4503 = mul(x = x_135, y = const_79_promoted)[name = string("op_4503")]; bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; tensor input_199 = concat(axis = var_4501, interleave = input_199_interleave_0, values = (x_135, var_4503))[name = string("input_199")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_4498_to_fp16 = const()[name = string("op_4498_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_4498_to_fp16, x = input_199)[name = string("normed_189_cast_fp16")]; tensor var_4508_split_sizes_0 = const()[name = string("op_4508_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4508_axis_0 = const()[name = string("op_4508_axis_0"), val = int32(-1)]; tensor var_4508_0, tensor var_4508_1 = split(axis = var_4508_axis_0, split_sizes = var_4508_split_sizes_0, x = normed_189_cast_fp16)[name = string("op_4508")]; tensor hidden_states_63 = mul(x = var_4508_0, y = layers_6_post_feedforward_layernorm_weight)[name = string("hidden_states_63")]; tensor hidden_states_65_cast_fp16 = add(x = x_133_cast_fp16, y = hidden_states_63)[name = string("hidden_states_65_cast_fp16")]; tensor per_layer_slice_13_begin_0 = const()[name = string("per_layer_slice_13_begin_0"), val = tensor([0, 0, 4608])]; tensor per_layer_slice_13_end_0 = const()[name = string("per_layer_slice_13_end_0"), val = tensor([1, 1, 4864])]; tensor per_layer_slice_13_end_mask_0 = const()[name = string("per_layer_slice_13_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_13_cast_fp16 = slice_by_index(begin = per_layer_slice_13_begin_0, end = per_layer_slice_13_end_0, end_mask = per_layer_slice_13_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_13_cast_fp16")]; tensor var_4536 = const()[name = string("op_4536"), val = tensor([0, 2, 1])]; tensor input_201_axes_0 = const()[name = string("input_201_axes_0"), val = tensor([2])]; tensor var_4537 = transpose(perm = var_4536, x = hidden_states_65_cast_fp16)[name = string("transpose_92")]; tensor input_201 = expand_dims(axes = input_201_axes_0, x = var_4537)[name = string("input_201")]; string gated_37_pad_type_0 = const()[name = string("gated_37_pad_type_0"), val = string("valid")]; tensor gated_37_strides_0 = const()[name = string("gated_37_strides_0"), val = tensor([1, 1])]; tensor gated_37_pad_0 = const()[name = string("gated_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_37_dilations_0 = const()[name = string("gated_37_dilations_0"), val = tensor([1, 1])]; int32 gated_37_groups_0 = const()[name = string("gated_37_groups_0"), val = int32(1)]; tensor gated_37 = conv(dilations = gated_37_dilations_0, groups = gated_37_groups_0, pad = gated_37_pad_0, pad_type = gated_37_pad_type_0, strides = gated_37_strides_0, weight = layers_6_per_layer_input_gate_weight_palettized, x = input_201)[name = string("gated_37")]; string gated_39_mode_0 = const()[name = string("gated_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_39 = gelu(mode = gated_39_mode_0, x = gated_37)[name = string("gated_39")]; tensor var_4556 = const()[name = string("op_4556"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_13_axes_0 = const()[name = string("per_layer_slice_conv_13_axes_0"), val = tensor([2])]; tensor var_4557_cast_fp16 = transpose(perm = var_4556, x = per_layer_slice_13_cast_fp16)[name = string("transpose_91")]; tensor per_layer_slice_conv_13_cast_fp16 = expand_dims(axes = per_layer_slice_conv_13_axes_0, x = var_4557_cast_fp16)[name = string("per_layer_slice_conv_13_cast_fp16")]; tensor input_203_cast_fp16 = mul(x = gated_39, y = per_layer_slice_conv_13_cast_fp16)[name = string("input_203_cast_fp16")]; string gated_41_pad_type_0 = const()[name = string("gated_41_pad_type_0"), val = string("valid")]; tensor gated_41_strides_0 = const()[name = string("gated_41_strides_0"), val = tensor([1, 1])]; tensor gated_41_pad_0 = const()[name = string("gated_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_41_dilations_0 = const()[name = string("gated_41_dilations_0"), val = tensor([1, 1])]; int32 gated_41_groups_0 = const()[name = string("gated_41_groups_0"), val = int32(1)]; tensor layers_6_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554363840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554691584))))[name = string("layers_6_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_41_cast_fp16 = conv(dilations = gated_41_dilations_0, groups = gated_41_groups_0, pad = gated_41_pad_0, pad_type = gated_41_pad_type_0, strides = gated_41_strides_0, weight = layers_6_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_203_cast_fp16)[name = string("gated_41_cast_fp16")]; tensor var_4573_axes_0 = const()[name = string("op_4573_axes_0"), val = tensor([2])]; tensor var_4573_cast_fp16 = squeeze(axes = var_4573_axes_0, x = gated_41_cast_fp16)[name = string("op_4573_cast_fp16")]; tensor var_4577 = const()[name = string("op_4577"), val = tensor([0, 2, 1])]; int32 var_4583 = const()[name = string("op_4583"), val = int32(-1)]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_137_cast_fp16 = transpose(perm = var_4577, x = var_4573_cast_fp16)[name = string("transpose_90")]; tensor var_4585_cast_fp16 = mul(x = x_137_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_4585_cast_fp16")]; bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; tensor input_205_cast_fp16 = concat(axis = var_4583, interleave = input_205_interleave_0, values = (x_137_cast_fp16, var_4585_cast_fp16))[name = string("input_205_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_4580_to_fp16 = const()[name = string("op_4580_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_4580_to_fp16, x = input_205_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor var_4590_split_sizes_0 = const()[name = string("op_4590_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4590_axis_0 = const()[name = string("op_4590_axis_0"), val = int32(-1)]; tensor var_4590_cast_fp16_0, tensor var_4590_cast_fp16_1 = split(axis = var_4590_axis_0, split_sizes = var_4590_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_4590_cast_fp16")]; tensor layers_6_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554694208)))]; tensor hidden_states_69_cast_fp16 = mul(x = var_4590_cast_fp16_0, y = layers_6_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_69_cast_fp16")]; tensor hidden_states_71_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_69_cast_fp16)[name = string("hidden_states_71_cast_fp16")]; tensor const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = tensor([0x1.16p-1])]; tensor x_139_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_81_promoted_to_fp16)[name = string("x_139_cast_fp16")]; tensor var_4602_axes_0 = const()[name = string("op_4602_axes_0"), val = tensor([0])]; tensor var_4602_cast_fp16 = squeeze(axes = var_4602_axes_0, x = K_sliding_out_11_cast_fp16)[name = string("op_4602_cast_fp16")]; tensor var_4604_axes_0 = const()[name = string("op_4604_axes_0"), val = tensor([0])]; tensor var_4604_cast_fp16 = squeeze(axes = var_4604_axes_0, x = V_sliding_out_11_cast_fp16)[name = string("op_4604_cast_fp16")]; tensor var_4607_begin_0 = const()[name = string("op_4607_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_4607_end_0 = const()[name = string("op_4607_end_0"), val = tensor([7, 2, 512, 512])]; tensor var_4607_end_mask_0 = const()[name = string("op_4607_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4607_squeeze_mask_0 = const()[name = string("op_4607_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4607_cast_fp16 = slice_by_index(begin = var_4607_begin_0, end = var_4607_end_0, end_mask = var_4607_end_mask_0, squeeze_mask = var_4607_squeeze_mask_0, x = K_sliding_in)[name = string("op_4607_cast_fp16")]; tensor K_sliding_slot_13_axes_0 = const()[name = string("K_sliding_slot_13_axes_0"), val = tensor([0])]; tensor K_sliding_slot_13_cast_fp16 = expand_dims(axes = K_sliding_slot_13_axes_0, x = var_4607_cast_fp16)[name = string("K_sliding_slot_13_cast_fp16")]; tensor var_4612_begin_0 = const()[name = string("op_4612_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_4612_end_0 = const()[name = string("op_4612_end_0"), val = tensor([7, 2, 512, 512])]; tensor var_4612_end_mask_0 = const()[name = string("op_4612_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4612_squeeze_mask_0 = const()[name = string("op_4612_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_4612_cast_fp16 = slice_by_index(begin = var_4612_begin_0, end = var_4612_end_0, end_mask = var_4612_end_mask_0, squeeze_mask = var_4612_squeeze_mask_0, x = V_sliding_in)[name = string("op_4612_cast_fp16")]; tensor V_sliding_slot_13_axes_0 = const()[name = string("V_sliding_slot_13_axes_0"), val = tensor([0])]; tensor V_sliding_slot_13_cast_fp16 = expand_dims(axes = V_sliding_slot_13_axes_0, x = var_4612_cast_fp16)[name = string("V_sliding_slot_13_cast_fp16")]; int32 var_4619 = const()[name = string("op_4619"), val = int32(-1)]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4621_cast_fp16 = mul(x = x_139_cast_fp16, y = const_82_promoted_to_fp16)[name = string("op_4621_cast_fp16")]; bool input_207_interleave_0 = const()[name = string("input_207_interleave_0"), val = bool(false)]; tensor input_207_cast_fp16 = concat(axis = var_4619, interleave = input_207_interleave_0, values = (x_139_cast_fp16, var_4621_cast_fp16))[name = string("input_207_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_4616_to_fp16 = const()[name = string("op_4616_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_4616_to_fp16, x = input_207_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_4626_split_sizes_0 = const()[name = string("op_4626_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4626_axis_0 = const()[name = string("op_4626_axis_0"), val = int32(-1)]; tensor var_4626_cast_fp16_0, tensor var_4626_cast_fp16_1 = split(axis = var_4626_axis_0, split_sizes = var_4626_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_4626_cast_fp16")]; tensor layers_7_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554699392)))]; tensor h_43_cast_fp16 = mul(x = var_4626_cast_fp16_0, y = layers_7_input_layernorm_weight_promoted_to_fp16)[name = string("h_43_cast_fp16")]; tensor var_4632 = const()[name = string("op_4632"), val = tensor([0, 2, 1])]; tensor var_4635_axes_0 = const()[name = string("op_4635_axes_0"), val = tensor([2])]; tensor var_4633_cast_fp16 = transpose(perm = var_4632, x = h_43_cast_fp16)[name = string("transpose_89")]; tensor var_4635_cast_fp16 = expand_dims(axes = var_4635_axes_0, x = var_4633_cast_fp16)[name = string("op_4635_cast_fp16")]; string var_4651_pad_type_0 = const()[name = string("op_4651_pad_type_0"), val = string("valid")]; tensor var_4651_strides_0 = const()[name = string("op_4651_strides_0"), val = tensor([1, 1])]; tensor var_4651_pad_0 = const()[name = string("op_4651_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4651_dilations_0 = const()[name = string("op_4651_dilations_0"), val = tensor([1, 1])]; int32 var_4651_groups_0 = const()[name = string("op_4651_groups_0"), val = int32(1)]; tensor var_4651 = conv(dilations = var_4651_dilations_0, groups = var_4651_groups_0, pad = var_4651_pad_0, pad_type = var_4651_pad_type_0, strides = var_4651_strides_0, weight = layers_7_self_attn_q_proj_weight_palettized, x = var_4635_cast_fp16)[name = string("op_4651")]; tensor var_4656 = const()[name = string("op_4656"), val = tensor([1, 8, 256, 1])]; tensor var_4657 = reshape(shape = var_4656, x = var_4651)[name = string("op_4657")]; tensor var_4662 = const()[name = string("op_4662"), val = tensor([0, 1, 3, 2])]; tensor var_4672 = const()[name = string("op_4672"), val = tensor([1, 8, 256])]; tensor var_4663 = transpose(perm = var_4662, x = var_4657)[name = string("transpose_88")]; tensor x_141 = reshape(shape = var_4672, x = var_4663)[name = string("x_141")]; int32 var_4678 = const()[name = string("op_4678"), val = int32(-1)]; fp16 const_83_promoted = const()[name = string("const_83_promoted"), val = fp16(-0x1p+0)]; tensor var_4680 = mul(x = x_141, y = const_83_promoted)[name = string("op_4680")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211 = concat(axis = var_4678, interleave = input_211_interleave_0, values = (x_141, var_4680))[name = string("input_211")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_4675_to_fp16 = const()[name = string("op_4675_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_4675_to_fp16, x = input_211)[name = string("normed_201_cast_fp16")]; tensor var_4685_split_sizes_0 = const()[name = string("op_4685_split_sizes_0"), val = tensor([256, 256])]; int32 var_4685_axis_0 = const()[name = string("op_4685_axis_0"), val = int32(-1)]; tensor var_4685_0, tensor var_4685_1 = split(axis = var_4685_axis_0, split_sizes = var_4685_split_sizes_0, x = normed_201_cast_fp16)[name = string("op_4685")]; tensor var_4687 = mul(x = var_4685_0, y = layers_7_self_attn_q_norm_weight)[name = string("op_4687")]; tensor var_4692 = const()[name = string("op_4692"), val = tensor([1, 8, 1, 256])]; tensor q_59 = reshape(shape = var_4692, x = var_4687)[name = string("q_59")]; tensor var_4694_cast_fp16 = mul(x = q_59, y = cos_s)[name = string("op_4694_cast_fp16")]; tensor var_4695_split_sizes_0 = const()[name = string("op_4695_split_sizes_0"), val = tensor([128, 128])]; int32 var_4695_axis_0 = const()[name = string("op_4695_axis_0"), val = int32(-1)]; tensor var_4695_0, tensor var_4695_1 = split(axis = var_4695_axis_0, split_sizes = var_4695_split_sizes_0, x = q_59)[name = string("op_4695")]; fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; tensor var_4697 = mul(x = var_4695_1, y = const_84_promoted)[name = string("op_4697")]; int32 var_4699 = const()[name = string("op_4699"), val = int32(-1)]; bool var_4700_interleave_0 = const()[name = string("op_4700_interleave_0"), val = bool(false)]; tensor var_4700 = concat(axis = var_4699, interleave = var_4700_interleave_0, values = (var_4697, var_4695_0))[name = string("op_4700")]; tensor var_4701_cast_fp16 = mul(x = var_4700, y = sin_s)[name = string("op_4701_cast_fp16")]; tensor q_63_cast_fp16 = add(x = var_4694_cast_fp16, y = var_4701_cast_fp16)[name = string("q_63_cast_fp16")]; string var_4714_pad_type_0 = const()[name = string("op_4714_pad_type_0"), val = string("valid")]; tensor var_4714_strides_0 = const()[name = string("op_4714_strides_0"), val = tensor([1, 1])]; tensor var_4714_pad_0 = const()[name = string("op_4714_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4714_dilations_0 = const()[name = string("op_4714_dilations_0"), val = tensor([1, 1])]; int32 var_4714_groups_0 = const()[name = string("op_4714_groups_0"), val = int32(1)]; tensor var_4714 = conv(dilations = var_4714_dilations_0, groups = var_4714_groups_0, pad = var_4714_pad_0, pad_type = var_4714_pad_type_0, strides = var_4714_strides_0, weight = layers_7_self_attn_k_proj_weight_palettized, x = var_4635_cast_fp16)[name = string("op_4714")]; tensor var_4719 = const()[name = string("op_4719"), val = tensor([1, 2, 256, 1])]; tensor var_4720 = reshape(shape = var_4719, x = var_4714)[name = string("op_4720")]; tensor var_4725 = const()[name = string("op_4725"), val = tensor([0, 1, 3, 2])]; string var_4742_pad_type_0 = const()[name = string("op_4742_pad_type_0"), val = string("valid")]; tensor var_4742_strides_0 = const()[name = string("op_4742_strides_0"), val = tensor([1, 1])]; tensor var_4742_pad_0 = const()[name = string("op_4742_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4742_dilations_0 = const()[name = string("op_4742_dilations_0"), val = tensor([1, 1])]; int32 var_4742_groups_0 = const()[name = string("op_4742_groups_0"), val = int32(1)]; tensor var_4742 = conv(dilations = var_4742_dilations_0, groups = var_4742_groups_0, pad = var_4742_pad_0, pad_type = var_4742_pad_type_0, strides = var_4742_strides_0, weight = layers_7_self_attn_v_proj_weight_palettized, x = var_4635_cast_fp16)[name = string("op_4742")]; tensor var_4747 = const()[name = string("op_4747"), val = tensor([1, 2, 256, 1])]; tensor var_4748 = reshape(shape = var_4747, x = var_4742)[name = string("op_4748")]; tensor var_4753 = const()[name = string("op_4753"), val = tensor([0, 1, 3, 2])]; tensor var_4763 = const()[name = string("op_4763"), val = tensor([1, 2, 256])]; tensor var_4726 = transpose(perm = var_4725, x = var_4720)[name = string("transpose_87")]; tensor x_143 = reshape(shape = var_4763, x = var_4726)[name = string("x_143")]; int32 var_4769 = const()[name = string("op_4769"), val = int32(-1)]; fp16 const_85_promoted = const()[name = string("const_85_promoted"), val = fp16(-0x1p+0)]; tensor var_4771 = mul(x = x_143, y = const_85_promoted)[name = string("op_4771")]; bool input_213_interleave_0 = const()[name = string("input_213_interleave_0"), val = bool(false)]; tensor input_213 = concat(axis = var_4769, interleave = input_213_interleave_0, values = (x_143, var_4771))[name = string("input_213")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_4766_to_fp16 = const()[name = string("op_4766_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_4766_to_fp16, x = input_213)[name = string("normed_205_cast_fp16")]; tensor var_4776_split_sizes_0 = const()[name = string("op_4776_split_sizes_0"), val = tensor([256, 256])]; int32 var_4776_axis_0 = const()[name = string("op_4776_axis_0"), val = int32(-1)]; tensor var_4776_0, tensor var_4776_1 = split(axis = var_4776_axis_0, split_sizes = var_4776_split_sizes_0, x = normed_205_cast_fp16)[name = string("op_4776")]; tensor var_4778 = mul(x = var_4776_0, y = layers_7_self_attn_k_norm_weight)[name = string("op_4778")]; tensor var_4783 = const()[name = string("op_4783"), val = tensor([1, 2, 1, 256])]; tensor q_61 = reshape(shape = var_4783, x = var_4778)[name = string("q_61")]; fp16 var_4785_promoted = const()[name = string("op_4785_promoted"), val = fp16(0x1p+1)]; tensor var_4754 = transpose(perm = var_4753, x = var_4748)[name = string("transpose_86")]; tensor var_4786 = pow(x = var_4754, y = var_4785_promoted)[name = string("op_4786")]; tensor var_4791_axes_0 = const()[name = string("op_4791_axes_0"), val = tensor([-1])]; bool var_4791_keep_dims_0 = const()[name = string("op_4791_keep_dims_0"), val = bool(true)]; tensor var_4791 = reduce_mean(axes = var_4791_axes_0, keep_dims = var_4791_keep_dims_0, x = var_4786)[name = string("op_4791")]; fp16 var_4793_to_fp16 = const()[name = string("op_4793_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_15_cast_fp16 = add(x = var_4791, y = var_4793_to_fp16)[name = string("mean_sq_15_cast_fp16")]; fp32 var_4795_epsilon_0 = const()[name = string("op_4795_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_4795_cast_fp16 = rsqrt(epsilon = var_4795_epsilon_0, x = mean_sq_15_cast_fp16)[name = string("op_4795_cast_fp16")]; tensor input_217_cast_fp16 = mul(x = var_4754, y = var_4795_cast_fp16)[name = string("input_217_cast_fp16")]; tensor var_4797_cast_fp16 = mul(x = q_61, y = cos_s)[name = string("op_4797_cast_fp16")]; tensor var_4798_split_sizes_0 = const()[name = string("op_4798_split_sizes_0"), val = tensor([128, 128])]; int32 var_4798_axis_0 = const()[name = string("op_4798_axis_0"), val = int32(-1)]; tensor var_4798_0, tensor var_4798_1 = split(axis = var_4798_axis_0, split_sizes = var_4798_split_sizes_0, x = q_61)[name = string("op_4798")]; fp16 const_86_promoted = const()[name = string("const_86_promoted"), val = fp16(-0x1p+0)]; tensor var_4800 = mul(x = var_4798_1, y = const_86_promoted)[name = string("op_4800")]; int32 var_4802 = const()[name = string("op_4802"), val = int32(-1)]; bool var_4803_interleave_0 = const()[name = string("op_4803_interleave_0"), val = bool(false)]; tensor var_4803 = concat(axis = var_4802, interleave = var_4803_interleave_0, values = (var_4800, var_4798_0))[name = string("op_4803")]; tensor var_4804_cast_fp16 = mul(x = var_4803, y = sin_s)[name = string("op_4804_cast_fp16")]; tensor input_215_cast_fp16 = add(x = var_4797_cast_fp16, y = var_4804_cast_fp16)[name = string("input_215_cast_fp16")]; tensor k_padded_13_pad_0 = const()[name = string("k_padded_13_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_13_mode_0 = const()[name = string("k_padded_13_mode_0"), val = string("constant")]; fp16 const_87_to_fp16 = const()[name = string("const_87_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_13_cast_fp16 = pad(constant_val = const_87_to_fp16, mode = k_padded_13_mode_0, pad = k_padded_13_pad_0, x = input_215_cast_fp16)[name = string("k_padded_13_cast_fp16")]; tensor v_padded_13_pad_0 = const()[name = string("v_padded_13_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_13_mode_0 = const()[name = string("v_padded_13_mode_0"), val = string("constant")]; fp16 const_88_to_fp16 = const()[name = string("const_88_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_13_cast_fp16 = pad(constant_val = const_88_to_fp16, mode = v_padded_13_mode_0, pad = v_padded_13_pad_0, x = input_217_cast_fp16)[name = string("v_padded_13_cast_fp16")]; tensor var_4833_begin_0 = const()[name = string("op_4833_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_4833_end_0 = const()[name = string("op_4833_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_4833_end_mask_0 = const()[name = string("op_4833_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4833_cast_fp16 = slice_by_index(begin = var_4833_begin_0, end = var_4833_end_0, end_mask = var_4833_end_mask_0, x = K_sliding_slot_13_cast_fp16)[name = string("op_4833_cast_fp16")]; int32 var_4840 = const()[name = string("op_4840"), val = int32(2)]; bool K_sliding_out_13_interleave_0 = const()[name = string("K_sliding_out_13_interleave_0"), val = bool(false)]; tensor K_sliding_out_13_cast_fp16 = concat(axis = var_4840, interleave = K_sliding_out_13_interleave_0, values = (var_4833_cast_fp16, k_padded_13_cast_fp16))[name = string("K_sliding_out_13_cast_fp16")]; tensor var_4856_begin_0 = const()[name = string("op_4856_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_4856_end_0 = const()[name = string("op_4856_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_4856_end_mask_0 = const()[name = string("op_4856_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4856_cast_fp16 = slice_by_index(begin = var_4856_begin_0, end = var_4856_end_0, end_mask = var_4856_end_mask_0, x = V_sliding_slot_13_cast_fp16)[name = string("op_4856_cast_fp16")]; int32 var_4863 = const()[name = string("op_4863"), val = int32(2)]; bool V_sliding_out_13_interleave_0 = const()[name = string("V_sliding_out_13_interleave_0"), val = bool(false)]; tensor V_sliding_out_13_cast_fp16 = concat(axis = var_4863, interleave = V_sliding_out_13_interleave_0, values = (var_4856_cast_fp16, v_padded_13_cast_fp16))[name = string("V_sliding_out_13_cast_fp16")]; tensor K_for_attn_15_begin_0 = const()[name = string("K_for_attn_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_15_end_0 = const()[name = string("K_for_attn_15_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_15_end_mask_0 = const()[name = string("K_for_attn_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_15_cast_fp16 = slice_by_index(begin = K_for_attn_15_begin_0, end = K_for_attn_15_end_0, end_mask = K_for_attn_15_end_mask_0, x = K_sliding_out_13_cast_fp16)[name = string("K_for_attn_15_cast_fp16")]; tensor V_for_attn_15_begin_0 = const()[name = string("V_for_attn_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_15_end_0 = const()[name = string("V_for_attn_15_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_15_end_mask_0 = const()[name = string("V_for_attn_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_15_cast_fp16 = slice_by_index(begin = V_for_attn_15_begin_0, end = V_for_attn_15_end_0, end_mask = V_for_attn_15_end_mask_0, x = V_sliding_out_13_cast_fp16)[name = string("V_for_attn_15_cast_fp16")]; tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = K_for_attn_15_cast_fp16)[name = string("transpose_85")]; tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; tensor concat_28 = const()[name = string("concat_28"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_28, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_29 = const()[name = string("concat_29"), val = tensor([-1, 1, 512, 256])]; tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_84")]; tensor reshape_29_cast_fp16 = reshape(shape = concat_29, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor transpose_55_perm_0 = const()[name = string("transpose_55_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = V_for_attn_15_cast_fp16)[name = string("transpose_83")]; tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; tensor concat_30 = const()[name = string("concat_30"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_30, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_31 = const()[name = string("concat_31"), val = tensor([-1, 1, 512, 256])]; tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_82")]; tensor reshape_31_cast_fp16 = reshape(shape = concat_31, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; tensor V_expanded_15_perm_0 = const()[name = string("V_expanded_15_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_29_transpose_x_0 = const()[name = string("attn_weights_29_transpose_x_0"), val = bool(false)]; bool attn_weights_29_transpose_y_0 = const()[name = string("attn_weights_29_transpose_y_0"), val = bool(false)]; tensor transpose_55_cast_fp16 = transpose(perm = transpose_55_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_81")]; tensor attn_weights_29_cast_fp16 = matmul(transpose_x = attn_weights_29_transpose_x_0, transpose_y = attn_weights_29_transpose_y_0, x = q_63_cast_fp16, y = transpose_55_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; tensor x_147_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask_sliding)[name = string("x_147_cast_fp16")]; tensor reduce_max_7_axes_0 = const()[name = string("reduce_max_7_axes_0"), val = tensor([-1])]; bool reduce_max_7_keep_dims_0 = const()[name = string("reduce_max_7_keep_dims_0"), val = bool(true)]; tensor reduce_max_7 = reduce_max(axes = reduce_max_7_axes_0, keep_dims = reduce_max_7_keep_dims_0, x = x_147_cast_fp16)[name = string("reduce_max_7")]; tensor var_4904 = sub(x = x_147_cast_fp16, y = reduce_max_7)[name = string("op_4904")]; tensor var_4910 = exp(x = var_4904)[name = string("op_4910")]; tensor var_4920_axes_0 = const()[name = string("op_4920_axes_0"), val = tensor([-1])]; bool var_4920_keep_dims_0 = const()[name = string("op_4920_keep_dims_0"), val = bool(true)]; tensor var_4920 = reduce_sum(axes = var_4920_axes_0, keep_dims = var_4920_keep_dims_0, x = var_4910)[name = string("op_4920")]; tensor var_4926_cast_fp16 = real_div(x = var_4910, y = var_4920)[name = string("op_4926_cast_fp16")]; bool attn_output_43_transpose_x_0 = const()[name = string("attn_output_43_transpose_x_0"), val = bool(false)]; bool attn_output_43_transpose_y_0 = const()[name = string("attn_output_43_transpose_y_0"), val = bool(false)]; tensor V_expanded_15_cast_fp16 = transpose(perm = V_expanded_15_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_80")]; tensor attn_output_43_cast_fp16 = matmul(transpose_x = attn_output_43_transpose_x_0, transpose_y = attn_output_43_transpose_y_0, x = var_4926_cast_fp16, y = V_expanded_15_cast_fp16)[name = string("attn_output_43_cast_fp16")]; tensor var_4937 = const()[name = string("op_4937"), val = tensor([0, 2, 1, 3])]; tensor var_4944 = const()[name = string("op_4944"), val = tensor([1, 1, -1])]; tensor var_4938_cast_fp16 = transpose(perm = var_4937, x = attn_output_43_cast_fp16)[name = string("transpose_79")]; tensor attn_output_45_cast_fp16 = reshape(shape = var_4944, x = var_4938_cast_fp16)[name = string("attn_output_45_cast_fp16")]; tensor var_4949 = const()[name = string("op_4949"), val = tensor([0, 2, 1])]; string var_4965_pad_type_0 = const()[name = string("op_4965_pad_type_0"), val = string("valid")]; int32 var_4965_groups_0 = const()[name = string("op_4965_groups_0"), val = int32(1)]; tensor var_4965_strides_0 = const()[name = string("op_4965_strides_0"), val = tensor([1])]; tensor var_4965_pad_0 = const()[name = string("op_4965_pad_0"), val = tensor([0, 0])]; tensor var_4965_dilations_0 = const()[name = string("op_4965_dilations_0"), val = tensor([1])]; tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554704576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557326080))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4950_cast_fp16 = transpose(perm = var_4949, x = attn_output_45_cast_fp16)[name = string("transpose_78")]; tensor var_4965_cast_fp16 = conv(dilations = var_4965_dilations_0, groups = var_4965_groups_0, pad = var_4965_pad_0, pad_type = var_4965_pad_type_0, strides = var_4965_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_4950_cast_fp16)[name = string("op_4965_cast_fp16")]; tensor var_4969 = const()[name = string("op_4969"), val = tensor([0, 2, 1])]; int32 var_4975 = const()[name = string("op_4975"), val = int32(-1)]; fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_151_cast_fp16 = transpose(perm = var_4969, x = var_4965_cast_fp16)[name = string("transpose_77")]; tensor var_4977_cast_fp16 = mul(x = x_151_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_4977_cast_fp16")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221_cast_fp16 = concat(axis = var_4975, interleave = input_221_interleave_0, values = (x_151_cast_fp16, var_4977_cast_fp16))[name = string("input_221_cast_fp16")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_4972_to_fp16 = const()[name = string("op_4972_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_4972_to_fp16, x = input_221_cast_fp16)[name = string("normed_209_cast_fp16")]; tensor var_4982_split_sizes_0 = const()[name = string("op_4982_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4982_axis_0 = const()[name = string("op_4982_axis_0"), val = int32(-1)]; tensor var_4982_cast_fp16_0, tensor var_4982_cast_fp16_1 = split(axis = var_4982_axis_0, split_sizes = var_4982_split_sizes_0, x = normed_209_cast_fp16)[name = string("op_4982_cast_fp16")]; tensor layers_7_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557328704)))]; tensor attn_output_47_cast_fp16 = mul(x = var_4982_cast_fp16_0, y = layers_7_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_47_cast_fp16")]; tensor x_153_cast_fp16 = add(x = x_139_cast_fp16, y = attn_output_47_cast_fp16)[name = string("x_153_cast_fp16")]; int32 var_4991 = const()[name = string("op_4991"), val = int32(-1)]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4993_cast_fp16 = mul(x = x_153_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_4993_cast_fp16")]; bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; tensor input_223_cast_fp16 = concat(axis = var_4991, interleave = input_223_interleave_0, values = (x_153_cast_fp16, var_4993_cast_fp16))[name = string("input_223_cast_fp16")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_4988_to_fp16 = const()[name = string("op_4988_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_4988_to_fp16, x = input_223_cast_fp16)[name = string("normed_213_cast_fp16")]; tensor var_4998_split_sizes_0 = const()[name = string("op_4998_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4998_axis_0 = const()[name = string("op_4998_axis_0"), val = int32(-1)]; tensor var_4998_cast_fp16_0, tensor var_4998_cast_fp16_1 = split(axis = var_4998_axis_0, split_sizes = var_4998_split_sizes_0, x = normed_213_cast_fp16)[name = string("op_4998_cast_fp16")]; tensor layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557333888)))]; tensor h_45_cast_fp16 = mul(x = var_4998_cast_fp16_0, y = layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_45_cast_fp16")]; tensor var_5009 = const()[name = string("op_5009"), val = tensor([0, 2, 1])]; tensor input_225_axes_0 = const()[name = string("input_225_axes_0"), val = tensor([2])]; tensor var_5010 = transpose(perm = var_5009, x = h_45_cast_fp16)[name = string("transpose_76")]; tensor input_225 = expand_dims(axes = input_225_axes_0, x = var_5010)[name = string("input_225")]; string gate_29_pad_type_0 = const()[name = string("gate_29_pad_type_0"), val = string("valid")]; tensor gate_29_strides_0 = const()[name = string("gate_29_strides_0"), val = tensor([1, 1])]; tensor gate_29_pad_0 = const()[name = string("gate_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_29_dilations_0 = const()[name = string("gate_29_dilations_0"), val = tensor([1, 1])]; int32 gate_29_groups_0 = const()[name = string("gate_29_groups_0"), val = int32(1)]; tensor gate_29 = conv(dilations = gate_29_dilations_0, groups = gate_29_groups_0, pad = gate_29_pad_0, pad_type = gate_29_pad_type_0, strides = gate_29_strides_0, weight = layers_7_mlp_gate_proj_weight_palettized, x = input_225)[name = string("gate_29")]; string up_15_pad_type_0 = const()[name = string("up_15_pad_type_0"), val = string("valid")]; tensor up_15_strides_0 = const()[name = string("up_15_strides_0"), val = tensor([1, 1])]; tensor up_15_pad_0 = const()[name = string("up_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_15_dilations_0 = const()[name = string("up_15_dilations_0"), val = tensor([1, 1])]; int32 up_15_groups_0 = const()[name = string("up_15_groups_0"), val = int32(1)]; tensor up_15 = conv(dilations = up_15_dilations_0, groups = up_15_groups_0, pad = up_15_pad_0, pad_type = up_15_pad_type_0, strides = up_15_strides_0, weight = layers_7_mlp_up_proj_weight_palettized, x = input_225)[name = string("up_15")]; string gate_31_mode_0 = const()[name = string("gate_31_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_31 = gelu(mode = gate_31_mode_0, x = gate_29)[name = string("gate_31")]; tensor input_227 = mul(x = gate_31, y = up_15)[name = string("input_227")]; string mlp_out_15_pad_type_0 = const()[name = string("mlp_out_15_pad_type_0"), val = string("valid")]; tensor mlp_out_15_strides_0 = const()[name = string("mlp_out_15_strides_0"), val = tensor([1, 1])]; tensor mlp_out_15_pad_0 = const()[name = string("mlp_out_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_15_dilations_0 = const()[name = string("mlp_out_15_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_15_groups_0 = const()[name = string("mlp_out_15_groups_0"), val = int32(1)]; tensor mlp_out_15 = conv(dilations = mlp_out_15_dilations_0, groups = mlp_out_15_groups_0, pad = mlp_out_15_pad_0, pad_type = mlp_out_15_pad_type_0, strides = mlp_out_15_strides_0, weight = layers_7_mlp_down_proj_weight_palettized, x = input_227)[name = string("mlp_out_15")]; tensor var_5050_axes_0 = const()[name = string("op_5050_axes_0"), val = tensor([2])]; tensor var_5050 = squeeze(axes = var_5050_axes_0, x = mlp_out_15)[name = string("op_5050")]; tensor var_5054 = const()[name = string("op_5054"), val = tensor([0, 2, 1])]; int32 var_5060 = const()[name = string("op_5060"), val = int32(-1)]; fp16 const_91_promoted = const()[name = string("const_91_promoted"), val = fp16(-0x1p+0)]; tensor x_155 = transpose(perm = var_5054, x = var_5050)[name = string("transpose_75")]; tensor var_5062 = mul(x = x_155, y = const_91_promoted)[name = string("op_5062")]; bool input_229_interleave_0 = const()[name = string("input_229_interleave_0"), val = bool(false)]; tensor input_229 = concat(axis = var_5060, interleave = input_229_interleave_0, values = (x_155, var_5062))[name = string("input_229")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_5057_to_fp16 = const()[name = string("op_5057_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_5057_to_fp16, x = input_229)[name = string("normed_217_cast_fp16")]; tensor var_5067_split_sizes_0 = const()[name = string("op_5067_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5067_axis_0 = const()[name = string("op_5067_axis_0"), val = int32(-1)]; tensor var_5067_0, tensor var_5067_1 = split(axis = var_5067_axis_0, split_sizes = var_5067_split_sizes_0, x = normed_217_cast_fp16)[name = string("op_5067")]; tensor hidden_states_73 = mul(x = var_5067_0, y = layers_7_post_feedforward_layernorm_weight)[name = string("hidden_states_73")]; tensor hidden_states_75_cast_fp16 = add(x = x_153_cast_fp16, y = hidden_states_73)[name = string("hidden_states_75_cast_fp16")]; tensor per_layer_slice_15_begin_0 = const()[name = string("per_layer_slice_15_begin_0"), val = tensor([0, 0, 4864])]; tensor per_layer_slice_15_end_0 = const()[name = string("per_layer_slice_15_end_0"), val = tensor([1, 1, 5120])]; tensor per_layer_slice_15_end_mask_0 = const()[name = string("per_layer_slice_15_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_15_cast_fp16 = slice_by_index(begin = per_layer_slice_15_begin_0, end = per_layer_slice_15_end_0, end_mask = per_layer_slice_15_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_15_cast_fp16")]; tensor var_5095 = const()[name = string("op_5095"), val = tensor([0, 2, 1])]; tensor input_231_axes_0 = const()[name = string("input_231_axes_0"), val = tensor([2])]; tensor var_5096 = transpose(perm = var_5095, x = hidden_states_75_cast_fp16)[name = string("transpose_74")]; tensor input_231 = expand_dims(axes = input_231_axes_0, x = var_5096)[name = string("input_231")]; string gated_43_pad_type_0 = const()[name = string("gated_43_pad_type_0"), val = string("valid")]; tensor gated_43_strides_0 = const()[name = string("gated_43_strides_0"), val = tensor([1, 1])]; tensor gated_43_pad_0 = const()[name = string("gated_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_43_dilations_0 = const()[name = string("gated_43_dilations_0"), val = tensor([1, 1])]; int32 gated_43_groups_0 = const()[name = string("gated_43_groups_0"), val = int32(1)]; tensor gated_43 = conv(dilations = gated_43_dilations_0, groups = gated_43_groups_0, pad = gated_43_pad_0, pad_type = gated_43_pad_type_0, strides = gated_43_strides_0, weight = layers_7_per_layer_input_gate_weight_palettized, x = input_231)[name = string("gated_43")]; string gated_45_mode_0 = const()[name = string("gated_45_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_45 = gelu(mode = gated_45_mode_0, x = gated_43)[name = string("gated_45")]; tensor var_5115 = const()[name = string("op_5115"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_15_axes_0 = const()[name = string("per_layer_slice_conv_15_axes_0"), val = tensor([2])]; tensor var_5116_cast_fp16 = transpose(perm = var_5115, x = per_layer_slice_15_cast_fp16)[name = string("transpose_73")]; tensor per_layer_slice_conv_15_cast_fp16 = expand_dims(axes = per_layer_slice_conv_15_axes_0, x = var_5116_cast_fp16)[name = string("per_layer_slice_conv_15_cast_fp16")]; tensor input_233_cast_fp16 = mul(x = gated_45, y = per_layer_slice_conv_15_cast_fp16)[name = string("input_233_cast_fp16")]; string gated_47_pad_type_0 = const()[name = string("gated_47_pad_type_0"), val = string("valid")]; tensor gated_47_strides_0 = const()[name = string("gated_47_strides_0"), val = tensor([1, 1])]; tensor gated_47_pad_0 = const()[name = string("gated_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_47_dilations_0 = const()[name = string("gated_47_dilations_0"), val = tensor([1, 1])]; int32 gated_47_groups_0 = const()[name = string("gated_47_groups_0"), val = int32(1)]; tensor layers_7_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557339072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557666816))))[name = string("layers_7_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_47_cast_fp16 = conv(dilations = gated_47_dilations_0, groups = gated_47_groups_0, pad = gated_47_pad_0, pad_type = gated_47_pad_type_0, strides = gated_47_strides_0, weight = layers_7_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_233_cast_fp16)[name = string("gated_47_cast_fp16")]; tensor var_5132_axes_0 = const()[name = string("op_5132_axes_0"), val = tensor([2])]; tensor var_5132_cast_fp16 = squeeze(axes = var_5132_axes_0, x = gated_47_cast_fp16)[name = string("op_5132_cast_fp16")]; tensor var_5136 = const()[name = string("op_5136"), val = tensor([0, 2, 1])]; int32 var_5142 = const()[name = string("op_5142"), val = int32(-1)]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_157_cast_fp16 = transpose(perm = var_5136, x = var_5132_cast_fp16)[name = string("transpose_72")]; tensor var_5144_cast_fp16 = mul(x = x_157_cast_fp16, y = const_92_promoted_to_fp16)[name = string("op_5144_cast_fp16")]; bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; tensor input_235_cast_fp16 = concat(axis = var_5142, interleave = input_235_interleave_0, values = (x_157_cast_fp16, var_5144_cast_fp16))[name = string("input_235_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_5139_to_fp16 = const()[name = string("op_5139_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_5139_to_fp16, x = input_235_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor var_5149_split_sizes_0 = const()[name = string("op_5149_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5149_axis_0 = const()[name = string("op_5149_axis_0"), val = int32(-1)]; tensor var_5149_cast_fp16_0, tensor var_5149_cast_fp16_1 = split(axis = var_5149_axis_0, split_sizes = var_5149_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_5149_cast_fp16")]; tensor layers_7_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557669440)))]; tensor hidden_states_79_cast_fp16 = mul(x = var_5149_cast_fp16_0, y = layers_7_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_79_cast_fp16")]; tensor hidden_states_81_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_79_cast_fp16)[name = string("hidden_states_81_cast_fp16")]; tensor const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = tensor([0x1.06p-1])]; tensor x_159_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_93_promoted_to_fp16)[name = string("x_159_cast_fp16")]; tensor var_5161_axes_0 = const()[name = string("op_5161_axes_0"), val = tensor([0])]; tensor var_5161_cast_fp16 = squeeze(axes = var_5161_axes_0, x = K_sliding_out_13_cast_fp16)[name = string("op_5161_cast_fp16")]; tensor var_5163_axes_0 = const()[name = string("op_5163_axes_0"), val = tensor([0])]; tensor var_5163_cast_fp16 = squeeze(axes = var_5163_axes_0, x = V_sliding_out_13_cast_fp16)[name = string("op_5163_cast_fp16")]; tensor var_5166_begin_0 = const()[name = string("op_5166_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5166_end_0 = const()[name = string("op_5166_end_0"), val = tensor([8, 2, 512, 512])]; tensor var_5166_end_mask_0 = const()[name = string("op_5166_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5166_squeeze_mask_0 = const()[name = string("op_5166_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5166_cast_fp16 = slice_by_index(begin = var_5166_begin_0, end = var_5166_end_0, end_mask = var_5166_end_mask_0, squeeze_mask = var_5166_squeeze_mask_0, x = K_sliding_in)[name = string("op_5166_cast_fp16")]; tensor K_sliding_slot_15_axes_0 = const()[name = string("K_sliding_slot_15_axes_0"), val = tensor([0])]; tensor K_sliding_slot_15_cast_fp16 = expand_dims(axes = K_sliding_slot_15_axes_0, x = var_5166_cast_fp16)[name = string("K_sliding_slot_15_cast_fp16")]; tensor var_5171_begin_0 = const()[name = string("op_5171_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5171_end_0 = const()[name = string("op_5171_end_0"), val = tensor([8, 2, 512, 512])]; tensor var_5171_end_mask_0 = const()[name = string("op_5171_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5171_squeeze_mask_0 = const()[name = string("op_5171_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5171_cast_fp16 = slice_by_index(begin = var_5171_begin_0, end = var_5171_end_0, end_mask = var_5171_end_mask_0, squeeze_mask = var_5171_squeeze_mask_0, x = V_sliding_in)[name = string("op_5171_cast_fp16")]; tensor V_sliding_slot_15_axes_0 = const()[name = string("V_sliding_slot_15_axes_0"), val = tensor([0])]; tensor V_sliding_slot_15_cast_fp16 = expand_dims(axes = V_sliding_slot_15_axes_0, x = var_5171_cast_fp16)[name = string("V_sliding_slot_15_cast_fp16")]; int32 var_5178 = const()[name = string("op_5178"), val = int32(-1)]; fp16 const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5180_cast_fp16 = mul(x = x_159_cast_fp16, y = const_94_promoted_to_fp16)[name = string("op_5180_cast_fp16")]; bool input_237_interleave_0 = const()[name = string("input_237_interleave_0"), val = bool(false)]; tensor input_237_cast_fp16 = concat(axis = var_5178, interleave = input_237_interleave_0, values = (x_159_cast_fp16, var_5180_cast_fp16))[name = string("input_237_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_5175_to_fp16 = const()[name = string("op_5175_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_5175_to_fp16, x = input_237_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_5185_split_sizes_0 = const()[name = string("op_5185_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5185_axis_0 = const()[name = string("op_5185_axis_0"), val = int32(-1)]; tensor var_5185_cast_fp16_0, tensor var_5185_cast_fp16_1 = split(axis = var_5185_axis_0, split_sizes = var_5185_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_5185_cast_fp16")]; tensor layers_8_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_8_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557674624)))]; tensor h_49_cast_fp16 = mul(x = var_5185_cast_fp16_0, y = layers_8_input_layernorm_weight_promoted_to_fp16)[name = string("h_49_cast_fp16")]; tensor var_5191 = const()[name = string("op_5191"), val = tensor([0, 2, 1])]; tensor var_5194_axes_0 = const()[name = string("op_5194_axes_0"), val = tensor([2])]; tensor var_5192_cast_fp16 = transpose(perm = var_5191, x = h_49_cast_fp16)[name = string("transpose_71")]; tensor var_5194_cast_fp16 = expand_dims(axes = var_5194_axes_0, x = var_5192_cast_fp16)[name = string("op_5194_cast_fp16")]; string var_5210_pad_type_0 = const()[name = string("op_5210_pad_type_0"), val = string("valid")]; tensor var_5210_strides_0 = const()[name = string("op_5210_strides_0"), val = tensor([1, 1])]; tensor var_5210_pad_0 = const()[name = string("op_5210_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5210_dilations_0 = const()[name = string("op_5210_dilations_0"), val = tensor([1, 1])]; int32 var_5210_groups_0 = const()[name = string("op_5210_groups_0"), val = int32(1)]; tensor var_5210 = conv(dilations = var_5210_dilations_0, groups = var_5210_groups_0, pad = var_5210_pad_0, pad_type = var_5210_pad_type_0, strides = var_5210_strides_0, weight = layers_8_self_attn_q_proj_weight_palettized, x = var_5194_cast_fp16)[name = string("op_5210")]; tensor var_5215 = const()[name = string("op_5215"), val = tensor([1, 8, 256, 1])]; tensor var_5216 = reshape(shape = var_5215, x = var_5210)[name = string("op_5216")]; tensor var_5221 = const()[name = string("op_5221"), val = tensor([0, 1, 3, 2])]; tensor var_5231 = const()[name = string("op_5231"), val = tensor([1, 8, 256])]; tensor var_5222 = transpose(perm = var_5221, x = var_5216)[name = string("transpose_70")]; tensor x_161 = reshape(shape = var_5231, x = var_5222)[name = string("x_161")]; int32 var_5237 = const()[name = string("op_5237"), val = int32(-1)]; fp16 const_95_promoted = const()[name = string("const_95_promoted"), val = fp16(-0x1p+0)]; tensor var_5239 = mul(x = x_161, y = const_95_promoted)[name = string("op_5239")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241 = concat(axis = var_5237, interleave = input_241_interleave_0, values = (x_161, var_5239))[name = string("input_241")]; tensor normed_229_axes_0 = const()[name = string("normed_229_axes_0"), val = tensor([-1])]; fp16 var_5234_to_fp16 = const()[name = string("op_5234_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_229_cast_fp16 = layer_norm(axes = normed_229_axes_0, epsilon = var_5234_to_fp16, x = input_241)[name = string("normed_229_cast_fp16")]; tensor var_5244_split_sizes_0 = const()[name = string("op_5244_split_sizes_0"), val = tensor([256, 256])]; int32 var_5244_axis_0 = const()[name = string("op_5244_axis_0"), val = int32(-1)]; tensor var_5244_0, tensor var_5244_1 = split(axis = var_5244_axis_0, split_sizes = var_5244_split_sizes_0, x = normed_229_cast_fp16)[name = string("op_5244")]; tensor var_5251 = const()[name = string("op_5251"), val = tensor([1, 8, 1, 256])]; tensor q_67 = reshape(shape = var_5251, x = var_5244_0)[name = string("q_67")]; tensor var_5253_cast_fp16 = mul(x = q_67, y = cos_s)[name = string("op_5253_cast_fp16")]; tensor var_5254_split_sizes_0 = const()[name = string("op_5254_split_sizes_0"), val = tensor([128, 128])]; int32 var_5254_axis_0 = const()[name = string("op_5254_axis_0"), val = int32(-1)]; tensor var_5254_0, tensor var_5254_1 = split(axis = var_5254_axis_0, split_sizes = var_5254_split_sizes_0, x = q_67)[name = string("op_5254")]; fp16 const_96_promoted = const()[name = string("const_96_promoted"), val = fp16(-0x1p+0)]; tensor var_5256 = mul(x = var_5254_1, y = const_96_promoted)[name = string("op_5256")]; int32 var_5258 = const()[name = string("op_5258"), val = int32(-1)]; bool var_5259_interleave_0 = const()[name = string("op_5259_interleave_0"), val = bool(false)]; tensor var_5259 = concat(axis = var_5258, interleave = var_5259_interleave_0, values = (var_5256, var_5254_0))[name = string("op_5259")]; tensor var_5260_cast_fp16 = mul(x = var_5259, y = sin_s)[name = string("op_5260_cast_fp16")]; tensor q_71_cast_fp16 = add(x = var_5253_cast_fp16, y = var_5260_cast_fp16)[name = string("q_71_cast_fp16")]; string var_5273_pad_type_0 = const()[name = string("op_5273_pad_type_0"), val = string("valid")]; tensor var_5273_strides_0 = const()[name = string("op_5273_strides_0"), val = tensor([1, 1])]; tensor var_5273_pad_0 = const()[name = string("op_5273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5273_dilations_0 = const()[name = string("op_5273_dilations_0"), val = tensor([1, 1])]; int32 var_5273_groups_0 = const()[name = string("op_5273_groups_0"), val = int32(1)]; tensor var_5273 = conv(dilations = var_5273_dilations_0, groups = var_5273_groups_0, pad = var_5273_pad_0, pad_type = var_5273_pad_type_0, strides = var_5273_strides_0, weight = layers_8_self_attn_k_proj_weight_palettized, x = var_5194_cast_fp16)[name = string("op_5273")]; tensor var_5278 = const()[name = string("op_5278"), val = tensor([1, 2, 256, 1])]; tensor var_5279 = reshape(shape = var_5278, x = var_5273)[name = string("op_5279")]; tensor var_5284 = const()[name = string("op_5284"), val = tensor([0, 1, 3, 2])]; string var_5301_pad_type_0 = const()[name = string("op_5301_pad_type_0"), val = string("valid")]; tensor var_5301_strides_0 = const()[name = string("op_5301_strides_0"), val = tensor([1, 1])]; tensor var_5301_pad_0 = const()[name = string("op_5301_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5301_dilations_0 = const()[name = string("op_5301_dilations_0"), val = tensor([1, 1])]; int32 var_5301_groups_0 = const()[name = string("op_5301_groups_0"), val = int32(1)]; tensor var_5301 = conv(dilations = var_5301_dilations_0, groups = var_5301_groups_0, pad = var_5301_pad_0, pad_type = var_5301_pad_type_0, strides = var_5301_strides_0, weight = layers_8_self_attn_v_proj_weight_palettized, x = var_5194_cast_fp16)[name = string("op_5301")]; tensor var_5306 = const()[name = string("op_5306"), val = tensor([1, 2, 256, 1])]; tensor var_5307 = reshape(shape = var_5306, x = var_5301)[name = string("op_5307")]; tensor var_5312 = const()[name = string("op_5312"), val = tensor([0, 1, 3, 2])]; tensor var_5322 = const()[name = string("op_5322"), val = tensor([1, 2, 256])]; tensor var_5285 = transpose(perm = var_5284, x = var_5279)[name = string("transpose_69")]; tensor x_163 = reshape(shape = var_5322, x = var_5285)[name = string("x_163")]; int32 var_5328 = const()[name = string("op_5328"), val = int32(-1)]; fp16 const_97_promoted = const()[name = string("const_97_promoted"), val = fp16(-0x1p+0)]; tensor var_5330 = mul(x = x_163, y = const_97_promoted)[name = string("op_5330")]; bool input_243_interleave_0 = const()[name = string("input_243_interleave_0"), val = bool(false)]; tensor input_243 = concat(axis = var_5328, interleave = input_243_interleave_0, values = (x_163, var_5330))[name = string("input_243")]; tensor normed_233_axes_0 = const()[name = string("normed_233_axes_0"), val = tensor([-1])]; fp16 var_5325_to_fp16 = const()[name = string("op_5325_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_233_cast_fp16 = layer_norm(axes = normed_233_axes_0, epsilon = var_5325_to_fp16, x = input_243)[name = string("normed_233_cast_fp16")]; tensor var_5335_split_sizes_0 = const()[name = string("op_5335_split_sizes_0"), val = tensor([256, 256])]; int32 var_5335_axis_0 = const()[name = string("op_5335_axis_0"), val = int32(-1)]; tensor var_5335_0, tensor var_5335_1 = split(axis = var_5335_axis_0, split_sizes = var_5335_split_sizes_0, x = normed_233_cast_fp16)[name = string("op_5335")]; tensor var_5337 = mul(x = var_5335_0, y = layers_8_self_attn_k_norm_weight)[name = string("op_5337")]; tensor var_5342 = const()[name = string("op_5342"), val = tensor([1, 2, 1, 256])]; tensor q_69 = reshape(shape = var_5342, x = var_5337)[name = string("q_69")]; fp16 var_5344_promoted = const()[name = string("op_5344_promoted"), val = fp16(0x1p+1)]; tensor var_5313 = transpose(perm = var_5312, x = var_5307)[name = string("transpose_68")]; tensor var_5345 = pow(x = var_5313, y = var_5344_promoted)[name = string("op_5345")]; tensor var_5350_axes_0 = const()[name = string("op_5350_axes_0"), val = tensor([-1])]; bool var_5350_keep_dims_0 = const()[name = string("op_5350_keep_dims_0"), val = bool(true)]; tensor var_5350 = reduce_mean(axes = var_5350_axes_0, keep_dims = var_5350_keep_dims_0, x = var_5345)[name = string("op_5350")]; fp16 var_5352_to_fp16 = const()[name = string("op_5352_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_17_cast_fp16 = add(x = var_5350, y = var_5352_to_fp16)[name = string("mean_sq_17_cast_fp16")]; fp32 var_5354_epsilon_0 = const()[name = string("op_5354_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_5354_cast_fp16 = rsqrt(epsilon = var_5354_epsilon_0, x = mean_sq_17_cast_fp16)[name = string("op_5354_cast_fp16")]; tensor input_247_cast_fp16 = mul(x = var_5313, y = var_5354_cast_fp16)[name = string("input_247_cast_fp16")]; tensor var_5356_cast_fp16 = mul(x = q_69, y = cos_s)[name = string("op_5356_cast_fp16")]; tensor var_5357_split_sizes_0 = const()[name = string("op_5357_split_sizes_0"), val = tensor([128, 128])]; int32 var_5357_axis_0 = const()[name = string("op_5357_axis_0"), val = int32(-1)]; tensor var_5357_0, tensor var_5357_1 = split(axis = var_5357_axis_0, split_sizes = var_5357_split_sizes_0, x = q_69)[name = string("op_5357")]; fp16 const_98_promoted = const()[name = string("const_98_promoted"), val = fp16(-0x1p+0)]; tensor var_5359 = mul(x = var_5357_1, y = const_98_promoted)[name = string("op_5359")]; int32 var_5361 = const()[name = string("op_5361"), val = int32(-1)]; bool var_5362_interleave_0 = const()[name = string("op_5362_interleave_0"), val = bool(false)]; tensor var_5362 = concat(axis = var_5361, interleave = var_5362_interleave_0, values = (var_5359, var_5357_0))[name = string("op_5362")]; tensor var_5363_cast_fp16 = mul(x = var_5362, y = sin_s)[name = string("op_5363_cast_fp16")]; tensor input_245_cast_fp16 = add(x = var_5356_cast_fp16, y = var_5363_cast_fp16)[name = string("input_245_cast_fp16")]; tensor k_padded_15_pad_0 = const()[name = string("k_padded_15_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_15_mode_0 = const()[name = string("k_padded_15_mode_0"), val = string("constant")]; fp16 const_99_to_fp16 = const()[name = string("const_99_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_15_cast_fp16 = pad(constant_val = const_99_to_fp16, mode = k_padded_15_mode_0, pad = k_padded_15_pad_0, x = input_245_cast_fp16)[name = string("k_padded_15_cast_fp16")]; tensor v_padded_15_pad_0 = const()[name = string("v_padded_15_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_15_mode_0 = const()[name = string("v_padded_15_mode_0"), val = string("constant")]; fp16 const_100_to_fp16 = const()[name = string("const_100_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_15_cast_fp16 = pad(constant_val = const_100_to_fp16, mode = v_padded_15_mode_0, pad = v_padded_15_pad_0, x = input_247_cast_fp16)[name = string("v_padded_15_cast_fp16")]; tensor var_5392_begin_0 = const()[name = string("op_5392_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_5392_end_0 = const()[name = string("op_5392_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5392_end_mask_0 = const()[name = string("op_5392_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5392_cast_fp16 = slice_by_index(begin = var_5392_begin_0, end = var_5392_end_0, end_mask = var_5392_end_mask_0, x = K_sliding_slot_15_cast_fp16)[name = string("op_5392_cast_fp16")]; int32 var_5399 = const()[name = string("op_5399"), val = int32(2)]; bool K_sliding_out_15_interleave_0 = const()[name = string("K_sliding_out_15_interleave_0"), val = bool(false)]; tensor K_sliding_out_15_cast_fp16 = concat(axis = var_5399, interleave = K_sliding_out_15_interleave_0, values = (var_5392_cast_fp16, k_padded_15_cast_fp16))[name = string("K_sliding_out_15_cast_fp16")]; tensor var_5415_begin_0 = const()[name = string("op_5415_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_5415_end_0 = const()[name = string("op_5415_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5415_end_mask_0 = const()[name = string("op_5415_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5415_cast_fp16 = slice_by_index(begin = var_5415_begin_0, end = var_5415_end_0, end_mask = var_5415_end_mask_0, x = V_sliding_slot_15_cast_fp16)[name = string("op_5415_cast_fp16")]; int32 var_5422 = const()[name = string("op_5422"), val = int32(2)]; bool V_sliding_out_15_interleave_0 = const()[name = string("V_sliding_out_15_interleave_0"), val = bool(false)]; tensor V_sliding_out_15_cast_fp16 = concat(axis = var_5422, interleave = V_sliding_out_15_interleave_0, values = (var_5415_cast_fp16, v_padded_15_cast_fp16))[name = string("V_sliding_out_15_cast_fp16")]; tensor K_for_attn_17_begin_0 = const()[name = string("K_for_attn_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_17_end_0 = const()[name = string("K_for_attn_17_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_17_end_mask_0 = const()[name = string("K_for_attn_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_17_cast_fp16 = slice_by_index(begin = K_for_attn_17_begin_0, end = K_for_attn_17_end_0, end_mask = K_for_attn_17_end_mask_0, x = K_sliding_out_15_cast_fp16)[name = string("K_for_attn_17_cast_fp16")]; tensor V_for_attn_17_begin_0 = const()[name = string("V_for_attn_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_17_end_0 = const()[name = string("V_for_attn_17_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_17_end_mask_0 = const()[name = string("V_for_attn_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_17_cast_fp16 = slice_by_index(begin = V_for_attn_17_begin_0, end = V_for_attn_17_end_0, end_mask = V_for_attn_17_end_mask_0, x = V_sliding_out_15_cast_fp16)[name = string("V_for_attn_17_cast_fp16")]; tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_16_reps_0 = const()[name = string("tile_16_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = K_for_attn_17_cast_fp16)[name = string("transpose_67")]; tensor tile_16_cast_fp16 = tile(reps = tile_16_reps_0, x = transpose_32_cast_fp16)[name = string("tile_16_cast_fp16")]; tensor concat_32 = const()[name = string("concat_32"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_32_cast_fp16 = reshape(shape = concat_32, x = tile_16_cast_fp16)[name = string("reshape_32_cast_fp16")]; tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_33 = const()[name = string("concat_33"), val = tensor([-1, 1, 512, 256])]; tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_66")]; tensor reshape_33_cast_fp16 = reshape(shape = concat_33, x = transpose_33_cast_fp16)[name = string("reshape_33_cast_fp16")]; tensor transpose_56_perm_0 = const()[name = string("transpose_56_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_17_reps_0 = const()[name = string("tile_17_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = V_for_attn_17_cast_fp16)[name = string("transpose_65")]; tensor tile_17_cast_fp16 = tile(reps = tile_17_reps_0, x = transpose_34_cast_fp16)[name = string("tile_17_cast_fp16")]; tensor concat_34 = const()[name = string("concat_34"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_34_cast_fp16 = reshape(shape = concat_34, x = tile_17_cast_fp16)[name = string("reshape_34_cast_fp16")]; tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([-1, 1, 512, 256])]; tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_34_cast_fp16)[name = string("transpose_64")]; tensor reshape_35_cast_fp16 = reshape(shape = concat_35, x = transpose_35_cast_fp16)[name = string("reshape_35_cast_fp16")]; tensor V_expanded_17_perm_0 = const()[name = string("V_expanded_17_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor transpose_56_cast_fp16 = transpose(perm = transpose_56_perm_0, x = reshape_33_cast_fp16)[name = string("transpose_63")]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = q_71_cast_fp16, y = transpose_56_cast_fp16)[name = string("attn_weights_33_cast_fp16")]; tensor x_167_cast_fp16 = add(x = attn_weights_33_cast_fp16, y = causal_mask_sliding)[name = string("x_167_cast_fp16")]; tensor reduce_max_8_axes_0 = const()[name = string("reduce_max_8_axes_0"), val = tensor([-1])]; bool reduce_max_8_keep_dims_0 = const()[name = string("reduce_max_8_keep_dims_0"), val = bool(true)]; tensor reduce_max_8 = reduce_max(axes = reduce_max_8_axes_0, keep_dims = reduce_max_8_keep_dims_0, x = x_167_cast_fp16)[name = string("reduce_max_8")]; tensor var_5463 = sub(x = x_167_cast_fp16, y = reduce_max_8)[name = string("op_5463")]; tensor var_5469 = exp(x = var_5463)[name = string("op_5469")]; tensor var_5479_axes_0 = const()[name = string("op_5479_axes_0"), val = tensor([-1])]; bool var_5479_keep_dims_0 = const()[name = string("op_5479_keep_dims_0"), val = bool(true)]; tensor var_5479 = reduce_sum(axes = var_5479_axes_0, keep_dims = var_5479_keep_dims_0, x = var_5469)[name = string("op_5479")]; tensor var_5485_cast_fp16 = real_div(x = var_5469, y = var_5479)[name = string("op_5485_cast_fp16")]; bool attn_output_49_transpose_x_0 = const()[name = string("attn_output_49_transpose_x_0"), val = bool(false)]; bool attn_output_49_transpose_y_0 = const()[name = string("attn_output_49_transpose_y_0"), val = bool(false)]; tensor V_expanded_17_cast_fp16 = transpose(perm = V_expanded_17_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_62")]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = var_5485_cast_fp16, y = V_expanded_17_cast_fp16)[name = string("attn_output_49_cast_fp16")]; tensor var_5496 = const()[name = string("op_5496"), val = tensor([0, 2, 1, 3])]; tensor var_5503 = const()[name = string("op_5503"), val = tensor([1, 1, -1])]; tensor var_5497_cast_fp16 = transpose(perm = var_5496, x = attn_output_49_cast_fp16)[name = string("transpose_61")]; tensor attn_output_51_cast_fp16 = reshape(shape = var_5503, x = var_5497_cast_fp16)[name = string("attn_output_51_cast_fp16")]; tensor var_5508 = const()[name = string("op_5508"), val = tensor([0, 2, 1])]; string var_5524_pad_type_0 = const()[name = string("op_5524_pad_type_0"), val = string("valid")]; int32 var_5524_groups_0 = const()[name = string("op_5524_groups_0"), val = int32(1)]; tensor var_5524_strides_0 = const()[name = string("op_5524_strides_0"), val = tensor([1])]; tensor var_5524_pad_0 = const()[name = string("op_5524_pad_0"), val = tensor([0, 0])]; tensor var_5524_dilations_0 = const()[name = string("op_5524_dilations_0"), val = tensor([1])]; tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557679808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560301312))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5509_cast_fp16 = transpose(perm = var_5508, x = attn_output_51_cast_fp16)[name = string("transpose_60")]; tensor var_5524_cast_fp16 = conv(dilations = var_5524_dilations_0, groups = var_5524_groups_0, pad = var_5524_pad_0, pad_type = var_5524_pad_type_0, strides = var_5524_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5509_cast_fp16)[name = string("op_5524_cast_fp16")]; tensor var_5528 = const()[name = string("op_5528"), val = tensor([0, 2, 1])]; int32 var_5534 = const()[name = string("op_5534"), val = int32(-1)]; fp16 const_101_promoted_to_fp16 = const()[name = string("const_101_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_171_cast_fp16 = transpose(perm = var_5528, x = var_5524_cast_fp16)[name = string("transpose_59")]; tensor var_5536_cast_fp16 = mul(x = x_171_cast_fp16, y = const_101_promoted_to_fp16)[name = string("op_5536_cast_fp16")]; bool input_251_interleave_0 = const()[name = string("input_251_interleave_0"), val = bool(false)]; tensor input_251_cast_fp16 = concat(axis = var_5534, interleave = input_251_interleave_0, values = (x_171_cast_fp16, var_5536_cast_fp16))[name = string("input_251_cast_fp16")]; tensor normed_237_axes_0 = const()[name = string("normed_237_axes_0"), val = tensor([-1])]; fp16 var_5531_to_fp16 = const()[name = string("op_5531_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_237_cast_fp16 = layer_norm(axes = normed_237_axes_0, epsilon = var_5531_to_fp16, x = input_251_cast_fp16)[name = string("normed_237_cast_fp16")]; tensor var_5541_split_sizes_0 = const()[name = string("op_5541_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5541_axis_0 = const()[name = string("op_5541_axis_0"), val = int32(-1)]; tensor var_5541_cast_fp16_0, tensor var_5541_cast_fp16_1 = split(axis = var_5541_axis_0, split_sizes = var_5541_split_sizes_0, x = normed_237_cast_fp16)[name = string("op_5541_cast_fp16")]; tensor layers_8_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_8_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560303936)))]; tensor attn_output_53_cast_fp16 = mul(x = var_5541_cast_fp16_0, y = layers_8_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_53_cast_fp16")]; tensor x_173_cast_fp16 = add(x = x_159_cast_fp16, y = attn_output_53_cast_fp16)[name = string("x_173_cast_fp16")]; int32 var_5550 = const()[name = string("op_5550"), val = int32(-1)]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5552_cast_fp16 = mul(x = x_173_cast_fp16, y = const_102_promoted_to_fp16)[name = string("op_5552_cast_fp16")]; bool input_253_interleave_0 = const()[name = string("input_253_interleave_0"), val = bool(false)]; tensor input_253_cast_fp16 = concat(axis = var_5550, interleave = input_253_interleave_0, values = (x_173_cast_fp16, var_5552_cast_fp16))[name = string("input_253_cast_fp16")]; tensor normed_241_axes_0 = const()[name = string("normed_241_axes_0"), val = tensor([-1])]; fp16 var_5547_to_fp16 = const()[name = string("op_5547_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_241_cast_fp16 = layer_norm(axes = normed_241_axes_0, epsilon = var_5547_to_fp16, x = input_253_cast_fp16)[name = string("normed_241_cast_fp16")]; tensor var_5557_split_sizes_0 = const()[name = string("op_5557_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5557_axis_0 = const()[name = string("op_5557_axis_0"), val = int32(-1)]; tensor var_5557_cast_fp16_0, tensor var_5557_cast_fp16_1 = split(axis = var_5557_axis_0, split_sizes = var_5557_split_sizes_0, x = normed_241_cast_fp16)[name = string("op_5557_cast_fp16")]; tensor layers_8_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_8_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560309120)))]; tensor h_51_cast_fp16 = mul(x = var_5557_cast_fp16_0, y = layers_8_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_51_cast_fp16")]; tensor var_5568 = const()[name = string("op_5568"), val = tensor([0, 2, 1])]; tensor input_255_axes_0 = const()[name = string("input_255_axes_0"), val = tensor([2])]; tensor var_5569 = transpose(perm = var_5568, x = h_51_cast_fp16)[name = string("transpose_58")]; tensor input_255 = expand_dims(axes = input_255_axes_0, x = var_5569)[name = string("input_255")]; string gate_33_pad_type_0 = const()[name = string("gate_33_pad_type_0"), val = string("valid")]; tensor gate_33_strides_0 = const()[name = string("gate_33_strides_0"), val = tensor([1, 1])]; tensor gate_33_pad_0 = const()[name = string("gate_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_33_dilations_0 = const()[name = string("gate_33_dilations_0"), val = tensor([1, 1])]; int32 gate_33_groups_0 = const()[name = string("gate_33_groups_0"), val = int32(1)]; tensor gate_33 = conv(dilations = gate_33_dilations_0, groups = gate_33_groups_0, pad = gate_33_pad_0, pad_type = gate_33_pad_type_0, strides = gate_33_strides_0, weight = layers_8_mlp_gate_proj_weight_palettized, x = input_255)[name = string("gate_33")]; string up_17_pad_type_0 = const()[name = string("up_17_pad_type_0"), val = string("valid")]; tensor up_17_strides_0 = const()[name = string("up_17_strides_0"), val = tensor([1, 1])]; tensor up_17_pad_0 = const()[name = string("up_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_17_dilations_0 = const()[name = string("up_17_dilations_0"), val = tensor([1, 1])]; int32 up_17_groups_0 = const()[name = string("up_17_groups_0"), val = int32(1)]; tensor up_17 = conv(dilations = up_17_dilations_0, groups = up_17_groups_0, pad = up_17_pad_0, pad_type = up_17_pad_type_0, strides = up_17_strides_0, weight = layers_8_mlp_up_proj_weight_palettized, x = input_255)[name = string("up_17")]; string gate_35_mode_0 = const()[name = string("gate_35_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_35 = gelu(mode = gate_35_mode_0, x = gate_33)[name = string("gate_35")]; tensor input_257 = mul(x = gate_35, y = up_17)[name = string("input_257")]; string mlp_out_17_pad_type_0 = const()[name = string("mlp_out_17_pad_type_0"), val = string("valid")]; tensor mlp_out_17_strides_0 = const()[name = string("mlp_out_17_strides_0"), val = tensor([1, 1])]; tensor mlp_out_17_pad_0 = const()[name = string("mlp_out_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_17_dilations_0 = const()[name = string("mlp_out_17_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_17_groups_0 = const()[name = string("mlp_out_17_groups_0"), val = int32(1)]; tensor mlp_out_17 = conv(dilations = mlp_out_17_dilations_0, groups = mlp_out_17_groups_0, pad = mlp_out_17_pad_0, pad_type = mlp_out_17_pad_type_0, strides = mlp_out_17_strides_0, weight = layers_8_mlp_down_proj_weight_palettized, x = input_257)[name = string("mlp_out_17")]; tensor var_5609_axes_0 = const()[name = string("op_5609_axes_0"), val = tensor([2])]; tensor var_5609 = squeeze(axes = var_5609_axes_0, x = mlp_out_17)[name = string("op_5609")]; tensor var_5613 = const()[name = string("op_5613"), val = tensor([0, 2, 1])]; int32 var_5619 = const()[name = string("op_5619"), val = int32(-1)]; fp16 const_103_promoted = const()[name = string("const_103_promoted"), val = fp16(-0x1p+0)]; tensor x_175 = transpose(perm = var_5613, x = var_5609)[name = string("transpose_57")]; tensor var_5621 = mul(x = x_175, y = const_103_promoted)[name = string("op_5621")]; bool input_259_interleave_0 = const()[name = string("input_259_interleave_0"), val = bool(false)]; tensor input_259 = concat(axis = var_5619, interleave = input_259_interleave_0, values = (x_175, var_5621))[name = string("input_259")]; tensor normed_245_axes_0 = const()[name = string("normed_245_axes_0"), val = tensor([-1])]; fp16 var_5616_to_fp16 = const()[name = string("op_5616_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_245_cast_fp16 = layer_norm(axes = normed_245_axes_0, epsilon = var_5616_to_fp16, x = input_259)[name = string("normed_245_cast_fp16")]; tensor var_5626_split_sizes_0 = const()[name = string("op_5626_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5626_axis_0 = const()[name = string("op_5626_axis_0"), val = int32(-1)]; tensor var_5626_0, tensor var_5626_1 = split(axis = var_5626_axis_0, split_sizes = var_5626_split_sizes_0, x = normed_245_cast_fp16)[name = string("op_5626")]; tensor hidden_states_83 = mul(x = var_5626_0, y = layers_8_post_feedforward_layernorm_weight)[name = string("hidden_states_83")]; tensor hidden_states_85_cast_fp16 = add(x = x_173_cast_fp16, y = hidden_states_83)[name = string("hidden_states_85_cast_fp16")]; tensor per_layer_slice_17_begin_0 = const()[name = string("per_layer_slice_17_begin_0"), val = tensor([0, 0, 5120])]; tensor per_layer_slice_17_end_0 = const()[name = string("per_layer_slice_17_end_0"), val = tensor([1, 1, 5376])]; tensor per_layer_slice_17_end_mask_0 = const()[name = string("per_layer_slice_17_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_17_cast_fp16 = slice_by_index(begin = per_layer_slice_17_begin_0, end = per_layer_slice_17_end_0, end_mask = per_layer_slice_17_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_17_cast_fp16")]; tensor var_5654 = const()[name = string("op_5654"), val = tensor([0, 2, 1])]; tensor input_261_axes_0 = const()[name = string("input_261_axes_0"), val = tensor([2])]; tensor var_5655 = transpose(perm = var_5654, x = hidden_states_85_cast_fp16)[name = string("transpose_56")]; tensor input_261 = expand_dims(axes = input_261_axes_0, x = var_5655)[name = string("input_261")]; string gated_49_pad_type_0 = const()[name = string("gated_49_pad_type_0"), val = string("valid")]; tensor gated_49_strides_0 = const()[name = string("gated_49_strides_0"), val = tensor([1, 1])]; tensor gated_49_pad_0 = const()[name = string("gated_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_49_dilations_0 = const()[name = string("gated_49_dilations_0"), val = tensor([1, 1])]; int32 gated_49_groups_0 = const()[name = string("gated_49_groups_0"), val = int32(1)]; tensor gated_49 = conv(dilations = gated_49_dilations_0, groups = gated_49_groups_0, pad = gated_49_pad_0, pad_type = gated_49_pad_type_0, strides = gated_49_strides_0, weight = layers_8_per_layer_input_gate_weight_palettized, x = input_261)[name = string("gated_49")]; string gated_51_mode_0 = const()[name = string("gated_51_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_51 = gelu(mode = gated_51_mode_0, x = gated_49)[name = string("gated_51")]; tensor var_5674 = const()[name = string("op_5674"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_17_axes_0 = const()[name = string("per_layer_slice_conv_17_axes_0"), val = tensor([2])]; tensor var_5675_cast_fp16 = transpose(perm = var_5674, x = per_layer_slice_17_cast_fp16)[name = string("transpose_55")]; tensor per_layer_slice_conv_17_cast_fp16 = expand_dims(axes = per_layer_slice_conv_17_axes_0, x = var_5675_cast_fp16)[name = string("per_layer_slice_conv_17_cast_fp16")]; tensor input_263_cast_fp16 = mul(x = gated_51, y = per_layer_slice_conv_17_cast_fp16)[name = string("input_263_cast_fp16")]; string gated_53_pad_type_0 = const()[name = string("gated_53_pad_type_0"), val = string("valid")]; tensor gated_53_strides_0 = const()[name = string("gated_53_strides_0"), val = tensor([1, 1])]; tensor gated_53_pad_0 = const()[name = string("gated_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_53_dilations_0 = const()[name = string("gated_53_dilations_0"), val = tensor([1, 1])]; int32 gated_53_groups_0 = const()[name = string("gated_53_groups_0"), val = int32(1)]; tensor layers_8_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560314304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560642048))))[name = string("layers_8_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_53_cast_fp16 = conv(dilations = gated_53_dilations_0, groups = gated_53_groups_0, pad = gated_53_pad_0, pad_type = gated_53_pad_type_0, strides = gated_53_strides_0, weight = layers_8_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_263_cast_fp16)[name = string("gated_53_cast_fp16")]; tensor var_5691_axes_0 = const()[name = string("op_5691_axes_0"), val = tensor([2])]; tensor var_5691_cast_fp16 = squeeze(axes = var_5691_axes_0, x = gated_53_cast_fp16)[name = string("op_5691_cast_fp16")]; tensor var_5695 = const()[name = string("op_5695"), val = tensor([0, 2, 1])]; int32 var_5701 = const()[name = string("op_5701"), val = int32(-1)]; fp16 const_104_promoted_to_fp16 = const()[name = string("const_104_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_177_cast_fp16 = transpose(perm = var_5695, x = var_5691_cast_fp16)[name = string("transpose_54")]; tensor var_5703_cast_fp16 = mul(x = x_177_cast_fp16, y = const_104_promoted_to_fp16)[name = string("op_5703_cast_fp16")]; bool input_265_interleave_0 = const()[name = string("input_265_interleave_0"), val = bool(false)]; tensor input_265_cast_fp16 = concat(axis = var_5701, interleave = input_265_interleave_0, values = (x_177_cast_fp16, var_5703_cast_fp16))[name = string("input_265_cast_fp16")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; fp16 var_5698_to_fp16 = const()[name = string("op_5698_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_5698_to_fp16, x = input_265_cast_fp16)[name = string("normed_249_cast_fp16")]; tensor var_5708_split_sizes_0 = const()[name = string("op_5708_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5708_axis_0 = const()[name = string("op_5708_axis_0"), val = int32(-1)]; tensor var_5708_cast_fp16_0, tensor var_5708_cast_fp16_1 = split(axis = var_5708_axis_0, split_sizes = var_5708_split_sizes_0, x = normed_249_cast_fp16)[name = string("op_5708_cast_fp16")]; tensor layers_8_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_8_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560644672)))]; tensor hidden_states_89_cast_fp16 = mul(x = var_5708_cast_fp16_0, y = layers_8_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_89_cast_fp16")]; tensor hidden_states_91_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_89_cast_fp16)[name = string("hidden_states_91_cast_fp16")]; tensor const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = tensor([0x1.bap-2])]; tensor x_179_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_105_promoted_to_fp16)[name = string("x_179_cast_fp16")]; tensor var_5720_axes_0 = const()[name = string("op_5720_axes_0"), val = tensor([0])]; tensor var_5720_cast_fp16 = squeeze(axes = var_5720_axes_0, x = K_sliding_out_15_cast_fp16)[name = string("op_5720_cast_fp16")]; tensor var_5722_axes_0 = const()[name = string("op_5722_axes_0"), val = tensor([0])]; tensor var_5722_cast_fp16 = squeeze(axes = var_5722_axes_0, x = V_sliding_out_15_cast_fp16)[name = string("op_5722_cast_fp16")]; tensor var_5725_begin_0 = const()[name = string("op_5725_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_5725_end_0 = const()[name = string("op_5725_end_0"), val = tensor([9, 2, 512, 512])]; tensor var_5725_end_mask_0 = const()[name = string("op_5725_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5725_squeeze_mask_0 = const()[name = string("op_5725_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5725_cast_fp16 = slice_by_index(begin = var_5725_begin_0, end = var_5725_end_0, end_mask = var_5725_end_mask_0, squeeze_mask = var_5725_squeeze_mask_0, x = K_sliding_in)[name = string("op_5725_cast_fp16")]; tensor K_sliding_slot_17_axes_0 = const()[name = string("K_sliding_slot_17_axes_0"), val = tensor([0])]; tensor K_sliding_slot_17_cast_fp16 = expand_dims(axes = K_sliding_slot_17_axes_0, x = var_5725_cast_fp16)[name = string("K_sliding_slot_17_cast_fp16")]; tensor var_5730_begin_0 = const()[name = string("op_5730_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_5730_end_0 = const()[name = string("op_5730_end_0"), val = tensor([9, 2, 512, 512])]; tensor var_5730_end_mask_0 = const()[name = string("op_5730_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5730_squeeze_mask_0 = const()[name = string("op_5730_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_5730_cast_fp16 = slice_by_index(begin = var_5730_begin_0, end = var_5730_end_0, end_mask = var_5730_end_mask_0, squeeze_mask = var_5730_squeeze_mask_0, x = V_sliding_in)[name = string("op_5730_cast_fp16")]; tensor V_sliding_slot_17_axes_0 = const()[name = string("V_sliding_slot_17_axes_0"), val = tensor([0])]; tensor V_sliding_slot_17_cast_fp16 = expand_dims(axes = V_sliding_slot_17_axes_0, x = var_5730_cast_fp16)[name = string("V_sliding_slot_17_cast_fp16")]; int32 var_5737 = const()[name = string("op_5737"), val = int32(-1)]; fp16 const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5739_cast_fp16 = mul(x = x_179_cast_fp16, y = const_106_promoted_to_fp16)[name = string("op_5739_cast_fp16")]; bool input_267_interleave_0 = const()[name = string("input_267_interleave_0"), val = bool(false)]; tensor input_267_cast_fp16 = concat(axis = var_5737, interleave = input_267_interleave_0, values = (x_179_cast_fp16, var_5739_cast_fp16))[name = string("input_267_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; fp16 var_5734_to_fp16 = const()[name = string("op_5734_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_5734_to_fp16, x = input_267_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor var_5744_split_sizes_0 = const()[name = string("op_5744_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5744_axis_0 = const()[name = string("op_5744_axis_0"), val = int32(-1)]; tensor var_5744_cast_fp16_0, tensor var_5744_cast_fp16_1 = split(axis = var_5744_axis_0, split_sizes = var_5744_split_sizes_0, x = normed_253_cast_fp16)[name = string("op_5744_cast_fp16")]; tensor layers_9_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_9_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560649856)))]; tensor h_55_cast_fp16 = mul(x = var_5744_cast_fp16_0, y = layers_9_input_layernorm_weight_promoted_to_fp16)[name = string("h_55_cast_fp16")]; tensor var_5750 = const()[name = string("op_5750"), val = tensor([0, 2, 1])]; tensor var_5753_axes_0 = const()[name = string("op_5753_axes_0"), val = tensor([2])]; tensor var_5751_cast_fp16 = transpose(perm = var_5750, x = h_55_cast_fp16)[name = string("transpose_53")]; tensor var_5753_cast_fp16 = expand_dims(axes = var_5753_axes_0, x = var_5751_cast_fp16)[name = string("op_5753_cast_fp16")]; string var_5769_pad_type_0 = const()[name = string("op_5769_pad_type_0"), val = string("valid")]; tensor var_5769_strides_0 = const()[name = string("op_5769_strides_0"), val = tensor([1, 1])]; tensor var_5769_pad_0 = const()[name = string("op_5769_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5769_dilations_0 = const()[name = string("op_5769_dilations_0"), val = tensor([1, 1])]; int32 var_5769_groups_0 = const()[name = string("op_5769_groups_0"), val = int32(1)]; tensor var_5769 = conv(dilations = var_5769_dilations_0, groups = var_5769_groups_0, pad = var_5769_pad_0, pad_type = var_5769_pad_type_0, strides = var_5769_strides_0, weight = layers_9_self_attn_q_proj_weight_palettized, x = var_5753_cast_fp16)[name = string("op_5769")]; tensor var_5774 = const()[name = string("op_5774"), val = tensor([1, 8, 256, 1])]; tensor var_5775 = reshape(shape = var_5774, x = var_5769)[name = string("op_5775")]; tensor var_5780 = const()[name = string("op_5780"), val = tensor([0, 1, 3, 2])]; tensor var_5790 = const()[name = string("op_5790"), val = tensor([1, 8, 256])]; tensor var_5781 = transpose(perm = var_5780, x = var_5775)[name = string("transpose_52")]; tensor x_181 = reshape(shape = var_5790, x = var_5781)[name = string("x_181")]; int32 var_5796 = const()[name = string("op_5796"), val = int32(-1)]; fp16 const_107_promoted = const()[name = string("const_107_promoted"), val = fp16(-0x1p+0)]; tensor var_5798 = mul(x = x_181, y = const_107_promoted)[name = string("op_5798")]; bool input_271_interleave_0 = const()[name = string("input_271_interleave_0"), val = bool(false)]; tensor input_271 = concat(axis = var_5796, interleave = input_271_interleave_0, values = (x_181, var_5798))[name = string("input_271")]; tensor normed_257_axes_0 = const()[name = string("normed_257_axes_0"), val = tensor([-1])]; fp16 var_5793_to_fp16 = const()[name = string("op_5793_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_257_cast_fp16 = layer_norm(axes = normed_257_axes_0, epsilon = var_5793_to_fp16, x = input_271)[name = string("normed_257_cast_fp16")]; tensor var_5803_split_sizes_0 = const()[name = string("op_5803_split_sizes_0"), val = tensor([256, 256])]; int32 var_5803_axis_0 = const()[name = string("op_5803_axis_0"), val = int32(-1)]; tensor var_5803_0, tensor var_5803_1 = split(axis = var_5803_axis_0, split_sizes = var_5803_split_sizes_0, x = normed_257_cast_fp16)[name = string("op_5803")]; tensor var_5805 = mul(x = var_5803_0, y = layers_9_self_attn_q_norm_weight)[name = string("op_5805")]; tensor var_5810 = const()[name = string("op_5810"), val = tensor([1, 8, 1, 256])]; tensor q_75 = reshape(shape = var_5810, x = var_5805)[name = string("q_75")]; tensor var_5812_cast_fp16 = mul(x = q_75, y = cos_s)[name = string("op_5812_cast_fp16")]; tensor var_5813_split_sizes_0 = const()[name = string("op_5813_split_sizes_0"), val = tensor([128, 128])]; int32 var_5813_axis_0 = const()[name = string("op_5813_axis_0"), val = int32(-1)]; tensor var_5813_0, tensor var_5813_1 = split(axis = var_5813_axis_0, split_sizes = var_5813_split_sizes_0, x = q_75)[name = string("op_5813")]; fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; tensor var_5815 = mul(x = var_5813_1, y = const_108_promoted)[name = string("op_5815")]; int32 var_5817 = const()[name = string("op_5817"), val = int32(-1)]; bool var_5818_interleave_0 = const()[name = string("op_5818_interleave_0"), val = bool(false)]; tensor var_5818 = concat(axis = var_5817, interleave = var_5818_interleave_0, values = (var_5815, var_5813_0))[name = string("op_5818")]; tensor var_5819_cast_fp16 = mul(x = var_5818, y = sin_s)[name = string("op_5819_cast_fp16")]; tensor q_79_cast_fp16 = add(x = var_5812_cast_fp16, y = var_5819_cast_fp16)[name = string("q_79_cast_fp16")]; string var_5832_pad_type_0 = const()[name = string("op_5832_pad_type_0"), val = string("valid")]; tensor var_5832_strides_0 = const()[name = string("op_5832_strides_0"), val = tensor([1, 1])]; tensor var_5832_pad_0 = const()[name = string("op_5832_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5832_dilations_0 = const()[name = string("op_5832_dilations_0"), val = tensor([1, 1])]; int32 var_5832_groups_0 = const()[name = string("op_5832_groups_0"), val = int32(1)]; tensor var_5832 = conv(dilations = var_5832_dilations_0, groups = var_5832_groups_0, pad = var_5832_pad_0, pad_type = var_5832_pad_type_0, strides = var_5832_strides_0, weight = layers_9_self_attn_k_proj_weight_palettized, x = var_5753_cast_fp16)[name = string("op_5832")]; tensor var_5837 = const()[name = string("op_5837"), val = tensor([1, 2, 256, 1])]; tensor var_5838 = reshape(shape = var_5837, x = var_5832)[name = string("op_5838")]; tensor var_5843 = const()[name = string("op_5843"), val = tensor([0, 1, 3, 2])]; string var_5860_pad_type_0 = const()[name = string("op_5860_pad_type_0"), val = string("valid")]; tensor var_5860_strides_0 = const()[name = string("op_5860_strides_0"), val = tensor([1, 1])]; tensor var_5860_pad_0 = const()[name = string("op_5860_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5860_dilations_0 = const()[name = string("op_5860_dilations_0"), val = tensor([1, 1])]; int32 var_5860_groups_0 = const()[name = string("op_5860_groups_0"), val = int32(1)]; tensor var_5860 = conv(dilations = var_5860_dilations_0, groups = var_5860_groups_0, pad = var_5860_pad_0, pad_type = var_5860_pad_type_0, strides = var_5860_strides_0, weight = layers_9_self_attn_v_proj_weight_palettized, x = var_5753_cast_fp16)[name = string("op_5860")]; tensor var_5865 = const()[name = string("op_5865"), val = tensor([1, 2, 256, 1])]; tensor var_5866 = reshape(shape = var_5865, x = var_5860)[name = string("op_5866")]; tensor var_5871 = const()[name = string("op_5871"), val = tensor([0, 1, 3, 2])]; tensor var_5881 = const()[name = string("op_5881"), val = tensor([1, 2, 256])]; tensor var_5844 = transpose(perm = var_5843, x = var_5838)[name = string("transpose_51")]; tensor x_183 = reshape(shape = var_5881, x = var_5844)[name = string("x_183")]; int32 var_5887 = const()[name = string("op_5887"), val = int32(-1)]; fp16 const_109_promoted = const()[name = string("const_109_promoted"), val = fp16(-0x1p+0)]; tensor var_5889 = mul(x = x_183, y = const_109_promoted)[name = string("op_5889")]; bool input_273_interleave_0 = const()[name = string("input_273_interleave_0"), val = bool(false)]; tensor input_273 = concat(axis = var_5887, interleave = input_273_interleave_0, values = (x_183, var_5889))[name = string("input_273")]; tensor normed_261_axes_0 = const()[name = string("normed_261_axes_0"), val = tensor([-1])]; fp16 var_5884_to_fp16 = const()[name = string("op_5884_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_261_cast_fp16 = layer_norm(axes = normed_261_axes_0, epsilon = var_5884_to_fp16, x = input_273)[name = string("normed_261_cast_fp16")]; tensor var_5894_split_sizes_0 = const()[name = string("op_5894_split_sizes_0"), val = tensor([256, 256])]; int32 var_5894_axis_0 = const()[name = string("op_5894_axis_0"), val = int32(-1)]; tensor var_5894_0, tensor var_5894_1 = split(axis = var_5894_axis_0, split_sizes = var_5894_split_sizes_0, x = normed_261_cast_fp16)[name = string("op_5894")]; tensor var_5896 = mul(x = var_5894_0, y = layers_9_self_attn_k_norm_weight)[name = string("op_5896")]; tensor var_5901 = const()[name = string("op_5901"), val = tensor([1, 2, 1, 256])]; tensor q_77 = reshape(shape = var_5901, x = var_5896)[name = string("q_77")]; fp16 var_5903_promoted = const()[name = string("op_5903_promoted"), val = fp16(0x1p+1)]; tensor var_5872 = transpose(perm = var_5871, x = var_5866)[name = string("transpose_50")]; tensor var_5904 = pow(x = var_5872, y = var_5903_promoted)[name = string("op_5904")]; tensor var_5909_axes_0 = const()[name = string("op_5909_axes_0"), val = tensor([-1])]; bool var_5909_keep_dims_0 = const()[name = string("op_5909_keep_dims_0"), val = bool(true)]; tensor var_5909 = reduce_mean(axes = var_5909_axes_0, keep_dims = var_5909_keep_dims_0, x = var_5904)[name = string("op_5909")]; fp16 var_5911_to_fp16 = const()[name = string("op_5911_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_19_cast_fp16 = add(x = var_5909, y = var_5911_to_fp16)[name = string("mean_sq_19_cast_fp16")]; fp32 var_5913_epsilon_0 = const()[name = string("op_5913_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_5913_cast_fp16 = rsqrt(epsilon = var_5913_epsilon_0, x = mean_sq_19_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor input_277_cast_fp16 = mul(x = var_5872, y = var_5913_cast_fp16)[name = string("input_277_cast_fp16")]; tensor var_5915_cast_fp16 = mul(x = q_77, y = cos_s)[name = string("op_5915_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([128, 128])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(-1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = q_77)[name = string("op_5916")]; fp16 const_110_promoted = const()[name = string("const_110_promoted"), val = fp16(-0x1p+0)]; tensor var_5918 = mul(x = var_5916_1, y = const_110_promoted)[name = string("op_5918")]; int32 var_5920 = const()[name = string("op_5920"), val = int32(-1)]; bool var_5921_interleave_0 = const()[name = string("op_5921_interleave_0"), val = bool(false)]; tensor var_5921 = concat(axis = var_5920, interleave = var_5921_interleave_0, values = (var_5918, var_5916_0))[name = string("op_5921")]; tensor var_5922_cast_fp16 = mul(x = var_5921, y = sin_s)[name = string("op_5922_cast_fp16")]; tensor input_275_cast_fp16 = add(x = var_5915_cast_fp16, y = var_5922_cast_fp16)[name = string("input_275_cast_fp16")]; tensor k_padded_17_pad_0 = const()[name = string("k_padded_17_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_17_mode_0 = const()[name = string("k_padded_17_mode_0"), val = string("constant")]; fp16 const_111_to_fp16 = const()[name = string("const_111_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_17_cast_fp16 = pad(constant_val = const_111_to_fp16, mode = k_padded_17_mode_0, pad = k_padded_17_pad_0, x = input_275_cast_fp16)[name = string("k_padded_17_cast_fp16")]; tensor v_padded_17_pad_0 = const()[name = string("v_padded_17_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_17_mode_0 = const()[name = string("v_padded_17_mode_0"), val = string("constant")]; fp16 const_112_to_fp16 = const()[name = string("const_112_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_17_cast_fp16 = pad(constant_val = const_112_to_fp16, mode = v_padded_17_mode_0, pad = v_padded_17_pad_0, x = input_277_cast_fp16)[name = string("v_padded_17_cast_fp16")]; tensor var_5951_begin_0 = const()[name = string("op_5951_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_5951_end_0 = const()[name = string("op_5951_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5951_end_mask_0 = const()[name = string("op_5951_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5951_cast_fp16 = slice_by_index(begin = var_5951_begin_0, end = var_5951_end_0, end_mask = var_5951_end_mask_0, x = K_sliding_slot_17_cast_fp16)[name = string("op_5951_cast_fp16")]; int32 var_5958 = const()[name = string("op_5958"), val = int32(2)]; bool K_sliding_out_17_interleave_0 = const()[name = string("K_sliding_out_17_interleave_0"), val = bool(false)]; tensor K_sliding_out_17_cast_fp16 = concat(axis = var_5958, interleave = K_sliding_out_17_interleave_0, values = (var_5951_cast_fp16, k_padded_17_cast_fp16))[name = string("K_sliding_out_17_cast_fp16")]; tensor var_5974_begin_0 = const()[name = string("op_5974_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_5974_end_0 = const()[name = string("op_5974_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5974_end_mask_0 = const()[name = string("op_5974_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5974_cast_fp16 = slice_by_index(begin = var_5974_begin_0, end = var_5974_end_0, end_mask = var_5974_end_mask_0, x = V_sliding_slot_17_cast_fp16)[name = string("op_5974_cast_fp16")]; int32 var_5981 = const()[name = string("op_5981"), val = int32(2)]; bool V_sliding_out_17_interleave_0 = const()[name = string("V_sliding_out_17_interleave_0"), val = bool(false)]; tensor V_sliding_out_17_cast_fp16 = concat(axis = var_5981, interleave = V_sliding_out_17_interleave_0, values = (var_5974_cast_fp16, v_padded_17_cast_fp16))[name = string("V_sliding_out_17_cast_fp16")]; tensor K_for_attn_19_begin_0 = const()[name = string("K_for_attn_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_19_end_0 = const()[name = string("K_for_attn_19_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_19_end_mask_0 = const()[name = string("K_for_attn_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_19_cast_fp16 = slice_by_index(begin = K_for_attn_19_begin_0, end = K_for_attn_19_end_0, end_mask = K_for_attn_19_end_mask_0, x = K_sliding_out_17_cast_fp16)[name = string("K_for_attn_19_cast_fp16")]; tensor V_for_attn_19_begin_0 = const()[name = string("V_for_attn_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_19_end_0 = const()[name = string("V_for_attn_19_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_19_end_mask_0 = const()[name = string("V_for_attn_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_19_cast_fp16 = slice_by_index(begin = V_for_attn_19_begin_0, end = V_for_attn_19_end_0, end_mask = V_for_attn_19_end_mask_0, x = V_sliding_out_17_cast_fp16)[name = string("V_for_attn_19_cast_fp16")]; tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_18_reps_0 = const()[name = string("tile_18_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = K_for_attn_19_cast_fp16)[name = string("transpose_49")]; tensor tile_18_cast_fp16 = tile(reps = tile_18_reps_0, x = transpose_36_cast_fp16)[name = string("tile_18_cast_fp16")]; tensor concat_36 = const()[name = string("concat_36"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_36_cast_fp16 = reshape(shape = concat_36, x = tile_18_cast_fp16)[name = string("reshape_36_cast_fp16")]; tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_37 = const()[name = string("concat_37"), val = tensor([-1, 1, 512, 256])]; tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_36_cast_fp16)[name = string("transpose_48")]; tensor reshape_37_cast_fp16 = reshape(shape = concat_37, x = transpose_37_cast_fp16)[name = string("reshape_37_cast_fp16")]; tensor transpose_57_perm_0 = const()[name = string("transpose_57_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_19_reps_0 = const()[name = string("tile_19_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = V_for_attn_19_cast_fp16)[name = string("transpose_47")]; tensor tile_19_cast_fp16 = tile(reps = tile_19_reps_0, x = transpose_38_cast_fp16)[name = string("tile_19_cast_fp16")]; tensor concat_38 = const()[name = string("concat_38"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_38_cast_fp16 = reshape(shape = concat_38, x = tile_19_cast_fp16)[name = string("reshape_38_cast_fp16")]; tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_39 = const()[name = string("concat_39"), val = tensor([-1, 1, 512, 256])]; tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_46")]; tensor reshape_39_cast_fp16 = reshape(shape = concat_39, x = transpose_39_cast_fp16)[name = string("reshape_39_cast_fp16")]; tensor V_expanded_19_perm_0 = const()[name = string("V_expanded_19_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_37_transpose_x_0 = const()[name = string("attn_weights_37_transpose_x_0"), val = bool(false)]; bool attn_weights_37_transpose_y_0 = const()[name = string("attn_weights_37_transpose_y_0"), val = bool(false)]; tensor transpose_57_cast_fp16 = transpose(perm = transpose_57_perm_0, x = reshape_37_cast_fp16)[name = string("transpose_45")]; tensor attn_weights_37_cast_fp16 = matmul(transpose_x = attn_weights_37_transpose_x_0, transpose_y = attn_weights_37_transpose_y_0, x = q_79_cast_fp16, y = transpose_57_cast_fp16)[name = string("attn_weights_37_cast_fp16")]; tensor x_187_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask_sliding)[name = string("x_187_cast_fp16")]; tensor reduce_max_9_axes_0 = const()[name = string("reduce_max_9_axes_0"), val = tensor([-1])]; bool reduce_max_9_keep_dims_0 = const()[name = string("reduce_max_9_keep_dims_0"), val = bool(true)]; tensor reduce_max_9 = reduce_max(axes = reduce_max_9_axes_0, keep_dims = reduce_max_9_keep_dims_0, x = x_187_cast_fp16)[name = string("reduce_max_9")]; tensor var_6022 = sub(x = x_187_cast_fp16, y = reduce_max_9)[name = string("op_6022")]; tensor var_6028 = exp(x = var_6022)[name = string("op_6028")]; tensor var_6038_axes_0 = const()[name = string("op_6038_axes_0"), val = tensor([-1])]; bool var_6038_keep_dims_0 = const()[name = string("op_6038_keep_dims_0"), val = bool(true)]; tensor var_6038 = reduce_sum(axes = var_6038_axes_0, keep_dims = var_6038_keep_dims_0, x = var_6028)[name = string("op_6038")]; tensor var_6044_cast_fp16 = real_div(x = var_6028, y = var_6038)[name = string("op_6044_cast_fp16")]; bool attn_output_55_transpose_x_0 = const()[name = string("attn_output_55_transpose_x_0"), val = bool(false)]; bool attn_output_55_transpose_y_0 = const()[name = string("attn_output_55_transpose_y_0"), val = bool(false)]; tensor V_expanded_19_cast_fp16 = transpose(perm = V_expanded_19_perm_0, x = reshape_39_cast_fp16)[name = string("transpose_44")]; tensor attn_output_55_cast_fp16 = matmul(transpose_x = attn_output_55_transpose_x_0, transpose_y = attn_output_55_transpose_y_0, x = var_6044_cast_fp16, y = V_expanded_19_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor var_6055 = const()[name = string("op_6055"), val = tensor([0, 2, 1, 3])]; tensor var_6062 = const()[name = string("op_6062"), val = tensor([1, 1, -1])]; tensor var_6056_cast_fp16 = transpose(perm = var_6055, x = attn_output_55_cast_fp16)[name = string("transpose_43")]; tensor attn_output_57_cast_fp16 = reshape(shape = var_6062, x = var_6056_cast_fp16)[name = string("attn_output_57_cast_fp16")]; tensor var_6067 = const()[name = string("op_6067"), val = tensor([0, 2, 1])]; string var_6083_pad_type_0 = const()[name = string("op_6083_pad_type_0"), val = string("valid")]; int32 var_6083_groups_0 = const()[name = string("op_6083_groups_0"), val = int32(1)]; tensor var_6083_strides_0 = const()[name = string("op_6083_strides_0"), val = tensor([1])]; tensor var_6083_pad_0 = const()[name = string("op_6083_pad_0"), val = tensor([0, 0])]; tensor var_6083_dilations_0 = const()[name = string("op_6083_dilations_0"), val = tensor([1])]; tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560655040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563276544))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6068_cast_fp16 = transpose(perm = var_6067, x = attn_output_57_cast_fp16)[name = string("transpose_42")]; tensor var_6083_cast_fp16 = conv(dilations = var_6083_dilations_0, groups = var_6083_groups_0, pad = var_6083_pad_0, pad_type = var_6083_pad_type_0, strides = var_6083_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_6068_cast_fp16)[name = string("op_6083_cast_fp16")]; tensor var_6087 = const()[name = string("op_6087"), val = tensor([0, 2, 1])]; int32 var_6093 = const()[name = string("op_6093"), val = int32(-1)]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_191_cast_fp16 = transpose(perm = var_6087, x = var_6083_cast_fp16)[name = string("transpose_41")]; tensor var_6095_cast_fp16 = mul(x = x_191_cast_fp16, y = const_113_promoted_to_fp16)[name = string("op_6095_cast_fp16")]; bool input_281_interleave_0 = const()[name = string("input_281_interleave_0"), val = bool(false)]; tensor input_281_cast_fp16 = concat(axis = var_6093, interleave = input_281_interleave_0, values = (x_191_cast_fp16, var_6095_cast_fp16))[name = string("input_281_cast_fp16")]; tensor normed_265_axes_0 = const()[name = string("normed_265_axes_0"), val = tensor([-1])]; fp16 var_6090_to_fp16 = const()[name = string("op_6090_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_265_cast_fp16 = layer_norm(axes = normed_265_axes_0, epsilon = var_6090_to_fp16, x = input_281_cast_fp16)[name = string("normed_265_cast_fp16")]; tensor var_6100_split_sizes_0 = const()[name = string("op_6100_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6100_axis_0 = const()[name = string("op_6100_axis_0"), val = int32(-1)]; tensor var_6100_cast_fp16_0, tensor var_6100_cast_fp16_1 = split(axis = var_6100_axis_0, split_sizes = var_6100_split_sizes_0, x = normed_265_cast_fp16)[name = string("op_6100_cast_fp16")]; tensor layers_9_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_9_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563279168)))]; tensor attn_output_59_cast_fp16 = mul(x = var_6100_cast_fp16_0, y = layers_9_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_59_cast_fp16")]; tensor x_193_cast_fp16 = add(x = x_179_cast_fp16, y = attn_output_59_cast_fp16)[name = string("x_193_cast_fp16")]; int32 var_6109 = const()[name = string("op_6109"), val = int32(-1)]; fp16 const_114_promoted_to_fp16 = const()[name = string("const_114_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6111_cast_fp16 = mul(x = x_193_cast_fp16, y = const_114_promoted_to_fp16)[name = string("op_6111_cast_fp16")]; bool input_283_interleave_0 = const()[name = string("input_283_interleave_0"), val = bool(false)]; tensor input_283_cast_fp16 = concat(axis = var_6109, interleave = input_283_interleave_0, values = (x_193_cast_fp16, var_6111_cast_fp16))[name = string("input_283_cast_fp16")]; tensor normed_269_axes_0 = const()[name = string("normed_269_axes_0"), val = tensor([-1])]; fp16 var_6106_to_fp16 = const()[name = string("op_6106_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_269_cast_fp16 = layer_norm(axes = normed_269_axes_0, epsilon = var_6106_to_fp16, x = input_283_cast_fp16)[name = string("normed_269_cast_fp16")]; tensor var_6116_split_sizes_0 = const()[name = string("op_6116_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6116_axis_0 = const()[name = string("op_6116_axis_0"), val = int32(-1)]; tensor var_6116_cast_fp16_0, tensor var_6116_cast_fp16_1 = split(axis = var_6116_axis_0, split_sizes = var_6116_split_sizes_0, x = normed_269_cast_fp16)[name = string("op_6116_cast_fp16")]; tensor layers_9_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_9_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563284352)))]; tensor h_57_cast_fp16 = mul(x = var_6116_cast_fp16_0, y = layers_9_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_57_cast_fp16")]; tensor var_6127 = const()[name = string("op_6127"), val = tensor([0, 2, 1])]; tensor input_285_axes_0 = const()[name = string("input_285_axes_0"), val = tensor([2])]; tensor var_6128 = transpose(perm = var_6127, x = h_57_cast_fp16)[name = string("transpose_40")]; tensor input_285 = expand_dims(axes = input_285_axes_0, x = var_6128)[name = string("input_285")]; string gate_37_pad_type_0 = const()[name = string("gate_37_pad_type_0"), val = string("valid")]; tensor gate_37_strides_0 = const()[name = string("gate_37_strides_0"), val = tensor([1, 1])]; tensor gate_37_pad_0 = const()[name = string("gate_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_37_dilations_0 = const()[name = string("gate_37_dilations_0"), val = tensor([1, 1])]; int32 gate_37_groups_0 = const()[name = string("gate_37_groups_0"), val = int32(1)]; tensor gate_37 = conv(dilations = gate_37_dilations_0, groups = gate_37_groups_0, pad = gate_37_pad_0, pad_type = gate_37_pad_type_0, strides = gate_37_strides_0, weight = layers_9_mlp_gate_proj_weight_palettized, x = input_285)[name = string("gate_37")]; string up_19_pad_type_0 = const()[name = string("up_19_pad_type_0"), val = string("valid")]; tensor up_19_strides_0 = const()[name = string("up_19_strides_0"), val = tensor([1, 1])]; tensor up_19_pad_0 = const()[name = string("up_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_19_dilations_0 = const()[name = string("up_19_dilations_0"), val = tensor([1, 1])]; int32 up_19_groups_0 = const()[name = string("up_19_groups_0"), val = int32(1)]; tensor up_19 = conv(dilations = up_19_dilations_0, groups = up_19_groups_0, pad = up_19_pad_0, pad_type = up_19_pad_type_0, strides = up_19_strides_0, weight = layers_9_mlp_up_proj_weight_palettized, x = input_285)[name = string("up_19")]; string gate_39_mode_0 = const()[name = string("gate_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_39 = gelu(mode = gate_39_mode_0, x = gate_37)[name = string("gate_39")]; tensor input_287 = mul(x = gate_39, y = up_19)[name = string("input_287")]; string mlp_out_19_pad_type_0 = const()[name = string("mlp_out_19_pad_type_0"), val = string("valid")]; tensor mlp_out_19_strides_0 = const()[name = string("mlp_out_19_strides_0"), val = tensor([1, 1])]; tensor mlp_out_19_pad_0 = const()[name = string("mlp_out_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_19_dilations_0 = const()[name = string("mlp_out_19_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_19_groups_0 = const()[name = string("mlp_out_19_groups_0"), val = int32(1)]; tensor mlp_out_19 = conv(dilations = mlp_out_19_dilations_0, groups = mlp_out_19_groups_0, pad = mlp_out_19_pad_0, pad_type = mlp_out_19_pad_type_0, strides = mlp_out_19_strides_0, weight = layers_9_mlp_down_proj_weight_palettized, x = input_287)[name = string("mlp_out_19")]; tensor var_6168_axes_0 = const()[name = string("op_6168_axes_0"), val = tensor([2])]; tensor var_6168 = squeeze(axes = var_6168_axes_0, x = mlp_out_19)[name = string("op_6168")]; tensor var_6172 = const()[name = string("op_6172"), val = tensor([0, 2, 1])]; int32 var_6178 = const()[name = string("op_6178"), val = int32(-1)]; fp16 const_115_promoted = const()[name = string("const_115_promoted"), val = fp16(-0x1p+0)]; tensor x_195 = transpose(perm = var_6172, x = var_6168)[name = string("transpose_39")]; tensor var_6180 = mul(x = x_195, y = const_115_promoted)[name = string("op_6180")]; bool input_289_interleave_0 = const()[name = string("input_289_interleave_0"), val = bool(false)]; tensor input_289 = concat(axis = var_6178, interleave = input_289_interleave_0, values = (x_195, var_6180))[name = string("input_289")]; tensor normed_273_axes_0 = const()[name = string("normed_273_axes_0"), val = tensor([-1])]; fp16 var_6175_to_fp16 = const()[name = string("op_6175_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_273_cast_fp16 = layer_norm(axes = normed_273_axes_0, epsilon = var_6175_to_fp16, x = input_289)[name = string("normed_273_cast_fp16")]; tensor var_6185_split_sizes_0 = const()[name = string("op_6185_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6185_axis_0 = const()[name = string("op_6185_axis_0"), val = int32(-1)]; tensor var_6185_0, tensor var_6185_1 = split(axis = var_6185_axis_0, split_sizes = var_6185_split_sizes_0, x = normed_273_cast_fp16)[name = string("op_6185")]; tensor hidden_states_93 = mul(x = var_6185_0, y = layers_9_post_feedforward_layernorm_weight)[name = string("hidden_states_93")]; tensor hidden_states_95_cast_fp16 = add(x = x_193_cast_fp16, y = hidden_states_93)[name = string("hidden_states_95_cast_fp16")]; tensor per_layer_slice_19_begin_0 = const()[name = string("per_layer_slice_19_begin_0"), val = tensor([0, 0, 5376])]; tensor per_layer_slice_19_end_0 = const()[name = string("per_layer_slice_19_end_0"), val = tensor([1, 1, 5632])]; tensor per_layer_slice_19_end_mask_0 = const()[name = string("per_layer_slice_19_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_19_cast_fp16 = slice_by_index(begin = per_layer_slice_19_begin_0, end = per_layer_slice_19_end_0, end_mask = per_layer_slice_19_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_19_cast_fp16")]; tensor var_6213 = const()[name = string("op_6213"), val = tensor([0, 2, 1])]; tensor input_291_axes_0 = const()[name = string("input_291_axes_0"), val = tensor([2])]; tensor var_6214 = transpose(perm = var_6213, x = hidden_states_95_cast_fp16)[name = string("transpose_38")]; tensor input_291 = expand_dims(axes = input_291_axes_0, x = var_6214)[name = string("input_291")]; string gated_55_pad_type_0 = const()[name = string("gated_55_pad_type_0"), val = string("valid")]; tensor gated_55_strides_0 = const()[name = string("gated_55_strides_0"), val = tensor([1, 1])]; tensor gated_55_pad_0 = const()[name = string("gated_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_55_dilations_0 = const()[name = string("gated_55_dilations_0"), val = tensor([1, 1])]; int32 gated_55_groups_0 = const()[name = string("gated_55_groups_0"), val = int32(1)]; tensor gated_55 = conv(dilations = gated_55_dilations_0, groups = gated_55_groups_0, pad = gated_55_pad_0, pad_type = gated_55_pad_type_0, strides = gated_55_strides_0, weight = layers_9_per_layer_input_gate_weight_palettized, x = input_291)[name = string("gated_55")]; string gated_57_mode_0 = const()[name = string("gated_57_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_57 = gelu(mode = gated_57_mode_0, x = gated_55)[name = string("gated_57")]; tensor var_6233 = const()[name = string("op_6233"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_19_axes_0 = const()[name = string("per_layer_slice_conv_19_axes_0"), val = tensor([2])]; tensor var_6234_cast_fp16 = transpose(perm = var_6233, x = per_layer_slice_19_cast_fp16)[name = string("transpose_37")]; tensor per_layer_slice_conv_19_cast_fp16 = expand_dims(axes = per_layer_slice_conv_19_axes_0, x = var_6234_cast_fp16)[name = string("per_layer_slice_conv_19_cast_fp16")]; tensor input_293_cast_fp16 = mul(x = gated_57, y = per_layer_slice_conv_19_cast_fp16)[name = string("input_293_cast_fp16")]; string gated_59_pad_type_0 = const()[name = string("gated_59_pad_type_0"), val = string("valid")]; tensor gated_59_strides_0 = const()[name = string("gated_59_strides_0"), val = tensor([1, 1])]; tensor gated_59_pad_0 = const()[name = string("gated_59_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_59_dilations_0 = const()[name = string("gated_59_dilations_0"), val = tensor([1, 1])]; int32 gated_59_groups_0 = const()[name = string("gated_59_groups_0"), val = int32(1)]; tensor layers_9_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563289536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563617280))))[name = string("layers_9_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_59_cast_fp16 = conv(dilations = gated_59_dilations_0, groups = gated_59_groups_0, pad = gated_59_pad_0, pad_type = gated_59_pad_type_0, strides = gated_59_strides_0, weight = layers_9_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_293_cast_fp16)[name = string("gated_59_cast_fp16")]; tensor var_6250_axes_0 = const()[name = string("op_6250_axes_0"), val = tensor([2])]; tensor var_6250_cast_fp16 = squeeze(axes = var_6250_axes_0, x = gated_59_cast_fp16)[name = string("op_6250_cast_fp16")]; tensor var_6254 = const()[name = string("op_6254"), val = tensor([0, 2, 1])]; int32 var_6260 = const()[name = string("op_6260"), val = int32(-1)]; fp16 const_116_promoted_to_fp16 = const()[name = string("const_116_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_197_cast_fp16 = transpose(perm = var_6254, x = var_6250_cast_fp16)[name = string("transpose_36")]; tensor var_6262_cast_fp16 = mul(x = x_197_cast_fp16, y = const_116_promoted_to_fp16)[name = string("op_6262_cast_fp16")]; bool input_295_interleave_0 = const()[name = string("input_295_interleave_0"), val = bool(false)]; tensor input_295_cast_fp16 = concat(axis = var_6260, interleave = input_295_interleave_0, values = (x_197_cast_fp16, var_6262_cast_fp16))[name = string("input_295_cast_fp16")]; tensor normed_277_axes_0 = const()[name = string("normed_277_axes_0"), val = tensor([-1])]; fp16 var_6257_to_fp16 = const()[name = string("op_6257_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_277_cast_fp16 = layer_norm(axes = normed_277_axes_0, epsilon = var_6257_to_fp16, x = input_295_cast_fp16)[name = string("normed_277_cast_fp16")]; tensor var_6267_split_sizes_0 = const()[name = string("op_6267_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6267_axis_0 = const()[name = string("op_6267_axis_0"), val = int32(-1)]; tensor var_6267_cast_fp16_0, tensor var_6267_cast_fp16_1 = split(axis = var_6267_axis_0, split_sizes = var_6267_split_sizes_0, x = normed_277_cast_fp16)[name = string("op_6267_cast_fp16")]; tensor layers_9_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_9_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563619904)))]; tensor hidden_states_99_cast_fp16 = mul(x = var_6267_cast_fp16_0, y = layers_9_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_99_cast_fp16")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_99_cast_fp16)[name = string("hidden_states_101_cast_fp16")]; tensor const_117_promoted_to_fp16 = const()[name = string("const_117_promoted_to_fp16"), val = tensor([0x1.d8p-2])]; tensor x_199_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_117_promoted_to_fp16)[name = string("x_199_cast_fp16")]; tensor var_6279_axes_0 = const()[name = string("op_6279_axes_0"), val = tensor([0])]; tensor var_6279_cast_fp16 = squeeze(axes = var_6279_axes_0, x = K_sliding_out_17_cast_fp16)[name = string("op_6279_cast_fp16")]; tensor var_6281_axes_0 = const()[name = string("op_6281_axes_0"), val = tensor([0])]; tensor var_6281_cast_fp16 = squeeze(axes = var_6281_axes_0, x = V_sliding_out_17_cast_fp16)[name = string("op_6281_cast_fp16")]; tensor var_6284_begin_0 = const()[name = string("op_6284_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_6284_end_0 = const()[name = string("op_6284_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_6284_end_mask_0 = const()[name = string("op_6284_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6284_squeeze_mask_0 = const()[name = string("op_6284_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6284_cast_fp16 = slice_by_index(begin = var_6284_begin_0, end = var_6284_end_0, end_mask = var_6284_end_mask_0, squeeze_mask = var_6284_squeeze_mask_0, x = K_sliding_in)[name = string("op_6284_cast_fp16")]; tensor K_sliding_slot_axes_0 = const()[name = string("K_sliding_slot_axes_0"), val = tensor([0])]; tensor K_sliding_slot_cast_fp16 = expand_dims(axes = K_sliding_slot_axes_0, x = var_6284_cast_fp16)[name = string("K_sliding_slot_cast_fp16")]; tensor var_6289_begin_0 = const()[name = string("op_6289_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_6289_end_0 = const()[name = string("op_6289_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_6289_end_mask_0 = const()[name = string("op_6289_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6289_squeeze_mask_0 = const()[name = string("op_6289_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6289_cast_fp16 = slice_by_index(begin = var_6289_begin_0, end = var_6289_end_0, end_mask = var_6289_end_mask_0, squeeze_mask = var_6289_squeeze_mask_0, x = V_sliding_in)[name = string("op_6289_cast_fp16")]; tensor V_sliding_slot_axes_0 = const()[name = string("V_sliding_slot_axes_0"), val = tensor([0])]; tensor V_sliding_slot_cast_fp16 = expand_dims(axes = V_sliding_slot_axes_0, x = var_6289_cast_fp16)[name = string("V_sliding_slot_cast_fp16")]; int32 var_6296 = const()[name = string("op_6296"), val = int32(-1)]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6298_cast_fp16 = mul(x = x_199_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_6298_cast_fp16")]; bool input_297_interleave_0 = const()[name = string("input_297_interleave_0"), val = bool(false)]; tensor input_297_cast_fp16 = concat(axis = var_6296, interleave = input_297_interleave_0, values = (x_199_cast_fp16, var_6298_cast_fp16))[name = string("input_297_cast_fp16")]; tensor normed_281_axes_0 = const()[name = string("normed_281_axes_0"), val = tensor([-1])]; fp16 var_6293_to_fp16 = const()[name = string("op_6293_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_281_cast_fp16 = layer_norm(axes = normed_281_axes_0, epsilon = var_6293_to_fp16, x = input_297_cast_fp16)[name = string("normed_281_cast_fp16")]; tensor var_6303_split_sizes_0 = const()[name = string("op_6303_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6303_axis_0 = const()[name = string("op_6303_axis_0"), val = int32(-1)]; tensor var_6303_cast_fp16_0, tensor var_6303_cast_fp16_1 = split(axis = var_6303_axis_0, split_sizes = var_6303_split_sizes_0, x = normed_281_cast_fp16)[name = string("op_6303_cast_fp16")]; tensor layers_10_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_10_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563625088)))]; tensor h_61_cast_fp16 = mul(x = var_6303_cast_fp16_0, y = layers_10_input_layernorm_weight_promoted_to_fp16)[name = string("h_61_cast_fp16")]; tensor var_6309 = const()[name = string("op_6309"), val = tensor([0, 2, 1])]; tensor var_6312_axes_0 = const()[name = string("op_6312_axes_0"), val = tensor([2])]; tensor var_6310_cast_fp16 = transpose(perm = var_6309, x = h_61_cast_fp16)[name = string("transpose_35")]; tensor var_6312_cast_fp16 = expand_dims(axes = var_6312_axes_0, x = var_6310_cast_fp16)[name = string("op_6312_cast_fp16")]; string var_6328_pad_type_0 = const()[name = string("op_6328_pad_type_0"), val = string("valid")]; tensor var_6328_strides_0 = const()[name = string("op_6328_strides_0"), val = tensor([1, 1])]; tensor var_6328_pad_0 = const()[name = string("op_6328_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6328_dilations_0 = const()[name = string("op_6328_dilations_0"), val = tensor([1, 1])]; int32 var_6328_groups_0 = const()[name = string("op_6328_groups_0"), val = int32(1)]; tensor var_6328 = conv(dilations = var_6328_dilations_0, groups = var_6328_groups_0, pad = var_6328_pad_0, pad_type = var_6328_pad_type_0, strides = var_6328_strides_0, weight = layers_10_self_attn_q_proj_weight_palettized, x = var_6312_cast_fp16)[name = string("op_6328")]; tensor var_6333 = const()[name = string("op_6333"), val = tensor([1, 8, 256, 1])]; tensor var_6334 = reshape(shape = var_6333, x = var_6328)[name = string("op_6334")]; tensor var_6339 = const()[name = string("op_6339"), val = tensor([0, 1, 3, 2])]; tensor var_6349 = const()[name = string("op_6349"), val = tensor([1, 8, 256])]; tensor var_6340 = transpose(perm = var_6339, x = var_6334)[name = string("transpose_34")]; tensor x_201 = reshape(shape = var_6349, x = var_6340)[name = string("x_201")]; int32 var_6355 = const()[name = string("op_6355"), val = int32(-1)]; fp16 const_119_promoted = const()[name = string("const_119_promoted"), val = fp16(-0x1p+0)]; tensor var_6357 = mul(x = x_201, y = const_119_promoted)[name = string("op_6357")]; bool input_301_interleave_0 = const()[name = string("input_301_interleave_0"), val = bool(false)]; tensor input_301 = concat(axis = var_6355, interleave = input_301_interleave_0, values = (x_201, var_6357))[name = string("input_301")]; tensor normed_285_axes_0 = const()[name = string("normed_285_axes_0"), val = tensor([-1])]; fp16 var_6352_to_fp16 = const()[name = string("op_6352_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_285_cast_fp16 = layer_norm(axes = normed_285_axes_0, epsilon = var_6352_to_fp16, x = input_301)[name = string("normed_285_cast_fp16")]; tensor var_6362_split_sizes_0 = const()[name = string("op_6362_split_sizes_0"), val = tensor([256, 256])]; int32 var_6362_axis_0 = const()[name = string("op_6362_axis_0"), val = int32(-1)]; tensor var_6362_0, tensor var_6362_1 = split(axis = var_6362_axis_0, split_sizes = var_6362_split_sizes_0, x = normed_285_cast_fp16)[name = string("op_6362")]; tensor var_6364 = mul(x = var_6362_0, y = layers_10_self_attn_q_norm_weight)[name = string("op_6364")]; tensor var_6369 = const()[name = string("op_6369"), val = tensor([1, 8, 1, 256])]; tensor q_83 = reshape(shape = var_6369, x = var_6364)[name = string("q_83")]; tensor var_6371_cast_fp16 = mul(x = q_83, y = cos_s)[name = string("op_6371_cast_fp16")]; tensor var_6372_split_sizes_0 = const()[name = string("op_6372_split_sizes_0"), val = tensor([128, 128])]; int32 var_6372_axis_0 = const()[name = string("op_6372_axis_0"), val = int32(-1)]; tensor var_6372_0, tensor var_6372_1 = split(axis = var_6372_axis_0, split_sizes = var_6372_split_sizes_0, x = q_83)[name = string("op_6372")]; fp16 const_120_promoted = const()[name = string("const_120_promoted"), val = fp16(-0x1p+0)]; tensor var_6374 = mul(x = var_6372_1, y = const_120_promoted)[name = string("op_6374")]; int32 var_6376 = const()[name = string("op_6376"), val = int32(-1)]; bool var_6377_interleave_0 = const()[name = string("op_6377_interleave_0"), val = bool(false)]; tensor var_6377 = concat(axis = var_6376, interleave = var_6377_interleave_0, values = (var_6374, var_6372_0))[name = string("op_6377")]; tensor var_6378_cast_fp16 = mul(x = var_6377, y = sin_s)[name = string("op_6378_cast_fp16")]; tensor q_87_cast_fp16 = add(x = var_6371_cast_fp16, y = var_6378_cast_fp16)[name = string("q_87_cast_fp16")]; string var_6391_pad_type_0 = const()[name = string("op_6391_pad_type_0"), val = string("valid")]; tensor var_6391_strides_0 = const()[name = string("op_6391_strides_0"), val = tensor([1, 1])]; tensor var_6391_pad_0 = const()[name = string("op_6391_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6391_dilations_0 = const()[name = string("op_6391_dilations_0"), val = tensor([1, 1])]; int32 var_6391_groups_0 = const()[name = string("op_6391_groups_0"), val = int32(1)]; tensor var_6391 = conv(dilations = var_6391_dilations_0, groups = var_6391_groups_0, pad = var_6391_pad_0, pad_type = var_6391_pad_type_0, strides = var_6391_strides_0, weight = layers_10_self_attn_k_proj_weight_palettized, x = var_6312_cast_fp16)[name = string("op_6391")]; tensor var_6396 = const()[name = string("op_6396"), val = tensor([1, 2, 256, 1])]; tensor var_6397 = reshape(shape = var_6396, x = var_6391)[name = string("op_6397")]; tensor var_6402 = const()[name = string("op_6402"), val = tensor([0, 1, 3, 2])]; string var_6419_pad_type_0 = const()[name = string("op_6419_pad_type_0"), val = string("valid")]; tensor var_6419_strides_0 = const()[name = string("op_6419_strides_0"), val = tensor([1, 1])]; tensor var_6419_pad_0 = const()[name = string("op_6419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6419_dilations_0 = const()[name = string("op_6419_dilations_0"), val = tensor([1, 1])]; int32 var_6419_groups_0 = const()[name = string("op_6419_groups_0"), val = int32(1)]; tensor var_6419 = conv(dilations = var_6419_dilations_0, groups = var_6419_groups_0, pad = var_6419_pad_0, pad_type = var_6419_pad_type_0, strides = var_6419_strides_0, weight = layers_10_self_attn_v_proj_weight_palettized, x = var_6312_cast_fp16)[name = string("op_6419")]; tensor var_6424 = const()[name = string("op_6424"), val = tensor([1, 2, 256, 1])]; tensor var_6425 = reshape(shape = var_6424, x = var_6419)[name = string("op_6425")]; tensor var_6430 = const()[name = string("op_6430"), val = tensor([0, 1, 3, 2])]; tensor var_6440 = const()[name = string("op_6440"), val = tensor([1, 2, 256])]; tensor var_6403 = transpose(perm = var_6402, x = var_6397)[name = string("transpose_33")]; tensor x_203 = reshape(shape = var_6440, x = var_6403)[name = string("x_203")]; int32 var_6446 = const()[name = string("op_6446"), val = int32(-1)]; fp16 const_121_promoted = const()[name = string("const_121_promoted"), val = fp16(-0x1p+0)]; tensor var_6448 = mul(x = x_203, y = const_121_promoted)[name = string("op_6448")]; bool input_303_interleave_0 = const()[name = string("input_303_interleave_0"), val = bool(false)]; tensor input_303 = concat(axis = var_6446, interleave = input_303_interleave_0, values = (x_203, var_6448))[name = string("input_303")]; tensor normed_289_axes_0 = const()[name = string("normed_289_axes_0"), val = tensor([-1])]; fp16 var_6443_to_fp16 = const()[name = string("op_6443_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_289_cast_fp16 = layer_norm(axes = normed_289_axes_0, epsilon = var_6443_to_fp16, x = input_303)[name = string("normed_289_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([256, 256])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-1)]; tensor var_6453_0, tensor var_6453_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = normed_289_cast_fp16)[name = string("op_6453")]; tensor var_6455 = mul(x = var_6453_0, y = layers_4_self_attn_k_norm_weight)[name = string("op_6455")]; tensor var_6460 = const()[name = string("op_6460"), val = tensor([1, 2, 1, 256])]; tensor q_85 = reshape(shape = var_6460, x = var_6455)[name = string("q_85")]; fp16 var_6462_promoted = const()[name = string("op_6462_promoted"), val = fp16(0x1p+1)]; tensor var_6431 = transpose(perm = var_6430, x = var_6425)[name = string("transpose_32")]; tensor var_6463 = pow(x = var_6431, y = var_6462_promoted)[name = string("op_6463")]; tensor var_6468_axes_0 = const()[name = string("op_6468_axes_0"), val = tensor([-1])]; bool var_6468_keep_dims_0 = const()[name = string("op_6468_keep_dims_0"), val = bool(true)]; tensor var_6468 = reduce_mean(axes = var_6468_axes_0, keep_dims = var_6468_keep_dims_0, x = var_6463)[name = string("op_6468")]; fp16 var_6470_to_fp16 = const()[name = string("op_6470_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_21_cast_fp16 = add(x = var_6468, y = var_6470_to_fp16)[name = string("mean_sq_21_cast_fp16")]; fp32 var_6472_epsilon_0 = const()[name = string("op_6472_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_6472_cast_fp16 = rsqrt(epsilon = var_6472_epsilon_0, x = mean_sq_21_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor input_307_cast_fp16 = mul(x = var_6431, y = var_6472_cast_fp16)[name = string("input_307_cast_fp16")]; tensor var_6474_cast_fp16 = mul(x = q_85, y = cos_s)[name = string("op_6474_cast_fp16")]; tensor var_6475_split_sizes_0 = const()[name = string("op_6475_split_sizes_0"), val = tensor([128, 128])]; int32 var_6475_axis_0 = const()[name = string("op_6475_axis_0"), val = int32(-1)]; tensor var_6475_0, tensor var_6475_1 = split(axis = var_6475_axis_0, split_sizes = var_6475_split_sizes_0, x = q_85)[name = string("op_6475")]; fp16 const_122_promoted = const()[name = string("const_122_promoted"), val = fp16(-0x1p+0)]; tensor var_6477 = mul(x = var_6475_1, y = const_122_promoted)[name = string("op_6477")]; int32 var_6479 = const()[name = string("op_6479"), val = int32(-1)]; bool var_6480_interleave_0 = const()[name = string("op_6480_interleave_0"), val = bool(false)]; tensor var_6480 = concat(axis = var_6479, interleave = var_6480_interleave_0, values = (var_6477, var_6475_0))[name = string("op_6480")]; tensor var_6481_cast_fp16 = mul(x = var_6480, y = sin_s)[name = string("op_6481_cast_fp16")]; tensor input_305_cast_fp16 = add(x = var_6474_cast_fp16, y = var_6481_cast_fp16)[name = string("input_305_cast_fp16")]; tensor k_padded_pad_0 = const()[name = string("k_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_mode_0 = const()[name = string("k_padded_mode_0"), val = string("constant")]; fp16 const_123_to_fp16 = const()[name = string("const_123_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_cast_fp16 = pad(constant_val = const_123_to_fp16, mode = k_padded_mode_0, pad = k_padded_pad_0, x = input_305_cast_fp16)[name = string("k_padded_cast_fp16")]; tensor v_padded_pad_0 = const()[name = string("v_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_mode_0 = const()[name = string("v_padded_mode_0"), val = string("constant")]; fp16 const_124_to_fp16 = const()[name = string("const_124_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_cast_fp16 = pad(constant_val = const_124_to_fp16, mode = v_padded_mode_0, pad = v_padded_pad_0, x = input_307_cast_fp16)[name = string("v_padded_cast_fp16")]; tensor var_6510_begin_0 = const()[name = string("op_6510_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_6510_end_0 = const()[name = string("op_6510_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_6510_end_mask_0 = const()[name = string("op_6510_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6510_cast_fp16 = slice_by_index(begin = var_6510_begin_0, end = var_6510_end_0, end_mask = var_6510_end_mask_0, x = K_sliding_slot_cast_fp16)[name = string("op_6510_cast_fp16")]; int32 var_6517 = const()[name = string("op_6517"), val = int32(2)]; bool K_sliding_out_interleave_0 = const()[name = string("K_sliding_out_interleave_0"), val = bool(false)]; tensor K_sliding_out_cast_fp16 = concat(axis = var_6517, interleave = K_sliding_out_interleave_0, values = (var_6510_cast_fp16, k_padded_cast_fp16))[name = string("K_sliding_out_cast_fp16")]; tensor var_6533_begin_0 = const()[name = string("op_6533_begin_0"), val = tensor([0, 0, 1, 0])]; tensor var_6533_end_0 = const()[name = string("op_6533_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_6533_end_mask_0 = const()[name = string("op_6533_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6533_cast_fp16 = slice_by_index(begin = var_6533_begin_0, end = var_6533_end_0, end_mask = var_6533_end_mask_0, x = V_sliding_slot_cast_fp16)[name = string("op_6533_cast_fp16")]; int32 var_6540 = const()[name = string("op_6540"), val = int32(2)]; bool V_sliding_out_interleave_0 = const()[name = string("V_sliding_out_interleave_0"), val = bool(false)]; tensor V_sliding_out_cast_fp16 = concat(axis = var_6540, interleave = V_sliding_out_interleave_0, values = (var_6533_cast_fp16, v_padded_cast_fp16))[name = string("V_sliding_out_cast_fp16")]; tensor K_for_attn_21_begin_0 = const()[name = string("K_for_attn_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_21_end_0 = const()[name = string("K_for_attn_21_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_21_end_mask_0 = const()[name = string("K_for_attn_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor kv13_k = slice_by_index(begin = K_for_attn_21_begin_0, end = K_for_attn_21_end_0, end_mask = K_for_attn_21_end_mask_0, x = K_sliding_out_cast_fp16)[name = string("K_for_attn_21_cast_fp16")]; tensor V_for_attn_21_begin_0 = const()[name = string("V_for_attn_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_21_end_0 = const()[name = string("V_for_attn_21_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_21_end_mask_0 = const()[name = string("V_for_attn_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor kv13_v = slice_by_index(begin = V_for_attn_21_begin_0, end = V_for_attn_21_end_0, end_mask = V_for_attn_21_end_mask_0, x = V_sliding_out_cast_fp16)[name = string("V_for_attn_21_cast_fp16")]; tensor transpose_40_perm_0 = const()[name = string("transpose_40_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_20_reps_0 = const()[name = string("tile_20_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_40_cast_fp16 = transpose(perm = transpose_40_perm_0, x = kv13_k)[name = string("transpose_31")]; tensor tile_20_cast_fp16 = tile(reps = tile_20_reps_0, x = transpose_40_cast_fp16)[name = string("tile_20_cast_fp16")]; tensor concat_40 = const()[name = string("concat_40"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_40_cast_fp16 = reshape(shape = concat_40, x = tile_20_cast_fp16)[name = string("reshape_40_cast_fp16")]; tensor transpose_41_perm_0 = const()[name = string("transpose_41_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_41 = const()[name = string("concat_41"), val = tensor([-1, 1, 512, 256])]; tensor transpose_41_cast_fp16 = transpose(perm = transpose_41_perm_0, x = reshape_40_cast_fp16)[name = string("transpose_30")]; tensor reshape_41_cast_fp16 = reshape(shape = concat_41, x = transpose_41_cast_fp16)[name = string("reshape_41_cast_fp16")]; tensor transpose_58_perm_0 = const()[name = string("transpose_58_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_42_perm_0 = const()[name = string("transpose_42_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_21_reps_0 = const()[name = string("tile_21_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_42_cast_fp16 = transpose(perm = transpose_42_perm_0, x = kv13_v)[name = string("transpose_29")]; tensor tile_21_cast_fp16 = tile(reps = tile_21_reps_0, x = transpose_42_cast_fp16)[name = string("tile_21_cast_fp16")]; tensor concat_42 = const()[name = string("concat_42"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_42_cast_fp16 = reshape(shape = concat_42, x = tile_21_cast_fp16)[name = string("reshape_42_cast_fp16")]; tensor transpose_43_perm_0 = const()[name = string("transpose_43_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_43 = const()[name = string("concat_43"), val = tensor([-1, 1, 512, 256])]; tensor transpose_43_cast_fp16 = transpose(perm = transpose_43_perm_0, x = reshape_42_cast_fp16)[name = string("transpose_28")]; tensor reshape_43_cast_fp16 = reshape(shape = concat_43, x = transpose_43_cast_fp16)[name = string("reshape_43_cast_fp16")]; tensor V_expanded_21_perm_0 = const()[name = string("V_expanded_21_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor transpose_58_cast_fp16 = transpose(perm = transpose_58_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_27")]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = q_87_cast_fp16, y = transpose_58_cast_fp16)[name = string("attn_weights_41_cast_fp16")]; tensor x_207_cast_fp16 = add(x = attn_weights_41_cast_fp16, y = causal_mask_sliding)[name = string("x_207_cast_fp16")]; tensor reduce_max_10_axes_0 = const()[name = string("reduce_max_10_axes_0"), val = tensor([-1])]; bool reduce_max_10_keep_dims_0 = const()[name = string("reduce_max_10_keep_dims_0"), val = bool(true)]; tensor reduce_max_10 = reduce_max(axes = reduce_max_10_axes_0, keep_dims = reduce_max_10_keep_dims_0, x = x_207_cast_fp16)[name = string("reduce_max_10")]; tensor var_6591 = sub(x = x_207_cast_fp16, y = reduce_max_10)[name = string("op_6591")]; tensor var_6597 = exp(x = var_6591)[name = string("op_6597")]; tensor var_6607_axes_0 = const()[name = string("op_6607_axes_0"), val = tensor([-1])]; bool var_6607_keep_dims_0 = const()[name = string("op_6607_keep_dims_0"), val = bool(true)]; tensor var_6607 = reduce_sum(axes = var_6607_axes_0, keep_dims = var_6607_keep_dims_0, x = var_6597)[name = string("op_6607")]; tensor var_6613_cast_fp16 = real_div(x = var_6597, y = var_6607)[name = string("op_6613_cast_fp16")]; bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; tensor V_expanded_21_cast_fp16 = transpose(perm = V_expanded_21_perm_0, x = reshape_43_cast_fp16)[name = string("transpose_26")]; tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = var_6613_cast_fp16, y = V_expanded_21_cast_fp16)[name = string("attn_output_61_cast_fp16")]; tensor var_6624 = const()[name = string("op_6624"), val = tensor([0, 2, 1, 3])]; tensor var_6631 = const()[name = string("op_6631"), val = tensor([1, 1, -1])]; tensor var_6625_cast_fp16 = transpose(perm = var_6624, x = attn_output_61_cast_fp16)[name = string("transpose_25")]; tensor attn_output_63_cast_fp16 = reshape(shape = var_6631, x = var_6625_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor var_6636 = const()[name = string("op_6636"), val = tensor([0, 2, 1])]; string var_6652_pad_type_0 = const()[name = string("op_6652_pad_type_0"), val = string("valid")]; int32 var_6652_groups_0 = const()[name = string("op_6652_groups_0"), val = int32(1)]; tensor var_6652_strides_0 = const()[name = string("op_6652_strides_0"), val = tensor([1])]; tensor var_6652_pad_0 = const()[name = string("op_6652_pad_0"), val = tensor([0, 0])]; tensor var_6652_dilations_0 = const()[name = string("op_6652_dilations_0"), val = tensor([1])]; tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563630272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566251776))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6637_cast_fp16 = transpose(perm = var_6636, x = attn_output_63_cast_fp16)[name = string("transpose_24")]; tensor var_6652_cast_fp16 = conv(dilations = var_6652_dilations_0, groups = var_6652_groups_0, pad = var_6652_pad_0, pad_type = var_6652_pad_type_0, strides = var_6652_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6637_cast_fp16)[name = string("op_6652_cast_fp16")]; tensor var_6656 = const()[name = string("op_6656"), val = tensor([0, 2, 1])]; int32 var_6662 = const()[name = string("op_6662"), val = int32(-1)]; fp16 const_125_promoted_to_fp16 = const()[name = string("const_125_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_211_cast_fp16 = transpose(perm = var_6656, x = var_6652_cast_fp16)[name = string("transpose_23")]; tensor var_6664_cast_fp16 = mul(x = x_211_cast_fp16, y = const_125_promoted_to_fp16)[name = string("op_6664_cast_fp16")]; bool input_311_interleave_0 = const()[name = string("input_311_interleave_0"), val = bool(false)]; tensor input_311_cast_fp16 = concat(axis = var_6662, interleave = input_311_interleave_0, values = (x_211_cast_fp16, var_6664_cast_fp16))[name = string("input_311_cast_fp16")]; tensor normed_293_axes_0 = const()[name = string("normed_293_axes_0"), val = tensor([-1])]; fp16 var_6659_to_fp16 = const()[name = string("op_6659_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_293_cast_fp16 = layer_norm(axes = normed_293_axes_0, epsilon = var_6659_to_fp16, x = input_311_cast_fp16)[name = string("normed_293_cast_fp16")]; tensor var_6669_split_sizes_0 = const()[name = string("op_6669_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6669_axis_0 = const()[name = string("op_6669_axis_0"), val = int32(-1)]; tensor var_6669_cast_fp16_0, tensor var_6669_cast_fp16_1 = split(axis = var_6669_axis_0, split_sizes = var_6669_split_sizes_0, x = normed_293_cast_fp16)[name = string("op_6669_cast_fp16")]; tensor layers_10_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_10_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566254400)))]; tensor attn_output_65_cast_fp16 = mul(x = var_6669_cast_fp16_0, y = layers_10_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_65_cast_fp16")]; tensor x_213_cast_fp16 = add(x = x_199_cast_fp16, y = attn_output_65_cast_fp16)[name = string("x_213_cast_fp16")]; int32 var_6678 = const()[name = string("op_6678"), val = int32(-1)]; fp16 const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6680_cast_fp16 = mul(x = x_213_cast_fp16, y = const_126_promoted_to_fp16)[name = string("op_6680_cast_fp16")]; bool input_313_interleave_0 = const()[name = string("input_313_interleave_0"), val = bool(false)]; tensor input_313_cast_fp16 = concat(axis = var_6678, interleave = input_313_interleave_0, values = (x_213_cast_fp16, var_6680_cast_fp16))[name = string("input_313_cast_fp16")]; tensor normed_297_axes_0 = const()[name = string("normed_297_axes_0"), val = tensor([-1])]; fp16 var_6675_to_fp16 = const()[name = string("op_6675_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_297_cast_fp16 = layer_norm(axes = normed_297_axes_0, epsilon = var_6675_to_fp16, x = input_313_cast_fp16)[name = string("normed_297_cast_fp16")]; tensor var_6685_split_sizes_0 = const()[name = string("op_6685_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6685_axis_0 = const()[name = string("op_6685_axis_0"), val = int32(-1)]; tensor var_6685_cast_fp16_0, tensor var_6685_cast_fp16_1 = split(axis = var_6685_axis_0, split_sizes = var_6685_split_sizes_0, x = normed_297_cast_fp16)[name = string("op_6685_cast_fp16")]; tensor layers_10_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_10_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566259584)))]; tensor h_63_cast_fp16 = mul(x = var_6685_cast_fp16_0, y = layers_10_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_63_cast_fp16")]; tensor var_6696 = const()[name = string("op_6696"), val = tensor([0, 2, 1])]; tensor input_315_axes_0 = const()[name = string("input_315_axes_0"), val = tensor([2])]; tensor var_6697 = transpose(perm = var_6696, x = h_63_cast_fp16)[name = string("transpose_22")]; tensor input_315 = expand_dims(axes = input_315_axes_0, x = var_6697)[name = string("input_315")]; string gate_41_pad_type_0 = const()[name = string("gate_41_pad_type_0"), val = string("valid")]; tensor gate_41_strides_0 = const()[name = string("gate_41_strides_0"), val = tensor([1, 1])]; tensor gate_41_pad_0 = const()[name = string("gate_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_41_dilations_0 = const()[name = string("gate_41_dilations_0"), val = tensor([1, 1])]; int32 gate_41_groups_0 = const()[name = string("gate_41_groups_0"), val = int32(1)]; tensor gate_41 = conv(dilations = gate_41_dilations_0, groups = gate_41_groups_0, pad = gate_41_pad_0, pad_type = gate_41_pad_type_0, strides = gate_41_strides_0, weight = layers_10_mlp_gate_proj_weight_palettized, x = input_315)[name = string("gate_41")]; string up_21_pad_type_0 = const()[name = string("up_21_pad_type_0"), val = string("valid")]; tensor up_21_strides_0 = const()[name = string("up_21_strides_0"), val = tensor([1, 1])]; tensor up_21_pad_0 = const()[name = string("up_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_21_dilations_0 = const()[name = string("up_21_dilations_0"), val = tensor([1, 1])]; int32 up_21_groups_0 = const()[name = string("up_21_groups_0"), val = int32(1)]; tensor up_21 = conv(dilations = up_21_dilations_0, groups = up_21_groups_0, pad = up_21_pad_0, pad_type = up_21_pad_type_0, strides = up_21_strides_0, weight = layers_10_mlp_up_proj_weight_palettized, x = input_315)[name = string("up_21")]; string gate_43_mode_0 = const()[name = string("gate_43_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_43 = gelu(mode = gate_43_mode_0, x = gate_41)[name = string("gate_43")]; tensor input_317 = mul(x = gate_43, y = up_21)[name = string("input_317")]; string mlp_out_21_pad_type_0 = const()[name = string("mlp_out_21_pad_type_0"), val = string("valid")]; tensor mlp_out_21_strides_0 = const()[name = string("mlp_out_21_strides_0"), val = tensor([1, 1])]; tensor mlp_out_21_pad_0 = const()[name = string("mlp_out_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_21_dilations_0 = const()[name = string("mlp_out_21_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_21_groups_0 = const()[name = string("mlp_out_21_groups_0"), val = int32(1)]; tensor mlp_out_21 = conv(dilations = mlp_out_21_dilations_0, groups = mlp_out_21_groups_0, pad = mlp_out_21_pad_0, pad_type = mlp_out_21_pad_type_0, strides = mlp_out_21_strides_0, weight = layers_10_mlp_down_proj_weight_palettized, x = input_317)[name = string("mlp_out_21")]; tensor var_6737_axes_0 = const()[name = string("op_6737_axes_0"), val = tensor([2])]; tensor var_6737 = squeeze(axes = var_6737_axes_0, x = mlp_out_21)[name = string("op_6737")]; tensor var_6741 = const()[name = string("op_6741"), val = tensor([0, 2, 1])]; int32 var_6747 = const()[name = string("op_6747"), val = int32(-1)]; fp16 const_127_promoted = const()[name = string("const_127_promoted"), val = fp16(-0x1p+0)]; tensor x_215 = transpose(perm = var_6741, x = var_6737)[name = string("transpose_21")]; tensor var_6749 = mul(x = x_215, y = const_127_promoted)[name = string("op_6749")]; bool input_319_interleave_0 = const()[name = string("input_319_interleave_0"), val = bool(false)]; tensor input_319 = concat(axis = var_6747, interleave = input_319_interleave_0, values = (x_215, var_6749))[name = string("input_319")]; tensor normed_301_axes_0 = const()[name = string("normed_301_axes_0"), val = tensor([-1])]; fp16 var_6744_to_fp16 = const()[name = string("op_6744_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_301_cast_fp16 = layer_norm(axes = normed_301_axes_0, epsilon = var_6744_to_fp16, x = input_319)[name = string("normed_301_cast_fp16")]; tensor var_6754_split_sizes_0 = const()[name = string("op_6754_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6754_axis_0 = const()[name = string("op_6754_axis_0"), val = int32(-1)]; tensor var_6754_0, tensor var_6754_1 = split(axis = var_6754_axis_0, split_sizes = var_6754_split_sizes_0, x = normed_301_cast_fp16)[name = string("op_6754")]; tensor hidden_states_103 = mul(x = var_6754_0, y = layers_10_post_feedforward_layernorm_weight)[name = string("hidden_states_103")]; tensor hidden_states_105_cast_fp16 = add(x = x_213_cast_fp16, y = hidden_states_103)[name = string("hidden_states_105_cast_fp16")]; tensor per_layer_slice_21_begin_0 = const()[name = string("per_layer_slice_21_begin_0"), val = tensor([0, 0, 5632])]; tensor per_layer_slice_21_end_0 = const()[name = string("per_layer_slice_21_end_0"), val = tensor([1, 1, 5888])]; tensor per_layer_slice_21_end_mask_0 = const()[name = string("per_layer_slice_21_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_21_cast_fp16 = slice_by_index(begin = per_layer_slice_21_begin_0, end = per_layer_slice_21_end_0, end_mask = per_layer_slice_21_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_21_cast_fp16")]; tensor var_6782 = const()[name = string("op_6782"), val = tensor([0, 2, 1])]; tensor input_321_axes_0 = const()[name = string("input_321_axes_0"), val = tensor([2])]; tensor var_6783 = transpose(perm = var_6782, x = hidden_states_105_cast_fp16)[name = string("transpose_20")]; tensor input_321 = expand_dims(axes = input_321_axes_0, x = var_6783)[name = string("input_321")]; string gated_61_pad_type_0 = const()[name = string("gated_61_pad_type_0"), val = string("valid")]; tensor gated_61_strides_0 = const()[name = string("gated_61_strides_0"), val = tensor([1, 1])]; tensor gated_61_pad_0 = const()[name = string("gated_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_61_dilations_0 = const()[name = string("gated_61_dilations_0"), val = tensor([1, 1])]; int32 gated_61_groups_0 = const()[name = string("gated_61_groups_0"), val = int32(1)]; tensor gated_61 = conv(dilations = gated_61_dilations_0, groups = gated_61_groups_0, pad = gated_61_pad_0, pad_type = gated_61_pad_type_0, strides = gated_61_strides_0, weight = layers_10_per_layer_input_gate_weight_palettized, x = input_321)[name = string("gated_61")]; string gated_63_mode_0 = const()[name = string("gated_63_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_63 = gelu(mode = gated_63_mode_0, x = gated_61)[name = string("gated_63")]; tensor var_6802 = const()[name = string("op_6802"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_21_axes_0 = const()[name = string("per_layer_slice_conv_21_axes_0"), val = tensor([2])]; tensor var_6803_cast_fp16 = transpose(perm = var_6802, x = per_layer_slice_21_cast_fp16)[name = string("transpose_19")]; tensor per_layer_slice_conv_21_cast_fp16 = expand_dims(axes = per_layer_slice_conv_21_axes_0, x = var_6803_cast_fp16)[name = string("per_layer_slice_conv_21_cast_fp16")]; tensor input_323_cast_fp16 = mul(x = gated_63, y = per_layer_slice_conv_21_cast_fp16)[name = string("input_323_cast_fp16")]; string gated_65_pad_type_0 = const()[name = string("gated_65_pad_type_0"), val = string("valid")]; tensor gated_65_strides_0 = const()[name = string("gated_65_strides_0"), val = tensor([1, 1])]; tensor gated_65_pad_0 = const()[name = string("gated_65_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_65_dilations_0 = const()[name = string("gated_65_dilations_0"), val = tensor([1, 1])]; int32 gated_65_groups_0 = const()[name = string("gated_65_groups_0"), val = int32(1)]; tensor layers_10_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566264768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566592512))))[name = string("layers_10_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_65_cast_fp16 = conv(dilations = gated_65_dilations_0, groups = gated_65_groups_0, pad = gated_65_pad_0, pad_type = gated_65_pad_type_0, strides = gated_65_strides_0, weight = layers_10_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_323_cast_fp16)[name = string("gated_65_cast_fp16")]; tensor var_6819_axes_0 = const()[name = string("op_6819_axes_0"), val = tensor([2])]; tensor var_6819_cast_fp16 = squeeze(axes = var_6819_axes_0, x = gated_65_cast_fp16)[name = string("op_6819_cast_fp16")]; tensor var_6823 = const()[name = string("op_6823"), val = tensor([0, 2, 1])]; int32 var_6829 = const()[name = string("op_6829"), val = int32(-1)]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_217_cast_fp16 = transpose(perm = var_6823, x = var_6819_cast_fp16)[name = string("transpose_18")]; tensor var_6831_cast_fp16 = mul(x = x_217_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_6831_cast_fp16")]; bool input_325_interleave_0 = const()[name = string("input_325_interleave_0"), val = bool(false)]; tensor input_325_cast_fp16 = concat(axis = var_6829, interleave = input_325_interleave_0, values = (x_217_cast_fp16, var_6831_cast_fp16))[name = string("input_325_cast_fp16")]; tensor normed_305_axes_0 = const()[name = string("normed_305_axes_0"), val = tensor([-1])]; fp16 var_6826_to_fp16 = const()[name = string("op_6826_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_305_cast_fp16 = layer_norm(axes = normed_305_axes_0, epsilon = var_6826_to_fp16, x = input_325_cast_fp16)[name = string("normed_305_cast_fp16")]; tensor var_6836_split_sizes_0 = const()[name = string("op_6836_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6836_axis_0 = const()[name = string("op_6836_axis_0"), val = int32(-1)]; tensor var_6836_cast_fp16_0, tensor var_6836_cast_fp16_1 = split(axis = var_6836_axis_0, split_sizes = var_6836_split_sizes_0, x = normed_305_cast_fp16)[name = string("op_6836_cast_fp16")]; tensor layers_10_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_10_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566595136)))]; tensor hidden_states_109_cast_fp16 = mul(x = var_6836_cast_fp16_0, y = layers_10_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_109_cast_fp16")]; tensor hidden_states_111_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_109_cast_fp16)[name = string("hidden_states_111_cast_fp16")]; tensor const_129_promoted_to_fp16 = const()[name = string("const_129_promoted_to_fp16"), val = tensor([0x1.42p-3])]; tensor x_219_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_129_promoted_to_fp16)[name = string("x_219_cast_fp16")]; tensor var_6848_axes_0 = const()[name = string("op_6848_axes_0"), val = tensor([0])]; tensor var_6848_cast_fp16 = squeeze(axes = var_6848_axes_0, x = K_sliding_out_cast_fp16)[name = string("op_6848_cast_fp16")]; tensor var_6850_axes_0 = const()[name = string("op_6850_axes_0"), val = tensor([0])]; tensor var_6850_cast_fp16 = squeeze(axes = var_6850_axes_0, x = V_sliding_out_cast_fp16)[name = string("op_6850_cast_fp16")]; tensor var_6853_begin_0 = const()[name = string("op_6853_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_6853_end_0 = const()[name = string("op_6853_end_0"), val = tensor([2, 2, 2048, 512])]; tensor var_6853_end_mask_0 = const()[name = string("op_6853_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6853_squeeze_mask_0 = const()[name = string("op_6853_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6853_cast_fp16 = slice_by_index(begin = var_6853_begin_0, end = var_6853_end_0, end_mask = var_6853_end_mask_0, squeeze_mask = var_6853_squeeze_mask_0, x = K_full_in)[name = string("op_6853_cast_fp16")]; tensor K_full_slot_axes_0 = const()[name = string("K_full_slot_axes_0"), val = tensor([0])]; tensor K_full_slot_cast_fp16 = expand_dims(axes = K_full_slot_axes_0, x = var_6853_cast_fp16)[name = string("K_full_slot_cast_fp16")]; tensor var_6858_begin_0 = const()[name = string("op_6858_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_6858_end_0 = const()[name = string("op_6858_end_0"), val = tensor([2, 2, 2048, 512])]; tensor var_6858_end_mask_0 = const()[name = string("op_6858_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6858_squeeze_mask_0 = const()[name = string("op_6858_squeeze_mask_0"), val = tensor([true, false, false, false])]; tensor var_6858_cast_fp16 = slice_by_index(begin = var_6858_begin_0, end = var_6858_end_0, end_mask = var_6858_end_mask_0, squeeze_mask = var_6858_squeeze_mask_0, x = V_full_in)[name = string("op_6858_cast_fp16")]; tensor V_full_slot_axes_0 = const()[name = string("V_full_slot_axes_0"), val = tensor([0])]; tensor V_full_slot_cast_fp16 = expand_dims(axes = V_full_slot_axes_0, x = var_6858_cast_fp16)[name = string("V_full_slot_cast_fp16")]; int32 var_6865 = const()[name = string("op_6865"), val = int32(-1)]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6867_cast_fp16 = mul(x = x_219_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_6867_cast_fp16")]; bool input_327_interleave_0 = const()[name = string("input_327_interleave_0"), val = bool(false)]; tensor input_327_cast_fp16 = concat(axis = var_6865, interleave = input_327_interleave_0, values = (x_219_cast_fp16, var_6867_cast_fp16))[name = string("input_327_cast_fp16")]; tensor normed_309_axes_0 = const()[name = string("normed_309_axes_0"), val = tensor([-1])]; fp16 var_6862_to_fp16 = const()[name = string("op_6862_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_309_cast_fp16 = layer_norm(axes = normed_309_axes_0, epsilon = var_6862_to_fp16, x = input_327_cast_fp16)[name = string("normed_309_cast_fp16")]; tensor var_6872_split_sizes_0 = const()[name = string("op_6872_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6872_axis_0 = const()[name = string("op_6872_axis_0"), val = int32(-1)]; tensor var_6872_cast_fp16_0, tensor var_6872_cast_fp16_1 = split(axis = var_6872_axis_0, split_sizes = var_6872_split_sizes_0, x = normed_309_cast_fp16)[name = string("op_6872_cast_fp16")]; tensor layers_11_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_11_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566600320)))]; tensor h_67_cast_fp16 = mul(x = var_6872_cast_fp16_0, y = layers_11_input_layernorm_weight_promoted_to_fp16)[name = string("h_67_cast_fp16")]; tensor var_6878 = const()[name = string("op_6878"), val = tensor([0, 2, 1])]; tensor var_6881_axes_0 = const()[name = string("op_6881_axes_0"), val = tensor([2])]; tensor var_6879_cast_fp16 = transpose(perm = var_6878, x = h_67_cast_fp16)[name = string("transpose_17")]; tensor var_6881_cast_fp16 = expand_dims(axes = var_6881_axes_0, x = var_6879_cast_fp16)[name = string("op_6881_cast_fp16")]; string var_6897_pad_type_0 = const()[name = string("op_6897_pad_type_0"), val = string("valid")]; tensor var_6897_strides_0 = const()[name = string("op_6897_strides_0"), val = tensor([1, 1])]; tensor var_6897_pad_0 = const()[name = string("op_6897_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6897_dilations_0 = const()[name = string("op_6897_dilations_0"), val = tensor([1, 1])]; int32 var_6897_groups_0 = const()[name = string("op_6897_groups_0"), val = int32(1)]; tensor var_6897 = conv(dilations = var_6897_dilations_0, groups = var_6897_groups_0, pad = var_6897_pad_0, pad_type = var_6897_pad_type_0, strides = var_6897_strides_0, weight = layers_11_self_attn_q_proj_weight_palettized, x = var_6881_cast_fp16)[name = string("op_6897")]; tensor var_6902 = const()[name = string("op_6902"), val = tensor([1, 8, 512, 1])]; tensor var_6903 = reshape(shape = var_6902, x = var_6897)[name = string("op_6903")]; tensor var_6908 = const()[name = string("op_6908"), val = tensor([0, 1, 3, 2])]; tensor var_6918 = const()[name = string("op_6918"), val = tensor([1, 8, 512])]; tensor var_6909 = transpose(perm = var_6908, x = var_6903)[name = string("transpose_16")]; tensor x_221 = reshape(shape = var_6918, x = var_6909)[name = string("x_221")]; int32 var_6924 = const()[name = string("op_6924"), val = int32(-1)]; fp16 const_131_promoted = const()[name = string("const_131_promoted"), val = fp16(-0x1p+0)]; tensor var_6926 = mul(x = x_221, y = const_131_promoted)[name = string("op_6926")]; bool input_331_interleave_0 = const()[name = string("input_331_interleave_0"), val = bool(false)]; tensor input_331 = concat(axis = var_6924, interleave = input_331_interleave_0, values = (x_221, var_6926))[name = string("input_331")]; tensor normed_313_axes_0 = const()[name = string("normed_313_axes_0"), val = tensor([-1])]; fp16 var_6921_to_fp16 = const()[name = string("op_6921_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_313_cast_fp16 = layer_norm(axes = normed_313_axes_0, epsilon = var_6921_to_fp16, x = input_331)[name = string("normed_313_cast_fp16")]; tensor var_6931_split_sizes_0 = const()[name = string("op_6931_split_sizes_0"), val = tensor([512, 512])]; int32 var_6931_axis_0 = const()[name = string("op_6931_axis_0"), val = int32(-1)]; tensor var_6931_0, tensor var_6931_1 = split(axis = var_6931_axis_0, split_sizes = var_6931_split_sizes_0, x = normed_313_cast_fp16)[name = string("op_6931")]; tensor var_6933 = mul(x = var_6931_0, y = layers_11_self_attn_q_norm_weight)[name = string("op_6933")]; tensor var_6938 = const()[name = string("op_6938"), val = tensor([1, 8, 1, 512])]; tensor q_91 = reshape(shape = var_6938, x = var_6933)[name = string("q_91")]; tensor var_6940_cast_fp16 = mul(x = q_91, y = cos_f)[name = string("op_6940_cast_fp16")]; tensor var_6941_split_sizes_0 = const()[name = string("op_6941_split_sizes_0"), val = tensor([256, 256])]; int32 var_6941_axis_0 = const()[name = string("op_6941_axis_0"), val = int32(-1)]; tensor var_6941_0, tensor var_6941_1 = split(axis = var_6941_axis_0, split_sizes = var_6941_split_sizes_0, x = q_91)[name = string("op_6941")]; fp16 const_132_promoted = const()[name = string("const_132_promoted"), val = fp16(-0x1p+0)]; tensor var_6943 = mul(x = var_6941_1, y = const_132_promoted)[name = string("op_6943")]; int32 var_6945 = const()[name = string("op_6945"), val = int32(-1)]; bool var_6946_interleave_0 = const()[name = string("op_6946_interleave_0"), val = bool(false)]; tensor var_6946 = concat(axis = var_6945, interleave = var_6946_interleave_0, values = (var_6943, var_6941_0))[name = string("op_6946")]; tensor var_6947_cast_fp16 = mul(x = var_6946, y = sin_f)[name = string("op_6947_cast_fp16")]; tensor q_cast_fp16 = add(x = var_6940_cast_fp16, y = var_6947_cast_fp16)[name = string("q_cast_fp16")]; string var_6960_pad_type_0 = const()[name = string("op_6960_pad_type_0"), val = string("valid")]; tensor var_6960_strides_0 = const()[name = string("op_6960_strides_0"), val = tensor([1, 1])]; tensor var_6960_pad_0 = const()[name = string("op_6960_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6960_dilations_0 = const()[name = string("op_6960_dilations_0"), val = tensor([1, 1])]; int32 var_6960_groups_0 = const()[name = string("op_6960_groups_0"), val = int32(1)]; tensor var_6960 = conv(dilations = var_6960_dilations_0, groups = var_6960_groups_0, pad = var_6960_pad_0, pad_type = var_6960_pad_type_0, strides = var_6960_strides_0, weight = layers_11_self_attn_k_proj_weight_palettized, x = var_6881_cast_fp16)[name = string("op_6960")]; tensor var_6965 = const()[name = string("op_6965"), val = tensor([1, 2, 512, 1])]; tensor var_6966 = reshape(shape = var_6965, x = var_6960)[name = string("op_6966")]; tensor var_6971 = const()[name = string("op_6971"), val = tensor([0, 1, 3, 2])]; string var_6988_pad_type_0 = const()[name = string("op_6988_pad_type_0"), val = string("valid")]; tensor var_6988_strides_0 = const()[name = string("op_6988_strides_0"), val = tensor([1, 1])]; tensor var_6988_pad_0 = const()[name = string("op_6988_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6988_dilations_0 = const()[name = string("op_6988_dilations_0"), val = tensor([1, 1])]; int32 var_6988_groups_0 = const()[name = string("op_6988_groups_0"), val = int32(1)]; tensor var_6988 = conv(dilations = var_6988_dilations_0, groups = var_6988_groups_0, pad = var_6988_pad_0, pad_type = var_6988_pad_type_0, strides = var_6988_strides_0, weight = layers_11_self_attn_v_proj_weight_palettized, x = var_6881_cast_fp16)[name = string("op_6988")]; tensor var_6993 = const()[name = string("op_6993"), val = tensor([1, 2, 512, 1])]; tensor var_6994 = reshape(shape = var_6993, x = var_6988)[name = string("op_6994")]; tensor var_6999 = const()[name = string("op_6999"), val = tensor([0, 1, 3, 2])]; tensor var_7009 = const()[name = string("op_7009"), val = tensor([1, 2, 512])]; tensor var_6972 = transpose(perm = var_6971, x = var_6966)[name = string("transpose_15")]; tensor x_223 = reshape(shape = var_7009, x = var_6972)[name = string("x_223")]; int32 var_7015 = const()[name = string("op_7015"), val = int32(-1)]; fp16 const_133_promoted = const()[name = string("const_133_promoted"), val = fp16(-0x1p+0)]; tensor var_7017 = mul(x = x_223, y = const_133_promoted)[name = string("op_7017")]; bool input_333_interleave_0 = const()[name = string("input_333_interleave_0"), val = bool(false)]; tensor input_333 = concat(axis = var_7015, interleave = input_333_interleave_0, values = (x_223, var_7017))[name = string("input_333")]; tensor normed_317_axes_0 = const()[name = string("normed_317_axes_0"), val = tensor([-1])]; fp16 var_7012_to_fp16 = const()[name = string("op_7012_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_317_cast_fp16 = layer_norm(axes = normed_317_axes_0, epsilon = var_7012_to_fp16, x = input_333)[name = string("normed_317_cast_fp16")]; tensor var_7022_split_sizes_0 = const()[name = string("op_7022_split_sizes_0"), val = tensor([512, 512])]; int32 var_7022_axis_0 = const()[name = string("op_7022_axis_0"), val = int32(-1)]; tensor var_7022_0, tensor var_7022_1 = split(axis = var_7022_axis_0, split_sizes = var_7022_split_sizes_0, x = normed_317_cast_fp16)[name = string("op_7022")]; tensor var_7024 = mul(x = var_7022_0, y = layers_11_self_attn_k_norm_weight)[name = string("op_7024")]; tensor var_7029 = const()[name = string("op_7029"), val = tensor([1, 2, 1, 512])]; tensor q_93 = reshape(shape = var_7029, x = var_7024)[name = string("q_93")]; fp16 var_7031_promoted = const()[name = string("op_7031_promoted"), val = fp16(0x1p+1)]; tensor var_7000 = transpose(perm = var_6999, x = var_6994)[name = string("transpose_14")]; tensor var_7032 = pow(x = var_7000, y = var_7031_promoted)[name = string("op_7032")]; tensor var_7037_axes_0 = const()[name = string("op_7037_axes_0"), val = tensor([-1])]; bool var_7037_keep_dims_0 = const()[name = string("op_7037_keep_dims_0"), val = bool(true)]; tensor var_7037 = reduce_mean(axes = var_7037_axes_0, keep_dims = var_7037_keep_dims_0, x = var_7032)[name = string("op_7037")]; fp16 var_7039_to_fp16 = const()[name = string("op_7039_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_cast_fp16 = add(x = var_7037, y = var_7039_to_fp16)[name = string("mean_sq_cast_fp16")]; fp32 var_7041_epsilon_0 = const()[name = string("op_7041_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_7041_cast_fp16 = rsqrt(epsilon = var_7041_epsilon_0, x = mean_sq_cast_fp16)[name = string("op_7041_cast_fp16")]; tensor v_cast_fp16 = mul(x = var_7000, y = var_7041_cast_fp16)[name = string("v_cast_fp16")]; tensor var_7043_cast_fp16 = mul(x = q_93, y = cos_f)[name = string("op_7043_cast_fp16")]; tensor var_7044_split_sizes_0 = const()[name = string("op_7044_split_sizes_0"), val = tensor([256, 256])]; int32 var_7044_axis_0 = const()[name = string("op_7044_axis_0"), val = int32(-1)]; tensor var_7044_0, tensor var_7044_1 = split(axis = var_7044_axis_0, split_sizes = var_7044_split_sizes_0, x = q_93)[name = string("op_7044")]; fp16 const_134_promoted = const()[name = string("const_134_promoted"), val = fp16(-0x1p+0)]; tensor var_7046 = mul(x = var_7044_1, y = const_134_promoted)[name = string("op_7046")]; int32 var_7048 = const()[name = string("op_7048"), val = int32(-1)]; bool var_7049_interleave_0 = const()[name = string("op_7049_interleave_0"), val = bool(false)]; tensor var_7049 = concat(axis = var_7048, interleave = var_7049_interleave_0, values = (var_7046, var_7044_0))[name = string("op_7049")]; tensor var_7050_cast_fp16 = mul(x = var_7049, y = sin_f)[name = string("op_7050_cast_fp16")]; tensor k_cast_fp16 = add(x = var_7043_cast_fp16, y = var_7050_cast_fp16)[name = string("k_cast_fp16")]; tensor var_7056_cast_fp16 = mul(x = K_full_slot_cast_fp16, y = var_3733_cast_fp16)[name = string("op_7056_cast_fp16")]; tensor var_7057_reps_0 = const()[name = string("op_7057_reps_0"), val = tensor([1, 1, 2048, 1])]; tensor var_7057_cast_fp16 = tile(reps = var_7057_reps_0, x = k_cast_fp16)[name = string("op_7057_cast_fp16")]; tensor var_7058_cast_fp16 = mul(x = var_7057_cast_fp16, y = update_mask)[name = string("op_7058_cast_fp16")]; tensor kv14_k = add(x = var_7056_cast_fp16, y = var_7058_cast_fp16)[name = string("K_full_out_cast_fp16")]; tensor var_7064_cast_fp16 = mul(x = V_full_slot_cast_fp16, y = var_3733_cast_fp16)[name = string("op_7064_cast_fp16")]; tensor var_7065_reps_0 = const()[name = string("op_7065_reps_0"), val = tensor([1, 1, 2048, 1])]; tensor var_7065_cast_fp16 = tile(reps = var_7065_reps_0, x = v_cast_fp16)[name = string("op_7065_cast_fp16")]; tensor var_7066_cast_fp16 = mul(x = var_7065_cast_fp16, y = update_mask)[name = string("op_7066_cast_fp16")]; tensor kv14_v = add(x = var_7064_cast_fp16, y = var_7066_cast_fp16)[name = string("V_full_out_cast_fp16")]; tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_22_reps_0 = const()[name = string("tile_22_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = kv14_k)[name = string("transpose_13")]; tensor tile_22_cast_fp16 = tile(reps = tile_22_reps_0, x = transpose_44_cast_fp16)[name = string("tile_22_cast_fp16")]; tensor concat_44 = const()[name = string("concat_44"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_44_cast_fp16 = reshape(shape = concat_44, x = tile_22_cast_fp16)[name = string("reshape_44_cast_fp16")]; tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_12")]; tensor reshape_45_cast_fp16 = reshape(shape = concat_45, x = transpose_45_cast_fp16)[name = string("reshape_45_cast_fp16")]; tensor transpose_59_perm_0 = const()[name = string("transpose_59_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_46_perm_0 = const()[name = string("transpose_46_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_23_reps_0 = const()[name = string("tile_23_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_46_cast_fp16 = transpose(perm = transpose_46_perm_0, x = kv14_v)[name = string("transpose_11")]; tensor tile_23_cast_fp16 = tile(reps = tile_23_reps_0, x = transpose_46_cast_fp16)[name = string("tile_23_cast_fp16")]; tensor concat_46 = const()[name = string("concat_46"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_46_cast_fp16 = reshape(shape = concat_46, x = tile_23_cast_fp16)[name = string("reshape_46_cast_fp16")]; tensor transpose_47_perm_0 = const()[name = string("transpose_47_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_47 = const()[name = string("concat_47"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_47_cast_fp16 = transpose(perm = transpose_47_perm_0, x = reshape_46_cast_fp16)[name = string("transpose_10")]; tensor reshape_47_cast_fp16 = reshape(shape = concat_47, x = transpose_47_cast_fp16)[name = string("reshape_47_cast_fp16")]; tensor V_expanded_perm_0 = const()[name = string("V_expanded_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_45_transpose_x_0 = const()[name = string("attn_weights_45_transpose_x_0"), val = bool(false)]; bool attn_weights_45_transpose_y_0 = const()[name = string("attn_weights_45_transpose_y_0"), val = bool(false)]; tensor transpose_59_cast_fp16 = transpose(perm = transpose_59_perm_0, x = reshape_45_cast_fp16)[name = string("transpose_9")]; tensor attn_weights_45_cast_fp16 = matmul(transpose_x = attn_weights_45_transpose_x_0, transpose_y = attn_weights_45_transpose_y_0, x = q_cast_fp16, y = transpose_59_cast_fp16)[name = string("attn_weights_45_cast_fp16")]; tensor x_227_cast_fp16 = add(x = attn_weights_45_cast_fp16, y = causal_mask_full)[name = string("x_227_cast_fp16")]; tensor reduce_max_11_axes_0 = const()[name = string("reduce_max_11_axes_0"), val = tensor([-1])]; bool reduce_max_11_keep_dims_0 = const()[name = string("reduce_max_11_keep_dims_0"), val = bool(true)]; tensor reduce_max_11 = reduce_max(axes = reduce_max_11_axes_0, keep_dims = reduce_max_11_keep_dims_0, x = x_227_cast_fp16)[name = string("reduce_max_11")]; tensor var_7118 = sub(x = x_227_cast_fp16, y = reduce_max_11)[name = string("op_7118")]; tensor var_7124 = exp(x = var_7118)[name = string("op_7124")]; tensor var_7134_axes_0 = const()[name = string("op_7134_axes_0"), val = tensor([-1])]; bool var_7134_keep_dims_0 = const()[name = string("op_7134_keep_dims_0"), val = bool(true)]; tensor var_7134 = reduce_sum(axes = var_7134_axes_0, keep_dims = var_7134_keep_dims_0, x = var_7124)[name = string("op_7134")]; tensor var_7140_cast_fp16 = real_div(x = var_7124, y = var_7134)[name = string("op_7140_cast_fp16")]; bool attn_output_67_transpose_x_0 = const()[name = string("attn_output_67_transpose_x_0"), val = bool(false)]; bool attn_output_67_transpose_y_0 = const()[name = string("attn_output_67_transpose_y_0"), val = bool(false)]; tensor V_expanded_cast_fp16 = transpose(perm = V_expanded_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_8")]; tensor attn_output_67_cast_fp16 = matmul(transpose_x = attn_output_67_transpose_x_0, transpose_y = attn_output_67_transpose_y_0, x = var_7140_cast_fp16, y = V_expanded_cast_fp16)[name = string("attn_output_67_cast_fp16")]; tensor var_7151 = const()[name = string("op_7151"), val = tensor([0, 2, 1, 3])]; tensor var_7158 = const()[name = string("op_7158"), val = tensor([1, 1, -1])]; tensor var_7152_cast_fp16 = transpose(perm = var_7151, x = attn_output_67_cast_fp16)[name = string("transpose_7")]; tensor attn_output_69_cast_fp16 = reshape(shape = var_7158, x = var_7152_cast_fp16)[name = string("attn_output_69_cast_fp16")]; tensor var_7163 = const()[name = string("op_7163"), val = tensor([0, 2, 1])]; string var_7179_pad_type_0 = const()[name = string("op_7179_pad_type_0"), val = string("valid")]; int32 var_7179_groups_0 = const()[name = string("op_7179_groups_0"), val = int32(1)]; tensor var_7179_strides_0 = const()[name = string("op_7179_strides_0"), val = tensor([1])]; tensor var_7179_pad_0 = const()[name = string("op_7179_pad_0"), val = tensor([0, 0])]; tensor var_7179_dilations_0 = const()[name = string("op_7179_dilations_0"), val = tensor([1])]; tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566605504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571848448))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7164_cast_fp16 = transpose(perm = var_7163, x = attn_output_69_cast_fp16)[name = string("transpose_6")]; tensor var_7179_cast_fp16 = conv(dilations = var_7179_dilations_0, groups = var_7179_groups_0, pad = var_7179_pad_0, pad_type = var_7179_pad_type_0, strides = var_7179_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_7164_cast_fp16)[name = string("op_7179_cast_fp16")]; tensor var_7183 = const()[name = string("op_7183"), val = tensor([0, 2, 1])]; int32 var_7189 = const()[name = string("op_7189"), val = int32(-1)]; fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_231_cast_fp16 = transpose(perm = var_7183, x = var_7179_cast_fp16)[name = string("transpose_5")]; tensor var_7191_cast_fp16 = mul(x = x_231_cast_fp16, y = const_135_promoted_to_fp16)[name = string("op_7191_cast_fp16")]; bool input_337_interleave_0 = const()[name = string("input_337_interleave_0"), val = bool(false)]; tensor input_337_cast_fp16 = concat(axis = var_7189, interleave = input_337_interleave_0, values = (x_231_cast_fp16, var_7191_cast_fp16))[name = string("input_337_cast_fp16")]; tensor normed_321_axes_0 = const()[name = string("normed_321_axes_0"), val = tensor([-1])]; fp16 var_7186_to_fp16 = const()[name = string("op_7186_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_321_cast_fp16 = layer_norm(axes = normed_321_axes_0, epsilon = var_7186_to_fp16, x = input_337_cast_fp16)[name = string("normed_321_cast_fp16")]; tensor var_7196_split_sizes_0 = const()[name = string("op_7196_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7196_axis_0 = const()[name = string("op_7196_axis_0"), val = int32(-1)]; tensor var_7196_cast_fp16_0, tensor var_7196_cast_fp16_1 = split(axis = var_7196_axis_0, split_sizes = var_7196_split_sizes_0, x = normed_321_cast_fp16)[name = string("op_7196_cast_fp16")]; tensor layers_11_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_11_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571851072)))]; tensor attn_output_cast_fp16 = mul(x = var_7196_cast_fp16_0, y = layers_11_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_cast_fp16")]; tensor x_233_cast_fp16 = add(x = x_219_cast_fp16, y = attn_output_cast_fp16)[name = string("x_233_cast_fp16")]; int32 var_7205 = const()[name = string("op_7205"), val = int32(-1)]; fp16 const_136_promoted_to_fp16 = const()[name = string("const_136_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7207_cast_fp16 = mul(x = x_233_cast_fp16, y = const_136_promoted_to_fp16)[name = string("op_7207_cast_fp16")]; bool input_339_interleave_0 = const()[name = string("input_339_interleave_0"), val = bool(false)]; tensor input_339_cast_fp16 = concat(axis = var_7205, interleave = input_339_interleave_0, values = (x_233_cast_fp16, var_7207_cast_fp16))[name = string("input_339_cast_fp16")]; tensor normed_325_axes_0 = const()[name = string("normed_325_axes_0"), val = tensor([-1])]; fp16 var_7202_to_fp16 = const()[name = string("op_7202_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_325_cast_fp16 = layer_norm(axes = normed_325_axes_0, epsilon = var_7202_to_fp16, x = input_339_cast_fp16)[name = string("normed_325_cast_fp16")]; tensor var_7212_split_sizes_0 = const()[name = string("op_7212_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7212_axis_0 = const()[name = string("op_7212_axis_0"), val = int32(-1)]; tensor var_7212_cast_fp16_0, tensor var_7212_cast_fp16_1 = split(axis = var_7212_axis_0, split_sizes = var_7212_split_sizes_0, x = normed_325_cast_fp16)[name = string("op_7212_cast_fp16")]; tensor layers_11_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_11_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571856256)))]; tensor h_69_cast_fp16 = mul(x = var_7212_cast_fp16_0, y = layers_11_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_69_cast_fp16")]; tensor var_7223 = const()[name = string("op_7223"), val = tensor([0, 2, 1])]; tensor input_341_axes_0 = const()[name = string("input_341_axes_0"), val = tensor([2])]; tensor var_7224 = transpose(perm = var_7223, x = h_69_cast_fp16)[name = string("transpose_4")]; tensor input_341 = expand_dims(axes = input_341_axes_0, x = var_7224)[name = string("input_341")]; string gate_45_pad_type_0 = const()[name = string("gate_45_pad_type_0"), val = string("valid")]; tensor gate_45_strides_0 = const()[name = string("gate_45_strides_0"), val = tensor([1, 1])]; tensor gate_45_pad_0 = const()[name = string("gate_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_45_dilations_0 = const()[name = string("gate_45_dilations_0"), val = tensor([1, 1])]; int32 gate_45_groups_0 = const()[name = string("gate_45_groups_0"), val = int32(1)]; tensor gate_45 = conv(dilations = gate_45_dilations_0, groups = gate_45_groups_0, pad = gate_45_pad_0, pad_type = gate_45_pad_type_0, strides = gate_45_strides_0, weight = layers_11_mlp_gate_proj_weight_palettized, x = input_341)[name = string("gate_45")]; string up_pad_type_0 = const()[name = string("up_pad_type_0"), val = string("valid")]; tensor up_strides_0 = const()[name = string("up_strides_0"), val = tensor([1, 1])]; tensor up_pad_0 = const()[name = string("up_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_dilations_0 = const()[name = string("up_dilations_0"), val = tensor([1, 1])]; int32 up_groups_0 = const()[name = string("up_groups_0"), val = int32(1)]; tensor up = conv(dilations = up_dilations_0, groups = up_groups_0, pad = up_pad_0, pad_type = up_pad_type_0, strides = up_strides_0, weight = layers_11_mlp_up_proj_weight_palettized, x = input_341)[name = string("up")]; string gate_mode_0 = const()[name = string("gate_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate = gelu(mode = gate_mode_0, x = gate_45)[name = string("gate")]; tensor input_343 = mul(x = gate, y = up)[name = string("input_343")]; string mlp_out_pad_type_0 = const()[name = string("mlp_out_pad_type_0"), val = string("valid")]; tensor mlp_out_strides_0 = const()[name = string("mlp_out_strides_0"), val = tensor([1, 1])]; tensor mlp_out_pad_0 = const()[name = string("mlp_out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_dilations_0 = const()[name = string("mlp_out_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_groups_0 = const()[name = string("mlp_out_groups_0"), val = int32(1)]; tensor mlp_out = conv(dilations = mlp_out_dilations_0, groups = mlp_out_groups_0, pad = mlp_out_pad_0, pad_type = mlp_out_pad_type_0, strides = mlp_out_strides_0, weight = layers_11_mlp_down_proj_weight_palettized, x = input_343)[name = string("mlp_out")]; tensor var_7264_axes_0 = const()[name = string("op_7264_axes_0"), val = tensor([2])]; tensor var_7264 = squeeze(axes = var_7264_axes_0, x = mlp_out)[name = string("op_7264")]; tensor var_7268 = const()[name = string("op_7268"), val = tensor([0, 2, 1])]; int32 var_7274 = const()[name = string("op_7274"), val = int32(-1)]; fp16 const_137_promoted = const()[name = string("const_137_promoted"), val = fp16(-0x1p+0)]; tensor x_235 = transpose(perm = var_7268, x = var_7264)[name = string("transpose_3")]; tensor var_7276 = mul(x = x_235, y = const_137_promoted)[name = string("op_7276")]; bool input_345_interleave_0 = const()[name = string("input_345_interleave_0"), val = bool(false)]; tensor input_345 = concat(axis = var_7274, interleave = input_345_interleave_0, values = (x_235, var_7276))[name = string("input_345")]; tensor normed_329_axes_0 = const()[name = string("normed_329_axes_0"), val = tensor([-1])]; fp16 var_7271_to_fp16 = const()[name = string("op_7271_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_329_cast_fp16 = layer_norm(axes = normed_329_axes_0, epsilon = var_7271_to_fp16, x = input_345)[name = string("normed_329_cast_fp16")]; tensor var_7281_split_sizes_0 = const()[name = string("op_7281_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7281_axis_0 = const()[name = string("op_7281_axis_0"), val = int32(-1)]; tensor var_7281_0, tensor var_7281_1 = split(axis = var_7281_axis_0, split_sizes = var_7281_split_sizes_0, x = normed_329_cast_fp16)[name = string("op_7281")]; tensor hidden_states_113 = mul(x = var_7281_0, y = layers_11_post_feedforward_layernorm_weight)[name = string("hidden_states_113")]; tensor hidden_states_115_cast_fp16 = add(x = x_233_cast_fp16, y = hidden_states_113)[name = string("hidden_states_115_cast_fp16")]; tensor per_layer_slice_begin_0 = const()[name = string("per_layer_slice_begin_0"), val = tensor([0, 0, 5888])]; tensor per_layer_slice_end_0 = const()[name = string("per_layer_slice_end_0"), val = tensor([1, 1, 6144])]; tensor per_layer_slice_end_mask_0 = const()[name = string("per_layer_slice_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_cast_fp16 = slice_by_index(begin = per_layer_slice_begin_0, end = per_layer_slice_end_0, end_mask = per_layer_slice_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_cast_fp16")]; tensor var_7309 = const()[name = string("op_7309"), val = tensor([0, 2, 1])]; tensor input_347_axes_0 = const()[name = string("input_347_axes_0"), val = tensor([2])]; tensor var_7310 = transpose(perm = var_7309, x = hidden_states_115_cast_fp16)[name = string("transpose_2")]; tensor input_347 = expand_dims(axes = input_347_axes_0, x = var_7310)[name = string("input_347")]; string gated_67_pad_type_0 = const()[name = string("gated_67_pad_type_0"), val = string("valid")]; tensor gated_67_strides_0 = const()[name = string("gated_67_strides_0"), val = tensor([1, 1])]; tensor gated_67_pad_0 = const()[name = string("gated_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_67_dilations_0 = const()[name = string("gated_67_dilations_0"), val = tensor([1, 1])]; int32 gated_67_groups_0 = const()[name = string("gated_67_groups_0"), val = int32(1)]; tensor gated_67 = conv(dilations = gated_67_dilations_0, groups = gated_67_groups_0, pad = gated_67_pad_0, pad_type = gated_67_pad_type_0, strides = gated_67_strides_0, weight = layers_11_per_layer_input_gate_weight_palettized, x = input_347)[name = string("gated_67")]; string gated_69_mode_0 = const()[name = string("gated_69_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_69 = gelu(mode = gated_69_mode_0, x = gated_67)[name = string("gated_69")]; tensor var_7329 = const()[name = string("op_7329"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_axes_0 = const()[name = string("per_layer_slice_conv_axes_0"), val = tensor([2])]; tensor var_7330_cast_fp16 = transpose(perm = var_7329, x = per_layer_slice_cast_fp16)[name = string("transpose_1")]; tensor per_layer_slice_conv_cast_fp16 = expand_dims(axes = per_layer_slice_conv_axes_0, x = var_7330_cast_fp16)[name = string("per_layer_slice_conv_cast_fp16")]; tensor input_349_cast_fp16 = mul(x = gated_69, y = per_layer_slice_conv_cast_fp16)[name = string("input_349_cast_fp16")]; string gated_pad_type_0 = const()[name = string("gated_pad_type_0"), val = string("valid")]; tensor gated_strides_0 = const()[name = string("gated_strides_0"), val = tensor([1, 1])]; tensor gated_pad_0 = const()[name = string("gated_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_dilations_0 = const()[name = string("gated_dilations_0"), val = tensor([1, 1])]; int32 gated_groups_0 = const()[name = string("gated_groups_0"), val = int32(1)]; tensor layers_11_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571861440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572189184))))[name = string("layers_11_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_cast_fp16 = conv(dilations = gated_dilations_0, groups = gated_groups_0, pad = gated_pad_0, pad_type = gated_pad_type_0, strides = gated_strides_0, weight = layers_11_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_349_cast_fp16)[name = string("gated_cast_fp16")]; tensor var_7346_axes_0 = const()[name = string("op_7346_axes_0"), val = tensor([2])]; tensor var_7346_cast_fp16 = squeeze(axes = var_7346_axes_0, x = gated_cast_fp16)[name = string("op_7346_cast_fp16")]; tensor var_7350 = const()[name = string("op_7350"), val = tensor([0, 2, 1])]; int32 var_7356 = const()[name = string("op_7356"), val = int32(-1)]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_cast_fp16 = transpose(perm = var_7350, x = var_7346_cast_fp16)[name = string("transpose_0")]; tensor var_7358_cast_fp16 = mul(x = x_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_7358_cast_fp16")]; bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; tensor input_cast_fp16 = concat(axis = var_7356, interleave = input_interleave_0, values = (x_cast_fp16, var_7358_cast_fp16))[name = string("input_cast_fp16")]; tensor normed_333_axes_0 = const()[name = string("normed_333_axes_0"), val = tensor([-1])]; fp16 var_7353_to_fp16 = const()[name = string("op_7353_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_333_cast_fp16 = layer_norm(axes = normed_333_axes_0, epsilon = var_7353_to_fp16, x = input_cast_fp16)[name = string("normed_333_cast_fp16")]; tensor var_7363_split_sizes_0 = const()[name = string("op_7363_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7363_axis_0 = const()[name = string("op_7363_axis_0"), val = int32(-1)]; tensor var_7363_cast_fp16_0, tensor var_7363_cast_fp16_1 = split(axis = var_7363_axis_0, split_sizes = var_7363_split_sizes_0, x = normed_333_cast_fp16)[name = string("op_7363_cast_fp16")]; tensor layers_11_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_11_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572191808)))]; tensor hidden_states_119_cast_fp16 = mul(x = var_7363_cast_fp16_0, y = layers_11_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_119_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_119_cast_fp16)[name = string("hidden_states_cast_fp16")]; tensor const_139_promoted_to_fp16 = const()[name = string("const_139_promoted_to_fp16"), val = tensor([0x1.0cp-4])]; tensor hidden_states_out = mul(x = hidden_states_cast_fp16, y = const_139_promoted_to_fp16)[name = string("op_7373_cast_fp16")]; tensor var_7375_axes_0 = const()[name = string("op_7375_axes_0"), val = tensor([0])]; tensor var_7375_cast_fp16 = squeeze(axes = var_7375_axes_0, x = kv14_k)[name = string("op_7375_cast_fp16")]; tensor var_7377_axes_0 = const()[name = string("op_7377_axes_0"), val = tensor([0])]; tensor var_7377_cast_fp16 = squeeze(axes = var_7377_axes_0, x = kv14_v)[name = string("op_7377_cast_fp16")]; int32 var_7380_axis_0 = const()[name = string("op_7380_axis_0"), val = int32(0)]; tensor K_sliding_out = stack(axis = var_7380_axis_0, values = (var_1290_cast_fp16, var_1849_cast_fp16, var_2408_cast_fp16, var_2967_cast_fp16, var_3526_cast_fp16, var_4602_cast_fp16, var_5161_cast_fp16, var_5720_cast_fp16, var_6279_cast_fp16, var_6848_cast_fp16))[name = string("op_7380_cast_fp16")]; int32 var_7383_axis_0 = const()[name = string("op_7383_axis_0"), val = int32(0)]; tensor V_sliding_out = stack(axis = var_7383_axis_0, values = (var_1292_cast_fp16, var_1851_cast_fp16, var_2410_cast_fp16, var_2969_cast_fp16, var_3528_cast_fp16, var_4604_cast_fp16, var_5163_cast_fp16, var_5722_cast_fp16, var_6281_cast_fp16, var_6850_cast_fp16))[name = string("op_7383_cast_fp16")]; int32 var_7386_axis_0 = const()[name = string("op_7386_axis_0"), val = int32(0)]; tensor K_full_out = stack(axis = var_7386_axis_0, values = (var_4043_cast_fp16, var_7375_cast_fp16))[name = string("op_7386_cast_fp16")]; int32 var_7389_axis_0 = const()[name = string("op_7389_axis_0"), val = int32(0)]; tensor V_full_out = stack(axis = var_7389_axis_0, values = (var_4045_cast_fp16, var_7377_cast_fp16))[name = string("op_7389_cast_fp16")]; } -> (hidden_states_out, K_sliding_out, V_sliding_out, K_full_out, V_full_out, kv13_k, kv13_v, kv14_k, kv14_v); func verify_qK(tensor K_full_in, tensor K_sliding_in, tensor V_full_in, tensor V_sliding_in, tensor causal_mask_full, tensor causal_mask_sliding, tensor cos_f, tensor cos_s, tensor hidden_states, tensor per_layer_combined, tensor sin_f, tensor sin_s, tensor update_indicator) { tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2621568))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_q_norm_weight = const()[name = string("layers_0_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2623680)))]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2624256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3279680))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3280256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3935680))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_self_attn_k_norm_weight = const()[name = string("layers_0_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3936256)))]; tensor layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3936832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17044096))))[name = string("layers_0_mlp_gate_proj_weight_palettized")]; tensor layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17054400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30161664))))[name = string("layers_0_mlp_up_proj_weight_palettized")]; tensor layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30171968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43279232))))[name = string("layers_0_mlp_down_proj_weight_palettized")]; tensor layers_0_post_feedforward_layernorm_weight = const()[name = string("layers_0_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43281856)))]; tensor layers_0_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43287040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43614784))))[name = string("layers_0_per_layer_input_gate_weight_palettized")]; tensor layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43615104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46236608))))[name = string("layers_1_self_attn_q_proj_weight_palettized")]; tensor layers_1_self_attn_q_norm_weight = const()[name = string("layers_1_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46238720)))]; tensor layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46239296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46894720))))[name = string("layers_1_self_attn_k_proj_weight_palettized")]; tensor layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46895296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47550720))))[name = string("layers_1_self_attn_v_proj_weight_palettized")]; tensor layers_1_self_attn_k_norm_weight = const()[name = string("layers_1_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47551296)))]; tensor layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47551872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60659136))))[name = string("layers_1_mlp_gate_proj_weight_palettized")]; tensor layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60669440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73776704))))[name = string("layers_1_mlp_up_proj_weight_palettized")]; tensor layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73787008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86894272))))[name = string("layers_1_mlp_down_proj_weight_palettized")]; tensor layers_1_post_feedforward_layernorm_weight = const()[name = string("layers_1_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86896896)))]; tensor layers_1_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86902080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87229824))))[name = string("layers_1_per_layer_input_gate_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87230144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89851648))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_q_norm_weight = const()[name = string("layers_2_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89853760)))]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89854336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90509760))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90510336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91165760))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_self_attn_k_norm_weight = const()[name = string("layers_2_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91166336)))]; tensor layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91166912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104274176))))[name = string("layers_2_mlp_gate_proj_weight_palettized")]; tensor layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104284480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117391744))))[name = string("layers_2_mlp_up_proj_weight_palettized")]; tensor layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117402048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130509312))))[name = string("layers_2_mlp_down_proj_weight_palettized")]; tensor layers_2_post_feedforward_layernorm_weight = const()[name = string("layers_2_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130511936)))]; tensor layers_2_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130517120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130844864))))[name = string("layers_2_per_layer_input_gate_weight_palettized")]; tensor layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130845184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133466688))))[name = string("layers_3_self_attn_q_proj_weight_palettized")]; tensor layers_3_self_attn_q_norm_weight = const()[name = string("layers_3_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133468800)))]; tensor layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133469376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134124800))))[name = string("layers_3_self_attn_k_proj_weight_palettized")]; tensor layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134125376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134780800))))[name = string("layers_3_self_attn_v_proj_weight_palettized")]; tensor layers_3_self_attn_k_norm_weight = const()[name = string("layers_3_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134781376)))]; tensor layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134781952))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147889216))))[name = string("layers_3_mlp_gate_proj_weight_palettized")]; tensor layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147899520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161006784))))[name = string("layers_3_mlp_up_proj_weight_palettized")]; tensor layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161017088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174124352))))[name = string("layers_3_mlp_down_proj_weight_palettized")]; tensor layers_3_post_feedforward_layernorm_weight = const()[name = string("layers_3_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174126976)))]; tensor layers_3_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174132160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174459904))))[name = string("layers_3_per_layer_input_gate_weight_palettized")]; tensor layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174460224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177081728))))[name = string("layers_4_self_attn_q_proj_weight_palettized")]; tensor layers_4_self_attn_q_norm_weight = const()[name = string("layers_4_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177083840)))]; tensor layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177084416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177739840))))[name = string("layers_4_self_attn_k_proj_weight_palettized")]; tensor layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(177740416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178395840))))[name = string("layers_4_self_attn_v_proj_weight_palettized")]; tensor layers_4_self_attn_k_norm_weight = const()[name = string("layers_4_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178396416)))]; tensor layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178396992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(191504256))))[name = string("layers_4_mlp_gate_proj_weight_palettized")]; tensor layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(191514560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204621824))))[name = string("layers_4_mlp_up_proj_weight_palettized")]; tensor layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204632128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217739392))))[name = string("layers_4_mlp_down_proj_weight_palettized")]; tensor layers_4_post_feedforward_layernorm_weight = const()[name = string("layers_4_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217742016)))]; tensor layers_4_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217747200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218074944))))[name = string("layers_4_per_layer_input_gate_weight_palettized")]; tensor layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218075264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223318208))))[name = string("layers_5_self_attn_q_proj_weight_palettized")]; tensor layers_5_self_attn_q_norm_weight = const()[name = string("layers_5_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223322368)))]; tensor layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223323456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224634240))))[name = string("layers_5_self_attn_k_proj_weight_palettized")]; tensor layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224635328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225946112))))[name = string("layers_5_self_attn_v_proj_weight_palettized")]; tensor layers_5_self_attn_k_norm_weight = const()[name = string("layers_5_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225947200)))]; tensor layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(225948288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239055552))))[name = string("layers_5_mlp_gate_proj_weight_palettized")]; tensor layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239065856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252173120))))[name = string("layers_5_mlp_up_proj_weight_palettized")]; tensor layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252183424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265290688))))[name = string("layers_5_mlp_down_proj_weight_palettized")]; tensor layers_5_post_feedforward_layernorm_weight = const()[name = string("layers_5_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265293312)))]; tensor layers_5_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265298496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265626240))))[name = string("layers_5_per_layer_input_gate_weight_palettized")]; tensor layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265626560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268248064))))[name = string("layers_6_self_attn_q_proj_weight_palettized")]; tensor layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268250176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268905600))))[name = string("layers_6_self_attn_k_proj_weight_palettized")]; tensor layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268906176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269561600))))[name = string("layers_6_self_attn_v_proj_weight_palettized")]; tensor layers_6_self_attn_k_norm_weight = const()[name = string("layers_6_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269562176)))]; tensor layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(269562752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282670016))))[name = string("layers_6_mlp_gate_proj_weight_palettized")]; tensor layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282680320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295787584))))[name = string("layers_6_mlp_up_proj_weight_palettized")]; tensor layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295797888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308905152))))[name = string("layers_6_mlp_down_proj_weight_palettized")]; tensor layers_6_post_feedforward_layernorm_weight = const()[name = string("layers_6_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308907776)))]; tensor layers_6_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308912960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(309240704))))[name = string("layers_6_per_layer_input_gate_weight_palettized")]; tensor layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(309241024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311862528))))[name = string("layers_7_self_attn_q_proj_weight_palettized")]; tensor layers_7_self_attn_q_norm_weight = const()[name = string("layers_7_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311864640)))]; tensor layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311865216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312520640))))[name = string("layers_7_self_attn_k_proj_weight_palettized")]; tensor layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312521216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313176640))))[name = string("layers_7_self_attn_v_proj_weight_palettized")]; tensor layers_7_self_attn_k_norm_weight = const()[name = string("layers_7_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313177216)))]; tensor layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313177792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326285056))))[name = string("layers_7_mlp_gate_proj_weight_palettized")]; tensor layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326295360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339402624))))[name = string("layers_7_mlp_up_proj_weight_palettized")]; tensor layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339412928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352520192))))[name = string("layers_7_mlp_down_proj_weight_palettized")]; tensor layers_7_post_feedforward_layernorm_weight = const()[name = string("layers_7_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352522816)))]; tensor layers_7_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352528000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352855744))))[name = string("layers_7_per_layer_input_gate_weight_palettized")]; tensor layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352856064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355477568))))[name = string("layers_8_self_attn_q_proj_weight_palettized")]; tensor layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355479680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356135104))))[name = string("layers_8_self_attn_k_proj_weight_palettized")]; tensor layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356135680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356791104))))[name = string("layers_8_self_attn_v_proj_weight_palettized")]; tensor layers_8_self_attn_k_norm_weight = const()[name = string("layers_8_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356791680)))]; tensor layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356792256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369899520))))[name = string("layers_8_mlp_gate_proj_weight_palettized")]; tensor layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369909824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(383017088))))[name = string("layers_8_mlp_up_proj_weight_palettized")]; tensor layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(383027392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396134656))))[name = string("layers_8_mlp_down_proj_weight_palettized")]; tensor layers_8_post_feedforward_layernorm_weight = const()[name = string("layers_8_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396137280)))]; tensor layers_8_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396142464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396470208))))[name = string("layers_8_per_layer_input_gate_weight_palettized")]; tensor layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(396470528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399092032))))[name = string("layers_9_self_attn_q_proj_weight_palettized")]; tensor layers_9_self_attn_q_norm_weight = const()[name = string("layers_9_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399094144)))]; tensor layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399094720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399750144))))[name = string("layers_9_self_attn_k_proj_weight_palettized")]; tensor layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399750720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400406144))))[name = string("layers_9_self_attn_v_proj_weight_palettized")]; tensor layers_9_self_attn_k_norm_weight = const()[name = string("layers_9_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400406720)))]; tensor layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400407296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413514560))))[name = string("layers_9_mlp_gate_proj_weight_palettized")]; tensor layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413524864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(426632128))))[name = string("layers_9_mlp_up_proj_weight_palettized")]; tensor layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(426642432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(439749696))))[name = string("layers_9_mlp_down_proj_weight_palettized")]; tensor layers_9_post_feedforward_layernorm_weight = const()[name = string("layers_9_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(439752320)))]; tensor layers_9_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(439757504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(440085248))))[name = string("layers_9_per_layer_input_gate_weight_palettized")]; tensor layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(440085568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(442707072))))[name = string("layers_10_self_attn_q_proj_weight_palettized")]; tensor layers_10_self_attn_q_norm_weight = const()[name = string("layers_10_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(442709184)))]; tensor layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(442709760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(443365184))))[name = string("layers_10_self_attn_k_proj_weight_palettized")]; tensor layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(443365760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(444021184))))[name = string("layers_10_self_attn_v_proj_weight_palettized")]; tensor layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(444021760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(457129024))))[name = string("layers_10_mlp_gate_proj_weight_palettized")]; tensor layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(457139328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(470246592))))[name = string("layers_10_mlp_up_proj_weight_palettized")]; tensor layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(470256896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483364160))))[name = string("layers_10_mlp_down_proj_weight_palettized")]; tensor layers_10_post_feedforward_layernorm_weight = const()[name = string("layers_10_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483366784)))]; tensor layers_10_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483371968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483699712))))[name = string("layers_10_per_layer_input_gate_weight_palettized")]; tensor layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483700032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488942976))))[name = string("layers_11_self_attn_q_proj_weight_palettized")]; tensor layers_11_self_attn_q_norm_weight = const()[name = string("layers_11_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488947136)))]; tensor layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488948224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490259008))))[name = string("layers_11_self_attn_k_proj_weight_palettized")]; tensor layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490260096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(491570880))))[name = string("layers_11_self_attn_v_proj_weight_palettized")]; tensor layers_11_self_attn_k_norm_weight = const()[name = string("layers_11_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(491571968)))]; tensor layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(491573056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(504680320))))[name = string("layers_11_mlp_gate_proj_weight_palettized")]; tensor layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(504690624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(517797888))))[name = string("layers_11_mlp_up_proj_weight_palettized")]; tensor layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(517808192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530915456))))[name = string("layers_11_mlp_down_proj_weight_palettized")]; tensor layers_11_post_feedforward_layernorm_weight = const()[name = string("layers_11_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530918080)))]; tensor layers_11_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530923264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531251008))))[name = string("layers_11_per_layer_input_gate_weight_palettized")]; int32 var_738 = const()[name = string("op_738"), val = int32(-1)]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_740_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_740_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_738, interleave = input_1_interleave_0, values = (hidden_states, var_740_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_735_to_fp16 = const()[name = string("op_735_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_735_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_745_split_sizes_0 = const()[name = string("op_745_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_745_axis_0 = const()[name = string("op_745_axis_0"), val = int32(-1)]; tensor var_745_cast_fp16_0, tensor var_745_cast_fp16_1 = split(axis = var_745_axis_0, split_sizes = var_745_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_745_cast_fp16")]; tensor layers_0_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531251328)))]; tensor h_1_cast_fp16 = mul(x = var_745_cast_fp16_0, y = layers_0_input_layernorm_weight_promoted_to_fp16)[name = string("h_1_cast_fp16")]; tensor var_751 = const()[name = string("op_751"), val = tensor([0, 2, 1])]; tensor var_754_axes_0 = const()[name = string("op_754_axes_0"), val = tensor([2])]; tensor var_752_cast_fp16 = transpose(perm = var_751, x = h_1_cast_fp16)[name = string("transpose_239")]; tensor var_754_cast_fp16 = expand_dims(axes = var_754_axes_0, x = var_752_cast_fp16)[name = string("op_754_cast_fp16")]; string q_1_pad_type_0 = const()[name = string("q_1_pad_type_0"), val = string("valid")]; tensor q_1_strides_0 = const()[name = string("q_1_strides_0"), val = tensor([1, 1])]; tensor q_1_pad_0 = const()[name = string("q_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_1_dilations_0 = const()[name = string("q_1_dilations_0"), val = tensor([1, 1])]; int32 q_1_groups_0 = const()[name = string("q_1_groups_0"), val = int32(1)]; tensor q_1 = conv(dilations = q_1_dilations_0, groups = q_1_groups_0, pad = q_1_pad_0, pad_type = q_1_pad_type_0, strides = q_1_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_754_cast_fp16)[name = string("q_1")]; tensor var_775 = const()[name = string("op_775"), val = tensor([1, 8, 256, 3])]; tensor var_776 = reshape(shape = var_775, x = q_1)[name = string("op_776")]; tensor transpose_48_perm_0 = const()[name = string("transpose_48_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_799 = const()[name = string("op_799"), val = tensor([3, 8, 256])]; tensor transpose_48 = transpose(perm = transpose_48_perm_0, x = var_776)[name = string("transpose_238")]; tensor x_1 = reshape(shape = var_799, x = transpose_48)[name = string("x_1")]; int32 var_805 = const()[name = string("op_805"), val = int32(-1)]; fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; tensor var_807 = mul(x = x_1, y = const_1_promoted)[name = string("op_807")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_805, interleave = input_5_interleave_0, values = (x_1, var_807))[name = string("input_5")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_802_to_fp16 = const()[name = string("op_802_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_802_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; tensor var_812_split_sizes_0 = const()[name = string("op_812_split_sizes_0"), val = tensor([256, 256])]; int32 var_812_axis_0 = const()[name = string("op_812_axis_0"), val = int32(-1)]; tensor var_812_0, tensor var_812_1 = split(axis = var_812_axis_0, split_sizes = var_812_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_812")]; tensor q_5 = mul(x = var_812_0, y = layers_0_self_attn_q_norm_weight)[name = string("q_5")]; tensor var_819 = const()[name = string("op_819"), val = tensor([1, 3, 8, 256])]; tensor var_820 = reshape(shape = var_819, x = q_5)[name = string("op_820")]; tensor var_825 = const()[name = string("op_825"), val = tensor([0, 2, 1, 3])]; tensor q_7 = transpose(perm = var_825, x = var_820)[name = string("transpose_237")]; tensor var_827_cast_fp16 = mul(x = q_7, y = cos_s)[name = string("op_827_cast_fp16")]; tensor var_828_split_sizes_0 = const()[name = string("op_828_split_sizes_0"), val = tensor([128, 128])]; int32 var_828_axis_0 = const()[name = string("op_828_axis_0"), val = int32(-1)]; tensor var_828_0, tensor var_828_1 = split(axis = var_828_axis_0, split_sizes = var_828_split_sizes_0, x = q_7)[name = string("op_828")]; fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; tensor var_830 = mul(x = var_828_1, y = const_2_promoted)[name = string("op_830")]; int32 var_832 = const()[name = string("op_832"), val = int32(-1)]; bool var_833_interleave_0 = const()[name = string("op_833_interleave_0"), val = bool(false)]; tensor var_833 = concat(axis = var_832, interleave = var_833_interleave_0, values = (var_830, var_828_0))[name = string("op_833")]; tensor var_834_cast_fp16 = mul(x = var_833, y = sin_s)[name = string("op_834_cast_fp16")]; tensor q_11_cast_fp16 = add(x = var_827_cast_fp16, y = var_834_cast_fp16)[name = string("q_11_cast_fp16")]; string k_1_pad_type_0 = const()[name = string("k_1_pad_type_0"), val = string("valid")]; tensor k_1_strides_0 = const()[name = string("k_1_strides_0"), val = tensor([1, 1])]; tensor k_1_pad_0 = const()[name = string("k_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_1_dilations_0 = const()[name = string("k_1_dilations_0"), val = tensor([1, 1])]; int32 k_1_groups_0 = const()[name = string("k_1_groups_0"), val = int32(1)]; tensor k_1 = conv(dilations = k_1_dilations_0, groups = k_1_groups_0, pad = k_1_pad_0, pad_type = k_1_pad_type_0, strides = k_1_strides_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_754_cast_fp16)[name = string("k_1")]; tensor var_852 = const()[name = string("op_852"), val = tensor([1, 2, 256, 3])]; tensor var_853 = reshape(shape = var_852, x = k_1)[name = string("op_853")]; tensor transpose_49_perm_0 = const()[name = string("transpose_49_perm_0"), val = tensor([0, 3, 1, 2])]; string v_1_pad_type_0 = const()[name = string("v_1_pad_type_0"), val = string("valid")]; tensor v_1_strides_0 = const()[name = string("v_1_strides_0"), val = tensor([1, 1])]; tensor v_1_pad_0 = const()[name = string("v_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_1_dilations_0 = const()[name = string("v_1_dilations_0"), val = tensor([1, 1])]; int32 v_1_groups_0 = const()[name = string("v_1_groups_0"), val = int32(1)]; tensor v_1 = conv(dilations = v_1_dilations_0, groups = v_1_groups_0, pad = v_1_pad_0, pad_type = v_1_pad_type_0, strides = v_1_strides_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_754_cast_fp16)[name = string("v_1")]; tensor var_880 = const()[name = string("op_880"), val = tensor([1, 2, 256, 3])]; tensor var_881 = reshape(shape = var_880, x = v_1)[name = string("op_881")]; tensor var_886 = const()[name = string("op_886"), val = tensor([0, 1, 3, 2])]; tensor var_904 = const()[name = string("op_904"), val = tensor([3, 2, 256])]; tensor transpose_49 = transpose(perm = transpose_49_perm_0, x = var_853)[name = string("transpose_236")]; tensor x_3 = reshape(shape = var_904, x = transpose_49)[name = string("x_3")]; int32 var_910 = const()[name = string("op_910"), val = int32(-1)]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_912 = mul(x = x_3, y = const_3_promoted)[name = string("op_912")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7 = concat(axis = var_910, interleave = input_7_interleave_0, values = (x_3, var_912))[name = string("input_7")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_907_to_fp16 = const()[name = string("op_907_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_907_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; tensor var_917_split_sizes_0 = const()[name = string("op_917_split_sizes_0"), val = tensor([256, 256])]; int32 var_917_axis_0 = const()[name = string("op_917_axis_0"), val = int32(-1)]; tensor var_917_0, tensor var_917_1 = split(axis = var_917_axis_0, split_sizes = var_917_split_sizes_0, x = normed_9_cast_fp16)[name = string("op_917")]; tensor k_5 = mul(x = var_917_0, y = layers_0_self_attn_k_norm_weight)[name = string("k_5")]; tensor var_924 = const()[name = string("op_924"), val = tensor([1, 3, 2, 256])]; tensor var_925 = reshape(shape = var_924, x = k_5)[name = string("op_925")]; tensor var_930 = const()[name = string("op_930"), val = tensor([0, 2, 1, 3])]; fp16 var_932_promoted = const()[name = string("op_932_promoted"), val = fp16(0x1p+1)]; tensor var_887 = transpose(perm = var_886, x = var_881)[name = string("transpose_235")]; tensor var_933 = pow(x = var_887, y = var_932_promoted)[name = string("op_933")]; tensor var_938_axes_0 = const()[name = string("op_938_axes_0"), val = tensor([-1])]; bool var_938_keep_dims_0 = const()[name = string("op_938_keep_dims_0"), val = bool(true)]; tensor var_938 = reduce_mean(axes = var_938_axes_0, keep_dims = var_938_keep_dims_0, x = var_933)[name = string("op_938")]; fp16 var_940_to_fp16 = const()[name = string("op_940_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_1_cast_fp16 = add(x = var_938, y = var_940_to_fp16)[name = string("mean_sq_1_cast_fp16")]; fp32 var_942_epsilon_0 = const()[name = string("op_942_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_942_cast_fp16 = rsqrt(epsilon = var_942_epsilon_0, x = mean_sq_1_cast_fp16)[name = string("op_942_cast_fp16")]; tensor input_11_cast_fp16 = mul(x = var_887, y = var_942_cast_fp16)[name = string("input_11_cast_fp16")]; tensor q_9 = transpose(perm = var_930, x = var_925)[name = string("transpose_234")]; tensor var_944_cast_fp16 = mul(x = q_9, y = cos_s)[name = string("op_944_cast_fp16")]; tensor var_945_split_sizes_0 = const()[name = string("op_945_split_sizes_0"), val = tensor([128, 128])]; int32 var_945_axis_0 = const()[name = string("op_945_axis_0"), val = int32(-1)]; tensor var_945_0, tensor var_945_1 = split(axis = var_945_axis_0, split_sizes = var_945_split_sizes_0, x = q_9)[name = string("op_945")]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_947 = mul(x = var_945_1, y = const_4_promoted)[name = string("op_947")]; int32 var_949 = const()[name = string("op_949"), val = int32(-1)]; bool var_950_interleave_0 = const()[name = string("op_950_interleave_0"), val = bool(false)]; tensor var_950 = concat(axis = var_949, interleave = var_950_interleave_0, values = (var_947, var_945_0))[name = string("op_950")]; tensor var_951_cast_fp16 = mul(x = var_950, y = sin_s)[name = string("op_951_cast_fp16")]; tensor input_9_cast_fp16 = add(x = var_944_cast_fp16, y = var_951_cast_fp16)[name = string("input_9_cast_fp16")]; tensor k_padded_1_pad_0 = const()[name = string("k_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_1_mode_0 = const()[name = string("k_padded_1_mode_0"), val = string("constant")]; fp16 const_5_to_fp16 = const()[name = string("const_5_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_1_cast_fp16 = pad(constant_val = const_5_to_fp16, mode = k_padded_1_mode_0, pad = k_padded_1_pad_0, x = input_9_cast_fp16)[name = string("k_padded_1_cast_fp16")]; tensor v_padded_1_pad_0 = const()[name = string("v_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_1_mode_0 = const()[name = string("v_padded_1_mode_0"), val = string("constant")]; fp16 const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_1_cast_fp16 = pad(constant_val = const_6_to_fp16, mode = v_padded_1_mode_0, pad = v_padded_1_pad_0, x = input_11_cast_fp16)[name = string("v_padded_1_cast_fp16")]; tensor slot_k_1_begin_0 = const()[name = string("slot_k_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slot_k_1_end_0 = const()[name = string("slot_k_1_end_0"), val = tensor([1, 2, 512, 512])]; tensor slot_k_1_end_mask_0 = const()[name = string("slot_k_1_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_1_cast_fp16 = slice_by_index(begin = slot_k_1_begin_0, end = slot_k_1_end_0, end_mask = slot_k_1_end_mask_0, x = K_sliding_in)[name = string("slot_k_1_cast_fp16")]; tensor slot_v_1_begin_0 = const()[name = string("slot_v_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slot_v_1_end_0 = const()[name = string("slot_v_1_end_0"), val = tensor([1, 2, 512, 512])]; tensor slot_v_1_end_mask_0 = const()[name = string("slot_v_1_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_1_cast_fp16 = slice_by_index(begin = slot_v_1_begin_0, end = slot_v_1_end_0, end_mask = slot_v_1_end_mask_0, x = V_sliding_in)[name = string("slot_v_1_cast_fp16")]; tensor var_990_begin_0 = const()[name = string("op_990_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_990_end_0 = const()[name = string("op_990_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_990_end_mask_0 = const()[name = string("op_990_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_990_cast_fp16 = slice_by_index(begin = var_990_begin_0, end = var_990_end_0, end_mask = var_990_end_mask_0, x = slot_k_1_cast_fp16)[name = string("op_990_cast_fp16")]; int32 var_997 = const()[name = string("op_997"), val = int32(2)]; bool new_k_1_interleave_0 = const()[name = string("new_k_1_interleave_0"), val = bool(false)]; tensor new_k_1_cast_fp16 = concat(axis = var_997, interleave = new_k_1_interleave_0, values = (var_990_cast_fp16, k_padded_1_cast_fp16))[name = string("new_k_1_cast_fp16")]; tensor var_1013_begin_0 = const()[name = string("op_1013_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_1013_end_0 = const()[name = string("op_1013_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1013_end_mask_0 = const()[name = string("op_1013_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1013_cast_fp16 = slice_by_index(begin = var_1013_begin_0, end = var_1013_end_0, end_mask = var_1013_end_mask_0, x = slot_v_1_cast_fp16)[name = string("op_1013_cast_fp16")]; int32 var_1020 = const()[name = string("op_1020"), val = int32(2)]; bool new_v_1_interleave_0 = const()[name = string("new_v_1_interleave_0"), val = bool(false)]; tensor new_v_1_cast_fp16 = concat(axis = var_1020, interleave = new_v_1_interleave_0, values = (var_1013_cast_fp16, v_padded_1_cast_fp16))[name = string("new_v_1_cast_fp16")]; tensor var_1031_begin_0 = const()[name = string("op_1031_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1031_end_0 = const()[name = string("op_1031_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_1031_end_mask_0 = const()[name = string("op_1031_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1031_cast_fp16 = slice_by_index(begin = var_1031_begin_0, end = var_1031_end_0, end_mask = var_1031_end_mask_0, x = K_sliding_in)[name = string("op_1031_cast_fp16")]; int32 var_1033 = const()[name = string("op_1033"), val = int32(0)]; bool K_sliding_out_1_interleave_0 = const()[name = string("K_sliding_out_1_interleave_0"), val = bool(false)]; tensor K_sliding_out_1_cast_fp16 = concat(axis = var_1033, interleave = K_sliding_out_1_interleave_0, values = (new_k_1_cast_fp16, var_1031_cast_fp16))[name = string("K_sliding_out_1_cast_fp16")]; tensor var_1044_begin_0 = const()[name = string("op_1044_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1044_end_0 = const()[name = string("op_1044_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_1044_end_mask_0 = const()[name = string("op_1044_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1044_cast_fp16 = slice_by_index(begin = var_1044_begin_0, end = var_1044_end_0, end_mask = var_1044_end_mask_0, x = V_sliding_in)[name = string("op_1044_cast_fp16")]; int32 var_1046 = const()[name = string("op_1046"), val = int32(0)]; bool V_sliding_out_1_interleave_0 = const()[name = string("V_sliding_out_1_interleave_0"), val = bool(false)]; tensor V_sliding_out_1_cast_fp16 = concat(axis = var_1046, interleave = V_sliding_out_1_interleave_0, values = (new_v_1_cast_fp16, var_1044_cast_fp16))[name = string("V_sliding_out_1_cast_fp16")]; tensor var_1052_begin_0 = const()[name = string("op_1052_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1052_end_0 = const()[name = string("op_1052_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1052_end_mask_0 = const()[name = string("op_1052_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1052_cast_fp16 = slice_by_index(begin = var_1052_begin_0, end = var_1052_end_0, end_mask = var_1052_end_mask_0, x = K_sliding_out_1_cast_fp16)[name = string("op_1052_cast_fp16")]; tensor K_for_attn_1_begin_0 = const()[name = string("K_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_1_end_0 = const()[name = string("K_for_attn_1_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_1_end_mask_0 = const()[name = string("K_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_1_cast_fp16 = slice_by_index(begin = K_for_attn_1_begin_0, end = K_for_attn_1_end_0, end_mask = K_for_attn_1_end_mask_0, x = var_1052_cast_fp16)[name = string("K_for_attn_1_cast_fp16")]; tensor var_1062_begin_0 = const()[name = string("op_1062_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1062_end_0 = const()[name = string("op_1062_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1062_end_mask_0 = const()[name = string("op_1062_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1062_cast_fp16 = slice_by_index(begin = var_1062_begin_0, end = var_1062_end_0, end_mask = var_1062_end_mask_0, x = V_sliding_out_1_cast_fp16)[name = string("op_1062_cast_fp16")]; tensor V_for_attn_1_begin_0 = const()[name = string("V_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_1_end_0 = const()[name = string("V_for_attn_1_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_1_end_mask_0 = const()[name = string("V_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_1_cast_fp16 = slice_by_index(begin = V_for_attn_1_begin_0, end = V_for_attn_1_end_0, end_mask = V_for_attn_1_end_mask_0, x = var_1062_cast_fp16)[name = string("V_for_attn_1_cast_fp16")]; tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = K_for_attn_1_cast_fp16)[name = string("transpose_233")]; tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; tensor concat_0 = const()[name = string("concat_0"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_0, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_1 = const()[name = string("concat_1"), val = tensor([-1, 1, 512, 256])]; tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_232")]; tensor reshape_1_cast_fp16 = reshape(shape = concat_1, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; tensor transpose_50_perm_0 = const()[name = string("transpose_50_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = V_for_attn_1_cast_fp16)[name = string("transpose_231")]; tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; tensor concat_2 = const()[name = string("concat_2"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_2, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_3 = const()[name = string("concat_3"), val = tensor([-1, 1, 512, 256])]; tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_230")]; tensor reshape_3_cast_fp16 = reshape(shape = concat_3, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor V_expanded_1_perm_0 = const()[name = string("V_expanded_1_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor transpose_50_cast_fp16 = transpose(perm = transpose_50_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_229")]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = q_11_cast_fp16, y = transpose_50_cast_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor x_7_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask_sliding)[name = string("x_7_cast_fp16")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = x_7_cast_fp16)[name = string("reduce_max_0")]; tensor var_1097 = sub(x = x_7_cast_fp16, y = reduce_max_0)[name = string("op_1097")]; tensor var_1103 = exp(x = var_1097)[name = string("op_1103")]; tensor var_1113_axes_0 = const()[name = string("op_1113_axes_0"), val = tensor([-1])]; bool var_1113_keep_dims_0 = const()[name = string("op_1113_keep_dims_0"), val = bool(true)]; tensor var_1113 = reduce_sum(axes = var_1113_axes_0, keep_dims = var_1113_keep_dims_0, x = var_1103)[name = string("op_1113")]; tensor var_1119_cast_fp16 = real_div(x = var_1103, y = var_1113)[name = string("op_1119_cast_fp16")]; bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; tensor V_expanded_1_cast_fp16 = transpose(perm = V_expanded_1_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_228")]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = var_1119_cast_fp16, y = V_expanded_1_cast_fp16)[name = string("attn_output_1_cast_fp16")]; tensor var_1130 = const()[name = string("op_1130"), val = tensor([0, 2, 1, 3])]; tensor var_1137 = const()[name = string("op_1137"), val = tensor([1, 3, -1])]; tensor var_1131_cast_fp16 = transpose(perm = var_1130, x = attn_output_1_cast_fp16)[name = string("transpose_227")]; tensor attn_output_3_cast_fp16 = reshape(shape = var_1137, x = var_1131_cast_fp16)[name = string("attn_output_3_cast_fp16")]; tensor var_1142 = const()[name = string("op_1142"), val = tensor([0, 2, 1])]; string var_1158_pad_type_0 = const()[name = string("op_1158_pad_type_0"), val = string("valid")]; int32 var_1158_groups_0 = const()[name = string("op_1158_groups_0"), val = int32(1)]; tensor var_1158_strides_0 = const()[name = string("op_1158_strides_0"), val = tensor([1])]; tensor var_1158_pad_0 = const()[name = string("op_1158_pad_0"), val = tensor([0, 0])]; tensor var_1158_dilations_0 = const()[name = string("op_1158_dilations_0"), val = tensor([1])]; tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531256512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533878016))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_1143_cast_fp16 = transpose(perm = var_1142, x = attn_output_3_cast_fp16)[name = string("transpose_226")]; tensor var_1158_cast_fp16 = conv(dilations = var_1158_dilations_0, groups = var_1158_groups_0, pad = var_1158_pad_0, pad_type = var_1158_pad_type_0, strides = var_1158_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1143_cast_fp16)[name = string("op_1158_cast_fp16")]; tensor var_1162 = const()[name = string("op_1162"), val = tensor([0, 2, 1])]; int32 var_1168 = const()[name = string("op_1168"), val = int32(-1)]; fp16 const_7_promoted_to_fp16 = const()[name = string("const_7_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_11_cast_fp16 = transpose(perm = var_1162, x = var_1158_cast_fp16)[name = string("transpose_225")]; tensor var_1170_cast_fp16 = mul(x = x_11_cast_fp16, y = const_7_promoted_to_fp16)[name = string("op_1170_cast_fp16")]; bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; tensor input_15_cast_fp16 = concat(axis = var_1168, interleave = input_15_interleave_0, values = (x_11_cast_fp16, var_1170_cast_fp16))[name = string("input_15_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_1165_to_fp16 = const()[name = string("op_1165_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1165_to_fp16, x = input_15_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor var_1175_split_sizes_0 = const()[name = string("op_1175_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1175_axis_0 = const()[name = string("op_1175_axis_0"), val = int32(-1)]; tensor var_1175_cast_fp16_0, tensor var_1175_cast_fp16_1 = split(axis = var_1175_axis_0, split_sizes = var_1175_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_1175_cast_fp16")]; tensor layers_0_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533880640)))]; tensor attn_output_5_cast_fp16 = mul(x = var_1175_cast_fp16_0, y = layers_0_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_5_cast_fp16")]; tensor x_13_cast_fp16 = add(x = hidden_states, y = attn_output_5_cast_fp16)[name = string("x_13_cast_fp16")]; int32 var_1184 = const()[name = string("op_1184"), val = int32(-1)]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1186_cast_fp16 = mul(x = x_13_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1186_cast_fp16")]; bool input_17_interleave_0 = const()[name = string("input_17_interleave_0"), val = bool(false)]; tensor input_17_cast_fp16 = concat(axis = var_1184, interleave = input_17_interleave_0, values = (x_13_cast_fp16, var_1186_cast_fp16))[name = string("input_17_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_1181_to_fp16 = const()[name = string("op_1181_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1181_to_fp16, x = input_17_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_1191_split_sizes_0 = const()[name = string("op_1191_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1191_axis_0 = const()[name = string("op_1191_axis_0"), val = int32(-1)]; tensor var_1191_cast_fp16_0, tensor var_1191_cast_fp16_1 = split(axis = var_1191_axis_0, split_sizes = var_1191_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_1191_cast_fp16")]; tensor layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533885824)))]; tensor h_3_cast_fp16 = mul(x = var_1191_cast_fp16_0, y = layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_3_cast_fp16")]; tensor var_1202 = const()[name = string("op_1202"), val = tensor([0, 2, 1])]; tensor input_19_axes_0 = const()[name = string("input_19_axes_0"), val = tensor([2])]; tensor var_1203 = transpose(perm = var_1202, x = h_3_cast_fp16)[name = string("transpose_224")]; tensor input_19 = expand_dims(axes = input_19_axes_0, x = var_1203)[name = string("input_19")]; string gate_1_pad_type_0 = const()[name = string("gate_1_pad_type_0"), val = string("valid")]; tensor gate_1_strides_0 = const()[name = string("gate_1_strides_0"), val = tensor([1, 1])]; tensor gate_1_pad_0 = const()[name = string("gate_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_1_dilations_0 = const()[name = string("gate_1_dilations_0"), val = tensor([1, 1])]; int32 gate_1_groups_0 = const()[name = string("gate_1_groups_0"), val = int32(1)]; tensor gate_1 = conv(dilations = gate_1_dilations_0, groups = gate_1_groups_0, pad = gate_1_pad_0, pad_type = gate_1_pad_type_0, strides = gate_1_strides_0, weight = layers_0_mlp_gate_proj_weight_palettized, x = input_19)[name = string("gate_1")]; string up_1_pad_type_0 = const()[name = string("up_1_pad_type_0"), val = string("valid")]; tensor up_1_strides_0 = const()[name = string("up_1_strides_0"), val = tensor([1, 1])]; tensor up_1_pad_0 = const()[name = string("up_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_1_dilations_0 = const()[name = string("up_1_dilations_0"), val = tensor([1, 1])]; int32 up_1_groups_0 = const()[name = string("up_1_groups_0"), val = int32(1)]; tensor up_1 = conv(dilations = up_1_dilations_0, groups = up_1_groups_0, pad = up_1_pad_0, pad_type = up_1_pad_type_0, strides = up_1_strides_0, weight = layers_0_mlp_up_proj_weight_palettized, x = input_19)[name = string("up_1")]; string gate_3_mode_0 = const()[name = string("gate_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_3 = gelu(mode = gate_3_mode_0, x = gate_1)[name = string("gate_3")]; tensor input_21 = mul(x = gate_3, y = up_1)[name = string("input_21")]; string mlp_out_1_pad_type_0 = const()[name = string("mlp_out_1_pad_type_0"), val = string("valid")]; tensor mlp_out_1_strides_0 = const()[name = string("mlp_out_1_strides_0"), val = tensor([1, 1])]; tensor mlp_out_1_pad_0 = const()[name = string("mlp_out_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_1_dilations_0 = const()[name = string("mlp_out_1_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_1_groups_0 = const()[name = string("mlp_out_1_groups_0"), val = int32(1)]; tensor mlp_out_1 = conv(dilations = mlp_out_1_dilations_0, groups = mlp_out_1_groups_0, pad = mlp_out_1_pad_0, pad_type = mlp_out_1_pad_type_0, strides = mlp_out_1_strides_0, weight = layers_0_mlp_down_proj_weight_palettized, x = input_21)[name = string("mlp_out_1")]; tensor var_1243_axes_0 = const()[name = string("op_1243_axes_0"), val = tensor([2])]; tensor var_1243 = squeeze(axes = var_1243_axes_0, x = mlp_out_1)[name = string("op_1243")]; tensor var_1247 = const()[name = string("op_1247"), val = tensor([0, 2, 1])]; int32 var_1253 = const()[name = string("op_1253"), val = int32(-1)]; fp16 const_9_promoted = const()[name = string("const_9_promoted"), val = fp16(-0x1p+0)]; tensor x_15 = transpose(perm = var_1247, x = var_1243)[name = string("transpose_223")]; tensor var_1255 = mul(x = x_15, y = const_9_promoted)[name = string("op_1255")]; bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; tensor input_23 = concat(axis = var_1253, interleave = input_23_interleave_0, values = (x_15, var_1255))[name = string("input_23")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_1250_to_fp16 = const()[name = string("op_1250_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1250_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; tensor var_1260_split_sizes_0 = const()[name = string("op_1260_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1260_axis_0 = const()[name = string("op_1260_axis_0"), val = int32(-1)]; tensor var_1260_0, tensor var_1260_1 = split(axis = var_1260_axis_0, split_sizes = var_1260_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_1260")]; tensor hidden_states_3 = mul(x = var_1260_0, y = layers_0_post_feedforward_layernorm_weight)[name = string("hidden_states_3")]; tensor hidden_states_5_cast_fp16 = add(x = x_13_cast_fp16, y = hidden_states_3)[name = string("hidden_states_5_cast_fp16")]; tensor per_layer_slice_1_begin_0 = const()[name = string("per_layer_slice_1_begin_0"), val = tensor([0, 0, 3072])]; tensor per_layer_slice_1_end_0 = const()[name = string("per_layer_slice_1_end_0"), val = tensor([1, 3, 3328])]; tensor per_layer_slice_1_end_mask_0 = const()[name = string("per_layer_slice_1_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_1_cast_fp16 = slice_by_index(begin = per_layer_slice_1_begin_0, end = per_layer_slice_1_end_0, end_mask = per_layer_slice_1_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_1_cast_fp16")]; tensor var_1288 = const()[name = string("op_1288"), val = tensor([0, 2, 1])]; tensor input_25_axes_0 = const()[name = string("input_25_axes_0"), val = tensor([2])]; tensor var_1289 = transpose(perm = var_1288, x = hidden_states_5_cast_fp16)[name = string("transpose_222")]; tensor input_25 = expand_dims(axes = input_25_axes_0, x = var_1289)[name = string("input_25")]; string gated_1_pad_type_0 = const()[name = string("gated_1_pad_type_0"), val = string("valid")]; tensor gated_1_strides_0 = const()[name = string("gated_1_strides_0"), val = tensor([1, 1])]; tensor gated_1_pad_0 = const()[name = string("gated_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_1_dilations_0 = const()[name = string("gated_1_dilations_0"), val = tensor([1, 1])]; int32 gated_1_groups_0 = const()[name = string("gated_1_groups_0"), val = int32(1)]; tensor gated_1 = conv(dilations = gated_1_dilations_0, groups = gated_1_groups_0, pad = gated_1_pad_0, pad_type = gated_1_pad_type_0, strides = gated_1_strides_0, weight = layers_0_per_layer_input_gate_weight_palettized, x = input_25)[name = string("gated_1")]; string gated_3_mode_0 = const()[name = string("gated_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_3 = gelu(mode = gated_3_mode_0, x = gated_1)[name = string("gated_3")]; tensor var_1308 = const()[name = string("op_1308"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_1_axes_0 = const()[name = string("per_layer_slice_conv_1_axes_0"), val = tensor([2])]; tensor var_1309_cast_fp16 = transpose(perm = var_1308, x = per_layer_slice_1_cast_fp16)[name = string("transpose_221")]; tensor per_layer_slice_conv_1_cast_fp16 = expand_dims(axes = per_layer_slice_conv_1_axes_0, x = var_1309_cast_fp16)[name = string("per_layer_slice_conv_1_cast_fp16")]; tensor input_27_cast_fp16 = mul(x = gated_3, y = per_layer_slice_conv_1_cast_fp16)[name = string("input_27_cast_fp16")]; string gated_5_pad_type_0 = const()[name = string("gated_5_pad_type_0"), val = string("valid")]; tensor gated_5_strides_0 = const()[name = string("gated_5_strides_0"), val = tensor([1, 1])]; tensor gated_5_pad_0 = const()[name = string("gated_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_5_dilations_0 = const()[name = string("gated_5_dilations_0"), val = tensor([1, 1])]; int32 gated_5_groups_0 = const()[name = string("gated_5_groups_0"), val = int32(1)]; tensor layers_0_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(533891008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534218752))))[name = string("layers_0_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_5_cast_fp16 = conv(dilations = gated_5_dilations_0, groups = gated_5_groups_0, pad = gated_5_pad_0, pad_type = gated_5_pad_type_0, strides = gated_5_strides_0, weight = layers_0_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_27_cast_fp16)[name = string("gated_5_cast_fp16")]; tensor var_1325_axes_0 = const()[name = string("op_1325_axes_0"), val = tensor([2])]; tensor var_1325_cast_fp16 = squeeze(axes = var_1325_axes_0, x = gated_5_cast_fp16)[name = string("op_1325_cast_fp16")]; tensor var_1329 = const()[name = string("op_1329"), val = tensor([0, 2, 1])]; int32 var_1335 = const()[name = string("op_1335"), val = int32(-1)]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_17_cast_fp16 = transpose(perm = var_1329, x = var_1325_cast_fp16)[name = string("transpose_220")]; tensor var_1337_cast_fp16 = mul(x = x_17_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1337_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_1335, interleave = input_29_interleave_0, values = (x_17_cast_fp16, var_1337_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_1332_to_fp16 = const()[name = string("op_1332_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1332_to_fp16, x = input_29_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor var_1342_split_sizes_0 = const()[name = string("op_1342_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1342_axis_0 = const()[name = string("op_1342_axis_0"), val = int32(-1)]; tensor var_1342_cast_fp16_0, tensor var_1342_cast_fp16_1 = split(axis = var_1342_axis_0, split_sizes = var_1342_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_1342_cast_fp16")]; tensor layers_0_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534221376)))]; tensor hidden_states_9_cast_fp16 = mul(x = var_1342_cast_fp16_0, y = layers_0_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_9_cast_fp16")]; tensor hidden_states_11_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_9_cast_fp16)[name = string("hidden_states_11_cast_fp16")]; tensor const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = tensor([0x1.7ep-1])]; tensor x_19_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_11_promoted_to_fp16)[name = string("x_19_cast_fp16")]; int32 var_1357 = const()[name = string("op_1357"), val = int32(-1)]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1359_cast_fp16 = mul(x = x_19_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_1359_cast_fp16")]; bool input_31_interleave_0 = const()[name = string("input_31_interleave_0"), val = bool(false)]; tensor input_31_cast_fp16 = concat(axis = var_1357, interleave = input_31_interleave_0, values = (x_19_cast_fp16, var_1359_cast_fp16))[name = string("input_31_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_1354_to_fp16 = const()[name = string("op_1354_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1354_to_fp16, x = input_31_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_1364_split_sizes_0 = const()[name = string("op_1364_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1364_axis_0 = const()[name = string("op_1364_axis_0"), val = int32(-1)]; tensor var_1364_cast_fp16_0, tensor var_1364_cast_fp16_1 = split(axis = var_1364_axis_0, split_sizes = var_1364_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_1364_cast_fp16")]; tensor layers_1_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534226560)))]; tensor h_7_cast_fp16 = mul(x = var_1364_cast_fp16_0, y = layers_1_input_layernorm_weight_promoted_to_fp16)[name = string("h_7_cast_fp16")]; tensor var_1370 = const()[name = string("op_1370"), val = tensor([0, 2, 1])]; tensor var_1373_axes_0 = const()[name = string("op_1373_axes_0"), val = tensor([2])]; tensor var_1371_cast_fp16 = transpose(perm = var_1370, x = h_7_cast_fp16)[name = string("transpose_219")]; tensor var_1373_cast_fp16 = expand_dims(axes = var_1373_axes_0, x = var_1371_cast_fp16)[name = string("op_1373_cast_fp16")]; string q_13_pad_type_0 = const()[name = string("q_13_pad_type_0"), val = string("valid")]; tensor q_13_strides_0 = const()[name = string("q_13_strides_0"), val = tensor([1, 1])]; tensor q_13_pad_0 = const()[name = string("q_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_13_dilations_0 = const()[name = string("q_13_dilations_0"), val = tensor([1, 1])]; int32 q_13_groups_0 = const()[name = string("q_13_groups_0"), val = int32(1)]; tensor q_13 = conv(dilations = q_13_dilations_0, groups = q_13_groups_0, pad = q_13_pad_0, pad_type = q_13_pad_type_0, strides = q_13_strides_0, weight = layers_1_self_attn_q_proj_weight_palettized, x = var_1373_cast_fp16)[name = string("q_13")]; tensor var_1394 = const()[name = string("op_1394"), val = tensor([1, 8, 256, 3])]; tensor var_1395 = reshape(shape = var_1394, x = q_13)[name = string("op_1395")]; tensor transpose_51_perm_0 = const()[name = string("transpose_51_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_1418 = const()[name = string("op_1418"), val = tensor([3, 8, 256])]; tensor transpose_51 = transpose(perm = transpose_51_perm_0, x = var_1395)[name = string("transpose_218")]; tensor x_21 = reshape(shape = var_1418, x = transpose_51)[name = string("x_21")]; int32 var_1424 = const()[name = string("op_1424"), val = int32(-1)]; fp16 const_13_promoted = const()[name = string("const_13_promoted"), val = fp16(-0x1p+0)]; tensor var_1426 = mul(x = x_21, y = const_13_promoted)[name = string("op_1426")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35 = concat(axis = var_1424, interleave = input_35_interleave_0, values = (x_21, var_1426))[name = string("input_35")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_1421_to_fp16 = const()[name = string("op_1421_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1421_to_fp16, x = input_35)[name = string("normed_33_cast_fp16")]; tensor var_1431_split_sizes_0 = const()[name = string("op_1431_split_sizes_0"), val = tensor([256, 256])]; int32 var_1431_axis_0 = const()[name = string("op_1431_axis_0"), val = int32(-1)]; tensor var_1431_0, tensor var_1431_1 = split(axis = var_1431_axis_0, split_sizes = var_1431_split_sizes_0, x = normed_33_cast_fp16)[name = string("op_1431")]; tensor q_17 = mul(x = var_1431_0, y = layers_1_self_attn_q_norm_weight)[name = string("q_17")]; tensor var_1438 = const()[name = string("op_1438"), val = tensor([1, 3, 8, 256])]; tensor var_1439 = reshape(shape = var_1438, x = q_17)[name = string("op_1439")]; tensor var_1444 = const()[name = string("op_1444"), val = tensor([0, 2, 1, 3])]; tensor q_19 = transpose(perm = var_1444, x = var_1439)[name = string("transpose_217")]; tensor var_1446_cast_fp16 = mul(x = q_19, y = cos_s)[name = string("op_1446_cast_fp16")]; tensor var_1447_split_sizes_0 = const()[name = string("op_1447_split_sizes_0"), val = tensor([128, 128])]; int32 var_1447_axis_0 = const()[name = string("op_1447_axis_0"), val = int32(-1)]; tensor var_1447_0, tensor var_1447_1 = split(axis = var_1447_axis_0, split_sizes = var_1447_split_sizes_0, x = q_19)[name = string("op_1447")]; fp16 const_14_promoted = const()[name = string("const_14_promoted"), val = fp16(-0x1p+0)]; tensor var_1449 = mul(x = var_1447_1, y = const_14_promoted)[name = string("op_1449")]; int32 var_1451 = const()[name = string("op_1451"), val = int32(-1)]; bool var_1452_interleave_0 = const()[name = string("op_1452_interleave_0"), val = bool(false)]; tensor var_1452 = concat(axis = var_1451, interleave = var_1452_interleave_0, values = (var_1449, var_1447_0))[name = string("op_1452")]; tensor var_1453_cast_fp16 = mul(x = var_1452, y = sin_s)[name = string("op_1453_cast_fp16")]; tensor q_23_cast_fp16 = add(x = var_1446_cast_fp16, y = var_1453_cast_fp16)[name = string("q_23_cast_fp16")]; string k_7_pad_type_0 = const()[name = string("k_7_pad_type_0"), val = string("valid")]; tensor k_7_strides_0 = const()[name = string("k_7_strides_0"), val = tensor([1, 1])]; tensor k_7_pad_0 = const()[name = string("k_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_7_dilations_0 = const()[name = string("k_7_dilations_0"), val = tensor([1, 1])]; int32 k_7_groups_0 = const()[name = string("k_7_groups_0"), val = int32(1)]; tensor k_7 = conv(dilations = k_7_dilations_0, groups = k_7_groups_0, pad = k_7_pad_0, pad_type = k_7_pad_type_0, strides = k_7_strides_0, weight = layers_1_self_attn_k_proj_weight_palettized, x = var_1373_cast_fp16)[name = string("k_7")]; tensor var_1471 = const()[name = string("op_1471"), val = tensor([1, 2, 256, 3])]; tensor var_1472 = reshape(shape = var_1471, x = k_7)[name = string("op_1472")]; tensor transpose_52_perm_0 = const()[name = string("transpose_52_perm_0"), val = tensor([0, 3, 1, 2])]; string v_3_pad_type_0 = const()[name = string("v_3_pad_type_0"), val = string("valid")]; tensor v_3_strides_0 = const()[name = string("v_3_strides_0"), val = tensor([1, 1])]; tensor v_3_pad_0 = const()[name = string("v_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_3_dilations_0 = const()[name = string("v_3_dilations_0"), val = tensor([1, 1])]; int32 v_3_groups_0 = const()[name = string("v_3_groups_0"), val = int32(1)]; tensor v_3 = conv(dilations = v_3_dilations_0, groups = v_3_groups_0, pad = v_3_pad_0, pad_type = v_3_pad_type_0, strides = v_3_strides_0, weight = layers_1_self_attn_v_proj_weight_palettized, x = var_1373_cast_fp16)[name = string("v_3")]; tensor var_1499 = const()[name = string("op_1499"), val = tensor([1, 2, 256, 3])]; tensor var_1500 = reshape(shape = var_1499, x = v_3)[name = string("op_1500")]; tensor var_1505 = const()[name = string("op_1505"), val = tensor([0, 1, 3, 2])]; tensor var_1523 = const()[name = string("op_1523"), val = tensor([3, 2, 256])]; tensor transpose_52 = transpose(perm = transpose_52_perm_0, x = var_1472)[name = string("transpose_216")]; tensor x_23 = reshape(shape = var_1523, x = transpose_52)[name = string("x_23")]; int32 var_1529 = const()[name = string("op_1529"), val = int32(-1)]; fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; tensor var_1531 = mul(x = x_23, y = const_15_promoted)[name = string("op_1531")]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37 = concat(axis = var_1529, interleave = input_37_interleave_0, values = (x_23, var_1531))[name = string("input_37")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_1526_to_fp16 = const()[name = string("op_1526_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1526_to_fp16, x = input_37)[name = string("normed_37_cast_fp16")]; tensor var_1536_split_sizes_0 = const()[name = string("op_1536_split_sizes_0"), val = tensor([256, 256])]; int32 var_1536_axis_0 = const()[name = string("op_1536_axis_0"), val = int32(-1)]; tensor var_1536_0, tensor var_1536_1 = split(axis = var_1536_axis_0, split_sizes = var_1536_split_sizes_0, x = normed_37_cast_fp16)[name = string("op_1536")]; tensor k_11 = mul(x = var_1536_0, y = layers_1_self_attn_k_norm_weight)[name = string("k_11")]; tensor var_1543 = const()[name = string("op_1543"), val = tensor([1, 3, 2, 256])]; tensor var_1544 = reshape(shape = var_1543, x = k_11)[name = string("op_1544")]; tensor var_1549 = const()[name = string("op_1549"), val = tensor([0, 2, 1, 3])]; fp16 var_1551_promoted = const()[name = string("op_1551_promoted"), val = fp16(0x1p+1)]; tensor var_1506 = transpose(perm = var_1505, x = var_1500)[name = string("transpose_215")]; tensor var_1552 = pow(x = var_1506, y = var_1551_promoted)[name = string("op_1552")]; tensor var_1557_axes_0 = const()[name = string("op_1557_axes_0"), val = tensor([-1])]; bool var_1557_keep_dims_0 = const()[name = string("op_1557_keep_dims_0"), val = bool(true)]; tensor var_1557 = reduce_mean(axes = var_1557_axes_0, keep_dims = var_1557_keep_dims_0, x = var_1552)[name = string("op_1557")]; fp16 var_1559_to_fp16 = const()[name = string("op_1559_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_3_cast_fp16 = add(x = var_1557, y = var_1559_to_fp16)[name = string("mean_sq_3_cast_fp16")]; fp32 var_1561_epsilon_0 = const()[name = string("op_1561_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_1561_cast_fp16 = rsqrt(epsilon = var_1561_epsilon_0, x = mean_sq_3_cast_fp16)[name = string("op_1561_cast_fp16")]; tensor input_41_cast_fp16 = mul(x = var_1506, y = var_1561_cast_fp16)[name = string("input_41_cast_fp16")]; tensor q_21 = transpose(perm = var_1549, x = var_1544)[name = string("transpose_214")]; tensor var_1563_cast_fp16 = mul(x = q_21, y = cos_s)[name = string("op_1563_cast_fp16")]; tensor var_1564_split_sizes_0 = const()[name = string("op_1564_split_sizes_0"), val = tensor([128, 128])]; int32 var_1564_axis_0 = const()[name = string("op_1564_axis_0"), val = int32(-1)]; tensor var_1564_0, tensor var_1564_1 = split(axis = var_1564_axis_0, split_sizes = var_1564_split_sizes_0, x = q_21)[name = string("op_1564")]; fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; tensor var_1566 = mul(x = var_1564_1, y = const_16_promoted)[name = string("op_1566")]; int32 var_1568 = const()[name = string("op_1568"), val = int32(-1)]; bool var_1569_interleave_0 = const()[name = string("op_1569_interleave_0"), val = bool(false)]; tensor var_1569 = concat(axis = var_1568, interleave = var_1569_interleave_0, values = (var_1566, var_1564_0))[name = string("op_1569")]; tensor var_1570_cast_fp16 = mul(x = var_1569, y = sin_s)[name = string("op_1570_cast_fp16")]; tensor input_39_cast_fp16 = add(x = var_1563_cast_fp16, y = var_1570_cast_fp16)[name = string("input_39_cast_fp16")]; tensor k_padded_3_pad_0 = const()[name = string("k_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_3_mode_0 = const()[name = string("k_padded_3_mode_0"), val = string("constant")]; fp16 const_17_to_fp16 = const()[name = string("const_17_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_3_cast_fp16 = pad(constant_val = const_17_to_fp16, mode = k_padded_3_mode_0, pad = k_padded_3_pad_0, x = input_39_cast_fp16)[name = string("k_padded_3_cast_fp16")]; tensor v_padded_3_pad_0 = const()[name = string("v_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_3_mode_0 = const()[name = string("v_padded_3_mode_0"), val = string("constant")]; fp16 const_18_to_fp16 = const()[name = string("const_18_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_3_cast_fp16 = pad(constant_val = const_18_to_fp16, mode = v_padded_3_mode_0, pad = v_padded_3_pad_0, x = input_41_cast_fp16)[name = string("v_padded_3_cast_fp16")]; tensor slot_k_3_begin_0 = const()[name = string("slot_k_3_begin_0"), val = tensor([1, 0, 0, 0])]; tensor slot_k_3_end_0 = const()[name = string("slot_k_3_end_0"), val = tensor([2, 2, 512, 512])]; tensor slot_k_3_end_mask_0 = const()[name = string("slot_k_3_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_3_cast_fp16 = slice_by_index(begin = slot_k_3_begin_0, end = slot_k_3_end_0, end_mask = slot_k_3_end_mask_0, x = K_sliding_out_1_cast_fp16)[name = string("slot_k_3_cast_fp16")]; tensor slot_v_3_begin_0 = const()[name = string("slot_v_3_begin_0"), val = tensor([1, 0, 0, 0])]; tensor slot_v_3_end_0 = const()[name = string("slot_v_3_end_0"), val = tensor([2, 2, 512, 512])]; tensor slot_v_3_end_mask_0 = const()[name = string("slot_v_3_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_3_cast_fp16 = slice_by_index(begin = slot_v_3_begin_0, end = slot_v_3_end_0, end_mask = slot_v_3_end_mask_0, x = V_sliding_out_1_cast_fp16)[name = string("slot_v_3_cast_fp16")]; tensor var_1609_begin_0 = const()[name = string("op_1609_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_1609_end_0 = const()[name = string("op_1609_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1609_end_mask_0 = const()[name = string("op_1609_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1609_cast_fp16 = slice_by_index(begin = var_1609_begin_0, end = var_1609_end_0, end_mask = var_1609_end_mask_0, x = slot_k_3_cast_fp16)[name = string("op_1609_cast_fp16")]; int32 var_1616 = const()[name = string("op_1616"), val = int32(2)]; bool new_k_3_interleave_0 = const()[name = string("new_k_3_interleave_0"), val = bool(false)]; tensor new_k_3_cast_fp16 = concat(axis = var_1616, interleave = new_k_3_interleave_0, values = (var_1609_cast_fp16, k_padded_3_cast_fp16))[name = string("new_k_3_cast_fp16")]; tensor var_1632_begin_0 = const()[name = string("op_1632_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_1632_end_0 = const()[name = string("op_1632_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_1632_end_mask_0 = const()[name = string("op_1632_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1632_cast_fp16 = slice_by_index(begin = var_1632_begin_0, end = var_1632_end_0, end_mask = var_1632_end_mask_0, x = slot_v_3_cast_fp16)[name = string("op_1632_cast_fp16")]; int32 var_1639 = const()[name = string("op_1639"), val = int32(2)]; bool new_v_3_interleave_0 = const()[name = string("new_v_3_interleave_0"), val = bool(false)]; tensor new_v_3_cast_fp16 = concat(axis = var_1639, interleave = new_v_3_interleave_0, values = (var_1632_cast_fp16, v_padded_3_cast_fp16))[name = string("new_v_3_cast_fp16")]; tensor var_1650_begin_0 = const()[name = string("op_1650_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1650_end_0 = const()[name = string("op_1650_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_1650_end_mask_0 = const()[name = string("op_1650_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1650_cast_fp16 = slice_by_index(begin = var_1650_begin_0, end = var_1650_end_0, end_mask = var_1650_end_mask_0, x = K_sliding_out_1_cast_fp16)[name = string("op_1650_cast_fp16")]; int32 var_1652 = const()[name = string("op_1652"), val = int32(0)]; bool K_sliding_out_3_interleave_0 = const()[name = string("K_sliding_out_3_interleave_0"), val = bool(false)]; tensor K_sliding_out_3_cast_fp16 = concat(axis = var_1652, interleave = K_sliding_out_3_interleave_0, values = (var_1052_cast_fp16, new_k_3_cast_fp16, var_1650_cast_fp16))[name = string("K_sliding_out_3_cast_fp16")]; tensor var_1663_begin_0 = const()[name = string("op_1663_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1663_end_0 = const()[name = string("op_1663_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_1663_end_mask_0 = const()[name = string("op_1663_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1663_cast_fp16 = slice_by_index(begin = var_1663_begin_0, end = var_1663_end_0, end_mask = var_1663_end_mask_0, x = V_sliding_out_1_cast_fp16)[name = string("op_1663_cast_fp16")]; int32 var_1665 = const()[name = string("op_1665"), val = int32(0)]; bool V_sliding_out_3_interleave_0 = const()[name = string("V_sliding_out_3_interleave_0"), val = bool(false)]; tensor V_sliding_out_3_cast_fp16 = concat(axis = var_1665, interleave = V_sliding_out_3_interleave_0, values = (var_1062_cast_fp16, new_v_3_cast_fp16, var_1663_cast_fp16))[name = string("V_sliding_out_3_cast_fp16")]; tensor var_1671_begin_0 = const()[name = string("op_1671_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1671_end_0 = const()[name = string("op_1671_end_0"), val = tensor([2, 2, 512, 512])]; tensor var_1671_end_mask_0 = const()[name = string("op_1671_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1671_cast_fp16 = slice_by_index(begin = var_1671_begin_0, end = var_1671_end_0, end_mask = var_1671_end_mask_0, x = K_sliding_out_3_cast_fp16)[name = string("op_1671_cast_fp16")]; tensor K_for_attn_3_begin_0 = const()[name = string("K_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_3_end_0 = const()[name = string("K_for_attn_3_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_3_end_mask_0 = const()[name = string("K_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_3_cast_fp16 = slice_by_index(begin = K_for_attn_3_begin_0, end = K_for_attn_3_end_0, end_mask = K_for_attn_3_end_mask_0, x = var_1671_cast_fp16)[name = string("K_for_attn_3_cast_fp16")]; tensor var_1681_begin_0 = const()[name = string("op_1681_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1681_end_0 = const()[name = string("op_1681_end_0"), val = tensor([2, 2, 512, 512])]; tensor var_1681_end_mask_0 = const()[name = string("op_1681_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1681_cast_fp16 = slice_by_index(begin = var_1681_begin_0, end = var_1681_end_0, end_mask = var_1681_end_mask_0, x = V_sliding_out_3_cast_fp16)[name = string("op_1681_cast_fp16")]; tensor V_for_attn_3_begin_0 = const()[name = string("V_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_3_end_0 = const()[name = string("V_for_attn_3_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_3_end_mask_0 = const()[name = string("V_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_3_cast_fp16 = slice_by_index(begin = V_for_attn_3_begin_0, end = V_for_attn_3_end_0, end_mask = V_for_attn_3_end_mask_0, x = var_1681_cast_fp16)[name = string("V_for_attn_3_cast_fp16")]; tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = K_for_attn_3_cast_fp16)[name = string("transpose_213")]; tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; tensor concat_4 = const()[name = string("concat_4"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_4, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_5 = const()[name = string("concat_5"), val = tensor([-1, 1, 512, 256])]; tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_212")]; tensor reshape_5_cast_fp16 = reshape(shape = concat_5, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor transpose_53_perm_0 = const()[name = string("transpose_53_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = V_for_attn_3_cast_fp16)[name = string("transpose_211")]; tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; tensor concat_6 = const()[name = string("concat_6"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_6, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_7 = const()[name = string("concat_7"), val = tensor([-1, 1, 512, 256])]; tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_210")]; tensor reshape_7_cast_fp16 = reshape(shape = concat_7, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; tensor V_expanded_3_perm_0 = const()[name = string("V_expanded_3_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_5_transpose_x_0 = const()[name = string("attn_weights_5_transpose_x_0"), val = bool(false)]; bool attn_weights_5_transpose_y_0 = const()[name = string("attn_weights_5_transpose_y_0"), val = bool(false)]; tensor transpose_53_cast_fp16 = transpose(perm = transpose_53_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_209")]; tensor attn_weights_5_cast_fp16 = matmul(transpose_x = attn_weights_5_transpose_x_0, transpose_y = attn_weights_5_transpose_y_0, x = q_23_cast_fp16, y = transpose_53_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; tensor x_27_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask_sliding)[name = string("x_27_cast_fp16")]; tensor reduce_max_1_axes_0 = const()[name = string("reduce_max_1_axes_0"), val = tensor([-1])]; bool reduce_max_1_keep_dims_0 = const()[name = string("reduce_max_1_keep_dims_0"), val = bool(true)]; tensor reduce_max_1 = reduce_max(axes = reduce_max_1_axes_0, keep_dims = reduce_max_1_keep_dims_0, x = x_27_cast_fp16)[name = string("reduce_max_1")]; tensor var_1716 = sub(x = x_27_cast_fp16, y = reduce_max_1)[name = string("op_1716")]; tensor var_1722 = exp(x = var_1716)[name = string("op_1722")]; tensor var_1732_axes_0 = const()[name = string("op_1732_axes_0"), val = tensor([-1])]; bool var_1732_keep_dims_0 = const()[name = string("op_1732_keep_dims_0"), val = bool(true)]; tensor var_1732 = reduce_sum(axes = var_1732_axes_0, keep_dims = var_1732_keep_dims_0, x = var_1722)[name = string("op_1732")]; tensor var_1738_cast_fp16 = real_div(x = var_1722, y = var_1732)[name = string("op_1738_cast_fp16")]; bool attn_output_7_transpose_x_0 = const()[name = string("attn_output_7_transpose_x_0"), val = bool(false)]; bool attn_output_7_transpose_y_0 = const()[name = string("attn_output_7_transpose_y_0"), val = bool(false)]; tensor V_expanded_3_cast_fp16 = transpose(perm = V_expanded_3_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_208")]; tensor attn_output_7_cast_fp16 = matmul(transpose_x = attn_output_7_transpose_x_0, transpose_y = attn_output_7_transpose_y_0, x = var_1738_cast_fp16, y = V_expanded_3_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor var_1749 = const()[name = string("op_1749"), val = tensor([0, 2, 1, 3])]; tensor var_1756 = const()[name = string("op_1756"), val = tensor([1, 3, -1])]; tensor var_1750_cast_fp16 = transpose(perm = var_1749, x = attn_output_7_cast_fp16)[name = string("transpose_207")]; tensor attn_output_9_cast_fp16 = reshape(shape = var_1756, x = var_1750_cast_fp16)[name = string("attn_output_9_cast_fp16")]; tensor var_1761 = const()[name = string("op_1761"), val = tensor([0, 2, 1])]; string var_1777_pad_type_0 = const()[name = string("op_1777_pad_type_0"), val = string("valid")]; int32 var_1777_groups_0 = const()[name = string("op_1777_groups_0"), val = int32(1)]; tensor var_1777_strides_0 = const()[name = string("op_1777_strides_0"), val = tensor([1])]; tensor var_1777_pad_0 = const()[name = string("op_1777_pad_0"), val = tensor([0, 0])]; tensor var_1777_dilations_0 = const()[name = string("op_1777_dilations_0"), val = tensor([1])]; tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(534231744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536853248))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_1762_cast_fp16 = transpose(perm = var_1761, x = attn_output_9_cast_fp16)[name = string("transpose_206")]; tensor var_1777_cast_fp16 = conv(dilations = var_1777_dilations_0, groups = var_1777_groups_0, pad = var_1777_pad_0, pad_type = var_1777_pad_type_0, strides = var_1777_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_1762_cast_fp16)[name = string("op_1777_cast_fp16")]; tensor var_1781 = const()[name = string("op_1781"), val = tensor([0, 2, 1])]; int32 var_1787 = const()[name = string("op_1787"), val = int32(-1)]; fp16 const_19_promoted_to_fp16 = const()[name = string("const_19_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_31_cast_fp16 = transpose(perm = var_1781, x = var_1777_cast_fp16)[name = string("transpose_205")]; tensor var_1789_cast_fp16 = mul(x = x_31_cast_fp16, y = const_19_promoted_to_fp16)[name = string("op_1789_cast_fp16")]; bool input_45_interleave_0 = const()[name = string("input_45_interleave_0"), val = bool(false)]; tensor input_45_cast_fp16 = concat(axis = var_1787, interleave = input_45_interleave_0, values = (x_31_cast_fp16, var_1789_cast_fp16))[name = string("input_45_cast_fp16")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_1784_to_fp16 = const()[name = string("op_1784_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1784_to_fp16, x = input_45_cast_fp16)[name = string("normed_41_cast_fp16")]; tensor var_1794_split_sizes_0 = const()[name = string("op_1794_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1794_axis_0 = const()[name = string("op_1794_axis_0"), val = int32(-1)]; tensor var_1794_cast_fp16_0, tensor var_1794_cast_fp16_1 = split(axis = var_1794_axis_0, split_sizes = var_1794_split_sizes_0, x = normed_41_cast_fp16)[name = string("op_1794_cast_fp16")]; tensor layers_1_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536855872)))]; tensor attn_output_11_cast_fp16 = mul(x = var_1794_cast_fp16_0, y = layers_1_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_11_cast_fp16")]; tensor x_33_cast_fp16 = add(x = x_19_cast_fp16, y = attn_output_11_cast_fp16)[name = string("x_33_cast_fp16")]; int32 var_1803 = const()[name = string("op_1803"), val = int32(-1)]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1805_cast_fp16 = mul(x = x_33_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1805_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_1803, interleave = input_47_interleave_0, values = (x_33_cast_fp16, var_1805_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_1800_to_fp16 = const()[name = string("op_1800_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_1800_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor var_1810_split_sizes_0 = const()[name = string("op_1810_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1810_axis_0 = const()[name = string("op_1810_axis_0"), val = int32(-1)]; tensor var_1810_cast_fp16_0, tensor var_1810_cast_fp16_1 = split(axis = var_1810_axis_0, split_sizes = var_1810_split_sizes_0, x = normed_45_cast_fp16)[name = string("op_1810_cast_fp16")]; tensor layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536861056)))]; tensor h_9_cast_fp16 = mul(x = var_1810_cast_fp16_0, y = layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_9_cast_fp16")]; tensor var_1821 = const()[name = string("op_1821"), val = tensor([0, 2, 1])]; tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; tensor var_1822 = transpose(perm = var_1821, x = h_9_cast_fp16)[name = string("transpose_204")]; tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_1822)[name = string("input_49")]; string gate_5_pad_type_0 = const()[name = string("gate_5_pad_type_0"), val = string("valid")]; tensor gate_5_strides_0 = const()[name = string("gate_5_strides_0"), val = tensor([1, 1])]; tensor gate_5_pad_0 = const()[name = string("gate_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_5_dilations_0 = const()[name = string("gate_5_dilations_0"), val = tensor([1, 1])]; int32 gate_5_groups_0 = const()[name = string("gate_5_groups_0"), val = int32(1)]; tensor gate_5 = conv(dilations = gate_5_dilations_0, groups = gate_5_groups_0, pad = gate_5_pad_0, pad_type = gate_5_pad_type_0, strides = gate_5_strides_0, weight = layers_1_mlp_gate_proj_weight_palettized, x = input_49)[name = string("gate_5")]; string up_3_pad_type_0 = const()[name = string("up_3_pad_type_0"), val = string("valid")]; tensor up_3_strides_0 = const()[name = string("up_3_strides_0"), val = tensor([1, 1])]; tensor up_3_pad_0 = const()[name = string("up_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_3_dilations_0 = const()[name = string("up_3_dilations_0"), val = tensor([1, 1])]; int32 up_3_groups_0 = const()[name = string("up_3_groups_0"), val = int32(1)]; tensor up_3 = conv(dilations = up_3_dilations_0, groups = up_3_groups_0, pad = up_3_pad_0, pad_type = up_3_pad_type_0, strides = up_3_strides_0, weight = layers_1_mlp_up_proj_weight_palettized, x = input_49)[name = string("up_3")]; string gate_7_mode_0 = const()[name = string("gate_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_7 = gelu(mode = gate_7_mode_0, x = gate_5)[name = string("gate_7")]; tensor input_51 = mul(x = gate_7, y = up_3)[name = string("input_51")]; string mlp_out_3_pad_type_0 = const()[name = string("mlp_out_3_pad_type_0"), val = string("valid")]; tensor mlp_out_3_strides_0 = const()[name = string("mlp_out_3_strides_0"), val = tensor([1, 1])]; tensor mlp_out_3_pad_0 = const()[name = string("mlp_out_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_3_dilations_0 = const()[name = string("mlp_out_3_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_3_groups_0 = const()[name = string("mlp_out_3_groups_0"), val = int32(1)]; tensor mlp_out_3 = conv(dilations = mlp_out_3_dilations_0, groups = mlp_out_3_groups_0, pad = mlp_out_3_pad_0, pad_type = mlp_out_3_pad_type_0, strides = mlp_out_3_strides_0, weight = layers_1_mlp_down_proj_weight_palettized, x = input_51)[name = string("mlp_out_3")]; tensor var_1862_axes_0 = const()[name = string("op_1862_axes_0"), val = tensor([2])]; tensor var_1862 = squeeze(axes = var_1862_axes_0, x = mlp_out_3)[name = string("op_1862")]; tensor var_1866 = const()[name = string("op_1866"), val = tensor([0, 2, 1])]; int32 var_1872 = const()[name = string("op_1872"), val = int32(-1)]; fp16 const_21_promoted = const()[name = string("const_21_promoted"), val = fp16(-0x1p+0)]; tensor x_35 = transpose(perm = var_1866, x = var_1862)[name = string("transpose_203")]; tensor var_1874 = mul(x = x_35, y = const_21_promoted)[name = string("op_1874")]; bool input_53_interleave_0 = const()[name = string("input_53_interleave_0"), val = bool(false)]; tensor input_53 = concat(axis = var_1872, interleave = input_53_interleave_0, values = (x_35, var_1874))[name = string("input_53")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_1869_to_fp16 = const()[name = string("op_1869_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_1869_to_fp16, x = input_53)[name = string("normed_49_cast_fp16")]; tensor var_1879_split_sizes_0 = const()[name = string("op_1879_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1879_axis_0 = const()[name = string("op_1879_axis_0"), val = int32(-1)]; tensor var_1879_0, tensor var_1879_1 = split(axis = var_1879_axis_0, split_sizes = var_1879_split_sizes_0, x = normed_49_cast_fp16)[name = string("op_1879")]; tensor hidden_states_13 = mul(x = var_1879_0, y = layers_1_post_feedforward_layernorm_weight)[name = string("hidden_states_13")]; tensor hidden_states_15_cast_fp16 = add(x = x_33_cast_fp16, y = hidden_states_13)[name = string("hidden_states_15_cast_fp16")]; tensor per_layer_slice_3_begin_0 = const()[name = string("per_layer_slice_3_begin_0"), val = tensor([0, 0, 3328])]; tensor per_layer_slice_3_end_0 = const()[name = string("per_layer_slice_3_end_0"), val = tensor([1, 3, 3584])]; tensor per_layer_slice_3_end_mask_0 = const()[name = string("per_layer_slice_3_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_3_cast_fp16 = slice_by_index(begin = per_layer_slice_3_begin_0, end = per_layer_slice_3_end_0, end_mask = per_layer_slice_3_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_3_cast_fp16")]; tensor var_1907 = const()[name = string("op_1907"), val = tensor([0, 2, 1])]; tensor input_55_axes_0 = const()[name = string("input_55_axes_0"), val = tensor([2])]; tensor var_1908 = transpose(perm = var_1907, x = hidden_states_15_cast_fp16)[name = string("transpose_202")]; tensor input_55 = expand_dims(axes = input_55_axes_0, x = var_1908)[name = string("input_55")]; string gated_7_pad_type_0 = const()[name = string("gated_7_pad_type_0"), val = string("valid")]; tensor gated_7_strides_0 = const()[name = string("gated_7_strides_0"), val = tensor([1, 1])]; tensor gated_7_pad_0 = const()[name = string("gated_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_7_dilations_0 = const()[name = string("gated_7_dilations_0"), val = tensor([1, 1])]; int32 gated_7_groups_0 = const()[name = string("gated_7_groups_0"), val = int32(1)]; tensor gated_7 = conv(dilations = gated_7_dilations_0, groups = gated_7_groups_0, pad = gated_7_pad_0, pad_type = gated_7_pad_type_0, strides = gated_7_strides_0, weight = layers_1_per_layer_input_gate_weight_palettized, x = input_55)[name = string("gated_7")]; string gated_9_mode_0 = const()[name = string("gated_9_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_9 = gelu(mode = gated_9_mode_0, x = gated_7)[name = string("gated_9")]; tensor var_1927 = const()[name = string("op_1927"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_3_axes_0 = const()[name = string("per_layer_slice_conv_3_axes_0"), val = tensor([2])]; tensor var_1928_cast_fp16 = transpose(perm = var_1927, x = per_layer_slice_3_cast_fp16)[name = string("transpose_201")]; tensor per_layer_slice_conv_3_cast_fp16 = expand_dims(axes = per_layer_slice_conv_3_axes_0, x = var_1928_cast_fp16)[name = string("per_layer_slice_conv_3_cast_fp16")]; tensor input_57_cast_fp16 = mul(x = gated_9, y = per_layer_slice_conv_3_cast_fp16)[name = string("input_57_cast_fp16")]; string gated_11_pad_type_0 = const()[name = string("gated_11_pad_type_0"), val = string("valid")]; tensor gated_11_strides_0 = const()[name = string("gated_11_strides_0"), val = tensor([1, 1])]; tensor gated_11_pad_0 = const()[name = string("gated_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_11_dilations_0 = const()[name = string("gated_11_dilations_0"), val = tensor([1, 1])]; int32 gated_11_groups_0 = const()[name = string("gated_11_groups_0"), val = int32(1)]; tensor layers_1_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(536866240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537193984))))[name = string("layers_1_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_11_cast_fp16 = conv(dilations = gated_11_dilations_0, groups = gated_11_groups_0, pad = gated_11_pad_0, pad_type = gated_11_pad_type_0, strides = gated_11_strides_0, weight = layers_1_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_57_cast_fp16)[name = string("gated_11_cast_fp16")]; tensor var_1944_axes_0 = const()[name = string("op_1944_axes_0"), val = tensor([2])]; tensor var_1944_cast_fp16 = squeeze(axes = var_1944_axes_0, x = gated_11_cast_fp16)[name = string("op_1944_cast_fp16")]; tensor var_1948 = const()[name = string("op_1948"), val = tensor([0, 2, 1])]; int32 var_1954 = const()[name = string("op_1954"), val = int32(-1)]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_37_cast_fp16 = transpose(perm = var_1948, x = var_1944_cast_fp16)[name = string("transpose_200")]; tensor var_1956_cast_fp16 = mul(x = x_37_cast_fp16, y = const_22_promoted_to_fp16)[name = string("op_1956_cast_fp16")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59_cast_fp16 = concat(axis = var_1954, interleave = input_59_interleave_0, values = (x_37_cast_fp16, var_1956_cast_fp16))[name = string("input_59_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_1951_to_fp16 = const()[name = string("op_1951_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_1951_to_fp16, x = input_59_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor var_1961_split_sizes_0 = const()[name = string("op_1961_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1961_axis_0 = const()[name = string("op_1961_axis_0"), val = int32(-1)]; tensor var_1961_cast_fp16_0, tensor var_1961_cast_fp16_1 = split(axis = var_1961_axis_0, split_sizes = var_1961_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_1961_cast_fp16")]; tensor layers_1_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537196608)))]; tensor hidden_states_19_cast_fp16 = mul(x = var_1961_cast_fp16_0, y = layers_1_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_19_cast_fp16")]; tensor hidden_states_21_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_19_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; tensor const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = tensor([0x1.6cp-1])]; tensor x_39_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_23_promoted_to_fp16)[name = string("x_39_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-1)]; fp16 const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1978_cast_fp16 = mul(x = x_39_cast_fp16, y = const_24_promoted_to_fp16)[name = string("op_1978_cast_fp16")]; bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; tensor input_61_cast_fp16 = concat(axis = var_1976, interleave = input_61_interleave_0, values = (x_39_cast_fp16, var_1978_cast_fp16))[name = string("input_61_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_1973_to_fp16 = const()[name = string("op_1973_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_1973_to_fp16, x = input_61_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_1983_split_sizes_0 = const()[name = string("op_1983_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_1983_axis_0 = const()[name = string("op_1983_axis_0"), val = int32(-1)]; tensor var_1983_cast_fp16_0, tensor var_1983_cast_fp16_1 = split(axis = var_1983_axis_0, split_sizes = var_1983_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_1983_cast_fp16")]; tensor layers_2_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537201792)))]; tensor h_13_cast_fp16 = mul(x = var_1983_cast_fp16_0, y = layers_2_input_layernorm_weight_promoted_to_fp16)[name = string("h_13_cast_fp16")]; tensor var_1989 = const()[name = string("op_1989"), val = tensor([0, 2, 1])]; tensor var_1992_axes_0 = const()[name = string("op_1992_axes_0"), val = tensor([2])]; tensor var_1990_cast_fp16 = transpose(perm = var_1989, x = h_13_cast_fp16)[name = string("transpose_199")]; tensor var_1992_cast_fp16 = expand_dims(axes = var_1992_axes_0, x = var_1990_cast_fp16)[name = string("op_1992_cast_fp16")]; string q_25_pad_type_0 = const()[name = string("q_25_pad_type_0"), val = string("valid")]; tensor q_25_strides_0 = const()[name = string("q_25_strides_0"), val = tensor([1, 1])]; tensor q_25_pad_0 = const()[name = string("q_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_25_dilations_0 = const()[name = string("q_25_dilations_0"), val = tensor([1, 1])]; int32 q_25_groups_0 = const()[name = string("q_25_groups_0"), val = int32(1)]; tensor q_25 = conv(dilations = q_25_dilations_0, groups = q_25_groups_0, pad = q_25_pad_0, pad_type = q_25_pad_type_0, strides = q_25_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_1992_cast_fp16)[name = string("q_25")]; tensor var_2013 = const()[name = string("op_2013"), val = tensor([1, 8, 256, 3])]; tensor var_2014 = reshape(shape = var_2013, x = q_25)[name = string("op_2014")]; tensor transpose_54_perm_0 = const()[name = string("transpose_54_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_2037 = const()[name = string("op_2037"), val = tensor([3, 8, 256])]; tensor transpose_54 = transpose(perm = transpose_54_perm_0, x = var_2014)[name = string("transpose_198")]; tensor x_41 = reshape(shape = var_2037, x = transpose_54)[name = string("x_41")]; int32 var_2043 = const()[name = string("op_2043"), val = int32(-1)]; fp16 const_25_promoted = const()[name = string("const_25_promoted"), val = fp16(-0x1p+0)]; tensor var_2045 = mul(x = x_41, y = const_25_promoted)[name = string("op_2045")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65 = concat(axis = var_2043, interleave = input_65_interleave_0, values = (x_41, var_2045))[name = string("input_65")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_2040_to_fp16 = const()[name = string("op_2040_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_2040_to_fp16, x = input_65)[name = string("normed_61_cast_fp16")]; tensor var_2050_split_sizes_0 = const()[name = string("op_2050_split_sizes_0"), val = tensor([256, 256])]; int32 var_2050_axis_0 = const()[name = string("op_2050_axis_0"), val = int32(-1)]; tensor var_2050_0, tensor var_2050_1 = split(axis = var_2050_axis_0, split_sizes = var_2050_split_sizes_0, x = normed_61_cast_fp16)[name = string("op_2050")]; tensor q_29 = mul(x = var_2050_0, y = layers_2_self_attn_q_norm_weight)[name = string("q_29")]; tensor var_2057 = const()[name = string("op_2057"), val = tensor([1, 3, 8, 256])]; tensor var_2058 = reshape(shape = var_2057, x = q_29)[name = string("op_2058")]; tensor var_2063 = const()[name = string("op_2063"), val = tensor([0, 2, 1, 3])]; tensor q_31 = transpose(perm = var_2063, x = var_2058)[name = string("transpose_197")]; tensor var_2065_cast_fp16 = mul(x = q_31, y = cos_s)[name = string("op_2065_cast_fp16")]; tensor var_2066_split_sizes_0 = const()[name = string("op_2066_split_sizes_0"), val = tensor([128, 128])]; int32 var_2066_axis_0 = const()[name = string("op_2066_axis_0"), val = int32(-1)]; tensor var_2066_0, tensor var_2066_1 = split(axis = var_2066_axis_0, split_sizes = var_2066_split_sizes_0, x = q_31)[name = string("op_2066")]; fp16 const_26_promoted = const()[name = string("const_26_promoted"), val = fp16(-0x1p+0)]; tensor var_2068 = mul(x = var_2066_1, y = const_26_promoted)[name = string("op_2068")]; int32 var_2070 = const()[name = string("op_2070"), val = int32(-1)]; bool var_2071_interleave_0 = const()[name = string("op_2071_interleave_0"), val = bool(false)]; tensor var_2071 = concat(axis = var_2070, interleave = var_2071_interleave_0, values = (var_2068, var_2066_0))[name = string("op_2071")]; tensor var_2072_cast_fp16 = mul(x = var_2071, y = sin_s)[name = string("op_2072_cast_fp16")]; tensor q_35_cast_fp16 = add(x = var_2065_cast_fp16, y = var_2072_cast_fp16)[name = string("q_35_cast_fp16")]; string k_13_pad_type_0 = const()[name = string("k_13_pad_type_0"), val = string("valid")]; tensor k_13_strides_0 = const()[name = string("k_13_strides_0"), val = tensor([1, 1])]; tensor k_13_pad_0 = const()[name = string("k_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_13_dilations_0 = const()[name = string("k_13_dilations_0"), val = tensor([1, 1])]; int32 k_13_groups_0 = const()[name = string("k_13_groups_0"), val = int32(1)]; tensor k_13 = conv(dilations = k_13_dilations_0, groups = k_13_groups_0, pad = k_13_pad_0, pad_type = k_13_pad_type_0, strides = k_13_strides_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_1992_cast_fp16)[name = string("k_13")]; tensor var_2090 = const()[name = string("op_2090"), val = tensor([1, 2, 256, 3])]; tensor var_2091 = reshape(shape = var_2090, x = k_13)[name = string("op_2091")]; tensor transpose_55_perm_0 = const()[name = string("transpose_55_perm_0"), val = tensor([0, 3, 1, 2])]; string v_5_pad_type_0 = const()[name = string("v_5_pad_type_0"), val = string("valid")]; tensor v_5_strides_0 = const()[name = string("v_5_strides_0"), val = tensor([1, 1])]; tensor v_5_pad_0 = const()[name = string("v_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_5_dilations_0 = const()[name = string("v_5_dilations_0"), val = tensor([1, 1])]; int32 v_5_groups_0 = const()[name = string("v_5_groups_0"), val = int32(1)]; tensor v_5 = conv(dilations = v_5_dilations_0, groups = v_5_groups_0, pad = v_5_pad_0, pad_type = v_5_pad_type_0, strides = v_5_strides_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_1992_cast_fp16)[name = string("v_5")]; tensor var_2118 = const()[name = string("op_2118"), val = tensor([1, 2, 256, 3])]; tensor var_2119 = reshape(shape = var_2118, x = v_5)[name = string("op_2119")]; tensor var_2124 = const()[name = string("op_2124"), val = tensor([0, 1, 3, 2])]; tensor var_2142 = const()[name = string("op_2142"), val = tensor([3, 2, 256])]; tensor transpose_55 = transpose(perm = transpose_55_perm_0, x = var_2091)[name = string("transpose_196")]; tensor x_43 = reshape(shape = var_2142, x = transpose_55)[name = string("x_43")]; int32 var_2148 = const()[name = string("op_2148"), val = int32(-1)]; fp16 const_27_promoted = const()[name = string("const_27_promoted"), val = fp16(-0x1p+0)]; tensor var_2150 = mul(x = x_43, y = const_27_promoted)[name = string("op_2150")]; bool input_67_interleave_0 = const()[name = string("input_67_interleave_0"), val = bool(false)]; tensor input_67 = concat(axis = var_2148, interleave = input_67_interleave_0, values = (x_43, var_2150))[name = string("input_67")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_2145_to_fp16 = const()[name = string("op_2145_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_2145_to_fp16, x = input_67)[name = string("normed_65_cast_fp16")]; tensor var_2155_split_sizes_0 = const()[name = string("op_2155_split_sizes_0"), val = tensor([256, 256])]; int32 var_2155_axis_0 = const()[name = string("op_2155_axis_0"), val = int32(-1)]; tensor var_2155_0, tensor var_2155_1 = split(axis = var_2155_axis_0, split_sizes = var_2155_split_sizes_0, x = normed_65_cast_fp16)[name = string("op_2155")]; tensor k_17 = mul(x = var_2155_0, y = layers_2_self_attn_k_norm_weight)[name = string("k_17")]; tensor var_2162 = const()[name = string("op_2162"), val = tensor([1, 3, 2, 256])]; tensor var_2163 = reshape(shape = var_2162, x = k_17)[name = string("op_2163")]; tensor var_2168 = const()[name = string("op_2168"), val = tensor([0, 2, 1, 3])]; fp16 var_2170_promoted = const()[name = string("op_2170_promoted"), val = fp16(0x1p+1)]; tensor var_2125 = transpose(perm = var_2124, x = var_2119)[name = string("transpose_195")]; tensor var_2171 = pow(x = var_2125, y = var_2170_promoted)[name = string("op_2171")]; tensor var_2176_axes_0 = const()[name = string("op_2176_axes_0"), val = tensor([-1])]; bool var_2176_keep_dims_0 = const()[name = string("op_2176_keep_dims_0"), val = bool(true)]; tensor var_2176 = reduce_mean(axes = var_2176_axes_0, keep_dims = var_2176_keep_dims_0, x = var_2171)[name = string("op_2176")]; fp16 var_2178_to_fp16 = const()[name = string("op_2178_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_5_cast_fp16 = add(x = var_2176, y = var_2178_to_fp16)[name = string("mean_sq_5_cast_fp16")]; fp32 var_2180_epsilon_0 = const()[name = string("op_2180_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2180_cast_fp16 = rsqrt(epsilon = var_2180_epsilon_0, x = mean_sq_5_cast_fp16)[name = string("op_2180_cast_fp16")]; tensor input_71_cast_fp16 = mul(x = var_2125, y = var_2180_cast_fp16)[name = string("input_71_cast_fp16")]; tensor q_33 = transpose(perm = var_2168, x = var_2163)[name = string("transpose_194")]; tensor var_2182_cast_fp16 = mul(x = q_33, y = cos_s)[name = string("op_2182_cast_fp16")]; tensor var_2183_split_sizes_0 = const()[name = string("op_2183_split_sizes_0"), val = tensor([128, 128])]; int32 var_2183_axis_0 = const()[name = string("op_2183_axis_0"), val = int32(-1)]; tensor var_2183_0, tensor var_2183_1 = split(axis = var_2183_axis_0, split_sizes = var_2183_split_sizes_0, x = q_33)[name = string("op_2183")]; fp16 const_28_promoted = const()[name = string("const_28_promoted"), val = fp16(-0x1p+0)]; tensor var_2185 = mul(x = var_2183_1, y = const_28_promoted)[name = string("op_2185")]; int32 var_2187 = const()[name = string("op_2187"), val = int32(-1)]; bool var_2188_interleave_0 = const()[name = string("op_2188_interleave_0"), val = bool(false)]; tensor var_2188 = concat(axis = var_2187, interleave = var_2188_interleave_0, values = (var_2185, var_2183_0))[name = string("op_2188")]; tensor var_2189_cast_fp16 = mul(x = var_2188, y = sin_s)[name = string("op_2189_cast_fp16")]; tensor input_69_cast_fp16 = add(x = var_2182_cast_fp16, y = var_2189_cast_fp16)[name = string("input_69_cast_fp16")]; tensor k_padded_5_pad_0 = const()[name = string("k_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_5_mode_0 = const()[name = string("k_padded_5_mode_0"), val = string("constant")]; fp16 const_29_to_fp16 = const()[name = string("const_29_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_5_cast_fp16 = pad(constant_val = const_29_to_fp16, mode = k_padded_5_mode_0, pad = k_padded_5_pad_0, x = input_69_cast_fp16)[name = string("k_padded_5_cast_fp16")]; tensor v_padded_5_pad_0 = const()[name = string("v_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_5_mode_0 = const()[name = string("v_padded_5_mode_0"), val = string("constant")]; fp16 const_30_to_fp16 = const()[name = string("const_30_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_5_cast_fp16 = pad(constant_val = const_30_to_fp16, mode = v_padded_5_mode_0, pad = v_padded_5_pad_0, x = input_71_cast_fp16)[name = string("v_padded_5_cast_fp16")]; tensor slot_k_5_begin_0 = const()[name = string("slot_k_5_begin_0"), val = tensor([2, 0, 0, 0])]; tensor slot_k_5_end_0 = const()[name = string("slot_k_5_end_0"), val = tensor([3, 2, 512, 512])]; tensor slot_k_5_end_mask_0 = const()[name = string("slot_k_5_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_5_cast_fp16 = slice_by_index(begin = slot_k_5_begin_0, end = slot_k_5_end_0, end_mask = slot_k_5_end_mask_0, x = K_sliding_out_3_cast_fp16)[name = string("slot_k_5_cast_fp16")]; tensor slot_v_5_begin_0 = const()[name = string("slot_v_5_begin_0"), val = tensor([2, 0, 0, 0])]; tensor slot_v_5_end_0 = const()[name = string("slot_v_5_end_0"), val = tensor([3, 2, 512, 512])]; tensor slot_v_5_end_mask_0 = const()[name = string("slot_v_5_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_5_cast_fp16 = slice_by_index(begin = slot_v_5_begin_0, end = slot_v_5_end_0, end_mask = slot_v_5_end_mask_0, x = V_sliding_out_3_cast_fp16)[name = string("slot_v_5_cast_fp16")]; tensor var_2228_begin_0 = const()[name = string("op_2228_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_2228_end_0 = const()[name = string("op_2228_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2228_end_mask_0 = const()[name = string("op_2228_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2228_cast_fp16 = slice_by_index(begin = var_2228_begin_0, end = var_2228_end_0, end_mask = var_2228_end_mask_0, x = slot_k_5_cast_fp16)[name = string("op_2228_cast_fp16")]; int32 var_2235 = const()[name = string("op_2235"), val = int32(2)]; bool new_k_5_interleave_0 = const()[name = string("new_k_5_interleave_0"), val = bool(false)]; tensor new_k_5_cast_fp16 = concat(axis = var_2235, interleave = new_k_5_interleave_0, values = (var_2228_cast_fp16, k_padded_5_cast_fp16))[name = string("new_k_5_cast_fp16")]; tensor var_2251_begin_0 = const()[name = string("op_2251_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_2251_end_0 = const()[name = string("op_2251_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2251_end_mask_0 = const()[name = string("op_2251_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2251_cast_fp16 = slice_by_index(begin = var_2251_begin_0, end = var_2251_end_0, end_mask = var_2251_end_mask_0, x = slot_v_5_cast_fp16)[name = string("op_2251_cast_fp16")]; int32 var_2258 = const()[name = string("op_2258"), val = int32(2)]; bool new_v_5_interleave_0 = const()[name = string("new_v_5_interleave_0"), val = bool(false)]; tensor new_v_5_cast_fp16 = concat(axis = var_2258, interleave = new_v_5_interleave_0, values = (var_2251_cast_fp16, v_padded_5_cast_fp16))[name = string("new_v_5_cast_fp16")]; tensor var_2264_begin_0 = const()[name = string("op_2264_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2264_end_0 = const()[name = string("op_2264_end_0"), val = tensor([2, 2, 512, 512])]; tensor var_2264_end_mask_0 = const()[name = string("op_2264_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2264_cast_fp16 = slice_by_index(begin = var_2264_begin_0, end = var_2264_end_0, end_mask = var_2264_end_mask_0, x = K_sliding_out_3_cast_fp16)[name = string("op_2264_cast_fp16")]; tensor var_2269_begin_0 = const()[name = string("op_2269_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2269_end_0 = const()[name = string("op_2269_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_2269_end_mask_0 = const()[name = string("op_2269_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2269_cast_fp16 = slice_by_index(begin = var_2269_begin_0, end = var_2269_end_0, end_mask = var_2269_end_mask_0, x = K_sliding_out_3_cast_fp16)[name = string("op_2269_cast_fp16")]; int32 var_2271 = const()[name = string("op_2271"), val = int32(0)]; bool K_sliding_out_5_interleave_0 = const()[name = string("K_sliding_out_5_interleave_0"), val = bool(false)]; tensor K_sliding_out_5_cast_fp16 = concat(axis = var_2271, interleave = K_sliding_out_5_interleave_0, values = (var_2264_cast_fp16, new_k_5_cast_fp16, var_2269_cast_fp16))[name = string("K_sliding_out_5_cast_fp16")]; tensor var_2277_begin_0 = const()[name = string("op_2277_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2277_end_0 = const()[name = string("op_2277_end_0"), val = tensor([2, 2, 512, 512])]; tensor var_2277_end_mask_0 = const()[name = string("op_2277_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2277_cast_fp16 = slice_by_index(begin = var_2277_begin_0, end = var_2277_end_0, end_mask = var_2277_end_mask_0, x = V_sliding_out_3_cast_fp16)[name = string("op_2277_cast_fp16")]; tensor var_2282_begin_0 = const()[name = string("op_2282_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2282_end_0 = const()[name = string("op_2282_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_2282_end_mask_0 = const()[name = string("op_2282_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2282_cast_fp16 = slice_by_index(begin = var_2282_begin_0, end = var_2282_end_0, end_mask = var_2282_end_mask_0, x = V_sliding_out_3_cast_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(0)]; bool V_sliding_out_5_interleave_0 = const()[name = string("V_sliding_out_5_interleave_0"), val = bool(false)]; tensor V_sliding_out_5_cast_fp16 = concat(axis = var_2284, interleave = V_sliding_out_5_interleave_0, values = (var_2277_cast_fp16, new_v_5_cast_fp16, var_2282_cast_fp16))[name = string("V_sliding_out_5_cast_fp16")]; tensor var_2290_begin_0 = const()[name = string("op_2290_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_2290_end_0 = const()[name = string("op_2290_end_0"), val = tensor([3, 2, 512, 512])]; tensor var_2290_end_mask_0 = const()[name = string("op_2290_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2290_cast_fp16 = slice_by_index(begin = var_2290_begin_0, end = var_2290_end_0, end_mask = var_2290_end_mask_0, x = K_sliding_out_5_cast_fp16)[name = string("op_2290_cast_fp16")]; tensor K_for_attn_5_begin_0 = const()[name = string("K_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_5_end_0 = const()[name = string("K_for_attn_5_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_5_end_mask_0 = const()[name = string("K_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_5_cast_fp16 = slice_by_index(begin = K_for_attn_5_begin_0, end = K_for_attn_5_end_0, end_mask = K_for_attn_5_end_mask_0, x = var_2290_cast_fp16)[name = string("K_for_attn_5_cast_fp16")]; tensor var_2300_begin_0 = const()[name = string("op_2300_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_2300_end_0 = const()[name = string("op_2300_end_0"), val = tensor([3, 2, 512, 512])]; tensor var_2300_end_mask_0 = const()[name = string("op_2300_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2300_cast_fp16 = slice_by_index(begin = var_2300_begin_0, end = var_2300_end_0, end_mask = var_2300_end_mask_0, x = V_sliding_out_5_cast_fp16)[name = string("op_2300_cast_fp16")]; tensor V_for_attn_5_begin_0 = const()[name = string("V_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_5_end_0 = const()[name = string("V_for_attn_5_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_5_end_mask_0 = const()[name = string("V_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_5_cast_fp16 = slice_by_index(begin = V_for_attn_5_begin_0, end = V_for_attn_5_end_0, end_mask = V_for_attn_5_end_mask_0, x = var_2300_cast_fp16)[name = string("V_for_attn_5_cast_fp16")]; tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = K_for_attn_5_cast_fp16)[name = string("transpose_193")]; tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; tensor concat_8 = const()[name = string("concat_8"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_8, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_9 = const()[name = string("concat_9"), val = tensor([-1, 1, 512, 256])]; tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_192")]; tensor reshape_9_cast_fp16 = reshape(shape = concat_9, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor transpose_56_perm_0 = const()[name = string("transpose_56_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = V_for_attn_5_cast_fp16)[name = string("transpose_191")]; tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; tensor concat_10 = const()[name = string("concat_10"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_10, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_11 = const()[name = string("concat_11"), val = tensor([-1, 1, 512, 256])]; tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_190")]; tensor reshape_11_cast_fp16 = reshape(shape = concat_11, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor V_expanded_5_perm_0 = const()[name = string("V_expanded_5_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor transpose_56_cast_fp16 = transpose(perm = transpose_56_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_189")]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = q_35_cast_fp16, y = transpose_56_cast_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor x_47_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask_sliding)[name = string("x_47_cast_fp16")]; tensor reduce_max_2_axes_0 = const()[name = string("reduce_max_2_axes_0"), val = tensor([-1])]; bool reduce_max_2_keep_dims_0 = const()[name = string("reduce_max_2_keep_dims_0"), val = bool(true)]; tensor reduce_max_2 = reduce_max(axes = reduce_max_2_axes_0, keep_dims = reduce_max_2_keep_dims_0, x = x_47_cast_fp16)[name = string("reduce_max_2")]; tensor var_2335 = sub(x = x_47_cast_fp16, y = reduce_max_2)[name = string("op_2335")]; tensor var_2341 = exp(x = var_2335)[name = string("op_2341")]; tensor var_2351_axes_0 = const()[name = string("op_2351_axes_0"), val = tensor([-1])]; bool var_2351_keep_dims_0 = const()[name = string("op_2351_keep_dims_0"), val = bool(true)]; tensor var_2351 = reduce_sum(axes = var_2351_axes_0, keep_dims = var_2351_keep_dims_0, x = var_2341)[name = string("op_2351")]; tensor var_2357_cast_fp16 = real_div(x = var_2341, y = var_2351)[name = string("op_2357_cast_fp16")]; bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; tensor V_expanded_5_cast_fp16 = transpose(perm = V_expanded_5_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_188")]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = var_2357_cast_fp16, y = V_expanded_5_cast_fp16)[name = string("attn_output_13_cast_fp16")]; tensor var_2368 = const()[name = string("op_2368"), val = tensor([0, 2, 1, 3])]; tensor var_2375 = const()[name = string("op_2375"), val = tensor([1, 3, -1])]; tensor var_2369_cast_fp16 = transpose(perm = var_2368, x = attn_output_13_cast_fp16)[name = string("transpose_187")]; tensor attn_output_15_cast_fp16 = reshape(shape = var_2375, x = var_2369_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor var_2380 = const()[name = string("op_2380"), val = tensor([0, 2, 1])]; string var_2396_pad_type_0 = const()[name = string("op_2396_pad_type_0"), val = string("valid")]; int32 var_2396_groups_0 = const()[name = string("op_2396_groups_0"), val = int32(1)]; tensor var_2396_strides_0 = const()[name = string("op_2396_strides_0"), val = tensor([1])]; tensor var_2396_pad_0 = const()[name = string("op_2396_pad_0"), val = tensor([0, 0])]; tensor var_2396_dilations_0 = const()[name = string("op_2396_dilations_0"), val = tensor([1])]; tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(537206976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539828480))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2381_cast_fp16 = transpose(perm = var_2380, x = attn_output_15_cast_fp16)[name = string("transpose_186")]; tensor var_2396_cast_fp16 = conv(dilations = var_2396_dilations_0, groups = var_2396_groups_0, pad = var_2396_pad_0, pad_type = var_2396_pad_type_0, strides = var_2396_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2381_cast_fp16)[name = string("op_2396_cast_fp16")]; tensor var_2400 = const()[name = string("op_2400"), val = tensor([0, 2, 1])]; int32 var_2406 = const()[name = string("op_2406"), val = int32(-1)]; fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_51_cast_fp16 = transpose(perm = var_2400, x = var_2396_cast_fp16)[name = string("transpose_185")]; tensor var_2408_cast_fp16 = mul(x = x_51_cast_fp16, y = const_31_promoted_to_fp16)[name = string("op_2408_cast_fp16")]; bool input_75_interleave_0 = const()[name = string("input_75_interleave_0"), val = bool(false)]; tensor input_75_cast_fp16 = concat(axis = var_2406, interleave = input_75_interleave_0, values = (x_51_cast_fp16, var_2408_cast_fp16))[name = string("input_75_cast_fp16")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_2403_to_fp16 = const()[name = string("op_2403_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_2403_to_fp16, x = input_75_cast_fp16)[name = string("normed_69_cast_fp16")]; tensor var_2413_split_sizes_0 = const()[name = string("op_2413_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2413_axis_0 = const()[name = string("op_2413_axis_0"), val = int32(-1)]; tensor var_2413_cast_fp16_0, tensor var_2413_cast_fp16_1 = split(axis = var_2413_axis_0, split_sizes = var_2413_split_sizes_0, x = normed_69_cast_fp16)[name = string("op_2413_cast_fp16")]; tensor layers_2_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539831104)))]; tensor attn_output_17_cast_fp16 = mul(x = var_2413_cast_fp16_0, y = layers_2_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_17_cast_fp16")]; tensor x_53_cast_fp16 = add(x = x_39_cast_fp16, y = attn_output_17_cast_fp16)[name = string("x_53_cast_fp16")]; int32 var_2422 = const()[name = string("op_2422"), val = int32(-1)]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2424_cast_fp16 = mul(x = x_53_cast_fp16, y = const_32_promoted_to_fp16)[name = string("op_2424_cast_fp16")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77_cast_fp16 = concat(axis = var_2422, interleave = input_77_interleave_0, values = (x_53_cast_fp16, var_2424_cast_fp16))[name = string("input_77_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_2419_to_fp16 = const()[name = string("op_2419_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_2419_to_fp16, x = input_77_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor var_2429_split_sizes_0 = const()[name = string("op_2429_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2429_axis_0 = const()[name = string("op_2429_axis_0"), val = int32(-1)]; tensor var_2429_cast_fp16_0, tensor var_2429_cast_fp16_1 = split(axis = var_2429_axis_0, split_sizes = var_2429_split_sizes_0, x = normed_73_cast_fp16)[name = string("op_2429_cast_fp16")]; tensor layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539836288)))]; tensor h_15_cast_fp16 = mul(x = var_2429_cast_fp16_0, y = layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_15_cast_fp16")]; tensor var_2440 = const()[name = string("op_2440"), val = tensor([0, 2, 1])]; tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; tensor var_2441 = transpose(perm = var_2440, x = h_15_cast_fp16)[name = string("transpose_184")]; tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_2441)[name = string("input_79")]; string gate_9_pad_type_0 = const()[name = string("gate_9_pad_type_0"), val = string("valid")]; tensor gate_9_strides_0 = const()[name = string("gate_9_strides_0"), val = tensor([1, 1])]; tensor gate_9_pad_0 = const()[name = string("gate_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_9_dilations_0 = const()[name = string("gate_9_dilations_0"), val = tensor([1, 1])]; int32 gate_9_groups_0 = const()[name = string("gate_9_groups_0"), val = int32(1)]; tensor gate_9 = conv(dilations = gate_9_dilations_0, groups = gate_9_groups_0, pad = gate_9_pad_0, pad_type = gate_9_pad_type_0, strides = gate_9_strides_0, weight = layers_2_mlp_gate_proj_weight_palettized, x = input_79)[name = string("gate_9")]; string up_5_pad_type_0 = const()[name = string("up_5_pad_type_0"), val = string("valid")]; tensor up_5_strides_0 = const()[name = string("up_5_strides_0"), val = tensor([1, 1])]; tensor up_5_pad_0 = const()[name = string("up_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_5_dilations_0 = const()[name = string("up_5_dilations_0"), val = tensor([1, 1])]; int32 up_5_groups_0 = const()[name = string("up_5_groups_0"), val = int32(1)]; tensor up_5 = conv(dilations = up_5_dilations_0, groups = up_5_groups_0, pad = up_5_pad_0, pad_type = up_5_pad_type_0, strides = up_5_strides_0, weight = layers_2_mlp_up_proj_weight_palettized, x = input_79)[name = string("up_5")]; string gate_11_mode_0 = const()[name = string("gate_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_11 = gelu(mode = gate_11_mode_0, x = gate_9)[name = string("gate_11")]; tensor input_81 = mul(x = gate_11, y = up_5)[name = string("input_81")]; string mlp_out_5_pad_type_0 = const()[name = string("mlp_out_5_pad_type_0"), val = string("valid")]; tensor mlp_out_5_strides_0 = const()[name = string("mlp_out_5_strides_0"), val = tensor([1, 1])]; tensor mlp_out_5_pad_0 = const()[name = string("mlp_out_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_5_dilations_0 = const()[name = string("mlp_out_5_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_5_groups_0 = const()[name = string("mlp_out_5_groups_0"), val = int32(1)]; tensor mlp_out_5 = conv(dilations = mlp_out_5_dilations_0, groups = mlp_out_5_groups_0, pad = mlp_out_5_pad_0, pad_type = mlp_out_5_pad_type_0, strides = mlp_out_5_strides_0, weight = layers_2_mlp_down_proj_weight_palettized, x = input_81)[name = string("mlp_out_5")]; tensor var_2481_axes_0 = const()[name = string("op_2481_axes_0"), val = tensor([2])]; tensor var_2481 = squeeze(axes = var_2481_axes_0, x = mlp_out_5)[name = string("op_2481")]; tensor var_2485 = const()[name = string("op_2485"), val = tensor([0, 2, 1])]; int32 var_2491 = const()[name = string("op_2491"), val = int32(-1)]; fp16 const_33_promoted = const()[name = string("const_33_promoted"), val = fp16(-0x1p+0)]; tensor x_55 = transpose(perm = var_2485, x = var_2481)[name = string("transpose_183")]; tensor var_2493 = mul(x = x_55, y = const_33_promoted)[name = string("op_2493")]; bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; tensor input_83 = concat(axis = var_2491, interleave = input_83_interleave_0, values = (x_55, var_2493))[name = string("input_83")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_2488_to_fp16 = const()[name = string("op_2488_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_2488_to_fp16, x = input_83)[name = string("normed_77_cast_fp16")]; tensor var_2498_split_sizes_0 = const()[name = string("op_2498_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2498_axis_0 = const()[name = string("op_2498_axis_0"), val = int32(-1)]; tensor var_2498_0, tensor var_2498_1 = split(axis = var_2498_axis_0, split_sizes = var_2498_split_sizes_0, x = normed_77_cast_fp16)[name = string("op_2498")]; tensor hidden_states_23 = mul(x = var_2498_0, y = layers_2_post_feedforward_layernorm_weight)[name = string("hidden_states_23")]; tensor hidden_states_25_cast_fp16 = add(x = x_53_cast_fp16, y = hidden_states_23)[name = string("hidden_states_25_cast_fp16")]; tensor per_layer_slice_5_begin_0 = const()[name = string("per_layer_slice_5_begin_0"), val = tensor([0, 0, 3584])]; tensor per_layer_slice_5_end_0 = const()[name = string("per_layer_slice_5_end_0"), val = tensor([1, 3, 3840])]; tensor per_layer_slice_5_end_mask_0 = const()[name = string("per_layer_slice_5_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_5_cast_fp16 = slice_by_index(begin = per_layer_slice_5_begin_0, end = per_layer_slice_5_end_0, end_mask = per_layer_slice_5_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_5_cast_fp16")]; tensor var_2526 = const()[name = string("op_2526"), val = tensor([0, 2, 1])]; tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; tensor var_2527 = transpose(perm = var_2526, x = hidden_states_25_cast_fp16)[name = string("transpose_182")]; tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_2527)[name = string("input_85")]; string gated_13_pad_type_0 = const()[name = string("gated_13_pad_type_0"), val = string("valid")]; tensor gated_13_strides_0 = const()[name = string("gated_13_strides_0"), val = tensor([1, 1])]; tensor gated_13_pad_0 = const()[name = string("gated_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_13_dilations_0 = const()[name = string("gated_13_dilations_0"), val = tensor([1, 1])]; int32 gated_13_groups_0 = const()[name = string("gated_13_groups_0"), val = int32(1)]; tensor gated_13 = conv(dilations = gated_13_dilations_0, groups = gated_13_groups_0, pad = gated_13_pad_0, pad_type = gated_13_pad_type_0, strides = gated_13_strides_0, weight = layers_2_per_layer_input_gate_weight_palettized, x = input_85)[name = string("gated_13")]; string gated_15_mode_0 = const()[name = string("gated_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_15 = gelu(mode = gated_15_mode_0, x = gated_13)[name = string("gated_15")]; tensor var_2546 = const()[name = string("op_2546"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_5_axes_0 = const()[name = string("per_layer_slice_conv_5_axes_0"), val = tensor([2])]; tensor var_2547_cast_fp16 = transpose(perm = var_2546, x = per_layer_slice_5_cast_fp16)[name = string("transpose_181")]; tensor per_layer_slice_conv_5_cast_fp16 = expand_dims(axes = per_layer_slice_conv_5_axes_0, x = var_2547_cast_fp16)[name = string("per_layer_slice_conv_5_cast_fp16")]; tensor input_87_cast_fp16 = mul(x = gated_15, y = per_layer_slice_conv_5_cast_fp16)[name = string("input_87_cast_fp16")]; string gated_17_pad_type_0 = const()[name = string("gated_17_pad_type_0"), val = string("valid")]; tensor gated_17_strides_0 = const()[name = string("gated_17_strides_0"), val = tensor([1, 1])]; tensor gated_17_pad_0 = const()[name = string("gated_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_17_dilations_0 = const()[name = string("gated_17_dilations_0"), val = tensor([1, 1])]; int32 gated_17_groups_0 = const()[name = string("gated_17_groups_0"), val = int32(1)]; tensor layers_2_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(539841472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540169216))))[name = string("layers_2_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_17_cast_fp16 = conv(dilations = gated_17_dilations_0, groups = gated_17_groups_0, pad = gated_17_pad_0, pad_type = gated_17_pad_type_0, strides = gated_17_strides_0, weight = layers_2_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_87_cast_fp16)[name = string("gated_17_cast_fp16")]; tensor var_2563_axes_0 = const()[name = string("op_2563_axes_0"), val = tensor([2])]; tensor var_2563_cast_fp16 = squeeze(axes = var_2563_axes_0, x = gated_17_cast_fp16)[name = string("op_2563_cast_fp16")]; tensor var_2567 = const()[name = string("op_2567"), val = tensor([0, 2, 1])]; int32 var_2573 = const()[name = string("op_2573"), val = int32(-1)]; fp16 const_34_promoted_to_fp16 = const()[name = string("const_34_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_57_cast_fp16 = transpose(perm = var_2567, x = var_2563_cast_fp16)[name = string("transpose_180")]; tensor var_2575_cast_fp16 = mul(x = x_57_cast_fp16, y = const_34_promoted_to_fp16)[name = string("op_2575_cast_fp16")]; bool input_89_interleave_0 = const()[name = string("input_89_interleave_0"), val = bool(false)]; tensor input_89_cast_fp16 = concat(axis = var_2573, interleave = input_89_interleave_0, values = (x_57_cast_fp16, var_2575_cast_fp16))[name = string("input_89_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_2570_to_fp16 = const()[name = string("op_2570_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_2570_to_fp16, x = input_89_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor var_2580_split_sizes_0 = const()[name = string("op_2580_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2580_axis_0 = const()[name = string("op_2580_axis_0"), val = int32(-1)]; tensor var_2580_cast_fp16_0, tensor var_2580_cast_fp16_1 = split(axis = var_2580_axis_0, split_sizes = var_2580_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_2580_cast_fp16")]; tensor layers_2_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540171840)))]; tensor hidden_states_29_cast_fp16 = mul(x = var_2580_cast_fp16_0, y = layers_2_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_29_cast_fp16")]; tensor hidden_states_31_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_29_cast_fp16)[name = string("hidden_states_31_cast_fp16")]; tensor const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = tensor([0x1.58p-1])]; tensor x_59_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_35_promoted_to_fp16)[name = string("x_59_cast_fp16")]; int32 var_2595 = const()[name = string("op_2595"), val = int32(-1)]; fp16 const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2597_cast_fp16 = mul(x = x_59_cast_fp16, y = const_36_promoted_to_fp16)[name = string("op_2597_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_2595, interleave = input_91_interleave_0, values = (x_59_cast_fp16, var_2597_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_2592_to_fp16 = const()[name = string("op_2592_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_2592_to_fp16, x = input_91_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_2602_split_sizes_0 = const()[name = string("op_2602_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_2602_axis_0 = const()[name = string("op_2602_axis_0"), val = int32(-1)]; tensor var_2602_cast_fp16_0, tensor var_2602_cast_fp16_1 = split(axis = var_2602_axis_0, split_sizes = var_2602_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_2602_cast_fp16")]; tensor layers_3_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540177024)))]; tensor h_19_cast_fp16 = mul(x = var_2602_cast_fp16_0, y = layers_3_input_layernorm_weight_promoted_to_fp16)[name = string("h_19_cast_fp16")]; tensor var_2608 = const()[name = string("op_2608"), val = tensor([0, 2, 1])]; tensor var_2611_axes_0 = const()[name = string("op_2611_axes_0"), val = tensor([2])]; tensor var_2609_cast_fp16 = transpose(perm = var_2608, x = h_19_cast_fp16)[name = string("transpose_179")]; tensor var_2611_cast_fp16 = expand_dims(axes = var_2611_axes_0, x = var_2609_cast_fp16)[name = string("op_2611_cast_fp16")]; string q_37_pad_type_0 = const()[name = string("q_37_pad_type_0"), val = string("valid")]; tensor q_37_strides_0 = const()[name = string("q_37_strides_0"), val = tensor([1, 1])]; tensor q_37_pad_0 = const()[name = string("q_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_37_dilations_0 = const()[name = string("q_37_dilations_0"), val = tensor([1, 1])]; int32 q_37_groups_0 = const()[name = string("q_37_groups_0"), val = int32(1)]; tensor q_37 = conv(dilations = q_37_dilations_0, groups = q_37_groups_0, pad = q_37_pad_0, pad_type = q_37_pad_type_0, strides = q_37_strides_0, weight = layers_3_self_attn_q_proj_weight_palettized, x = var_2611_cast_fp16)[name = string("q_37")]; tensor var_2632 = const()[name = string("op_2632"), val = tensor([1, 8, 256, 3])]; tensor var_2633 = reshape(shape = var_2632, x = q_37)[name = string("op_2633")]; tensor transpose_57_perm_0 = const()[name = string("transpose_57_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_2656 = const()[name = string("op_2656"), val = tensor([3, 8, 256])]; tensor transpose_57 = transpose(perm = transpose_57_perm_0, x = var_2633)[name = string("transpose_178")]; tensor x_61 = reshape(shape = var_2656, x = transpose_57)[name = string("x_61")]; int32 var_2662 = const()[name = string("op_2662"), val = int32(-1)]; fp16 const_37_promoted = const()[name = string("const_37_promoted"), val = fp16(-0x1p+0)]; tensor var_2664 = mul(x = x_61, y = const_37_promoted)[name = string("op_2664")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95 = concat(axis = var_2662, interleave = input_95_interleave_0, values = (x_61, var_2664))[name = string("input_95")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_2659_to_fp16 = const()[name = string("op_2659_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_2659_to_fp16, x = input_95)[name = string("normed_89_cast_fp16")]; tensor var_2669_split_sizes_0 = const()[name = string("op_2669_split_sizes_0"), val = tensor([256, 256])]; int32 var_2669_axis_0 = const()[name = string("op_2669_axis_0"), val = int32(-1)]; tensor var_2669_0, tensor var_2669_1 = split(axis = var_2669_axis_0, split_sizes = var_2669_split_sizes_0, x = normed_89_cast_fp16)[name = string("op_2669")]; tensor q_41 = mul(x = var_2669_0, y = layers_3_self_attn_q_norm_weight)[name = string("q_41")]; tensor var_2676 = const()[name = string("op_2676"), val = tensor([1, 3, 8, 256])]; tensor var_2677 = reshape(shape = var_2676, x = q_41)[name = string("op_2677")]; tensor var_2682 = const()[name = string("op_2682"), val = tensor([0, 2, 1, 3])]; tensor q_43 = transpose(perm = var_2682, x = var_2677)[name = string("transpose_177")]; tensor var_2684_cast_fp16 = mul(x = q_43, y = cos_s)[name = string("op_2684_cast_fp16")]; tensor var_2685_split_sizes_0 = const()[name = string("op_2685_split_sizes_0"), val = tensor([128, 128])]; int32 var_2685_axis_0 = const()[name = string("op_2685_axis_0"), val = int32(-1)]; tensor var_2685_0, tensor var_2685_1 = split(axis = var_2685_axis_0, split_sizes = var_2685_split_sizes_0, x = q_43)[name = string("op_2685")]; fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; tensor var_2687 = mul(x = var_2685_1, y = const_38_promoted)[name = string("op_2687")]; int32 var_2689 = const()[name = string("op_2689"), val = int32(-1)]; bool var_2690_interleave_0 = const()[name = string("op_2690_interleave_0"), val = bool(false)]; tensor var_2690 = concat(axis = var_2689, interleave = var_2690_interleave_0, values = (var_2687, var_2685_0))[name = string("op_2690")]; tensor var_2691_cast_fp16 = mul(x = var_2690, y = sin_s)[name = string("op_2691_cast_fp16")]; tensor q_47_cast_fp16 = add(x = var_2684_cast_fp16, y = var_2691_cast_fp16)[name = string("q_47_cast_fp16")]; string k_19_pad_type_0 = const()[name = string("k_19_pad_type_0"), val = string("valid")]; tensor k_19_strides_0 = const()[name = string("k_19_strides_0"), val = tensor([1, 1])]; tensor k_19_pad_0 = const()[name = string("k_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_19_dilations_0 = const()[name = string("k_19_dilations_0"), val = tensor([1, 1])]; int32 k_19_groups_0 = const()[name = string("k_19_groups_0"), val = int32(1)]; tensor k_19 = conv(dilations = k_19_dilations_0, groups = k_19_groups_0, pad = k_19_pad_0, pad_type = k_19_pad_type_0, strides = k_19_strides_0, weight = layers_3_self_attn_k_proj_weight_palettized, x = var_2611_cast_fp16)[name = string("k_19")]; tensor var_2709 = const()[name = string("op_2709"), val = tensor([1, 2, 256, 3])]; tensor var_2710 = reshape(shape = var_2709, x = k_19)[name = string("op_2710")]; tensor transpose_58_perm_0 = const()[name = string("transpose_58_perm_0"), val = tensor([0, 3, 1, 2])]; string v_7_pad_type_0 = const()[name = string("v_7_pad_type_0"), val = string("valid")]; tensor v_7_strides_0 = const()[name = string("v_7_strides_0"), val = tensor([1, 1])]; tensor v_7_pad_0 = const()[name = string("v_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_7_dilations_0 = const()[name = string("v_7_dilations_0"), val = tensor([1, 1])]; int32 v_7_groups_0 = const()[name = string("v_7_groups_0"), val = int32(1)]; tensor v_7 = conv(dilations = v_7_dilations_0, groups = v_7_groups_0, pad = v_7_pad_0, pad_type = v_7_pad_type_0, strides = v_7_strides_0, weight = layers_3_self_attn_v_proj_weight_palettized, x = var_2611_cast_fp16)[name = string("v_7")]; tensor var_2737 = const()[name = string("op_2737"), val = tensor([1, 2, 256, 3])]; tensor var_2738 = reshape(shape = var_2737, x = v_7)[name = string("op_2738")]; tensor var_2743 = const()[name = string("op_2743"), val = tensor([0, 1, 3, 2])]; tensor var_2761 = const()[name = string("op_2761"), val = tensor([3, 2, 256])]; tensor transpose_58 = transpose(perm = transpose_58_perm_0, x = var_2710)[name = string("transpose_176")]; tensor x_63 = reshape(shape = var_2761, x = transpose_58)[name = string("x_63")]; int32 var_2767 = const()[name = string("op_2767"), val = int32(-1)]; fp16 const_39_promoted = const()[name = string("const_39_promoted"), val = fp16(-0x1p+0)]; tensor var_2769 = mul(x = x_63, y = const_39_promoted)[name = string("op_2769")]; bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; tensor input_97 = concat(axis = var_2767, interleave = input_97_interleave_0, values = (x_63, var_2769))[name = string("input_97")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_2764_to_fp16 = const()[name = string("op_2764_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_2764_to_fp16, x = input_97)[name = string("normed_93_cast_fp16")]; tensor var_2774_split_sizes_0 = const()[name = string("op_2774_split_sizes_0"), val = tensor([256, 256])]; int32 var_2774_axis_0 = const()[name = string("op_2774_axis_0"), val = int32(-1)]; tensor var_2774_0, tensor var_2774_1 = split(axis = var_2774_axis_0, split_sizes = var_2774_split_sizes_0, x = normed_93_cast_fp16)[name = string("op_2774")]; tensor k_23 = mul(x = var_2774_0, y = layers_3_self_attn_k_norm_weight)[name = string("k_23")]; tensor var_2781 = const()[name = string("op_2781"), val = tensor([1, 3, 2, 256])]; tensor var_2782 = reshape(shape = var_2781, x = k_23)[name = string("op_2782")]; tensor var_2787 = const()[name = string("op_2787"), val = tensor([0, 2, 1, 3])]; fp16 var_2789_promoted = const()[name = string("op_2789_promoted"), val = fp16(0x1p+1)]; tensor var_2744 = transpose(perm = var_2743, x = var_2738)[name = string("transpose_175")]; tensor var_2790 = pow(x = var_2744, y = var_2789_promoted)[name = string("op_2790")]; tensor var_2795_axes_0 = const()[name = string("op_2795_axes_0"), val = tensor([-1])]; bool var_2795_keep_dims_0 = const()[name = string("op_2795_keep_dims_0"), val = bool(true)]; tensor var_2795 = reduce_mean(axes = var_2795_axes_0, keep_dims = var_2795_keep_dims_0, x = var_2790)[name = string("op_2795")]; fp16 var_2797_to_fp16 = const()[name = string("op_2797_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_7_cast_fp16 = add(x = var_2795, y = var_2797_to_fp16)[name = string("mean_sq_7_cast_fp16")]; fp32 var_2799_epsilon_0 = const()[name = string("op_2799_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2799_cast_fp16 = rsqrt(epsilon = var_2799_epsilon_0, x = mean_sq_7_cast_fp16)[name = string("op_2799_cast_fp16")]; tensor input_101_cast_fp16 = mul(x = var_2744, y = var_2799_cast_fp16)[name = string("input_101_cast_fp16")]; tensor q_45 = transpose(perm = var_2787, x = var_2782)[name = string("transpose_174")]; tensor var_2801_cast_fp16 = mul(x = q_45, y = cos_s)[name = string("op_2801_cast_fp16")]; tensor var_2802_split_sizes_0 = const()[name = string("op_2802_split_sizes_0"), val = tensor([128, 128])]; int32 var_2802_axis_0 = const()[name = string("op_2802_axis_0"), val = int32(-1)]; tensor var_2802_0, tensor var_2802_1 = split(axis = var_2802_axis_0, split_sizes = var_2802_split_sizes_0, x = q_45)[name = string("op_2802")]; fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; tensor var_2804 = mul(x = var_2802_1, y = const_40_promoted)[name = string("op_2804")]; int32 var_2806 = const()[name = string("op_2806"), val = int32(-1)]; bool var_2807_interleave_0 = const()[name = string("op_2807_interleave_0"), val = bool(false)]; tensor var_2807 = concat(axis = var_2806, interleave = var_2807_interleave_0, values = (var_2804, var_2802_0))[name = string("op_2807")]; tensor var_2808_cast_fp16 = mul(x = var_2807, y = sin_s)[name = string("op_2808_cast_fp16")]; tensor input_99_cast_fp16 = add(x = var_2801_cast_fp16, y = var_2808_cast_fp16)[name = string("input_99_cast_fp16")]; tensor k_padded_7_pad_0 = const()[name = string("k_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_7_mode_0 = const()[name = string("k_padded_7_mode_0"), val = string("constant")]; fp16 const_41_to_fp16 = const()[name = string("const_41_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_7_cast_fp16 = pad(constant_val = const_41_to_fp16, mode = k_padded_7_mode_0, pad = k_padded_7_pad_0, x = input_99_cast_fp16)[name = string("k_padded_7_cast_fp16")]; tensor v_padded_7_pad_0 = const()[name = string("v_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_7_mode_0 = const()[name = string("v_padded_7_mode_0"), val = string("constant")]; fp16 const_42_to_fp16 = const()[name = string("const_42_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_7_cast_fp16 = pad(constant_val = const_42_to_fp16, mode = v_padded_7_mode_0, pad = v_padded_7_pad_0, x = input_101_cast_fp16)[name = string("v_padded_7_cast_fp16")]; tensor slot_k_7_begin_0 = const()[name = string("slot_k_7_begin_0"), val = tensor([3, 0, 0, 0])]; tensor slot_k_7_end_0 = const()[name = string("slot_k_7_end_0"), val = tensor([4, 2, 512, 512])]; tensor slot_k_7_end_mask_0 = const()[name = string("slot_k_7_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_7_cast_fp16 = slice_by_index(begin = slot_k_7_begin_0, end = slot_k_7_end_0, end_mask = slot_k_7_end_mask_0, x = K_sliding_out_5_cast_fp16)[name = string("slot_k_7_cast_fp16")]; tensor slot_v_7_begin_0 = const()[name = string("slot_v_7_begin_0"), val = tensor([3, 0, 0, 0])]; tensor slot_v_7_end_0 = const()[name = string("slot_v_7_end_0"), val = tensor([4, 2, 512, 512])]; tensor slot_v_7_end_mask_0 = const()[name = string("slot_v_7_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_7_cast_fp16 = slice_by_index(begin = slot_v_7_begin_0, end = slot_v_7_end_0, end_mask = slot_v_7_end_mask_0, x = V_sliding_out_5_cast_fp16)[name = string("slot_v_7_cast_fp16")]; tensor var_2847_begin_0 = const()[name = string("op_2847_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_2847_end_0 = const()[name = string("op_2847_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2847_end_mask_0 = const()[name = string("op_2847_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2847_cast_fp16 = slice_by_index(begin = var_2847_begin_0, end = var_2847_end_0, end_mask = var_2847_end_mask_0, x = slot_k_7_cast_fp16)[name = string("op_2847_cast_fp16")]; int32 var_2854 = const()[name = string("op_2854"), val = int32(2)]; bool new_k_7_interleave_0 = const()[name = string("new_k_7_interleave_0"), val = bool(false)]; tensor new_k_7_cast_fp16 = concat(axis = var_2854, interleave = new_k_7_interleave_0, values = (var_2847_cast_fp16, k_padded_7_cast_fp16))[name = string("new_k_7_cast_fp16")]; tensor var_2870_begin_0 = const()[name = string("op_2870_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_2870_end_0 = const()[name = string("op_2870_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_2870_end_mask_0 = const()[name = string("op_2870_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2870_cast_fp16 = slice_by_index(begin = var_2870_begin_0, end = var_2870_end_0, end_mask = var_2870_end_mask_0, x = slot_v_7_cast_fp16)[name = string("op_2870_cast_fp16")]; int32 var_2877 = const()[name = string("op_2877"), val = int32(2)]; bool new_v_7_interleave_0 = const()[name = string("new_v_7_interleave_0"), val = bool(false)]; tensor new_v_7_cast_fp16 = concat(axis = var_2877, interleave = new_v_7_interleave_0, values = (var_2870_cast_fp16, v_padded_7_cast_fp16))[name = string("new_v_7_cast_fp16")]; tensor var_2883_begin_0 = const()[name = string("op_2883_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2883_end_0 = const()[name = string("op_2883_end_0"), val = tensor([3, 2, 512, 512])]; tensor var_2883_end_mask_0 = const()[name = string("op_2883_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2883_cast_fp16 = slice_by_index(begin = var_2883_begin_0, end = var_2883_end_0, end_mask = var_2883_end_mask_0, x = K_sliding_out_5_cast_fp16)[name = string("op_2883_cast_fp16")]; tensor var_2888_begin_0 = const()[name = string("op_2888_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2888_end_0 = const()[name = string("op_2888_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_2888_end_mask_0 = const()[name = string("op_2888_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2888_cast_fp16 = slice_by_index(begin = var_2888_begin_0, end = var_2888_end_0, end_mask = var_2888_end_mask_0, x = K_sliding_out_5_cast_fp16)[name = string("op_2888_cast_fp16")]; int32 var_2890 = const()[name = string("op_2890"), val = int32(0)]; bool K_sliding_out_7_interleave_0 = const()[name = string("K_sliding_out_7_interleave_0"), val = bool(false)]; tensor K_sliding_out_7_cast_fp16 = concat(axis = var_2890, interleave = K_sliding_out_7_interleave_0, values = (var_2883_cast_fp16, new_k_7_cast_fp16, var_2888_cast_fp16))[name = string("K_sliding_out_7_cast_fp16")]; tensor var_2896_begin_0 = const()[name = string("op_2896_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2896_end_0 = const()[name = string("op_2896_end_0"), val = tensor([3, 2, 512, 512])]; tensor var_2896_end_mask_0 = const()[name = string("op_2896_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2896_cast_fp16 = slice_by_index(begin = var_2896_begin_0, end = var_2896_end_0, end_mask = var_2896_end_mask_0, x = V_sliding_out_5_cast_fp16)[name = string("op_2896_cast_fp16")]; tensor var_2901_begin_0 = const()[name = string("op_2901_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2901_end_0 = const()[name = string("op_2901_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_2901_end_mask_0 = const()[name = string("op_2901_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2901_cast_fp16 = slice_by_index(begin = var_2901_begin_0, end = var_2901_end_0, end_mask = var_2901_end_mask_0, x = V_sliding_out_5_cast_fp16)[name = string("op_2901_cast_fp16")]; int32 var_2903 = const()[name = string("op_2903"), val = int32(0)]; bool V_sliding_out_7_interleave_0 = const()[name = string("V_sliding_out_7_interleave_0"), val = bool(false)]; tensor V_sliding_out_7_cast_fp16 = concat(axis = var_2903, interleave = V_sliding_out_7_interleave_0, values = (var_2896_cast_fp16, new_v_7_cast_fp16, var_2901_cast_fp16))[name = string("V_sliding_out_7_cast_fp16")]; tensor var_2909_begin_0 = const()[name = string("op_2909_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2909_end_0 = const()[name = string("op_2909_end_0"), val = tensor([4, 2, 512, 512])]; tensor var_2909_end_mask_0 = const()[name = string("op_2909_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2909_cast_fp16 = slice_by_index(begin = var_2909_begin_0, end = var_2909_end_0, end_mask = var_2909_end_mask_0, x = K_sliding_out_7_cast_fp16)[name = string("op_2909_cast_fp16")]; tensor K_for_attn_7_begin_0 = const()[name = string("K_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_7_end_0 = const()[name = string("K_for_attn_7_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_7_end_mask_0 = const()[name = string("K_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_7_cast_fp16 = slice_by_index(begin = K_for_attn_7_begin_0, end = K_for_attn_7_end_0, end_mask = K_for_attn_7_end_mask_0, x = var_2909_cast_fp16)[name = string("K_for_attn_7_cast_fp16")]; tensor var_2919_begin_0 = const()[name = string("op_2919_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2919_end_0 = const()[name = string("op_2919_end_0"), val = tensor([4, 2, 512, 512])]; tensor var_2919_end_mask_0 = const()[name = string("op_2919_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2919_cast_fp16 = slice_by_index(begin = var_2919_begin_0, end = var_2919_end_0, end_mask = var_2919_end_mask_0, x = V_sliding_out_7_cast_fp16)[name = string("op_2919_cast_fp16")]; tensor V_for_attn_7_begin_0 = const()[name = string("V_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_7_end_0 = const()[name = string("V_for_attn_7_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_7_end_mask_0 = const()[name = string("V_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_7_cast_fp16 = slice_by_index(begin = V_for_attn_7_begin_0, end = V_for_attn_7_end_0, end_mask = V_for_attn_7_end_mask_0, x = var_2919_cast_fp16)[name = string("V_for_attn_7_cast_fp16")]; tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = K_for_attn_7_cast_fp16)[name = string("transpose_173")]; tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; tensor concat_12 = const()[name = string("concat_12"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_12, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_13 = const()[name = string("concat_13"), val = tensor([-1, 1, 512, 256])]; tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_172")]; tensor reshape_13_cast_fp16 = reshape(shape = concat_13, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; tensor transpose_59_perm_0 = const()[name = string("transpose_59_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = V_for_attn_7_cast_fp16)[name = string("transpose_171")]; tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; tensor concat_14 = const()[name = string("concat_14"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_14, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_15 = const()[name = string("concat_15"), val = tensor([-1, 1, 512, 256])]; tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_170")]; tensor reshape_15_cast_fp16 = reshape(shape = concat_15, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor V_expanded_7_perm_0 = const()[name = string("V_expanded_7_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_13_transpose_x_0 = const()[name = string("attn_weights_13_transpose_x_0"), val = bool(false)]; bool attn_weights_13_transpose_y_0 = const()[name = string("attn_weights_13_transpose_y_0"), val = bool(false)]; tensor transpose_59_cast_fp16 = transpose(perm = transpose_59_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_169")]; tensor attn_weights_13_cast_fp16 = matmul(transpose_x = attn_weights_13_transpose_x_0, transpose_y = attn_weights_13_transpose_y_0, x = q_47_cast_fp16, y = transpose_59_cast_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor x_67_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask_sliding)[name = string("x_67_cast_fp16")]; tensor reduce_max_3_axes_0 = const()[name = string("reduce_max_3_axes_0"), val = tensor([-1])]; bool reduce_max_3_keep_dims_0 = const()[name = string("reduce_max_3_keep_dims_0"), val = bool(true)]; tensor reduce_max_3 = reduce_max(axes = reduce_max_3_axes_0, keep_dims = reduce_max_3_keep_dims_0, x = x_67_cast_fp16)[name = string("reduce_max_3")]; tensor var_2954 = sub(x = x_67_cast_fp16, y = reduce_max_3)[name = string("op_2954")]; tensor var_2960 = exp(x = var_2954)[name = string("op_2960")]; tensor var_2970_axes_0 = const()[name = string("op_2970_axes_0"), val = tensor([-1])]; bool var_2970_keep_dims_0 = const()[name = string("op_2970_keep_dims_0"), val = bool(true)]; tensor var_2970 = reduce_sum(axes = var_2970_axes_0, keep_dims = var_2970_keep_dims_0, x = var_2960)[name = string("op_2970")]; tensor var_2976_cast_fp16 = real_div(x = var_2960, y = var_2970)[name = string("op_2976_cast_fp16")]; bool attn_output_19_transpose_x_0 = const()[name = string("attn_output_19_transpose_x_0"), val = bool(false)]; bool attn_output_19_transpose_y_0 = const()[name = string("attn_output_19_transpose_y_0"), val = bool(false)]; tensor V_expanded_7_cast_fp16 = transpose(perm = V_expanded_7_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_168")]; tensor attn_output_19_cast_fp16 = matmul(transpose_x = attn_output_19_transpose_x_0, transpose_y = attn_output_19_transpose_y_0, x = var_2976_cast_fp16, y = V_expanded_7_cast_fp16)[name = string("attn_output_19_cast_fp16")]; tensor var_2987 = const()[name = string("op_2987"), val = tensor([0, 2, 1, 3])]; tensor var_2994 = const()[name = string("op_2994"), val = tensor([1, 3, -1])]; tensor var_2988_cast_fp16 = transpose(perm = var_2987, x = attn_output_19_cast_fp16)[name = string("transpose_167")]; tensor attn_output_21_cast_fp16 = reshape(shape = var_2994, x = var_2988_cast_fp16)[name = string("attn_output_21_cast_fp16")]; tensor var_2999 = const()[name = string("op_2999"), val = tensor([0, 2, 1])]; string var_3015_pad_type_0 = const()[name = string("op_3015_pad_type_0"), val = string("valid")]; int32 var_3015_groups_0 = const()[name = string("op_3015_groups_0"), val = int32(1)]; tensor var_3015_strides_0 = const()[name = string("op_3015_strides_0"), val = tensor([1])]; tensor var_3015_pad_0 = const()[name = string("op_3015_pad_0"), val = tensor([0, 0])]; tensor var_3015_dilations_0 = const()[name = string("op_3015_dilations_0"), val = tensor([1])]; tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(540182208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542803712))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3000_cast_fp16 = transpose(perm = var_2999, x = attn_output_21_cast_fp16)[name = string("transpose_166")]; tensor var_3015_cast_fp16 = conv(dilations = var_3015_dilations_0, groups = var_3015_groups_0, pad = var_3015_pad_0, pad_type = var_3015_pad_type_0, strides = var_3015_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_3000_cast_fp16)[name = string("op_3015_cast_fp16")]; tensor var_3019 = const()[name = string("op_3019"), val = tensor([0, 2, 1])]; int32 var_3025 = const()[name = string("op_3025"), val = int32(-1)]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_71_cast_fp16 = transpose(perm = var_3019, x = var_3015_cast_fp16)[name = string("transpose_165")]; tensor var_3027_cast_fp16 = mul(x = x_71_cast_fp16, y = const_43_promoted_to_fp16)[name = string("op_3027_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_3025, interleave = input_105_interleave_0, values = (x_71_cast_fp16, var_3027_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_3022_to_fp16 = const()[name = string("op_3022_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_3022_to_fp16, x = input_105_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor var_3032_split_sizes_0 = const()[name = string("op_3032_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3032_axis_0 = const()[name = string("op_3032_axis_0"), val = int32(-1)]; tensor var_3032_cast_fp16_0, tensor var_3032_cast_fp16_1 = split(axis = var_3032_axis_0, split_sizes = var_3032_split_sizes_0, x = normed_97_cast_fp16)[name = string("op_3032_cast_fp16")]; tensor layers_3_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542806336)))]; tensor attn_output_23_cast_fp16 = mul(x = var_3032_cast_fp16_0, y = layers_3_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_23_cast_fp16")]; tensor x_73_cast_fp16 = add(x = x_59_cast_fp16, y = attn_output_23_cast_fp16)[name = string("x_73_cast_fp16")]; int32 var_3041 = const()[name = string("op_3041"), val = int32(-1)]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3043_cast_fp16 = mul(x = x_73_cast_fp16, y = const_44_promoted_to_fp16)[name = string("op_3043_cast_fp16")]; bool input_107_interleave_0 = const()[name = string("input_107_interleave_0"), val = bool(false)]; tensor input_107_cast_fp16 = concat(axis = var_3041, interleave = input_107_interleave_0, values = (x_73_cast_fp16, var_3043_cast_fp16))[name = string("input_107_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_3038_to_fp16 = const()[name = string("op_3038_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_3038_to_fp16, x = input_107_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor var_3048_split_sizes_0 = const()[name = string("op_3048_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3048_axis_0 = const()[name = string("op_3048_axis_0"), val = int32(-1)]; tensor var_3048_cast_fp16_0, tensor var_3048_cast_fp16_1 = split(axis = var_3048_axis_0, split_sizes = var_3048_split_sizes_0, x = normed_101_cast_fp16)[name = string("op_3048_cast_fp16")]; tensor layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542811520)))]; tensor h_21_cast_fp16 = mul(x = var_3048_cast_fp16_0, y = layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_21_cast_fp16")]; tensor var_3059 = const()[name = string("op_3059"), val = tensor([0, 2, 1])]; tensor input_109_axes_0 = const()[name = string("input_109_axes_0"), val = tensor([2])]; tensor var_3060 = transpose(perm = var_3059, x = h_21_cast_fp16)[name = string("transpose_164")]; tensor input_109 = expand_dims(axes = input_109_axes_0, x = var_3060)[name = string("input_109")]; string gate_13_pad_type_0 = const()[name = string("gate_13_pad_type_0"), val = string("valid")]; tensor gate_13_strides_0 = const()[name = string("gate_13_strides_0"), val = tensor([1, 1])]; tensor gate_13_pad_0 = const()[name = string("gate_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_13_dilations_0 = const()[name = string("gate_13_dilations_0"), val = tensor([1, 1])]; int32 gate_13_groups_0 = const()[name = string("gate_13_groups_0"), val = int32(1)]; tensor gate_13 = conv(dilations = gate_13_dilations_0, groups = gate_13_groups_0, pad = gate_13_pad_0, pad_type = gate_13_pad_type_0, strides = gate_13_strides_0, weight = layers_3_mlp_gate_proj_weight_palettized, x = input_109)[name = string("gate_13")]; string up_7_pad_type_0 = const()[name = string("up_7_pad_type_0"), val = string("valid")]; tensor up_7_strides_0 = const()[name = string("up_7_strides_0"), val = tensor([1, 1])]; tensor up_7_pad_0 = const()[name = string("up_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_7_dilations_0 = const()[name = string("up_7_dilations_0"), val = tensor([1, 1])]; int32 up_7_groups_0 = const()[name = string("up_7_groups_0"), val = int32(1)]; tensor up_7 = conv(dilations = up_7_dilations_0, groups = up_7_groups_0, pad = up_7_pad_0, pad_type = up_7_pad_type_0, strides = up_7_strides_0, weight = layers_3_mlp_up_proj_weight_palettized, x = input_109)[name = string("up_7")]; string gate_15_mode_0 = const()[name = string("gate_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_15 = gelu(mode = gate_15_mode_0, x = gate_13)[name = string("gate_15")]; tensor input_111 = mul(x = gate_15, y = up_7)[name = string("input_111")]; string mlp_out_7_pad_type_0 = const()[name = string("mlp_out_7_pad_type_0"), val = string("valid")]; tensor mlp_out_7_strides_0 = const()[name = string("mlp_out_7_strides_0"), val = tensor([1, 1])]; tensor mlp_out_7_pad_0 = const()[name = string("mlp_out_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_7_dilations_0 = const()[name = string("mlp_out_7_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_7_groups_0 = const()[name = string("mlp_out_7_groups_0"), val = int32(1)]; tensor mlp_out_7 = conv(dilations = mlp_out_7_dilations_0, groups = mlp_out_7_groups_0, pad = mlp_out_7_pad_0, pad_type = mlp_out_7_pad_type_0, strides = mlp_out_7_strides_0, weight = layers_3_mlp_down_proj_weight_palettized, x = input_111)[name = string("mlp_out_7")]; tensor var_3100_axes_0 = const()[name = string("op_3100_axes_0"), val = tensor([2])]; tensor var_3100 = squeeze(axes = var_3100_axes_0, x = mlp_out_7)[name = string("op_3100")]; tensor var_3104 = const()[name = string("op_3104"), val = tensor([0, 2, 1])]; int32 var_3110 = const()[name = string("op_3110"), val = int32(-1)]; fp16 const_45_promoted = const()[name = string("const_45_promoted"), val = fp16(-0x1p+0)]; tensor x_75 = transpose(perm = var_3104, x = var_3100)[name = string("transpose_163")]; tensor var_3112 = mul(x = x_75, y = const_45_promoted)[name = string("op_3112")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113 = concat(axis = var_3110, interleave = input_113_interleave_0, values = (x_75, var_3112))[name = string("input_113")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_3107_to_fp16 = const()[name = string("op_3107_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_3107_to_fp16, x = input_113)[name = string("normed_105_cast_fp16")]; tensor var_3117_split_sizes_0 = const()[name = string("op_3117_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3117_axis_0 = const()[name = string("op_3117_axis_0"), val = int32(-1)]; tensor var_3117_0, tensor var_3117_1 = split(axis = var_3117_axis_0, split_sizes = var_3117_split_sizes_0, x = normed_105_cast_fp16)[name = string("op_3117")]; tensor hidden_states_33 = mul(x = var_3117_0, y = layers_3_post_feedforward_layernorm_weight)[name = string("hidden_states_33")]; tensor hidden_states_35_cast_fp16 = add(x = x_73_cast_fp16, y = hidden_states_33)[name = string("hidden_states_35_cast_fp16")]; tensor per_layer_slice_7_begin_0 = const()[name = string("per_layer_slice_7_begin_0"), val = tensor([0, 0, 3840])]; tensor per_layer_slice_7_end_0 = const()[name = string("per_layer_slice_7_end_0"), val = tensor([1, 3, 4096])]; tensor per_layer_slice_7_end_mask_0 = const()[name = string("per_layer_slice_7_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_7_cast_fp16 = slice_by_index(begin = per_layer_slice_7_begin_0, end = per_layer_slice_7_end_0, end_mask = per_layer_slice_7_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_7_cast_fp16")]; tensor var_3145 = const()[name = string("op_3145"), val = tensor([0, 2, 1])]; tensor input_115_axes_0 = const()[name = string("input_115_axes_0"), val = tensor([2])]; tensor var_3146 = transpose(perm = var_3145, x = hidden_states_35_cast_fp16)[name = string("transpose_162")]; tensor input_115 = expand_dims(axes = input_115_axes_0, x = var_3146)[name = string("input_115")]; string gated_19_pad_type_0 = const()[name = string("gated_19_pad_type_0"), val = string("valid")]; tensor gated_19_strides_0 = const()[name = string("gated_19_strides_0"), val = tensor([1, 1])]; tensor gated_19_pad_0 = const()[name = string("gated_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_19_dilations_0 = const()[name = string("gated_19_dilations_0"), val = tensor([1, 1])]; int32 gated_19_groups_0 = const()[name = string("gated_19_groups_0"), val = int32(1)]; tensor gated_19 = conv(dilations = gated_19_dilations_0, groups = gated_19_groups_0, pad = gated_19_pad_0, pad_type = gated_19_pad_type_0, strides = gated_19_strides_0, weight = layers_3_per_layer_input_gate_weight_palettized, x = input_115)[name = string("gated_19")]; string gated_21_mode_0 = const()[name = string("gated_21_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_21 = gelu(mode = gated_21_mode_0, x = gated_19)[name = string("gated_21")]; tensor var_3165 = const()[name = string("op_3165"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_7_axes_0 = const()[name = string("per_layer_slice_conv_7_axes_0"), val = tensor([2])]; tensor var_3166_cast_fp16 = transpose(perm = var_3165, x = per_layer_slice_7_cast_fp16)[name = string("transpose_161")]; tensor per_layer_slice_conv_7_cast_fp16 = expand_dims(axes = per_layer_slice_conv_7_axes_0, x = var_3166_cast_fp16)[name = string("per_layer_slice_conv_7_cast_fp16")]; tensor input_117_cast_fp16 = mul(x = gated_21, y = per_layer_slice_conv_7_cast_fp16)[name = string("input_117_cast_fp16")]; string gated_23_pad_type_0 = const()[name = string("gated_23_pad_type_0"), val = string("valid")]; tensor gated_23_strides_0 = const()[name = string("gated_23_strides_0"), val = tensor([1, 1])]; tensor gated_23_pad_0 = const()[name = string("gated_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_23_dilations_0 = const()[name = string("gated_23_dilations_0"), val = tensor([1, 1])]; int32 gated_23_groups_0 = const()[name = string("gated_23_groups_0"), val = int32(1)]; tensor layers_3_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542816704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543144448))))[name = string("layers_3_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_23_cast_fp16 = conv(dilations = gated_23_dilations_0, groups = gated_23_groups_0, pad = gated_23_pad_0, pad_type = gated_23_pad_type_0, strides = gated_23_strides_0, weight = layers_3_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_117_cast_fp16)[name = string("gated_23_cast_fp16")]; tensor var_3182_axes_0 = const()[name = string("op_3182_axes_0"), val = tensor([2])]; tensor var_3182_cast_fp16 = squeeze(axes = var_3182_axes_0, x = gated_23_cast_fp16)[name = string("op_3182_cast_fp16")]; tensor var_3186 = const()[name = string("op_3186"), val = tensor([0, 2, 1])]; int32 var_3192 = const()[name = string("op_3192"), val = int32(-1)]; fp16 const_46_promoted_to_fp16 = const()[name = string("const_46_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_77_cast_fp16 = transpose(perm = var_3186, x = var_3182_cast_fp16)[name = string("transpose_160")]; tensor var_3194_cast_fp16 = mul(x = x_77_cast_fp16, y = const_46_promoted_to_fp16)[name = string("op_3194_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_3192, interleave = input_119_interleave_0, values = (x_77_cast_fp16, var_3194_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_3189_to_fp16 = const()[name = string("op_3189_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_3189_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor var_3199_split_sizes_0 = const()[name = string("op_3199_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3199_axis_0 = const()[name = string("op_3199_axis_0"), val = int32(-1)]; tensor var_3199_cast_fp16_0, tensor var_3199_cast_fp16_1 = split(axis = var_3199_axis_0, split_sizes = var_3199_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_3199_cast_fp16")]; tensor layers_3_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543147072)))]; tensor hidden_states_39_cast_fp16 = mul(x = var_3199_cast_fp16_0, y = layers_3_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_39_cast_fp16")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; tensor const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = tensor([0x1.14p-1])]; tensor x_79_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_47_promoted_to_fp16)[name = string("x_79_cast_fp16")]; int32 var_3214 = const()[name = string("op_3214"), val = int32(-1)]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3216_cast_fp16 = mul(x = x_79_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_3216_cast_fp16")]; bool input_121_interleave_0 = const()[name = string("input_121_interleave_0"), val = bool(false)]; tensor input_121_cast_fp16 = concat(axis = var_3214, interleave = input_121_interleave_0, values = (x_79_cast_fp16, var_3216_cast_fp16))[name = string("input_121_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_3211_to_fp16 = const()[name = string("op_3211_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_3211_to_fp16, x = input_121_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_3221_split_sizes_0 = const()[name = string("op_3221_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3221_axis_0 = const()[name = string("op_3221_axis_0"), val = int32(-1)]; tensor var_3221_cast_fp16_0, tensor var_3221_cast_fp16_1 = split(axis = var_3221_axis_0, split_sizes = var_3221_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_3221_cast_fp16")]; tensor layers_4_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543152256)))]; tensor h_25_cast_fp16 = mul(x = var_3221_cast_fp16_0, y = layers_4_input_layernorm_weight_promoted_to_fp16)[name = string("h_25_cast_fp16")]; tensor var_3227 = const()[name = string("op_3227"), val = tensor([0, 2, 1])]; tensor var_3230_axes_0 = const()[name = string("op_3230_axes_0"), val = tensor([2])]; tensor var_3228_cast_fp16 = transpose(perm = var_3227, x = h_25_cast_fp16)[name = string("transpose_159")]; tensor var_3230_cast_fp16 = expand_dims(axes = var_3230_axes_0, x = var_3228_cast_fp16)[name = string("op_3230_cast_fp16")]; string q_49_pad_type_0 = const()[name = string("q_49_pad_type_0"), val = string("valid")]; tensor q_49_strides_0 = const()[name = string("q_49_strides_0"), val = tensor([1, 1])]; tensor q_49_pad_0 = const()[name = string("q_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_49_dilations_0 = const()[name = string("q_49_dilations_0"), val = tensor([1, 1])]; int32 q_49_groups_0 = const()[name = string("q_49_groups_0"), val = int32(1)]; tensor q_49 = conv(dilations = q_49_dilations_0, groups = q_49_groups_0, pad = q_49_pad_0, pad_type = q_49_pad_type_0, strides = q_49_strides_0, weight = layers_4_self_attn_q_proj_weight_palettized, x = var_3230_cast_fp16)[name = string("q_49")]; tensor var_3251 = const()[name = string("op_3251"), val = tensor([1, 8, 256, 3])]; tensor var_3252 = reshape(shape = var_3251, x = q_49)[name = string("op_3252")]; tensor transpose_60_perm_0 = const()[name = string("transpose_60_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_3275 = const()[name = string("op_3275"), val = tensor([3, 8, 256])]; tensor transpose_60 = transpose(perm = transpose_60_perm_0, x = var_3252)[name = string("transpose_158")]; tensor x_81 = reshape(shape = var_3275, x = transpose_60)[name = string("x_81")]; int32 var_3281 = const()[name = string("op_3281"), val = int32(-1)]; fp16 const_49_promoted = const()[name = string("const_49_promoted"), val = fp16(-0x1p+0)]; tensor var_3283 = mul(x = x_81, y = const_49_promoted)[name = string("op_3283")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125 = concat(axis = var_3281, interleave = input_125_interleave_0, values = (x_81, var_3283))[name = string("input_125")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_3278_to_fp16 = const()[name = string("op_3278_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_3278_to_fp16, x = input_125)[name = string("normed_117_cast_fp16")]; tensor var_3288_split_sizes_0 = const()[name = string("op_3288_split_sizes_0"), val = tensor([256, 256])]; int32 var_3288_axis_0 = const()[name = string("op_3288_axis_0"), val = int32(-1)]; tensor var_3288_0, tensor var_3288_1 = split(axis = var_3288_axis_0, split_sizes = var_3288_split_sizes_0, x = normed_117_cast_fp16)[name = string("op_3288")]; tensor q_53 = mul(x = var_3288_0, y = layers_4_self_attn_q_norm_weight)[name = string("q_53")]; tensor var_3295 = const()[name = string("op_3295"), val = tensor([1, 3, 8, 256])]; tensor var_3296 = reshape(shape = var_3295, x = q_53)[name = string("op_3296")]; tensor var_3301 = const()[name = string("op_3301"), val = tensor([0, 2, 1, 3])]; tensor q_55 = transpose(perm = var_3301, x = var_3296)[name = string("transpose_157")]; tensor var_3303_cast_fp16 = mul(x = q_55, y = cos_s)[name = string("op_3303_cast_fp16")]; tensor var_3304_split_sizes_0 = const()[name = string("op_3304_split_sizes_0"), val = tensor([128, 128])]; int32 var_3304_axis_0 = const()[name = string("op_3304_axis_0"), val = int32(-1)]; tensor var_3304_0, tensor var_3304_1 = split(axis = var_3304_axis_0, split_sizes = var_3304_split_sizes_0, x = q_55)[name = string("op_3304")]; fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; tensor var_3306 = mul(x = var_3304_1, y = const_50_promoted)[name = string("op_3306")]; int32 var_3308 = const()[name = string("op_3308"), val = int32(-1)]; bool var_3309_interleave_0 = const()[name = string("op_3309_interleave_0"), val = bool(false)]; tensor var_3309 = concat(axis = var_3308, interleave = var_3309_interleave_0, values = (var_3306, var_3304_0))[name = string("op_3309")]; tensor var_3310_cast_fp16 = mul(x = var_3309, y = sin_s)[name = string("op_3310_cast_fp16")]; tensor q_59_cast_fp16 = add(x = var_3303_cast_fp16, y = var_3310_cast_fp16)[name = string("q_59_cast_fp16")]; string k_25_pad_type_0 = const()[name = string("k_25_pad_type_0"), val = string("valid")]; tensor k_25_strides_0 = const()[name = string("k_25_strides_0"), val = tensor([1, 1])]; tensor k_25_pad_0 = const()[name = string("k_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_25_dilations_0 = const()[name = string("k_25_dilations_0"), val = tensor([1, 1])]; int32 k_25_groups_0 = const()[name = string("k_25_groups_0"), val = int32(1)]; tensor k_25 = conv(dilations = k_25_dilations_0, groups = k_25_groups_0, pad = k_25_pad_0, pad_type = k_25_pad_type_0, strides = k_25_strides_0, weight = layers_4_self_attn_k_proj_weight_palettized, x = var_3230_cast_fp16)[name = string("k_25")]; tensor var_3328 = const()[name = string("op_3328"), val = tensor([1, 2, 256, 3])]; tensor var_3329 = reshape(shape = var_3328, x = k_25)[name = string("op_3329")]; tensor transpose_61_perm_0 = const()[name = string("transpose_61_perm_0"), val = tensor([0, 3, 1, 2])]; string v_9_pad_type_0 = const()[name = string("v_9_pad_type_0"), val = string("valid")]; tensor v_9_strides_0 = const()[name = string("v_9_strides_0"), val = tensor([1, 1])]; tensor v_9_pad_0 = const()[name = string("v_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_9_dilations_0 = const()[name = string("v_9_dilations_0"), val = tensor([1, 1])]; int32 v_9_groups_0 = const()[name = string("v_9_groups_0"), val = int32(1)]; tensor v_9 = conv(dilations = v_9_dilations_0, groups = v_9_groups_0, pad = v_9_pad_0, pad_type = v_9_pad_type_0, strides = v_9_strides_0, weight = layers_4_self_attn_v_proj_weight_palettized, x = var_3230_cast_fp16)[name = string("v_9")]; tensor var_3356 = const()[name = string("op_3356"), val = tensor([1, 2, 256, 3])]; tensor var_3357 = reshape(shape = var_3356, x = v_9)[name = string("op_3357")]; tensor var_3362 = const()[name = string("op_3362"), val = tensor([0, 1, 3, 2])]; tensor var_3380 = const()[name = string("op_3380"), val = tensor([3, 2, 256])]; tensor transpose_61 = transpose(perm = transpose_61_perm_0, x = var_3329)[name = string("transpose_156")]; tensor x_83 = reshape(shape = var_3380, x = transpose_61)[name = string("x_83")]; int32 var_3386 = const()[name = string("op_3386"), val = int32(-1)]; fp16 const_51_promoted = const()[name = string("const_51_promoted"), val = fp16(-0x1p+0)]; tensor var_3388 = mul(x = x_83, y = const_51_promoted)[name = string("op_3388")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127 = concat(axis = var_3386, interleave = input_127_interleave_0, values = (x_83, var_3388))[name = string("input_127")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_3383_to_fp16 = const()[name = string("op_3383_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_3383_to_fp16, x = input_127)[name = string("normed_121_cast_fp16")]; tensor var_3393_split_sizes_0 = const()[name = string("op_3393_split_sizes_0"), val = tensor([256, 256])]; int32 var_3393_axis_0 = const()[name = string("op_3393_axis_0"), val = int32(-1)]; tensor var_3393_0, tensor var_3393_1 = split(axis = var_3393_axis_0, split_sizes = var_3393_split_sizes_0, x = normed_121_cast_fp16)[name = string("op_3393")]; tensor k_29 = mul(x = var_3393_0, y = layers_4_self_attn_k_norm_weight)[name = string("k_29")]; tensor var_3400 = const()[name = string("op_3400"), val = tensor([1, 3, 2, 256])]; tensor var_3401 = reshape(shape = var_3400, x = k_29)[name = string("op_3401")]; tensor var_3406 = const()[name = string("op_3406"), val = tensor([0, 2, 1, 3])]; fp16 var_3408_promoted = const()[name = string("op_3408_promoted"), val = fp16(0x1p+1)]; tensor var_3363 = transpose(perm = var_3362, x = var_3357)[name = string("transpose_155")]; tensor var_3409 = pow(x = var_3363, y = var_3408_promoted)[name = string("op_3409")]; tensor var_3414_axes_0 = const()[name = string("op_3414_axes_0"), val = tensor([-1])]; bool var_3414_keep_dims_0 = const()[name = string("op_3414_keep_dims_0"), val = bool(true)]; tensor var_3414 = reduce_mean(axes = var_3414_axes_0, keep_dims = var_3414_keep_dims_0, x = var_3409)[name = string("op_3414")]; fp16 var_3416_to_fp16 = const()[name = string("op_3416_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_9_cast_fp16 = add(x = var_3414, y = var_3416_to_fp16)[name = string("mean_sq_9_cast_fp16")]; fp32 var_3418_epsilon_0 = const()[name = string("op_3418_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3418_cast_fp16 = rsqrt(epsilon = var_3418_epsilon_0, x = mean_sq_9_cast_fp16)[name = string("op_3418_cast_fp16")]; tensor input_131_cast_fp16 = mul(x = var_3363, y = var_3418_cast_fp16)[name = string("input_131_cast_fp16")]; tensor q_57 = transpose(perm = var_3406, x = var_3401)[name = string("transpose_154")]; tensor var_3420_cast_fp16 = mul(x = q_57, y = cos_s)[name = string("op_3420_cast_fp16")]; tensor var_3421_split_sizes_0 = const()[name = string("op_3421_split_sizes_0"), val = tensor([128, 128])]; int32 var_3421_axis_0 = const()[name = string("op_3421_axis_0"), val = int32(-1)]; tensor var_3421_0, tensor var_3421_1 = split(axis = var_3421_axis_0, split_sizes = var_3421_split_sizes_0, x = q_57)[name = string("op_3421")]; fp16 const_52_promoted = const()[name = string("const_52_promoted"), val = fp16(-0x1p+0)]; tensor var_3423 = mul(x = var_3421_1, y = const_52_promoted)[name = string("op_3423")]; int32 var_3425 = const()[name = string("op_3425"), val = int32(-1)]; bool var_3426_interleave_0 = const()[name = string("op_3426_interleave_0"), val = bool(false)]; tensor var_3426 = concat(axis = var_3425, interleave = var_3426_interleave_0, values = (var_3423, var_3421_0))[name = string("op_3426")]; tensor var_3427_cast_fp16 = mul(x = var_3426, y = sin_s)[name = string("op_3427_cast_fp16")]; tensor input_129_cast_fp16 = add(x = var_3420_cast_fp16, y = var_3427_cast_fp16)[name = string("input_129_cast_fp16")]; tensor k_padded_9_pad_0 = const()[name = string("k_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_9_mode_0 = const()[name = string("k_padded_9_mode_0"), val = string("constant")]; fp16 const_53_to_fp16 = const()[name = string("const_53_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_9_cast_fp16 = pad(constant_val = const_53_to_fp16, mode = k_padded_9_mode_0, pad = k_padded_9_pad_0, x = input_129_cast_fp16)[name = string("k_padded_9_cast_fp16")]; tensor v_padded_9_pad_0 = const()[name = string("v_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_9_mode_0 = const()[name = string("v_padded_9_mode_0"), val = string("constant")]; fp16 const_54_to_fp16 = const()[name = string("const_54_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_9_cast_fp16 = pad(constant_val = const_54_to_fp16, mode = v_padded_9_mode_0, pad = v_padded_9_pad_0, x = input_131_cast_fp16)[name = string("v_padded_9_cast_fp16")]; tensor slot_k_9_begin_0 = const()[name = string("slot_k_9_begin_0"), val = tensor([4, 0, 0, 0])]; tensor slot_k_9_end_0 = const()[name = string("slot_k_9_end_0"), val = tensor([5, 2, 512, 512])]; tensor slot_k_9_end_mask_0 = const()[name = string("slot_k_9_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_9_cast_fp16 = slice_by_index(begin = slot_k_9_begin_0, end = slot_k_9_end_0, end_mask = slot_k_9_end_mask_0, x = K_sliding_out_7_cast_fp16)[name = string("slot_k_9_cast_fp16")]; tensor slot_v_9_begin_0 = const()[name = string("slot_v_9_begin_0"), val = tensor([4, 0, 0, 0])]; tensor slot_v_9_end_0 = const()[name = string("slot_v_9_end_0"), val = tensor([5, 2, 512, 512])]; tensor slot_v_9_end_mask_0 = const()[name = string("slot_v_9_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_9_cast_fp16 = slice_by_index(begin = slot_v_9_begin_0, end = slot_v_9_end_0, end_mask = slot_v_9_end_mask_0, x = V_sliding_out_7_cast_fp16)[name = string("slot_v_9_cast_fp16")]; tensor var_3466_begin_0 = const()[name = string("op_3466_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_3466_end_0 = const()[name = string("op_3466_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_3466_end_mask_0 = const()[name = string("op_3466_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3466_cast_fp16 = slice_by_index(begin = var_3466_begin_0, end = var_3466_end_0, end_mask = var_3466_end_mask_0, x = slot_k_9_cast_fp16)[name = string("op_3466_cast_fp16")]; int32 var_3473 = const()[name = string("op_3473"), val = int32(2)]; bool new_k_9_interleave_0 = const()[name = string("new_k_9_interleave_0"), val = bool(false)]; tensor new_k_9_cast_fp16 = concat(axis = var_3473, interleave = new_k_9_interleave_0, values = (var_3466_cast_fp16, k_padded_9_cast_fp16))[name = string("new_k_9_cast_fp16")]; tensor var_3489_begin_0 = const()[name = string("op_3489_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_3489_end_0 = const()[name = string("op_3489_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_3489_end_mask_0 = const()[name = string("op_3489_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3489_cast_fp16 = slice_by_index(begin = var_3489_begin_0, end = var_3489_end_0, end_mask = var_3489_end_mask_0, x = slot_v_9_cast_fp16)[name = string("op_3489_cast_fp16")]; int32 var_3496 = const()[name = string("op_3496"), val = int32(2)]; bool new_v_9_interleave_0 = const()[name = string("new_v_9_interleave_0"), val = bool(false)]; tensor new_v_9_cast_fp16 = concat(axis = var_3496, interleave = new_v_9_interleave_0, values = (var_3489_cast_fp16, v_padded_9_cast_fp16))[name = string("new_v_9_cast_fp16")]; tensor var_3502_begin_0 = const()[name = string("op_3502_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3502_end_0 = const()[name = string("op_3502_end_0"), val = tensor([4, 2, 512, 512])]; tensor var_3502_end_mask_0 = const()[name = string("op_3502_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3502_cast_fp16 = slice_by_index(begin = var_3502_begin_0, end = var_3502_end_0, end_mask = var_3502_end_mask_0, x = K_sliding_out_7_cast_fp16)[name = string("op_3502_cast_fp16")]; tensor var_3507_begin_0 = const()[name = string("op_3507_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_3507_end_0 = const()[name = string("op_3507_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_3507_end_mask_0 = const()[name = string("op_3507_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3507_cast_fp16 = slice_by_index(begin = var_3507_begin_0, end = var_3507_end_0, end_mask = var_3507_end_mask_0, x = K_sliding_out_7_cast_fp16)[name = string("op_3507_cast_fp16")]; int32 var_3509 = const()[name = string("op_3509"), val = int32(0)]; bool K_sliding_out_9_interleave_0 = const()[name = string("K_sliding_out_9_interleave_0"), val = bool(false)]; tensor K_sliding_out_9_cast_fp16 = concat(axis = var_3509, interleave = K_sliding_out_9_interleave_0, values = (var_3502_cast_fp16, new_k_9_cast_fp16, var_3507_cast_fp16))[name = string("K_sliding_out_9_cast_fp16")]; tensor var_3515_begin_0 = const()[name = string("op_3515_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3515_end_0 = const()[name = string("op_3515_end_0"), val = tensor([4, 2, 512, 512])]; tensor var_3515_end_mask_0 = const()[name = string("op_3515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3515_cast_fp16 = slice_by_index(begin = var_3515_begin_0, end = var_3515_end_0, end_mask = var_3515_end_mask_0, x = V_sliding_out_7_cast_fp16)[name = string("op_3515_cast_fp16")]; tensor var_3520_begin_0 = const()[name = string("op_3520_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_3520_end_0 = const()[name = string("op_3520_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_3520_end_mask_0 = const()[name = string("op_3520_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3520_cast_fp16 = slice_by_index(begin = var_3520_begin_0, end = var_3520_end_0, end_mask = var_3520_end_mask_0, x = V_sliding_out_7_cast_fp16)[name = string("op_3520_cast_fp16")]; int32 var_3522 = const()[name = string("op_3522"), val = int32(0)]; bool V_sliding_out_9_interleave_0 = const()[name = string("V_sliding_out_9_interleave_0"), val = bool(false)]; tensor V_sliding_out_9_cast_fp16 = concat(axis = var_3522, interleave = V_sliding_out_9_interleave_0, values = (var_3515_cast_fp16, new_v_9_cast_fp16, var_3520_cast_fp16))[name = string("V_sliding_out_9_cast_fp16")]; tensor var_3528_begin_0 = const()[name = string("op_3528_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_3528_end_0 = const()[name = string("op_3528_end_0"), val = tensor([5, 2, 512, 512])]; tensor var_3528_end_mask_0 = const()[name = string("op_3528_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3528_cast_fp16 = slice_by_index(begin = var_3528_begin_0, end = var_3528_end_0, end_mask = var_3528_end_mask_0, x = K_sliding_out_9_cast_fp16)[name = string("op_3528_cast_fp16")]; tensor K_for_attn_9_begin_0 = const()[name = string("K_for_attn_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_9_end_0 = const()[name = string("K_for_attn_9_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_9_end_mask_0 = const()[name = string("K_for_attn_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_9_cast_fp16 = slice_by_index(begin = K_for_attn_9_begin_0, end = K_for_attn_9_end_0, end_mask = K_for_attn_9_end_mask_0, x = var_3528_cast_fp16)[name = string("K_for_attn_9_cast_fp16")]; tensor var_3538_begin_0 = const()[name = string("op_3538_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_3538_end_0 = const()[name = string("op_3538_end_0"), val = tensor([5, 2, 512, 512])]; tensor var_3538_end_mask_0 = const()[name = string("op_3538_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3538_cast_fp16 = slice_by_index(begin = var_3538_begin_0, end = var_3538_end_0, end_mask = var_3538_end_mask_0, x = V_sliding_out_9_cast_fp16)[name = string("op_3538_cast_fp16")]; tensor V_for_attn_9_begin_0 = const()[name = string("V_for_attn_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_9_end_0 = const()[name = string("V_for_attn_9_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_9_end_mask_0 = const()[name = string("V_for_attn_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_9_cast_fp16 = slice_by_index(begin = V_for_attn_9_begin_0, end = V_for_attn_9_end_0, end_mask = V_for_attn_9_end_mask_0, x = var_3538_cast_fp16)[name = string("V_for_attn_9_cast_fp16")]; tensor transpose_16_perm_0 = const()[name = string("transpose_16_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_8_reps_0 = const()[name = string("tile_8_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_16_cast_fp16 = transpose(perm = transpose_16_perm_0, x = K_for_attn_9_cast_fp16)[name = string("transpose_153")]; tensor tile_8_cast_fp16 = tile(reps = tile_8_reps_0, x = transpose_16_cast_fp16)[name = string("tile_8_cast_fp16")]; tensor concat_16 = const()[name = string("concat_16"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_16, x = tile_8_cast_fp16)[name = string("reshape_16_cast_fp16")]; tensor transpose_17_perm_0 = const()[name = string("transpose_17_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_17 = const()[name = string("concat_17"), val = tensor([-1, 1, 512, 256])]; tensor transpose_17_cast_fp16 = transpose(perm = transpose_17_perm_0, x = reshape_16_cast_fp16)[name = string("transpose_152")]; tensor reshape_17_cast_fp16 = reshape(shape = concat_17, x = transpose_17_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor transpose_62_perm_0 = const()[name = string("transpose_62_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_18_perm_0 = const()[name = string("transpose_18_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_9_reps_0 = const()[name = string("tile_9_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_18_cast_fp16 = transpose(perm = transpose_18_perm_0, x = V_for_attn_9_cast_fp16)[name = string("transpose_151")]; tensor tile_9_cast_fp16 = tile(reps = tile_9_reps_0, x = transpose_18_cast_fp16)[name = string("tile_9_cast_fp16")]; tensor concat_18 = const()[name = string("concat_18"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_18, x = tile_9_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor transpose_19_perm_0 = const()[name = string("transpose_19_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_19 = const()[name = string("concat_19"), val = tensor([-1, 1, 512, 256])]; tensor transpose_19_cast_fp16 = transpose(perm = transpose_19_perm_0, x = reshape_18_cast_fp16)[name = string("transpose_150")]; tensor reshape_19_cast_fp16 = reshape(shape = concat_19, x = transpose_19_cast_fp16)[name = string("reshape_19_cast_fp16")]; tensor V_expanded_9_perm_0 = const()[name = string("V_expanded_9_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor transpose_62_cast_fp16 = transpose(perm = transpose_62_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_149")]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = q_59_cast_fp16, y = transpose_62_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; tensor x_87_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask_sliding)[name = string("x_87_cast_fp16")]; tensor reduce_max_4_axes_0 = const()[name = string("reduce_max_4_axes_0"), val = tensor([-1])]; bool reduce_max_4_keep_dims_0 = const()[name = string("reduce_max_4_keep_dims_0"), val = bool(true)]; tensor reduce_max_4 = reduce_max(axes = reduce_max_4_axes_0, keep_dims = reduce_max_4_keep_dims_0, x = x_87_cast_fp16)[name = string("reduce_max_4")]; tensor var_3573 = sub(x = x_87_cast_fp16, y = reduce_max_4)[name = string("op_3573")]; tensor var_3579 = exp(x = var_3573)[name = string("op_3579")]; tensor var_3589_axes_0 = const()[name = string("op_3589_axes_0"), val = tensor([-1])]; bool var_3589_keep_dims_0 = const()[name = string("op_3589_keep_dims_0"), val = bool(true)]; tensor var_3589 = reduce_sum(axes = var_3589_axes_0, keep_dims = var_3589_keep_dims_0, x = var_3579)[name = string("op_3589")]; tensor var_3595_cast_fp16 = real_div(x = var_3579, y = var_3589)[name = string("op_3595_cast_fp16")]; bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; tensor V_expanded_9_cast_fp16 = transpose(perm = V_expanded_9_perm_0, x = reshape_19_cast_fp16)[name = string("transpose_148")]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = var_3595_cast_fp16, y = V_expanded_9_cast_fp16)[name = string("attn_output_25_cast_fp16")]; tensor var_3606 = const()[name = string("op_3606"), val = tensor([0, 2, 1, 3])]; tensor var_3613 = const()[name = string("op_3613"), val = tensor([1, 3, -1])]; tensor var_3607_cast_fp16 = transpose(perm = var_3606, x = attn_output_25_cast_fp16)[name = string("transpose_147")]; tensor attn_output_27_cast_fp16 = reshape(shape = var_3613, x = var_3607_cast_fp16)[name = string("attn_output_27_cast_fp16")]; tensor var_3618 = const()[name = string("op_3618"), val = tensor([0, 2, 1])]; string var_3634_pad_type_0 = const()[name = string("op_3634_pad_type_0"), val = string("valid")]; int32 var_3634_groups_0 = const()[name = string("op_3634_groups_0"), val = int32(1)]; tensor var_3634_strides_0 = const()[name = string("op_3634_strides_0"), val = tensor([1])]; tensor var_3634_pad_0 = const()[name = string("op_3634_pad_0"), val = tensor([0, 0])]; tensor var_3634_dilations_0 = const()[name = string("op_3634_dilations_0"), val = tensor([1])]; tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(543157440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545778944))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3619_cast_fp16 = transpose(perm = var_3618, x = attn_output_27_cast_fp16)[name = string("transpose_146")]; tensor var_3634_cast_fp16 = conv(dilations = var_3634_dilations_0, groups = var_3634_groups_0, pad = var_3634_pad_0, pad_type = var_3634_pad_type_0, strides = var_3634_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3619_cast_fp16)[name = string("op_3634_cast_fp16")]; tensor var_3638 = const()[name = string("op_3638"), val = tensor([0, 2, 1])]; int32 var_3644 = const()[name = string("op_3644"), val = int32(-1)]; fp16 const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_91_cast_fp16 = transpose(perm = var_3638, x = var_3634_cast_fp16)[name = string("transpose_145")]; tensor var_3646_cast_fp16 = mul(x = x_91_cast_fp16, y = const_55_promoted_to_fp16)[name = string("op_3646_cast_fp16")]; bool input_135_interleave_0 = const()[name = string("input_135_interleave_0"), val = bool(false)]; tensor input_135_cast_fp16 = concat(axis = var_3644, interleave = input_135_interleave_0, values = (x_91_cast_fp16, var_3646_cast_fp16))[name = string("input_135_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_3641_to_fp16 = const()[name = string("op_3641_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_3641_to_fp16, x = input_135_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor var_3651_split_sizes_0 = const()[name = string("op_3651_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3651_axis_0 = const()[name = string("op_3651_axis_0"), val = int32(-1)]; tensor var_3651_cast_fp16_0, tensor var_3651_cast_fp16_1 = split(axis = var_3651_axis_0, split_sizes = var_3651_split_sizes_0, x = normed_125_cast_fp16)[name = string("op_3651_cast_fp16")]; tensor layers_4_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545781568)))]; tensor attn_output_29_cast_fp16 = mul(x = var_3651_cast_fp16_0, y = layers_4_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_29_cast_fp16")]; tensor x_93_cast_fp16 = add(x = x_79_cast_fp16, y = attn_output_29_cast_fp16)[name = string("x_93_cast_fp16")]; int32 var_3660 = const()[name = string("op_3660"), val = int32(-1)]; fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3662_cast_fp16 = mul(x = x_93_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_3662_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_3660, interleave = input_137_interleave_0, values = (x_93_cast_fp16, var_3662_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_3657_to_fp16 = const()[name = string("op_3657_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_3657_to_fp16, x = input_137_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor var_3667_split_sizes_0 = const()[name = string("op_3667_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3667_axis_0 = const()[name = string("op_3667_axis_0"), val = int32(-1)]; tensor var_3667_cast_fp16_0, tensor var_3667_cast_fp16_1 = split(axis = var_3667_axis_0, split_sizes = var_3667_split_sizes_0, x = normed_129_cast_fp16)[name = string("op_3667_cast_fp16")]; tensor layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545786752)))]; tensor h_27_cast_fp16 = mul(x = var_3667_cast_fp16_0, y = layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_27_cast_fp16")]; tensor var_3678 = const()[name = string("op_3678"), val = tensor([0, 2, 1])]; tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; tensor var_3679 = transpose(perm = var_3678, x = h_27_cast_fp16)[name = string("transpose_144")]; tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_3679)[name = string("input_139")]; string gate_17_pad_type_0 = const()[name = string("gate_17_pad_type_0"), val = string("valid")]; tensor gate_17_strides_0 = const()[name = string("gate_17_strides_0"), val = tensor([1, 1])]; tensor gate_17_pad_0 = const()[name = string("gate_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_17_dilations_0 = const()[name = string("gate_17_dilations_0"), val = tensor([1, 1])]; int32 gate_17_groups_0 = const()[name = string("gate_17_groups_0"), val = int32(1)]; tensor gate_17 = conv(dilations = gate_17_dilations_0, groups = gate_17_groups_0, pad = gate_17_pad_0, pad_type = gate_17_pad_type_0, strides = gate_17_strides_0, weight = layers_4_mlp_gate_proj_weight_palettized, x = input_139)[name = string("gate_17")]; string up_9_pad_type_0 = const()[name = string("up_9_pad_type_0"), val = string("valid")]; tensor up_9_strides_0 = const()[name = string("up_9_strides_0"), val = tensor([1, 1])]; tensor up_9_pad_0 = const()[name = string("up_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_9_dilations_0 = const()[name = string("up_9_dilations_0"), val = tensor([1, 1])]; int32 up_9_groups_0 = const()[name = string("up_9_groups_0"), val = int32(1)]; tensor up_9 = conv(dilations = up_9_dilations_0, groups = up_9_groups_0, pad = up_9_pad_0, pad_type = up_9_pad_type_0, strides = up_9_strides_0, weight = layers_4_mlp_up_proj_weight_palettized, x = input_139)[name = string("up_9")]; string gate_19_mode_0 = const()[name = string("gate_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_19 = gelu(mode = gate_19_mode_0, x = gate_17)[name = string("gate_19")]; tensor input_141 = mul(x = gate_19, y = up_9)[name = string("input_141")]; string mlp_out_9_pad_type_0 = const()[name = string("mlp_out_9_pad_type_0"), val = string("valid")]; tensor mlp_out_9_strides_0 = const()[name = string("mlp_out_9_strides_0"), val = tensor([1, 1])]; tensor mlp_out_9_pad_0 = const()[name = string("mlp_out_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_9_dilations_0 = const()[name = string("mlp_out_9_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_9_groups_0 = const()[name = string("mlp_out_9_groups_0"), val = int32(1)]; tensor mlp_out_9 = conv(dilations = mlp_out_9_dilations_0, groups = mlp_out_9_groups_0, pad = mlp_out_9_pad_0, pad_type = mlp_out_9_pad_type_0, strides = mlp_out_9_strides_0, weight = layers_4_mlp_down_proj_weight_palettized, x = input_141)[name = string("mlp_out_9")]; tensor var_3719_axes_0 = const()[name = string("op_3719_axes_0"), val = tensor([2])]; tensor var_3719 = squeeze(axes = var_3719_axes_0, x = mlp_out_9)[name = string("op_3719")]; tensor var_3723 = const()[name = string("op_3723"), val = tensor([0, 2, 1])]; int32 var_3729 = const()[name = string("op_3729"), val = int32(-1)]; fp16 const_57_promoted = const()[name = string("const_57_promoted"), val = fp16(-0x1p+0)]; tensor x_95 = transpose(perm = var_3723, x = var_3719)[name = string("transpose_143")]; tensor var_3731 = mul(x = x_95, y = const_57_promoted)[name = string("op_3731")]; bool input_143_interleave_0 = const()[name = string("input_143_interleave_0"), val = bool(false)]; tensor input_143 = concat(axis = var_3729, interleave = input_143_interleave_0, values = (x_95, var_3731))[name = string("input_143")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_3726_to_fp16 = const()[name = string("op_3726_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_3726_to_fp16, x = input_143)[name = string("normed_133_cast_fp16")]; tensor var_3736_split_sizes_0 = const()[name = string("op_3736_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3736_axis_0 = const()[name = string("op_3736_axis_0"), val = int32(-1)]; tensor var_3736_0, tensor var_3736_1 = split(axis = var_3736_axis_0, split_sizes = var_3736_split_sizes_0, x = normed_133_cast_fp16)[name = string("op_3736")]; tensor hidden_states_43 = mul(x = var_3736_0, y = layers_4_post_feedforward_layernorm_weight)[name = string("hidden_states_43")]; tensor hidden_states_45_cast_fp16 = add(x = x_93_cast_fp16, y = hidden_states_43)[name = string("hidden_states_45_cast_fp16")]; tensor per_layer_slice_9_begin_0 = const()[name = string("per_layer_slice_9_begin_0"), val = tensor([0, 0, 4096])]; tensor per_layer_slice_9_end_0 = const()[name = string("per_layer_slice_9_end_0"), val = tensor([1, 3, 4352])]; tensor per_layer_slice_9_end_mask_0 = const()[name = string("per_layer_slice_9_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_9_cast_fp16 = slice_by_index(begin = per_layer_slice_9_begin_0, end = per_layer_slice_9_end_0, end_mask = per_layer_slice_9_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_9_cast_fp16")]; tensor var_3764 = const()[name = string("op_3764"), val = tensor([0, 2, 1])]; tensor input_145_axes_0 = const()[name = string("input_145_axes_0"), val = tensor([2])]; tensor var_3765 = transpose(perm = var_3764, x = hidden_states_45_cast_fp16)[name = string("transpose_142")]; tensor input_145 = expand_dims(axes = input_145_axes_0, x = var_3765)[name = string("input_145")]; string gated_25_pad_type_0 = const()[name = string("gated_25_pad_type_0"), val = string("valid")]; tensor gated_25_strides_0 = const()[name = string("gated_25_strides_0"), val = tensor([1, 1])]; tensor gated_25_pad_0 = const()[name = string("gated_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_25_dilations_0 = const()[name = string("gated_25_dilations_0"), val = tensor([1, 1])]; int32 gated_25_groups_0 = const()[name = string("gated_25_groups_0"), val = int32(1)]; tensor gated_25 = conv(dilations = gated_25_dilations_0, groups = gated_25_groups_0, pad = gated_25_pad_0, pad_type = gated_25_pad_type_0, strides = gated_25_strides_0, weight = layers_4_per_layer_input_gate_weight_palettized, x = input_145)[name = string("gated_25")]; string gated_27_mode_0 = const()[name = string("gated_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_27 = gelu(mode = gated_27_mode_0, x = gated_25)[name = string("gated_27")]; tensor var_3784 = const()[name = string("op_3784"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_9_axes_0 = const()[name = string("per_layer_slice_conv_9_axes_0"), val = tensor([2])]; tensor var_3785_cast_fp16 = transpose(perm = var_3784, x = per_layer_slice_9_cast_fp16)[name = string("transpose_141")]; tensor per_layer_slice_conv_9_cast_fp16 = expand_dims(axes = per_layer_slice_conv_9_axes_0, x = var_3785_cast_fp16)[name = string("per_layer_slice_conv_9_cast_fp16")]; tensor input_147_cast_fp16 = mul(x = gated_27, y = per_layer_slice_conv_9_cast_fp16)[name = string("input_147_cast_fp16")]; string gated_29_pad_type_0 = const()[name = string("gated_29_pad_type_0"), val = string("valid")]; tensor gated_29_strides_0 = const()[name = string("gated_29_strides_0"), val = tensor([1, 1])]; tensor gated_29_pad_0 = const()[name = string("gated_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_29_dilations_0 = const()[name = string("gated_29_dilations_0"), val = tensor([1, 1])]; int32 gated_29_groups_0 = const()[name = string("gated_29_groups_0"), val = int32(1)]; tensor layers_4_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(545791936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546119680))))[name = string("layers_4_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_29_cast_fp16 = conv(dilations = gated_29_dilations_0, groups = gated_29_groups_0, pad = gated_29_pad_0, pad_type = gated_29_pad_type_0, strides = gated_29_strides_0, weight = layers_4_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_147_cast_fp16)[name = string("gated_29_cast_fp16")]; tensor var_3801_axes_0 = const()[name = string("op_3801_axes_0"), val = tensor([2])]; tensor var_3801_cast_fp16 = squeeze(axes = var_3801_axes_0, x = gated_29_cast_fp16)[name = string("op_3801_cast_fp16")]; tensor var_3805 = const()[name = string("op_3805"), val = tensor([0, 2, 1])]; int32 var_3811 = const()[name = string("op_3811"), val = int32(-1)]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_97_cast_fp16 = transpose(perm = var_3805, x = var_3801_cast_fp16)[name = string("transpose_140")]; tensor var_3813_cast_fp16 = mul(x = x_97_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_3813_cast_fp16")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149_cast_fp16 = concat(axis = var_3811, interleave = input_149_interleave_0, values = (x_97_cast_fp16, var_3813_cast_fp16))[name = string("input_149_cast_fp16")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_3808_to_fp16 = const()[name = string("op_3808_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_3808_to_fp16, x = input_149_cast_fp16)[name = string("normed_137_cast_fp16")]; tensor var_3818_split_sizes_0 = const()[name = string("op_3818_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3818_axis_0 = const()[name = string("op_3818_axis_0"), val = int32(-1)]; tensor var_3818_cast_fp16_0, tensor var_3818_cast_fp16_1 = split(axis = var_3818_axis_0, split_sizes = var_3818_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_3818_cast_fp16")]; tensor layers_4_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546122304)))]; tensor hidden_states_49_cast_fp16 = mul(x = var_3818_cast_fp16_0, y = layers_4_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_49_cast_fp16")]; tensor hidden_states_51_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_49_cast_fp16)[name = string("hidden_states_51_cast_fp16")]; tensor const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = tensor([0x1.46p-1])]; tensor x_99_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_59_promoted_to_fp16)[name = string("x_99_cast_fp16")]; int32 var_3833 = const()[name = string("op_3833"), val = int32(-1)]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3835_cast_fp16 = mul(x = x_99_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_3835_cast_fp16")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151_cast_fp16 = concat(axis = var_3833, interleave = input_151_interleave_0, values = (x_99_cast_fp16, var_3835_cast_fp16))[name = string("input_151_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_3830_to_fp16 = const()[name = string("op_3830_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_3830_to_fp16, x = input_151_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_3840_split_sizes_0 = const()[name = string("op_3840_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_3840_axis_0 = const()[name = string("op_3840_axis_0"), val = int32(-1)]; tensor var_3840_cast_fp16_0, tensor var_3840_cast_fp16_1 = split(axis = var_3840_axis_0, split_sizes = var_3840_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_3840_cast_fp16")]; tensor layers_5_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546127488)))]; tensor h_31_cast_fp16 = mul(x = var_3840_cast_fp16_0, y = layers_5_input_layernorm_weight_promoted_to_fp16)[name = string("h_31_cast_fp16")]; tensor var_3846 = const()[name = string("op_3846"), val = tensor([0, 2, 1])]; tensor var_3849_axes_0 = const()[name = string("op_3849_axes_0"), val = tensor([2])]; tensor var_3847_cast_fp16 = transpose(perm = var_3846, x = h_31_cast_fp16)[name = string("transpose_139")]; tensor var_3849_cast_fp16 = expand_dims(axes = var_3849_axes_0, x = var_3847_cast_fp16)[name = string("op_3849_cast_fp16")]; string q_61_pad_type_0 = const()[name = string("q_61_pad_type_0"), val = string("valid")]; tensor q_61_strides_0 = const()[name = string("q_61_strides_0"), val = tensor([1, 1])]; tensor q_61_pad_0 = const()[name = string("q_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_61_dilations_0 = const()[name = string("q_61_dilations_0"), val = tensor([1, 1])]; int32 q_61_groups_0 = const()[name = string("q_61_groups_0"), val = int32(1)]; tensor q_61 = conv(dilations = q_61_dilations_0, groups = q_61_groups_0, pad = q_61_pad_0, pad_type = q_61_pad_type_0, strides = q_61_strides_0, weight = layers_5_self_attn_q_proj_weight_palettized, x = var_3849_cast_fp16)[name = string("q_61")]; tensor var_3870 = const()[name = string("op_3870"), val = tensor([1, 8, 512, 3])]; tensor var_3871 = reshape(shape = var_3870, x = q_61)[name = string("op_3871")]; tensor transpose_63_perm_0 = const()[name = string("transpose_63_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_3894 = const()[name = string("op_3894"), val = tensor([3, 8, 512])]; tensor transpose_63 = transpose(perm = transpose_63_perm_0, x = var_3871)[name = string("transpose_138")]; tensor x_101 = reshape(shape = var_3894, x = transpose_63)[name = string("x_101")]; int32 var_3900 = const()[name = string("op_3900"), val = int32(-1)]; fp16 const_61_promoted = const()[name = string("const_61_promoted"), val = fp16(-0x1p+0)]; tensor var_3902 = mul(x = x_101, y = const_61_promoted)[name = string("op_3902")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155 = concat(axis = var_3900, interleave = input_155_interleave_0, values = (x_101, var_3902))[name = string("input_155")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_3897_to_fp16 = const()[name = string("op_3897_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_3897_to_fp16, x = input_155)[name = string("normed_145_cast_fp16")]; tensor var_3907_split_sizes_0 = const()[name = string("op_3907_split_sizes_0"), val = tensor([512, 512])]; int32 var_3907_axis_0 = const()[name = string("op_3907_axis_0"), val = int32(-1)]; tensor var_3907_0, tensor var_3907_1 = split(axis = var_3907_axis_0, split_sizes = var_3907_split_sizes_0, x = normed_145_cast_fp16)[name = string("op_3907")]; tensor q_65 = mul(x = var_3907_0, y = layers_5_self_attn_q_norm_weight)[name = string("q_65")]; tensor var_3914 = const()[name = string("op_3914"), val = tensor([1, 3, 8, 512])]; tensor var_3915 = reshape(shape = var_3914, x = q_65)[name = string("op_3915")]; tensor var_3920 = const()[name = string("op_3920"), val = tensor([0, 2, 1, 3])]; tensor q_67 = transpose(perm = var_3920, x = var_3915)[name = string("transpose_137")]; tensor var_3922_cast_fp16 = mul(x = q_67, y = cos_f)[name = string("op_3922_cast_fp16")]; tensor var_3923_split_sizes_0 = const()[name = string("op_3923_split_sizes_0"), val = tensor([256, 256])]; int32 var_3923_axis_0 = const()[name = string("op_3923_axis_0"), val = int32(-1)]; tensor var_3923_0, tensor var_3923_1 = split(axis = var_3923_axis_0, split_sizes = var_3923_split_sizes_0, x = q_67)[name = string("op_3923")]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_3925 = mul(x = var_3923_1, y = const_62_promoted)[name = string("op_3925")]; int32 var_3927 = const()[name = string("op_3927"), val = int32(-1)]; bool var_3928_interleave_0 = const()[name = string("op_3928_interleave_0"), val = bool(false)]; tensor var_3928 = concat(axis = var_3927, interleave = var_3928_interleave_0, values = (var_3925, var_3923_0))[name = string("op_3928")]; tensor var_3929_cast_fp16 = mul(x = var_3928, y = sin_f)[name = string("op_3929_cast_fp16")]; tensor q_71_cast_fp16 = add(x = var_3922_cast_fp16, y = var_3929_cast_fp16)[name = string("q_71_cast_fp16")]; string k_31_pad_type_0 = const()[name = string("k_31_pad_type_0"), val = string("valid")]; tensor k_31_strides_0 = const()[name = string("k_31_strides_0"), val = tensor([1, 1])]; tensor k_31_pad_0 = const()[name = string("k_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_31_dilations_0 = const()[name = string("k_31_dilations_0"), val = tensor([1, 1])]; int32 k_31_groups_0 = const()[name = string("k_31_groups_0"), val = int32(1)]; tensor k_31 = conv(dilations = k_31_dilations_0, groups = k_31_groups_0, pad = k_31_pad_0, pad_type = k_31_pad_type_0, strides = k_31_strides_0, weight = layers_5_self_attn_k_proj_weight_palettized, x = var_3849_cast_fp16)[name = string("k_31")]; tensor var_3947 = const()[name = string("op_3947"), val = tensor([1, 2, 512, 3])]; tensor var_3948 = reshape(shape = var_3947, x = k_31)[name = string("op_3948")]; tensor transpose_64_perm_0 = const()[name = string("transpose_64_perm_0"), val = tensor([0, 3, 1, 2])]; string v_11_pad_type_0 = const()[name = string("v_11_pad_type_0"), val = string("valid")]; tensor v_11_strides_0 = const()[name = string("v_11_strides_0"), val = tensor([1, 1])]; tensor v_11_pad_0 = const()[name = string("v_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_11_dilations_0 = const()[name = string("v_11_dilations_0"), val = tensor([1, 1])]; int32 v_11_groups_0 = const()[name = string("v_11_groups_0"), val = int32(1)]; tensor v_11 = conv(dilations = v_11_dilations_0, groups = v_11_groups_0, pad = v_11_pad_0, pad_type = v_11_pad_type_0, strides = v_11_strides_0, weight = layers_5_self_attn_v_proj_weight_palettized, x = var_3849_cast_fp16)[name = string("v_11")]; tensor var_3975 = const()[name = string("op_3975"), val = tensor([1, 2, 512, 3])]; tensor var_3976 = reshape(shape = var_3975, x = v_11)[name = string("op_3976")]; tensor var_3981 = const()[name = string("op_3981"), val = tensor([0, 1, 3, 2])]; tensor var_3999 = const()[name = string("op_3999"), val = tensor([3, 2, 512])]; tensor transpose_64 = transpose(perm = transpose_64_perm_0, x = var_3948)[name = string("transpose_136")]; tensor x_103 = reshape(shape = var_3999, x = transpose_64)[name = string("x_103")]; int32 var_4005 = const()[name = string("op_4005"), val = int32(-1)]; fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; tensor var_4007 = mul(x = x_103, y = const_63_promoted)[name = string("op_4007")]; bool input_157_interleave_0 = const()[name = string("input_157_interleave_0"), val = bool(false)]; tensor input_157 = concat(axis = var_4005, interleave = input_157_interleave_0, values = (x_103, var_4007))[name = string("input_157")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_4002_to_fp16 = const()[name = string("op_4002_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_4002_to_fp16, x = input_157)[name = string("normed_149_cast_fp16")]; tensor var_4012_split_sizes_0 = const()[name = string("op_4012_split_sizes_0"), val = tensor([512, 512])]; int32 var_4012_axis_0 = const()[name = string("op_4012_axis_0"), val = int32(-1)]; tensor var_4012_0, tensor var_4012_1 = split(axis = var_4012_axis_0, split_sizes = var_4012_split_sizes_0, x = normed_149_cast_fp16)[name = string("op_4012")]; tensor k_35 = mul(x = var_4012_0, y = layers_5_self_attn_k_norm_weight)[name = string("k_35")]; tensor var_4019 = const()[name = string("op_4019"), val = tensor([1, 3, 2, 512])]; tensor var_4020 = reshape(shape = var_4019, x = k_35)[name = string("op_4020")]; tensor var_4025 = const()[name = string("op_4025"), val = tensor([0, 2, 1, 3])]; fp16 var_4027_promoted = const()[name = string("op_4027_promoted"), val = fp16(0x1p+1)]; tensor var_3982 = transpose(perm = var_3981, x = var_3976)[name = string("transpose_135")]; tensor var_4028 = pow(x = var_3982, y = var_4027_promoted)[name = string("op_4028")]; tensor var_4033_axes_0 = const()[name = string("op_4033_axes_0"), val = tensor([-1])]; bool var_4033_keep_dims_0 = const()[name = string("op_4033_keep_dims_0"), val = bool(true)]; tensor var_4033 = reduce_mean(axes = var_4033_axes_0, keep_dims = var_4033_keep_dims_0, x = var_4028)[name = string("op_4033")]; fp16 var_4035_to_fp16 = const()[name = string("op_4035_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_11_cast_fp16 = add(x = var_4033, y = var_4035_to_fp16)[name = string("mean_sq_11_cast_fp16")]; fp32 var_4037_epsilon_0 = const()[name = string("op_4037_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_4037_cast_fp16 = rsqrt(epsilon = var_4037_epsilon_0, x = mean_sq_11_cast_fp16)[name = string("op_4037_cast_fp16")]; tensor v_13_cast_fp16 = mul(x = var_3982, y = var_4037_cast_fp16)[name = string("v_13_cast_fp16")]; tensor q_69 = transpose(perm = var_4025, x = var_4020)[name = string("transpose_134")]; tensor var_4039_cast_fp16 = mul(x = q_69, y = cos_f)[name = string("op_4039_cast_fp16")]; tensor var_4040_split_sizes_0 = const()[name = string("op_4040_split_sizes_0"), val = tensor([256, 256])]; int32 var_4040_axis_0 = const()[name = string("op_4040_axis_0"), val = int32(-1)]; tensor var_4040_0, tensor var_4040_1 = split(axis = var_4040_axis_0, split_sizes = var_4040_split_sizes_0, x = q_69)[name = string("op_4040")]; fp16 const_64_promoted = const()[name = string("const_64_promoted"), val = fp16(-0x1p+0)]; tensor var_4042 = mul(x = var_4040_1, y = const_64_promoted)[name = string("op_4042")]; int32 var_4044 = const()[name = string("op_4044"), val = int32(-1)]; bool var_4045_interleave_0 = const()[name = string("op_4045_interleave_0"), val = bool(false)]; tensor var_4045 = concat(axis = var_4044, interleave = var_4045_interleave_0, values = (var_4042, var_4040_0))[name = string("op_4045")]; tensor var_4046_cast_fp16 = mul(x = var_4045, y = sin_f)[name = string("op_4046_cast_fp16")]; tensor k_37_cast_fp16 = add(x = var_4039_cast_fp16, y = var_4046_cast_fp16)[name = string("k_37_cast_fp16")]; tensor var_4055_reps_0 = const()[name = string("op_4055_reps_0"), val = tensor([1, 2, 1, 1])]; tensor var_4055_cast_fp16 = tile(reps = var_4055_reps_0, x = update_indicator)[name = string("op_4055_cast_fp16")]; bool k_scattered_1_transpose_x_0 = const()[name = string("k_scattered_1_transpose_x_0"), val = bool(false)]; bool k_scattered_1_transpose_y_0 = const()[name = string("k_scattered_1_transpose_y_0"), val = bool(false)]; tensor k_scattered_1_cast_fp16 = matmul(transpose_x = k_scattered_1_transpose_x_0, transpose_y = k_scattered_1_transpose_y_0, x = var_4055_cast_fp16, y = k_37_cast_fp16)[name = string("k_scattered_1_cast_fp16")]; bool v_scattered_1_transpose_x_0 = const()[name = string("v_scattered_1_transpose_x_0"), val = bool(false)]; bool v_scattered_1_transpose_y_0 = const()[name = string("v_scattered_1_transpose_y_0"), val = bool(false)]; tensor v_scattered_1_cast_fp16 = matmul(transpose_x = v_scattered_1_transpose_x_0, transpose_y = v_scattered_1_transpose_y_0, x = var_4055_cast_fp16, y = v_13_cast_fp16)[name = string("v_scattered_1_cast_fp16")]; tensor var_4069_axes_0 = const()[name = string("op_4069_axes_0"), val = tensor([-1])]; bool var_4069_keep_dims_0 = const()[name = string("op_4069_keep_dims_0"), val = bool(true)]; tensor var_4069_cast_fp16 = reduce_sum(axes = var_4069_axes_0, keep_dims = var_4069_keep_dims_0, x = update_indicator)[name = string("op_4069_cast_fp16")]; tensor slot_k_11_begin_0 = const()[name = string("slot_k_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slot_k_11_end_0 = const()[name = string("slot_k_11_end_0"), val = tensor([1, 2, 2048, 512])]; tensor slot_k_11_end_mask_0 = const()[name = string("slot_k_11_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_11_cast_fp16 = slice_by_index(begin = slot_k_11_begin_0, end = slot_k_11_end_0, end_mask = slot_k_11_end_mask_0, x = K_full_in)[name = string("slot_k_11_cast_fp16")]; tensor slot_v_11_begin_0 = const()[name = string("slot_v_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor slot_v_11_end_0 = const()[name = string("slot_v_11_end_0"), val = tensor([1, 2, 2048, 512])]; tensor slot_v_11_end_mask_0 = const()[name = string("slot_v_11_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_11_cast_fp16 = slice_by_index(begin = slot_v_11_begin_0, end = slot_v_11_end_0, end_mask = slot_v_11_end_mask_0, x = V_full_in)[name = string("slot_v_11_cast_fp16")]; fp16 var_4080_promoted_to_fp16 = const()[name = string("op_4080_promoted_to_fp16"), val = fp16(0x1p+0)]; tensor var_4082_cast_fp16 = sub(x = var_4080_promoted_to_fp16, y = var_4069_cast_fp16)[name = string("op_4082_cast_fp16")]; tensor var_4083_cast_fp16 = mul(x = slot_k_11_cast_fp16, y = var_4082_cast_fp16)[name = string("op_4083_cast_fp16")]; tensor new_k_11_cast_fp16 = add(x = var_4083_cast_fp16, y = k_scattered_1_cast_fp16)[name = string("new_k_11_cast_fp16")]; tensor var_4089_cast_fp16 = mul(x = slot_v_11_cast_fp16, y = var_4082_cast_fp16)[name = string("op_4089_cast_fp16")]; tensor new_v_11_cast_fp16 = add(x = var_4089_cast_fp16, y = v_scattered_1_cast_fp16)[name = string("new_v_11_cast_fp16")]; tensor var_4101_begin_0 = const()[name = string("op_4101_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_4101_end_0 = const()[name = string("op_4101_end_0"), val = tensor([2, 2, 2048, 512])]; tensor var_4101_end_mask_0 = const()[name = string("op_4101_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4101_cast_fp16 = slice_by_index(begin = var_4101_begin_0, end = var_4101_end_0, end_mask = var_4101_end_mask_0, x = K_full_in)[name = string("op_4101_cast_fp16")]; int32 var_4103 = const()[name = string("op_4103"), val = int32(0)]; bool K_full_out_1_interleave_0 = const()[name = string("K_full_out_1_interleave_0"), val = bool(false)]; tensor K_full_out_1_cast_fp16 = concat(axis = var_4103, interleave = K_full_out_1_interleave_0, values = (new_k_11_cast_fp16, var_4101_cast_fp16))[name = string("K_full_out_1_cast_fp16")]; tensor var_4114_begin_0 = const()[name = string("op_4114_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_4114_end_0 = const()[name = string("op_4114_end_0"), val = tensor([2, 2, 2048, 512])]; tensor var_4114_end_mask_0 = const()[name = string("op_4114_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4114_cast_fp16 = slice_by_index(begin = var_4114_begin_0, end = var_4114_end_0, end_mask = var_4114_end_mask_0, x = V_full_in)[name = string("op_4114_cast_fp16")]; int32 var_4116 = const()[name = string("op_4116"), val = int32(0)]; bool V_full_out_1_interleave_0 = const()[name = string("V_full_out_1_interleave_0"), val = bool(false)]; tensor V_full_out_1_cast_fp16 = concat(axis = var_4116, interleave = V_full_out_1_interleave_0, values = (new_v_11_cast_fp16, var_4114_cast_fp16))[name = string("V_full_out_1_cast_fp16")]; tensor var_4122_begin_0 = const()[name = string("op_4122_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4122_end_0 = const()[name = string("op_4122_end_0"), val = tensor([1, 2, 2048, 512])]; tensor var_4122_end_mask_0 = const()[name = string("op_4122_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4122_cast_fp16 = slice_by_index(begin = var_4122_begin_0, end = var_4122_end_0, end_mask = var_4122_end_mask_0, x = K_full_out_1_cast_fp16)[name = string("op_4122_cast_fp16")]; tensor var_4132_begin_0 = const()[name = string("op_4132_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4132_end_0 = const()[name = string("op_4132_end_0"), val = tensor([1, 2, 2048, 512])]; tensor var_4132_end_mask_0 = const()[name = string("op_4132_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4132_cast_fp16 = slice_by_index(begin = var_4132_begin_0, end = var_4132_end_0, end_mask = var_4132_end_mask_0, x = V_full_out_1_cast_fp16)[name = string("op_4132_cast_fp16")]; tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = var_4122_cast_fp16)[name = string("transpose_133")]; tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; tensor concat_22 = const()[name = string("concat_22"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_22, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_23 = const()[name = string("concat_23"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_132")]; tensor reshape_21_cast_fp16 = reshape(shape = concat_23, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor transpose_65_perm_0 = const()[name = string("transpose_65_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = var_4132_cast_fp16)[name = string("transpose_131")]; tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; tensor concat_24 = const()[name = string("concat_24"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_24, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_25 = const()[name = string("concat_25"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_130")]; tensor reshape_23_cast_fp16 = reshape(shape = concat_25, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor V_expanded_11_perm_0 = const()[name = string("V_expanded_11_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_21_transpose_x_0 = const()[name = string("attn_weights_21_transpose_x_0"), val = bool(false)]; bool attn_weights_21_transpose_y_0 = const()[name = string("attn_weights_21_transpose_y_0"), val = bool(false)]; tensor transpose_65_cast_fp16 = transpose(perm = transpose_65_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_129")]; tensor attn_weights_21_cast_fp16 = matmul(transpose_x = attn_weights_21_transpose_x_0, transpose_y = attn_weights_21_transpose_y_0, x = q_71_cast_fp16, y = transpose_65_cast_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor x_107_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask_full)[name = string("x_107_cast_fp16")]; tensor reduce_max_5_axes_0 = const()[name = string("reduce_max_5_axes_0"), val = tensor([-1])]; bool reduce_max_5_keep_dims_0 = const()[name = string("reduce_max_5_keep_dims_0"), val = bool(true)]; tensor reduce_max_5 = reduce_max(axes = reduce_max_5_axes_0, keep_dims = reduce_max_5_keep_dims_0, x = x_107_cast_fp16)[name = string("reduce_max_5")]; tensor var_4167 = sub(x = x_107_cast_fp16, y = reduce_max_5)[name = string("op_4167")]; tensor var_4173 = exp(x = var_4167)[name = string("op_4173")]; tensor var_4183_axes_0 = const()[name = string("op_4183_axes_0"), val = tensor([-1])]; bool var_4183_keep_dims_0 = const()[name = string("op_4183_keep_dims_0"), val = bool(true)]; tensor var_4183 = reduce_sum(axes = var_4183_axes_0, keep_dims = var_4183_keep_dims_0, x = var_4173)[name = string("op_4183")]; tensor var_4189_cast_fp16 = real_div(x = var_4173, y = var_4183)[name = string("op_4189_cast_fp16")]; bool attn_output_31_transpose_x_0 = const()[name = string("attn_output_31_transpose_x_0"), val = bool(false)]; bool attn_output_31_transpose_y_0 = const()[name = string("attn_output_31_transpose_y_0"), val = bool(false)]; tensor V_expanded_11_cast_fp16 = transpose(perm = V_expanded_11_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_128")]; tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = var_4189_cast_fp16, y = V_expanded_11_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor var_4200 = const()[name = string("op_4200"), val = tensor([0, 2, 1, 3])]; tensor var_4207 = const()[name = string("op_4207"), val = tensor([1, 3, -1])]; tensor var_4201_cast_fp16 = transpose(perm = var_4200, x = attn_output_31_cast_fp16)[name = string("transpose_127")]; tensor attn_output_33_cast_fp16 = reshape(shape = var_4207, x = var_4201_cast_fp16)[name = string("attn_output_33_cast_fp16")]; tensor var_4212 = const()[name = string("op_4212"), val = tensor([0, 2, 1])]; string var_4228_pad_type_0 = const()[name = string("op_4228_pad_type_0"), val = string("valid")]; int32 var_4228_groups_0 = const()[name = string("op_4228_groups_0"), val = int32(1)]; tensor var_4228_strides_0 = const()[name = string("op_4228_strides_0"), val = tensor([1])]; tensor var_4228_pad_0 = const()[name = string("op_4228_pad_0"), val = tensor([0, 0])]; tensor var_4228_dilations_0 = const()[name = string("op_4228_dilations_0"), val = tensor([1])]; tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(546132672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551375616))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4213_cast_fp16 = transpose(perm = var_4212, x = attn_output_33_cast_fp16)[name = string("transpose_126")]; tensor var_4228_cast_fp16 = conv(dilations = var_4228_dilations_0, groups = var_4228_groups_0, pad = var_4228_pad_0, pad_type = var_4228_pad_type_0, strides = var_4228_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_4213_cast_fp16)[name = string("op_4228_cast_fp16")]; tensor var_4232 = const()[name = string("op_4232"), val = tensor([0, 2, 1])]; int32 var_4238 = const()[name = string("op_4238"), val = int32(-1)]; fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_111_cast_fp16 = transpose(perm = var_4232, x = var_4228_cast_fp16)[name = string("transpose_125")]; tensor var_4240_cast_fp16 = mul(x = x_111_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_4240_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_4238, interleave = input_161_interleave_0, values = (x_111_cast_fp16, var_4240_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_4235_to_fp16 = const()[name = string("op_4235_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_4235_to_fp16, x = input_161_cast_fp16)[name = string("normed_153_cast_fp16")]; tensor var_4245_split_sizes_0 = const()[name = string("op_4245_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4245_axis_0 = const()[name = string("op_4245_axis_0"), val = int32(-1)]; tensor var_4245_cast_fp16_0, tensor var_4245_cast_fp16_1 = split(axis = var_4245_axis_0, split_sizes = var_4245_split_sizes_0, x = normed_153_cast_fp16)[name = string("op_4245_cast_fp16")]; tensor layers_5_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551378240)))]; tensor attn_output_35_cast_fp16 = mul(x = var_4245_cast_fp16_0, y = layers_5_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_35_cast_fp16")]; tensor x_113_cast_fp16 = add(x = x_99_cast_fp16, y = attn_output_35_cast_fp16)[name = string("x_113_cast_fp16")]; int32 var_4254 = const()[name = string("op_4254"), val = int32(-1)]; fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4256_cast_fp16 = mul(x = x_113_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_4256_cast_fp16")]; bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; tensor input_163_cast_fp16 = concat(axis = var_4254, interleave = input_163_interleave_0, values = (x_113_cast_fp16, var_4256_cast_fp16))[name = string("input_163_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_4251_to_fp16 = const()[name = string("op_4251_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_4251_to_fp16, x = input_163_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor var_4261_split_sizes_0 = const()[name = string("op_4261_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4261_axis_0 = const()[name = string("op_4261_axis_0"), val = int32(-1)]; tensor var_4261_cast_fp16_0, tensor var_4261_cast_fp16_1 = split(axis = var_4261_axis_0, split_sizes = var_4261_split_sizes_0, x = normed_157_cast_fp16)[name = string("op_4261_cast_fp16")]; tensor layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551383424)))]; tensor h_33_cast_fp16 = mul(x = var_4261_cast_fp16_0, y = layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_33_cast_fp16")]; tensor var_4272 = const()[name = string("op_4272"), val = tensor([0, 2, 1])]; tensor input_165_axes_0 = const()[name = string("input_165_axes_0"), val = tensor([2])]; tensor var_4273 = transpose(perm = var_4272, x = h_33_cast_fp16)[name = string("transpose_124")]; tensor input_165 = expand_dims(axes = input_165_axes_0, x = var_4273)[name = string("input_165")]; string gate_21_pad_type_0 = const()[name = string("gate_21_pad_type_0"), val = string("valid")]; tensor gate_21_strides_0 = const()[name = string("gate_21_strides_0"), val = tensor([1, 1])]; tensor gate_21_pad_0 = const()[name = string("gate_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_21_dilations_0 = const()[name = string("gate_21_dilations_0"), val = tensor([1, 1])]; int32 gate_21_groups_0 = const()[name = string("gate_21_groups_0"), val = int32(1)]; tensor gate_21 = conv(dilations = gate_21_dilations_0, groups = gate_21_groups_0, pad = gate_21_pad_0, pad_type = gate_21_pad_type_0, strides = gate_21_strides_0, weight = layers_5_mlp_gate_proj_weight_palettized, x = input_165)[name = string("gate_21")]; string up_11_pad_type_0 = const()[name = string("up_11_pad_type_0"), val = string("valid")]; tensor up_11_strides_0 = const()[name = string("up_11_strides_0"), val = tensor([1, 1])]; tensor up_11_pad_0 = const()[name = string("up_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_11_dilations_0 = const()[name = string("up_11_dilations_0"), val = tensor([1, 1])]; int32 up_11_groups_0 = const()[name = string("up_11_groups_0"), val = int32(1)]; tensor up_11 = conv(dilations = up_11_dilations_0, groups = up_11_groups_0, pad = up_11_pad_0, pad_type = up_11_pad_type_0, strides = up_11_strides_0, weight = layers_5_mlp_up_proj_weight_palettized, x = input_165)[name = string("up_11")]; string gate_23_mode_0 = const()[name = string("gate_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_23 = gelu(mode = gate_23_mode_0, x = gate_21)[name = string("gate_23")]; tensor input_167 = mul(x = gate_23, y = up_11)[name = string("input_167")]; string mlp_out_11_pad_type_0 = const()[name = string("mlp_out_11_pad_type_0"), val = string("valid")]; tensor mlp_out_11_strides_0 = const()[name = string("mlp_out_11_strides_0"), val = tensor([1, 1])]; tensor mlp_out_11_pad_0 = const()[name = string("mlp_out_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_11_dilations_0 = const()[name = string("mlp_out_11_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_11_groups_0 = const()[name = string("mlp_out_11_groups_0"), val = int32(1)]; tensor mlp_out_11 = conv(dilations = mlp_out_11_dilations_0, groups = mlp_out_11_groups_0, pad = mlp_out_11_pad_0, pad_type = mlp_out_11_pad_type_0, strides = mlp_out_11_strides_0, weight = layers_5_mlp_down_proj_weight_palettized, x = input_167)[name = string("mlp_out_11")]; tensor var_4313_axes_0 = const()[name = string("op_4313_axes_0"), val = tensor([2])]; tensor var_4313 = squeeze(axes = var_4313_axes_0, x = mlp_out_11)[name = string("op_4313")]; tensor var_4317 = const()[name = string("op_4317"), val = tensor([0, 2, 1])]; int32 var_4323 = const()[name = string("op_4323"), val = int32(-1)]; fp16 const_67_promoted = const()[name = string("const_67_promoted"), val = fp16(-0x1p+0)]; tensor x_115 = transpose(perm = var_4317, x = var_4313)[name = string("transpose_123")]; tensor var_4325 = mul(x = x_115, y = const_67_promoted)[name = string("op_4325")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169 = concat(axis = var_4323, interleave = input_169_interleave_0, values = (x_115, var_4325))[name = string("input_169")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_4320_to_fp16 = const()[name = string("op_4320_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_4320_to_fp16, x = input_169)[name = string("normed_161_cast_fp16")]; tensor var_4330_split_sizes_0 = const()[name = string("op_4330_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4330_axis_0 = const()[name = string("op_4330_axis_0"), val = int32(-1)]; tensor var_4330_0, tensor var_4330_1 = split(axis = var_4330_axis_0, split_sizes = var_4330_split_sizes_0, x = normed_161_cast_fp16)[name = string("op_4330")]; tensor hidden_states_53 = mul(x = var_4330_0, y = layers_5_post_feedforward_layernorm_weight)[name = string("hidden_states_53")]; tensor hidden_states_55_cast_fp16 = add(x = x_113_cast_fp16, y = hidden_states_53)[name = string("hidden_states_55_cast_fp16")]; tensor per_layer_slice_11_begin_0 = const()[name = string("per_layer_slice_11_begin_0"), val = tensor([0, 0, 4352])]; tensor per_layer_slice_11_end_0 = const()[name = string("per_layer_slice_11_end_0"), val = tensor([1, 3, 4608])]; tensor per_layer_slice_11_end_mask_0 = const()[name = string("per_layer_slice_11_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_11_cast_fp16 = slice_by_index(begin = per_layer_slice_11_begin_0, end = per_layer_slice_11_end_0, end_mask = per_layer_slice_11_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_11_cast_fp16")]; tensor var_4358 = const()[name = string("op_4358"), val = tensor([0, 2, 1])]; tensor input_171_axes_0 = const()[name = string("input_171_axes_0"), val = tensor([2])]; tensor var_4359 = transpose(perm = var_4358, x = hidden_states_55_cast_fp16)[name = string("transpose_122")]; tensor input_171 = expand_dims(axes = input_171_axes_0, x = var_4359)[name = string("input_171")]; string gated_31_pad_type_0 = const()[name = string("gated_31_pad_type_0"), val = string("valid")]; tensor gated_31_strides_0 = const()[name = string("gated_31_strides_0"), val = tensor([1, 1])]; tensor gated_31_pad_0 = const()[name = string("gated_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_31_dilations_0 = const()[name = string("gated_31_dilations_0"), val = tensor([1, 1])]; int32 gated_31_groups_0 = const()[name = string("gated_31_groups_0"), val = int32(1)]; tensor gated_31 = conv(dilations = gated_31_dilations_0, groups = gated_31_groups_0, pad = gated_31_pad_0, pad_type = gated_31_pad_type_0, strides = gated_31_strides_0, weight = layers_5_per_layer_input_gate_weight_palettized, x = input_171)[name = string("gated_31")]; string gated_33_mode_0 = const()[name = string("gated_33_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_33 = gelu(mode = gated_33_mode_0, x = gated_31)[name = string("gated_33")]; tensor var_4378 = const()[name = string("op_4378"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_11_axes_0 = const()[name = string("per_layer_slice_conv_11_axes_0"), val = tensor([2])]; tensor var_4379_cast_fp16 = transpose(perm = var_4378, x = per_layer_slice_11_cast_fp16)[name = string("transpose_121")]; tensor per_layer_slice_conv_11_cast_fp16 = expand_dims(axes = per_layer_slice_conv_11_axes_0, x = var_4379_cast_fp16)[name = string("per_layer_slice_conv_11_cast_fp16")]; tensor input_173_cast_fp16 = mul(x = gated_33, y = per_layer_slice_conv_11_cast_fp16)[name = string("input_173_cast_fp16")]; string gated_35_pad_type_0 = const()[name = string("gated_35_pad_type_0"), val = string("valid")]; tensor gated_35_strides_0 = const()[name = string("gated_35_strides_0"), val = tensor([1, 1])]; tensor gated_35_pad_0 = const()[name = string("gated_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_35_dilations_0 = const()[name = string("gated_35_dilations_0"), val = tensor([1, 1])]; int32 gated_35_groups_0 = const()[name = string("gated_35_groups_0"), val = int32(1)]; tensor layers_5_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551388608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551716352))))[name = string("layers_5_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_35_cast_fp16 = conv(dilations = gated_35_dilations_0, groups = gated_35_groups_0, pad = gated_35_pad_0, pad_type = gated_35_pad_type_0, strides = gated_35_strides_0, weight = layers_5_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_173_cast_fp16)[name = string("gated_35_cast_fp16")]; tensor var_4395_axes_0 = const()[name = string("op_4395_axes_0"), val = tensor([2])]; tensor var_4395_cast_fp16 = squeeze(axes = var_4395_axes_0, x = gated_35_cast_fp16)[name = string("op_4395_cast_fp16")]; tensor var_4399 = const()[name = string("op_4399"), val = tensor([0, 2, 1])]; int32 var_4405 = const()[name = string("op_4405"), val = int32(-1)]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_117_cast_fp16 = transpose(perm = var_4399, x = var_4395_cast_fp16)[name = string("transpose_120")]; tensor var_4407_cast_fp16 = mul(x = x_117_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_4407_cast_fp16")]; bool input_175_interleave_0 = const()[name = string("input_175_interleave_0"), val = bool(false)]; tensor input_175_cast_fp16 = concat(axis = var_4405, interleave = input_175_interleave_0, values = (x_117_cast_fp16, var_4407_cast_fp16))[name = string("input_175_cast_fp16")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_4402_to_fp16 = const()[name = string("op_4402_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_4402_to_fp16, x = input_175_cast_fp16)[name = string("normed_165_cast_fp16")]; tensor var_4412_split_sizes_0 = const()[name = string("op_4412_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4412_axis_0 = const()[name = string("op_4412_axis_0"), val = int32(-1)]; tensor var_4412_cast_fp16_0, tensor var_4412_cast_fp16_1 = split(axis = var_4412_axis_0, split_sizes = var_4412_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_4412_cast_fp16")]; tensor layers_5_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551718976)))]; tensor hidden_states_59_cast_fp16 = mul(x = var_4412_cast_fp16_0, y = layers_5_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_59_cast_fp16")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_59_cast_fp16)[name = string("hidden_states_61_cast_fp16")]; tensor const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = tensor([0x1.b2p-2])]; tensor x_119_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_69_promoted_to_fp16)[name = string("x_119_cast_fp16")]; int32 var_4427 = const()[name = string("op_4427"), val = int32(-1)]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4429_cast_fp16 = mul(x = x_119_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_4429_cast_fp16")]; bool input_177_interleave_0 = const()[name = string("input_177_interleave_0"), val = bool(false)]; tensor input_177_cast_fp16 = concat(axis = var_4427, interleave = input_177_interleave_0, values = (x_119_cast_fp16, var_4429_cast_fp16))[name = string("input_177_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_4424_to_fp16 = const()[name = string("op_4424_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_4424_to_fp16, x = input_177_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_4434_split_sizes_0 = const()[name = string("op_4434_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4434_axis_0 = const()[name = string("op_4434_axis_0"), val = int32(-1)]; tensor var_4434_cast_fp16_0, tensor var_4434_cast_fp16_1 = split(axis = var_4434_axis_0, split_sizes = var_4434_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_4434_cast_fp16")]; tensor layers_6_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551724160)))]; tensor h_37_cast_fp16 = mul(x = var_4434_cast_fp16_0, y = layers_6_input_layernorm_weight_promoted_to_fp16)[name = string("h_37_cast_fp16")]; tensor var_4440 = const()[name = string("op_4440"), val = tensor([0, 2, 1])]; tensor var_4443_axes_0 = const()[name = string("op_4443_axes_0"), val = tensor([2])]; tensor var_4441_cast_fp16 = transpose(perm = var_4440, x = h_37_cast_fp16)[name = string("transpose_119")]; tensor var_4443_cast_fp16 = expand_dims(axes = var_4443_axes_0, x = var_4441_cast_fp16)[name = string("op_4443_cast_fp16")]; string q_73_pad_type_0 = const()[name = string("q_73_pad_type_0"), val = string("valid")]; tensor q_73_strides_0 = const()[name = string("q_73_strides_0"), val = tensor([1, 1])]; tensor q_73_pad_0 = const()[name = string("q_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_73_dilations_0 = const()[name = string("q_73_dilations_0"), val = tensor([1, 1])]; int32 q_73_groups_0 = const()[name = string("q_73_groups_0"), val = int32(1)]; tensor q_73 = conv(dilations = q_73_dilations_0, groups = q_73_groups_0, pad = q_73_pad_0, pad_type = q_73_pad_type_0, strides = q_73_strides_0, weight = layers_6_self_attn_q_proj_weight_palettized, x = var_4443_cast_fp16)[name = string("q_73")]; tensor var_4464 = const()[name = string("op_4464"), val = tensor([1, 8, 256, 3])]; tensor var_4465 = reshape(shape = var_4464, x = q_73)[name = string("op_4465")]; tensor transpose_66_perm_0 = const()[name = string("transpose_66_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_4488 = const()[name = string("op_4488"), val = tensor([3, 8, 256])]; tensor transpose_66 = transpose(perm = transpose_66_perm_0, x = var_4465)[name = string("transpose_118")]; tensor x_121 = reshape(shape = var_4488, x = transpose_66)[name = string("x_121")]; int32 var_4494 = const()[name = string("op_4494"), val = int32(-1)]; fp16 const_71_promoted = const()[name = string("const_71_promoted"), val = fp16(-0x1p+0)]; tensor var_4496 = mul(x = x_121, y = const_71_promoted)[name = string("op_4496")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181 = concat(axis = var_4494, interleave = input_181_interleave_0, values = (x_121, var_4496))[name = string("input_181")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_4491_to_fp16 = const()[name = string("op_4491_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_4491_to_fp16, x = input_181)[name = string("normed_173_cast_fp16")]; tensor var_4501_split_sizes_0 = const()[name = string("op_4501_split_sizes_0"), val = tensor([256, 256])]; int32 var_4501_axis_0 = const()[name = string("op_4501_axis_0"), val = int32(-1)]; tensor var_4501_0, tensor var_4501_1 = split(axis = var_4501_axis_0, split_sizes = var_4501_split_sizes_0, x = normed_173_cast_fp16)[name = string("op_4501")]; tensor q_77 = mul(x = var_4501_0, y = layers_2_self_attn_q_norm_weight)[name = string("q_77")]; tensor var_4508 = const()[name = string("op_4508"), val = tensor([1, 3, 8, 256])]; tensor var_4509 = reshape(shape = var_4508, x = q_77)[name = string("op_4509")]; tensor var_4514 = const()[name = string("op_4514"), val = tensor([0, 2, 1, 3])]; tensor q_79 = transpose(perm = var_4514, x = var_4509)[name = string("transpose_117")]; tensor var_4516_cast_fp16 = mul(x = q_79, y = cos_s)[name = string("op_4516_cast_fp16")]; tensor var_4517_split_sizes_0 = const()[name = string("op_4517_split_sizes_0"), val = tensor([128, 128])]; int32 var_4517_axis_0 = const()[name = string("op_4517_axis_0"), val = int32(-1)]; tensor var_4517_0, tensor var_4517_1 = split(axis = var_4517_axis_0, split_sizes = var_4517_split_sizes_0, x = q_79)[name = string("op_4517")]; fp16 const_72_promoted = const()[name = string("const_72_promoted"), val = fp16(-0x1p+0)]; tensor var_4519 = mul(x = var_4517_1, y = const_72_promoted)[name = string("op_4519")]; int32 var_4521 = const()[name = string("op_4521"), val = int32(-1)]; bool var_4522_interleave_0 = const()[name = string("op_4522_interleave_0"), val = bool(false)]; tensor var_4522 = concat(axis = var_4521, interleave = var_4522_interleave_0, values = (var_4519, var_4517_0))[name = string("op_4522")]; tensor var_4523_cast_fp16 = mul(x = var_4522, y = sin_s)[name = string("op_4523_cast_fp16")]; tensor q_83_cast_fp16 = add(x = var_4516_cast_fp16, y = var_4523_cast_fp16)[name = string("q_83_cast_fp16")]; string k_39_pad_type_0 = const()[name = string("k_39_pad_type_0"), val = string("valid")]; tensor k_39_strides_0 = const()[name = string("k_39_strides_0"), val = tensor([1, 1])]; tensor k_39_pad_0 = const()[name = string("k_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_39_dilations_0 = const()[name = string("k_39_dilations_0"), val = tensor([1, 1])]; int32 k_39_groups_0 = const()[name = string("k_39_groups_0"), val = int32(1)]; tensor k_39 = conv(dilations = k_39_dilations_0, groups = k_39_groups_0, pad = k_39_pad_0, pad_type = k_39_pad_type_0, strides = k_39_strides_0, weight = layers_6_self_attn_k_proj_weight_palettized, x = var_4443_cast_fp16)[name = string("k_39")]; tensor var_4541 = const()[name = string("op_4541"), val = tensor([1, 2, 256, 3])]; tensor var_4542 = reshape(shape = var_4541, x = k_39)[name = string("op_4542")]; tensor transpose_67_perm_0 = const()[name = string("transpose_67_perm_0"), val = tensor([0, 3, 1, 2])]; string v_15_pad_type_0 = const()[name = string("v_15_pad_type_0"), val = string("valid")]; tensor v_15_strides_0 = const()[name = string("v_15_strides_0"), val = tensor([1, 1])]; tensor v_15_pad_0 = const()[name = string("v_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_15_dilations_0 = const()[name = string("v_15_dilations_0"), val = tensor([1, 1])]; int32 v_15_groups_0 = const()[name = string("v_15_groups_0"), val = int32(1)]; tensor v_15 = conv(dilations = v_15_dilations_0, groups = v_15_groups_0, pad = v_15_pad_0, pad_type = v_15_pad_type_0, strides = v_15_strides_0, weight = layers_6_self_attn_v_proj_weight_palettized, x = var_4443_cast_fp16)[name = string("v_15")]; tensor var_4569 = const()[name = string("op_4569"), val = tensor([1, 2, 256, 3])]; tensor var_4570 = reshape(shape = var_4569, x = v_15)[name = string("op_4570")]; tensor var_4575 = const()[name = string("op_4575"), val = tensor([0, 1, 3, 2])]; tensor var_4593 = const()[name = string("op_4593"), val = tensor([3, 2, 256])]; tensor transpose_67 = transpose(perm = transpose_67_perm_0, x = var_4542)[name = string("transpose_116")]; tensor x_123 = reshape(shape = var_4593, x = transpose_67)[name = string("x_123")]; int32 var_4599 = const()[name = string("op_4599"), val = int32(-1)]; fp16 const_73_promoted = const()[name = string("const_73_promoted"), val = fp16(-0x1p+0)]; tensor var_4601 = mul(x = x_123, y = const_73_promoted)[name = string("op_4601")]; bool input_183_interleave_0 = const()[name = string("input_183_interleave_0"), val = bool(false)]; tensor input_183 = concat(axis = var_4599, interleave = input_183_interleave_0, values = (x_123, var_4601))[name = string("input_183")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_4596_to_fp16 = const()[name = string("op_4596_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_4596_to_fp16, x = input_183)[name = string("normed_177_cast_fp16")]; tensor var_4606_split_sizes_0 = const()[name = string("op_4606_split_sizes_0"), val = tensor([256, 256])]; int32 var_4606_axis_0 = const()[name = string("op_4606_axis_0"), val = int32(-1)]; tensor var_4606_0, tensor var_4606_1 = split(axis = var_4606_axis_0, split_sizes = var_4606_split_sizes_0, x = normed_177_cast_fp16)[name = string("op_4606")]; tensor k_43 = mul(x = var_4606_0, y = layers_6_self_attn_k_norm_weight)[name = string("k_43")]; tensor var_4613 = const()[name = string("op_4613"), val = tensor([1, 3, 2, 256])]; tensor var_4614 = reshape(shape = var_4613, x = k_43)[name = string("op_4614")]; tensor var_4619 = const()[name = string("op_4619"), val = tensor([0, 2, 1, 3])]; fp16 var_4621_promoted = const()[name = string("op_4621_promoted"), val = fp16(0x1p+1)]; tensor var_4576 = transpose(perm = var_4575, x = var_4570)[name = string("transpose_115")]; tensor var_4622 = pow(x = var_4576, y = var_4621_promoted)[name = string("op_4622")]; tensor var_4627_axes_0 = const()[name = string("op_4627_axes_0"), val = tensor([-1])]; bool var_4627_keep_dims_0 = const()[name = string("op_4627_keep_dims_0"), val = bool(true)]; tensor var_4627 = reduce_mean(axes = var_4627_axes_0, keep_dims = var_4627_keep_dims_0, x = var_4622)[name = string("op_4627")]; fp16 var_4629_to_fp16 = const()[name = string("op_4629_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_13_cast_fp16 = add(x = var_4627, y = var_4629_to_fp16)[name = string("mean_sq_13_cast_fp16")]; fp32 var_4631_epsilon_0 = const()[name = string("op_4631_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_4631_cast_fp16 = rsqrt(epsilon = var_4631_epsilon_0, x = mean_sq_13_cast_fp16)[name = string("op_4631_cast_fp16")]; tensor input_187_cast_fp16 = mul(x = var_4576, y = var_4631_cast_fp16)[name = string("input_187_cast_fp16")]; tensor q_81 = transpose(perm = var_4619, x = var_4614)[name = string("transpose_114")]; tensor var_4633_cast_fp16 = mul(x = q_81, y = cos_s)[name = string("op_4633_cast_fp16")]; tensor var_4634_split_sizes_0 = const()[name = string("op_4634_split_sizes_0"), val = tensor([128, 128])]; int32 var_4634_axis_0 = const()[name = string("op_4634_axis_0"), val = int32(-1)]; tensor var_4634_0, tensor var_4634_1 = split(axis = var_4634_axis_0, split_sizes = var_4634_split_sizes_0, x = q_81)[name = string("op_4634")]; fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; tensor var_4636 = mul(x = var_4634_1, y = const_74_promoted)[name = string("op_4636")]; int32 var_4638 = const()[name = string("op_4638"), val = int32(-1)]; bool var_4639_interleave_0 = const()[name = string("op_4639_interleave_0"), val = bool(false)]; tensor var_4639 = concat(axis = var_4638, interleave = var_4639_interleave_0, values = (var_4636, var_4634_0))[name = string("op_4639")]; tensor var_4640_cast_fp16 = mul(x = var_4639, y = sin_s)[name = string("op_4640_cast_fp16")]; tensor input_185_cast_fp16 = add(x = var_4633_cast_fp16, y = var_4640_cast_fp16)[name = string("input_185_cast_fp16")]; tensor k_padded_11_pad_0 = const()[name = string("k_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_11_mode_0 = const()[name = string("k_padded_11_mode_0"), val = string("constant")]; fp16 const_75_to_fp16 = const()[name = string("const_75_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_11_cast_fp16 = pad(constant_val = const_75_to_fp16, mode = k_padded_11_mode_0, pad = k_padded_11_pad_0, x = input_185_cast_fp16)[name = string("k_padded_11_cast_fp16")]; tensor v_padded_11_pad_0 = const()[name = string("v_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_11_mode_0 = const()[name = string("v_padded_11_mode_0"), val = string("constant")]; fp16 const_76_to_fp16 = const()[name = string("const_76_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_11_cast_fp16 = pad(constant_val = const_76_to_fp16, mode = v_padded_11_mode_0, pad = v_padded_11_pad_0, x = input_187_cast_fp16)[name = string("v_padded_11_cast_fp16")]; tensor slot_k_13_begin_0 = const()[name = string("slot_k_13_begin_0"), val = tensor([5, 0, 0, 0])]; tensor slot_k_13_end_0 = const()[name = string("slot_k_13_end_0"), val = tensor([6, 2, 512, 512])]; tensor slot_k_13_end_mask_0 = const()[name = string("slot_k_13_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_13_cast_fp16 = slice_by_index(begin = slot_k_13_begin_0, end = slot_k_13_end_0, end_mask = slot_k_13_end_mask_0, x = K_sliding_out_9_cast_fp16)[name = string("slot_k_13_cast_fp16")]; tensor slot_v_13_begin_0 = const()[name = string("slot_v_13_begin_0"), val = tensor([5, 0, 0, 0])]; tensor slot_v_13_end_0 = const()[name = string("slot_v_13_end_0"), val = tensor([6, 2, 512, 512])]; tensor slot_v_13_end_mask_0 = const()[name = string("slot_v_13_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_13_cast_fp16 = slice_by_index(begin = slot_v_13_begin_0, end = slot_v_13_end_0, end_mask = slot_v_13_end_mask_0, x = V_sliding_out_9_cast_fp16)[name = string("slot_v_13_cast_fp16")]; tensor var_4679_begin_0 = const()[name = string("op_4679_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_4679_end_0 = const()[name = string("op_4679_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_4679_end_mask_0 = const()[name = string("op_4679_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4679_cast_fp16 = slice_by_index(begin = var_4679_begin_0, end = var_4679_end_0, end_mask = var_4679_end_mask_0, x = slot_k_13_cast_fp16)[name = string("op_4679_cast_fp16")]; int32 var_4686 = const()[name = string("op_4686"), val = int32(2)]; bool new_k_13_interleave_0 = const()[name = string("new_k_13_interleave_0"), val = bool(false)]; tensor new_k_13_cast_fp16 = concat(axis = var_4686, interleave = new_k_13_interleave_0, values = (var_4679_cast_fp16, k_padded_11_cast_fp16))[name = string("new_k_13_cast_fp16")]; tensor var_4702_begin_0 = const()[name = string("op_4702_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_4702_end_0 = const()[name = string("op_4702_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_4702_end_mask_0 = const()[name = string("op_4702_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4702_cast_fp16 = slice_by_index(begin = var_4702_begin_0, end = var_4702_end_0, end_mask = var_4702_end_mask_0, x = slot_v_13_cast_fp16)[name = string("op_4702_cast_fp16")]; int32 var_4709 = const()[name = string("op_4709"), val = int32(2)]; bool new_v_13_interleave_0 = const()[name = string("new_v_13_interleave_0"), val = bool(false)]; tensor new_v_13_cast_fp16 = concat(axis = var_4709, interleave = new_v_13_interleave_0, values = (var_4702_cast_fp16, v_padded_11_cast_fp16))[name = string("new_v_13_cast_fp16")]; tensor var_4715_begin_0 = const()[name = string("op_4715_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4715_end_0 = const()[name = string("op_4715_end_0"), val = tensor([5, 2, 512, 512])]; tensor var_4715_end_mask_0 = const()[name = string("op_4715_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4715_cast_fp16 = slice_by_index(begin = var_4715_begin_0, end = var_4715_end_0, end_mask = var_4715_end_mask_0, x = K_sliding_out_9_cast_fp16)[name = string("op_4715_cast_fp16")]; tensor var_4720_begin_0 = const()[name = string("op_4720_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_4720_end_0 = const()[name = string("op_4720_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_4720_end_mask_0 = const()[name = string("op_4720_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4720_cast_fp16 = slice_by_index(begin = var_4720_begin_0, end = var_4720_end_0, end_mask = var_4720_end_mask_0, x = K_sliding_out_9_cast_fp16)[name = string("op_4720_cast_fp16")]; int32 var_4722 = const()[name = string("op_4722"), val = int32(0)]; bool K_sliding_out_11_interleave_0 = const()[name = string("K_sliding_out_11_interleave_0"), val = bool(false)]; tensor K_sliding_out_11_cast_fp16 = concat(axis = var_4722, interleave = K_sliding_out_11_interleave_0, values = (var_4715_cast_fp16, new_k_13_cast_fp16, var_4720_cast_fp16))[name = string("K_sliding_out_11_cast_fp16")]; tensor var_4728_begin_0 = const()[name = string("op_4728_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4728_end_0 = const()[name = string("op_4728_end_0"), val = tensor([5, 2, 512, 512])]; tensor var_4728_end_mask_0 = const()[name = string("op_4728_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4728_cast_fp16 = slice_by_index(begin = var_4728_begin_0, end = var_4728_end_0, end_mask = var_4728_end_mask_0, x = V_sliding_out_9_cast_fp16)[name = string("op_4728_cast_fp16")]; tensor var_4733_begin_0 = const()[name = string("op_4733_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_4733_end_0 = const()[name = string("op_4733_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_4733_end_mask_0 = const()[name = string("op_4733_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4733_cast_fp16 = slice_by_index(begin = var_4733_begin_0, end = var_4733_end_0, end_mask = var_4733_end_mask_0, x = V_sliding_out_9_cast_fp16)[name = string("op_4733_cast_fp16")]; int32 var_4735 = const()[name = string("op_4735"), val = int32(0)]; bool V_sliding_out_11_interleave_0 = const()[name = string("V_sliding_out_11_interleave_0"), val = bool(false)]; tensor V_sliding_out_11_cast_fp16 = concat(axis = var_4735, interleave = V_sliding_out_11_interleave_0, values = (var_4728_cast_fp16, new_v_13_cast_fp16, var_4733_cast_fp16))[name = string("V_sliding_out_11_cast_fp16")]; tensor var_4741_begin_0 = const()[name = string("op_4741_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_4741_end_0 = const()[name = string("op_4741_end_0"), val = tensor([6, 2, 512, 512])]; tensor var_4741_end_mask_0 = const()[name = string("op_4741_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4741_cast_fp16 = slice_by_index(begin = var_4741_begin_0, end = var_4741_end_0, end_mask = var_4741_end_mask_0, x = K_sliding_out_11_cast_fp16)[name = string("op_4741_cast_fp16")]; tensor K_for_attn_13_begin_0 = const()[name = string("K_for_attn_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_13_end_0 = const()[name = string("K_for_attn_13_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_13_end_mask_0 = const()[name = string("K_for_attn_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_13_cast_fp16 = slice_by_index(begin = K_for_attn_13_begin_0, end = K_for_attn_13_end_0, end_mask = K_for_attn_13_end_mask_0, x = var_4741_cast_fp16)[name = string("K_for_attn_13_cast_fp16")]; tensor var_4751_begin_0 = const()[name = string("op_4751_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_4751_end_0 = const()[name = string("op_4751_end_0"), val = tensor([6, 2, 512, 512])]; tensor var_4751_end_mask_0 = const()[name = string("op_4751_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4751_cast_fp16 = slice_by_index(begin = var_4751_begin_0, end = var_4751_end_0, end_mask = var_4751_end_mask_0, x = V_sliding_out_11_cast_fp16)[name = string("op_4751_cast_fp16")]; tensor V_for_attn_13_begin_0 = const()[name = string("V_for_attn_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_13_end_0 = const()[name = string("V_for_attn_13_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_13_end_mask_0 = const()[name = string("V_for_attn_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_13_cast_fp16 = slice_by_index(begin = V_for_attn_13_begin_0, end = V_for_attn_13_end_0, end_mask = V_for_attn_13_end_mask_0, x = var_4751_cast_fp16)[name = string("V_for_attn_13_cast_fp16")]; tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = K_for_attn_13_cast_fp16)[name = string("transpose_113")]; tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; tensor concat_26 = const()[name = string("concat_26"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_26, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_27 = const()[name = string("concat_27"), val = tensor([-1, 1, 512, 256])]; tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_112")]; tensor reshape_25_cast_fp16 = reshape(shape = concat_27, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; tensor transpose_68_perm_0 = const()[name = string("transpose_68_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = V_for_attn_13_cast_fp16)[name = string("transpose_111")]; tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; tensor concat_28 = const()[name = string("concat_28"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_28, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_29 = const()[name = string("concat_29"), val = tensor([-1, 1, 512, 256])]; tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_110")]; tensor reshape_27_cast_fp16 = reshape(shape = concat_29, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor V_expanded_13_perm_0 = const()[name = string("V_expanded_13_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor transpose_68_cast_fp16 = transpose(perm = transpose_68_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_109")]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = q_83_cast_fp16, y = transpose_68_cast_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor x_127_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask_sliding)[name = string("x_127_cast_fp16")]; tensor reduce_max_6_axes_0 = const()[name = string("reduce_max_6_axes_0"), val = tensor([-1])]; bool reduce_max_6_keep_dims_0 = const()[name = string("reduce_max_6_keep_dims_0"), val = bool(true)]; tensor reduce_max_6 = reduce_max(axes = reduce_max_6_axes_0, keep_dims = reduce_max_6_keep_dims_0, x = x_127_cast_fp16)[name = string("reduce_max_6")]; tensor var_4786 = sub(x = x_127_cast_fp16, y = reduce_max_6)[name = string("op_4786")]; tensor var_4792 = exp(x = var_4786)[name = string("op_4792")]; tensor var_4802_axes_0 = const()[name = string("op_4802_axes_0"), val = tensor([-1])]; bool var_4802_keep_dims_0 = const()[name = string("op_4802_keep_dims_0"), val = bool(true)]; tensor var_4802 = reduce_sum(axes = var_4802_axes_0, keep_dims = var_4802_keep_dims_0, x = var_4792)[name = string("op_4802")]; tensor var_4808_cast_fp16 = real_div(x = var_4792, y = var_4802)[name = string("op_4808_cast_fp16")]; bool attn_output_37_transpose_x_0 = const()[name = string("attn_output_37_transpose_x_0"), val = bool(false)]; bool attn_output_37_transpose_y_0 = const()[name = string("attn_output_37_transpose_y_0"), val = bool(false)]; tensor V_expanded_13_cast_fp16 = transpose(perm = V_expanded_13_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_108")]; tensor attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = var_4808_cast_fp16, y = V_expanded_13_cast_fp16)[name = string("attn_output_37_cast_fp16")]; tensor var_4819 = const()[name = string("op_4819"), val = tensor([0, 2, 1, 3])]; tensor var_4826 = const()[name = string("op_4826"), val = tensor([1, 3, -1])]; tensor var_4820_cast_fp16 = transpose(perm = var_4819, x = attn_output_37_cast_fp16)[name = string("transpose_107")]; tensor attn_output_39_cast_fp16 = reshape(shape = var_4826, x = var_4820_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor var_4831 = const()[name = string("op_4831"), val = tensor([0, 2, 1])]; string var_4847_pad_type_0 = const()[name = string("op_4847_pad_type_0"), val = string("valid")]; int32 var_4847_groups_0 = const()[name = string("op_4847_groups_0"), val = int32(1)]; tensor var_4847_strides_0 = const()[name = string("op_4847_strides_0"), val = tensor([1])]; tensor var_4847_pad_0 = const()[name = string("op_4847_pad_0"), val = tensor([0, 0])]; tensor var_4847_dilations_0 = const()[name = string("op_4847_dilations_0"), val = tensor([1])]; tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(551729344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554350848))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4832_cast_fp16 = transpose(perm = var_4831, x = attn_output_39_cast_fp16)[name = string("transpose_106")]; tensor var_4847_cast_fp16 = conv(dilations = var_4847_dilations_0, groups = var_4847_groups_0, pad = var_4847_pad_0, pad_type = var_4847_pad_type_0, strides = var_4847_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4832_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4851 = const()[name = string("op_4851"), val = tensor([0, 2, 1])]; int32 var_4857 = const()[name = string("op_4857"), val = int32(-1)]; fp16 const_77_promoted_to_fp16 = const()[name = string("const_77_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_131_cast_fp16 = transpose(perm = var_4851, x = var_4847_cast_fp16)[name = string("transpose_105")]; tensor var_4859_cast_fp16 = mul(x = x_131_cast_fp16, y = const_77_promoted_to_fp16)[name = string("op_4859_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_4857, interleave = input_191_interleave_0, values = (x_131_cast_fp16, var_4859_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_4854_to_fp16 = const()[name = string("op_4854_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_4854_to_fp16, x = input_191_cast_fp16)[name = string("normed_181_cast_fp16")]; tensor var_4864_split_sizes_0 = const()[name = string("op_4864_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4864_axis_0 = const()[name = string("op_4864_axis_0"), val = int32(-1)]; tensor var_4864_cast_fp16_0, tensor var_4864_cast_fp16_1 = split(axis = var_4864_axis_0, split_sizes = var_4864_split_sizes_0, x = normed_181_cast_fp16)[name = string("op_4864_cast_fp16")]; tensor layers_6_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554353472)))]; tensor attn_output_41_cast_fp16 = mul(x = var_4864_cast_fp16_0, y = layers_6_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_41_cast_fp16")]; tensor x_133_cast_fp16 = add(x = x_119_cast_fp16, y = attn_output_41_cast_fp16)[name = string("x_133_cast_fp16")]; int32 var_4873 = const()[name = string("op_4873"), val = int32(-1)]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4875_cast_fp16 = mul(x = x_133_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_4875_cast_fp16")]; bool input_193_interleave_0 = const()[name = string("input_193_interleave_0"), val = bool(false)]; tensor input_193_cast_fp16 = concat(axis = var_4873, interleave = input_193_interleave_0, values = (x_133_cast_fp16, var_4875_cast_fp16))[name = string("input_193_cast_fp16")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_4870_to_fp16 = const()[name = string("op_4870_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_4870_to_fp16, x = input_193_cast_fp16)[name = string("normed_185_cast_fp16")]; tensor var_4880_split_sizes_0 = const()[name = string("op_4880_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4880_axis_0 = const()[name = string("op_4880_axis_0"), val = int32(-1)]; tensor var_4880_cast_fp16_0, tensor var_4880_cast_fp16_1 = split(axis = var_4880_axis_0, split_sizes = var_4880_split_sizes_0, x = normed_185_cast_fp16)[name = string("op_4880_cast_fp16")]; tensor layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554358656)))]; tensor h_39_cast_fp16 = mul(x = var_4880_cast_fp16_0, y = layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_39_cast_fp16")]; tensor var_4891 = const()[name = string("op_4891"), val = tensor([0, 2, 1])]; tensor input_195_axes_0 = const()[name = string("input_195_axes_0"), val = tensor([2])]; tensor var_4892 = transpose(perm = var_4891, x = h_39_cast_fp16)[name = string("transpose_104")]; tensor input_195 = expand_dims(axes = input_195_axes_0, x = var_4892)[name = string("input_195")]; string gate_25_pad_type_0 = const()[name = string("gate_25_pad_type_0"), val = string("valid")]; tensor gate_25_strides_0 = const()[name = string("gate_25_strides_0"), val = tensor([1, 1])]; tensor gate_25_pad_0 = const()[name = string("gate_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_25_dilations_0 = const()[name = string("gate_25_dilations_0"), val = tensor([1, 1])]; int32 gate_25_groups_0 = const()[name = string("gate_25_groups_0"), val = int32(1)]; tensor gate_25 = conv(dilations = gate_25_dilations_0, groups = gate_25_groups_0, pad = gate_25_pad_0, pad_type = gate_25_pad_type_0, strides = gate_25_strides_0, weight = layers_6_mlp_gate_proj_weight_palettized, x = input_195)[name = string("gate_25")]; string up_13_pad_type_0 = const()[name = string("up_13_pad_type_0"), val = string("valid")]; tensor up_13_strides_0 = const()[name = string("up_13_strides_0"), val = tensor([1, 1])]; tensor up_13_pad_0 = const()[name = string("up_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_13_dilations_0 = const()[name = string("up_13_dilations_0"), val = tensor([1, 1])]; int32 up_13_groups_0 = const()[name = string("up_13_groups_0"), val = int32(1)]; tensor up_13 = conv(dilations = up_13_dilations_0, groups = up_13_groups_0, pad = up_13_pad_0, pad_type = up_13_pad_type_0, strides = up_13_strides_0, weight = layers_6_mlp_up_proj_weight_palettized, x = input_195)[name = string("up_13")]; string gate_27_mode_0 = const()[name = string("gate_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_27 = gelu(mode = gate_27_mode_0, x = gate_25)[name = string("gate_27")]; tensor input_197 = mul(x = gate_27, y = up_13)[name = string("input_197")]; string mlp_out_13_pad_type_0 = const()[name = string("mlp_out_13_pad_type_0"), val = string("valid")]; tensor mlp_out_13_strides_0 = const()[name = string("mlp_out_13_strides_0"), val = tensor([1, 1])]; tensor mlp_out_13_pad_0 = const()[name = string("mlp_out_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_13_dilations_0 = const()[name = string("mlp_out_13_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_13_groups_0 = const()[name = string("mlp_out_13_groups_0"), val = int32(1)]; tensor mlp_out_13 = conv(dilations = mlp_out_13_dilations_0, groups = mlp_out_13_groups_0, pad = mlp_out_13_pad_0, pad_type = mlp_out_13_pad_type_0, strides = mlp_out_13_strides_0, weight = layers_6_mlp_down_proj_weight_palettized, x = input_197)[name = string("mlp_out_13")]; tensor var_4932_axes_0 = const()[name = string("op_4932_axes_0"), val = tensor([2])]; tensor var_4932 = squeeze(axes = var_4932_axes_0, x = mlp_out_13)[name = string("op_4932")]; tensor var_4936 = const()[name = string("op_4936"), val = tensor([0, 2, 1])]; int32 var_4942 = const()[name = string("op_4942"), val = int32(-1)]; fp16 const_79_promoted = const()[name = string("const_79_promoted"), val = fp16(-0x1p+0)]; tensor x_135 = transpose(perm = var_4936, x = var_4932)[name = string("transpose_103")]; tensor var_4944 = mul(x = x_135, y = const_79_promoted)[name = string("op_4944")]; bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; tensor input_199 = concat(axis = var_4942, interleave = input_199_interleave_0, values = (x_135, var_4944))[name = string("input_199")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_4939_to_fp16 = const()[name = string("op_4939_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_4939_to_fp16, x = input_199)[name = string("normed_189_cast_fp16")]; tensor var_4949_split_sizes_0 = const()[name = string("op_4949_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_4949_axis_0 = const()[name = string("op_4949_axis_0"), val = int32(-1)]; tensor var_4949_0, tensor var_4949_1 = split(axis = var_4949_axis_0, split_sizes = var_4949_split_sizes_0, x = normed_189_cast_fp16)[name = string("op_4949")]; tensor hidden_states_63 = mul(x = var_4949_0, y = layers_6_post_feedforward_layernorm_weight)[name = string("hidden_states_63")]; tensor hidden_states_65_cast_fp16 = add(x = x_133_cast_fp16, y = hidden_states_63)[name = string("hidden_states_65_cast_fp16")]; tensor per_layer_slice_13_begin_0 = const()[name = string("per_layer_slice_13_begin_0"), val = tensor([0, 0, 4608])]; tensor per_layer_slice_13_end_0 = const()[name = string("per_layer_slice_13_end_0"), val = tensor([1, 3, 4864])]; tensor per_layer_slice_13_end_mask_0 = const()[name = string("per_layer_slice_13_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_13_cast_fp16 = slice_by_index(begin = per_layer_slice_13_begin_0, end = per_layer_slice_13_end_0, end_mask = per_layer_slice_13_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_13_cast_fp16")]; tensor var_4977 = const()[name = string("op_4977"), val = tensor([0, 2, 1])]; tensor input_201_axes_0 = const()[name = string("input_201_axes_0"), val = tensor([2])]; tensor var_4978 = transpose(perm = var_4977, x = hidden_states_65_cast_fp16)[name = string("transpose_102")]; tensor input_201 = expand_dims(axes = input_201_axes_0, x = var_4978)[name = string("input_201")]; string gated_37_pad_type_0 = const()[name = string("gated_37_pad_type_0"), val = string("valid")]; tensor gated_37_strides_0 = const()[name = string("gated_37_strides_0"), val = tensor([1, 1])]; tensor gated_37_pad_0 = const()[name = string("gated_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_37_dilations_0 = const()[name = string("gated_37_dilations_0"), val = tensor([1, 1])]; int32 gated_37_groups_0 = const()[name = string("gated_37_groups_0"), val = int32(1)]; tensor gated_37 = conv(dilations = gated_37_dilations_0, groups = gated_37_groups_0, pad = gated_37_pad_0, pad_type = gated_37_pad_type_0, strides = gated_37_strides_0, weight = layers_6_per_layer_input_gate_weight_palettized, x = input_201)[name = string("gated_37")]; string gated_39_mode_0 = const()[name = string("gated_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_39 = gelu(mode = gated_39_mode_0, x = gated_37)[name = string("gated_39")]; tensor var_4997 = const()[name = string("op_4997"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_13_axes_0 = const()[name = string("per_layer_slice_conv_13_axes_0"), val = tensor([2])]; tensor var_4998_cast_fp16 = transpose(perm = var_4997, x = per_layer_slice_13_cast_fp16)[name = string("transpose_101")]; tensor per_layer_slice_conv_13_cast_fp16 = expand_dims(axes = per_layer_slice_conv_13_axes_0, x = var_4998_cast_fp16)[name = string("per_layer_slice_conv_13_cast_fp16")]; tensor input_203_cast_fp16 = mul(x = gated_39, y = per_layer_slice_conv_13_cast_fp16)[name = string("input_203_cast_fp16")]; string gated_41_pad_type_0 = const()[name = string("gated_41_pad_type_0"), val = string("valid")]; tensor gated_41_strides_0 = const()[name = string("gated_41_strides_0"), val = tensor([1, 1])]; tensor gated_41_pad_0 = const()[name = string("gated_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_41_dilations_0 = const()[name = string("gated_41_dilations_0"), val = tensor([1, 1])]; int32 gated_41_groups_0 = const()[name = string("gated_41_groups_0"), val = int32(1)]; tensor layers_6_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554363840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554691584))))[name = string("layers_6_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_41_cast_fp16 = conv(dilations = gated_41_dilations_0, groups = gated_41_groups_0, pad = gated_41_pad_0, pad_type = gated_41_pad_type_0, strides = gated_41_strides_0, weight = layers_6_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_203_cast_fp16)[name = string("gated_41_cast_fp16")]; tensor var_5014_axes_0 = const()[name = string("op_5014_axes_0"), val = tensor([2])]; tensor var_5014_cast_fp16 = squeeze(axes = var_5014_axes_0, x = gated_41_cast_fp16)[name = string("op_5014_cast_fp16")]; tensor var_5018 = const()[name = string("op_5018"), val = tensor([0, 2, 1])]; int32 var_5024 = const()[name = string("op_5024"), val = int32(-1)]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_137_cast_fp16 = transpose(perm = var_5018, x = var_5014_cast_fp16)[name = string("transpose_100")]; tensor var_5026_cast_fp16 = mul(x = x_137_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_5026_cast_fp16")]; bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; tensor input_205_cast_fp16 = concat(axis = var_5024, interleave = input_205_interleave_0, values = (x_137_cast_fp16, var_5026_cast_fp16))[name = string("input_205_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_5021_to_fp16 = const()[name = string("op_5021_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_5021_to_fp16, x = input_205_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor var_5031_split_sizes_0 = const()[name = string("op_5031_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5031_axis_0 = const()[name = string("op_5031_axis_0"), val = int32(-1)]; tensor var_5031_cast_fp16_0, tensor var_5031_cast_fp16_1 = split(axis = var_5031_axis_0, split_sizes = var_5031_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_5031_cast_fp16")]; tensor layers_6_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554694208)))]; tensor hidden_states_69_cast_fp16 = mul(x = var_5031_cast_fp16_0, y = layers_6_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_69_cast_fp16")]; tensor hidden_states_71_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_69_cast_fp16)[name = string("hidden_states_71_cast_fp16")]; tensor const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = tensor([0x1.16p-1])]; tensor x_139_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_81_promoted_to_fp16)[name = string("x_139_cast_fp16")]; int32 var_5046 = const()[name = string("op_5046"), val = int32(-1)]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5048_cast_fp16 = mul(x = x_139_cast_fp16, y = const_82_promoted_to_fp16)[name = string("op_5048_cast_fp16")]; bool input_207_interleave_0 = const()[name = string("input_207_interleave_0"), val = bool(false)]; tensor input_207_cast_fp16 = concat(axis = var_5046, interleave = input_207_interleave_0, values = (x_139_cast_fp16, var_5048_cast_fp16))[name = string("input_207_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_5043_to_fp16 = const()[name = string("op_5043_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_5043_to_fp16, x = input_207_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_5053_split_sizes_0 = const()[name = string("op_5053_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5053_axis_0 = const()[name = string("op_5053_axis_0"), val = int32(-1)]; tensor var_5053_cast_fp16_0, tensor var_5053_cast_fp16_1 = split(axis = var_5053_axis_0, split_sizes = var_5053_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_5053_cast_fp16")]; tensor layers_7_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554699392)))]; tensor h_43_cast_fp16 = mul(x = var_5053_cast_fp16_0, y = layers_7_input_layernorm_weight_promoted_to_fp16)[name = string("h_43_cast_fp16")]; tensor var_5059 = const()[name = string("op_5059"), val = tensor([0, 2, 1])]; tensor var_5062_axes_0 = const()[name = string("op_5062_axes_0"), val = tensor([2])]; tensor var_5060_cast_fp16 = transpose(perm = var_5059, x = h_43_cast_fp16)[name = string("transpose_99")]; tensor var_5062_cast_fp16 = expand_dims(axes = var_5062_axes_0, x = var_5060_cast_fp16)[name = string("op_5062_cast_fp16")]; string q_85_pad_type_0 = const()[name = string("q_85_pad_type_0"), val = string("valid")]; tensor q_85_strides_0 = const()[name = string("q_85_strides_0"), val = tensor([1, 1])]; tensor q_85_pad_0 = const()[name = string("q_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_85_dilations_0 = const()[name = string("q_85_dilations_0"), val = tensor([1, 1])]; int32 q_85_groups_0 = const()[name = string("q_85_groups_0"), val = int32(1)]; tensor q_85 = conv(dilations = q_85_dilations_0, groups = q_85_groups_0, pad = q_85_pad_0, pad_type = q_85_pad_type_0, strides = q_85_strides_0, weight = layers_7_self_attn_q_proj_weight_palettized, x = var_5062_cast_fp16)[name = string("q_85")]; tensor var_5083 = const()[name = string("op_5083"), val = tensor([1, 8, 256, 3])]; tensor var_5084 = reshape(shape = var_5083, x = q_85)[name = string("op_5084")]; tensor transpose_69_perm_0 = const()[name = string("transpose_69_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_5107 = const()[name = string("op_5107"), val = tensor([3, 8, 256])]; tensor transpose_69 = transpose(perm = transpose_69_perm_0, x = var_5084)[name = string("transpose_98")]; tensor x_141 = reshape(shape = var_5107, x = transpose_69)[name = string("x_141")]; int32 var_5113 = const()[name = string("op_5113"), val = int32(-1)]; fp16 const_83_promoted = const()[name = string("const_83_promoted"), val = fp16(-0x1p+0)]; tensor var_5115 = mul(x = x_141, y = const_83_promoted)[name = string("op_5115")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211 = concat(axis = var_5113, interleave = input_211_interleave_0, values = (x_141, var_5115))[name = string("input_211")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_5110_to_fp16 = const()[name = string("op_5110_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_5110_to_fp16, x = input_211)[name = string("normed_201_cast_fp16")]; tensor var_5120_split_sizes_0 = const()[name = string("op_5120_split_sizes_0"), val = tensor([256, 256])]; int32 var_5120_axis_0 = const()[name = string("op_5120_axis_0"), val = int32(-1)]; tensor var_5120_0, tensor var_5120_1 = split(axis = var_5120_axis_0, split_sizes = var_5120_split_sizes_0, x = normed_201_cast_fp16)[name = string("op_5120")]; tensor q_89 = mul(x = var_5120_0, y = layers_7_self_attn_q_norm_weight)[name = string("q_89")]; tensor var_5127 = const()[name = string("op_5127"), val = tensor([1, 3, 8, 256])]; tensor var_5128 = reshape(shape = var_5127, x = q_89)[name = string("op_5128")]; tensor var_5133 = const()[name = string("op_5133"), val = tensor([0, 2, 1, 3])]; tensor q_91 = transpose(perm = var_5133, x = var_5128)[name = string("transpose_97")]; tensor var_5135_cast_fp16 = mul(x = q_91, y = cos_s)[name = string("op_5135_cast_fp16")]; tensor var_5136_split_sizes_0 = const()[name = string("op_5136_split_sizes_0"), val = tensor([128, 128])]; int32 var_5136_axis_0 = const()[name = string("op_5136_axis_0"), val = int32(-1)]; tensor var_5136_0, tensor var_5136_1 = split(axis = var_5136_axis_0, split_sizes = var_5136_split_sizes_0, x = q_91)[name = string("op_5136")]; fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; tensor var_5138 = mul(x = var_5136_1, y = const_84_promoted)[name = string("op_5138")]; int32 var_5140 = const()[name = string("op_5140"), val = int32(-1)]; bool var_5141_interleave_0 = const()[name = string("op_5141_interleave_0"), val = bool(false)]; tensor var_5141 = concat(axis = var_5140, interleave = var_5141_interleave_0, values = (var_5138, var_5136_0))[name = string("op_5141")]; tensor var_5142_cast_fp16 = mul(x = var_5141, y = sin_s)[name = string("op_5142_cast_fp16")]; tensor q_95_cast_fp16 = add(x = var_5135_cast_fp16, y = var_5142_cast_fp16)[name = string("q_95_cast_fp16")]; string k_45_pad_type_0 = const()[name = string("k_45_pad_type_0"), val = string("valid")]; tensor k_45_strides_0 = const()[name = string("k_45_strides_0"), val = tensor([1, 1])]; tensor k_45_pad_0 = const()[name = string("k_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_45_dilations_0 = const()[name = string("k_45_dilations_0"), val = tensor([1, 1])]; int32 k_45_groups_0 = const()[name = string("k_45_groups_0"), val = int32(1)]; tensor k_45 = conv(dilations = k_45_dilations_0, groups = k_45_groups_0, pad = k_45_pad_0, pad_type = k_45_pad_type_0, strides = k_45_strides_0, weight = layers_7_self_attn_k_proj_weight_palettized, x = var_5062_cast_fp16)[name = string("k_45")]; tensor var_5160 = const()[name = string("op_5160"), val = tensor([1, 2, 256, 3])]; tensor var_5161 = reshape(shape = var_5160, x = k_45)[name = string("op_5161")]; tensor transpose_70_perm_0 = const()[name = string("transpose_70_perm_0"), val = tensor([0, 3, 1, 2])]; string v_17_pad_type_0 = const()[name = string("v_17_pad_type_0"), val = string("valid")]; tensor v_17_strides_0 = const()[name = string("v_17_strides_0"), val = tensor([1, 1])]; tensor v_17_pad_0 = const()[name = string("v_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_17_dilations_0 = const()[name = string("v_17_dilations_0"), val = tensor([1, 1])]; int32 v_17_groups_0 = const()[name = string("v_17_groups_0"), val = int32(1)]; tensor v_17 = conv(dilations = v_17_dilations_0, groups = v_17_groups_0, pad = v_17_pad_0, pad_type = v_17_pad_type_0, strides = v_17_strides_0, weight = layers_7_self_attn_v_proj_weight_palettized, x = var_5062_cast_fp16)[name = string("v_17")]; tensor var_5188 = const()[name = string("op_5188"), val = tensor([1, 2, 256, 3])]; tensor var_5189 = reshape(shape = var_5188, x = v_17)[name = string("op_5189")]; tensor var_5194 = const()[name = string("op_5194"), val = tensor([0, 1, 3, 2])]; tensor var_5212 = const()[name = string("op_5212"), val = tensor([3, 2, 256])]; tensor transpose_70 = transpose(perm = transpose_70_perm_0, x = var_5161)[name = string("transpose_96")]; tensor x_143 = reshape(shape = var_5212, x = transpose_70)[name = string("x_143")]; int32 var_5218 = const()[name = string("op_5218"), val = int32(-1)]; fp16 const_85_promoted = const()[name = string("const_85_promoted"), val = fp16(-0x1p+0)]; tensor var_5220 = mul(x = x_143, y = const_85_promoted)[name = string("op_5220")]; bool input_213_interleave_0 = const()[name = string("input_213_interleave_0"), val = bool(false)]; tensor input_213 = concat(axis = var_5218, interleave = input_213_interleave_0, values = (x_143, var_5220))[name = string("input_213")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_5215_to_fp16 = const()[name = string("op_5215_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_5215_to_fp16, x = input_213)[name = string("normed_205_cast_fp16")]; tensor var_5225_split_sizes_0 = const()[name = string("op_5225_split_sizes_0"), val = tensor([256, 256])]; int32 var_5225_axis_0 = const()[name = string("op_5225_axis_0"), val = int32(-1)]; tensor var_5225_0, tensor var_5225_1 = split(axis = var_5225_axis_0, split_sizes = var_5225_split_sizes_0, x = normed_205_cast_fp16)[name = string("op_5225")]; tensor k_49 = mul(x = var_5225_0, y = layers_7_self_attn_k_norm_weight)[name = string("k_49")]; tensor var_5232 = const()[name = string("op_5232"), val = tensor([1, 3, 2, 256])]; tensor var_5233 = reshape(shape = var_5232, x = k_49)[name = string("op_5233")]; tensor var_5238 = const()[name = string("op_5238"), val = tensor([0, 2, 1, 3])]; fp16 var_5240_promoted = const()[name = string("op_5240_promoted"), val = fp16(0x1p+1)]; tensor var_5195 = transpose(perm = var_5194, x = var_5189)[name = string("transpose_95")]; tensor var_5241 = pow(x = var_5195, y = var_5240_promoted)[name = string("op_5241")]; tensor var_5246_axes_0 = const()[name = string("op_5246_axes_0"), val = tensor([-1])]; bool var_5246_keep_dims_0 = const()[name = string("op_5246_keep_dims_0"), val = bool(true)]; tensor var_5246 = reduce_mean(axes = var_5246_axes_0, keep_dims = var_5246_keep_dims_0, x = var_5241)[name = string("op_5246")]; fp16 var_5248_to_fp16 = const()[name = string("op_5248_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_15_cast_fp16 = add(x = var_5246, y = var_5248_to_fp16)[name = string("mean_sq_15_cast_fp16")]; fp32 var_5250_epsilon_0 = const()[name = string("op_5250_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_5250_cast_fp16 = rsqrt(epsilon = var_5250_epsilon_0, x = mean_sq_15_cast_fp16)[name = string("op_5250_cast_fp16")]; tensor input_217_cast_fp16 = mul(x = var_5195, y = var_5250_cast_fp16)[name = string("input_217_cast_fp16")]; tensor q_93 = transpose(perm = var_5238, x = var_5233)[name = string("transpose_94")]; tensor var_5252_cast_fp16 = mul(x = q_93, y = cos_s)[name = string("op_5252_cast_fp16")]; tensor var_5253_split_sizes_0 = const()[name = string("op_5253_split_sizes_0"), val = tensor([128, 128])]; int32 var_5253_axis_0 = const()[name = string("op_5253_axis_0"), val = int32(-1)]; tensor var_5253_0, tensor var_5253_1 = split(axis = var_5253_axis_0, split_sizes = var_5253_split_sizes_0, x = q_93)[name = string("op_5253")]; fp16 const_86_promoted = const()[name = string("const_86_promoted"), val = fp16(-0x1p+0)]; tensor var_5255 = mul(x = var_5253_1, y = const_86_promoted)[name = string("op_5255")]; int32 var_5257 = const()[name = string("op_5257"), val = int32(-1)]; bool var_5258_interleave_0 = const()[name = string("op_5258_interleave_0"), val = bool(false)]; tensor var_5258 = concat(axis = var_5257, interleave = var_5258_interleave_0, values = (var_5255, var_5253_0))[name = string("op_5258")]; tensor var_5259_cast_fp16 = mul(x = var_5258, y = sin_s)[name = string("op_5259_cast_fp16")]; tensor input_215_cast_fp16 = add(x = var_5252_cast_fp16, y = var_5259_cast_fp16)[name = string("input_215_cast_fp16")]; tensor k_padded_13_pad_0 = const()[name = string("k_padded_13_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_13_mode_0 = const()[name = string("k_padded_13_mode_0"), val = string("constant")]; fp16 const_87_to_fp16 = const()[name = string("const_87_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_13_cast_fp16 = pad(constant_val = const_87_to_fp16, mode = k_padded_13_mode_0, pad = k_padded_13_pad_0, x = input_215_cast_fp16)[name = string("k_padded_13_cast_fp16")]; tensor v_padded_13_pad_0 = const()[name = string("v_padded_13_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_13_mode_0 = const()[name = string("v_padded_13_mode_0"), val = string("constant")]; fp16 const_88_to_fp16 = const()[name = string("const_88_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_13_cast_fp16 = pad(constant_val = const_88_to_fp16, mode = v_padded_13_mode_0, pad = v_padded_13_pad_0, x = input_217_cast_fp16)[name = string("v_padded_13_cast_fp16")]; tensor slot_k_15_begin_0 = const()[name = string("slot_k_15_begin_0"), val = tensor([6, 0, 0, 0])]; tensor slot_k_15_end_0 = const()[name = string("slot_k_15_end_0"), val = tensor([7, 2, 512, 512])]; tensor slot_k_15_end_mask_0 = const()[name = string("slot_k_15_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_15_cast_fp16 = slice_by_index(begin = slot_k_15_begin_0, end = slot_k_15_end_0, end_mask = slot_k_15_end_mask_0, x = K_sliding_out_11_cast_fp16)[name = string("slot_k_15_cast_fp16")]; tensor slot_v_15_begin_0 = const()[name = string("slot_v_15_begin_0"), val = tensor([6, 0, 0, 0])]; tensor slot_v_15_end_0 = const()[name = string("slot_v_15_end_0"), val = tensor([7, 2, 512, 512])]; tensor slot_v_15_end_mask_0 = const()[name = string("slot_v_15_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_15_cast_fp16 = slice_by_index(begin = slot_v_15_begin_0, end = slot_v_15_end_0, end_mask = slot_v_15_end_mask_0, x = V_sliding_out_11_cast_fp16)[name = string("slot_v_15_cast_fp16")]; tensor var_5298_begin_0 = const()[name = string("op_5298_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_5298_end_0 = const()[name = string("op_5298_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5298_end_mask_0 = const()[name = string("op_5298_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5298_cast_fp16 = slice_by_index(begin = var_5298_begin_0, end = var_5298_end_0, end_mask = var_5298_end_mask_0, x = slot_k_15_cast_fp16)[name = string("op_5298_cast_fp16")]; int32 var_5305 = const()[name = string("op_5305"), val = int32(2)]; bool new_k_15_interleave_0 = const()[name = string("new_k_15_interleave_0"), val = bool(false)]; tensor new_k_15_cast_fp16 = concat(axis = var_5305, interleave = new_k_15_interleave_0, values = (var_5298_cast_fp16, k_padded_13_cast_fp16))[name = string("new_k_15_cast_fp16")]; tensor var_5321_begin_0 = const()[name = string("op_5321_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_5321_end_0 = const()[name = string("op_5321_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5321_end_mask_0 = const()[name = string("op_5321_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5321_cast_fp16 = slice_by_index(begin = var_5321_begin_0, end = var_5321_end_0, end_mask = var_5321_end_mask_0, x = slot_v_15_cast_fp16)[name = string("op_5321_cast_fp16")]; int32 var_5328 = const()[name = string("op_5328"), val = int32(2)]; bool new_v_15_interleave_0 = const()[name = string("new_v_15_interleave_0"), val = bool(false)]; tensor new_v_15_cast_fp16 = concat(axis = var_5328, interleave = new_v_15_interleave_0, values = (var_5321_cast_fp16, v_padded_13_cast_fp16))[name = string("new_v_15_cast_fp16")]; tensor var_5334_begin_0 = const()[name = string("op_5334_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5334_end_0 = const()[name = string("op_5334_end_0"), val = tensor([6, 2, 512, 512])]; tensor var_5334_end_mask_0 = const()[name = string("op_5334_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5334_cast_fp16 = slice_by_index(begin = var_5334_begin_0, end = var_5334_end_0, end_mask = var_5334_end_mask_0, x = K_sliding_out_11_cast_fp16)[name = string("op_5334_cast_fp16")]; tensor var_5339_begin_0 = const()[name = string("op_5339_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5339_end_0 = const()[name = string("op_5339_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_5339_end_mask_0 = const()[name = string("op_5339_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5339_cast_fp16 = slice_by_index(begin = var_5339_begin_0, end = var_5339_end_0, end_mask = var_5339_end_mask_0, x = K_sliding_out_11_cast_fp16)[name = string("op_5339_cast_fp16")]; int32 var_5341 = const()[name = string("op_5341"), val = int32(0)]; bool K_sliding_out_13_interleave_0 = const()[name = string("K_sliding_out_13_interleave_0"), val = bool(false)]; tensor K_sliding_out_13_cast_fp16 = concat(axis = var_5341, interleave = K_sliding_out_13_interleave_0, values = (var_5334_cast_fp16, new_k_15_cast_fp16, var_5339_cast_fp16))[name = string("K_sliding_out_13_cast_fp16")]; tensor var_5347_begin_0 = const()[name = string("op_5347_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5347_end_0 = const()[name = string("op_5347_end_0"), val = tensor([6, 2, 512, 512])]; tensor var_5347_end_mask_0 = const()[name = string("op_5347_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5347_cast_fp16 = slice_by_index(begin = var_5347_begin_0, end = var_5347_end_0, end_mask = var_5347_end_mask_0, x = V_sliding_out_11_cast_fp16)[name = string("op_5347_cast_fp16")]; tensor var_5352_begin_0 = const()[name = string("op_5352_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5352_end_0 = const()[name = string("op_5352_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_5352_end_mask_0 = const()[name = string("op_5352_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5352_cast_fp16 = slice_by_index(begin = var_5352_begin_0, end = var_5352_end_0, end_mask = var_5352_end_mask_0, x = V_sliding_out_11_cast_fp16)[name = string("op_5352_cast_fp16")]; int32 var_5354 = const()[name = string("op_5354"), val = int32(0)]; bool V_sliding_out_13_interleave_0 = const()[name = string("V_sliding_out_13_interleave_0"), val = bool(false)]; tensor V_sliding_out_13_cast_fp16 = concat(axis = var_5354, interleave = V_sliding_out_13_interleave_0, values = (var_5347_cast_fp16, new_v_15_cast_fp16, var_5352_cast_fp16))[name = string("V_sliding_out_13_cast_fp16")]; tensor var_5360_begin_0 = const()[name = string("op_5360_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_5360_end_0 = const()[name = string("op_5360_end_0"), val = tensor([7, 2, 512, 512])]; tensor var_5360_end_mask_0 = const()[name = string("op_5360_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5360_cast_fp16 = slice_by_index(begin = var_5360_begin_0, end = var_5360_end_0, end_mask = var_5360_end_mask_0, x = K_sliding_out_13_cast_fp16)[name = string("op_5360_cast_fp16")]; tensor K_for_attn_15_begin_0 = const()[name = string("K_for_attn_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_15_end_0 = const()[name = string("K_for_attn_15_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_15_end_mask_0 = const()[name = string("K_for_attn_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_15_cast_fp16 = slice_by_index(begin = K_for_attn_15_begin_0, end = K_for_attn_15_end_0, end_mask = K_for_attn_15_end_mask_0, x = var_5360_cast_fp16)[name = string("K_for_attn_15_cast_fp16")]; tensor var_5370_begin_0 = const()[name = string("op_5370_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_5370_end_0 = const()[name = string("op_5370_end_0"), val = tensor([7, 2, 512, 512])]; tensor var_5370_end_mask_0 = const()[name = string("op_5370_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5370_cast_fp16 = slice_by_index(begin = var_5370_begin_0, end = var_5370_end_0, end_mask = var_5370_end_mask_0, x = V_sliding_out_13_cast_fp16)[name = string("op_5370_cast_fp16")]; tensor V_for_attn_15_begin_0 = const()[name = string("V_for_attn_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_15_end_0 = const()[name = string("V_for_attn_15_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_15_end_mask_0 = const()[name = string("V_for_attn_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_15_cast_fp16 = slice_by_index(begin = V_for_attn_15_begin_0, end = V_for_attn_15_end_0, end_mask = V_for_attn_15_end_mask_0, x = var_5370_cast_fp16)[name = string("V_for_attn_15_cast_fp16")]; tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = K_for_attn_15_cast_fp16)[name = string("transpose_93")]; tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; tensor concat_30 = const()[name = string("concat_30"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_30, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_31 = const()[name = string("concat_31"), val = tensor([-1, 1, 512, 256])]; tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_92")]; tensor reshape_29_cast_fp16 = reshape(shape = concat_31, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor transpose_71_perm_0 = const()[name = string("transpose_71_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = V_for_attn_15_cast_fp16)[name = string("transpose_91")]; tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; tensor concat_32 = const()[name = string("concat_32"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_32, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_33 = const()[name = string("concat_33"), val = tensor([-1, 1, 512, 256])]; tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_90")]; tensor reshape_31_cast_fp16 = reshape(shape = concat_33, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; tensor V_expanded_15_perm_0 = const()[name = string("V_expanded_15_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_29_transpose_x_0 = const()[name = string("attn_weights_29_transpose_x_0"), val = bool(false)]; bool attn_weights_29_transpose_y_0 = const()[name = string("attn_weights_29_transpose_y_0"), val = bool(false)]; tensor transpose_71_cast_fp16 = transpose(perm = transpose_71_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_89")]; tensor attn_weights_29_cast_fp16 = matmul(transpose_x = attn_weights_29_transpose_x_0, transpose_y = attn_weights_29_transpose_y_0, x = q_95_cast_fp16, y = transpose_71_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; tensor x_147_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask_sliding)[name = string("x_147_cast_fp16")]; tensor reduce_max_7_axes_0 = const()[name = string("reduce_max_7_axes_0"), val = tensor([-1])]; bool reduce_max_7_keep_dims_0 = const()[name = string("reduce_max_7_keep_dims_0"), val = bool(true)]; tensor reduce_max_7 = reduce_max(axes = reduce_max_7_axes_0, keep_dims = reduce_max_7_keep_dims_0, x = x_147_cast_fp16)[name = string("reduce_max_7")]; tensor var_5405 = sub(x = x_147_cast_fp16, y = reduce_max_7)[name = string("op_5405")]; tensor var_5411 = exp(x = var_5405)[name = string("op_5411")]; tensor var_5421_axes_0 = const()[name = string("op_5421_axes_0"), val = tensor([-1])]; bool var_5421_keep_dims_0 = const()[name = string("op_5421_keep_dims_0"), val = bool(true)]; tensor var_5421 = reduce_sum(axes = var_5421_axes_0, keep_dims = var_5421_keep_dims_0, x = var_5411)[name = string("op_5421")]; tensor var_5427_cast_fp16 = real_div(x = var_5411, y = var_5421)[name = string("op_5427_cast_fp16")]; bool attn_output_43_transpose_x_0 = const()[name = string("attn_output_43_transpose_x_0"), val = bool(false)]; bool attn_output_43_transpose_y_0 = const()[name = string("attn_output_43_transpose_y_0"), val = bool(false)]; tensor V_expanded_15_cast_fp16 = transpose(perm = V_expanded_15_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_88")]; tensor attn_output_43_cast_fp16 = matmul(transpose_x = attn_output_43_transpose_x_0, transpose_y = attn_output_43_transpose_y_0, x = var_5427_cast_fp16, y = V_expanded_15_cast_fp16)[name = string("attn_output_43_cast_fp16")]; tensor var_5438 = const()[name = string("op_5438"), val = tensor([0, 2, 1, 3])]; tensor var_5445 = const()[name = string("op_5445"), val = tensor([1, 3, -1])]; tensor var_5439_cast_fp16 = transpose(perm = var_5438, x = attn_output_43_cast_fp16)[name = string("transpose_87")]; tensor attn_output_45_cast_fp16 = reshape(shape = var_5445, x = var_5439_cast_fp16)[name = string("attn_output_45_cast_fp16")]; tensor var_5450 = const()[name = string("op_5450"), val = tensor([0, 2, 1])]; string var_5466_pad_type_0 = const()[name = string("op_5466_pad_type_0"), val = string("valid")]; int32 var_5466_groups_0 = const()[name = string("op_5466_groups_0"), val = int32(1)]; tensor var_5466_strides_0 = const()[name = string("op_5466_strides_0"), val = tensor([1])]; tensor var_5466_pad_0 = const()[name = string("op_5466_pad_0"), val = tensor([0, 0])]; tensor var_5466_dilations_0 = const()[name = string("op_5466_dilations_0"), val = tensor([1])]; tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554704576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557326080))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5451_cast_fp16 = transpose(perm = var_5450, x = attn_output_45_cast_fp16)[name = string("transpose_86")]; tensor var_5466_cast_fp16 = conv(dilations = var_5466_dilations_0, groups = var_5466_groups_0, pad = var_5466_pad_0, pad_type = var_5466_pad_type_0, strides = var_5466_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_5451_cast_fp16)[name = string("op_5466_cast_fp16")]; tensor var_5470 = const()[name = string("op_5470"), val = tensor([0, 2, 1])]; int32 var_5476 = const()[name = string("op_5476"), val = int32(-1)]; fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_151_cast_fp16 = transpose(perm = var_5470, x = var_5466_cast_fp16)[name = string("transpose_85")]; tensor var_5478_cast_fp16 = mul(x = x_151_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_5478_cast_fp16")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221_cast_fp16 = concat(axis = var_5476, interleave = input_221_interleave_0, values = (x_151_cast_fp16, var_5478_cast_fp16))[name = string("input_221_cast_fp16")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_5473_to_fp16 = const()[name = string("op_5473_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_5473_to_fp16, x = input_221_cast_fp16)[name = string("normed_209_cast_fp16")]; tensor var_5483_split_sizes_0 = const()[name = string("op_5483_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5483_axis_0 = const()[name = string("op_5483_axis_0"), val = int32(-1)]; tensor var_5483_cast_fp16_0, tensor var_5483_cast_fp16_1 = split(axis = var_5483_axis_0, split_sizes = var_5483_split_sizes_0, x = normed_209_cast_fp16)[name = string("op_5483_cast_fp16")]; tensor layers_7_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557328704)))]; tensor attn_output_47_cast_fp16 = mul(x = var_5483_cast_fp16_0, y = layers_7_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_47_cast_fp16")]; tensor x_153_cast_fp16 = add(x = x_139_cast_fp16, y = attn_output_47_cast_fp16)[name = string("x_153_cast_fp16")]; int32 var_5492 = const()[name = string("op_5492"), val = int32(-1)]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5494_cast_fp16 = mul(x = x_153_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_5494_cast_fp16")]; bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; tensor input_223_cast_fp16 = concat(axis = var_5492, interleave = input_223_interleave_0, values = (x_153_cast_fp16, var_5494_cast_fp16))[name = string("input_223_cast_fp16")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_5489_to_fp16 = const()[name = string("op_5489_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_5489_to_fp16, x = input_223_cast_fp16)[name = string("normed_213_cast_fp16")]; tensor var_5499_split_sizes_0 = const()[name = string("op_5499_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5499_axis_0 = const()[name = string("op_5499_axis_0"), val = int32(-1)]; tensor var_5499_cast_fp16_0, tensor var_5499_cast_fp16_1 = split(axis = var_5499_axis_0, split_sizes = var_5499_split_sizes_0, x = normed_213_cast_fp16)[name = string("op_5499_cast_fp16")]; tensor layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557333888)))]; tensor h_45_cast_fp16 = mul(x = var_5499_cast_fp16_0, y = layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_45_cast_fp16")]; tensor var_5510 = const()[name = string("op_5510"), val = tensor([0, 2, 1])]; tensor input_225_axes_0 = const()[name = string("input_225_axes_0"), val = tensor([2])]; tensor var_5511 = transpose(perm = var_5510, x = h_45_cast_fp16)[name = string("transpose_84")]; tensor input_225 = expand_dims(axes = input_225_axes_0, x = var_5511)[name = string("input_225")]; string gate_29_pad_type_0 = const()[name = string("gate_29_pad_type_0"), val = string("valid")]; tensor gate_29_strides_0 = const()[name = string("gate_29_strides_0"), val = tensor([1, 1])]; tensor gate_29_pad_0 = const()[name = string("gate_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_29_dilations_0 = const()[name = string("gate_29_dilations_0"), val = tensor([1, 1])]; int32 gate_29_groups_0 = const()[name = string("gate_29_groups_0"), val = int32(1)]; tensor gate_29 = conv(dilations = gate_29_dilations_0, groups = gate_29_groups_0, pad = gate_29_pad_0, pad_type = gate_29_pad_type_0, strides = gate_29_strides_0, weight = layers_7_mlp_gate_proj_weight_palettized, x = input_225)[name = string("gate_29")]; string up_15_pad_type_0 = const()[name = string("up_15_pad_type_0"), val = string("valid")]; tensor up_15_strides_0 = const()[name = string("up_15_strides_0"), val = tensor([1, 1])]; tensor up_15_pad_0 = const()[name = string("up_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_15_dilations_0 = const()[name = string("up_15_dilations_0"), val = tensor([1, 1])]; int32 up_15_groups_0 = const()[name = string("up_15_groups_0"), val = int32(1)]; tensor up_15 = conv(dilations = up_15_dilations_0, groups = up_15_groups_0, pad = up_15_pad_0, pad_type = up_15_pad_type_0, strides = up_15_strides_0, weight = layers_7_mlp_up_proj_weight_palettized, x = input_225)[name = string("up_15")]; string gate_31_mode_0 = const()[name = string("gate_31_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_31 = gelu(mode = gate_31_mode_0, x = gate_29)[name = string("gate_31")]; tensor input_227 = mul(x = gate_31, y = up_15)[name = string("input_227")]; string mlp_out_15_pad_type_0 = const()[name = string("mlp_out_15_pad_type_0"), val = string("valid")]; tensor mlp_out_15_strides_0 = const()[name = string("mlp_out_15_strides_0"), val = tensor([1, 1])]; tensor mlp_out_15_pad_0 = const()[name = string("mlp_out_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_15_dilations_0 = const()[name = string("mlp_out_15_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_15_groups_0 = const()[name = string("mlp_out_15_groups_0"), val = int32(1)]; tensor mlp_out_15 = conv(dilations = mlp_out_15_dilations_0, groups = mlp_out_15_groups_0, pad = mlp_out_15_pad_0, pad_type = mlp_out_15_pad_type_0, strides = mlp_out_15_strides_0, weight = layers_7_mlp_down_proj_weight_palettized, x = input_227)[name = string("mlp_out_15")]; tensor var_5551_axes_0 = const()[name = string("op_5551_axes_0"), val = tensor([2])]; tensor var_5551 = squeeze(axes = var_5551_axes_0, x = mlp_out_15)[name = string("op_5551")]; tensor var_5555 = const()[name = string("op_5555"), val = tensor([0, 2, 1])]; int32 var_5561 = const()[name = string("op_5561"), val = int32(-1)]; fp16 const_91_promoted = const()[name = string("const_91_promoted"), val = fp16(-0x1p+0)]; tensor x_155 = transpose(perm = var_5555, x = var_5551)[name = string("transpose_83")]; tensor var_5563 = mul(x = x_155, y = const_91_promoted)[name = string("op_5563")]; bool input_229_interleave_0 = const()[name = string("input_229_interleave_0"), val = bool(false)]; tensor input_229 = concat(axis = var_5561, interleave = input_229_interleave_0, values = (x_155, var_5563))[name = string("input_229")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_5558_to_fp16 = const()[name = string("op_5558_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_5558_to_fp16, x = input_229)[name = string("normed_217_cast_fp16")]; tensor var_5568_split_sizes_0 = const()[name = string("op_5568_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5568_axis_0 = const()[name = string("op_5568_axis_0"), val = int32(-1)]; tensor var_5568_0, tensor var_5568_1 = split(axis = var_5568_axis_0, split_sizes = var_5568_split_sizes_0, x = normed_217_cast_fp16)[name = string("op_5568")]; tensor hidden_states_73 = mul(x = var_5568_0, y = layers_7_post_feedforward_layernorm_weight)[name = string("hidden_states_73")]; tensor hidden_states_75_cast_fp16 = add(x = x_153_cast_fp16, y = hidden_states_73)[name = string("hidden_states_75_cast_fp16")]; tensor per_layer_slice_15_begin_0 = const()[name = string("per_layer_slice_15_begin_0"), val = tensor([0, 0, 4864])]; tensor per_layer_slice_15_end_0 = const()[name = string("per_layer_slice_15_end_0"), val = tensor([1, 3, 5120])]; tensor per_layer_slice_15_end_mask_0 = const()[name = string("per_layer_slice_15_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_15_cast_fp16 = slice_by_index(begin = per_layer_slice_15_begin_0, end = per_layer_slice_15_end_0, end_mask = per_layer_slice_15_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_15_cast_fp16")]; tensor var_5596 = const()[name = string("op_5596"), val = tensor([0, 2, 1])]; tensor input_231_axes_0 = const()[name = string("input_231_axes_0"), val = tensor([2])]; tensor var_5597 = transpose(perm = var_5596, x = hidden_states_75_cast_fp16)[name = string("transpose_82")]; tensor input_231 = expand_dims(axes = input_231_axes_0, x = var_5597)[name = string("input_231")]; string gated_43_pad_type_0 = const()[name = string("gated_43_pad_type_0"), val = string("valid")]; tensor gated_43_strides_0 = const()[name = string("gated_43_strides_0"), val = tensor([1, 1])]; tensor gated_43_pad_0 = const()[name = string("gated_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_43_dilations_0 = const()[name = string("gated_43_dilations_0"), val = tensor([1, 1])]; int32 gated_43_groups_0 = const()[name = string("gated_43_groups_0"), val = int32(1)]; tensor gated_43 = conv(dilations = gated_43_dilations_0, groups = gated_43_groups_0, pad = gated_43_pad_0, pad_type = gated_43_pad_type_0, strides = gated_43_strides_0, weight = layers_7_per_layer_input_gate_weight_palettized, x = input_231)[name = string("gated_43")]; string gated_45_mode_0 = const()[name = string("gated_45_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_45 = gelu(mode = gated_45_mode_0, x = gated_43)[name = string("gated_45")]; tensor var_5616 = const()[name = string("op_5616"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_15_axes_0 = const()[name = string("per_layer_slice_conv_15_axes_0"), val = tensor([2])]; tensor var_5617_cast_fp16 = transpose(perm = var_5616, x = per_layer_slice_15_cast_fp16)[name = string("transpose_81")]; tensor per_layer_slice_conv_15_cast_fp16 = expand_dims(axes = per_layer_slice_conv_15_axes_0, x = var_5617_cast_fp16)[name = string("per_layer_slice_conv_15_cast_fp16")]; tensor input_233_cast_fp16 = mul(x = gated_45, y = per_layer_slice_conv_15_cast_fp16)[name = string("input_233_cast_fp16")]; string gated_47_pad_type_0 = const()[name = string("gated_47_pad_type_0"), val = string("valid")]; tensor gated_47_strides_0 = const()[name = string("gated_47_strides_0"), val = tensor([1, 1])]; tensor gated_47_pad_0 = const()[name = string("gated_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_47_dilations_0 = const()[name = string("gated_47_dilations_0"), val = tensor([1, 1])]; int32 gated_47_groups_0 = const()[name = string("gated_47_groups_0"), val = int32(1)]; tensor layers_7_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557339072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557666816))))[name = string("layers_7_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_47_cast_fp16 = conv(dilations = gated_47_dilations_0, groups = gated_47_groups_0, pad = gated_47_pad_0, pad_type = gated_47_pad_type_0, strides = gated_47_strides_0, weight = layers_7_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_233_cast_fp16)[name = string("gated_47_cast_fp16")]; tensor var_5633_axes_0 = const()[name = string("op_5633_axes_0"), val = tensor([2])]; tensor var_5633_cast_fp16 = squeeze(axes = var_5633_axes_0, x = gated_47_cast_fp16)[name = string("op_5633_cast_fp16")]; tensor var_5637 = const()[name = string("op_5637"), val = tensor([0, 2, 1])]; int32 var_5643 = const()[name = string("op_5643"), val = int32(-1)]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_157_cast_fp16 = transpose(perm = var_5637, x = var_5633_cast_fp16)[name = string("transpose_80")]; tensor var_5645_cast_fp16 = mul(x = x_157_cast_fp16, y = const_92_promoted_to_fp16)[name = string("op_5645_cast_fp16")]; bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; tensor input_235_cast_fp16 = concat(axis = var_5643, interleave = input_235_interleave_0, values = (x_157_cast_fp16, var_5645_cast_fp16))[name = string("input_235_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_5640_to_fp16 = const()[name = string("op_5640_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_5640_to_fp16, x = input_235_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor var_5650_split_sizes_0 = const()[name = string("op_5650_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5650_axis_0 = const()[name = string("op_5650_axis_0"), val = int32(-1)]; tensor var_5650_cast_fp16_0, tensor var_5650_cast_fp16_1 = split(axis = var_5650_axis_0, split_sizes = var_5650_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_5650_cast_fp16")]; tensor layers_7_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557669440)))]; tensor hidden_states_79_cast_fp16 = mul(x = var_5650_cast_fp16_0, y = layers_7_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_79_cast_fp16")]; tensor hidden_states_81_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_79_cast_fp16)[name = string("hidden_states_81_cast_fp16")]; tensor const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = tensor([0x1.06p-1])]; tensor x_159_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_93_promoted_to_fp16)[name = string("x_159_cast_fp16")]; int32 var_5665 = const()[name = string("op_5665"), val = int32(-1)]; fp16 const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5667_cast_fp16 = mul(x = x_159_cast_fp16, y = const_94_promoted_to_fp16)[name = string("op_5667_cast_fp16")]; bool input_237_interleave_0 = const()[name = string("input_237_interleave_0"), val = bool(false)]; tensor input_237_cast_fp16 = concat(axis = var_5665, interleave = input_237_interleave_0, values = (x_159_cast_fp16, var_5667_cast_fp16))[name = string("input_237_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_5662_to_fp16 = const()[name = string("op_5662_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_5662_to_fp16, x = input_237_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_5672_split_sizes_0 = const()[name = string("op_5672_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_5672_axis_0 = const()[name = string("op_5672_axis_0"), val = int32(-1)]; tensor var_5672_cast_fp16_0, tensor var_5672_cast_fp16_1 = split(axis = var_5672_axis_0, split_sizes = var_5672_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_5672_cast_fp16")]; tensor layers_8_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_8_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557674624)))]; tensor h_49_cast_fp16 = mul(x = var_5672_cast_fp16_0, y = layers_8_input_layernorm_weight_promoted_to_fp16)[name = string("h_49_cast_fp16")]; tensor var_5678 = const()[name = string("op_5678"), val = tensor([0, 2, 1])]; tensor var_5681_axes_0 = const()[name = string("op_5681_axes_0"), val = tensor([2])]; tensor var_5679_cast_fp16 = transpose(perm = var_5678, x = h_49_cast_fp16)[name = string("transpose_79")]; tensor var_5681_cast_fp16 = expand_dims(axes = var_5681_axes_0, x = var_5679_cast_fp16)[name = string("op_5681_cast_fp16")]; string q_97_pad_type_0 = const()[name = string("q_97_pad_type_0"), val = string("valid")]; tensor q_97_strides_0 = const()[name = string("q_97_strides_0"), val = tensor([1, 1])]; tensor q_97_pad_0 = const()[name = string("q_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_97_dilations_0 = const()[name = string("q_97_dilations_0"), val = tensor([1, 1])]; int32 q_97_groups_0 = const()[name = string("q_97_groups_0"), val = int32(1)]; tensor q_97 = conv(dilations = q_97_dilations_0, groups = q_97_groups_0, pad = q_97_pad_0, pad_type = q_97_pad_type_0, strides = q_97_strides_0, weight = layers_8_self_attn_q_proj_weight_palettized, x = var_5681_cast_fp16)[name = string("q_97")]; tensor var_5702 = const()[name = string("op_5702"), val = tensor([1, 8, 256, 3])]; tensor var_5703 = reshape(shape = var_5702, x = q_97)[name = string("op_5703")]; tensor transpose_72_perm_0 = const()[name = string("transpose_72_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_5726 = const()[name = string("op_5726"), val = tensor([3, 8, 256])]; tensor transpose_72 = transpose(perm = transpose_72_perm_0, x = var_5703)[name = string("transpose_78")]; tensor x_161 = reshape(shape = var_5726, x = transpose_72)[name = string("x_161")]; int32 var_5732 = const()[name = string("op_5732"), val = int32(-1)]; fp16 const_95_promoted = const()[name = string("const_95_promoted"), val = fp16(-0x1p+0)]; tensor var_5734 = mul(x = x_161, y = const_95_promoted)[name = string("op_5734")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241 = concat(axis = var_5732, interleave = input_241_interleave_0, values = (x_161, var_5734))[name = string("input_241")]; tensor normed_229_axes_0 = const()[name = string("normed_229_axes_0"), val = tensor([-1])]; fp16 var_5729_to_fp16 = const()[name = string("op_5729_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_229_cast_fp16 = layer_norm(axes = normed_229_axes_0, epsilon = var_5729_to_fp16, x = input_241)[name = string("normed_229_cast_fp16")]; tensor var_5739_split_sizes_0 = const()[name = string("op_5739_split_sizes_0"), val = tensor([256, 256])]; int32 var_5739_axis_0 = const()[name = string("op_5739_axis_0"), val = int32(-1)]; tensor var_5739_0, tensor var_5739_1 = split(axis = var_5739_axis_0, split_sizes = var_5739_split_sizes_0, x = normed_229_cast_fp16)[name = string("op_5739")]; tensor var_5746 = const()[name = string("op_5746"), val = tensor([1, 3, 8, 256])]; tensor var_5747 = reshape(shape = var_5746, x = var_5739_0)[name = string("op_5747")]; tensor var_5752 = const()[name = string("op_5752"), val = tensor([0, 2, 1, 3])]; tensor q_103 = transpose(perm = var_5752, x = var_5747)[name = string("transpose_77")]; tensor var_5754_cast_fp16 = mul(x = q_103, y = cos_s)[name = string("op_5754_cast_fp16")]; tensor var_5755_split_sizes_0 = const()[name = string("op_5755_split_sizes_0"), val = tensor([128, 128])]; int32 var_5755_axis_0 = const()[name = string("op_5755_axis_0"), val = int32(-1)]; tensor var_5755_0, tensor var_5755_1 = split(axis = var_5755_axis_0, split_sizes = var_5755_split_sizes_0, x = q_103)[name = string("op_5755")]; fp16 const_96_promoted = const()[name = string("const_96_promoted"), val = fp16(-0x1p+0)]; tensor var_5757 = mul(x = var_5755_1, y = const_96_promoted)[name = string("op_5757")]; int32 var_5759 = const()[name = string("op_5759"), val = int32(-1)]; bool var_5760_interleave_0 = const()[name = string("op_5760_interleave_0"), val = bool(false)]; tensor var_5760 = concat(axis = var_5759, interleave = var_5760_interleave_0, values = (var_5757, var_5755_0))[name = string("op_5760")]; tensor var_5761_cast_fp16 = mul(x = var_5760, y = sin_s)[name = string("op_5761_cast_fp16")]; tensor q_107_cast_fp16 = add(x = var_5754_cast_fp16, y = var_5761_cast_fp16)[name = string("q_107_cast_fp16")]; string k_51_pad_type_0 = const()[name = string("k_51_pad_type_0"), val = string("valid")]; tensor k_51_strides_0 = const()[name = string("k_51_strides_0"), val = tensor([1, 1])]; tensor k_51_pad_0 = const()[name = string("k_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_51_dilations_0 = const()[name = string("k_51_dilations_0"), val = tensor([1, 1])]; int32 k_51_groups_0 = const()[name = string("k_51_groups_0"), val = int32(1)]; tensor k_51 = conv(dilations = k_51_dilations_0, groups = k_51_groups_0, pad = k_51_pad_0, pad_type = k_51_pad_type_0, strides = k_51_strides_0, weight = layers_8_self_attn_k_proj_weight_palettized, x = var_5681_cast_fp16)[name = string("k_51")]; tensor var_5779 = const()[name = string("op_5779"), val = tensor([1, 2, 256, 3])]; tensor var_5780 = reshape(shape = var_5779, x = k_51)[name = string("op_5780")]; tensor transpose_73_perm_0 = const()[name = string("transpose_73_perm_0"), val = tensor([0, 3, 1, 2])]; string v_19_pad_type_0 = const()[name = string("v_19_pad_type_0"), val = string("valid")]; tensor v_19_strides_0 = const()[name = string("v_19_strides_0"), val = tensor([1, 1])]; tensor v_19_pad_0 = const()[name = string("v_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_19_dilations_0 = const()[name = string("v_19_dilations_0"), val = tensor([1, 1])]; int32 v_19_groups_0 = const()[name = string("v_19_groups_0"), val = int32(1)]; tensor v_19 = conv(dilations = v_19_dilations_0, groups = v_19_groups_0, pad = v_19_pad_0, pad_type = v_19_pad_type_0, strides = v_19_strides_0, weight = layers_8_self_attn_v_proj_weight_palettized, x = var_5681_cast_fp16)[name = string("v_19")]; tensor var_5807 = const()[name = string("op_5807"), val = tensor([1, 2, 256, 3])]; tensor var_5808 = reshape(shape = var_5807, x = v_19)[name = string("op_5808")]; tensor var_5813 = const()[name = string("op_5813"), val = tensor([0, 1, 3, 2])]; tensor var_5831 = const()[name = string("op_5831"), val = tensor([3, 2, 256])]; tensor transpose_73 = transpose(perm = transpose_73_perm_0, x = var_5780)[name = string("transpose_76")]; tensor x_163 = reshape(shape = var_5831, x = transpose_73)[name = string("x_163")]; int32 var_5837 = const()[name = string("op_5837"), val = int32(-1)]; fp16 const_97_promoted = const()[name = string("const_97_promoted"), val = fp16(-0x1p+0)]; tensor var_5839 = mul(x = x_163, y = const_97_promoted)[name = string("op_5839")]; bool input_243_interleave_0 = const()[name = string("input_243_interleave_0"), val = bool(false)]; tensor input_243 = concat(axis = var_5837, interleave = input_243_interleave_0, values = (x_163, var_5839))[name = string("input_243")]; tensor normed_233_axes_0 = const()[name = string("normed_233_axes_0"), val = tensor([-1])]; fp16 var_5834_to_fp16 = const()[name = string("op_5834_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_233_cast_fp16 = layer_norm(axes = normed_233_axes_0, epsilon = var_5834_to_fp16, x = input_243)[name = string("normed_233_cast_fp16")]; tensor var_5844_split_sizes_0 = const()[name = string("op_5844_split_sizes_0"), val = tensor([256, 256])]; int32 var_5844_axis_0 = const()[name = string("op_5844_axis_0"), val = int32(-1)]; tensor var_5844_0, tensor var_5844_1 = split(axis = var_5844_axis_0, split_sizes = var_5844_split_sizes_0, x = normed_233_cast_fp16)[name = string("op_5844")]; tensor k_55 = mul(x = var_5844_0, y = layers_8_self_attn_k_norm_weight)[name = string("k_55")]; tensor var_5851 = const()[name = string("op_5851"), val = tensor([1, 3, 2, 256])]; tensor var_5852 = reshape(shape = var_5851, x = k_55)[name = string("op_5852")]; tensor var_5857 = const()[name = string("op_5857"), val = tensor([0, 2, 1, 3])]; fp16 var_5859_promoted = const()[name = string("op_5859_promoted"), val = fp16(0x1p+1)]; tensor var_5814 = transpose(perm = var_5813, x = var_5808)[name = string("transpose_75")]; tensor var_5860 = pow(x = var_5814, y = var_5859_promoted)[name = string("op_5860")]; tensor var_5865_axes_0 = const()[name = string("op_5865_axes_0"), val = tensor([-1])]; bool var_5865_keep_dims_0 = const()[name = string("op_5865_keep_dims_0"), val = bool(true)]; tensor var_5865 = reduce_mean(axes = var_5865_axes_0, keep_dims = var_5865_keep_dims_0, x = var_5860)[name = string("op_5865")]; fp16 var_5867_to_fp16 = const()[name = string("op_5867_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_17_cast_fp16 = add(x = var_5865, y = var_5867_to_fp16)[name = string("mean_sq_17_cast_fp16")]; fp32 var_5869_epsilon_0 = const()[name = string("op_5869_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_5869_cast_fp16 = rsqrt(epsilon = var_5869_epsilon_0, x = mean_sq_17_cast_fp16)[name = string("op_5869_cast_fp16")]; tensor input_247_cast_fp16 = mul(x = var_5814, y = var_5869_cast_fp16)[name = string("input_247_cast_fp16")]; tensor q_105 = transpose(perm = var_5857, x = var_5852)[name = string("transpose_74")]; tensor var_5871_cast_fp16 = mul(x = q_105, y = cos_s)[name = string("op_5871_cast_fp16")]; tensor var_5872_split_sizes_0 = const()[name = string("op_5872_split_sizes_0"), val = tensor([128, 128])]; int32 var_5872_axis_0 = const()[name = string("op_5872_axis_0"), val = int32(-1)]; tensor var_5872_0, tensor var_5872_1 = split(axis = var_5872_axis_0, split_sizes = var_5872_split_sizes_0, x = q_105)[name = string("op_5872")]; fp16 const_98_promoted = const()[name = string("const_98_promoted"), val = fp16(-0x1p+0)]; tensor var_5874 = mul(x = var_5872_1, y = const_98_promoted)[name = string("op_5874")]; int32 var_5876 = const()[name = string("op_5876"), val = int32(-1)]; bool var_5877_interleave_0 = const()[name = string("op_5877_interleave_0"), val = bool(false)]; tensor var_5877 = concat(axis = var_5876, interleave = var_5877_interleave_0, values = (var_5874, var_5872_0))[name = string("op_5877")]; tensor var_5878_cast_fp16 = mul(x = var_5877, y = sin_s)[name = string("op_5878_cast_fp16")]; tensor input_245_cast_fp16 = add(x = var_5871_cast_fp16, y = var_5878_cast_fp16)[name = string("input_245_cast_fp16")]; tensor k_padded_15_pad_0 = const()[name = string("k_padded_15_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_15_mode_0 = const()[name = string("k_padded_15_mode_0"), val = string("constant")]; fp16 const_99_to_fp16 = const()[name = string("const_99_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_15_cast_fp16 = pad(constant_val = const_99_to_fp16, mode = k_padded_15_mode_0, pad = k_padded_15_pad_0, x = input_245_cast_fp16)[name = string("k_padded_15_cast_fp16")]; tensor v_padded_15_pad_0 = const()[name = string("v_padded_15_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_15_mode_0 = const()[name = string("v_padded_15_mode_0"), val = string("constant")]; fp16 const_100_to_fp16 = const()[name = string("const_100_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_15_cast_fp16 = pad(constant_val = const_100_to_fp16, mode = v_padded_15_mode_0, pad = v_padded_15_pad_0, x = input_247_cast_fp16)[name = string("v_padded_15_cast_fp16")]; tensor slot_k_17_begin_0 = const()[name = string("slot_k_17_begin_0"), val = tensor([7, 0, 0, 0])]; tensor slot_k_17_end_0 = const()[name = string("slot_k_17_end_0"), val = tensor([8, 2, 512, 512])]; tensor slot_k_17_end_mask_0 = const()[name = string("slot_k_17_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_17_cast_fp16 = slice_by_index(begin = slot_k_17_begin_0, end = slot_k_17_end_0, end_mask = slot_k_17_end_mask_0, x = K_sliding_out_13_cast_fp16)[name = string("slot_k_17_cast_fp16")]; tensor slot_v_17_begin_0 = const()[name = string("slot_v_17_begin_0"), val = tensor([7, 0, 0, 0])]; tensor slot_v_17_end_0 = const()[name = string("slot_v_17_end_0"), val = tensor([8, 2, 512, 512])]; tensor slot_v_17_end_mask_0 = const()[name = string("slot_v_17_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_17_cast_fp16 = slice_by_index(begin = slot_v_17_begin_0, end = slot_v_17_end_0, end_mask = slot_v_17_end_mask_0, x = V_sliding_out_13_cast_fp16)[name = string("slot_v_17_cast_fp16")]; tensor var_5917_begin_0 = const()[name = string("op_5917_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_5917_end_0 = const()[name = string("op_5917_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5917_end_mask_0 = const()[name = string("op_5917_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5917_cast_fp16 = slice_by_index(begin = var_5917_begin_0, end = var_5917_end_0, end_mask = var_5917_end_mask_0, x = slot_k_17_cast_fp16)[name = string("op_5917_cast_fp16")]; int32 var_5924 = const()[name = string("op_5924"), val = int32(2)]; bool new_k_17_interleave_0 = const()[name = string("new_k_17_interleave_0"), val = bool(false)]; tensor new_k_17_cast_fp16 = concat(axis = var_5924, interleave = new_k_17_interleave_0, values = (var_5917_cast_fp16, k_padded_15_cast_fp16))[name = string("new_k_17_cast_fp16")]; tensor var_5940_begin_0 = const()[name = string("op_5940_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_5940_end_0 = const()[name = string("op_5940_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_5940_end_mask_0 = const()[name = string("op_5940_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5940_cast_fp16 = slice_by_index(begin = var_5940_begin_0, end = var_5940_end_0, end_mask = var_5940_end_mask_0, x = slot_v_17_cast_fp16)[name = string("op_5940_cast_fp16")]; int32 var_5947 = const()[name = string("op_5947"), val = int32(2)]; bool new_v_17_interleave_0 = const()[name = string("new_v_17_interleave_0"), val = bool(false)]; tensor new_v_17_cast_fp16 = concat(axis = var_5947, interleave = new_v_17_interleave_0, values = (var_5940_cast_fp16, v_padded_15_cast_fp16))[name = string("new_v_17_cast_fp16")]; tensor var_5953_begin_0 = const()[name = string("op_5953_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5953_end_0 = const()[name = string("op_5953_end_0"), val = tensor([7, 2, 512, 512])]; tensor var_5953_end_mask_0 = const()[name = string("op_5953_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5953_cast_fp16 = slice_by_index(begin = var_5953_begin_0, end = var_5953_end_0, end_mask = var_5953_end_mask_0, x = K_sliding_out_13_cast_fp16)[name = string("op_5953_cast_fp16")]; tensor var_5958_begin_0 = const()[name = string("op_5958_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_5958_end_0 = const()[name = string("op_5958_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_5958_end_mask_0 = const()[name = string("op_5958_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5958_cast_fp16 = slice_by_index(begin = var_5958_begin_0, end = var_5958_end_0, end_mask = var_5958_end_mask_0, x = K_sliding_out_13_cast_fp16)[name = string("op_5958_cast_fp16")]; int32 var_5960 = const()[name = string("op_5960"), val = int32(0)]; bool K_sliding_out_15_interleave_0 = const()[name = string("K_sliding_out_15_interleave_0"), val = bool(false)]; tensor K_sliding_out_15_cast_fp16 = concat(axis = var_5960, interleave = K_sliding_out_15_interleave_0, values = (var_5953_cast_fp16, new_k_17_cast_fp16, var_5958_cast_fp16))[name = string("K_sliding_out_15_cast_fp16")]; tensor var_5966_begin_0 = const()[name = string("op_5966_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5966_end_0 = const()[name = string("op_5966_end_0"), val = tensor([7, 2, 512, 512])]; tensor var_5966_end_mask_0 = const()[name = string("op_5966_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5966_cast_fp16 = slice_by_index(begin = var_5966_begin_0, end = var_5966_end_0, end_mask = var_5966_end_mask_0, x = V_sliding_out_13_cast_fp16)[name = string("op_5966_cast_fp16")]; tensor var_5971_begin_0 = const()[name = string("op_5971_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_5971_end_0 = const()[name = string("op_5971_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_5971_end_mask_0 = const()[name = string("op_5971_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5971_cast_fp16 = slice_by_index(begin = var_5971_begin_0, end = var_5971_end_0, end_mask = var_5971_end_mask_0, x = V_sliding_out_13_cast_fp16)[name = string("op_5971_cast_fp16")]; int32 var_5973 = const()[name = string("op_5973"), val = int32(0)]; bool V_sliding_out_15_interleave_0 = const()[name = string("V_sliding_out_15_interleave_0"), val = bool(false)]; tensor V_sliding_out_15_cast_fp16 = concat(axis = var_5973, interleave = V_sliding_out_15_interleave_0, values = (var_5966_cast_fp16, new_v_17_cast_fp16, var_5971_cast_fp16))[name = string("V_sliding_out_15_cast_fp16")]; tensor var_5979_begin_0 = const()[name = string("op_5979_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5979_end_0 = const()[name = string("op_5979_end_0"), val = tensor([8, 2, 512, 512])]; tensor var_5979_end_mask_0 = const()[name = string("op_5979_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5979_cast_fp16 = slice_by_index(begin = var_5979_begin_0, end = var_5979_end_0, end_mask = var_5979_end_mask_0, x = K_sliding_out_15_cast_fp16)[name = string("op_5979_cast_fp16")]; tensor K_for_attn_17_begin_0 = const()[name = string("K_for_attn_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_17_end_0 = const()[name = string("K_for_attn_17_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_17_end_mask_0 = const()[name = string("K_for_attn_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_17_cast_fp16 = slice_by_index(begin = K_for_attn_17_begin_0, end = K_for_attn_17_end_0, end_mask = K_for_attn_17_end_mask_0, x = var_5979_cast_fp16)[name = string("K_for_attn_17_cast_fp16")]; tensor var_5989_begin_0 = const()[name = string("op_5989_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_5989_end_0 = const()[name = string("op_5989_end_0"), val = tensor([8, 2, 512, 512])]; tensor var_5989_end_mask_0 = const()[name = string("op_5989_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5989_cast_fp16 = slice_by_index(begin = var_5989_begin_0, end = var_5989_end_0, end_mask = var_5989_end_mask_0, x = V_sliding_out_15_cast_fp16)[name = string("op_5989_cast_fp16")]; tensor V_for_attn_17_begin_0 = const()[name = string("V_for_attn_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_17_end_0 = const()[name = string("V_for_attn_17_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_17_end_mask_0 = const()[name = string("V_for_attn_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_17_cast_fp16 = slice_by_index(begin = V_for_attn_17_begin_0, end = V_for_attn_17_end_0, end_mask = V_for_attn_17_end_mask_0, x = var_5989_cast_fp16)[name = string("V_for_attn_17_cast_fp16")]; tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_16_reps_0 = const()[name = string("tile_16_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = K_for_attn_17_cast_fp16)[name = string("transpose_73")]; tensor tile_16_cast_fp16 = tile(reps = tile_16_reps_0, x = transpose_32_cast_fp16)[name = string("tile_16_cast_fp16")]; tensor concat_34 = const()[name = string("concat_34"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_32_cast_fp16 = reshape(shape = concat_34, x = tile_16_cast_fp16)[name = string("reshape_32_cast_fp16")]; tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([-1, 1, 512, 256])]; tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_72")]; tensor reshape_33_cast_fp16 = reshape(shape = concat_35, x = transpose_33_cast_fp16)[name = string("reshape_33_cast_fp16")]; tensor transpose_74_perm_0 = const()[name = string("transpose_74_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_17_reps_0 = const()[name = string("tile_17_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = V_for_attn_17_cast_fp16)[name = string("transpose_71")]; tensor tile_17_cast_fp16 = tile(reps = tile_17_reps_0, x = transpose_34_cast_fp16)[name = string("tile_17_cast_fp16")]; tensor concat_36 = const()[name = string("concat_36"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_34_cast_fp16 = reshape(shape = concat_36, x = tile_17_cast_fp16)[name = string("reshape_34_cast_fp16")]; tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_37 = const()[name = string("concat_37"), val = tensor([-1, 1, 512, 256])]; tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_34_cast_fp16)[name = string("transpose_70")]; tensor reshape_35_cast_fp16 = reshape(shape = concat_37, x = transpose_35_cast_fp16)[name = string("reshape_35_cast_fp16")]; tensor V_expanded_17_perm_0 = const()[name = string("V_expanded_17_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor transpose_74_cast_fp16 = transpose(perm = transpose_74_perm_0, x = reshape_33_cast_fp16)[name = string("transpose_69")]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = q_107_cast_fp16, y = transpose_74_cast_fp16)[name = string("attn_weights_33_cast_fp16")]; tensor x_167_cast_fp16 = add(x = attn_weights_33_cast_fp16, y = causal_mask_sliding)[name = string("x_167_cast_fp16")]; tensor reduce_max_8_axes_0 = const()[name = string("reduce_max_8_axes_0"), val = tensor([-1])]; bool reduce_max_8_keep_dims_0 = const()[name = string("reduce_max_8_keep_dims_0"), val = bool(true)]; tensor reduce_max_8 = reduce_max(axes = reduce_max_8_axes_0, keep_dims = reduce_max_8_keep_dims_0, x = x_167_cast_fp16)[name = string("reduce_max_8")]; tensor var_6024 = sub(x = x_167_cast_fp16, y = reduce_max_8)[name = string("op_6024")]; tensor var_6030 = exp(x = var_6024)[name = string("op_6030")]; tensor var_6040_axes_0 = const()[name = string("op_6040_axes_0"), val = tensor([-1])]; bool var_6040_keep_dims_0 = const()[name = string("op_6040_keep_dims_0"), val = bool(true)]; tensor var_6040 = reduce_sum(axes = var_6040_axes_0, keep_dims = var_6040_keep_dims_0, x = var_6030)[name = string("op_6040")]; tensor var_6046_cast_fp16 = real_div(x = var_6030, y = var_6040)[name = string("op_6046_cast_fp16")]; bool attn_output_49_transpose_x_0 = const()[name = string("attn_output_49_transpose_x_0"), val = bool(false)]; bool attn_output_49_transpose_y_0 = const()[name = string("attn_output_49_transpose_y_0"), val = bool(false)]; tensor V_expanded_17_cast_fp16 = transpose(perm = V_expanded_17_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_68")]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = var_6046_cast_fp16, y = V_expanded_17_cast_fp16)[name = string("attn_output_49_cast_fp16")]; tensor var_6057 = const()[name = string("op_6057"), val = tensor([0, 2, 1, 3])]; tensor var_6064 = const()[name = string("op_6064"), val = tensor([1, 3, -1])]; tensor var_6058_cast_fp16 = transpose(perm = var_6057, x = attn_output_49_cast_fp16)[name = string("transpose_67")]; tensor attn_output_51_cast_fp16 = reshape(shape = var_6064, x = var_6058_cast_fp16)[name = string("attn_output_51_cast_fp16")]; tensor var_6069 = const()[name = string("op_6069"), val = tensor([0, 2, 1])]; string var_6085_pad_type_0 = const()[name = string("op_6085_pad_type_0"), val = string("valid")]; int32 var_6085_groups_0 = const()[name = string("op_6085_groups_0"), val = int32(1)]; tensor var_6085_strides_0 = const()[name = string("op_6085_strides_0"), val = tensor([1])]; tensor var_6085_pad_0 = const()[name = string("op_6085_pad_0"), val = tensor([0, 0])]; tensor var_6085_dilations_0 = const()[name = string("op_6085_dilations_0"), val = tensor([1])]; tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557679808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560301312))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6070_cast_fp16 = transpose(perm = var_6069, x = attn_output_51_cast_fp16)[name = string("transpose_66")]; tensor var_6085_cast_fp16 = conv(dilations = var_6085_dilations_0, groups = var_6085_groups_0, pad = var_6085_pad_0, pad_type = var_6085_pad_type_0, strides = var_6085_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_6070_cast_fp16)[name = string("op_6085_cast_fp16")]; tensor var_6089 = const()[name = string("op_6089"), val = tensor([0, 2, 1])]; int32 var_6095 = const()[name = string("op_6095"), val = int32(-1)]; fp16 const_101_promoted_to_fp16 = const()[name = string("const_101_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_171_cast_fp16 = transpose(perm = var_6089, x = var_6085_cast_fp16)[name = string("transpose_65")]; tensor var_6097_cast_fp16 = mul(x = x_171_cast_fp16, y = const_101_promoted_to_fp16)[name = string("op_6097_cast_fp16")]; bool input_251_interleave_0 = const()[name = string("input_251_interleave_0"), val = bool(false)]; tensor input_251_cast_fp16 = concat(axis = var_6095, interleave = input_251_interleave_0, values = (x_171_cast_fp16, var_6097_cast_fp16))[name = string("input_251_cast_fp16")]; tensor normed_237_axes_0 = const()[name = string("normed_237_axes_0"), val = tensor([-1])]; fp16 var_6092_to_fp16 = const()[name = string("op_6092_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_237_cast_fp16 = layer_norm(axes = normed_237_axes_0, epsilon = var_6092_to_fp16, x = input_251_cast_fp16)[name = string("normed_237_cast_fp16")]; tensor var_6102_split_sizes_0 = const()[name = string("op_6102_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6102_axis_0 = const()[name = string("op_6102_axis_0"), val = int32(-1)]; tensor var_6102_cast_fp16_0, tensor var_6102_cast_fp16_1 = split(axis = var_6102_axis_0, split_sizes = var_6102_split_sizes_0, x = normed_237_cast_fp16)[name = string("op_6102_cast_fp16")]; tensor layers_8_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_8_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560303936)))]; tensor attn_output_53_cast_fp16 = mul(x = var_6102_cast_fp16_0, y = layers_8_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_53_cast_fp16")]; tensor x_173_cast_fp16 = add(x = x_159_cast_fp16, y = attn_output_53_cast_fp16)[name = string("x_173_cast_fp16")]; int32 var_6111 = const()[name = string("op_6111"), val = int32(-1)]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6113_cast_fp16 = mul(x = x_173_cast_fp16, y = const_102_promoted_to_fp16)[name = string("op_6113_cast_fp16")]; bool input_253_interleave_0 = const()[name = string("input_253_interleave_0"), val = bool(false)]; tensor input_253_cast_fp16 = concat(axis = var_6111, interleave = input_253_interleave_0, values = (x_173_cast_fp16, var_6113_cast_fp16))[name = string("input_253_cast_fp16")]; tensor normed_241_axes_0 = const()[name = string("normed_241_axes_0"), val = tensor([-1])]; fp16 var_6108_to_fp16 = const()[name = string("op_6108_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_241_cast_fp16 = layer_norm(axes = normed_241_axes_0, epsilon = var_6108_to_fp16, x = input_253_cast_fp16)[name = string("normed_241_cast_fp16")]; tensor var_6118_split_sizes_0 = const()[name = string("op_6118_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6118_axis_0 = const()[name = string("op_6118_axis_0"), val = int32(-1)]; tensor var_6118_cast_fp16_0, tensor var_6118_cast_fp16_1 = split(axis = var_6118_axis_0, split_sizes = var_6118_split_sizes_0, x = normed_241_cast_fp16)[name = string("op_6118_cast_fp16")]; tensor layers_8_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_8_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560309120)))]; tensor h_51_cast_fp16 = mul(x = var_6118_cast_fp16_0, y = layers_8_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_51_cast_fp16")]; tensor var_6129 = const()[name = string("op_6129"), val = tensor([0, 2, 1])]; tensor input_255_axes_0 = const()[name = string("input_255_axes_0"), val = tensor([2])]; tensor var_6130 = transpose(perm = var_6129, x = h_51_cast_fp16)[name = string("transpose_64")]; tensor input_255 = expand_dims(axes = input_255_axes_0, x = var_6130)[name = string("input_255")]; string gate_33_pad_type_0 = const()[name = string("gate_33_pad_type_0"), val = string("valid")]; tensor gate_33_strides_0 = const()[name = string("gate_33_strides_0"), val = tensor([1, 1])]; tensor gate_33_pad_0 = const()[name = string("gate_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_33_dilations_0 = const()[name = string("gate_33_dilations_0"), val = tensor([1, 1])]; int32 gate_33_groups_0 = const()[name = string("gate_33_groups_0"), val = int32(1)]; tensor gate_33 = conv(dilations = gate_33_dilations_0, groups = gate_33_groups_0, pad = gate_33_pad_0, pad_type = gate_33_pad_type_0, strides = gate_33_strides_0, weight = layers_8_mlp_gate_proj_weight_palettized, x = input_255)[name = string("gate_33")]; string up_17_pad_type_0 = const()[name = string("up_17_pad_type_0"), val = string("valid")]; tensor up_17_strides_0 = const()[name = string("up_17_strides_0"), val = tensor([1, 1])]; tensor up_17_pad_0 = const()[name = string("up_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_17_dilations_0 = const()[name = string("up_17_dilations_0"), val = tensor([1, 1])]; int32 up_17_groups_0 = const()[name = string("up_17_groups_0"), val = int32(1)]; tensor up_17 = conv(dilations = up_17_dilations_0, groups = up_17_groups_0, pad = up_17_pad_0, pad_type = up_17_pad_type_0, strides = up_17_strides_0, weight = layers_8_mlp_up_proj_weight_palettized, x = input_255)[name = string("up_17")]; string gate_35_mode_0 = const()[name = string("gate_35_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_35 = gelu(mode = gate_35_mode_0, x = gate_33)[name = string("gate_35")]; tensor input_257 = mul(x = gate_35, y = up_17)[name = string("input_257")]; string mlp_out_17_pad_type_0 = const()[name = string("mlp_out_17_pad_type_0"), val = string("valid")]; tensor mlp_out_17_strides_0 = const()[name = string("mlp_out_17_strides_0"), val = tensor([1, 1])]; tensor mlp_out_17_pad_0 = const()[name = string("mlp_out_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_17_dilations_0 = const()[name = string("mlp_out_17_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_17_groups_0 = const()[name = string("mlp_out_17_groups_0"), val = int32(1)]; tensor mlp_out_17 = conv(dilations = mlp_out_17_dilations_0, groups = mlp_out_17_groups_0, pad = mlp_out_17_pad_0, pad_type = mlp_out_17_pad_type_0, strides = mlp_out_17_strides_0, weight = layers_8_mlp_down_proj_weight_palettized, x = input_257)[name = string("mlp_out_17")]; tensor var_6170_axes_0 = const()[name = string("op_6170_axes_0"), val = tensor([2])]; tensor var_6170 = squeeze(axes = var_6170_axes_0, x = mlp_out_17)[name = string("op_6170")]; tensor var_6174 = const()[name = string("op_6174"), val = tensor([0, 2, 1])]; int32 var_6180 = const()[name = string("op_6180"), val = int32(-1)]; fp16 const_103_promoted = const()[name = string("const_103_promoted"), val = fp16(-0x1p+0)]; tensor x_175 = transpose(perm = var_6174, x = var_6170)[name = string("transpose_63")]; tensor var_6182 = mul(x = x_175, y = const_103_promoted)[name = string("op_6182")]; bool input_259_interleave_0 = const()[name = string("input_259_interleave_0"), val = bool(false)]; tensor input_259 = concat(axis = var_6180, interleave = input_259_interleave_0, values = (x_175, var_6182))[name = string("input_259")]; tensor normed_245_axes_0 = const()[name = string("normed_245_axes_0"), val = tensor([-1])]; fp16 var_6177_to_fp16 = const()[name = string("op_6177_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_245_cast_fp16 = layer_norm(axes = normed_245_axes_0, epsilon = var_6177_to_fp16, x = input_259)[name = string("normed_245_cast_fp16")]; tensor var_6187_split_sizes_0 = const()[name = string("op_6187_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6187_axis_0 = const()[name = string("op_6187_axis_0"), val = int32(-1)]; tensor var_6187_0, tensor var_6187_1 = split(axis = var_6187_axis_0, split_sizes = var_6187_split_sizes_0, x = normed_245_cast_fp16)[name = string("op_6187")]; tensor hidden_states_83 = mul(x = var_6187_0, y = layers_8_post_feedforward_layernorm_weight)[name = string("hidden_states_83")]; tensor hidden_states_85_cast_fp16 = add(x = x_173_cast_fp16, y = hidden_states_83)[name = string("hidden_states_85_cast_fp16")]; tensor per_layer_slice_17_begin_0 = const()[name = string("per_layer_slice_17_begin_0"), val = tensor([0, 0, 5120])]; tensor per_layer_slice_17_end_0 = const()[name = string("per_layer_slice_17_end_0"), val = tensor([1, 3, 5376])]; tensor per_layer_slice_17_end_mask_0 = const()[name = string("per_layer_slice_17_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_17_cast_fp16 = slice_by_index(begin = per_layer_slice_17_begin_0, end = per_layer_slice_17_end_0, end_mask = per_layer_slice_17_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_17_cast_fp16")]; tensor var_6215 = const()[name = string("op_6215"), val = tensor([0, 2, 1])]; tensor input_261_axes_0 = const()[name = string("input_261_axes_0"), val = tensor([2])]; tensor var_6216 = transpose(perm = var_6215, x = hidden_states_85_cast_fp16)[name = string("transpose_62")]; tensor input_261 = expand_dims(axes = input_261_axes_0, x = var_6216)[name = string("input_261")]; string gated_49_pad_type_0 = const()[name = string("gated_49_pad_type_0"), val = string("valid")]; tensor gated_49_strides_0 = const()[name = string("gated_49_strides_0"), val = tensor([1, 1])]; tensor gated_49_pad_0 = const()[name = string("gated_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_49_dilations_0 = const()[name = string("gated_49_dilations_0"), val = tensor([1, 1])]; int32 gated_49_groups_0 = const()[name = string("gated_49_groups_0"), val = int32(1)]; tensor gated_49 = conv(dilations = gated_49_dilations_0, groups = gated_49_groups_0, pad = gated_49_pad_0, pad_type = gated_49_pad_type_0, strides = gated_49_strides_0, weight = layers_8_per_layer_input_gate_weight_palettized, x = input_261)[name = string("gated_49")]; string gated_51_mode_0 = const()[name = string("gated_51_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_51 = gelu(mode = gated_51_mode_0, x = gated_49)[name = string("gated_51")]; tensor var_6235 = const()[name = string("op_6235"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_17_axes_0 = const()[name = string("per_layer_slice_conv_17_axes_0"), val = tensor([2])]; tensor var_6236_cast_fp16 = transpose(perm = var_6235, x = per_layer_slice_17_cast_fp16)[name = string("transpose_61")]; tensor per_layer_slice_conv_17_cast_fp16 = expand_dims(axes = per_layer_slice_conv_17_axes_0, x = var_6236_cast_fp16)[name = string("per_layer_slice_conv_17_cast_fp16")]; tensor input_263_cast_fp16 = mul(x = gated_51, y = per_layer_slice_conv_17_cast_fp16)[name = string("input_263_cast_fp16")]; string gated_53_pad_type_0 = const()[name = string("gated_53_pad_type_0"), val = string("valid")]; tensor gated_53_strides_0 = const()[name = string("gated_53_strides_0"), val = tensor([1, 1])]; tensor gated_53_pad_0 = const()[name = string("gated_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_53_dilations_0 = const()[name = string("gated_53_dilations_0"), val = tensor([1, 1])]; int32 gated_53_groups_0 = const()[name = string("gated_53_groups_0"), val = int32(1)]; tensor layers_8_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560314304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560642048))))[name = string("layers_8_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_53_cast_fp16 = conv(dilations = gated_53_dilations_0, groups = gated_53_groups_0, pad = gated_53_pad_0, pad_type = gated_53_pad_type_0, strides = gated_53_strides_0, weight = layers_8_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_263_cast_fp16)[name = string("gated_53_cast_fp16")]; tensor var_6252_axes_0 = const()[name = string("op_6252_axes_0"), val = tensor([2])]; tensor var_6252_cast_fp16 = squeeze(axes = var_6252_axes_0, x = gated_53_cast_fp16)[name = string("op_6252_cast_fp16")]; tensor var_6256 = const()[name = string("op_6256"), val = tensor([0, 2, 1])]; int32 var_6262 = const()[name = string("op_6262"), val = int32(-1)]; fp16 const_104_promoted_to_fp16 = const()[name = string("const_104_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_177_cast_fp16 = transpose(perm = var_6256, x = var_6252_cast_fp16)[name = string("transpose_60")]; tensor var_6264_cast_fp16 = mul(x = x_177_cast_fp16, y = const_104_promoted_to_fp16)[name = string("op_6264_cast_fp16")]; bool input_265_interleave_0 = const()[name = string("input_265_interleave_0"), val = bool(false)]; tensor input_265_cast_fp16 = concat(axis = var_6262, interleave = input_265_interleave_0, values = (x_177_cast_fp16, var_6264_cast_fp16))[name = string("input_265_cast_fp16")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; fp16 var_6259_to_fp16 = const()[name = string("op_6259_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_6259_to_fp16, x = input_265_cast_fp16)[name = string("normed_249_cast_fp16")]; tensor var_6269_split_sizes_0 = const()[name = string("op_6269_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6269_axis_0 = const()[name = string("op_6269_axis_0"), val = int32(-1)]; tensor var_6269_cast_fp16_0, tensor var_6269_cast_fp16_1 = split(axis = var_6269_axis_0, split_sizes = var_6269_split_sizes_0, x = normed_249_cast_fp16)[name = string("op_6269_cast_fp16")]; tensor layers_8_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_8_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560644672)))]; tensor hidden_states_89_cast_fp16 = mul(x = var_6269_cast_fp16_0, y = layers_8_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_89_cast_fp16")]; tensor hidden_states_91_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_89_cast_fp16)[name = string("hidden_states_91_cast_fp16")]; tensor const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = tensor([0x1.bap-2])]; tensor x_179_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_105_promoted_to_fp16)[name = string("x_179_cast_fp16")]; int32 var_6284 = const()[name = string("op_6284"), val = int32(-1)]; fp16 const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6286_cast_fp16 = mul(x = x_179_cast_fp16, y = const_106_promoted_to_fp16)[name = string("op_6286_cast_fp16")]; bool input_267_interleave_0 = const()[name = string("input_267_interleave_0"), val = bool(false)]; tensor input_267_cast_fp16 = concat(axis = var_6284, interleave = input_267_interleave_0, values = (x_179_cast_fp16, var_6286_cast_fp16))[name = string("input_267_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; fp16 var_6281_to_fp16 = const()[name = string("op_6281_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_6281_to_fp16, x = input_267_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor var_6291_split_sizes_0 = const()[name = string("op_6291_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6291_axis_0 = const()[name = string("op_6291_axis_0"), val = int32(-1)]; tensor var_6291_cast_fp16_0, tensor var_6291_cast_fp16_1 = split(axis = var_6291_axis_0, split_sizes = var_6291_split_sizes_0, x = normed_253_cast_fp16)[name = string("op_6291_cast_fp16")]; tensor layers_9_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_9_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560649856)))]; tensor h_55_cast_fp16 = mul(x = var_6291_cast_fp16_0, y = layers_9_input_layernorm_weight_promoted_to_fp16)[name = string("h_55_cast_fp16")]; tensor var_6297 = const()[name = string("op_6297"), val = tensor([0, 2, 1])]; tensor var_6300_axes_0 = const()[name = string("op_6300_axes_0"), val = tensor([2])]; tensor var_6298_cast_fp16 = transpose(perm = var_6297, x = h_55_cast_fp16)[name = string("transpose_59")]; tensor var_6300_cast_fp16 = expand_dims(axes = var_6300_axes_0, x = var_6298_cast_fp16)[name = string("op_6300_cast_fp16")]; string q_109_pad_type_0 = const()[name = string("q_109_pad_type_0"), val = string("valid")]; tensor q_109_strides_0 = const()[name = string("q_109_strides_0"), val = tensor([1, 1])]; tensor q_109_pad_0 = const()[name = string("q_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_109_dilations_0 = const()[name = string("q_109_dilations_0"), val = tensor([1, 1])]; int32 q_109_groups_0 = const()[name = string("q_109_groups_0"), val = int32(1)]; tensor q_109 = conv(dilations = q_109_dilations_0, groups = q_109_groups_0, pad = q_109_pad_0, pad_type = q_109_pad_type_0, strides = q_109_strides_0, weight = layers_9_self_attn_q_proj_weight_palettized, x = var_6300_cast_fp16)[name = string("q_109")]; tensor var_6321 = const()[name = string("op_6321"), val = tensor([1, 8, 256, 3])]; tensor var_6322 = reshape(shape = var_6321, x = q_109)[name = string("op_6322")]; tensor transpose_75_perm_0 = const()[name = string("transpose_75_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_6345 = const()[name = string("op_6345"), val = tensor([3, 8, 256])]; tensor transpose_75 = transpose(perm = transpose_75_perm_0, x = var_6322)[name = string("transpose_58")]; tensor x_181 = reshape(shape = var_6345, x = transpose_75)[name = string("x_181")]; int32 var_6351 = const()[name = string("op_6351"), val = int32(-1)]; fp16 const_107_promoted = const()[name = string("const_107_promoted"), val = fp16(-0x1p+0)]; tensor var_6353 = mul(x = x_181, y = const_107_promoted)[name = string("op_6353")]; bool input_271_interleave_0 = const()[name = string("input_271_interleave_0"), val = bool(false)]; tensor input_271 = concat(axis = var_6351, interleave = input_271_interleave_0, values = (x_181, var_6353))[name = string("input_271")]; tensor normed_257_axes_0 = const()[name = string("normed_257_axes_0"), val = tensor([-1])]; fp16 var_6348_to_fp16 = const()[name = string("op_6348_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_257_cast_fp16 = layer_norm(axes = normed_257_axes_0, epsilon = var_6348_to_fp16, x = input_271)[name = string("normed_257_cast_fp16")]; tensor var_6358_split_sizes_0 = const()[name = string("op_6358_split_sizes_0"), val = tensor([256, 256])]; int32 var_6358_axis_0 = const()[name = string("op_6358_axis_0"), val = int32(-1)]; tensor var_6358_0, tensor var_6358_1 = split(axis = var_6358_axis_0, split_sizes = var_6358_split_sizes_0, x = normed_257_cast_fp16)[name = string("op_6358")]; tensor q_113 = mul(x = var_6358_0, y = layers_9_self_attn_q_norm_weight)[name = string("q_113")]; tensor var_6365 = const()[name = string("op_6365"), val = tensor([1, 3, 8, 256])]; tensor var_6366 = reshape(shape = var_6365, x = q_113)[name = string("op_6366")]; tensor var_6371 = const()[name = string("op_6371"), val = tensor([0, 2, 1, 3])]; tensor q_115 = transpose(perm = var_6371, x = var_6366)[name = string("transpose_57")]; tensor var_6373_cast_fp16 = mul(x = q_115, y = cos_s)[name = string("op_6373_cast_fp16")]; tensor var_6374_split_sizes_0 = const()[name = string("op_6374_split_sizes_0"), val = tensor([128, 128])]; int32 var_6374_axis_0 = const()[name = string("op_6374_axis_0"), val = int32(-1)]; tensor var_6374_0, tensor var_6374_1 = split(axis = var_6374_axis_0, split_sizes = var_6374_split_sizes_0, x = q_115)[name = string("op_6374")]; fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; tensor var_6376 = mul(x = var_6374_1, y = const_108_promoted)[name = string("op_6376")]; int32 var_6378 = const()[name = string("op_6378"), val = int32(-1)]; bool var_6379_interleave_0 = const()[name = string("op_6379_interleave_0"), val = bool(false)]; tensor var_6379 = concat(axis = var_6378, interleave = var_6379_interleave_0, values = (var_6376, var_6374_0))[name = string("op_6379")]; tensor var_6380_cast_fp16 = mul(x = var_6379, y = sin_s)[name = string("op_6380_cast_fp16")]; tensor q_119_cast_fp16 = add(x = var_6373_cast_fp16, y = var_6380_cast_fp16)[name = string("q_119_cast_fp16")]; string k_57_pad_type_0 = const()[name = string("k_57_pad_type_0"), val = string("valid")]; tensor k_57_strides_0 = const()[name = string("k_57_strides_0"), val = tensor([1, 1])]; tensor k_57_pad_0 = const()[name = string("k_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_57_dilations_0 = const()[name = string("k_57_dilations_0"), val = tensor([1, 1])]; int32 k_57_groups_0 = const()[name = string("k_57_groups_0"), val = int32(1)]; tensor k_57 = conv(dilations = k_57_dilations_0, groups = k_57_groups_0, pad = k_57_pad_0, pad_type = k_57_pad_type_0, strides = k_57_strides_0, weight = layers_9_self_attn_k_proj_weight_palettized, x = var_6300_cast_fp16)[name = string("k_57")]; tensor var_6398 = const()[name = string("op_6398"), val = tensor([1, 2, 256, 3])]; tensor var_6399 = reshape(shape = var_6398, x = k_57)[name = string("op_6399")]; tensor transpose_76_perm_0 = const()[name = string("transpose_76_perm_0"), val = tensor([0, 3, 1, 2])]; string v_21_pad_type_0 = const()[name = string("v_21_pad_type_0"), val = string("valid")]; tensor v_21_strides_0 = const()[name = string("v_21_strides_0"), val = tensor([1, 1])]; tensor v_21_pad_0 = const()[name = string("v_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_21_dilations_0 = const()[name = string("v_21_dilations_0"), val = tensor([1, 1])]; int32 v_21_groups_0 = const()[name = string("v_21_groups_0"), val = int32(1)]; tensor v_21 = conv(dilations = v_21_dilations_0, groups = v_21_groups_0, pad = v_21_pad_0, pad_type = v_21_pad_type_0, strides = v_21_strides_0, weight = layers_9_self_attn_v_proj_weight_palettized, x = var_6300_cast_fp16)[name = string("v_21")]; tensor var_6426 = const()[name = string("op_6426"), val = tensor([1, 2, 256, 3])]; tensor var_6427 = reshape(shape = var_6426, x = v_21)[name = string("op_6427")]; tensor var_6432 = const()[name = string("op_6432"), val = tensor([0, 1, 3, 2])]; tensor var_6450 = const()[name = string("op_6450"), val = tensor([3, 2, 256])]; tensor transpose_76 = transpose(perm = transpose_76_perm_0, x = var_6399)[name = string("transpose_56")]; tensor x_183 = reshape(shape = var_6450, x = transpose_76)[name = string("x_183")]; int32 var_6456 = const()[name = string("op_6456"), val = int32(-1)]; fp16 const_109_promoted = const()[name = string("const_109_promoted"), val = fp16(-0x1p+0)]; tensor var_6458 = mul(x = x_183, y = const_109_promoted)[name = string("op_6458")]; bool input_273_interleave_0 = const()[name = string("input_273_interleave_0"), val = bool(false)]; tensor input_273 = concat(axis = var_6456, interleave = input_273_interleave_0, values = (x_183, var_6458))[name = string("input_273")]; tensor normed_261_axes_0 = const()[name = string("normed_261_axes_0"), val = tensor([-1])]; fp16 var_6453_to_fp16 = const()[name = string("op_6453_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_261_cast_fp16 = layer_norm(axes = normed_261_axes_0, epsilon = var_6453_to_fp16, x = input_273)[name = string("normed_261_cast_fp16")]; tensor var_6463_split_sizes_0 = const()[name = string("op_6463_split_sizes_0"), val = tensor([256, 256])]; int32 var_6463_axis_0 = const()[name = string("op_6463_axis_0"), val = int32(-1)]; tensor var_6463_0, tensor var_6463_1 = split(axis = var_6463_axis_0, split_sizes = var_6463_split_sizes_0, x = normed_261_cast_fp16)[name = string("op_6463")]; tensor k_61 = mul(x = var_6463_0, y = layers_9_self_attn_k_norm_weight)[name = string("k_61")]; tensor var_6470 = const()[name = string("op_6470"), val = tensor([1, 3, 2, 256])]; tensor var_6471 = reshape(shape = var_6470, x = k_61)[name = string("op_6471")]; tensor var_6476 = const()[name = string("op_6476"), val = tensor([0, 2, 1, 3])]; fp16 var_6478_promoted = const()[name = string("op_6478_promoted"), val = fp16(0x1p+1)]; tensor var_6433 = transpose(perm = var_6432, x = var_6427)[name = string("transpose_55")]; tensor var_6479 = pow(x = var_6433, y = var_6478_promoted)[name = string("op_6479")]; tensor var_6484_axes_0 = const()[name = string("op_6484_axes_0"), val = tensor([-1])]; bool var_6484_keep_dims_0 = const()[name = string("op_6484_keep_dims_0"), val = bool(true)]; tensor var_6484 = reduce_mean(axes = var_6484_axes_0, keep_dims = var_6484_keep_dims_0, x = var_6479)[name = string("op_6484")]; fp16 var_6486_to_fp16 = const()[name = string("op_6486_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_19_cast_fp16 = add(x = var_6484, y = var_6486_to_fp16)[name = string("mean_sq_19_cast_fp16")]; fp32 var_6488_epsilon_0 = const()[name = string("op_6488_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_6488_cast_fp16 = rsqrt(epsilon = var_6488_epsilon_0, x = mean_sq_19_cast_fp16)[name = string("op_6488_cast_fp16")]; tensor input_277_cast_fp16 = mul(x = var_6433, y = var_6488_cast_fp16)[name = string("input_277_cast_fp16")]; tensor q_117 = transpose(perm = var_6476, x = var_6471)[name = string("transpose_54")]; tensor var_6490_cast_fp16 = mul(x = q_117, y = cos_s)[name = string("op_6490_cast_fp16")]; tensor var_6491_split_sizes_0 = const()[name = string("op_6491_split_sizes_0"), val = tensor([128, 128])]; int32 var_6491_axis_0 = const()[name = string("op_6491_axis_0"), val = int32(-1)]; tensor var_6491_0, tensor var_6491_1 = split(axis = var_6491_axis_0, split_sizes = var_6491_split_sizes_0, x = q_117)[name = string("op_6491")]; fp16 const_110_promoted = const()[name = string("const_110_promoted"), val = fp16(-0x1p+0)]; tensor var_6493 = mul(x = var_6491_1, y = const_110_promoted)[name = string("op_6493")]; int32 var_6495 = const()[name = string("op_6495"), val = int32(-1)]; bool var_6496_interleave_0 = const()[name = string("op_6496_interleave_0"), val = bool(false)]; tensor var_6496 = concat(axis = var_6495, interleave = var_6496_interleave_0, values = (var_6493, var_6491_0))[name = string("op_6496")]; tensor var_6497_cast_fp16 = mul(x = var_6496, y = sin_s)[name = string("op_6497_cast_fp16")]; tensor input_275_cast_fp16 = add(x = var_6490_cast_fp16, y = var_6497_cast_fp16)[name = string("input_275_cast_fp16")]; tensor k_padded_17_pad_0 = const()[name = string("k_padded_17_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_17_mode_0 = const()[name = string("k_padded_17_mode_0"), val = string("constant")]; fp16 const_111_to_fp16 = const()[name = string("const_111_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_17_cast_fp16 = pad(constant_val = const_111_to_fp16, mode = k_padded_17_mode_0, pad = k_padded_17_pad_0, x = input_275_cast_fp16)[name = string("k_padded_17_cast_fp16")]; tensor v_padded_17_pad_0 = const()[name = string("v_padded_17_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_17_mode_0 = const()[name = string("v_padded_17_mode_0"), val = string("constant")]; fp16 const_112_to_fp16 = const()[name = string("const_112_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_17_cast_fp16 = pad(constant_val = const_112_to_fp16, mode = v_padded_17_mode_0, pad = v_padded_17_pad_0, x = input_277_cast_fp16)[name = string("v_padded_17_cast_fp16")]; tensor slot_k_19_begin_0 = const()[name = string("slot_k_19_begin_0"), val = tensor([8, 0, 0, 0])]; tensor slot_k_19_end_0 = const()[name = string("slot_k_19_end_0"), val = tensor([9, 2, 512, 512])]; tensor slot_k_19_end_mask_0 = const()[name = string("slot_k_19_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_k_19_cast_fp16 = slice_by_index(begin = slot_k_19_begin_0, end = slot_k_19_end_0, end_mask = slot_k_19_end_mask_0, x = K_sliding_out_15_cast_fp16)[name = string("slot_k_19_cast_fp16")]; tensor slot_v_19_begin_0 = const()[name = string("slot_v_19_begin_0"), val = tensor([8, 0, 0, 0])]; tensor slot_v_19_end_0 = const()[name = string("slot_v_19_end_0"), val = tensor([9, 2, 512, 512])]; tensor slot_v_19_end_mask_0 = const()[name = string("slot_v_19_end_mask_0"), val = tensor([false, true, true, true])]; tensor slot_v_19_cast_fp16 = slice_by_index(begin = slot_v_19_begin_0, end = slot_v_19_end_0, end_mask = slot_v_19_end_mask_0, x = V_sliding_out_15_cast_fp16)[name = string("slot_v_19_cast_fp16")]; tensor var_6536_begin_0 = const()[name = string("op_6536_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_6536_end_0 = const()[name = string("op_6536_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_6536_end_mask_0 = const()[name = string("op_6536_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6536_cast_fp16 = slice_by_index(begin = var_6536_begin_0, end = var_6536_end_0, end_mask = var_6536_end_mask_0, x = slot_k_19_cast_fp16)[name = string("op_6536_cast_fp16")]; int32 var_6543 = const()[name = string("op_6543"), val = int32(2)]; bool new_k_19_interleave_0 = const()[name = string("new_k_19_interleave_0"), val = bool(false)]; tensor new_k_19_cast_fp16 = concat(axis = var_6543, interleave = new_k_19_interleave_0, values = (var_6536_cast_fp16, k_padded_17_cast_fp16))[name = string("new_k_19_cast_fp16")]; tensor var_6559_begin_0 = const()[name = string("op_6559_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_6559_end_0 = const()[name = string("op_6559_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_6559_end_mask_0 = const()[name = string("op_6559_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6559_cast_fp16 = slice_by_index(begin = var_6559_begin_0, end = var_6559_end_0, end_mask = var_6559_end_mask_0, x = slot_v_19_cast_fp16)[name = string("op_6559_cast_fp16")]; int32 var_6566 = const()[name = string("op_6566"), val = int32(2)]; bool new_v_19_interleave_0 = const()[name = string("new_v_19_interleave_0"), val = bool(false)]; tensor new_v_19_cast_fp16 = concat(axis = var_6566, interleave = new_v_19_interleave_0, values = (var_6559_cast_fp16, v_padded_17_cast_fp16))[name = string("new_v_19_cast_fp16")]; tensor var_6572_begin_0 = const()[name = string("op_6572_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6572_end_0 = const()[name = string("op_6572_end_0"), val = tensor([8, 2, 512, 512])]; tensor var_6572_end_mask_0 = const()[name = string("op_6572_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6572_cast_fp16 = slice_by_index(begin = var_6572_begin_0, end = var_6572_end_0, end_mask = var_6572_end_mask_0, x = K_sliding_out_15_cast_fp16)[name = string("op_6572_cast_fp16")]; tensor var_6577_begin_0 = const()[name = string("op_6577_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_6577_end_0 = const()[name = string("op_6577_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_6577_end_mask_0 = const()[name = string("op_6577_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6577_cast_fp16 = slice_by_index(begin = var_6577_begin_0, end = var_6577_end_0, end_mask = var_6577_end_mask_0, x = K_sliding_out_15_cast_fp16)[name = string("op_6577_cast_fp16")]; int32 var_6579 = const()[name = string("op_6579"), val = int32(0)]; bool K_sliding_out_17_interleave_0 = const()[name = string("K_sliding_out_17_interleave_0"), val = bool(false)]; tensor K_sliding_out_17_cast_fp16 = concat(axis = var_6579, interleave = K_sliding_out_17_interleave_0, values = (var_6572_cast_fp16, new_k_19_cast_fp16, var_6577_cast_fp16))[name = string("K_sliding_out_17_cast_fp16")]; tensor var_6585_begin_0 = const()[name = string("op_6585_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6585_end_0 = const()[name = string("op_6585_end_0"), val = tensor([8, 2, 512, 512])]; tensor var_6585_end_mask_0 = const()[name = string("op_6585_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6585_cast_fp16 = slice_by_index(begin = var_6585_begin_0, end = var_6585_end_0, end_mask = var_6585_end_mask_0, x = V_sliding_out_15_cast_fp16)[name = string("op_6585_cast_fp16")]; tensor var_6590_begin_0 = const()[name = string("op_6590_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_6590_end_0 = const()[name = string("op_6590_end_0"), val = tensor([10, 2, 512, 512])]; tensor var_6590_end_mask_0 = const()[name = string("op_6590_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6590_cast_fp16 = slice_by_index(begin = var_6590_begin_0, end = var_6590_end_0, end_mask = var_6590_end_mask_0, x = V_sliding_out_15_cast_fp16)[name = string("op_6590_cast_fp16")]; int32 var_6592 = const()[name = string("op_6592"), val = int32(0)]; bool V_sliding_out_17_interleave_0 = const()[name = string("V_sliding_out_17_interleave_0"), val = bool(false)]; tensor V_sliding_out_17_cast_fp16 = concat(axis = var_6592, interleave = V_sliding_out_17_interleave_0, values = (var_6585_cast_fp16, new_v_19_cast_fp16, var_6590_cast_fp16))[name = string("V_sliding_out_17_cast_fp16")]; tensor var_6598_begin_0 = const()[name = string("op_6598_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_6598_end_0 = const()[name = string("op_6598_end_0"), val = tensor([9, 2, 512, 512])]; tensor var_6598_end_mask_0 = const()[name = string("op_6598_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6598_cast_fp16 = slice_by_index(begin = var_6598_begin_0, end = var_6598_end_0, end_mask = var_6598_end_mask_0, x = K_sliding_out_17_cast_fp16)[name = string("op_6598_cast_fp16")]; tensor K_for_attn_19_begin_0 = const()[name = string("K_for_attn_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_19_end_0 = const()[name = string("K_for_attn_19_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_19_end_mask_0 = const()[name = string("K_for_attn_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_for_attn_19_cast_fp16 = slice_by_index(begin = K_for_attn_19_begin_0, end = K_for_attn_19_end_0, end_mask = K_for_attn_19_end_mask_0, x = var_6598_cast_fp16)[name = string("K_for_attn_19_cast_fp16")]; tensor var_6608_begin_0 = const()[name = string("op_6608_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_6608_end_0 = const()[name = string("op_6608_end_0"), val = tensor([9, 2, 512, 512])]; tensor var_6608_end_mask_0 = const()[name = string("op_6608_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6608_cast_fp16 = slice_by_index(begin = var_6608_begin_0, end = var_6608_end_0, end_mask = var_6608_end_mask_0, x = V_sliding_out_17_cast_fp16)[name = string("op_6608_cast_fp16")]; tensor V_for_attn_19_begin_0 = const()[name = string("V_for_attn_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_19_end_0 = const()[name = string("V_for_attn_19_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_19_end_mask_0 = const()[name = string("V_for_attn_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_19_cast_fp16 = slice_by_index(begin = V_for_attn_19_begin_0, end = V_for_attn_19_end_0, end_mask = V_for_attn_19_end_mask_0, x = var_6608_cast_fp16)[name = string("V_for_attn_19_cast_fp16")]; tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_18_reps_0 = const()[name = string("tile_18_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = K_for_attn_19_cast_fp16)[name = string("transpose_53")]; tensor tile_18_cast_fp16 = tile(reps = tile_18_reps_0, x = transpose_36_cast_fp16)[name = string("tile_18_cast_fp16")]; tensor concat_38 = const()[name = string("concat_38"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_36_cast_fp16 = reshape(shape = concat_38, x = tile_18_cast_fp16)[name = string("reshape_36_cast_fp16")]; tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_39 = const()[name = string("concat_39"), val = tensor([-1, 1, 512, 256])]; tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_36_cast_fp16)[name = string("transpose_52")]; tensor reshape_37_cast_fp16 = reshape(shape = concat_39, x = transpose_37_cast_fp16)[name = string("reshape_37_cast_fp16")]; tensor transpose_77_perm_0 = const()[name = string("transpose_77_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_19_reps_0 = const()[name = string("tile_19_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = V_for_attn_19_cast_fp16)[name = string("transpose_51")]; tensor tile_19_cast_fp16 = tile(reps = tile_19_reps_0, x = transpose_38_cast_fp16)[name = string("tile_19_cast_fp16")]; tensor concat_40 = const()[name = string("concat_40"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_38_cast_fp16 = reshape(shape = concat_40, x = tile_19_cast_fp16)[name = string("reshape_38_cast_fp16")]; tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_41 = const()[name = string("concat_41"), val = tensor([-1, 1, 512, 256])]; tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_50")]; tensor reshape_39_cast_fp16 = reshape(shape = concat_41, x = transpose_39_cast_fp16)[name = string("reshape_39_cast_fp16")]; tensor V_expanded_19_perm_0 = const()[name = string("V_expanded_19_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_37_transpose_x_0 = const()[name = string("attn_weights_37_transpose_x_0"), val = bool(false)]; bool attn_weights_37_transpose_y_0 = const()[name = string("attn_weights_37_transpose_y_0"), val = bool(false)]; tensor transpose_77_cast_fp16 = transpose(perm = transpose_77_perm_0, x = reshape_37_cast_fp16)[name = string("transpose_49")]; tensor attn_weights_37_cast_fp16 = matmul(transpose_x = attn_weights_37_transpose_x_0, transpose_y = attn_weights_37_transpose_y_0, x = q_119_cast_fp16, y = transpose_77_cast_fp16)[name = string("attn_weights_37_cast_fp16")]; tensor x_187_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask_sliding)[name = string("x_187_cast_fp16")]; tensor reduce_max_9_axes_0 = const()[name = string("reduce_max_9_axes_0"), val = tensor([-1])]; bool reduce_max_9_keep_dims_0 = const()[name = string("reduce_max_9_keep_dims_0"), val = bool(true)]; tensor reduce_max_9 = reduce_max(axes = reduce_max_9_axes_0, keep_dims = reduce_max_9_keep_dims_0, x = x_187_cast_fp16)[name = string("reduce_max_9")]; tensor var_6643 = sub(x = x_187_cast_fp16, y = reduce_max_9)[name = string("op_6643")]; tensor var_6649 = exp(x = var_6643)[name = string("op_6649")]; tensor var_6659_axes_0 = const()[name = string("op_6659_axes_0"), val = tensor([-1])]; bool var_6659_keep_dims_0 = const()[name = string("op_6659_keep_dims_0"), val = bool(true)]; tensor var_6659 = reduce_sum(axes = var_6659_axes_0, keep_dims = var_6659_keep_dims_0, x = var_6649)[name = string("op_6659")]; tensor var_6665_cast_fp16 = real_div(x = var_6649, y = var_6659)[name = string("op_6665_cast_fp16")]; bool attn_output_55_transpose_x_0 = const()[name = string("attn_output_55_transpose_x_0"), val = bool(false)]; bool attn_output_55_transpose_y_0 = const()[name = string("attn_output_55_transpose_y_0"), val = bool(false)]; tensor V_expanded_19_cast_fp16 = transpose(perm = V_expanded_19_perm_0, x = reshape_39_cast_fp16)[name = string("transpose_48")]; tensor attn_output_55_cast_fp16 = matmul(transpose_x = attn_output_55_transpose_x_0, transpose_y = attn_output_55_transpose_y_0, x = var_6665_cast_fp16, y = V_expanded_19_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor var_6676 = const()[name = string("op_6676"), val = tensor([0, 2, 1, 3])]; tensor var_6683 = const()[name = string("op_6683"), val = tensor([1, 3, -1])]; tensor var_6677_cast_fp16 = transpose(perm = var_6676, x = attn_output_55_cast_fp16)[name = string("transpose_47")]; tensor attn_output_57_cast_fp16 = reshape(shape = var_6683, x = var_6677_cast_fp16)[name = string("attn_output_57_cast_fp16")]; tensor var_6688 = const()[name = string("op_6688"), val = tensor([0, 2, 1])]; string var_6704_pad_type_0 = const()[name = string("op_6704_pad_type_0"), val = string("valid")]; int32 var_6704_groups_0 = const()[name = string("op_6704_groups_0"), val = int32(1)]; tensor var_6704_strides_0 = const()[name = string("op_6704_strides_0"), val = tensor([1])]; tensor var_6704_pad_0 = const()[name = string("op_6704_pad_0"), val = tensor([0, 0])]; tensor var_6704_dilations_0 = const()[name = string("op_6704_dilations_0"), val = tensor([1])]; tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(560655040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563276544))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6689_cast_fp16 = transpose(perm = var_6688, x = attn_output_57_cast_fp16)[name = string("transpose_46")]; tensor var_6704_cast_fp16 = conv(dilations = var_6704_dilations_0, groups = var_6704_groups_0, pad = var_6704_pad_0, pad_type = var_6704_pad_type_0, strides = var_6704_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_6689_cast_fp16)[name = string("op_6704_cast_fp16")]; tensor var_6708 = const()[name = string("op_6708"), val = tensor([0, 2, 1])]; int32 var_6714 = const()[name = string("op_6714"), val = int32(-1)]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_191_cast_fp16 = transpose(perm = var_6708, x = var_6704_cast_fp16)[name = string("transpose_45")]; tensor var_6716_cast_fp16 = mul(x = x_191_cast_fp16, y = const_113_promoted_to_fp16)[name = string("op_6716_cast_fp16")]; bool input_281_interleave_0 = const()[name = string("input_281_interleave_0"), val = bool(false)]; tensor input_281_cast_fp16 = concat(axis = var_6714, interleave = input_281_interleave_0, values = (x_191_cast_fp16, var_6716_cast_fp16))[name = string("input_281_cast_fp16")]; tensor normed_265_axes_0 = const()[name = string("normed_265_axes_0"), val = tensor([-1])]; fp16 var_6711_to_fp16 = const()[name = string("op_6711_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_265_cast_fp16 = layer_norm(axes = normed_265_axes_0, epsilon = var_6711_to_fp16, x = input_281_cast_fp16)[name = string("normed_265_cast_fp16")]; tensor var_6721_split_sizes_0 = const()[name = string("op_6721_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6721_axis_0 = const()[name = string("op_6721_axis_0"), val = int32(-1)]; tensor var_6721_cast_fp16_0, tensor var_6721_cast_fp16_1 = split(axis = var_6721_axis_0, split_sizes = var_6721_split_sizes_0, x = normed_265_cast_fp16)[name = string("op_6721_cast_fp16")]; tensor layers_9_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_9_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563279168)))]; tensor attn_output_59_cast_fp16 = mul(x = var_6721_cast_fp16_0, y = layers_9_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_59_cast_fp16")]; tensor x_193_cast_fp16 = add(x = x_179_cast_fp16, y = attn_output_59_cast_fp16)[name = string("x_193_cast_fp16")]; int32 var_6730 = const()[name = string("op_6730"), val = int32(-1)]; fp16 const_114_promoted_to_fp16 = const()[name = string("const_114_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6732_cast_fp16 = mul(x = x_193_cast_fp16, y = const_114_promoted_to_fp16)[name = string("op_6732_cast_fp16")]; bool input_283_interleave_0 = const()[name = string("input_283_interleave_0"), val = bool(false)]; tensor input_283_cast_fp16 = concat(axis = var_6730, interleave = input_283_interleave_0, values = (x_193_cast_fp16, var_6732_cast_fp16))[name = string("input_283_cast_fp16")]; tensor normed_269_axes_0 = const()[name = string("normed_269_axes_0"), val = tensor([-1])]; fp16 var_6727_to_fp16 = const()[name = string("op_6727_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_269_cast_fp16 = layer_norm(axes = normed_269_axes_0, epsilon = var_6727_to_fp16, x = input_283_cast_fp16)[name = string("normed_269_cast_fp16")]; tensor var_6737_split_sizes_0 = const()[name = string("op_6737_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6737_axis_0 = const()[name = string("op_6737_axis_0"), val = int32(-1)]; tensor var_6737_cast_fp16_0, tensor var_6737_cast_fp16_1 = split(axis = var_6737_axis_0, split_sizes = var_6737_split_sizes_0, x = normed_269_cast_fp16)[name = string("op_6737_cast_fp16")]; tensor layers_9_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_9_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563284352)))]; tensor h_57_cast_fp16 = mul(x = var_6737_cast_fp16_0, y = layers_9_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_57_cast_fp16")]; tensor var_6748 = const()[name = string("op_6748"), val = tensor([0, 2, 1])]; tensor input_285_axes_0 = const()[name = string("input_285_axes_0"), val = tensor([2])]; tensor var_6749 = transpose(perm = var_6748, x = h_57_cast_fp16)[name = string("transpose_44")]; tensor input_285 = expand_dims(axes = input_285_axes_0, x = var_6749)[name = string("input_285")]; string gate_37_pad_type_0 = const()[name = string("gate_37_pad_type_0"), val = string("valid")]; tensor gate_37_strides_0 = const()[name = string("gate_37_strides_0"), val = tensor([1, 1])]; tensor gate_37_pad_0 = const()[name = string("gate_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_37_dilations_0 = const()[name = string("gate_37_dilations_0"), val = tensor([1, 1])]; int32 gate_37_groups_0 = const()[name = string("gate_37_groups_0"), val = int32(1)]; tensor gate_37 = conv(dilations = gate_37_dilations_0, groups = gate_37_groups_0, pad = gate_37_pad_0, pad_type = gate_37_pad_type_0, strides = gate_37_strides_0, weight = layers_9_mlp_gate_proj_weight_palettized, x = input_285)[name = string("gate_37")]; string up_19_pad_type_0 = const()[name = string("up_19_pad_type_0"), val = string("valid")]; tensor up_19_strides_0 = const()[name = string("up_19_strides_0"), val = tensor([1, 1])]; tensor up_19_pad_0 = const()[name = string("up_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_19_dilations_0 = const()[name = string("up_19_dilations_0"), val = tensor([1, 1])]; int32 up_19_groups_0 = const()[name = string("up_19_groups_0"), val = int32(1)]; tensor up_19 = conv(dilations = up_19_dilations_0, groups = up_19_groups_0, pad = up_19_pad_0, pad_type = up_19_pad_type_0, strides = up_19_strides_0, weight = layers_9_mlp_up_proj_weight_palettized, x = input_285)[name = string("up_19")]; string gate_39_mode_0 = const()[name = string("gate_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_39 = gelu(mode = gate_39_mode_0, x = gate_37)[name = string("gate_39")]; tensor input_287 = mul(x = gate_39, y = up_19)[name = string("input_287")]; string mlp_out_19_pad_type_0 = const()[name = string("mlp_out_19_pad_type_0"), val = string("valid")]; tensor mlp_out_19_strides_0 = const()[name = string("mlp_out_19_strides_0"), val = tensor([1, 1])]; tensor mlp_out_19_pad_0 = const()[name = string("mlp_out_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_19_dilations_0 = const()[name = string("mlp_out_19_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_19_groups_0 = const()[name = string("mlp_out_19_groups_0"), val = int32(1)]; tensor mlp_out_19 = conv(dilations = mlp_out_19_dilations_0, groups = mlp_out_19_groups_0, pad = mlp_out_19_pad_0, pad_type = mlp_out_19_pad_type_0, strides = mlp_out_19_strides_0, weight = layers_9_mlp_down_proj_weight_palettized, x = input_287)[name = string("mlp_out_19")]; tensor var_6789_axes_0 = const()[name = string("op_6789_axes_0"), val = tensor([2])]; tensor var_6789 = squeeze(axes = var_6789_axes_0, x = mlp_out_19)[name = string("op_6789")]; tensor var_6793 = const()[name = string("op_6793"), val = tensor([0, 2, 1])]; int32 var_6799 = const()[name = string("op_6799"), val = int32(-1)]; fp16 const_115_promoted = const()[name = string("const_115_promoted"), val = fp16(-0x1p+0)]; tensor x_195 = transpose(perm = var_6793, x = var_6789)[name = string("transpose_43")]; tensor var_6801 = mul(x = x_195, y = const_115_promoted)[name = string("op_6801")]; bool input_289_interleave_0 = const()[name = string("input_289_interleave_0"), val = bool(false)]; tensor input_289 = concat(axis = var_6799, interleave = input_289_interleave_0, values = (x_195, var_6801))[name = string("input_289")]; tensor normed_273_axes_0 = const()[name = string("normed_273_axes_0"), val = tensor([-1])]; fp16 var_6796_to_fp16 = const()[name = string("op_6796_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_273_cast_fp16 = layer_norm(axes = normed_273_axes_0, epsilon = var_6796_to_fp16, x = input_289)[name = string("normed_273_cast_fp16")]; tensor var_6806_split_sizes_0 = const()[name = string("op_6806_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6806_axis_0 = const()[name = string("op_6806_axis_0"), val = int32(-1)]; tensor var_6806_0, tensor var_6806_1 = split(axis = var_6806_axis_0, split_sizes = var_6806_split_sizes_0, x = normed_273_cast_fp16)[name = string("op_6806")]; tensor hidden_states_93 = mul(x = var_6806_0, y = layers_9_post_feedforward_layernorm_weight)[name = string("hidden_states_93")]; tensor hidden_states_95_cast_fp16 = add(x = x_193_cast_fp16, y = hidden_states_93)[name = string("hidden_states_95_cast_fp16")]; tensor per_layer_slice_19_begin_0 = const()[name = string("per_layer_slice_19_begin_0"), val = tensor([0, 0, 5376])]; tensor per_layer_slice_19_end_0 = const()[name = string("per_layer_slice_19_end_0"), val = tensor([1, 3, 5632])]; tensor per_layer_slice_19_end_mask_0 = const()[name = string("per_layer_slice_19_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_19_cast_fp16 = slice_by_index(begin = per_layer_slice_19_begin_0, end = per_layer_slice_19_end_0, end_mask = per_layer_slice_19_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_19_cast_fp16")]; tensor var_6834 = const()[name = string("op_6834"), val = tensor([0, 2, 1])]; tensor input_291_axes_0 = const()[name = string("input_291_axes_0"), val = tensor([2])]; tensor var_6835 = transpose(perm = var_6834, x = hidden_states_95_cast_fp16)[name = string("transpose_42")]; tensor input_291 = expand_dims(axes = input_291_axes_0, x = var_6835)[name = string("input_291")]; string gated_55_pad_type_0 = const()[name = string("gated_55_pad_type_0"), val = string("valid")]; tensor gated_55_strides_0 = const()[name = string("gated_55_strides_0"), val = tensor([1, 1])]; tensor gated_55_pad_0 = const()[name = string("gated_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_55_dilations_0 = const()[name = string("gated_55_dilations_0"), val = tensor([1, 1])]; int32 gated_55_groups_0 = const()[name = string("gated_55_groups_0"), val = int32(1)]; tensor gated_55 = conv(dilations = gated_55_dilations_0, groups = gated_55_groups_0, pad = gated_55_pad_0, pad_type = gated_55_pad_type_0, strides = gated_55_strides_0, weight = layers_9_per_layer_input_gate_weight_palettized, x = input_291)[name = string("gated_55")]; string gated_57_mode_0 = const()[name = string("gated_57_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_57 = gelu(mode = gated_57_mode_0, x = gated_55)[name = string("gated_57")]; tensor var_6854 = const()[name = string("op_6854"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_19_axes_0 = const()[name = string("per_layer_slice_conv_19_axes_0"), val = tensor([2])]; tensor var_6855_cast_fp16 = transpose(perm = var_6854, x = per_layer_slice_19_cast_fp16)[name = string("transpose_41")]; tensor per_layer_slice_conv_19_cast_fp16 = expand_dims(axes = per_layer_slice_conv_19_axes_0, x = var_6855_cast_fp16)[name = string("per_layer_slice_conv_19_cast_fp16")]; tensor input_293_cast_fp16 = mul(x = gated_57, y = per_layer_slice_conv_19_cast_fp16)[name = string("input_293_cast_fp16")]; string gated_59_pad_type_0 = const()[name = string("gated_59_pad_type_0"), val = string("valid")]; tensor gated_59_strides_0 = const()[name = string("gated_59_strides_0"), val = tensor([1, 1])]; tensor gated_59_pad_0 = const()[name = string("gated_59_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_59_dilations_0 = const()[name = string("gated_59_dilations_0"), val = tensor([1, 1])]; int32 gated_59_groups_0 = const()[name = string("gated_59_groups_0"), val = int32(1)]; tensor layers_9_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563289536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563617280))))[name = string("layers_9_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_59_cast_fp16 = conv(dilations = gated_59_dilations_0, groups = gated_59_groups_0, pad = gated_59_pad_0, pad_type = gated_59_pad_type_0, strides = gated_59_strides_0, weight = layers_9_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_293_cast_fp16)[name = string("gated_59_cast_fp16")]; tensor var_6871_axes_0 = const()[name = string("op_6871_axes_0"), val = tensor([2])]; tensor var_6871_cast_fp16 = squeeze(axes = var_6871_axes_0, x = gated_59_cast_fp16)[name = string("op_6871_cast_fp16")]; tensor var_6875 = const()[name = string("op_6875"), val = tensor([0, 2, 1])]; int32 var_6881 = const()[name = string("op_6881"), val = int32(-1)]; fp16 const_116_promoted_to_fp16 = const()[name = string("const_116_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_197_cast_fp16 = transpose(perm = var_6875, x = var_6871_cast_fp16)[name = string("transpose_40")]; tensor var_6883_cast_fp16 = mul(x = x_197_cast_fp16, y = const_116_promoted_to_fp16)[name = string("op_6883_cast_fp16")]; bool input_295_interleave_0 = const()[name = string("input_295_interleave_0"), val = bool(false)]; tensor input_295_cast_fp16 = concat(axis = var_6881, interleave = input_295_interleave_0, values = (x_197_cast_fp16, var_6883_cast_fp16))[name = string("input_295_cast_fp16")]; tensor normed_277_axes_0 = const()[name = string("normed_277_axes_0"), val = tensor([-1])]; fp16 var_6878_to_fp16 = const()[name = string("op_6878_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_277_cast_fp16 = layer_norm(axes = normed_277_axes_0, epsilon = var_6878_to_fp16, x = input_295_cast_fp16)[name = string("normed_277_cast_fp16")]; tensor var_6888_split_sizes_0 = const()[name = string("op_6888_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6888_axis_0 = const()[name = string("op_6888_axis_0"), val = int32(-1)]; tensor var_6888_cast_fp16_0, tensor var_6888_cast_fp16_1 = split(axis = var_6888_axis_0, split_sizes = var_6888_split_sizes_0, x = normed_277_cast_fp16)[name = string("op_6888_cast_fp16")]; tensor layers_9_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_9_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563619904)))]; tensor hidden_states_99_cast_fp16 = mul(x = var_6888_cast_fp16_0, y = layers_9_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_99_cast_fp16")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_99_cast_fp16)[name = string("hidden_states_101_cast_fp16")]; tensor const_117_promoted_to_fp16 = const()[name = string("const_117_promoted_to_fp16"), val = tensor([0x1.d8p-2])]; tensor x_199_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_117_promoted_to_fp16)[name = string("x_199_cast_fp16")]; int32 var_6903 = const()[name = string("op_6903"), val = int32(-1)]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6905_cast_fp16 = mul(x = x_199_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_6905_cast_fp16")]; bool input_297_interleave_0 = const()[name = string("input_297_interleave_0"), val = bool(false)]; tensor input_297_cast_fp16 = concat(axis = var_6903, interleave = input_297_interleave_0, values = (x_199_cast_fp16, var_6905_cast_fp16))[name = string("input_297_cast_fp16")]; tensor normed_281_axes_0 = const()[name = string("normed_281_axes_0"), val = tensor([-1])]; fp16 var_6900_to_fp16 = const()[name = string("op_6900_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_281_cast_fp16 = layer_norm(axes = normed_281_axes_0, epsilon = var_6900_to_fp16, x = input_297_cast_fp16)[name = string("normed_281_cast_fp16")]; tensor var_6910_split_sizes_0 = const()[name = string("op_6910_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_6910_axis_0 = const()[name = string("op_6910_axis_0"), val = int32(-1)]; tensor var_6910_cast_fp16_0, tensor var_6910_cast_fp16_1 = split(axis = var_6910_axis_0, split_sizes = var_6910_split_sizes_0, x = normed_281_cast_fp16)[name = string("op_6910_cast_fp16")]; tensor layers_10_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_10_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563625088)))]; tensor h_61_cast_fp16 = mul(x = var_6910_cast_fp16_0, y = layers_10_input_layernorm_weight_promoted_to_fp16)[name = string("h_61_cast_fp16")]; tensor var_6916 = const()[name = string("op_6916"), val = tensor([0, 2, 1])]; tensor var_6919_axes_0 = const()[name = string("op_6919_axes_0"), val = tensor([2])]; tensor var_6917_cast_fp16 = transpose(perm = var_6916, x = h_61_cast_fp16)[name = string("transpose_39")]; tensor var_6919_cast_fp16 = expand_dims(axes = var_6919_axes_0, x = var_6917_cast_fp16)[name = string("op_6919_cast_fp16")]; string q_121_pad_type_0 = const()[name = string("q_121_pad_type_0"), val = string("valid")]; tensor q_121_strides_0 = const()[name = string("q_121_strides_0"), val = tensor([1, 1])]; tensor q_121_pad_0 = const()[name = string("q_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_121_dilations_0 = const()[name = string("q_121_dilations_0"), val = tensor([1, 1])]; int32 q_121_groups_0 = const()[name = string("q_121_groups_0"), val = int32(1)]; tensor q_121 = conv(dilations = q_121_dilations_0, groups = q_121_groups_0, pad = q_121_pad_0, pad_type = q_121_pad_type_0, strides = q_121_strides_0, weight = layers_10_self_attn_q_proj_weight_palettized, x = var_6919_cast_fp16)[name = string("q_121")]; tensor var_6940 = const()[name = string("op_6940"), val = tensor([1, 8, 256, 3])]; tensor var_6941 = reshape(shape = var_6940, x = q_121)[name = string("op_6941")]; tensor transpose_78_perm_0 = const()[name = string("transpose_78_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_6964 = const()[name = string("op_6964"), val = tensor([3, 8, 256])]; tensor transpose_78 = transpose(perm = transpose_78_perm_0, x = var_6941)[name = string("transpose_38")]; tensor x_201 = reshape(shape = var_6964, x = transpose_78)[name = string("x_201")]; int32 var_6970 = const()[name = string("op_6970"), val = int32(-1)]; fp16 const_119_promoted = const()[name = string("const_119_promoted"), val = fp16(-0x1p+0)]; tensor var_6972 = mul(x = x_201, y = const_119_promoted)[name = string("op_6972")]; bool input_301_interleave_0 = const()[name = string("input_301_interleave_0"), val = bool(false)]; tensor input_301 = concat(axis = var_6970, interleave = input_301_interleave_0, values = (x_201, var_6972))[name = string("input_301")]; tensor normed_285_axes_0 = const()[name = string("normed_285_axes_0"), val = tensor([-1])]; fp16 var_6967_to_fp16 = const()[name = string("op_6967_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_285_cast_fp16 = layer_norm(axes = normed_285_axes_0, epsilon = var_6967_to_fp16, x = input_301)[name = string("normed_285_cast_fp16")]; tensor var_6977_split_sizes_0 = const()[name = string("op_6977_split_sizes_0"), val = tensor([256, 256])]; int32 var_6977_axis_0 = const()[name = string("op_6977_axis_0"), val = int32(-1)]; tensor var_6977_0, tensor var_6977_1 = split(axis = var_6977_axis_0, split_sizes = var_6977_split_sizes_0, x = normed_285_cast_fp16)[name = string("op_6977")]; tensor q_125 = mul(x = var_6977_0, y = layers_10_self_attn_q_norm_weight)[name = string("q_125")]; tensor var_6984 = const()[name = string("op_6984"), val = tensor([1, 3, 8, 256])]; tensor var_6985 = reshape(shape = var_6984, x = q_125)[name = string("op_6985")]; tensor var_6990 = const()[name = string("op_6990"), val = tensor([0, 2, 1, 3])]; tensor q_127 = transpose(perm = var_6990, x = var_6985)[name = string("transpose_37")]; tensor var_6992_cast_fp16 = mul(x = q_127, y = cos_s)[name = string("op_6992_cast_fp16")]; tensor var_6993_split_sizes_0 = const()[name = string("op_6993_split_sizes_0"), val = tensor([128, 128])]; int32 var_6993_axis_0 = const()[name = string("op_6993_axis_0"), val = int32(-1)]; tensor var_6993_0, tensor var_6993_1 = split(axis = var_6993_axis_0, split_sizes = var_6993_split_sizes_0, x = q_127)[name = string("op_6993")]; fp16 const_120_promoted = const()[name = string("const_120_promoted"), val = fp16(-0x1p+0)]; tensor var_6995 = mul(x = var_6993_1, y = const_120_promoted)[name = string("op_6995")]; int32 var_6997 = const()[name = string("op_6997"), val = int32(-1)]; bool var_6998_interleave_0 = const()[name = string("op_6998_interleave_0"), val = bool(false)]; tensor var_6998 = concat(axis = var_6997, interleave = var_6998_interleave_0, values = (var_6995, var_6993_0))[name = string("op_6998")]; tensor var_6999_cast_fp16 = mul(x = var_6998, y = sin_s)[name = string("op_6999_cast_fp16")]; tensor q_131_cast_fp16 = add(x = var_6992_cast_fp16, y = var_6999_cast_fp16)[name = string("q_131_cast_fp16")]; string k_63_pad_type_0 = const()[name = string("k_63_pad_type_0"), val = string("valid")]; tensor k_63_strides_0 = const()[name = string("k_63_strides_0"), val = tensor([1, 1])]; tensor k_63_pad_0 = const()[name = string("k_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_63_dilations_0 = const()[name = string("k_63_dilations_0"), val = tensor([1, 1])]; int32 k_63_groups_0 = const()[name = string("k_63_groups_0"), val = int32(1)]; tensor k_63 = conv(dilations = k_63_dilations_0, groups = k_63_groups_0, pad = k_63_pad_0, pad_type = k_63_pad_type_0, strides = k_63_strides_0, weight = layers_10_self_attn_k_proj_weight_palettized, x = var_6919_cast_fp16)[name = string("k_63")]; tensor var_7017 = const()[name = string("op_7017"), val = tensor([1, 2, 256, 3])]; tensor var_7018 = reshape(shape = var_7017, x = k_63)[name = string("op_7018")]; tensor transpose_79_perm_0 = const()[name = string("transpose_79_perm_0"), val = tensor([0, 3, 1, 2])]; string v_23_pad_type_0 = const()[name = string("v_23_pad_type_0"), val = string("valid")]; tensor v_23_strides_0 = const()[name = string("v_23_strides_0"), val = tensor([1, 1])]; tensor v_23_pad_0 = const()[name = string("v_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_23_dilations_0 = const()[name = string("v_23_dilations_0"), val = tensor([1, 1])]; int32 v_23_groups_0 = const()[name = string("v_23_groups_0"), val = int32(1)]; tensor v_23 = conv(dilations = v_23_dilations_0, groups = v_23_groups_0, pad = v_23_pad_0, pad_type = v_23_pad_type_0, strides = v_23_strides_0, weight = layers_10_self_attn_v_proj_weight_palettized, x = var_6919_cast_fp16)[name = string("v_23")]; tensor var_7045 = const()[name = string("op_7045"), val = tensor([1, 2, 256, 3])]; tensor var_7046 = reshape(shape = var_7045, x = v_23)[name = string("op_7046")]; tensor var_7051 = const()[name = string("op_7051"), val = tensor([0, 1, 3, 2])]; tensor var_7069 = const()[name = string("op_7069"), val = tensor([3, 2, 256])]; tensor transpose_79 = transpose(perm = transpose_79_perm_0, x = var_7018)[name = string("transpose_36")]; tensor x_203 = reshape(shape = var_7069, x = transpose_79)[name = string("x_203")]; int32 var_7075 = const()[name = string("op_7075"), val = int32(-1)]; fp16 const_121_promoted = const()[name = string("const_121_promoted"), val = fp16(-0x1p+0)]; tensor var_7077 = mul(x = x_203, y = const_121_promoted)[name = string("op_7077")]; bool input_303_interleave_0 = const()[name = string("input_303_interleave_0"), val = bool(false)]; tensor input_303 = concat(axis = var_7075, interleave = input_303_interleave_0, values = (x_203, var_7077))[name = string("input_303")]; tensor normed_289_axes_0 = const()[name = string("normed_289_axes_0"), val = tensor([-1])]; fp16 var_7072_to_fp16 = const()[name = string("op_7072_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_289_cast_fp16 = layer_norm(axes = normed_289_axes_0, epsilon = var_7072_to_fp16, x = input_303)[name = string("normed_289_cast_fp16")]; tensor var_7082_split_sizes_0 = const()[name = string("op_7082_split_sizes_0"), val = tensor([256, 256])]; int32 var_7082_axis_0 = const()[name = string("op_7082_axis_0"), val = int32(-1)]; tensor var_7082_0, tensor var_7082_1 = split(axis = var_7082_axis_0, split_sizes = var_7082_split_sizes_0, x = normed_289_cast_fp16)[name = string("op_7082")]; tensor k_67 = mul(x = var_7082_0, y = layers_4_self_attn_k_norm_weight)[name = string("k_67")]; tensor var_7089 = const()[name = string("op_7089"), val = tensor([1, 3, 2, 256])]; tensor var_7090 = reshape(shape = var_7089, x = k_67)[name = string("op_7090")]; tensor var_7095 = const()[name = string("op_7095"), val = tensor([0, 2, 1, 3])]; fp16 var_7097_promoted = const()[name = string("op_7097_promoted"), val = fp16(0x1p+1)]; tensor var_7052 = transpose(perm = var_7051, x = var_7046)[name = string("transpose_35")]; tensor var_7098 = pow(x = var_7052, y = var_7097_promoted)[name = string("op_7098")]; tensor var_7103_axes_0 = const()[name = string("op_7103_axes_0"), val = tensor([-1])]; bool var_7103_keep_dims_0 = const()[name = string("op_7103_keep_dims_0"), val = bool(true)]; tensor var_7103 = reduce_mean(axes = var_7103_axes_0, keep_dims = var_7103_keep_dims_0, x = var_7098)[name = string("op_7103")]; fp16 var_7105_to_fp16 = const()[name = string("op_7105_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_21_cast_fp16 = add(x = var_7103, y = var_7105_to_fp16)[name = string("mean_sq_21_cast_fp16")]; fp32 var_7107_epsilon_0 = const()[name = string("op_7107_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_7107_cast_fp16 = rsqrt(epsilon = var_7107_epsilon_0, x = mean_sq_21_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor input_307_cast_fp16 = mul(x = var_7052, y = var_7107_cast_fp16)[name = string("input_307_cast_fp16")]; tensor q_129 = transpose(perm = var_7095, x = var_7090)[name = string("transpose_34")]; tensor var_7109_cast_fp16 = mul(x = q_129, y = cos_s)[name = string("op_7109_cast_fp16")]; tensor var_7110_split_sizes_0 = const()[name = string("op_7110_split_sizes_0"), val = tensor([128, 128])]; int32 var_7110_axis_0 = const()[name = string("op_7110_axis_0"), val = int32(-1)]; tensor var_7110_0, tensor var_7110_1 = split(axis = var_7110_axis_0, split_sizes = var_7110_split_sizes_0, x = q_129)[name = string("op_7110")]; fp16 const_122_promoted = const()[name = string("const_122_promoted"), val = fp16(-0x1p+0)]; tensor var_7112 = mul(x = var_7110_1, y = const_122_promoted)[name = string("op_7112")]; int32 var_7114 = const()[name = string("op_7114"), val = int32(-1)]; bool var_7115_interleave_0 = const()[name = string("op_7115_interleave_0"), val = bool(false)]; tensor var_7115 = concat(axis = var_7114, interleave = var_7115_interleave_0, values = (var_7112, var_7110_0))[name = string("op_7115")]; tensor var_7116_cast_fp16 = mul(x = var_7115, y = sin_s)[name = string("op_7116_cast_fp16")]; tensor input_305_cast_fp16 = add(x = var_7109_cast_fp16, y = var_7116_cast_fp16)[name = string("input_305_cast_fp16")]; tensor k_padded_pad_0 = const()[name = string("k_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_mode_0 = const()[name = string("k_padded_mode_0"), val = string("constant")]; fp16 const_123_to_fp16 = const()[name = string("const_123_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_cast_fp16 = pad(constant_val = const_123_to_fp16, mode = k_padded_mode_0, pad = k_padded_pad_0, x = input_305_cast_fp16)[name = string("k_padded_cast_fp16")]; tensor v_padded_pad_0 = const()[name = string("v_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_mode_0 = const()[name = string("v_padded_mode_0"), val = string("constant")]; fp16 const_124_to_fp16 = const()[name = string("const_124_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_cast_fp16 = pad(constant_val = const_124_to_fp16, mode = v_padded_mode_0, pad = v_padded_pad_0, x = input_307_cast_fp16)[name = string("v_padded_cast_fp16")]; tensor slot_k_21_begin_0 = const()[name = string("slot_k_21_begin_0"), val = tensor([9, 0, 0, 0])]; tensor slot_k_21_end_0 = const()[name = string("slot_k_21_end_0"), val = tensor([1, 2, 512, 512])]; tensor slot_k_21_end_mask_0 = const()[name = string("slot_k_21_end_mask_0"), val = tensor([true, true, true, true])]; tensor slot_k_21_cast_fp16 = slice_by_index(begin = slot_k_21_begin_0, end = slot_k_21_end_0, end_mask = slot_k_21_end_mask_0, x = K_sliding_out_17_cast_fp16)[name = string("slot_k_21_cast_fp16")]; tensor slot_v_21_begin_0 = const()[name = string("slot_v_21_begin_0"), val = tensor([9, 0, 0, 0])]; tensor slot_v_21_end_0 = const()[name = string("slot_v_21_end_0"), val = tensor([1, 2, 512, 512])]; tensor slot_v_21_end_mask_0 = const()[name = string("slot_v_21_end_mask_0"), val = tensor([true, true, true, true])]; tensor slot_v_21_cast_fp16 = slice_by_index(begin = slot_v_21_begin_0, end = slot_v_21_end_0, end_mask = slot_v_21_end_mask_0, x = V_sliding_out_17_cast_fp16)[name = string("slot_v_21_cast_fp16")]; tensor var_7155_begin_0 = const()[name = string("op_7155_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_7155_end_0 = const()[name = string("op_7155_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_7155_end_mask_0 = const()[name = string("op_7155_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7155_cast_fp16 = slice_by_index(begin = var_7155_begin_0, end = var_7155_end_0, end_mask = var_7155_end_mask_0, x = slot_k_21_cast_fp16)[name = string("op_7155_cast_fp16")]; int32 var_7162 = const()[name = string("op_7162"), val = int32(2)]; bool new_k_21_interleave_0 = const()[name = string("new_k_21_interleave_0"), val = bool(false)]; tensor new_k_21_cast_fp16 = concat(axis = var_7162, interleave = new_k_21_interleave_0, values = (var_7155_cast_fp16, k_padded_cast_fp16))[name = string("new_k_21_cast_fp16")]; tensor var_7178_begin_0 = const()[name = string("op_7178_begin_0"), val = tensor([0, 0, 3, 0])]; tensor var_7178_end_0 = const()[name = string("op_7178_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_7178_end_mask_0 = const()[name = string("op_7178_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7178_cast_fp16 = slice_by_index(begin = var_7178_begin_0, end = var_7178_end_0, end_mask = var_7178_end_mask_0, x = slot_v_21_cast_fp16)[name = string("op_7178_cast_fp16")]; int32 var_7185 = const()[name = string("op_7185"), val = int32(2)]; bool new_v_21_interleave_0 = const()[name = string("new_v_21_interleave_0"), val = bool(false)]; tensor new_v_21_cast_fp16 = concat(axis = var_7185, interleave = new_v_21_interleave_0, values = (var_7178_cast_fp16, v_padded_cast_fp16))[name = string("new_v_21_cast_fp16")]; tensor var_7191_begin_0 = const()[name = string("op_7191_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7191_end_0 = const()[name = string("op_7191_end_0"), val = tensor([9, 2, 512, 512])]; tensor var_7191_end_mask_0 = const()[name = string("op_7191_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7191_cast_fp16 = slice_by_index(begin = var_7191_begin_0, end = var_7191_end_0, end_mask = var_7191_end_mask_0, x = K_sliding_out_17_cast_fp16)[name = string("op_7191_cast_fp16")]; int32 var_7198 = const()[name = string("op_7198"), val = int32(0)]; bool K_sliding_out_interleave_0 = const()[name = string("K_sliding_out_interleave_0"), val = bool(false)]; tensor K_sliding_out = concat(axis = var_7198, interleave = K_sliding_out_interleave_0, values = (var_7191_cast_fp16, new_k_21_cast_fp16))[name = string("K_sliding_out_cast_fp16")]; tensor var_7204_begin_0 = const()[name = string("op_7204_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7204_end_0 = const()[name = string("op_7204_end_0"), val = tensor([9, 2, 512, 512])]; tensor var_7204_end_mask_0 = const()[name = string("op_7204_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7204_cast_fp16 = slice_by_index(begin = var_7204_begin_0, end = var_7204_end_0, end_mask = var_7204_end_mask_0, x = V_sliding_out_17_cast_fp16)[name = string("op_7204_cast_fp16")]; int32 var_7211 = const()[name = string("op_7211"), val = int32(0)]; bool V_sliding_out_interleave_0 = const()[name = string("V_sliding_out_interleave_0"), val = bool(false)]; tensor V_sliding_out = concat(axis = var_7211, interleave = V_sliding_out_interleave_0, values = (var_7204_cast_fp16, new_v_21_cast_fp16))[name = string("V_sliding_out_cast_fp16")]; tensor var_7217_begin_0 = const()[name = string("op_7217_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_7217_end_0 = const()[name = string("op_7217_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_7217_end_mask_0 = const()[name = string("op_7217_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7217_cast_fp16 = slice_by_index(begin = var_7217_begin_0, end = var_7217_end_0, end_mask = var_7217_end_mask_0, x = K_sliding_out)[name = string("op_7217_cast_fp16")]; tensor K_for_attn_21_begin_0 = const()[name = string("K_for_attn_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_for_attn_21_end_0 = const()[name = string("K_for_attn_21_end_0"), val = tensor([1, 2, 512, 256])]; tensor K_for_attn_21_end_mask_0 = const()[name = string("K_for_attn_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor kv13_k = slice_by_index(begin = K_for_attn_21_begin_0, end = K_for_attn_21_end_0, end_mask = K_for_attn_21_end_mask_0, x = var_7217_cast_fp16)[name = string("K_for_attn_21_cast_fp16")]; tensor var_7227_begin_0 = const()[name = string("op_7227_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_7227_end_0 = const()[name = string("op_7227_end_0"), val = tensor([1, 2, 512, 512])]; tensor var_7227_end_mask_0 = const()[name = string("op_7227_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7227_cast_fp16 = slice_by_index(begin = var_7227_begin_0, end = var_7227_end_0, end_mask = var_7227_end_mask_0, x = V_sliding_out)[name = string("op_7227_cast_fp16")]; tensor V_for_attn_21_begin_0 = const()[name = string("V_for_attn_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_21_end_0 = const()[name = string("V_for_attn_21_end_0"), val = tensor([1, 2, 512, 256])]; tensor V_for_attn_21_end_mask_0 = const()[name = string("V_for_attn_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor kv13_v = slice_by_index(begin = V_for_attn_21_begin_0, end = V_for_attn_21_end_0, end_mask = V_for_attn_21_end_mask_0, x = var_7227_cast_fp16)[name = string("V_for_attn_21_cast_fp16")]; tensor transpose_40_perm_0 = const()[name = string("transpose_40_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_20_reps_0 = const()[name = string("tile_20_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_40_cast_fp16 = transpose(perm = transpose_40_perm_0, x = kv13_k)[name = string("transpose_33")]; tensor tile_20_cast_fp16 = tile(reps = tile_20_reps_0, x = transpose_40_cast_fp16)[name = string("tile_20_cast_fp16")]; tensor concat_42 = const()[name = string("concat_42"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_40_cast_fp16 = reshape(shape = concat_42, x = tile_20_cast_fp16)[name = string("reshape_40_cast_fp16")]; tensor transpose_41_perm_0 = const()[name = string("transpose_41_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_43 = const()[name = string("concat_43"), val = tensor([-1, 1, 512, 256])]; tensor transpose_41_cast_fp16 = transpose(perm = transpose_41_perm_0, x = reshape_40_cast_fp16)[name = string("transpose_32")]; tensor reshape_41_cast_fp16 = reshape(shape = concat_43, x = transpose_41_cast_fp16)[name = string("reshape_41_cast_fp16")]; tensor transpose_80_perm_0 = const()[name = string("transpose_80_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_42_perm_0 = const()[name = string("transpose_42_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_21_reps_0 = const()[name = string("tile_21_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_42_cast_fp16 = transpose(perm = transpose_42_perm_0, x = kv13_v)[name = string("transpose_31")]; tensor tile_21_cast_fp16 = tile(reps = tile_21_reps_0, x = transpose_42_cast_fp16)[name = string("tile_21_cast_fp16")]; tensor concat_44 = const()[name = string("concat_44"), val = tensor([4, 2, 1, 512, 256])]; tensor reshape_42_cast_fp16 = reshape(shape = concat_44, x = tile_21_cast_fp16)[name = string("reshape_42_cast_fp16")]; tensor transpose_43_perm_0 = const()[name = string("transpose_43_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 512, 256])]; tensor transpose_43_cast_fp16 = transpose(perm = transpose_43_perm_0, x = reshape_42_cast_fp16)[name = string("transpose_30")]; tensor reshape_43_cast_fp16 = reshape(shape = concat_45, x = transpose_43_cast_fp16)[name = string("reshape_43_cast_fp16")]; tensor V_expanded_21_perm_0 = const()[name = string("V_expanded_21_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor transpose_80_cast_fp16 = transpose(perm = transpose_80_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_29")]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = q_131_cast_fp16, y = transpose_80_cast_fp16)[name = string("attn_weights_41_cast_fp16")]; tensor x_207_cast_fp16 = add(x = attn_weights_41_cast_fp16, y = causal_mask_sliding)[name = string("x_207_cast_fp16")]; tensor reduce_max_10_axes_0 = const()[name = string("reduce_max_10_axes_0"), val = tensor([-1])]; bool reduce_max_10_keep_dims_0 = const()[name = string("reduce_max_10_keep_dims_0"), val = bool(true)]; tensor reduce_max_10 = reduce_max(axes = reduce_max_10_axes_0, keep_dims = reduce_max_10_keep_dims_0, x = x_207_cast_fp16)[name = string("reduce_max_10")]; tensor var_7282 = sub(x = x_207_cast_fp16, y = reduce_max_10)[name = string("op_7282")]; tensor var_7288 = exp(x = var_7282)[name = string("op_7288")]; tensor var_7298_axes_0 = const()[name = string("op_7298_axes_0"), val = tensor([-1])]; bool var_7298_keep_dims_0 = const()[name = string("op_7298_keep_dims_0"), val = bool(true)]; tensor var_7298 = reduce_sum(axes = var_7298_axes_0, keep_dims = var_7298_keep_dims_0, x = var_7288)[name = string("op_7298")]; tensor var_7304_cast_fp16 = real_div(x = var_7288, y = var_7298)[name = string("op_7304_cast_fp16")]; bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; tensor V_expanded_21_cast_fp16 = transpose(perm = V_expanded_21_perm_0, x = reshape_43_cast_fp16)[name = string("transpose_28")]; tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = var_7304_cast_fp16, y = V_expanded_21_cast_fp16)[name = string("attn_output_61_cast_fp16")]; tensor var_7315 = const()[name = string("op_7315"), val = tensor([0, 2, 1, 3])]; tensor var_7322 = const()[name = string("op_7322"), val = tensor([1, 3, -1])]; tensor var_7316_cast_fp16 = transpose(perm = var_7315, x = attn_output_61_cast_fp16)[name = string("transpose_27")]; tensor attn_output_63_cast_fp16 = reshape(shape = var_7322, x = var_7316_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor var_7327 = const()[name = string("op_7327"), val = tensor([0, 2, 1])]; string var_7343_pad_type_0 = const()[name = string("op_7343_pad_type_0"), val = string("valid")]; int32 var_7343_groups_0 = const()[name = string("op_7343_groups_0"), val = int32(1)]; tensor var_7343_strides_0 = const()[name = string("op_7343_strides_0"), val = tensor([1])]; tensor var_7343_pad_0 = const()[name = string("op_7343_pad_0"), val = tensor([0, 0])]; tensor var_7343_dilations_0 = const()[name = string("op_7343_dilations_0"), val = tensor([1])]; tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(563630272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566251776))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7328_cast_fp16 = transpose(perm = var_7327, x = attn_output_63_cast_fp16)[name = string("transpose_26")]; tensor var_7343_cast_fp16 = conv(dilations = var_7343_dilations_0, groups = var_7343_groups_0, pad = var_7343_pad_0, pad_type = var_7343_pad_type_0, strides = var_7343_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_7328_cast_fp16)[name = string("op_7343_cast_fp16")]; tensor var_7347 = const()[name = string("op_7347"), val = tensor([0, 2, 1])]; int32 var_7353 = const()[name = string("op_7353"), val = int32(-1)]; fp16 const_125_promoted_to_fp16 = const()[name = string("const_125_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_211_cast_fp16 = transpose(perm = var_7347, x = var_7343_cast_fp16)[name = string("transpose_25")]; tensor var_7355_cast_fp16 = mul(x = x_211_cast_fp16, y = const_125_promoted_to_fp16)[name = string("op_7355_cast_fp16")]; bool input_311_interleave_0 = const()[name = string("input_311_interleave_0"), val = bool(false)]; tensor input_311_cast_fp16 = concat(axis = var_7353, interleave = input_311_interleave_0, values = (x_211_cast_fp16, var_7355_cast_fp16))[name = string("input_311_cast_fp16")]; tensor normed_293_axes_0 = const()[name = string("normed_293_axes_0"), val = tensor([-1])]; fp16 var_7350_to_fp16 = const()[name = string("op_7350_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_293_cast_fp16 = layer_norm(axes = normed_293_axes_0, epsilon = var_7350_to_fp16, x = input_311_cast_fp16)[name = string("normed_293_cast_fp16")]; tensor var_7360_split_sizes_0 = const()[name = string("op_7360_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7360_axis_0 = const()[name = string("op_7360_axis_0"), val = int32(-1)]; tensor var_7360_cast_fp16_0, tensor var_7360_cast_fp16_1 = split(axis = var_7360_axis_0, split_sizes = var_7360_split_sizes_0, x = normed_293_cast_fp16)[name = string("op_7360_cast_fp16")]; tensor layers_10_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_10_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566254400)))]; tensor attn_output_65_cast_fp16 = mul(x = var_7360_cast_fp16_0, y = layers_10_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_65_cast_fp16")]; tensor x_213_cast_fp16 = add(x = x_199_cast_fp16, y = attn_output_65_cast_fp16)[name = string("x_213_cast_fp16")]; int32 var_7369 = const()[name = string("op_7369"), val = int32(-1)]; fp16 const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7371_cast_fp16 = mul(x = x_213_cast_fp16, y = const_126_promoted_to_fp16)[name = string("op_7371_cast_fp16")]; bool input_313_interleave_0 = const()[name = string("input_313_interleave_0"), val = bool(false)]; tensor input_313_cast_fp16 = concat(axis = var_7369, interleave = input_313_interleave_0, values = (x_213_cast_fp16, var_7371_cast_fp16))[name = string("input_313_cast_fp16")]; tensor normed_297_axes_0 = const()[name = string("normed_297_axes_0"), val = tensor([-1])]; fp16 var_7366_to_fp16 = const()[name = string("op_7366_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_297_cast_fp16 = layer_norm(axes = normed_297_axes_0, epsilon = var_7366_to_fp16, x = input_313_cast_fp16)[name = string("normed_297_cast_fp16")]; tensor var_7376_split_sizes_0 = const()[name = string("op_7376_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7376_axis_0 = const()[name = string("op_7376_axis_0"), val = int32(-1)]; tensor var_7376_cast_fp16_0, tensor var_7376_cast_fp16_1 = split(axis = var_7376_axis_0, split_sizes = var_7376_split_sizes_0, x = normed_297_cast_fp16)[name = string("op_7376_cast_fp16")]; tensor layers_10_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_10_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566259584)))]; tensor h_63_cast_fp16 = mul(x = var_7376_cast_fp16_0, y = layers_10_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_63_cast_fp16")]; tensor var_7387 = const()[name = string("op_7387"), val = tensor([0, 2, 1])]; tensor input_315_axes_0 = const()[name = string("input_315_axes_0"), val = tensor([2])]; tensor var_7388 = transpose(perm = var_7387, x = h_63_cast_fp16)[name = string("transpose_24")]; tensor input_315 = expand_dims(axes = input_315_axes_0, x = var_7388)[name = string("input_315")]; string gate_41_pad_type_0 = const()[name = string("gate_41_pad_type_0"), val = string("valid")]; tensor gate_41_strides_0 = const()[name = string("gate_41_strides_0"), val = tensor([1, 1])]; tensor gate_41_pad_0 = const()[name = string("gate_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_41_dilations_0 = const()[name = string("gate_41_dilations_0"), val = tensor([1, 1])]; int32 gate_41_groups_0 = const()[name = string("gate_41_groups_0"), val = int32(1)]; tensor gate_41 = conv(dilations = gate_41_dilations_0, groups = gate_41_groups_0, pad = gate_41_pad_0, pad_type = gate_41_pad_type_0, strides = gate_41_strides_0, weight = layers_10_mlp_gate_proj_weight_palettized, x = input_315)[name = string("gate_41")]; string up_21_pad_type_0 = const()[name = string("up_21_pad_type_0"), val = string("valid")]; tensor up_21_strides_0 = const()[name = string("up_21_strides_0"), val = tensor([1, 1])]; tensor up_21_pad_0 = const()[name = string("up_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_21_dilations_0 = const()[name = string("up_21_dilations_0"), val = tensor([1, 1])]; int32 up_21_groups_0 = const()[name = string("up_21_groups_0"), val = int32(1)]; tensor up_21 = conv(dilations = up_21_dilations_0, groups = up_21_groups_0, pad = up_21_pad_0, pad_type = up_21_pad_type_0, strides = up_21_strides_0, weight = layers_10_mlp_up_proj_weight_palettized, x = input_315)[name = string("up_21")]; string gate_43_mode_0 = const()[name = string("gate_43_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_43 = gelu(mode = gate_43_mode_0, x = gate_41)[name = string("gate_43")]; tensor input_317 = mul(x = gate_43, y = up_21)[name = string("input_317")]; string mlp_out_21_pad_type_0 = const()[name = string("mlp_out_21_pad_type_0"), val = string("valid")]; tensor mlp_out_21_strides_0 = const()[name = string("mlp_out_21_strides_0"), val = tensor([1, 1])]; tensor mlp_out_21_pad_0 = const()[name = string("mlp_out_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_21_dilations_0 = const()[name = string("mlp_out_21_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_21_groups_0 = const()[name = string("mlp_out_21_groups_0"), val = int32(1)]; tensor mlp_out_21 = conv(dilations = mlp_out_21_dilations_0, groups = mlp_out_21_groups_0, pad = mlp_out_21_pad_0, pad_type = mlp_out_21_pad_type_0, strides = mlp_out_21_strides_0, weight = layers_10_mlp_down_proj_weight_palettized, x = input_317)[name = string("mlp_out_21")]; tensor var_7428_axes_0 = const()[name = string("op_7428_axes_0"), val = tensor([2])]; tensor var_7428 = squeeze(axes = var_7428_axes_0, x = mlp_out_21)[name = string("op_7428")]; tensor var_7432 = const()[name = string("op_7432"), val = tensor([0, 2, 1])]; int32 var_7438 = const()[name = string("op_7438"), val = int32(-1)]; fp16 const_127_promoted = const()[name = string("const_127_promoted"), val = fp16(-0x1p+0)]; tensor x_215 = transpose(perm = var_7432, x = var_7428)[name = string("transpose_23")]; tensor var_7440 = mul(x = x_215, y = const_127_promoted)[name = string("op_7440")]; bool input_319_interleave_0 = const()[name = string("input_319_interleave_0"), val = bool(false)]; tensor input_319 = concat(axis = var_7438, interleave = input_319_interleave_0, values = (x_215, var_7440))[name = string("input_319")]; tensor normed_301_axes_0 = const()[name = string("normed_301_axes_0"), val = tensor([-1])]; fp16 var_7435_to_fp16 = const()[name = string("op_7435_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_301_cast_fp16 = layer_norm(axes = normed_301_axes_0, epsilon = var_7435_to_fp16, x = input_319)[name = string("normed_301_cast_fp16")]; tensor var_7445_split_sizes_0 = const()[name = string("op_7445_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7445_axis_0 = const()[name = string("op_7445_axis_0"), val = int32(-1)]; tensor var_7445_0, tensor var_7445_1 = split(axis = var_7445_axis_0, split_sizes = var_7445_split_sizes_0, x = normed_301_cast_fp16)[name = string("op_7445")]; tensor hidden_states_103 = mul(x = var_7445_0, y = layers_10_post_feedforward_layernorm_weight)[name = string("hidden_states_103")]; tensor hidden_states_105_cast_fp16 = add(x = x_213_cast_fp16, y = hidden_states_103)[name = string("hidden_states_105_cast_fp16")]; tensor per_layer_slice_21_begin_0 = const()[name = string("per_layer_slice_21_begin_0"), val = tensor([0, 0, 5632])]; tensor per_layer_slice_21_end_0 = const()[name = string("per_layer_slice_21_end_0"), val = tensor([1, 3, 5888])]; tensor per_layer_slice_21_end_mask_0 = const()[name = string("per_layer_slice_21_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_21_cast_fp16 = slice_by_index(begin = per_layer_slice_21_begin_0, end = per_layer_slice_21_end_0, end_mask = per_layer_slice_21_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_21_cast_fp16")]; tensor var_7473 = const()[name = string("op_7473"), val = tensor([0, 2, 1])]; tensor input_321_axes_0 = const()[name = string("input_321_axes_0"), val = tensor([2])]; tensor var_7474 = transpose(perm = var_7473, x = hidden_states_105_cast_fp16)[name = string("transpose_22")]; tensor input_321 = expand_dims(axes = input_321_axes_0, x = var_7474)[name = string("input_321")]; string gated_61_pad_type_0 = const()[name = string("gated_61_pad_type_0"), val = string("valid")]; tensor gated_61_strides_0 = const()[name = string("gated_61_strides_0"), val = tensor([1, 1])]; tensor gated_61_pad_0 = const()[name = string("gated_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_61_dilations_0 = const()[name = string("gated_61_dilations_0"), val = tensor([1, 1])]; int32 gated_61_groups_0 = const()[name = string("gated_61_groups_0"), val = int32(1)]; tensor gated_61 = conv(dilations = gated_61_dilations_0, groups = gated_61_groups_0, pad = gated_61_pad_0, pad_type = gated_61_pad_type_0, strides = gated_61_strides_0, weight = layers_10_per_layer_input_gate_weight_palettized, x = input_321)[name = string("gated_61")]; string gated_63_mode_0 = const()[name = string("gated_63_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_63 = gelu(mode = gated_63_mode_0, x = gated_61)[name = string("gated_63")]; tensor var_7493 = const()[name = string("op_7493"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_21_axes_0 = const()[name = string("per_layer_slice_conv_21_axes_0"), val = tensor([2])]; tensor var_7494_cast_fp16 = transpose(perm = var_7493, x = per_layer_slice_21_cast_fp16)[name = string("transpose_21")]; tensor per_layer_slice_conv_21_cast_fp16 = expand_dims(axes = per_layer_slice_conv_21_axes_0, x = var_7494_cast_fp16)[name = string("per_layer_slice_conv_21_cast_fp16")]; tensor input_323_cast_fp16 = mul(x = gated_63, y = per_layer_slice_conv_21_cast_fp16)[name = string("input_323_cast_fp16")]; string gated_65_pad_type_0 = const()[name = string("gated_65_pad_type_0"), val = string("valid")]; tensor gated_65_strides_0 = const()[name = string("gated_65_strides_0"), val = tensor([1, 1])]; tensor gated_65_pad_0 = const()[name = string("gated_65_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_65_dilations_0 = const()[name = string("gated_65_dilations_0"), val = tensor([1, 1])]; int32 gated_65_groups_0 = const()[name = string("gated_65_groups_0"), val = int32(1)]; tensor layers_10_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566264768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566592512))))[name = string("layers_10_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_65_cast_fp16 = conv(dilations = gated_65_dilations_0, groups = gated_65_groups_0, pad = gated_65_pad_0, pad_type = gated_65_pad_type_0, strides = gated_65_strides_0, weight = layers_10_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_323_cast_fp16)[name = string("gated_65_cast_fp16")]; tensor var_7510_axes_0 = const()[name = string("op_7510_axes_0"), val = tensor([2])]; tensor var_7510_cast_fp16 = squeeze(axes = var_7510_axes_0, x = gated_65_cast_fp16)[name = string("op_7510_cast_fp16")]; tensor var_7514 = const()[name = string("op_7514"), val = tensor([0, 2, 1])]; int32 var_7520 = const()[name = string("op_7520"), val = int32(-1)]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_217_cast_fp16 = transpose(perm = var_7514, x = var_7510_cast_fp16)[name = string("transpose_20")]; tensor var_7522_cast_fp16 = mul(x = x_217_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_7522_cast_fp16")]; bool input_325_interleave_0 = const()[name = string("input_325_interleave_0"), val = bool(false)]; tensor input_325_cast_fp16 = concat(axis = var_7520, interleave = input_325_interleave_0, values = (x_217_cast_fp16, var_7522_cast_fp16))[name = string("input_325_cast_fp16")]; tensor normed_305_axes_0 = const()[name = string("normed_305_axes_0"), val = tensor([-1])]; fp16 var_7517_to_fp16 = const()[name = string("op_7517_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_305_cast_fp16 = layer_norm(axes = normed_305_axes_0, epsilon = var_7517_to_fp16, x = input_325_cast_fp16)[name = string("normed_305_cast_fp16")]; tensor var_7527_split_sizes_0 = const()[name = string("op_7527_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7527_axis_0 = const()[name = string("op_7527_axis_0"), val = int32(-1)]; tensor var_7527_cast_fp16_0, tensor var_7527_cast_fp16_1 = split(axis = var_7527_axis_0, split_sizes = var_7527_split_sizes_0, x = normed_305_cast_fp16)[name = string("op_7527_cast_fp16")]; tensor layers_10_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_10_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566595136)))]; tensor hidden_states_109_cast_fp16 = mul(x = var_7527_cast_fp16_0, y = layers_10_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_109_cast_fp16")]; tensor hidden_states_111_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_109_cast_fp16)[name = string("hidden_states_111_cast_fp16")]; tensor const_129_promoted_to_fp16 = const()[name = string("const_129_promoted_to_fp16"), val = tensor([0x1.42p-3])]; tensor x_219_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_129_promoted_to_fp16)[name = string("x_219_cast_fp16")]; int32 var_7542 = const()[name = string("op_7542"), val = int32(-1)]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7544_cast_fp16 = mul(x = x_219_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_7544_cast_fp16")]; bool input_327_interleave_0 = const()[name = string("input_327_interleave_0"), val = bool(false)]; tensor input_327_cast_fp16 = concat(axis = var_7542, interleave = input_327_interleave_0, values = (x_219_cast_fp16, var_7544_cast_fp16))[name = string("input_327_cast_fp16")]; tensor normed_309_axes_0 = const()[name = string("normed_309_axes_0"), val = tensor([-1])]; fp16 var_7539_to_fp16 = const()[name = string("op_7539_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_309_cast_fp16 = layer_norm(axes = normed_309_axes_0, epsilon = var_7539_to_fp16, x = input_327_cast_fp16)[name = string("normed_309_cast_fp16")]; tensor var_7549_split_sizes_0 = const()[name = string("op_7549_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7549_axis_0 = const()[name = string("op_7549_axis_0"), val = int32(-1)]; tensor var_7549_cast_fp16_0, tensor var_7549_cast_fp16_1 = split(axis = var_7549_axis_0, split_sizes = var_7549_split_sizes_0, x = normed_309_cast_fp16)[name = string("op_7549_cast_fp16")]; tensor layers_11_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_11_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566600320)))]; tensor h_67_cast_fp16 = mul(x = var_7549_cast_fp16_0, y = layers_11_input_layernorm_weight_promoted_to_fp16)[name = string("h_67_cast_fp16")]; tensor var_7555 = const()[name = string("op_7555"), val = tensor([0, 2, 1])]; tensor var_7558_axes_0 = const()[name = string("op_7558_axes_0"), val = tensor([2])]; tensor var_7556_cast_fp16 = transpose(perm = var_7555, x = h_67_cast_fp16)[name = string("transpose_19")]; tensor var_7558_cast_fp16 = expand_dims(axes = var_7558_axes_0, x = var_7556_cast_fp16)[name = string("op_7558_cast_fp16")]; string q_133_pad_type_0 = const()[name = string("q_133_pad_type_0"), val = string("valid")]; tensor q_133_strides_0 = const()[name = string("q_133_strides_0"), val = tensor([1, 1])]; tensor q_133_pad_0 = const()[name = string("q_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor q_133_dilations_0 = const()[name = string("q_133_dilations_0"), val = tensor([1, 1])]; int32 q_133_groups_0 = const()[name = string("q_133_groups_0"), val = int32(1)]; tensor q_133 = conv(dilations = q_133_dilations_0, groups = q_133_groups_0, pad = q_133_pad_0, pad_type = q_133_pad_type_0, strides = q_133_strides_0, weight = layers_11_self_attn_q_proj_weight_palettized, x = var_7558_cast_fp16)[name = string("q_133")]; tensor var_7579 = const()[name = string("op_7579"), val = tensor([1, 8, 512, 3])]; tensor var_7580 = reshape(shape = var_7579, x = q_133)[name = string("op_7580")]; tensor transpose_81_perm_0 = const()[name = string("transpose_81_perm_0"), val = tensor([0, 3, 1, 2])]; tensor var_7603 = const()[name = string("op_7603"), val = tensor([3, 8, 512])]; tensor transpose_81 = transpose(perm = transpose_81_perm_0, x = var_7580)[name = string("transpose_18")]; tensor x_221 = reshape(shape = var_7603, x = transpose_81)[name = string("x_221")]; int32 var_7609 = const()[name = string("op_7609"), val = int32(-1)]; fp16 const_131_promoted = const()[name = string("const_131_promoted"), val = fp16(-0x1p+0)]; tensor var_7611 = mul(x = x_221, y = const_131_promoted)[name = string("op_7611")]; bool input_331_interleave_0 = const()[name = string("input_331_interleave_0"), val = bool(false)]; tensor input_331 = concat(axis = var_7609, interleave = input_331_interleave_0, values = (x_221, var_7611))[name = string("input_331")]; tensor normed_313_axes_0 = const()[name = string("normed_313_axes_0"), val = tensor([-1])]; fp16 var_7606_to_fp16 = const()[name = string("op_7606_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_313_cast_fp16 = layer_norm(axes = normed_313_axes_0, epsilon = var_7606_to_fp16, x = input_331)[name = string("normed_313_cast_fp16")]; tensor var_7616_split_sizes_0 = const()[name = string("op_7616_split_sizes_0"), val = tensor([512, 512])]; int32 var_7616_axis_0 = const()[name = string("op_7616_axis_0"), val = int32(-1)]; tensor var_7616_0, tensor var_7616_1 = split(axis = var_7616_axis_0, split_sizes = var_7616_split_sizes_0, x = normed_313_cast_fp16)[name = string("op_7616")]; tensor q_137 = mul(x = var_7616_0, y = layers_11_self_attn_q_norm_weight)[name = string("q_137")]; tensor var_7623 = const()[name = string("op_7623"), val = tensor([1, 3, 8, 512])]; tensor var_7624 = reshape(shape = var_7623, x = q_137)[name = string("op_7624")]; tensor var_7629 = const()[name = string("op_7629"), val = tensor([0, 2, 1, 3])]; tensor q_139 = transpose(perm = var_7629, x = var_7624)[name = string("transpose_17")]; tensor var_7631_cast_fp16 = mul(x = q_139, y = cos_f)[name = string("op_7631_cast_fp16")]; tensor var_7632_split_sizes_0 = const()[name = string("op_7632_split_sizes_0"), val = tensor([256, 256])]; int32 var_7632_axis_0 = const()[name = string("op_7632_axis_0"), val = int32(-1)]; tensor var_7632_0, tensor var_7632_1 = split(axis = var_7632_axis_0, split_sizes = var_7632_split_sizes_0, x = q_139)[name = string("op_7632")]; fp16 const_132_promoted = const()[name = string("const_132_promoted"), val = fp16(-0x1p+0)]; tensor var_7634 = mul(x = var_7632_1, y = const_132_promoted)[name = string("op_7634")]; int32 var_7636 = const()[name = string("op_7636"), val = int32(-1)]; bool var_7637_interleave_0 = const()[name = string("op_7637_interleave_0"), val = bool(false)]; tensor var_7637 = concat(axis = var_7636, interleave = var_7637_interleave_0, values = (var_7634, var_7632_0))[name = string("op_7637")]; tensor var_7638_cast_fp16 = mul(x = var_7637, y = sin_f)[name = string("op_7638_cast_fp16")]; tensor q_cast_fp16 = add(x = var_7631_cast_fp16, y = var_7638_cast_fp16)[name = string("q_cast_fp16")]; string k_69_pad_type_0 = const()[name = string("k_69_pad_type_0"), val = string("valid")]; tensor k_69_strides_0 = const()[name = string("k_69_strides_0"), val = tensor([1, 1])]; tensor k_69_pad_0 = const()[name = string("k_69_pad_0"), val = tensor([0, 0, 0, 0])]; tensor k_69_dilations_0 = const()[name = string("k_69_dilations_0"), val = tensor([1, 1])]; int32 k_69_groups_0 = const()[name = string("k_69_groups_0"), val = int32(1)]; tensor k_69 = conv(dilations = k_69_dilations_0, groups = k_69_groups_0, pad = k_69_pad_0, pad_type = k_69_pad_type_0, strides = k_69_strides_0, weight = layers_11_self_attn_k_proj_weight_palettized, x = var_7558_cast_fp16)[name = string("k_69")]; tensor var_7656 = const()[name = string("op_7656"), val = tensor([1, 2, 512, 3])]; tensor var_7657 = reshape(shape = var_7656, x = k_69)[name = string("op_7657")]; tensor transpose_82_perm_0 = const()[name = string("transpose_82_perm_0"), val = tensor([0, 3, 1, 2])]; string v_25_pad_type_0 = const()[name = string("v_25_pad_type_0"), val = string("valid")]; tensor v_25_strides_0 = const()[name = string("v_25_strides_0"), val = tensor([1, 1])]; tensor v_25_pad_0 = const()[name = string("v_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor v_25_dilations_0 = const()[name = string("v_25_dilations_0"), val = tensor([1, 1])]; int32 v_25_groups_0 = const()[name = string("v_25_groups_0"), val = int32(1)]; tensor v_25 = conv(dilations = v_25_dilations_0, groups = v_25_groups_0, pad = v_25_pad_0, pad_type = v_25_pad_type_0, strides = v_25_strides_0, weight = layers_11_self_attn_v_proj_weight_palettized, x = var_7558_cast_fp16)[name = string("v_25")]; tensor var_7684 = const()[name = string("op_7684"), val = tensor([1, 2, 512, 3])]; tensor var_7685 = reshape(shape = var_7684, x = v_25)[name = string("op_7685")]; tensor var_7690 = const()[name = string("op_7690"), val = tensor([0, 1, 3, 2])]; tensor var_7708 = const()[name = string("op_7708"), val = tensor([3, 2, 512])]; tensor transpose_82 = transpose(perm = transpose_82_perm_0, x = var_7657)[name = string("transpose_16")]; tensor x_223 = reshape(shape = var_7708, x = transpose_82)[name = string("x_223")]; int32 var_7714 = const()[name = string("op_7714"), val = int32(-1)]; fp16 const_133_promoted = const()[name = string("const_133_promoted"), val = fp16(-0x1p+0)]; tensor var_7716 = mul(x = x_223, y = const_133_promoted)[name = string("op_7716")]; bool input_333_interleave_0 = const()[name = string("input_333_interleave_0"), val = bool(false)]; tensor input_333 = concat(axis = var_7714, interleave = input_333_interleave_0, values = (x_223, var_7716))[name = string("input_333")]; tensor normed_317_axes_0 = const()[name = string("normed_317_axes_0"), val = tensor([-1])]; fp16 var_7711_to_fp16 = const()[name = string("op_7711_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_317_cast_fp16 = layer_norm(axes = normed_317_axes_0, epsilon = var_7711_to_fp16, x = input_333)[name = string("normed_317_cast_fp16")]; tensor var_7721_split_sizes_0 = const()[name = string("op_7721_split_sizes_0"), val = tensor([512, 512])]; int32 var_7721_axis_0 = const()[name = string("op_7721_axis_0"), val = int32(-1)]; tensor var_7721_0, tensor var_7721_1 = split(axis = var_7721_axis_0, split_sizes = var_7721_split_sizes_0, x = normed_317_cast_fp16)[name = string("op_7721")]; tensor k_73 = mul(x = var_7721_0, y = layers_11_self_attn_k_norm_weight)[name = string("k_73")]; tensor var_7728 = const()[name = string("op_7728"), val = tensor([1, 3, 2, 512])]; tensor var_7729 = reshape(shape = var_7728, x = k_73)[name = string("op_7729")]; tensor var_7734 = const()[name = string("op_7734"), val = tensor([0, 2, 1, 3])]; fp16 var_7736_promoted = const()[name = string("op_7736_promoted"), val = fp16(0x1p+1)]; tensor var_7691 = transpose(perm = var_7690, x = var_7685)[name = string("transpose_15")]; tensor var_7737 = pow(x = var_7691, y = var_7736_promoted)[name = string("op_7737")]; tensor var_7742_axes_0 = const()[name = string("op_7742_axes_0"), val = tensor([-1])]; bool var_7742_keep_dims_0 = const()[name = string("op_7742_keep_dims_0"), val = bool(true)]; tensor var_7742 = reduce_mean(axes = var_7742_axes_0, keep_dims = var_7742_keep_dims_0, x = var_7737)[name = string("op_7742")]; fp16 var_7744_to_fp16 = const()[name = string("op_7744_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_cast_fp16 = add(x = var_7742, y = var_7744_to_fp16)[name = string("mean_sq_cast_fp16")]; fp32 var_7746_epsilon_0 = const()[name = string("op_7746_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_7746_cast_fp16 = rsqrt(epsilon = var_7746_epsilon_0, x = mean_sq_cast_fp16)[name = string("op_7746_cast_fp16")]; tensor v_cast_fp16 = mul(x = var_7691, y = var_7746_cast_fp16)[name = string("v_cast_fp16")]; tensor q_141 = transpose(perm = var_7734, x = var_7729)[name = string("transpose_14")]; tensor var_7748_cast_fp16 = mul(x = q_141, y = cos_f)[name = string("op_7748_cast_fp16")]; tensor var_7749_split_sizes_0 = const()[name = string("op_7749_split_sizes_0"), val = tensor([256, 256])]; int32 var_7749_axis_0 = const()[name = string("op_7749_axis_0"), val = int32(-1)]; tensor var_7749_0, tensor var_7749_1 = split(axis = var_7749_axis_0, split_sizes = var_7749_split_sizes_0, x = q_141)[name = string("op_7749")]; fp16 const_134_promoted = const()[name = string("const_134_promoted"), val = fp16(-0x1p+0)]; tensor var_7751 = mul(x = var_7749_1, y = const_134_promoted)[name = string("op_7751")]; int32 var_7753 = const()[name = string("op_7753"), val = int32(-1)]; bool var_7754_interleave_0 = const()[name = string("op_7754_interleave_0"), val = bool(false)]; tensor var_7754 = concat(axis = var_7753, interleave = var_7754_interleave_0, values = (var_7751, var_7749_0))[name = string("op_7754")]; tensor var_7755_cast_fp16 = mul(x = var_7754, y = sin_f)[name = string("op_7755_cast_fp16")]; tensor k_cast_fp16 = add(x = var_7748_cast_fp16, y = var_7755_cast_fp16)[name = string("k_cast_fp16")]; bool k_scattered_transpose_x_0 = const()[name = string("k_scattered_transpose_x_0"), val = bool(false)]; bool k_scattered_transpose_y_0 = const()[name = string("k_scattered_transpose_y_0"), val = bool(false)]; tensor k_scattered_cast_fp16 = matmul(transpose_x = k_scattered_transpose_x_0, transpose_y = k_scattered_transpose_y_0, x = var_4055_cast_fp16, y = k_cast_fp16)[name = string("k_scattered_cast_fp16")]; bool v_scattered_transpose_x_0 = const()[name = string("v_scattered_transpose_x_0"), val = bool(false)]; bool v_scattered_transpose_y_0 = const()[name = string("v_scattered_transpose_y_0"), val = bool(false)]; tensor v_scattered_cast_fp16 = matmul(transpose_x = v_scattered_transpose_x_0, transpose_y = v_scattered_transpose_y_0, x = var_4055_cast_fp16, y = v_cast_fp16)[name = string("v_scattered_cast_fp16")]; tensor slot_k_begin_0 = const()[name = string("slot_k_begin_0"), val = tensor([1, 0, 0, 0])]; tensor slot_k_end_0 = const()[name = string("slot_k_end_0"), val = tensor([1, 2, 2048, 512])]; tensor slot_k_end_mask_0 = const()[name = string("slot_k_end_mask_0"), val = tensor([true, true, true, true])]; tensor slot_k_cast_fp16 = slice_by_index(begin = slot_k_begin_0, end = slot_k_end_0, end_mask = slot_k_end_mask_0, x = K_full_out_1_cast_fp16)[name = string("slot_k_cast_fp16")]; tensor slot_v_begin_0 = const()[name = string("slot_v_begin_0"), val = tensor([1, 0, 0, 0])]; tensor slot_v_end_0 = const()[name = string("slot_v_end_0"), val = tensor([1, 2, 2048, 512])]; tensor slot_v_end_mask_0 = const()[name = string("slot_v_end_mask_0"), val = tensor([true, true, true, true])]; tensor slot_v_cast_fp16 = slice_by_index(begin = slot_v_begin_0, end = slot_v_end_0, end_mask = slot_v_end_mask_0, x = V_full_out_1_cast_fp16)[name = string("slot_v_cast_fp16")]; tensor var_7792_cast_fp16 = mul(x = slot_k_cast_fp16, y = var_4082_cast_fp16)[name = string("op_7792_cast_fp16")]; tensor new_k_cast_fp16 = add(x = var_7792_cast_fp16, y = k_scattered_cast_fp16)[name = string("new_k_cast_fp16")]; tensor var_7798_cast_fp16 = mul(x = slot_v_cast_fp16, y = var_4082_cast_fp16)[name = string("op_7798_cast_fp16")]; tensor new_v_cast_fp16 = add(x = var_7798_cast_fp16, y = v_scattered_cast_fp16)[name = string("new_v_cast_fp16")]; int32 var_7812 = const()[name = string("op_7812"), val = int32(0)]; bool K_full_out_interleave_0 = const()[name = string("K_full_out_interleave_0"), val = bool(false)]; tensor K_full_out = concat(axis = var_7812, interleave = K_full_out_interleave_0, values = (var_4122_cast_fp16, new_k_cast_fp16))[name = string("K_full_out_cast_fp16")]; int32 var_7825 = const()[name = string("op_7825"), val = int32(0)]; bool V_full_out_interleave_0 = const()[name = string("V_full_out_interleave_0"), val = bool(false)]; tensor V_full_out = concat(axis = var_7825, interleave = V_full_out_interleave_0, values = (var_4132_cast_fp16, new_v_cast_fp16))[name = string("V_full_out_cast_fp16")]; tensor var_7831_begin_0 = const()[name = string("op_7831_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_7831_end_0 = const()[name = string("op_7831_end_0"), val = tensor([1, 2, 2048, 512])]; tensor var_7831_end_mask_0 = const()[name = string("op_7831_end_mask_0"), val = tensor([true, true, true, true])]; tensor kv14_k = slice_by_index(begin = var_7831_begin_0, end = var_7831_end_0, end_mask = var_7831_end_mask_0, x = K_full_out)[name = string("op_7831_cast_fp16")]; tensor var_7841_begin_0 = const()[name = string("op_7841_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_7841_end_0 = const()[name = string("op_7841_end_0"), val = tensor([1, 2, 2048, 512])]; tensor var_7841_end_mask_0 = const()[name = string("op_7841_end_mask_0"), val = tensor([true, true, true, true])]; tensor kv14_v = slice_by_index(begin = var_7841_begin_0, end = var_7841_end_0, end_mask = var_7841_end_mask_0, x = V_full_out)[name = string("op_7841_cast_fp16")]; tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_22_reps_0 = const()[name = string("tile_22_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = kv14_k)[name = string("transpose_13")]; tensor tile_22_cast_fp16 = tile(reps = tile_22_reps_0, x = transpose_44_cast_fp16)[name = string("tile_22_cast_fp16")]; tensor concat_48 = const()[name = string("concat_48"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_44_cast_fp16 = reshape(shape = concat_48, x = tile_22_cast_fp16)[name = string("reshape_44_cast_fp16")]; tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_49 = const()[name = string("concat_49"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_12")]; tensor reshape_45_cast_fp16 = reshape(shape = concat_49, x = transpose_45_cast_fp16)[name = string("reshape_45_cast_fp16")]; tensor transpose_83_perm_0 = const()[name = string("transpose_83_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_46_perm_0 = const()[name = string("transpose_46_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_23_reps_0 = const()[name = string("tile_23_reps_0"), val = tensor([4, 1, 1, 1])]; tensor transpose_46_cast_fp16 = transpose(perm = transpose_46_perm_0, x = kv14_v)[name = string("transpose_11")]; tensor tile_23_cast_fp16 = tile(reps = tile_23_reps_0, x = transpose_46_cast_fp16)[name = string("tile_23_cast_fp16")]; tensor concat_50 = const()[name = string("concat_50"), val = tensor([4, 2, 1, 2048, 512])]; tensor reshape_46_cast_fp16 = reshape(shape = concat_50, x = tile_23_cast_fp16)[name = string("reshape_46_cast_fp16")]; tensor transpose_47_perm_0 = const()[name = string("transpose_47_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_51 = const()[name = string("concat_51"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_47_cast_fp16 = transpose(perm = transpose_47_perm_0, x = reshape_46_cast_fp16)[name = string("transpose_10")]; tensor reshape_47_cast_fp16 = reshape(shape = concat_51, x = transpose_47_cast_fp16)[name = string("reshape_47_cast_fp16")]; tensor V_expanded_perm_0 = const()[name = string("V_expanded_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_45_transpose_x_0 = const()[name = string("attn_weights_45_transpose_x_0"), val = bool(false)]; bool attn_weights_45_transpose_y_0 = const()[name = string("attn_weights_45_transpose_y_0"), val = bool(false)]; tensor transpose_83_cast_fp16 = transpose(perm = transpose_83_perm_0, x = reshape_45_cast_fp16)[name = string("transpose_9")]; tensor attn_weights_45_cast_fp16 = matmul(transpose_x = attn_weights_45_transpose_x_0, transpose_y = attn_weights_45_transpose_y_0, x = q_cast_fp16, y = transpose_83_cast_fp16)[name = string("attn_weights_45_cast_fp16")]; tensor x_227_cast_fp16 = add(x = attn_weights_45_cast_fp16, y = causal_mask_full)[name = string("x_227_cast_fp16")]; tensor reduce_max_11_axes_0 = const()[name = string("reduce_max_11_axes_0"), val = tensor([-1])]; bool reduce_max_11_keep_dims_0 = const()[name = string("reduce_max_11_keep_dims_0"), val = bool(true)]; tensor reduce_max_11 = reduce_max(axes = reduce_max_11_axes_0, keep_dims = reduce_max_11_keep_dims_0, x = x_227_cast_fp16)[name = string("reduce_max_11")]; tensor var_7896 = sub(x = x_227_cast_fp16, y = reduce_max_11)[name = string("op_7896")]; tensor var_7902 = exp(x = var_7896)[name = string("op_7902")]; tensor var_7912_axes_0 = const()[name = string("op_7912_axes_0"), val = tensor([-1])]; bool var_7912_keep_dims_0 = const()[name = string("op_7912_keep_dims_0"), val = bool(true)]; tensor var_7912 = reduce_sum(axes = var_7912_axes_0, keep_dims = var_7912_keep_dims_0, x = var_7902)[name = string("op_7912")]; tensor var_7918_cast_fp16 = real_div(x = var_7902, y = var_7912)[name = string("op_7918_cast_fp16")]; bool attn_output_67_transpose_x_0 = const()[name = string("attn_output_67_transpose_x_0"), val = bool(false)]; bool attn_output_67_transpose_y_0 = const()[name = string("attn_output_67_transpose_y_0"), val = bool(false)]; tensor V_expanded_cast_fp16 = transpose(perm = V_expanded_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_8")]; tensor attn_output_67_cast_fp16 = matmul(transpose_x = attn_output_67_transpose_x_0, transpose_y = attn_output_67_transpose_y_0, x = var_7918_cast_fp16, y = V_expanded_cast_fp16)[name = string("attn_output_67_cast_fp16")]; tensor var_7929 = const()[name = string("op_7929"), val = tensor([0, 2, 1, 3])]; tensor var_7936 = const()[name = string("op_7936"), val = tensor([1, 3, -1])]; tensor var_7930_cast_fp16 = transpose(perm = var_7929, x = attn_output_67_cast_fp16)[name = string("transpose_7")]; tensor attn_output_69_cast_fp16 = reshape(shape = var_7936, x = var_7930_cast_fp16)[name = string("attn_output_69_cast_fp16")]; tensor var_7941 = const()[name = string("op_7941"), val = tensor([0, 2, 1])]; string var_7957_pad_type_0 = const()[name = string("op_7957_pad_type_0"), val = string("valid")]; int32 var_7957_groups_0 = const()[name = string("op_7957_groups_0"), val = int32(1)]; tensor var_7957_strides_0 = const()[name = string("op_7957_strides_0"), val = tensor([1])]; tensor var_7957_pad_0 = const()[name = string("op_7957_pad_0"), val = tensor([0, 0])]; tensor var_7957_dilations_0 = const()[name = string("op_7957_dilations_0"), val = tensor([1])]; tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(566605504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571848448))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7942_cast_fp16 = transpose(perm = var_7941, x = attn_output_69_cast_fp16)[name = string("transpose_6")]; tensor var_7957_cast_fp16 = conv(dilations = var_7957_dilations_0, groups = var_7957_groups_0, pad = var_7957_pad_0, pad_type = var_7957_pad_type_0, strides = var_7957_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_7942_cast_fp16)[name = string("op_7957_cast_fp16")]; tensor var_7961 = const()[name = string("op_7961"), val = tensor([0, 2, 1])]; int32 var_7967 = const()[name = string("op_7967"), val = int32(-1)]; fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_231_cast_fp16 = transpose(perm = var_7961, x = var_7957_cast_fp16)[name = string("transpose_5")]; tensor var_7969_cast_fp16 = mul(x = x_231_cast_fp16, y = const_135_promoted_to_fp16)[name = string("op_7969_cast_fp16")]; bool input_337_interleave_0 = const()[name = string("input_337_interleave_0"), val = bool(false)]; tensor input_337_cast_fp16 = concat(axis = var_7967, interleave = input_337_interleave_0, values = (x_231_cast_fp16, var_7969_cast_fp16))[name = string("input_337_cast_fp16")]; tensor normed_321_axes_0 = const()[name = string("normed_321_axes_0"), val = tensor([-1])]; fp16 var_7964_to_fp16 = const()[name = string("op_7964_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_321_cast_fp16 = layer_norm(axes = normed_321_axes_0, epsilon = var_7964_to_fp16, x = input_337_cast_fp16)[name = string("normed_321_cast_fp16")]; tensor var_7974_split_sizes_0 = const()[name = string("op_7974_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7974_axis_0 = const()[name = string("op_7974_axis_0"), val = int32(-1)]; tensor var_7974_cast_fp16_0, tensor var_7974_cast_fp16_1 = split(axis = var_7974_axis_0, split_sizes = var_7974_split_sizes_0, x = normed_321_cast_fp16)[name = string("op_7974_cast_fp16")]; tensor layers_11_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_11_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571851072)))]; tensor attn_output_cast_fp16 = mul(x = var_7974_cast_fp16_0, y = layers_11_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_cast_fp16")]; tensor x_233_cast_fp16 = add(x = x_219_cast_fp16, y = attn_output_cast_fp16)[name = string("x_233_cast_fp16")]; int32 var_7983 = const()[name = string("op_7983"), val = int32(-1)]; fp16 const_136_promoted_to_fp16 = const()[name = string("const_136_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7985_cast_fp16 = mul(x = x_233_cast_fp16, y = const_136_promoted_to_fp16)[name = string("op_7985_cast_fp16")]; bool input_339_interleave_0 = const()[name = string("input_339_interleave_0"), val = bool(false)]; tensor input_339_cast_fp16 = concat(axis = var_7983, interleave = input_339_interleave_0, values = (x_233_cast_fp16, var_7985_cast_fp16))[name = string("input_339_cast_fp16")]; tensor normed_325_axes_0 = const()[name = string("normed_325_axes_0"), val = tensor([-1])]; fp16 var_7980_to_fp16 = const()[name = string("op_7980_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_325_cast_fp16 = layer_norm(axes = normed_325_axes_0, epsilon = var_7980_to_fp16, x = input_339_cast_fp16)[name = string("normed_325_cast_fp16")]; tensor var_7990_split_sizes_0 = const()[name = string("op_7990_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_7990_axis_0 = const()[name = string("op_7990_axis_0"), val = int32(-1)]; tensor var_7990_cast_fp16_0, tensor var_7990_cast_fp16_1 = split(axis = var_7990_axis_0, split_sizes = var_7990_split_sizes_0, x = normed_325_cast_fp16)[name = string("op_7990_cast_fp16")]; tensor layers_11_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_11_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571856256)))]; tensor h_69_cast_fp16 = mul(x = var_7990_cast_fp16_0, y = layers_11_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("h_69_cast_fp16")]; tensor var_8001 = const()[name = string("op_8001"), val = tensor([0, 2, 1])]; tensor input_341_axes_0 = const()[name = string("input_341_axes_0"), val = tensor([2])]; tensor var_8002 = transpose(perm = var_8001, x = h_69_cast_fp16)[name = string("transpose_4")]; tensor input_341 = expand_dims(axes = input_341_axes_0, x = var_8002)[name = string("input_341")]; string gate_45_pad_type_0 = const()[name = string("gate_45_pad_type_0"), val = string("valid")]; tensor gate_45_strides_0 = const()[name = string("gate_45_strides_0"), val = tensor([1, 1])]; tensor gate_45_pad_0 = const()[name = string("gate_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_45_dilations_0 = const()[name = string("gate_45_dilations_0"), val = tensor([1, 1])]; int32 gate_45_groups_0 = const()[name = string("gate_45_groups_0"), val = int32(1)]; tensor gate_45 = conv(dilations = gate_45_dilations_0, groups = gate_45_groups_0, pad = gate_45_pad_0, pad_type = gate_45_pad_type_0, strides = gate_45_strides_0, weight = layers_11_mlp_gate_proj_weight_palettized, x = input_341)[name = string("gate_45")]; string up_pad_type_0 = const()[name = string("up_pad_type_0"), val = string("valid")]; tensor up_strides_0 = const()[name = string("up_strides_0"), val = tensor([1, 1])]; tensor up_pad_0 = const()[name = string("up_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_dilations_0 = const()[name = string("up_dilations_0"), val = tensor([1, 1])]; int32 up_groups_0 = const()[name = string("up_groups_0"), val = int32(1)]; tensor up = conv(dilations = up_dilations_0, groups = up_groups_0, pad = up_pad_0, pad_type = up_pad_type_0, strides = up_strides_0, weight = layers_11_mlp_up_proj_weight_palettized, x = input_341)[name = string("up")]; string gate_mode_0 = const()[name = string("gate_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate = gelu(mode = gate_mode_0, x = gate_45)[name = string("gate")]; tensor input_343 = mul(x = gate, y = up)[name = string("input_343")]; string mlp_out_pad_type_0 = const()[name = string("mlp_out_pad_type_0"), val = string("valid")]; tensor mlp_out_strides_0 = const()[name = string("mlp_out_strides_0"), val = tensor([1, 1])]; tensor mlp_out_pad_0 = const()[name = string("mlp_out_pad_0"), val = tensor([0, 0, 0, 0])]; tensor mlp_out_dilations_0 = const()[name = string("mlp_out_dilations_0"), val = tensor([1, 1])]; int32 mlp_out_groups_0 = const()[name = string("mlp_out_groups_0"), val = int32(1)]; tensor mlp_out = conv(dilations = mlp_out_dilations_0, groups = mlp_out_groups_0, pad = mlp_out_pad_0, pad_type = mlp_out_pad_type_0, strides = mlp_out_strides_0, weight = layers_11_mlp_down_proj_weight_palettized, x = input_343)[name = string("mlp_out")]; tensor var_8042_axes_0 = const()[name = string("op_8042_axes_0"), val = tensor([2])]; tensor var_8042 = squeeze(axes = var_8042_axes_0, x = mlp_out)[name = string("op_8042")]; tensor var_8046 = const()[name = string("op_8046"), val = tensor([0, 2, 1])]; int32 var_8052 = const()[name = string("op_8052"), val = int32(-1)]; fp16 const_137_promoted = const()[name = string("const_137_promoted"), val = fp16(-0x1p+0)]; tensor x_235 = transpose(perm = var_8046, x = var_8042)[name = string("transpose_3")]; tensor var_8054 = mul(x = x_235, y = const_137_promoted)[name = string("op_8054")]; bool input_345_interleave_0 = const()[name = string("input_345_interleave_0"), val = bool(false)]; tensor input_345 = concat(axis = var_8052, interleave = input_345_interleave_0, values = (x_235, var_8054))[name = string("input_345")]; tensor normed_329_axes_0 = const()[name = string("normed_329_axes_0"), val = tensor([-1])]; fp16 var_8049_to_fp16 = const()[name = string("op_8049_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_329_cast_fp16 = layer_norm(axes = normed_329_axes_0, epsilon = var_8049_to_fp16, x = input_345)[name = string("normed_329_cast_fp16")]; tensor var_8059_split_sizes_0 = const()[name = string("op_8059_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_8059_axis_0 = const()[name = string("op_8059_axis_0"), val = int32(-1)]; tensor var_8059_0, tensor var_8059_1 = split(axis = var_8059_axis_0, split_sizes = var_8059_split_sizes_0, x = normed_329_cast_fp16)[name = string("op_8059")]; tensor hidden_states_113 = mul(x = var_8059_0, y = layers_11_post_feedforward_layernorm_weight)[name = string("hidden_states_113")]; tensor hidden_states_115_cast_fp16 = add(x = x_233_cast_fp16, y = hidden_states_113)[name = string("hidden_states_115_cast_fp16")]; tensor per_layer_slice_begin_0 = const()[name = string("per_layer_slice_begin_0"), val = tensor([0, 0, 5888])]; tensor per_layer_slice_end_0 = const()[name = string("per_layer_slice_end_0"), val = tensor([1, 3, 6144])]; tensor per_layer_slice_end_mask_0 = const()[name = string("per_layer_slice_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_cast_fp16 = slice_by_index(begin = per_layer_slice_begin_0, end = per_layer_slice_end_0, end_mask = per_layer_slice_end_mask_0, x = per_layer_combined)[name = string("per_layer_slice_cast_fp16")]; tensor var_8087 = const()[name = string("op_8087"), val = tensor([0, 2, 1])]; tensor input_347_axes_0 = const()[name = string("input_347_axes_0"), val = tensor([2])]; tensor var_8088 = transpose(perm = var_8087, x = hidden_states_115_cast_fp16)[name = string("transpose_2")]; tensor input_347 = expand_dims(axes = input_347_axes_0, x = var_8088)[name = string("input_347")]; string gated_67_pad_type_0 = const()[name = string("gated_67_pad_type_0"), val = string("valid")]; tensor gated_67_strides_0 = const()[name = string("gated_67_strides_0"), val = tensor([1, 1])]; tensor gated_67_pad_0 = const()[name = string("gated_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_67_dilations_0 = const()[name = string("gated_67_dilations_0"), val = tensor([1, 1])]; int32 gated_67_groups_0 = const()[name = string("gated_67_groups_0"), val = int32(1)]; tensor gated_67 = conv(dilations = gated_67_dilations_0, groups = gated_67_groups_0, pad = gated_67_pad_0, pad_type = gated_67_pad_type_0, strides = gated_67_strides_0, weight = layers_11_per_layer_input_gate_weight_palettized, x = input_347)[name = string("gated_67")]; string gated_69_mode_0 = const()[name = string("gated_69_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_69 = gelu(mode = gated_69_mode_0, x = gated_67)[name = string("gated_69")]; tensor var_8107 = const()[name = string("op_8107"), val = tensor([0, 2, 1])]; tensor per_layer_slice_conv_axes_0 = const()[name = string("per_layer_slice_conv_axes_0"), val = tensor([2])]; tensor var_8108_cast_fp16 = transpose(perm = var_8107, x = per_layer_slice_cast_fp16)[name = string("transpose_1")]; tensor per_layer_slice_conv_cast_fp16 = expand_dims(axes = per_layer_slice_conv_axes_0, x = var_8108_cast_fp16)[name = string("per_layer_slice_conv_cast_fp16")]; tensor input_349_cast_fp16 = mul(x = gated_69, y = per_layer_slice_conv_cast_fp16)[name = string("input_349_cast_fp16")]; string gated_pad_type_0 = const()[name = string("gated_pad_type_0"), val = string("valid")]; tensor gated_strides_0 = const()[name = string("gated_strides_0"), val = tensor([1, 1])]; tensor gated_pad_0 = const()[name = string("gated_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gated_dilations_0 = const()[name = string("gated_dilations_0"), val = tensor([1, 1])]; int32 gated_groups_0 = const()[name = string("gated_groups_0"), val = int32(1)]; tensor layers_11_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(571861440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572189184))))[name = string("layers_11_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor gated_cast_fp16 = conv(dilations = gated_dilations_0, groups = gated_groups_0, pad = gated_pad_0, pad_type = gated_pad_type_0, strides = gated_strides_0, weight = layers_11_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_349_cast_fp16)[name = string("gated_cast_fp16")]; tensor var_8124_axes_0 = const()[name = string("op_8124_axes_0"), val = tensor([2])]; tensor var_8124_cast_fp16 = squeeze(axes = var_8124_axes_0, x = gated_cast_fp16)[name = string("op_8124_cast_fp16")]; tensor var_8128 = const()[name = string("op_8128"), val = tensor([0, 2, 1])]; int32 var_8134 = const()[name = string("op_8134"), val = int32(-1)]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_cast_fp16 = transpose(perm = var_8128, x = var_8124_cast_fp16)[name = string("transpose_0")]; tensor var_8136_cast_fp16 = mul(x = x_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_8136_cast_fp16")]; bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; tensor input_cast_fp16 = concat(axis = var_8134, interleave = input_interleave_0, values = (x_cast_fp16, var_8136_cast_fp16))[name = string("input_cast_fp16")]; tensor normed_333_axes_0 = const()[name = string("normed_333_axes_0"), val = tensor([-1])]; fp16 var_8131_to_fp16 = const()[name = string("op_8131_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_333_cast_fp16 = layer_norm(axes = normed_333_axes_0, epsilon = var_8131_to_fp16, x = input_cast_fp16)[name = string("normed_333_cast_fp16")]; tensor var_8141_split_sizes_0 = const()[name = string("op_8141_split_sizes_0"), val = tensor([2560, 2560])]; int32 var_8141_axis_0 = const()[name = string("op_8141_axis_0"), val = int32(-1)]; tensor var_8141_cast_fp16_0, tensor var_8141_cast_fp16_1 = split(axis = var_8141_axis_0, split_sizes = var_8141_split_sizes_0, x = normed_333_cast_fp16)[name = string("op_8141_cast_fp16")]; tensor layers_11_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_11_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572191808)))]; tensor hidden_states_119_cast_fp16 = mul(x = var_8141_cast_fp16_0, y = layers_11_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_119_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_119_cast_fp16)[name = string("hidden_states_cast_fp16")]; tensor const_139_promoted_to_fp16 = const()[name = string("const_139_promoted_to_fp16"), val = tensor([0x1.0cp-4])]; tensor hidden_states_out = mul(x = hidden_states_cast_fp16, y = const_139_promoted_to_fp16)[name = string("op_8151_cast_fp16")]; } -> (hidden_states_out, K_sliding_out, V_sliding_out, K_full_out, V_full_out, kv13_k, kv13_v, kv14_k, kv14_v); }