program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}})] { func main(tensor attention_mask, tensor input_ids) { tensor encoder_layers_0_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589952))))[name = string("encoder_layers_0_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_0_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(591552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(788224))))[name = string("encoder_layers_0_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_0_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(788800))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(985472))))[name = string("encoder_layers_0_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_0_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(986048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1870848))))[name = string("encoder_layers_0_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_0_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1873216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2758016))))[name = string("encoder_layers_0_mlp_up_proj_weight_quantized")]; tensor encoder_layers_0_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2760384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3645184))))[name = string("encoder_layers_0_mlp_down_proj_weight_quantized")]; tensor encoder_layers_1_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3646784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4236672))))[name = string("encoder_layers_1_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_1_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4238272))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4434944))))[name = string("encoder_layers_1_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_1_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4435520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4632192))))[name = string("encoder_layers_1_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_1_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4632768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5517568))))[name = string("encoder_layers_1_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_1_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5519936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6404736))))[name = string("encoder_layers_1_mlp_up_proj_weight_quantized")]; tensor encoder_layers_1_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6407104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7291904))))[name = string("encoder_layers_1_mlp_down_proj_weight_quantized")]; tensor encoder_layers_2_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7293504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7883392))))[name = string("encoder_layers_2_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_2_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7884992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8081664))))[name = string("encoder_layers_2_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_2_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8082240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8278912))))[name = string("encoder_layers_2_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_2_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8279488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9164288))))[name = string("encoder_layers_2_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_2_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9166656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10051456))))[name = string("encoder_layers_2_mlp_up_proj_weight_quantized")]; tensor encoder_layers_2_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10053824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10938624))))[name = string("encoder_layers_2_mlp_down_proj_weight_quantized")]; tensor encoder_layers_3_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10940224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11530112))))[name = string("encoder_layers_3_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_3_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11531712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11728384))))[name = string("encoder_layers_3_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_3_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11728960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11925632))))[name = string("encoder_layers_3_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_3_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11926208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12811008))))[name = string("encoder_layers_3_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_3_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12813376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13698176))))[name = string("encoder_layers_3_mlp_up_proj_weight_quantized")]; tensor encoder_layers_3_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13700544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14585344))))[name = string("encoder_layers_3_mlp_down_proj_weight_quantized")]; tensor encoder_layers_4_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14586944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15176832))))[name = string("encoder_layers_4_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_4_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15178432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15375104))))[name = string("encoder_layers_4_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_4_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15375680))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15572352))))[name = string("encoder_layers_4_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_4_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15572928))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16457728))))[name = string("encoder_layers_4_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_4_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16460096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17344896))))[name = string("encoder_layers_4_mlp_up_proj_weight_quantized")]; tensor encoder_layers_4_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17347264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18232064))))[name = string("encoder_layers_4_mlp_down_proj_weight_quantized")]; tensor encoder_layers_5_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18233664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18823552))))[name = string("encoder_layers_5_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_5_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18825152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19021824))))[name = string("encoder_layers_5_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_5_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19022400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19219072))))[name = string("encoder_layers_5_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_5_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19219648))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20104448))))[name = string("encoder_layers_5_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_5_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20106816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20991616))))[name = string("encoder_layers_5_mlp_up_proj_weight_quantized")]; tensor encoder_layers_5_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20993984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21878784))))[name = string("encoder_layers_5_mlp_down_proj_weight_quantized")]; tensor encoder_layers_6_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21880384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22470272))))[name = string("encoder_layers_6_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_6_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22471872))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22668544))))[name = string("encoder_layers_6_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_6_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22669120))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22865792))))[name = string("encoder_layers_6_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_6_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22866368))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23751168))))[name = string("encoder_layers_6_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_6_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23753536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24638336))))[name = string("encoder_layers_6_mlp_up_proj_weight_quantized")]; tensor encoder_layers_6_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24640704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25525504))))[name = string("encoder_layers_6_mlp_down_proj_weight_quantized")]; tensor encoder_layers_7_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25527104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26116992))))[name = string("encoder_layers_7_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_7_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26118592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26315264))))[name = string("encoder_layers_7_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_7_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26315840))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26512512))))[name = string("encoder_layers_7_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_7_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26513088))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27397888))))[name = string("encoder_layers_7_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_7_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27400256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28285056))))[name = string("encoder_layers_7_mlp_up_proj_weight_quantized")]; tensor encoder_layers_7_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28287424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29172224))))[name = string("encoder_layers_7_mlp_down_proj_weight_quantized")]; tensor encoder_layers_8_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29173824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29763712))))[name = string("encoder_layers_8_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_8_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29765312))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29961984))))[name = string("encoder_layers_8_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_8_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29962560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30159232))))[name = string("encoder_layers_8_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_8_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30159808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31044608))))[name = string("encoder_layers_8_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_8_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31046976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31931776))))[name = string("encoder_layers_8_mlp_up_proj_weight_quantized")]; tensor encoder_layers_8_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31934144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(32818944))))[name = string("encoder_layers_8_mlp_down_proj_weight_quantized")]; tensor encoder_layers_9_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(32820544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33410432))))[name = string("encoder_layers_9_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_9_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33412032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33608704))))[name = string("encoder_layers_9_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_9_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33609280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33805952))))[name = string("encoder_layers_9_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_9_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33806528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34691328))))[name = string("encoder_layers_9_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_9_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34693696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35578496))))[name = string("encoder_layers_9_mlp_up_proj_weight_quantized")]; tensor encoder_layers_9_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35580864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36465664))))[name = string("encoder_layers_9_mlp_down_proj_weight_quantized")]; tensor encoder_layers_10_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36467264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37057152))))[name = string("encoder_layers_10_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_10_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37058752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37255424))))[name = string("encoder_layers_10_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_10_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37256000))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37452672))))[name = string("encoder_layers_10_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_10_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37453248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(38338048))))[name = string("encoder_layers_10_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_10_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(38340416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39225216))))[name = string("encoder_layers_10_mlp_up_proj_weight_quantized")]; tensor encoder_layers_10_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39227584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40112384))))[name = string("encoder_layers_10_mlp_down_proj_weight_quantized")]; tensor encoder_layers_11_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40113984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40703872))))[name = string("encoder_layers_11_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_11_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40705472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40902144))))[name = string("encoder_layers_11_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_11_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40902720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41099392))))[name = string("encoder_layers_11_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_11_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41099968))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41984768))))[name = string("encoder_layers_11_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_11_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41987136))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(42871936))))[name = string("encoder_layers_11_mlp_up_proj_weight_quantized")]; tensor encoder_layers_11_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(42874304))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43759104))))[name = string("encoder_layers_11_mlp_down_proj_weight_quantized")]; tensor encoder_layers_12_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43760704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44350592))))[name = string("encoder_layers_12_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_12_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44352192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44548864))))[name = string("encoder_layers_12_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_12_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44549440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44746112))))[name = string("encoder_layers_12_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_12_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44746688))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(45631488))))[name = string("encoder_layers_12_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_12_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(45633856))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46518656))))[name = string("encoder_layers_12_mlp_up_proj_weight_quantized")]; tensor encoder_layers_12_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46521024))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47405824))))[name = string("encoder_layers_12_mlp_down_proj_weight_quantized")]; tensor encoder_layers_13_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47407424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47997312))))[name = string("encoder_layers_13_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_13_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47998912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48195584))))[name = string("encoder_layers_13_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_13_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48196160))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48392832))))[name = string("encoder_layers_13_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_13_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48393408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49278208))))[name = string("encoder_layers_13_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_13_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49280576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50165376))))[name = string("encoder_layers_13_mlp_up_proj_weight_quantized")]; tensor encoder_layers_13_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50167744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51052544))))[name = string("encoder_layers_13_mlp_down_proj_weight_quantized")]; tensor encoder_layers_14_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51054144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51644032))))[name = string("encoder_layers_14_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_14_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51645632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51842304))))[name = string("encoder_layers_14_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_14_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51842880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52039552))))[name = string("encoder_layers_14_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_14_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52040128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52924928))))[name = string("encoder_layers_14_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_14_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52927296))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53812096))))[name = string("encoder_layers_14_mlp_up_proj_weight_quantized")]; tensor encoder_layers_14_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53814464))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(54699264))))[name = string("encoder_layers_14_mlp_down_proj_weight_quantized")]; tensor encoder_layers_15_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(54700864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55290752))))[name = string("encoder_layers_15_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_15_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55292352))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55489024))))[name = string("encoder_layers_15_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_15_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55489600))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55686272))))[name = string("encoder_layers_15_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_15_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55686848))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56571648))))[name = string("encoder_layers_15_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_15_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56574016))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57458816))))[name = string("encoder_layers_15_mlp_up_proj_weight_quantized")]; tensor encoder_layers_15_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57461184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58345984))))[name = string("encoder_layers_15_mlp_down_proj_weight_quantized")]; tensor encoder_layers_16_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58347584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58937472))))[name = string("encoder_layers_16_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_16_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58939072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59135744))))[name = string("encoder_layers_16_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_16_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59136320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59332992))))[name = string("encoder_layers_16_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_16_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59333568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60218368))))[name = string("encoder_layers_16_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_16_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60220736))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61105536))))[name = string("encoder_layers_16_mlp_up_proj_weight_quantized")]; tensor encoder_layers_16_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61107904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61992704))))[name = string("encoder_layers_16_mlp_down_proj_weight_quantized")]; tensor encoder_layers_17_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61994304))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62584192))))[name = string("encoder_layers_17_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_17_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62585792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62782464))))[name = string("encoder_layers_17_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_17_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62783040))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62979712))))[name = string("encoder_layers_17_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_17_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62980288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(63865088))))[name = string("encoder_layers_17_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_17_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(63867456))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64752256))))[name = string("encoder_layers_17_mlp_up_proj_weight_quantized")]; tensor encoder_layers_17_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64754624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65639424))))[name = string("encoder_layers_17_mlp_down_proj_weight_quantized")]; tensor encoder_layers_18_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65641024))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66230912))))[name = string("encoder_layers_18_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_18_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66232512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66429184))))[name = string("encoder_layers_18_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_18_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66429760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66626432))))[name = string("encoder_layers_18_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_18_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66627008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67511808))))[name = string("encoder_layers_18_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_18_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67514176))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68398976))))[name = string("encoder_layers_18_mlp_up_proj_weight_quantized")]; tensor encoder_layers_18_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68401344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69286144))))[name = string("encoder_layers_18_mlp_down_proj_weight_quantized")]; tensor encoder_layers_19_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69287744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69877632))))[name = string("encoder_layers_19_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_19_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69879232))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70075904))))[name = string("encoder_layers_19_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_19_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70076480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70273152))))[name = string("encoder_layers_19_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_19_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70273728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71158528))))[name = string("encoder_layers_19_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_19_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71160896))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72045696))))[name = string("encoder_layers_19_mlp_up_proj_weight_quantized")]; tensor encoder_layers_19_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72048064))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72932864))))[name = string("encoder_layers_19_mlp_down_proj_weight_quantized")]; tensor encoder_layers_20_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72934464))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73524352))))[name = string("encoder_layers_20_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_20_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73525952))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73722624))))[name = string("encoder_layers_20_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_20_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73723200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73919872))))[name = string("encoder_layers_20_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_20_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73920448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74805248))))[name = string("encoder_layers_20_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_20_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74807616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75692416))))[name = string("encoder_layers_20_mlp_up_proj_weight_quantized")]; tensor encoder_layers_20_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75694784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76579584))))[name = string("encoder_layers_20_mlp_down_proj_weight_quantized")]; tensor encoder_layers_21_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76581184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77171072))))[name = string("encoder_layers_21_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_21_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77172672))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77369344))))[name = string("encoder_layers_21_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_21_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77369920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77566592))))[name = string("encoder_layers_21_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_21_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77567168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78451968))))[name = string("encoder_layers_21_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_21_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78454336))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(79339136))))[name = string("encoder_layers_21_mlp_up_proj_weight_quantized")]; tensor encoder_layers_21_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(79341504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80226304))))[name = string("encoder_layers_21_mlp_down_proj_weight_quantized")]; tensor encoder_layers_22_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80227904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80817792))))[name = string("encoder_layers_22_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_22_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80819392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81016064))))[name = string("encoder_layers_22_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_22_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81016640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81213312))))[name = string("encoder_layers_22_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_22_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81213888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82098688))))[name = string("encoder_layers_22_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_22_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82101056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82985856))))[name = string("encoder_layers_22_mlp_up_proj_weight_quantized")]; tensor encoder_layers_22_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82988224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(83873024))))[name = string("encoder_layers_22_mlp_down_proj_weight_quantized")]; tensor encoder_layers_23_self_attn_q_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(83874624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84464512))))[name = string("encoder_layers_23_self_attn_q_proj_weight_quantized")]; tensor encoder_layers_23_self_attn_k_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84466112))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84662784))))[name = string("encoder_layers_23_self_attn_k_proj_weight_quantized")]; tensor encoder_layers_23_self_attn_v_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84663360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84860032))))[name = string("encoder_layers_23_self_attn_v_proj_weight_quantized")]; tensor encoder_layers_23_mlp_gate_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84860608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85745408))))[name = string("encoder_layers_23_mlp_gate_proj_weight_quantized")]; tensor encoder_layers_23_mlp_up_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85747776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86632576))))[name = string("encoder_layers_23_mlp_up_proj_weight_quantized")]; tensor encoder_layers_23_mlp_down_proj_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86634944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87519744))))[name = string("encoder_layers_23_mlp_down_proj_weight_quantized")]; tensor dense1_bias = const()[name = string("dense1_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87521344)))]; tensor dense1_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87527552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89886912))))[name = string("dense1_weight_quantized")]; tensor dense2_bias = const()[name = string("dense2_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89893120)))]; tensor dense2_weight_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89894720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92254080))))[name = string("dense2_weight_quantized")]; int32 var_22 = const()[name = string("op_22"), val = int32(-1)]; int32 var_80_batch_dims_0 = const()[name = string("op_80_batch_dims_0"), val = int32(0)]; bool var_80_validate_indices_0 = const()[name = string("op_80_validate_indices_0"), val = bool(false)]; tensor encoder_embed_tokens_weight_to_fp16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92255680))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(293582336))))[name = string("encoder_embed_tokens_weight_to_fp16_quantized")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = input_ids, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(262144)]; tensor add_0 = add(x = input_ids, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = input_ids, b = add_0, cond = greater_equal_0)[name = string("select_0")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor greater_equal_0_1 = greater_equal(x = select_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(262144)]; tensor add_0_1 = add(x = select_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = select_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_80_cast_fp16_axis_0 = const()[name = string("op_80_cast_fp16_axis_0"), val = int32(0)]; tensor op_80_cast_fp16 = gather(axis = op_80_cast_fp16_axis_0, batch_dims = var_80_batch_dims_0, indices = select_0_1, validate_indices = var_80_validate_indices_0, x = encoder_embed_tokens_weight_to_fp16_quantized)[name = string("op_80_cast_fp16")]; fp16 var_82_to_fp16 = const()[name = string("op_82_to_fp16"), val = fp16(0x1.bb8p+4)]; tensor x_1_cast_fp16 = mul(x = op_80_cast_fp16, y = var_82_to_fp16)[name = string("x_1_cast_fp16")]; tensor cos_1_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294106688))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("cos_1_quantized")]; tensor sin_1_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294172288))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("sin_1_quantized")]; tensor cos_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294237888))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("cos_quantized")]; tensor sin_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294303488))), scale = tensor([[[[0x1.02p-7]]]]))[name = string("sin_quantized")]; fp16 var_17_to_fp16 = const()[name = string("op_17_to_fp16"), val = fp16(0x1p+0)]; tensor var_92_cast_fp16 = sub(x = var_17_to_fp16, y = attention_mask)[name = string("op_92_cast_fp16")]; fp16 var_94_to_fp16 = const()[name = string("op_94_to_fp16"), val = fp16(-0x1.388p+13)]; tensor key_pad_1_cast_fp16 = mul(x = var_92_cast_fp16, y = var_94_to_fp16)[name = string("key_pad_1_cast_fp16")]; tensor var_96 = const()[name = string("op_96"), val = tensor([1, 1, 1, 256])]; tensor key_pad_cast_fp16 = reshape(shape = var_96, x = key_pad_1_cast_fp16)[name = string("key_pad_cast_fp16")]; tensor full_mask_reps_0 = const()[name = string("full_mask_reps_0"), val = tensor([1, 1, 256, 1])]; tensor full_mask_cast_fp16 = tile(reps = full_mask_reps_0, x = key_pad_cast_fp16)[name = string("full_mask_cast_fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_127_cast_fp16 = mul(x = x_1_cast_fp16, y = const_0_promoted_to_fp16)[name = string("op_127_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_22, interleave = input_1_interleave_0, values = (x_1_cast_fp16, var_127_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_8_to_fp16 = const()[name = string("op_8_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_8_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_132_split_sizes_0 = const()[name = string("op_132_split_sizes_0"), val = tensor([768, 768])]; int32 var_132_axis_0 = const()[name = string("op_132_axis_0"), val = int32(-1)]; tensor var_132_cast_fp16_0, tensor var_132_cast_fp16_1 = split(axis = var_132_axis_0, split_sizes = var_132_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_132_cast_fp16")]; tensor var_136_to_fp16 = const()[name = string("op_136_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294369088)))]; tensor out_1_cast_fp16 = mul(x = var_132_cast_fp16_0, y = var_136_to_fp16)[name = string("out_1_cast_fp16")]; tensor var_142 = const()[name = string("op_142"), val = tensor([0, 2, 1])]; tensor var_144_axes_0 = const()[name = string("op_144_axes_0"), val = tensor([2])]; tensor var_143_cast_fp16 = transpose(perm = var_142, x = out_1_cast_fp16)[name = string("transpose_215")]; tensor var_144_cast_fp16 = expand_dims(axes = var_144_axes_0, x = var_143_cast_fp16)[name = string("op_144_cast_fp16")]; string var_151_pad_type_0 = const()[name = string("op_151_pad_type_0"), val = string("valid")]; tensor var_151_strides_0 = const()[name = string("op_151_strides_0"), val = tensor([1, 1])]; tensor var_151_pad_0 = const()[name = string("op_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_151_dilations_0 = const()[name = string("op_151_dilations_0"), val = tensor([1, 1])]; int32 var_151_groups_0 = const()[name = string("op_151_groups_0"), val = int32(1)]; tensor var_151 = conv(dilations = var_151_dilations_0, groups = var_151_groups_0, pad = var_151_pad_0, pad_type = var_151_pad_type_0, strides = var_151_strides_0, weight = encoder_layers_0_self_attn_q_proj_weight_quantized, x = var_144_cast_fp16)[name = string("op_151")]; tensor var_152 = const()[name = string("op_152"), val = tensor([1, 3, 256, 256])]; tensor var_153 = reshape(shape = var_152, x = var_151)[name = string("op_153")]; tensor var_154 = const()[name = string("op_154"), val = tensor([0, 1, 3, 2])]; string var_161_pad_type_0 = const()[name = string("op_161_pad_type_0"), val = string("valid")]; tensor var_161_strides_0 = const()[name = string("op_161_strides_0"), val = tensor([1, 1])]; tensor var_161_pad_0 = const()[name = string("op_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_161_dilations_0 = const()[name = string("op_161_dilations_0"), val = tensor([1, 1])]; int32 var_161_groups_0 = const()[name = string("op_161_groups_0"), val = int32(1)]; tensor var_161 = conv(dilations = var_161_dilations_0, groups = var_161_groups_0, pad = var_161_pad_0, pad_type = var_161_pad_type_0, strides = var_161_strides_0, weight = encoder_layers_0_self_attn_k_proj_weight_quantized, x = var_144_cast_fp16)[name = string("op_161")]; tensor var_162 = const()[name = string("op_162"), val = tensor([1, 1, 256, 256])]; tensor var_163 = reshape(shape = var_162, x = var_161)[name = string("op_163")]; tensor var_164 = const()[name = string("op_164"), val = tensor([0, 1, 3, 2])]; string var_171_pad_type_0 = const()[name = string("op_171_pad_type_0"), val = string("valid")]; tensor var_171_strides_0 = const()[name = string("op_171_strides_0"), val = tensor([1, 1])]; tensor var_171_pad_0 = const()[name = string("op_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_171_dilations_0 = const()[name = string("op_171_dilations_0"), val = tensor([1, 1])]; int32 var_171_groups_0 = const()[name = string("op_171_groups_0"), val = int32(1)]; tensor var_171 = conv(dilations = var_171_dilations_0, groups = var_171_groups_0, pad = var_171_pad_0, pad_type = var_171_pad_type_0, strides = var_171_strides_0, weight = encoder_layers_0_self_attn_v_proj_weight_quantized, x = var_144_cast_fp16)[name = string("op_171")]; tensor var_172 = const()[name = string("op_172"), val = tensor([1, 1, 256, 256])]; tensor var_173 = reshape(shape = var_172, x = var_171)[name = string("op_173")]; tensor var_174 = const()[name = string("op_174"), val = tensor([0, 1, 3, 2])]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_1 = transpose(perm = var_154, x = var_153)[name = string("transpose_214")]; tensor var_180_cast_fp16 = mul(x = q_1, y = const_2_promoted_to_fp16)[name = string("op_180_cast_fp16")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5_cast_fp16 = concat(axis = var_22, interleave = input_5_interleave_0, values = (q_1, var_180_cast_fp16))[name = string("input_5_cast_fp16")]; tensor normed_7_axes_0 = const()[name = string("normed_7_axes_0"), val = tensor([-1])]; tensor normed_7_cast_fp16 = layer_norm(axes = normed_7_axes_0, epsilon = var_8_to_fp16, x = input_5_cast_fp16)[name = string("normed_7_cast_fp16")]; tensor var_185_split_sizes_0 = const()[name = string("op_185_split_sizes_0"), val = tensor([256, 256])]; int32 var_185_axis_0 = const()[name = string("op_185_axis_0"), val = int32(-1)]; tensor var_185_cast_fp16_0, tensor var_185_cast_fp16_1 = split(axis = var_185_axis_0, split_sizes = var_185_split_sizes_0, x = normed_7_cast_fp16)[name = string("op_185_cast_fp16")]; tensor var_189_to_fp16 = const()[name = string("op_189_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294370688)))]; tensor out_3_cast_fp16 = mul(x = var_185_cast_fp16_0, y = var_189_to_fp16)[name = string("out_3_cast_fp16")]; fp16 const_4_promoted_to_fp16 = const()[name = string("const_4_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_1 = transpose(perm = var_164, x = var_163)[name = string("transpose_213")]; tensor var_196_cast_fp16 = mul(x = k_1, y = const_4_promoted_to_fp16)[name = string("op_196_cast_fp16")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7_cast_fp16 = concat(axis = var_22, interleave = input_7_interleave_0, values = (k_1, var_196_cast_fp16))[name = string("input_7_cast_fp16")]; tensor normed_11_axes_0 = const()[name = string("normed_11_axes_0"), val = tensor([-1])]; tensor normed_11_cast_fp16 = layer_norm(axes = normed_11_axes_0, epsilon = var_8_to_fp16, x = input_7_cast_fp16)[name = string("normed_11_cast_fp16")]; tensor var_201_split_sizes_0 = const()[name = string("op_201_split_sizes_0"), val = tensor([256, 256])]; int32 var_201_axis_0 = const()[name = string("op_201_axis_0"), val = int32(-1)]; tensor var_201_cast_fp16_0, tensor var_201_cast_fp16_1 = split(axis = var_201_axis_0, split_sizes = var_201_split_sizes_0, x = normed_11_cast_fp16)[name = string("op_201_cast_fp16")]; tensor var_205_to_fp16 = const()[name = string("op_205_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294371264)))]; tensor out_5_cast_fp16 = mul(x = var_201_cast_fp16_0, y = var_205_to_fp16)[name = string("out_5_cast_fp16")]; tensor var_208 = mul(x = out_3_cast_fp16, y = cos_1_quantized)[name = string("op_208")]; tensor var_209_split_sizes_0 = const()[name = string("op_209_split_sizes_0"), val = tensor([128, 128])]; int32 var_209_axis_0 = const()[name = string("op_209_axis_0"), val = int32(-1)]; tensor var_209_0, tensor var_209_1 = split(axis = var_209_axis_0, split_sizes = var_209_split_sizes_0, x = out_3_cast_fp16)[name = string("op_209")]; fp16 const_6_promoted = const()[name = string("const_6_promoted"), val = fp16(-0x1p+0)]; tensor var_211 = mul(x = var_209_1, y = const_6_promoted)[name = string("op_211")]; bool var_213_interleave_0 = const()[name = string("op_213_interleave_0"), val = bool(false)]; tensor var_213 = concat(axis = var_22, interleave = var_213_interleave_0, values = (var_211, var_209_0))[name = string("op_213")]; tensor var_214 = mul(x = var_213, y = sin_1_quantized)[name = string("op_214")]; tensor q_5 = add(x = var_208, y = var_214)[name = string("q_5")]; tensor var_216 = mul(x = out_5_cast_fp16, y = cos_1_quantized)[name = string("op_216")]; tensor var_217_split_sizes_0 = const()[name = string("op_217_split_sizes_0"), val = tensor([128, 128])]; int32 var_217_axis_0 = const()[name = string("op_217_axis_0"), val = int32(-1)]; tensor var_217_0, tensor var_217_1 = split(axis = var_217_axis_0, split_sizes = var_217_split_sizes_0, x = out_5_cast_fp16)[name = string("op_217")]; fp16 const_7_promoted = const()[name = string("const_7_promoted"), val = fp16(-0x1p+0)]; tensor var_219 = mul(x = var_217_1, y = const_7_promoted)[name = string("op_219")]; bool var_221_interleave_0 = const()[name = string("op_221_interleave_0"), val = bool(false)]; tensor var_221 = concat(axis = var_22, interleave = var_221_interleave_0, values = (var_219, var_217_0))[name = string("op_221")]; tensor var_222 = mul(x = var_221, y = sin_1_quantized)[name = string("op_222")]; tensor hidden_states_1 = add(x = var_216, y = var_222)[name = string("hidden_states_1")]; tensor hidden_states_3_axes_0 = const()[name = string("hidden_states_3_axes_0"), val = tensor([2])]; tensor hidden_states_3 = expand_dims(axes = hidden_states_3_axes_0, x = hidden_states_1)[name = string("hidden_states_3")]; tensor var_225 = const()[name = string("op_225"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_5 = tile(reps = var_225, x = hidden_states_3)[name = string("hidden_states_5")]; tensor var_227 = const()[name = string("op_227"), val = tensor([1, 3, 256, 256])]; tensor k_5 = reshape(shape = var_227, x = hidden_states_5)[name = string("k_5")]; tensor hidden_states_9_axes_0 = const()[name = string("hidden_states_9_axes_0"), val = tensor([2])]; tensor hidden_states_7 = transpose(perm = var_174, x = var_173)[name = string("transpose_212")]; tensor hidden_states_9 = expand_dims(axes = hidden_states_9_axes_0, x = hidden_states_7)[name = string("hidden_states_9")]; tensor var_230 = const()[name = string("op_230"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_11 = tile(reps = var_230, x = hidden_states_9)[name = string("hidden_states_11")]; tensor var_232 = const()[name = string("op_232"), val = tensor([1, 3, 256, 256])]; tensor v_1 = reshape(shape = var_232, x = hidden_states_11)[name = string("v_1")]; bool var_237_transpose_x_1 = const()[name = string("op_237_transpose_x_1"), val = bool(false)]; bool var_237_transpose_y_1 = const()[name = string("op_237_transpose_y_1"), val = bool(true)]; tensor var_237_cast_fp16 = matmul(transpose_x = var_237_transpose_x_1, transpose_y = var_237_transpose_y_1, x = q_5, y = k_5)[name = string("op_237_cast_fp16")]; fp16 var_238_to_fp16 = const()[name = string("op_238_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_1_cast_fp16 = mul(x = var_237_cast_fp16, y = var_238_to_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor attn_weights_3_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor var_242_cast_fp16 = softmax(axis = var_22, x = attn_weights_3_cast_fp16)[name = string("op_242_cast_fp16")]; bool var_246_transpose_x_0 = const()[name = string("op_246_transpose_x_0"), val = bool(false)]; bool var_246_transpose_y_0 = const()[name = string("op_246_transpose_y_0"), val = bool(false)]; tensor var_246_cast_fp16 = matmul(transpose_x = var_246_transpose_x_0, transpose_y = var_246_transpose_y_0, x = var_242_cast_fp16, y = v_1)[name = string("op_246_cast_fp16")]; tensor var_248 = const()[name = string("op_248"), val = tensor([0, 2, 1, 3])]; tensor var_251 = const()[name = string("op_251"), val = tensor([1, 256, 768])]; tensor var_249 = transpose(perm = var_248, x = var_246_cast_fp16)[name = string("transpose_211")]; tensor attn_out_3 = reshape(shape = var_251, x = var_249)[name = string("attn_out_3")]; tensor var_253 = const()[name = string("op_253"), val = tensor([0, 2, 1])]; tensor squeeze_0_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294371840))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294961728))))[name = string("squeeze_0_quantized")]; string var_262_pad_type_0 = const()[name = string("op_262_pad_type_0"), val = string("valid")]; int32 var_262_groups_0 = const()[name = string("op_262_groups_0"), val = int32(1)]; tensor var_262_strides_0 = const()[name = string("op_262_strides_0"), val = tensor([1])]; tensor var_262_pad_0 = const()[name = string("op_262_pad_0"), val = tensor([0, 0])]; tensor var_262_dilations_0 = const()[name = string("op_262_dilations_0"), val = tensor([1])]; tensor var_254 = transpose(perm = var_253, x = attn_out_3)[name = string("transpose_210")]; tensor var_262 = conv(dilations = var_262_dilations_0, groups = var_262_groups_0, pad = var_262_pad_0, pad_type = var_262_pad_type_0, strides = var_262_strides_0, weight = squeeze_0_quantized, x = var_254)[name = string("op_262")]; tensor var_263 = const()[name = string("op_263"), val = tensor([0, 2, 1])]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_9 = transpose(perm = var_263, x = var_262)[name = string("transpose_209")]; tensor var_267_cast_fp16 = mul(x = x_9, y = const_8_promoted_to_fp16)[name = string("op_267_cast_fp16")]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11_cast_fp16 = concat(axis = var_22, interleave = input_11_interleave_0, values = (x_9, var_267_cast_fp16))[name = string("input_11_cast_fp16")]; tensor normed_15_axes_0 = const()[name = string("normed_15_axes_0"), val = tensor([-1])]; tensor normed_15_cast_fp16 = layer_norm(axes = normed_15_axes_0, epsilon = var_8_to_fp16, x = input_11_cast_fp16)[name = string("normed_15_cast_fp16")]; tensor var_272_split_sizes_0 = const()[name = string("op_272_split_sizes_0"), val = tensor([768, 768])]; int32 var_272_axis_0 = const()[name = string("op_272_axis_0"), val = int32(-1)]; tensor var_272_cast_fp16_0, tensor var_272_cast_fp16_1 = split(axis = var_272_axis_0, split_sizes = var_272_split_sizes_0, x = normed_15_cast_fp16)[name = string("op_272_cast_fp16")]; tensor var_276_to_fp16 = const()[name = string("op_276_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294963328)))]; tensor out_7_cast_fp16 = mul(x = var_272_cast_fp16_0, y = var_276_to_fp16)[name = string("out_7_cast_fp16")]; tensor x_11_cast_fp16 = add(x = x_1_cast_fp16, y = out_7_cast_fp16)[name = string("x_11_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_283_cast_fp16 = mul(x = x_11_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_283_cast_fp16")]; bool input_13_interleave_0 = const()[name = string("input_13_interleave_0"), val = bool(false)]; tensor input_13_cast_fp16 = concat(axis = var_22, interleave = input_13_interleave_0, values = (x_11_cast_fp16, var_283_cast_fp16))[name = string("input_13_cast_fp16")]; tensor normed_19_axes_0 = const()[name = string("normed_19_axes_0"), val = tensor([-1])]; tensor normed_19_cast_fp16 = layer_norm(axes = normed_19_axes_0, epsilon = var_8_to_fp16, x = input_13_cast_fp16)[name = string("normed_19_cast_fp16")]; tensor var_288_split_sizes_0 = const()[name = string("op_288_split_sizes_0"), val = tensor([768, 768])]; int32 var_288_axis_0 = const()[name = string("op_288_axis_0"), val = int32(-1)]; tensor var_288_cast_fp16_0, tensor var_288_cast_fp16_1 = split(axis = var_288_axis_0, split_sizes = var_288_split_sizes_0, x = normed_19_cast_fp16)[name = string("op_288_cast_fp16")]; tensor var_292_to_fp16 = const()[name = string("op_292_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294964928)))]; tensor out_9_cast_fp16 = mul(x = var_288_cast_fp16_0, y = var_292_to_fp16)[name = string("out_9_cast_fp16")]; tensor var_299 = const()[name = string("op_299"), val = tensor([0, 2, 1])]; tensor input_15_axes_0 = const()[name = string("input_15_axes_0"), val = tensor([2])]; tensor var_300 = transpose(perm = var_299, x = out_9_cast_fp16)[name = string("transpose_208")]; tensor input_15 = expand_dims(axes = input_15_axes_0, x = var_300)[name = string("input_15")]; string gate_1_pad_type_0 = const()[name = string("gate_1_pad_type_0"), val = string("valid")]; tensor gate_1_strides_0 = const()[name = string("gate_1_strides_0"), val = tensor([1, 1])]; tensor gate_1_pad_0 = const()[name = string("gate_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_1_dilations_0 = const()[name = string("gate_1_dilations_0"), val = tensor([1, 1])]; int32 gate_1_groups_0 = const()[name = string("gate_1_groups_0"), val = int32(1)]; tensor gate_1 = conv(dilations = gate_1_dilations_0, groups = gate_1_groups_0, pad = gate_1_pad_0, pad_type = gate_1_pad_type_0, strides = gate_1_strides_0, weight = encoder_layers_0_mlp_gate_proj_weight_quantized, x = input_15)[name = string("gate_1")]; string up_1_pad_type_0 = const()[name = string("up_1_pad_type_0"), val = string("valid")]; tensor up_1_strides_0 = const()[name = string("up_1_strides_0"), val = tensor([1, 1])]; tensor up_1_pad_0 = const()[name = string("up_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_1_dilations_0 = const()[name = string("up_1_dilations_0"), val = tensor([1, 1])]; int32 up_1_groups_0 = const()[name = string("up_1_groups_0"), val = int32(1)]; tensor up_1 = conv(dilations = up_1_dilations_0, groups = up_1_groups_0, pad = up_1_pad_0, pad_type = up_1_pad_type_0, strides = up_1_strides_0, weight = encoder_layers_0_mlp_up_proj_weight_quantized, x = input_15)[name = string("up_1")]; string gate_3_mode_0 = const()[name = string("gate_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_3 = gelu(mode = gate_3_mode_0, x = gate_1)[name = string("gate_3")]; tensor input_17 = mul(x = gate_3, y = up_1)[name = string("input_17")]; string var_321_pad_type_0 = const()[name = string("op_321_pad_type_0"), val = string("valid")]; tensor var_321_strides_0 = const()[name = string("op_321_strides_0"), val = tensor([1, 1])]; tensor var_321_pad_0 = const()[name = string("op_321_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_321_dilations_0 = const()[name = string("op_321_dilations_0"), val = tensor([1, 1])]; int32 var_321_groups_0 = const()[name = string("op_321_groups_0"), val = int32(1)]; tensor var_321 = conv(dilations = var_321_dilations_0, groups = var_321_groups_0, pad = var_321_pad_0, pad_type = var_321_pad_type_0, strides = var_321_strides_0, weight = encoder_layers_0_mlp_down_proj_weight_quantized, x = input_17)[name = string("op_321")]; tensor var_322_axes_0 = const()[name = string("op_322_axes_0"), val = tensor([2])]; tensor var_322 = squeeze(axes = var_322_axes_0, x = var_321)[name = string("op_322")]; tensor var_323 = const()[name = string("op_323"), val = tensor([0, 2, 1])]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_15 = transpose(perm = var_323, x = var_322)[name = string("transpose_207")]; tensor var_327_cast_fp16 = mul(x = x_15, y = const_12_promoted_to_fp16)[name = string("op_327_cast_fp16")]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_22, interleave = input_19_interleave_0, values = (x_15, var_327_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_8_to_fp16, x = input_19_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor var_332_split_sizes_0 = const()[name = string("op_332_split_sizes_0"), val = tensor([768, 768])]; int32 var_332_axis_0 = const()[name = string("op_332_axis_0"), val = int32(-1)]; tensor var_332_cast_fp16_0, tensor var_332_cast_fp16_1 = split(axis = var_332_axis_0, split_sizes = var_332_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_332_cast_fp16")]; tensor var_336_to_fp16 = const()[name = string("op_336_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294966528)))]; tensor out_11_cast_fp16 = mul(x = var_332_cast_fp16_0, y = var_336_to_fp16)[name = string("out_11_cast_fp16")]; tensor x_17_cast_fp16 = add(x = x_11_cast_fp16, y = out_11_cast_fp16)[name = string("x_17_cast_fp16")]; fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_365_cast_fp16 = mul(x = x_17_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_365_cast_fp16")]; bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; tensor input_21_cast_fp16 = concat(axis = var_22, interleave = input_21_interleave_0, values = (x_17_cast_fp16, var_365_cast_fp16))[name = string("input_21_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_8_to_fp16, x = input_21_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_370_split_sizes_0 = const()[name = string("op_370_split_sizes_0"), val = tensor([768, 768])]; int32 var_370_axis_0 = const()[name = string("op_370_axis_0"), val = int32(-1)]; tensor var_370_cast_fp16_0, tensor var_370_cast_fp16_1 = split(axis = var_370_axis_0, split_sizes = var_370_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_370_cast_fp16")]; tensor var_374_to_fp16 = const()[name = string("op_374_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294968128)))]; tensor out_13_cast_fp16 = mul(x = var_370_cast_fp16_0, y = var_374_to_fp16)[name = string("out_13_cast_fp16")]; tensor var_380 = const()[name = string("op_380"), val = tensor([0, 2, 1])]; tensor var_382_axes_0 = const()[name = string("op_382_axes_0"), val = tensor([2])]; tensor var_381_cast_fp16 = transpose(perm = var_380, x = out_13_cast_fp16)[name = string("transpose_206")]; tensor var_382_cast_fp16 = expand_dims(axes = var_382_axes_0, x = var_381_cast_fp16)[name = string("op_382_cast_fp16")]; string var_389_pad_type_0 = const()[name = string("op_389_pad_type_0"), val = string("valid")]; tensor var_389_strides_0 = const()[name = string("op_389_strides_0"), val = tensor([1, 1])]; tensor var_389_pad_0 = const()[name = string("op_389_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_389_dilations_0 = const()[name = string("op_389_dilations_0"), val = tensor([1, 1])]; int32 var_389_groups_0 = const()[name = string("op_389_groups_0"), val = int32(1)]; tensor var_389 = conv(dilations = var_389_dilations_0, groups = var_389_groups_0, pad = var_389_pad_0, pad_type = var_389_pad_type_0, strides = var_389_strides_0, weight = encoder_layers_1_self_attn_q_proj_weight_quantized, x = var_382_cast_fp16)[name = string("op_389")]; tensor var_390 = const()[name = string("op_390"), val = tensor([1, 3, 256, 256])]; tensor var_391 = reshape(shape = var_390, x = var_389)[name = string("op_391")]; tensor var_392 = const()[name = string("op_392"), val = tensor([0, 1, 3, 2])]; string var_399_pad_type_0 = const()[name = string("op_399_pad_type_0"), val = string("valid")]; tensor var_399_strides_0 = const()[name = string("op_399_strides_0"), val = tensor([1, 1])]; tensor var_399_pad_0 = const()[name = string("op_399_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_399_dilations_0 = const()[name = string("op_399_dilations_0"), val = tensor([1, 1])]; int32 var_399_groups_0 = const()[name = string("op_399_groups_0"), val = int32(1)]; tensor var_399 = conv(dilations = var_399_dilations_0, groups = var_399_groups_0, pad = var_399_pad_0, pad_type = var_399_pad_type_0, strides = var_399_strides_0, weight = encoder_layers_1_self_attn_k_proj_weight_quantized, x = var_382_cast_fp16)[name = string("op_399")]; tensor var_400 = const()[name = string("op_400"), val = tensor([1, 1, 256, 256])]; tensor var_401 = reshape(shape = var_400, x = var_399)[name = string("op_401")]; tensor var_402 = const()[name = string("op_402"), val = tensor([0, 1, 3, 2])]; string var_409_pad_type_0 = const()[name = string("op_409_pad_type_0"), val = string("valid")]; tensor var_409_strides_0 = const()[name = string("op_409_strides_0"), val = tensor([1, 1])]; tensor var_409_pad_0 = const()[name = string("op_409_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_409_dilations_0 = const()[name = string("op_409_dilations_0"), val = tensor([1, 1])]; int32 var_409_groups_0 = const()[name = string("op_409_groups_0"), val = int32(1)]; tensor var_409 = conv(dilations = var_409_dilations_0, groups = var_409_groups_0, pad = var_409_pad_0, pad_type = var_409_pad_type_0, strides = var_409_strides_0, weight = encoder_layers_1_self_attn_v_proj_weight_quantized, x = var_382_cast_fp16)[name = string("op_409")]; tensor var_410 = const()[name = string("op_410"), val = tensor([1, 1, 256, 256])]; tensor var_411 = reshape(shape = var_410, x = var_409)[name = string("op_411")]; tensor var_412 = const()[name = string("op_412"), val = tensor([0, 1, 3, 2])]; fp16 const_16_promoted_to_fp16 = const()[name = string("const_16_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_7 = transpose(perm = var_392, x = var_391)[name = string("transpose_205")]; tensor var_418_cast_fp16 = mul(x = q_7, y = const_16_promoted_to_fp16)[name = string("op_418_cast_fp16")]; bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; tensor input_25_cast_fp16 = concat(axis = var_22, interleave = input_25_interleave_0, values = (q_7, var_418_cast_fp16))[name = string("input_25_cast_fp16")]; tensor normed_35_axes_0 = const()[name = string("normed_35_axes_0"), val = tensor([-1])]; tensor normed_35_cast_fp16 = layer_norm(axes = normed_35_axes_0, epsilon = var_8_to_fp16, x = input_25_cast_fp16)[name = string("normed_35_cast_fp16")]; tensor var_423_split_sizes_0 = const()[name = string("op_423_split_sizes_0"), val = tensor([256, 256])]; int32 var_423_axis_0 = const()[name = string("op_423_axis_0"), val = int32(-1)]; tensor var_423_cast_fp16_0, tensor var_423_cast_fp16_1 = split(axis = var_423_axis_0, split_sizes = var_423_split_sizes_0, x = normed_35_cast_fp16)[name = string("op_423_cast_fp16")]; tensor var_427_to_fp16 = const()[name = string("op_427_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294969728)))]; tensor out_15_cast_fp16 = mul(x = var_423_cast_fp16_0, y = var_427_to_fp16)[name = string("out_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_7 = transpose(perm = var_402, x = var_401)[name = string("transpose_204")]; tensor var_434_cast_fp16 = mul(x = k_7, y = const_18_promoted_to_fp16)[name = string("op_434_cast_fp16")]; bool input_27_interleave_0 = const()[name = string("input_27_interleave_0"), val = bool(false)]; tensor input_27_cast_fp16 = concat(axis = var_22, interleave = input_27_interleave_0, values = (k_7, var_434_cast_fp16))[name = string("input_27_cast_fp16")]; tensor normed_39_axes_0 = const()[name = string("normed_39_axes_0"), val = tensor([-1])]; tensor normed_39_cast_fp16 = layer_norm(axes = normed_39_axes_0, epsilon = var_8_to_fp16, x = input_27_cast_fp16)[name = string("normed_39_cast_fp16")]; tensor var_439_split_sizes_0 = const()[name = string("op_439_split_sizes_0"), val = tensor([256, 256])]; int32 var_439_axis_0 = const()[name = string("op_439_axis_0"), val = int32(-1)]; tensor var_439_cast_fp16_0, tensor var_439_cast_fp16_1 = split(axis = var_439_axis_0, split_sizes = var_439_split_sizes_0, x = normed_39_cast_fp16)[name = string("op_439_cast_fp16")]; tensor var_443_to_fp16 = const()[name = string("op_443_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294970304)))]; tensor out_17_cast_fp16 = mul(x = var_439_cast_fp16_0, y = var_443_to_fp16)[name = string("out_17_cast_fp16")]; tensor var_446 = mul(x = out_15_cast_fp16, y = cos_1_quantized)[name = string("op_446")]; tensor var_447_split_sizes_0 = const()[name = string("op_447_split_sizes_0"), val = tensor([128, 128])]; int32 var_447_axis_0 = const()[name = string("op_447_axis_0"), val = int32(-1)]; tensor var_447_0, tensor var_447_1 = split(axis = var_447_axis_0, split_sizes = var_447_split_sizes_0, x = out_15_cast_fp16)[name = string("op_447")]; fp16 const_20_promoted = const()[name = string("const_20_promoted"), val = fp16(-0x1p+0)]; tensor var_449 = mul(x = var_447_1, y = const_20_promoted)[name = string("op_449")]; bool var_451_interleave_0 = const()[name = string("op_451_interleave_0"), val = bool(false)]; tensor var_451 = concat(axis = var_22, interleave = var_451_interleave_0, values = (var_449, var_447_0))[name = string("op_451")]; tensor var_452 = mul(x = var_451, y = sin_1_quantized)[name = string("op_452")]; tensor q_11 = add(x = var_446, y = var_452)[name = string("q_11")]; tensor var_454 = mul(x = out_17_cast_fp16, y = cos_1_quantized)[name = string("op_454")]; tensor var_455_split_sizes_0 = const()[name = string("op_455_split_sizes_0"), val = tensor([128, 128])]; int32 var_455_axis_0 = const()[name = string("op_455_axis_0"), val = int32(-1)]; tensor var_455_0, tensor var_455_1 = split(axis = var_455_axis_0, split_sizes = var_455_split_sizes_0, x = out_17_cast_fp16)[name = string("op_455")]; fp16 const_21_promoted = const()[name = string("const_21_promoted"), val = fp16(-0x1p+0)]; tensor var_457 = mul(x = var_455_1, y = const_21_promoted)[name = string("op_457")]; bool var_459_interleave_0 = const()[name = string("op_459_interleave_0"), val = bool(false)]; tensor var_459 = concat(axis = var_22, interleave = var_459_interleave_0, values = (var_457, var_455_0))[name = string("op_459")]; tensor var_460 = mul(x = var_459, y = sin_1_quantized)[name = string("op_460")]; tensor hidden_states_13 = add(x = var_454, y = var_460)[name = string("hidden_states_13")]; tensor hidden_states_15_axes_0 = const()[name = string("hidden_states_15_axes_0"), val = tensor([2])]; tensor hidden_states_15 = expand_dims(axes = hidden_states_15_axes_0, x = hidden_states_13)[name = string("hidden_states_15")]; tensor var_463 = const()[name = string("op_463"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_17 = tile(reps = var_463, x = hidden_states_15)[name = string("hidden_states_17")]; tensor var_465 = const()[name = string("op_465"), val = tensor([1, 3, 256, 256])]; tensor k_11 = reshape(shape = var_465, x = hidden_states_17)[name = string("k_11")]; tensor hidden_states_21_axes_0 = const()[name = string("hidden_states_21_axes_0"), val = tensor([2])]; tensor hidden_states_19 = transpose(perm = var_412, x = var_411)[name = string("transpose_203")]; tensor hidden_states_21 = expand_dims(axes = hidden_states_21_axes_0, x = hidden_states_19)[name = string("hidden_states_21")]; tensor var_468 = const()[name = string("op_468"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_23 = tile(reps = var_468, x = hidden_states_21)[name = string("hidden_states_23")]; tensor var_470 = const()[name = string("op_470"), val = tensor([1, 3, 256, 256])]; tensor v_3 = reshape(shape = var_470, x = hidden_states_23)[name = string("v_3")]; bool var_475_transpose_x_1 = const()[name = string("op_475_transpose_x_1"), val = bool(false)]; bool var_475_transpose_y_1 = const()[name = string("op_475_transpose_y_1"), val = bool(true)]; tensor var_475_cast_fp16 = matmul(transpose_x = var_475_transpose_x_1, transpose_y = var_475_transpose_y_1, x = q_11, y = k_11)[name = string("op_475_cast_fp16")]; fp16 var_476_to_fp16 = const()[name = string("op_476_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_7_cast_fp16 = mul(x = var_475_cast_fp16, y = var_476_to_fp16)[name = string("attn_weights_7_cast_fp16")]; tensor attn_weights_9_cast_fp16 = add(x = attn_weights_7_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor var_480_cast_fp16 = softmax(axis = var_22, x = attn_weights_9_cast_fp16)[name = string("op_480_cast_fp16")]; bool var_484_transpose_x_0 = const()[name = string("op_484_transpose_x_0"), val = bool(false)]; bool var_484_transpose_y_0 = const()[name = string("op_484_transpose_y_0"), val = bool(false)]; tensor var_484_cast_fp16 = matmul(transpose_x = var_484_transpose_x_0, transpose_y = var_484_transpose_y_0, x = var_480_cast_fp16, y = v_3)[name = string("op_484_cast_fp16")]; tensor var_486 = const()[name = string("op_486"), val = tensor([0, 2, 1, 3])]; tensor var_489 = const()[name = string("op_489"), val = tensor([1, 256, 768])]; tensor var_487 = transpose(perm = var_486, x = var_484_cast_fp16)[name = string("transpose_202")]; tensor attn_out_9 = reshape(shape = var_489, x = var_487)[name = string("attn_out_9")]; tensor var_491 = const()[name = string("op_491"), val = tensor([0, 2, 1])]; tensor squeeze_1_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294970880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295560768))))[name = string("squeeze_1_quantized")]; string var_500_pad_type_0 = const()[name = string("op_500_pad_type_0"), val = string("valid")]; int32 var_500_groups_0 = const()[name = string("op_500_groups_0"), val = int32(1)]; tensor var_500_strides_0 = const()[name = string("op_500_strides_0"), val = tensor([1])]; tensor var_500_pad_0 = const()[name = string("op_500_pad_0"), val = tensor([0, 0])]; tensor var_500_dilations_0 = const()[name = string("op_500_dilations_0"), val = tensor([1])]; tensor var_492 = transpose(perm = var_491, x = attn_out_9)[name = string("transpose_201")]; tensor var_500 = conv(dilations = var_500_dilations_0, groups = var_500_groups_0, pad = var_500_pad_0, pad_type = var_500_pad_type_0, strides = var_500_strides_0, weight = squeeze_1_quantized, x = var_492)[name = string("op_500")]; tensor var_501 = const()[name = string("op_501"), val = tensor([0, 2, 1])]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_25 = transpose(perm = var_501, x = var_500)[name = string("transpose_200")]; tensor var_505_cast_fp16 = mul(x = x_25, y = const_22_promoted_to_fp16)[name = string("op_505_cast_fp16")]; bool input_31_interleave_0 = const()[name = string("input_31_interleave_0"), val = bool(false)]; tensor input_31_cast_fp16 = concat(axis = var_22, interleave = input_31_interleave_0, values = (x_25, var_505_cast_fp16))[name = string("input_31_cast_fp16")]; tensor normed_43_axes_0 = const()[name = string("normed_43_axes_0"), val = tensor([-1])]; tensor normed_43_cast_fp16 = layer_norm(axes = normed_43_axes_0, epsilon = var_8_to_fp16, x = input_31_cast_fp16)[name = string("normed_43_cast_fp16")]; tensor var_510_split_sizes_0 = const()[name = string("op_510_split_sizes_0"), val = tensor([768, 768])]; int32 var_510_axis_0 = const()[name = string("op_510_axis_0"), val = int32(-1)]; tensor var_510_cast_fp16_0, tensor var_510_cast_fp16_1 = split(axis = var_510_axis_0, split_sizes = var_510_split_sizes_0, x = normed_43_cast_fp16)[name = string("op_510_cast_fp16")]; tensor var_514_to_fp16 = const()[name = string("op_514_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295562368)))]; tensor out_19_cast_fp16 = mul(x = var_510_cast_fp16_0, y = var_514_to_fp16)[name = string("out_19_cast_fp16")]; tensor x_27_cast_fp16 = add(x = x_17_cast_fp16, y = out_19_cast_fp16)[name = string("x_27_cast_fp16")]; fp16 const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_521_cast_fp16 = mul(x = x_27_cast_fp16, y = const_24_promoted_to_fp16)[name = string("op_521_cast_fp16")]; bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; tensor input_33_cast_fp16 = concat(axis = var_22, interleave = input_33_interleave_0, values = (x_27_cast_fp16, var_521_cast_fp16))[name = string("input_33_cast_fp16")]; tensor normed_47_axes_0 = const()[name = string("normed_47_axes_0"), val = tensor([-1])]; tensor normed_47_cast_fp16 = layer_norm(axes = normed_47_axes_0, epsilon = var_8_to_fp16, x = input_33_cast_fp16)[name = string("normed_47_cast_fp16")]; tensor var_526_split_sizes_0 = const()[name = string("op_526_split_sizes_0"), val = tensor([768, 768])]; int32 var_526_axis_0 = const()[name = string("op_526_axis_0"), val = int32(-1)]; tensor var_526_cast_fp16_0, tensor var_526_cast_fp16_1 = split(axis = var_526_axis_0, split_sizes = var_526_split_sizes_0, x = normed_47_cast_fp16)[name = string("op_526_cast_fp16")]; tensor var_530_to_fp16 = const()[name = string("op_530_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295563968)))]; tensor out_21_cast_fp16 = mul(x = var_526_cast_fp16_0, y = var_530_to_fp16)[name = string("out_21_cast_fp16")]; tensor var_537 = const()[name = string("op_537"), val = tensor([0, 2, 1])]; tensor input_35_axes_0 = const()[name = string("input_35_axes_0"), val = tensor([2])]; tensor var_538 = transpose(perm = var_537, x = out_21_cast_fp16)[name = string("transpose_199")]; tensor input_35 = expand_dims(axes = input_35_axes_0, x = var_538)[name = string("input_35")]; string gate_5_pad_type_0 = const()[name = string("gate_5_pad_type_0"), val = string("valid")]; tensor gate_5_strides_0 = const()[name = string("gate_5_strides_0"), val = tensor([1, 1])]; tensor gate_5_pad_0 = const()[name = string("gate_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_5_dilations_0 = const()[name = string("gate_5_dilations_0"), val = tensor([1, 1])]; int32 gate_5_groups_0 = const()[name = string("gate_5_groups_0"), val = int32(1)]; tensor gate_5 = conv(dilations = gate_5_dilations_0, groups = gate_5_groups_0, pad = gate_5_pad_0, pad_type = gate_5_pad_type_0, strides = gate_5_strides_0, weight = encoder_layers_1_mlp_gate_proj_weight_quantized, x = input_35)[name = string("gate_5")]; string up_3_pad_type_0 = const()[name = string("up_3_pad_type_0"), val = string("valid")]; tensor up_3_strides_0 = const()[name = string("up_3_strides_0"), val = tensor([1, 1])]; tensor up_3_pad_0 = const()[name = string("up_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_3_dilations_0 = const()[name = string("up_3_dilations_0"), val = tensor([1, 1])]; int32 up_3_groups_0 = const()[name = string("up_3_groups_0"), val = int32(1)]; tensor up_3 = conv(dilations = up_3_dilations_0, groups = up_3_groups_0, pad = up_3_pad_0, pad_type = up_3_pad_type_0, strides = up_3_strides_0, weight = encoder_layers_1_mlp_up_proj_weight_quantized, x = input_35)[name = string("up_3")]; string gate_7_mode_0 = const()[name = string("gate_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_7 = gelu(mode = gate_7_mode_0, x = gate_5)[name = string("gate_7")]; tensor input_37 = mul(x = gate_7, y = up_3)[name = string("input_37")]; string var_559_pad_type_0 = const()[name = string("op_559_pad_type_0"), val = string("valid")]; tensor var_559_strides_0 = const()[name = string("op_559_strides_0"), val = tensor([1, 1])]; tensor var_559_pad_0 = const()[name = string("op_559_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_559_dilations_0 = const()[name = string("op_559_dilations_0"), val = tensor([1, 1])]; int32 var_559_groups_0 = const()[name = string("op_559_groups_0"), val = int32(1)]; tensor var_559 = conv(dilations = var_559_dilations_0, groups = var_559_groups_0, pad = var_559_pad_0, pad_type = var_559_pad_type_0, strides = var_559_strides_0, weight = encoder_layers_1_mlp_down_proj_weight_quantized, x = input_37)[name = string("op_559")]; tensor var_560_axes_0 = const()[name = string("op_560_axes_0"), val = tensor([2])]; tensor var_560 = squeeze(axes = var_560_axes_0, x = var_559)[name = string("op_560")]; tensor var_561 = const()[name = string("op_561"), val = tensor([0, 2, 1])]; fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_31 = transpose(perm = var_561, x = var_560)[name = string("transpose_198")]; tensor var_565_cast_fp16 = mul(x = x_31, y = const_26_promoted_to_fp16)[name = string("op_565_cast_fp16")]; bool input_39_interleave_0 = const()[name = string("input_39_interleave_0"), val = bool(false)]; tensor input_39_cast_fp16 = concat(axis = var_22, interleave = input_39_interleave_0, values = (x_31, var_565_cast_fp16))[name = string("input_39_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_8_to_fp16, x = input_39_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor var_570_split_sizes_0 = const()[name = string("op_570_split_sizes_0"), val = tensor([768, 768])]; int32 var_570_axis_0 = const()[name = string("op_570_axis_0"), val = int32(-1)]; tensor var_570_cast_fp16_0, tensor var_570_cast_fp16_1 = split(axis = var_570_axis_0, split_sizes = var_570_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_570_cast_fp16")]; tensor var_574_to_fp16 = const()[name = string("op_574_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295565568)))]; tensor out_23_cast_fp16 = mul(x = var_570_cast_fp16_0, y = var_574_to_fp16)[name = string("out_23_cast_fp16")]; tensor x_33_cast_fp16 = add(x = x_27_cast_fp16, y = out_23_cast_fp16)[name = string("x_33_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_603_cast_fp16 = mul(x = x_33_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_603_cast_fp16")]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41_cast_fp16 = concat(axis = var_22, interleave = input_41_interleave_0, values = (x_33_cast_fp16, var_603_cast_fp16))[name = string("input_41_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_8_to_fp16, x = input_41_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_608_split_sizes_0 = const()[name = string("op_608_split_sizes_0"), val = tensor([768, 768])]; int32 var_608_axis_0 = const()[name = string("op_608_axis_0"), val = int32(-1)]; tensor var_608_cast_fp16_0, tensor var_608_cast_fp16_1 = split(axis = var_608_axis_0, split_sizes = var_608_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_608_cast_fp16")]; tensor var_612_to_fp16 = const()[name = string("op_612_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295567168)))]; tensor out_25_cast_fp16 = mul(x = var_608_cast_fp16_0, y = var_612_to_fp16)[name = string("out_25_cast_fp16")]; tensor var_618 = const()[name = string("op_618"), val = tensor([0, 2, 1])]; tensor var_620_axes_0 = const()[name = string("op_620_axes_0"), val = tensor([2])]; tensor var_619_cast_fp16 = transpose(perm = var_618, x = out_25_cast_fp16)[name = string("transpose_197")]; tensor var_620_cast_fp16 = expand_dims(axes = var_620_axes_0, x = var_619_cast_fp16)[name = string("op_620_cast_fp16")]; string var_627_pad_type_0 = const()[name = string("op_627_pad_type_0"), val = string("valid")]; tensor var_627_strides_0 = const()[name = string("op_627_strides_0"), val = tensor([1, 1])]; tensor var_627_pad_0 = const()[name = string("op_627_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_627_dilations_0 = const()[name = string("op_627_dilations_0"), val = tensor([1, 1])]; int32 var_627_groups_0 = const()[name = string("op_627_groups_0"), val = int32(1)]; tensor var_627 = conv(dilations = var_627_dilations_0, groups = var_627_groups_0, pad = var_627_pad_0, pad_type = var_627_pad_type_0, strides = var_627_strides_0, weight = encoder_layers_2_self_attn_q_proj_weight_quantized, x = var_620_cast_fp16)[name = string("op_627")]; tensor var_628 = const()[name = string("op_628"), val = tensor([1, 3, 256, 256])]; tensor var_629 = reshape(shape = var_628, x = var_627)[name = string("op_629")]; tensor var_630 = const()[name = string("op_630"), val = tensor([0, 1, 3, 2])]; string var_637_pad_type_0 = const()[name = string("op_637_pad_type_0"), val = string("valid")]; tensor var_637_strides_0 = const()[name = string("op_637_strides_0"), val = tensor([1, 1])]; tensor var_637_pad_0 = const()[name = string("op_637_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_637_dilations_0 = const()[name = string("op_637_dilations_0"), val = tensor([1, 1])]; int32 var_637_groups_0 = const()[name = string("op_637_groups_0"), val = int32(1)]; tensor var_637 = conv(dilations = var_637_dilations_0, groups = var_637_groups_0, pad = var_637_pad_0, pad_type = var_637_pad_type_0, strides = var_637_strides_0, weight = encoder_layers_2_self_attn_k_proj_weight_quantized, x = var_620_cast_fp16)[name = string("op_637")]; tensor var_638 = const()[name = string("op_638"), val = tensor([1, 1, 256, 256])]; tensor var_639 = reshape(shape = var_638, x = var_637)[name = string("op_639")]; tensor var_640 = const()[name = string("op_640"), val = tensor([0, 1, 3, 2])]; string var_647_pad_type_0 = const()[name = string("op_647_pad_type_0"), val = string("valid")]; tensor var_647_strides_0 = const()[name = string("op_647_strides_0"), val = tensor([1, 1])]; tensor var_647_pad_0 = const()[name = string("op_647_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_647_dilations_0 = const()[name = string("op_647_dilations_0"), val = tensor([1, 1])]; int32 var_647_groups_0 = const()[name = string("op_647_groups_0"), val = int32(1)]; tensor var_647 = conv(dilations = var_647_dilations_0, groups = var_647_groups_0, pad = var_647_pad_0, pad_type = var_647_pad_type_0, strides = var_647_strides_0, weight = encoder_layers_2_self_attn_v_proj_weight_quantized, x = var_620_cast_fp16)[name = string("op_647")]; tensor var_648 = const()[name = string("op_648"), val = tensor([1, 1, 256, 256])]; tensor var_649 = reshape(shape = var_648, x = var_647)[name = string("op_649")]; tensor var_650 = const()[name = string("op_650"), val = tensor([0, 1, 3, 2])]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_13 = transpose(perm = var_630, x = var_629)[name = string("transpose_196")]; tensor var_656_cast_fp16 = mul(x = q_13, y = const_30_promoted_to_fp16)[name = string("op_656_cast_fp16")]; bool input_45_interleave_0 = const()[name = string("input_45_interleave_0"), val = bool(false)]; tensor input_45_cast_fp16 = concat(axis = var_22, interleave = input_45_interleave_0, values = (q_13, var_656_cast_fp16))[name = string("input_45_cast_fp16")]; tensor normed_63_axes_0 = const()[name = string("normed_63_axes_0"), val = tensor([-1])]; tensor normed_63_cast_fp16 = layer_norm(axes = normed_63_axes_0, epsilon = var_8_to_fp16, x = input_45_cast_fp16)[name = string("normed_63_cast_fp16")]; tensor var_661_split_sizes_0 = const()[name = string("op_661_split_sizes_0"), val = tensor([256, 256])]; int32 var_661_axis_0 = const()[name = string("op_661_axis_0"), val = int32(-1)]; tensor var_661_cast_fp16_0, tensor var_661_cast_fp16_1 = split(axis = var_661_axis_0, split_sizes = var_661_split_sizes_0, x = normed_63_cast_fp16)[name = string("op_661_cast_fp16")]; tensor var_665_to_fp16 = const()[name = string("op_665_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295568768)))]; tensor out_27_cast_fp16 = mul(x = var_661_cast_fp16_0, y = var_665_to_fp16)[name = string("out_27_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_13 = transpose(perm = var_640, x = var_639)[name = string("transpose_195")]; tensor var_672_cast_fp16 = mul(x = k_13, y = const_32_promoted_to_fp16)[name = string("op_672_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_22, interleave = input_47_interleave_0, values = (k_13, var_672_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_67_axes_0 = const()[name = string("normed_67_axes_0"), val = tensor([-1])]; tensor normed_67_cast_fp16 = layer_norm(axes = normed_67_axes_0, epsilon = var_8_to_fp16, x = input_47_cast_fp16)[name = string("normed_67_cast_fp16")]; tensor var_677_split_sizes_0 = const()[name = string("op_677_split_sizes_0"), val = tensor([256, 256])]; int32 var_677_axis_0 = const()[name = string("op_677_axis_0"), val = int32(-1)]; tensor var_677_cast_fp16_0, tensor var_677_cast_fp16_1 = split(axis = var_677_axis_0, split_sizes = var_677_split_sizes_0, x = normed_67_cast_fp16)[name = string("op_677_cast_fp16")]; tensor var_681_to_fp16 = const()[name = string("op_681_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295569344)))]; tensor out_29_cast_fp16 = mul(x = var_677_cast_fp16_0, y = var_681_to_fp16)[name = string("out_29_cast_fp16")]; tensor var_684 = mul(x = out_27_cast_fp16, y = cos_1_quantized)[name = string("op_684")]; tensor var_685_split_sizes_0 = const()[name = string("op_685_split_sizes_0"), val = tensor([128, 128])]; int32 var_685_axis_0 = const()[name = string("op_685_axis_0"), val = int32(-1)]; tensor var_685_0, tensor var_685_1 = split(axis = var_685_axis_0, split_sizes = var_685_split_sizes_0, x = out_27_cast_fp16)[name = string("op_685")]; fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; tensor var_687 = mul(x = var_685_1, y = const_34_promoted)[name = string("op_687")]; bool var_689_interleave_0 = const()[name = string("op_689_interleave_0"), val = bool(false)]; tensor var_689 = concat(axis = var_22, interleave = var_689_interleave_0, values = (var_687, var_685_0))[name = string("op_689")]; tensor var_690 = mul(x = var_689, y = sin_1_quantized)[name = string("op_690")]; tensor q_17 = add(x = var_684, y = var_690)[name = string("q_17")]; tensor var_692 = mul(x = out_29_cast_fp16, y = cos_1_quantized)[name = string("op_692")]; tensor var_693_split_sizes_0 = const()[name = string("op_693_split_sizes_0"), val = tensor([128, 128])]; int32 var_693_axis_0 = const()[name = string("op_693_axis_0"), val = int32(-1)]; tensor var_693_0, tensor var_693_1 = split(axis = var_693_axis_0, split_sizes = var_693_split_sizes_0, x = out_29_cast_fp16)[name = string("op_693")]; fp16 const_35_promoted = const()[name = string("const_35_promoted"), val = fp16(-0x1p+0)]; tensor var_695 = mul(x = var_693_1, y = const_35_promoted)[name = string("op_695")]; bool var_697_interleave_0 = const()[name = string("op_697_interleave_0"), val = bool(false)]; tensor var_697 = concat(axis = var_22, interleave = var_697_interleave_0, values = (var_695, var_693_0))[name = string("op_697")]; tensor var_698 = mul(x = var_697, y = sin_1_quantized)[name = string("op_698")]; tensor hidden_states_25 = add(x = var_692, y = var_698)[name = string("hidden_states_25")]; tensor hidden_states_27_axes_0 = const()[name = string("hidden_states_27_axes_0"), val = tensor([2])]; tensor hidden_states_27 = expand_dims(axes = hidden_states_27_axes_0, x = hidden_states_25)[name = string("hidden_states_27")]; tensor var_701 = const()[name = string("op_701"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_29 = tile(reps = var_701, x = hidden_states_27)[name = string("hidden_states_29")]; tensor var_703 = const()[name = string("op_703"), val = tensor([1, 3, 256, 256])]; tensor k_17 = reshape(shape = var_703, x = hidden_states_29)[name = string("k_17")]; tensor hidden_states_33_axes_0 = const()[name = string("hidden_states_33_axes_0"), val = tensor([2])]; tensor hidden_states_31 = transpose(perm = var_650, x = var_649)[name = string("transpose_194")]; tensor hidden_states_33 = expand_dims(axes = hidden_states_33_axes_0, x = hidden_states_31)[name = string("hidden_states_33")]; tensor var_706 = const()[name = string("op_706"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_35 = tile(reps = var_706, x = hidden_states_33)[name = string("hidden_states_35")]; tensor var_708 = const()[name = string("op_708"), val = tensor([1, 3, 256, 256])]; tensor v_5 = reshape(shape = var_708, x = hidden_states_35)[name = string("v_5")]; bool var_713_transpose_x_1 = const()[name = string("op_713_transpose_x_1"), val = bool(false)]; bool var_713_transpose_y_1 = const()[name = string("op_713_transpose_y_1"), val = bool(true)]; tensor var_713_cast_fp16 = matmul(transpose_x = var_713_transpose_x_1, transpose_y = var_713_transpose_y_1, x = q_17, y = k_17)[name = string("op_713_cast_fp16")]; fp16 var_714_to_fp16 = const()[name = string("op_714_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_13_cast_fp16 = mul(x = var_713_cast_fp16, y = var_714_to_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor attn_weights_15_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; tensor var_718_cast_fp16 = softmax(axis = var_22, x = attn_weights_15_cast_fp16)[name = string("op_718_cast_fp16")]; bool var_722_transpose_x_0 = const()[name = string("op_722_transpose_x_0"), val = bool(false)]; bool var_722_transpose_y_0 = const()[name = string("op_722_transpose_y_0"), val = bool(false)]; tensor var_722_cast_fp16 = matmul(transpose_x = var_722_transpose_x_0, transpose_y = var_722_transpose_y_0, x = var_718_cast_fp16, y = v_5)[name = string("op_722_cast_fp16")]; tensor var_724 = const()[name = string("op_724"), val = tensor([0, 2, 1, 3])]; tensor var_727 = const()[name = string("op_727"), val = tensor([1, 256, 768])]; tensor var_725 = transpose(perm = var_724, x = var_722_cast_fp16)[name = string("transpose_193")]; tensor attn_out_15 = reshape(shape = var_727, x = var_725)[name = string("attn_out_15")]; tensor var_729 = const()[name = string("op_729"), val = tensor([0, 2, 1])]; tensor squeeze_2_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295569920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296159808))))[name = string("squeeze_2_quantized")]; string var_738_pad_type_0 = const()[name = string("op_738_pad_type_0"), val = string("valid")]; int32 var_738_groups_0 = const()[name = string("op_738_groups_0"), val = int32(1)]; tensor var_738_strides_0 = const()[name = string("op_738_strides_0"), val = tensor([1])]; tensor var_738_pad_0 = const()[name = string("op_738_pad_0"), val = tensor([0, 0])]; tensor var_738_dilations_0 = const()[name = string("op_738_dilations_0"), val = tensor([1])]; tensor var_730 = transpose(perm = var_729, x = attn_out_15)[name = string("transpose_192")]; tensor var_738 = conv(dilations = var_738_dilations_0, groups = var_738_groups_0, pad = var_738_pad_0, pad_type = var_738_pad_type_0, strides = var_738_strides_0, weight = squeeze_2_quantized, x = var_730)[name = string("op_738")]; tensor var_739 = const()[name = string("op_739"), val = tensor([0, 2, 1])]; fp16 const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_41 = transpose(perm = var_739, x = var_738)[name = string("transpose_191")]; tensor var_743_cast_fp16 = mul(x = x_41, y = const_36_promoted_to_fp16)[name = string("op_743_cast_fp16")]; bool input_51_interleave_0 = const()[name = string("input_51_interleave_0"), val = bool(false)]; tensor input_51_cast_fp16 = concat(axis = var_22, interleave = input_51_interleave_0, values = (x_41, var_743_cast_fp16))[name = string("input_51_cast_fp16")]; tensor normed_71_axes_0 = const()[name = string("normed_71_axes_0"), val = tensor([-1])]; tensor normed_71_cast_fp16 = layer_norm(axes = normed_71_axes_0, epsilon = var_8_to_fp16, x = input_51_cast_fp16)[name = string("normed_71_cast_fp16")]; tensor var_748_split_sizes_0 = const()[name = string("op_748_split_sizes_0"), val = tensor([768, 768])]; int32 var_748_axis_0 = const()[name = string("op_748_axis_0"), val = int32(-1)]; tensor var_748_cast_fp16_0, tensor var_748_cast_fp16_1 = split(axis = var_748_axis_0, split_sizes = var_748_split_sizes_0, x = normed_71_cast_fp16)[name = string("op_748_cast_fp16")]; tensor var_752_to_fp16 = const()[name = string("op_752_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296161408)))]; tensor out_31_cast_fp16 = mul(x = var_748_cast_fp16_0, y = var_752_to_fp16)[name = string("out_31_cast_fp16")]; tensor x_43_cast_fp16 = add(x = x_33_cast_fp16, y = out_31_cast_fp16)[name = string("x_43_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_759_cast_fp16 = mul(x = x_43_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_759_cast_fp16")]; bool input_53_interleave_0 = const()[name = string("input_53_interleave_0"), val = bool(false)]; tensor input_53_cast_fp16 = concat(axis = var_22, interleave = input_53_interleave_0, values = (x_43_cast_fp16, var_759_cast_fp16))[name = string("input_53_cast_fp16")]; tensor normed_75_axes_0 = const()[name = string("normed_75_axes_0"), val = tensor([-1])]; tensor normed_75_cast_fp16 = layer_norm(axes = normed_75_axes_0, epsilon = var_8_to_fp16, x = input_53_cast_fp16)[name = string("normed_75_cast_fp16")]; tensor var_764_split_sizes_0 = const()[name = string("op_764_split_sizes_0"), val = tensor([768, 768])]; int32 var_764_axis_0 = const()[name = string("op_764_axis_0"), val = int32(-1)]; tensor var_764_cast_fp16_0, tensor var_764_cast_fp16_1 = split(axis = var_764_axis_0, split_sizes = var_764_split_sizes_0, x = normed_75_cast_fp16)[name = string("op_764_cast_fp16")]; tensor var_768_to_fp16 = const()[name = string("op_768_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296163008)))]; tensor out_33_cast_fp16 = mul(x = var_764_cast_fp16_0, y = var_768_to_fp16)[name = string("out_33_cast_fp16")]; tensor var_775 = const()[name = string("op_775"), val = tensor([0, 2, 1])]; tensor input_55_axes_0 = const()[name = string("input_55_axes_0"), val = tensor([2])]; tensor var_776 = transpose(perm = var_775, x = out_33_cast_fp16)[name = string("transpose_190")]; tensor input_55 = expand_dims(axes = input_55_axes_0, x = var_776)[name = string("input_55")]; string gate_9_pad_type_0 = const()[name = string("gate_9_pad_type_0"), val = string("valid")]; tensor gate_9_strides_0 = const()[name = string("gate_9_strides_0"), val = tensor([1, 1])]; tensor gate_9_pad_0 = const()[name = string("gate_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_9_dilations_0 = const()[name = string("gate_9_dilations_0"), val = tensor([1, 1])]; int32 gate_9_groups_0 = const()[name = string("gate_9_groups_0"), val = int32(1)]; tensor gate_9 = conv(dilations = gate_9_dilations_0, groups = gate_9_groups_0, pad = gate_9_pad_0, pad_type = gate_9_pad_type_0, strides = gate_9_strides_0, weight = encoder_layers_2_mlp_gate_proj_weight_quantized, x = input_55)[name = string("gate_9")]; string up_5_pad_type_0 = const()[name = string("up_5_pad_type_0"), val = string("valid")]; tensor up_5_strides_0 = const()[name = string("up_5_strides_0"), val = tensor([1, 1])]; tensor up_5_pad_0 = const()[name = string("up_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_5_dilations_0 = const()[name = string("up_5_dilations_0"), val = tensor([1, 1])]; int32 up_5_groups_0 = const()[name = string("up_5_groups_0"), val = int32(1)]; tensor up_5 = conv(dilations = up_5_dilations_0, groups = up_5_groups_0, pad = up_5_pad_0, pad_type = up_5_pad_type_0, strides = up_5_strides_0, weight = encoder_layers_2_mlp_up_proj_weight_quantized, x = input_55)[name = string("up_5")]; string gate_11_mode_0 = const()[name = string("gate_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_11 = gelu(mode = gate_11_mode_0, x = gate_9)[name = string("gate_11")]; tensor input_57 = mul(x = gate_11, y = up_5)[name = string("input_57")]; string var_797_pad_type_0 = const()[name = string("op_797_pad_type_0"), val = string("valid")]; tensor var_797_strides_0 = const()[name = string("op_797_strides_0"), val = tensor([1, 1])]; tensor var_797_pad_0 = const()[name = string("op_797_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_797_dilations_0 = const()[name = string("op_797_dilations_0"), val = tensor([1, 1])]; int32 var_797_groups_0 = const()[name = string("op_797_groups_0"), val = int32(1)]; tensor var_797 = conv(dilations = var_797_dilations_0, groups = var_797_groups_0, pad = var_797_pad_0, pad_type = var_797_pad_type_0, strides = var_797_strides_0, weight = encoder_layers_2_mlp_down_proj_weight_quantized, x = input_57)[name = string("op_797")]; tensor var_798_axes_0 = const()[name = string("op_798_axes_0"), val = tensor([2])]; tensor var_798 = squeeze(axes = var_798_axes_0, x = var_797)[name = string("op_798")]; tensor var_799 = const()[name = string("op_799"), val = tensor([0, 2, 1])]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_47 = transpose(perm = var_799, x = var_798)[name = string("transpose_189")]; tensor var_803_cast_fp16 = mul(x = x_47, y = const_40_promoted_to_fp16)[name = string("op_803_cast_fp16")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59_cast_fp16 = concat(axis = var_22, interleave = input_59_interleave_0, values = (x_47, var_803_cast_fp16))[name = string("input_59_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_8_to_fp16, x = input_59_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor var_808_split_sizes_0 = const()[name = string("op_808_split_sizes_0"), val = tensor([768, 768])]; int32 var_808_axis_0 = const()[name = string("op_808_axis_0"), val = int32(-1)]; tensor var_808_cast_fp16_0, tensor var_808_cast_fp16_1 = split(axis = var_808_axis_0, split_sizes = var_808_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_808_cast_fp16")]; tensor var_812_to_fp16 = const()[name = string("op_812_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296164608)))]; tensor out_35_cast_fp16 = mul(x = var_808_cast_fp16_0, y = var_812_to_fp16)[name = string("out_35_cast_fp16")]; tensor x_49_cast_fp16 = add(x = x_43_cast_fp16, y = out_35_cast_fp16)[name = string("x_49_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_841_cast_fp16 = mul(x = x_49_cast_fp16, y = const_42_promoted_to_fp16)[name = string("op_841_cast_fp16")]; bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; tensor input_61_cast_fp16 = concat(axis = var_22, interleave = input_61_interleave_0, values = (x_49_cast_fp16, var_841_cast_fp16))[name = string("input_61_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_8_to_fp16, x = input_61_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_846_split_sizes_0 = const()[name = string("op_846_split_sizes_0"), val = tensor([768, 768])]; int32 var_846_axis_0 = const()[name = string("op_846_axis_0"), val = int32(-1)]; tensor var_846_cast_fp16_0, tensor var_846_cast_fp16_1 = split(axis = var_846_axis_0, split_sizes = var_846_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_846_cast_fp16")]; tensor var_850_to_fp16 = const()[name = string("op_850_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296166208)))]; tensor out_37_cast_fp16 = mul(x = var_846_cast_fp16_0, y = var_850_to_fp16)[name = string("out_37_cast_fp16")]; tensor var_856 = const()[name = string("op_856"), val = tensor([0, 2, 1])]; tensor var_858_axes_0 = const()[name = string("op_858_axes_0"), val = tensor([2])]; tensor var_857_cast_fp16 = transpose(perm = var_856, x = out_37_cast_fp16)[name = string("transpose_188")]; tensor var_858_cast_fp16 = expand_dims(axes = var_858_axes_0, x = var_857_cast_fp16)[name = string("op_858_cast_fp16")]; string var_865_pad_type_0 = const()[name = string("op_865_pad_type_0"), val = string("valid")]; tensor var_865_strides_0 = const()[name = string("op_865_strides_0"), val = tensor([1, 1])]; tensor var_865_pad_0 = const()[name = string("op_865_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_865_dilations_0 = const()[name = string("op_865_dilations_0"), val = tensor([1, 1])]; int32 var_865_groups_0 = const()[name = string("op_865_groups_0"), val = int32(1)]; tensor var_865 = conv(dilations = var_865_dilations_0, groups = var_865_groups_0, pad = var_865_pad_0, pad_type = var_865_pad_type_0, strides = var_865_strides_0, weight = encoder_layers_3_self_attn_q_proj_weight_quantized, x = var_858_cast_fp16)[name = string("op_865")]; tensor var_866 = const()[name = string("op_866"), val = tensor([1, 3, 256, 256])]; tensor var_867 = reshape(shape = var_866, x = var_865)[name = string("op_867")]; tensor var_868 = const()[name = string("op_868"), val = tensor([0, 1, 3, 2])]; string var_875_pad_type_0 = const()[name = string("op_875_pad_type_0"), val = string("valid")]; tensor var_875_strides_0 = const()[name = string("op_875_strides_0"), val = tensor([1, 1])]; tensor var_875_pad_0 = const()[name = string("op_875_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_875_dilations_0 = const()[name = string("op_875_dilations_0"), val = tensor([1, 1])]; int32 var_875_groups_0 = const()[name = string("op_875_groups_0"), val = int32(1)]; tensor var_875 = conv(dilations = var_875_dilations_0, groups = var_875_groups_0, pad = var_875_pad_0, pad_type = var_875_pad_type_0, strides = var_875_strides_0, weight = encoder_layers_3_self_attn_k_proj_weight_quantized, x = var_858_cast_fp16)[name = string("op_875")]; tensor var_876 = const()[name = string("op_876"), val = tensor([1, 1, 256, 256])]; tensor var_877 = reshape(shape = var_876, x = var_875)[name = string("op_877")]; tensor var_878 = const()[name = string("op_878"), val = tensor([0, 1, 3, 2])]; string var_885_pad_type_0 = const()[name = string("op_885_pad_type_0"), val = string("valid")]; tensor var_885_strides_0 = const()[name = string("op_885_strides_0"), val = tensor([1, 1])]; tensor var_885_pad_0 = const()[name = string("op_885_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_885_dilations_0 = const()[name = string("op_885_dilations_0"), val = tensor([1, 1])]; int32 var_885_groups_0 = const()[name = string("op_885_groups_0"), val = int32(1)]; tensor var_885 = conv(dilations = var_885_dilations_0, groups = var_885_groups_0, pad = var_885_pad_0, pad_type = var_885_pad_type_0, strides = var_885_strides_0, weight = encoder_layers_3_self_attn_v_proj_weight_quantized, x = var_858_cast_fp16)[name = string("op_885")]; tensor var_886 = const()[name = string("op_886"), val = tensor([1, 1, 256, 256])]; tensor var_887 = reshape(shape = var_886, x = var_885)[name = string("op_887")]; tensor var_888 = const()[name = string("op_888"), val = tensor([0, 1, 3, 2])]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_19 = transpose(perm = var_868, x = var_867)[name = string("transpose_187")]; tensor var_894_cast_fp16 = mul(x = q_19, y = const_44_promoted_to_fp16)[name = string("op_894_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_22, interleave = input_65_interleave_0, values = (q_19, var_894_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_91_axes_0 = const()[name = string("normed_91_axes_0"), val = tensor([-1])]; tensor normed_91_cast_fp16 = layer_norm(axes = normed_91_axes_0, epsilon = var_8_to_fp16, x = input_65_cast_fp16)[name = string("normed_91_cast_fp16")]; tensor var_899_split_sizes_0 = const()[name = string("op_899_split_sizes_0"), val = tensor([256, 256])]; int32 var_899_axis_0 = const()[name = string("op_899_axis_0"), val = int32(-1)]; tensor var_899_cast_fp16_0, tensor var_899_cast_fp16_1 = split(axis = var_899_axis_0, split_sizes = var_899_split_sizes_0, x = normed_91_cast_fp16)[name = string("op_899_cast_fp16")]; tensor var_903_to_fp16 = const()[name = string("op_903_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296167808)))]; tensor out_39_cast_fp16 = mul(x = var_899_cast_fp16_0, y = var_903_to_fp16)[name = string("out_39_cast_fp16")]; fp16 const_46_promoted_to_fp16 = const()[name = string("const_46_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_19 = transpose(perm = var_878, x = var_877)[name = string("transpose_186")]; tensor var_910_cast_fp16 = mul(x = k_19, y = const_46_promoted_to_fp16)[name = string("op_910_cast_fp16")]; bool input_67_interleave_0 = const()[name = string("input_67_interleave_0"), val = bool(false)]; tensor input_67_cast_fp16 = concat(axis = var_22, interleave = input_67_interleave_0, values = (k_19, var_910_cast_fp16))[name = string("input_67_cast_fp16")]; tensor normed_95_axes_0 = const()[name = string("normed_95_axes_0"), val = tensor([-1])]; tensor normed_95_cast_fp16 = layer_norm(axes = normed_95_axes_0, epsilon = var_8_to_fp16, x = input_67_cast_fp16)[name = string("normed_95_cast_fp16")]; tensor var_915_split_sizes_0 = const()[name = string("op_915_split_sizes_0"), val = tensor([256, 256])]; int32 var_915_axis_0 = const()[name = string("op_915_axis_0"), val = int32(-1)]; tensor var_915_cast_fp16_0, tensor var_915_cast_fp16_1 = split(axis = var_915_axis_0, split_sizes = var_915_split_sizes_0, x = normed_95_cast_fp16)[name = string("op_915_cast_fp16")]; tensor var_919_to_fp16 = const()[name = string("op_919_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296168384)))]; tensor out_41_cast_fp16 = mul(x = var_915_cast_fp16_0, y = var_919_to_fp16)[name = string("out_41_cast_fp16")]; tensor var_922 = mul(x = out_39_cast_fp16, y = cos_1_quantized)[name = string("op_922")]; tensor var_923_split_sizes_0 = const()[name = string("op_923_split_sizes_0"), val = tensor([128, 128])]; int32 var_923_axis_0 = const()[name = string("op_923_axis_0"), val = int32(-1)]; tensor var_923_0, tensor var_923_1 = split(axis = var_923_axis_0, split_sizes = var_923_split_sizes_0, x = out_39_cast_fp16)[name = string("op_923")]; fp16 const_48_promoted = const()[name = string("const_48_promoted"), val = fp16(-0x1p+0)]; tensor var_925 = mul(x = var_923_1, y = const_48_promoted)[name = string("op_925")]; bool var_927_interleave_0 = const()[name = string("op_927_interleave_0"), val = bool(false)]; tensor var_927 = concat(axis = var_22, interleave = var_927_interleave_0, values = (var_925, var_923_0))[name = string("op_927")]; tensor var_928 = mul(x = var_927, y = sin_1_quantized)[name = string("op_928")]; tensor q_23 = add(x = var_922, y = var_928)[name = string("q_23")]; tensor var_930 = mul(x = out_41_cast_fp16, y = cos_1_quantized)[name = string("op_930")]; tensor var_931_split_sizes_0 = const()[name = string("op_931_split_sizes_0"), val = tensor([128, 128])]; int32 var_931_axis_0 = const()[name = string("op_931_axis_0"), val = int32(-1)]; tensor var_931_0, tensor var_931_1 = split(axis = var_931_axis_0, split_sizes = var_931_split_sizes_0, x = out_41_cast_fp16)[name = string("op_931")]; fp16 const_49_promoted = const()[name = string("const_49_promoted"), val = fp16(-0x1p+0)]; tensor var_933 = mul(x = var_931_1, y = const_49_promoted)[name = string("op_933")]; bool var_935_interleave_0 = const()[name = string("op_935_interleave_0"), val = bool(false)]; tensor var_935 = concat(axis = var_22, interleave = var_935_interleave_0, values = (var_933, var_931_0))[name = string("op_935")]; tensor var_936 = mul(x = var_935, y = sin_1_quantized)[name = string("op_936")]; tensor hidden_states_37 = add(x = var_930, y = var_936)[name = string("hidden_states_37")]; tensor hidden_states_39_axes_0 = const()[name = string("hidden_states_39_axes_0"), val = tensor([2])]; tensor hidden_states_39 = expand_dims(axes = hidden_states_39_axes_0, x = hidden_states_37)[name = string("hidden_states_39")]; tensor var_939 = const()[name = string("op_939"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_41 = tile(reps = var_939, x = hidden_states_39)[name = string("hidden_states_41")]; tensor var_941 = const()[name = string("op_941"), val = tensor([1, 3, 256, 256])]; tensor k_23 = reshape(shape = var_941, x = hidden_states_41)[name = string("k_23")]; tensor hidden_states_45_axes_0 = const()[name = string("hidden_states_45_axes_0"), val = tensor([2])]; tensor hidden_states_43 = transpose(perm = var_888, x = var_887)[name = string("transpose_185")]; tensor hidden_states_45 = expand_dims(axes = hidden_states_45_axes_0, x = hidden_states_43)[name = string("hidden_states_45")]; tensor var_944 = const()[name = string("op_944"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_47 = tile(reps = var_944, x = hidden_states_45)[name = string("hidden_states_47")]; tensor var_946 = const()[name = string("op_946"), val = tensor([1, 3, 256, 256])]; tensor v_7 = reshape(shape = var_946, x = hidden_states_47)[name = string("v_7")]; bool var_951_transpose_x_1 = const()[name = string("op_951_transpose_x_1"), val = bool(false)]; bool var_951_transpose_y_1 = const()[name = string("op_951_transpose_y_1"), val = bool(true)]; tensor var_951_cast_fp16 = matmul(transpose_x = var_951_transpose_x_1, transpose_y = var_951_transpose_y_1, x = q_23, y = k_23)[name = string("op_951_cast_fp16")]; fp16 var_952_to_fp16 = const()[name = string("op_952_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_19_cast_fp16 = mul(x = var_951_cast_fp16, y = var_952_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor var_956_cast_fp16 = softmax(axis = var_22, x = attn_weights_21_cast_fp16)[name = string("op_956_cast_fp16")]; bool var_960_transpose_x_0 = const()[name = string("op_960_transpose_x_0"), val = bool(false)]; bool var_960_transpose_y_0 = const()[name = string("op_960_transpose_y_0"), val = bool(false)]; tensor var_960_cast_fp16 = matmul(transpose_x = var_960_transpose_x_0, transpose_y = var_960_transpose_y_0, x = var_956_cast_fp16, y = v_7)[name = string("op_960_cast_fp16")]; tensor var_962 = const()[name = string("op_962"), val = tensor([0, 2, 1, 3])]; tensor var_965 = const()[name = string("op_965"), val = tensor([1, 256, 768])]; tensor var_963 = transpose(perm = var_962, x = var_960_cast_fp16)[name = string("transpose_184")]; tensor attn_out_21 = reshape(shape = var_965, x = var_963)[name = string("attn_out_21")]; tensor var_967 = const()[name = string("op_967"), val = tensor([0, 2, 1])]; tensor squeeze_3_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296168960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296758848))))[name = string("squeeze_3_quantized")]; string var_976_pad_type_0 = const()[name = string("op_976_pad_type_0"), val = string("valid")]; int32 var_976_groups_0 = const()[name = string("op_976_groups_0"), val = int32(1)]; tensor var_976_strides_0 = const()[name = string("op_976_strides_0"), val = tensor([1])]; tensor var_976_pad_0 = const()[name = string("op_976_pad_0"), val = tensor([0, 0])]; tensor var_976_dilations_0 = const()[name = string("op_976_dilations_0"), val = tensor([1])]; tensor var_968 = transpose(perm = var_967, x = attn_out_21)[name = string("transpose_183")]; tensor var_976 = conv(dilations = var_976_dilations_0, groups = var_976_groups_0, pad = var_976_pad_0, pad_type = var_976_pad_type_0, strides = var_976_strides_0, weight = squeeze_3_quantized, x = var_968)[name = string("op_976")]; tensor var_977 = const()[name = string("op_977"), val = tensor([0, 2, 1])]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_57 = transpose(perm = var_977, x = var_976)[name = string("transpose_182")]; tensor var_981_cast_fp16 = mul(x = x_57, y = const_50_promoted_to_fp16)[name = string("op_981_cast_fp16")]; bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; tensor input_71_cast_fp16 = concat(axis = var_22, interleave = input_71_interleave_0, values = (x_57, var_981_cast_fp16))[name = string("input_71_cast_fp16")]; tensor normed_99_axes_0 = const()[name = string("normed_99_axes_0"), val = tensor([-1])]; tensor normed_99_cast_fp16 = layer_norm(axes = normed_99_axes_0, epsilon = var_8_to_fp16, x = input_71_cast_fp16)[name = string("normed_99_cast_fp16")]; tensor var_986_split_sizes_0 = const()[name = string("op_986_split_sizes_0"), val = tensor([768, 768])]; int32 var_986_axis_0 = const()[name = string("op_986_axis_0"), val = int32(-1)]; tensor var_986_cast_fp16_0, tensor var_986_cast_fp16_1 = split(axis = var_986_axis_0, split_sizes = var_986_split_sizes_0, x = normed_99_cast_fp16)[name = string("op_986_cast_fp16")]; tensor var_990_to_fp16 = const()[name = string("op_990_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296760448)))]; tensor out_43_cast_fp16 = mul(x = var_986_cast_fp16_0, y = var_990_to_fp16)[name = string("out_43_cast_fp16")]; tensor x_59_cast_fp16 = add(x = x_49_cast_fp16, y = out_43_cast_fp16)[name = string("x_59_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_997_cast_fp16 = mul(x = x_59_cast_fp16, y = const_52_promoted_to_fp16)[name = string("op_997_cast_fp16")]; bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; tensor input_73_cast_fp16 = concat(axis = var_22, interleave = input_73_interleave_0, values = (x_59_cast_fp16, var_997_cast_fp16))[name = string("input_73_cast_fp16")]; tensor normed_103_axes_0 = const()[name = string("normed_103_axes_0"), val = tensor([-1])]; tensor normed_103_cast_fp16 = layer_norm(axes = normed_103_axes_0, epsilon = var_8_to_fp16, x = input_73_cast_fp16)[name = string("normed_103_cast_fp16")]; tensor var_1002_split_sizes_0 = const()[name = string("op_1002_split_sizes_0"), val = tensor([768, 768])]; int32 var_1002_axis_0 = const()[name = string("op_1002_axis_0"), val = int32(-1)]; tensor var_1002_cast_fp16_0, tensor var_1002_cast_fp16_1 = split(axis = var_1002_axis_0, split_sizes = var_1002_split_sizes_0, x = normed_103_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor var_1006_to_fp16 = const()[name = string("op_1006_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296762048)))]; tensor out_45_cast_fp16 = mul(x = var_1002_cast_fp16_0, y = var_1006_to_fp16)[name = string("out_45_cast_fp16")]; tensor var_1013 = const()[name = string("op_1013"), val = tensor([0, 2, 1])]; tensor input_75_axes_0 = const()[name = string("input_75_axes_0"), val = tensor([2])]; tensor var_1014 = transpose(perm = var_1013, x = out_45_cast_fp16)[name = string("transpose_181")]; tensor input_75 = expand_dims(axes = input_75_axes_0, x = var_1014)[name = string("input_75")]; string gate_13_pad_type_0 = const()[name = string("gate_13_pad_type_0"), val = string("valid")]; tensor gate_13_strides_0 = const()[name = string("gate_13_strides_0"), val = tensor([1, 1])]; tensor gate_13_pad_0 = const()[name = string("gate_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_13_dilations_0 = const()[name = string("gate_13_dilations_0"), val = tensor([1, 1])]; int32 gate_13_groups_0 = const()[name = string("gate_13_groups_0"), val = int32(1)]; tensor gate_13 = conv(dilations = gate_13_dilations_0, groups = gate_13_groups_0, pad = gate_13_pad_0, pad_type = gate_13_pad_type_0, strides = gate_13_strides_0, weight = encoder_layers_3_mlp_gate_proj_weight_quantized, x = input_75)[name = string("gate_13")]; string up_7_pad_type_0 = const()[name = string("up_7_pad_type_0"), val = string("valid")]; tensor up_7_strides_0 = const()[name = string("up_7_strides_0"), val = tensor([1, 1])]; tensor up_7_pad_0 = const()[name = string("up_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_7_dilations_0 = const()[name = string("up_7_dilations_0"), val = tensor([1, 1])]; int32 up_7_groups_0 = const()[name = string("up_7_groups_0"), val = int32(1)]; tensor up_7 = conv(dilations = up_7_dilations_0, groups = up_7_groups_0, pad = up_7_pad_0, pad_type = up_7_pad_type_0, strides = up_7_strides_0, weight = encoder_layers_3_mlp_up_proj_weight_quantized, x = input_75)[name = string("up_7")]; string gate_15_mode_0 = const()[name = string("gate_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_15 = gelu(mode = gate_15_mode_0, x = gate_13)[name = string("gate_15")]; tensor input_77 = mul(x = gate_15, y = up_7)[name = string("input_77")]; string var_1035_pad_type_0 = const()[name = string("op_1035_pad_type_0"), val = string("valid")]; tensor var_1035_strides_0 = const()[name = string("op_1035_strides_0"), val = tensor([1, 1])]; tensor var_1035_pad_0 = const()[name = string("op_1035_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1035_dilations_0 = const()[name = string("op_1035_dilations_0"), val = tensor([1, 1])]; int32 var_1035_groups_0 = const()[name = string("op_1035_groups_0"), val = int32(1)]; tensor var_1035 = conv(dilations = var_1035_dilations_0, groups = var_1035_groups_0, pad = var_1035_pad_0, pad_type = var_1035_pad_type_0, strides = var_1035_strides_0, weight = encoder_layers_3_mlp_down_proj_weight_quantized, x = input_77)[name = string("op_1035")]; tensor var_1036_axes_0 = const()[name = string("op_1036_axes_0"), val = tensor([2])]; tensor var_1036 = squeeze(axes = var_1036_axes_0, x = var_1035)[name = string("op_1036")]; tensor var_1037 = const()[name = string("op_1037"), val = tensor([0, 2, 1])]; fp16 const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_63 = transpose(perm = var_1037, x = var_1036)[name = string("transpose_180")]; tensor var_1041_cast_fp16 = mul(x = x_63, y = const_54_promoted_to_fp16)[name = string("op_1041_cast_fp16")]; bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; tensor input_79_cast_fp16 = concat(axis = var_22, interleave = input_79_interleave_0, values = (x_63, var_1041_cast_fp16))[name = string("input_79_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_8_to_fp16, x = input_79_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor var_1046_split_sizes_0 = const()[name = string("op_1046_split_sizes_0"), val = tensor([768, 768])]; int32 var_1046_axis_0 = const()[name = string("op_1046_axis_0"), val = int32(-1)]; tensor var_1046_cast_fp16_0, tensor var_1046_cast_fp16_1 = split(axis = var_1046_axis_0, split_sizes = var_1046_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_1046_cast_fp16")]; tensor var_1050_to_fp16 = const()[name = string("op_1050_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296763648)))]; tensor out_47_cast_fp16 = mul(x = var_1046_cast_fp16_0, y = var_1050_to_fp16)[name = string("out_47_cast_fp16")]; tensor x_65_cast_fp16 = add(x = x_59_cast_fp16, y = out_47_cast_fp16)[name = string("x_65_cast_fp16")]; fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1079_cast_fp16 = mul(x = x_65_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_1079_cast_fp16")]; bool input_81_interleave_0 = const()[name = string("input_81_interleave_0"), val = bool(false)]; tensor input_81_cast_fp16 = concat(axis = var_22, interleave = input_81_interleave_0, values = (x_65_cast_fp16, var_1079_cast_fp16))[name = string("input_81_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_8_to_fp16, x = input_81_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_1084_split_sizes_0 = const()[name = string("op_1084_split_sizes_0"), val = tensor([768, 768])]; int32 var_1084_axis_0 = const()[name = string("op_1084_axis_0"), val = int32(-1)]; tensor var_1084_cast_fp16_0, tensor var_1084_cast_fp16_1 = split(axis = var_1084_axis_0, split_sizes = var_1084_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_1084_cast_fp16")]; tensor var_1088_to_fp16 = const()[name = string("op_1088_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296765248)))]; tensor out_49_cast_fp16 = mul(x = var_1084_cast_fp16_0, y = var_1088_to_fp16)[name = string("out_49_cast_fp16")]; tensor var_1094 = const()[name = string("op_1094"), val = tensor([0, 2, 1])]; tensor var_1096_axes_0 = const()[name = string("op_1096_axes_0"), val = tensor([2])]; tensor var_1095_cast_fp16 = transpose(perm = var_1094, x = out_49_cast_fp16)[name = string("transpose_179")]; tensor var_1096_cast_fp16 = expand_dims(axes = var_1096_axes_0, x = var_1095_cast_fp16)[name = string("op_1096_cast_fp16")]; string var_1103_pad_type_0 = const()[name = string("op_1103_pad_type_0"), val = string("valid")]; tensor var_1103_strides_0 = const()[name = string("op_1103_strides_0"), val = tensor([1, 1])]; tensor var_1103_pad_0 = const()[name = string("op_1103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1103_dilations_0 = const()[name = string("op_1103_dilations_0"), val = tensor([1, 1])]; int32 var_1103_groups_0 = const()[name = string("op_1103_groups_0"), val = int32(1)]; tensor var_1103 = conv(dilations = var_1103_dilations_0, groups = var_1103_groups_0, pad = var_1103_pad_0, pad_type = var_1103_pad_type_0, strides = var_1103_strides_0, weight = encoder_layers_4_self_attn_q_proj_weight_quantized, x = var_1096_cast_fp16)[name = string("op_1103")]; tensor var_1104 = const()[name = string("op_1104"), val = tensor([1, 3, 256, 256])]; tensor var_1105 = reshape(shape = var_1104, x = var_1103)[name = string("op_1105")]; tensor var_1106 = const()[name = string("op_1106"), val = tensor([0, 1, 3, 2])]; string var_1113_pad_type_0 = const()[name = string("op_1113_pad_type_0"), val = string("valid")]; tensor var_1113_strides_0 = const()[name = string("op_1113_strides_0"), val = tensor([1, 1])]; tensor var_1113_pad_0 = const()[name = string("op_1113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1113_dilations_0 = const()[name = string("op_1113_dilations_0"), val = tensor([1, 1])]; int32 var_1113_groups_0 = const()[name = string("op_1113_groups_0"), val = int32(1)]; tensor var_1113 = conv(dilations = var_1113_dilations_0, groups = var_1113_groups_0, pad = var_1113_pad_0, pad_type = var_1113_pad_type_0, strides = var_1113_strides_0, weight = encoder_layers_4_self_attn_k_proj_weight_quantized, x = var_1096_cast_fp16)[name = string("op_1113")]; tensor var_1114 = const()[name = string("op_1114"), val = tensor([1, 1, 256, 256])]; tensor var_1115 = reshape(shape = var_1114, x = var_1113)[name = string("op_1115")]; tensor var_1116 = const()[name = string("op_1116"), val = tensor([0, 1, 3, 2])]; string var_1123_pad_type_0 = const()[name = string("op_1123_pad_type_0"), val = string("valid")]; tensor var_1123_strides_0 = const()[name = string("op_1123_strides_0"), val = tensor([1, 1])]; tensor var_1123_pad_0 = const()[name = string("op_1123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1123_dilations_0 = const()[name = string("op_1123_dilations_0"), val = tensor([1, 1])]; int32 var_1123_groups_0 = const()[name = string("op_1123_groups_0"), val = int32(1)]; tensor var_1123 = conv(dilations = var_1123_dilations_0, groups = var_1123_groups_0, pad = var_1123_pad_0, pad_type = var_1123_pad_type_0, strides = var_1123_strides_0, weight = encoder_layers_4_self_attn_v_proj_weight_quantized, x = var_1096_cast_fp16)[name = string("op_1123")]; tensor var_1124 = const()[name = string("op_1124"), val = tensor([1, 1, 256, 256])]; tensor var_1125 = reshape(shape = var_1124, x = var_1123)[name = string("op_1125")]; tensor var_1126 = const()[name = string("op_1126"), val = tensor([0, 1, 3, 2])]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_25 = transpose(perm = var_1106, x = var_1105)[name = string("transpose_178")]; tensor var_1132_cast_fp16 = mul(x = q_25, y = const_58_promoted_to_fp16)[name = string("op_1132_cast_fp16")]; bool input_85_interleave_0 = const()[name = string("input_85_interleave_0"), val = bool(false)]; tensor input_85_cast_fp16 = concat(axis = var_22, interleave = input_85_interleave_0, values = (q_25, var_1132_cast_fp16))[name = string("input_85_cast_fp16")]; tensor normed_119_axes_0 = const()[name = string("normed_119_axes_0"), val = tensor([-1])]; tensor normed_119_cast_fp16 = layer_norm(axes = normed_119_axes_0, epsilon = var_8_to_fp16, x = input_85_cast_fp16)[name = string("normed_119_cast_fp16")]; tensor var_1137_split_sizes_0 = const()[name = string("op_1137_split_sizes_0"), val = tensor([256, 256])]; int32 var_1137_axis_0 = const()[name = string("op_1137_axis_0"), val = int32(-1)]; tensor var_1137_cast_fp16_0, tensor var_1137_cast_fp16_1 = split(axis = var_1137_axis_0, split_sizes = var_1137_split_sizes_0, x = normed_119_cast_fp16)[name = string("op_1137_cast_fp16")]; tensor var_1141_to_fp16 = const()[name = string("op_1141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296766848)))]; tensor out_51_cast_fp16 = mul(x = var_1137_cast_fp16_0, y = var_1141_to_fp16)[name = string("out_51_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_25 = transpose(perm = var_1116, x = var_1115)[name = string("transpose_177")]; tensor var_1148_cast_fp16 = mul(x = k_25, y = const_60_promoted_to_fp16)[name = string("op_1148_cast_fp16")]; bool input_87_interleave_0 = const()[name = string("input_87_interleave_0"), val = bool(false)]; tensor input_87_cast_fp16 = concat(axis = var_22, interleave = input_87_interleave_0, values = (k_25, var_1148_cast_fp16))[name = string("input_87_cast_fp16")]; tensor normed_123_axes_0 = const()[name = string("normed_123_axes_0"), val = tensor([-1])]; tensor normed_123_cast_fp16 = layer_norm(axes = normed_123_axes_0, epsilon = var_8_to_fp16, x = input_87_cast_fp16)[name = string("normed_123_cast_fp16")]; tensor var_1153_split_sizes_0 = const()[name = string("op_1153_split_sizes_0"), val = tensor([256, 256])]; int32 var_1153_axis_0 = const()[name = string("op_1153_axis_0"), val = int32(-1)]; tensor var_1153_cast_fp16_0, tensor var_1153_cast_fp16_1 = split(axis = var_1153_axis_0, split_sizes = var_1153_split_sizes_0, x = normed_123_cast_fp16)[name = string("op_1153_cast_fp16")]; tensor var_1157_to_fp16 = const()[name = string("op_1157_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296767424)))]; tensor out_53_cast_fp16 = mul(x = var_1153_cast_fp16_0, y = var_1157_to_fp16)[name = string("out_53_cast_fp16")]; tensor var_1160 = mul(x = out_51_cast_fp16, y = cos_1_quantized)[name = string("op_1160")]; tensor var_1161_split_sizes_0 = const()[name = string("op_1161_split_sizes_0"), val = tensor([128, 128])]; int32 var_1161_axis_0 = const()[name = string("op_1161_axis_0"), val = int32(-1)]; tensor var_1161_0, tensor var_1161_1 = split(axis = var_1161_axis_0, split_sizes = var_1161_split_sizes_0, x = out_51_cast_fp16)[name = string("op_1161")]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_1163 = mul(x = var_1161_1, y = const_62_promoted)[name = string("op_1163")]; bool var_1165_interleave_0 = const()[name = string("op_1165_interleave_0"), val = bool(false)]; tensor var_1165 = concat(axis = var_22, interleave = var_1165_interleave_0, values = (var_1163, var_1161_0))[name = string("op_1165")]; tensor var_1166 = mul(x = var_1165, y = sin_1_quantized)[name = string("op_1166")]; tensor q_29 = add(x = var_1160, y = var_1166)[name = string("q_29")]; tensor var_1168 = mul(x = out_53_cast_fp16, y = cos_1_quantized)[name = string("op_1168")]; tensor var_1169_split_sizes_0 = const()[name = string("op_1169_split_sizes_0"), val = tensor([128, 128])]; int32 var_1169_axis_0 = const()[name = string("op_1169_axis_0"), val = int32(-1)]; tensor var_1169_0, tensor var_1169_1 = split(axis = var_1169_axis_0, split_sizes = var_1169_split_sizes_0, x = out_53_cast_fp16)[name = string("op_1169")]; fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; tensor var_1171 = mul(x = var_1169_1, y = const_63_promoted)[name = string("op_1171")]; bool var_1173_interleave_0 = const()[name = string("op_1173_interleave_0"), val = bool(false)]; tensor var_1173 = concat(axis = var_22, interleave = var_1173_interleave_0, values = (var_1171, var_1169_0))[name = string("op_1173")]; tensor var_1174 = mul(x = var_1173, y = sin_1_quantized)[name = string("op_1174")]; tensor hidden_states_49 = add(x = var_1168, y = var_1174)[name = string("hidden_states_49")]; tensor hidden_states_51_axes_0 = const()[name = string("hidden_states_51_axes_0"), val = tensor([2])]; tensor hidden_states_51 = expand_dims(axes = hidden_states_51_axes_0, x = hidden_states_49)[name = string("hidden_states_51")]; tensor var_1177 = const()[name = string("op_1177"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_53 = tile(reps = var_1177, x = hidden_states_51)[name = string("hidden_states_53")]; tensor var_1179 = const()[name = string("op_1179"), val = tensor([1, 3, 256, 256])]; tensor k_29 = reshape(shape = var_1179, x = hidden_states_53)[name = string("k_29")]; tensor hidden_states_57_axes_0 = const()[name = string("hidden_states_57_axes_0"), val = tensor([2])]; tensor hidden_states_55 = transpose(perm = var_1126, x = var_1125)[name = string("transpose_176")]; tensor hidden_states_57 = expand_dims(axes = hidden_states_57_axes_0, x = hidden_states_55)[name = string("hidden_states_57")]; tensor var_1182 = const()[name = string("op_1182"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_59 = tile(reps = var_1182, x = hidden_states_57)[name = string("hidden_states_59")]; tensor var_1184 = const()[name = string("op_1184"), val = tensor([1, 3, 256, 256])]; tensor v_9 = reshape(shape = var_1184, x = hidden_states_59)[name = string("v_9")]; bool var_1189_transpose_x_1 = const()[name = string("op_1189_transpose_x_1"), val = bool(false)]; bool var_1189_transpose_y_1 = const()[name = string("op_1189_transpose_y_1"), val = bool(true)]; tensor var_1189_cast_fp16 = matmul(transpose_x = var_1189_transpose_x_1, transpose_y = var_1189_transpose_y_1, x = q_29, y = k_29)[name = string("op_1189_cast_fp16")]; fp16 var_1190_to_fp16 = const()[name = string("op_1190_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_25_cast_fp16 = mul(x = var_1189_cast_fp16, y = var_1190_to_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor attn_weights_27_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor var_1194_cast_fp16 = softmax(axis = var_22, x = attn_weights_27_cast_fp16)[name = string("op_1194_cast_fp16")]; bool var_1198_transpose_x_0 = const()[name = string("op_1198_transpose_x_0"), val = bool(false)]; bool var_1198_transpose_y_0 = const()[name = string("op_1198_transpose_y_0"), val = bool(false)]; tensor var_1198_cast_fp16 = matmul(transpose_x = var_1198_transpose_x_0, transpose_y = var_1198_transpose_y_0, x = var_1194_cast_fp16, y = v_9)[name = string("op_1198_cast_fp16")]; tensor var_1200 = const()[name = string("op_1200"), val = tensor([0, 2, 1, 3])]; tensor var_1203 = const()[name = string("op_1203"), val = tensor([1, 256, 768])]; tensor var_1201 = transpose(perm = var_1200, x = var_1198_cast_fp16)[name = string("transpose_175")]; tensor attn_out_27 = reshape(shape = var_1203, x = var_1201)[name = string("attn_out_27")]; tensor var_1205 = const()[name = string("op_1205"), val = tensor([0, 2, 1])]; tensor squeeze_4_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296768000))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297357888))))[name = string("squeeze_4_quantized")]; string var_1214_pad_type_0 = const()[name = string("op_1214_pad_type_0"), val = string("valid")]; int32 var_1214_groups_0 = const()[name = string("op_1214_groups_0"), val = int32(1)]; tensor var_1214_strides_0 = const()[name = string("op_1214_strides_0"), val = tensor([1])]; tensor var_1214_pad_0 = const()[name = string("op_1214_pad_0"), val = tensor([0, 0])]; tensor var_1214_dilations_0 = const()[name = string("op_1214_dilations_0"), val = tensor([1])]; tensor var_1206 = transpose(perm = var_1205, x = attn_out_27)[name = string("transpose_174")]; tensor var_1214 = conv(dilations = var_1214_dilations_0, groups = var_1214_groups_0, pad = var_1214_pad_0, pad_type = var_1214_pad_type_0, strides = var_1214_strides_0, weight = squeeze_4_quantized, x = var_1206)[name = string("op_1214")]; tensor var_1215 = const()[name = string("op_1215"), val = tensor([0, 2, 1])]; fp16 const_64_promoted_to_fp16 = const()[name = string("const_64_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_73 = transpose(perm = var_1215, x = var_1214)[name = string("transpose_173")]; tensor var_1219_cast_fp16 = mul(x = x_73, y = const_64_promoted_to_fp16)[name = string("op_1219_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_22, interleave = input_91_interleave_0, values = (x_73, var_1219_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_127_axes_0 = const()[name = string("normed_127_axes_0"), val = tensor([-1])]; tensor normed_127_cast_fp16 = layer_norm(axes = normed_127_axes_0, epsilon = var_8_to_fp16, x = input_91_cast_fp16)[name = string("normed_127_cast_fp16")]; tensor var_1224_split_sizes_0 = const()[name = string("op_1224_split_sizes_0"), val = tensor([768, 768])]; int32 var_1224_axis_0 = const()[name = string("op_1224_axis_0"), val = int32(-1)]; tensor var_1224_cast_fp16_0, tensor var_1224_cast_fp16_1 = split(axis = var_1224_axis_0, split_sizes = var_1224_split_sizes_0, x = normed_127_cast_fp16)[name = string("op_1224_cast_fp16")]; tensor var_1228_to_fp16 = const()[name = string("op_1228_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297359488)))]; tensor out_55_cast_fp16 = mul(x = var_1224_cast_fp16_0, y = var_1228_to_fp16)[name = string("out_55_cast_fp16")]; tensor x_75_cast_fp16 = add(x = x_65_cast_fp16, y = out_55_cast_fp16)[name = string("x_75_cast_fp16")]; fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1235_cast_fp16 = mul(x = x_75_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_1235_cast_fp16")]; bool input_93_interleave_0 = const()[name = string("input_93_interleave_0"), val = bool(false)]; tensor input_93_cast_fp16 = concat(axis = var_22, interleave = input_93_interleave_0, values = (x_75_cast_fp16, var_1235_cast_fp16))[name = string("input_93_cast_fp16")]; tensor normed_131_axes_0 = const()[name = string("normed_131_axes_0"), val = tensor([-1])]; tensor normed_131_cast_fp16 = layer_norm(axes = normed_131_axes_0, epsilon = var_8_to_fp16, x = input_93_cast_fp16)[name = string("normed_131_cast_fp16")]; tensor var_1240_split_sizes_0 = const()[name = string("op_1240_split_sizes_0"), val = tensor([768, 768])]; int32 var_1240_axis_0 = const()[name = string("op_1240_axis_0"), val = int32(-1)]; tensor var_1240_cast_fp16_0, tensor var_1240_cast_fp16_1 = split(axis = var_1240_axis_0, split_sizes = var_1240_split_sizes_0, x = normed_131_cast_fp16)[name = string("op_1240_cast_fp16")]; tensor var_1244_to_fp16 = const()[name = string("op_1244_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297361088)))]; tensor out_57_cast_fp16 = mul(x = var_1240_cast_fp16_0, y = var_1244_to_fp16)[name = string("out_57_cast_fp16")]; tensor var_1251 = const()[name = string("op_1251"), val = tensor([0, 2, 1])]; tensor input_95_axes_0 = const()[name = string("input_95_axes_0"), val = tensor([2])]; tensor var_1252 = transpose(perm = var_1251, x = out_57_cast_fp16)[name = string("transpose_172")]; tensor input_95 = expand_dims(axes = input_95_axes_0, x = var_1252)[name = string("input_95")]; string gate_17_pad_type_0 = const()[name = string("gate_17_pad_type_0"), val = string("valid")]; tensor gate_17_strides_0 = const()[name = string("gate_17_strides_0"), val = tensor([1, 1])]; tensor gate_17_pad_0 = const()[name = string("gate_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_17_dilations_0 = const()[name = string("gate_17_dilations_0"), val = tensor([1, 1])]; int32 gate_17_groups_0 = const()[name = string("gate_17_groups_0"), val = int32(1)]; tensor gate_17 = conv(dilations = gate_17_dilations_0, groups = gate_17_groups_0, pad = gate_17_pad_0, pad_type = gate_17_pad_type_0, strides = gate_17_strides_0, weight = encoder_layers_4_mlp_gate_proj_weight_quantized, x = input_95)[name = string("gate_17")]; string up_9_pad_type_0 = const()[name = string("up_9_pad_type_0"), val = string("valid")]; tensor up_9_strides_0 = const()[name = string("up_9_strides_0"), val = tensor([1, 1])]; tensor up_9_pad_0 = const()[name = string("up_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_9_dilations_0 = const()[name = string("up_9_dilations_0"), val = tensor([1, 1])]; int32 up_9_groups_0 = const()[name = string("up_9_groups_0"), val = int32(1)]; tensor up_9 = conv(dilations = up_9_dilations_0, groups = up_9_groups_0, pad = up_9_pad_0, pad_type = up_9_pad_type_0, strides = up_9_strides_0, weight = encoder_layers_4_mlp_up_proj_weight_quantized, x = input_95)[name = string("up_9")]; string gate_19_mode_0 = const()[name = string("gate_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_19 = gelu(mode = gate_19_mode_0, x = gate_17)[name = string("gate_19")]; tensor input_97 = mul(x = gate_19, y = up_9)[name = string("input_97")]; string var_1273_pad_type_0 = const()[name = string("op_1273_pad_type_0"), val = string("valid")]; tensor var_1273_strides_0 = const()[name = string("op_1273_strides_0"), val = tensor([1, 1])]; tensor var_1273_pad_0 = const()[name = string("op_1273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1273_dilations_0 = const()[name = string("op_1273_dilations_0"), val = tensor([1, 1])]; int32 var_1273_groups_0 = const()[name = string("op_1273_groups_0"), val = int32(1)]; tensor var_1273 = conv(dilations = var_1273_dilations_0, groups = var_1273_groups_0, pad = var_1273_pad_0, pad_type = var_1273_pad_type_0, strides = var_1273_strides_0, weight = encoder_layers_4_mlp_down_proj_weight_quantized, x = input_97)[name = string("op_1273")]; tensor var_1274_axes_0 = const()[name = string("op_1274_axes_0"), val = tensor([2])]; tensor var_1274 = squeeze(axes = var_1274_axes_0, x = var_1273)[name = string("op_1274")]; tensor var_1275 = const()[name = string("op_1275"), val = tensor([0, 2, 1])]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_79 = transpose(perm = var_1275, x = var_1274)[name = string("transpose_171")]; tensor var_1279_cast_fp16 = mul(x = x_79, y = const_68_promoted_to_fp16)[name = string("op_1279_cast_fp16")]; bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; tensor input_99_cast_fp16 = concat(axis = var_22, interleave = input_99_interleave_0, values = (x_79, var_1279_cast_fp16))[name = string("input_99_cast_fp16")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_8_to_fp16, x = input_99_cast_fp16)[name = string("normed_137_cast_fp16")]; tensor var_1284_split_sizes_0 = const()[name = string("op_1284_split_sizes_0"), val = tensor([768, 768])]; int32 var_1284_axis_0 = const()[name = string("op_1284_axis_0"), val = int32(-1)]; tensor var_1284_cast_fp16_0, tensor var_1284_cast_fp16_1 = split(axis = var_1284_axis_0, split_sizes = var_1284_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_1284_cast_fp16")]; tensor var_1288_to_fp16 = const()[name = string("op_1288_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297362688)))]; tensor out_59_cast_fp16 = mul(x = var_1284_cast_fp16_0, y = var_1288_to_fp16)[name = string("out_59_cast_fp16")]; tensor x_81_cast_fp16 = add(x = x_75_cast_fp16, y = out_59_cast_fp16)[name = string("x_81_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1317_cast_fp16 = mul(x = x_81_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_1317_cast_fp16")]; bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; tensor input_101_cast_fp16 = concat(axis = var_22, interleave = input_101_interleave_0, values = (x_81_cast_fp16, var_1317_cast_fp16))[name = string("input_101_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_8_to_fp16, x = input_101_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_1322_split_sizes_0 = const()[name = string("op_1322_split_sizes_0"), val = tensor([768, 768])]; int32 var_1322_axis_0 = const()[name = string("op_1322_axis_0"), val = int32(-1)]; tensor var_1322_cast_fp16_0, tensor var_1322_cast_fp16_1 = split(axis = var_1322_axis_0, split_sizes = var_1322_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_1322_cast_fp16")]; tensor var_1326_to_fp16 = const()[name = string("op_1326_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297364288)))]; tensor out_61_cast_fp16 = mul(x = var_1322_cast_fp16_0, y = var_1326_to_fp16)[name = string("out_61_cast_fp16")]; tensor var_1332 = const()[name = string("op_1332"), val = tensor([0, 2, 1])]; tensor var_1334_axes_0 = const()[name = string("op_1334_axes_0"), val = tensor([2])]; tensor var_1333_cast_fp16 = transpose(perm = var_1332, x = out_61_cast_fp16)[name = string("transpose_170")]; tensor var_1334_cast_fp16 = expand_dims(axes = var_1334_axes_0, x = var_1333_cast_fp16)[name = string("op_1334_cast_fp16")]; string var_1341_pad_type_0 = const()[name = string("op_1341_pad_type_0"), val = string("valid")]; tensor var_1341_strides_0 = const()[name = string("op_1341_strides_0"), val = tensor([1, 1])]; tensor var_1341_pad_0 = const()[name = string("op_1341_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1341_dilations_0 = const()[name = string("op_1341_dilations_0"), val = tensor([1, 1])]; int32 var_1341_groups_0 = const()[name = string("op_1341_groups_0"), val = int32(1)]; tensor var_1341 = conv(dilations = var_1341_dilations_0, groups = var_1341_groups_0, pad = var_1341_pad_0, pad_type = var_1341_pad_type_0, strides = var_1341_strides_0, weight = encoder_layers_5_self_attn_q_proj_weight_quantized, x = var_1334_cast_fp16)[name = string("op_1341")]; tensor var_1342 = const()[name = string("op_1342"), val = tensor([1, 3, 256, 256])]; tensor var_1343 = reshape(shape = var_1342, x = var_1341)[name = string("op_1343")]; tensor var_1344 = const()[name = string("op_1344"), val = tensor([0, 1, 3, 2])]; string var_1351_pad_type_0 = const()[name = string("op_1351_pad_type_0"), val = string("valid")]; tensor var_1351_strides_0 = const()[name = string("op_1351_strides_0"), val = tensor([1, 1])]; tensor var_1351_pad_0 = const()[name = string("op_1351_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1351_dilations_0 = const()[name = string("op_1351_dilations_0"), val = tensor([1, 1])]; int32 var_1351_groups_0 = const()[name = string("op_1351_groups_0"), val = int32(1)]; tensor var_1351 = conv(dilations = var_1351_dilations_0, groups = var_1351_groups_0, pad = var_1351_pad_0, pad_type = var_1351_pad_type_0, strides = var_1351_strides_0, weight = encoder_layers_5_self_attn_k_proj_weight_quantized, x = var_1334_cast_fp16)[name = string("op_1351")]; tensor var_1352 = const()[name = string("op_1352"), val = tensor([1, 1, 256, 256])]; tensor var_1353 = reshape(shape = var_1352, x = var_1351)[name = string("op_1353")]; tensor var_1354 = const()[name = string("op_1354"), val = tensor([0, 1, 3, 2])]; string var_1361_pad_type_0 = const()[name = string("op_1361_pad_type_0"), val = string("valid")]; tensor var_1361_strides_0 = const()[name = string("op_1361_strides_0"), val = tensor([1, 1])]; tensor var_1361_pad_0 = const()[name = string("op_1361_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1361_dilations_0 = const()[name = string("op_1361_dilations_0"), val = tensor([1, 1])]; int32 var_1361_groups_0 = const()[name = string("op_1361_groups_0"), val = int32(1)]; tensor var_1361 = conv(dilations = var_1361_dilations_0, groups = var_1361_groups_0, pad = var_1361_pad_0, pad_type = var_1361_pad_type_0, strides = var_1361_strides_0, weight = encoder_layers_5_self_attn_v_proj_weight_quantized, x = var_1334_cast_fp16)[name = string("op_1361")]; tensor var_1362 = const()[name = string("op_1362"), val = tensor([1, 1, 256, 256])]; tensor var_1363 = reshape(shape = var_1362, x = var_1361)[name = string("op_1363")]; tensor var_1364 = const()[name = string("op_1364"), val = tensor([0, 1, 3, 2])]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_31 = transpose(perm = var_1344, x = var_1343)[name = string("transpose_169")]; tensor var_1370_cast_fp16 = mul(x = q_31, y = const_72_promoted_to_fp16)[name = string("op_1370_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_22, interleave = input_105_interleave_0, values = (q_31, var_1370_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_147_axes_0 = const()[name = string("normed_147_axes_0"), val = tensor([-1])]; tensor normed_147_cast_fp16 = layer_norm(axes = normed_147_axes_0, epsilon = var_8_to_fp16, x = input_105_cast_fp16)[name = string("normed_147_cast_fp16")]; tensor var_1375_split_sizes_0 = const()[name = string("op_1375_split_sizes_0"), val = tensor([256, 256])]; int32 var_1375_axis_0 = const()[name = string("op_1375_axis_0"), val = int32(-1)]; tensor var_1375_cast_fp16_0, tensor var_1375_cast_fp16_1 = split(axis = var_1375_axis_0, split_sizes = var_1375_split_sizes_0, x = normed_147_cast_fp16)[name = string("op_1375_cast_fp16")]; tensor var_1379_to_fp16 = const()[name = string("op_1379_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297365888)))]; tensor out_63_cast_fp16 = mul(x = var_1375_cast_fp16_0, y = var_1379_to_fp16)[name = string("out_63_cast_fp16")]; fp16 const_74_promoted_to_fp16 = const()[name = string("const_74_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_31 = transpose(perm = var_1354, x = var_1353)[name = string("transpose_168")]; tensor var_1386_cast_fp16 = mul(x = k_31, y = const_74_promoted_to_fp16)[name = string("op_1386_cast_fp16")]; bool input_107_interleave_0 = const()[name = string("input_107_interleave_0"), val = bool(false)]; tensor input_107_cast_fp16 = concat(axis = var_22, interleave = input_107_interleave_0, values = (k_31, var_1386_cast_fp16))[name = string("input_107_cast_fp16")]; tensor normed_151_axes_0 = const()[name = string("normed_151_axes_0"), val = tensor([-1])]; tensor normed_151_cast_fp16 = layer_norm(axes = normed_151_axes_0, epsilon = var_8_to_fp16, x = input_107_cast_fp16)[name = string("normed_151_cast_fp16")]; tensor var_1391_split_sizes_0 = const()[name = string("op_1391_split_sizes_0"), val = tensor([256, 256])]; int32 var_1391_axis_0 = const()[name = string("op_1391_axis_0"), val = int32(-1)]; tensor var_1391_cast_fp16_0, tensor var_1391_cast_fp16_1 = split(axis = var_1391_axis_0, split_sizes = var_1391_split_sizes_0, x = normed_151_cast_fp16)[name = string("op_1391_cast_fp16")]; tensor var_1395_to_fp16 = const()[name = string("op_1395_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297366464)))]; tensor out_65_cast_fp16 = mul(x = var_1391_cast_fp16_0, y = var_1395_to_fp16)[name = string("out_65_cast_fp16")]; tensor var_1398 = mul(x = out_63_cast_fp16, y = cos_quantized)[name = string("op_1398")]; tensor var_1399_split_sizes_0 = const()[name = string("op_1399_split_sizes_0"), val = tensor([128, 128])]; int32 var_1399_axis_0 = const()[name = string("op_1399_axis_0"), val = int32(-1)]; tensor var_1399_0, tensor var_1399_1 = split(axis = var_1399_axis_0, split_sizes = var_1399_split_sizes_0, x = out_63_cast_fp16)[name = string("op_1399")]; fp16 const_76_promoted = const()[name = string("const_76_promoted"), val = fp16(-0x1p+0)]; tensor var_1401 = mul(x = var_1399_1, y = const_76_promoted)[name = string("op_1401")]; bool var_1403_interleave_0 = const()[name = string("op_1403_interleave_0"), val = bool(false)]; tensor var_1403 = concat(axis = var_22, interleave = var_1403_interleave_0, values = (var_1401, var_1399_0))[name = string("op_1403")]; tensor var_1404 = mul(x = var_1403, y = sin_quantized)[name = string("op_1404")]; tensor q_35 = add(x = var_1398, y = var_1404)[name = string("q_35")]; tensor var_1406 = mul(x = out_65_cast_fp16, y = cos_quantized)[name = string("op_1406")]; tensor var_1407_split_sizes_0 = const()[name = string("op_1407_split_sizes_0"), val = tensor([128, 128])]; int32 var_1407_axis_0 = const()[name = string("op_1407_axis_0"), val = int32(-1)]; tensor var_1407_0, tensor var_1407_1 = split(axis = var_1407_axis_0, split_sizes = var_1407_split_sizes_0, x = out_65_cast_fp16)[name = string("op_1407")]; fp16 const_77_promoted = const()[name = string("const_77_promoted"), val = fp16(-0x1p+0)]; tensor var_1409 = mul(x = var_1407_1, y = const_77_promoted)[name = string("op_1409")]; bool var_1411_interleave_0 = const()[name = string("op_1411_interleave_0"), val = bool(false)]; tensor var_1411 = concat(axis = var_22, interleave = var_1411_interleave_0, values = (var_1409, var_1407_0))[name = string("op_1411")]; tensor var_1412 = mul(x = var_1411, y = sin_quantized)[name = string("op_1412")]; tensor hidden_states_61 = add(x = var_1406, y = var_1412)[name = string("hidden_states_61")]; tensor hidden_states_63_axes_0 = const()[name = string("hidden_states_63_axes_0"), val = tensor([2])]; tensor hidden_states_63 = expand_dims(axes = hidden_states_63_axes_0, x = hidden_states_61)[name = string("hidden_states_63")]; tensor var_1415 = const()[name = string("op_1415"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_65 = tile(reps = var_1415, x = hidden_states_63)[name = string("hidden_states_65")]; tensor var_1417 = const()[name = string("op_1417"), val = tensor([1, 3, 256, 256])]; tensor k_35 = reshape(shape = var_1417, x = hidden_states_65)[name = string("k_35")]; tensor hidden_states_69_axes_0 = const()[name = string("hidden_states_69_axes_0"), val = tensor([2])]; tensor hidden_states_67 = transpose(perm = var_1364, x = var_1363)[name = string("transpose_167")]; tensor hidden_states_69 = expand_dims(axes = hidden_states_69_axes_0, x = hidden_states_67)[name = string("hidden_states_69")]; tensor var_1420 = const()[name = string("op_1420"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_71 = tile(reps = var_1420, x = hidden_states_69)[name = string("hidden_states_71")]; tensor var_1422 = const()[name = string("op_1422"), val = tensor([1, 3, 256, 256])]; tensor v_11 = reshape(shape = var_1422, x = hidden_states_71)[name = string("v_11")]; bool var_1427_transpose_x_1 = const()[name = string("op_1427_transpose_x_1"), val = bool(false)]; bool var_1427_transpose_y_1 = const()[name = string("op_1427_transpose_y_1"), val = bool(true)]; tensor var_1427_cast_fp16 = matmul(transpose_x = var_1427_transpose_x_1, transpose_y = var_1427_transpose_y_1, x = q_35, y = k_35)[name = string("op_1427_cast_fp16")]; fp16 var_1428_to_fp16 = const()[name = string("op_1428_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_31_cast_fp16 = mul(x = var_1427_cast_fp16, y = var_1428_to_fp16)[name = string("attn_weights_31_cast_fp16")]; tensor attn_weights_33_cast_fp16 = add(x = attn_weights_31_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_33_cast_fp16")]; tensor var_1432_cast_fp16 = softmax(axis = var_22, x = attn_weights_33_cast_fp16)[name = string("op_1432_cast_fp16")]; bool var_1436_transpose_x_0 = const()[name = string("op_1436_transpose_x_0"), val = bool(false)]; bool var_1436_transpose_y_0 = const()[name = string("op_1436_transpose_y_0"), val = bool(false)]; tensor var_1436_cast_fp16 = matmul(transpose_x = var_1436_transpose_x_0, transpose_y = var_1436_transpose_y_0, x = var_1432_cast_fp16, y = v_11)[name = string("op_1436_cast_fp16")]; tensor var_1438 = const()[name = string("op_1438"), val = tensor([0, 2, 1, 3])]; tensor var_1441 = const()[name = string("op_1441"), val = tensor([1, 256, 768])]; tensor var_1439 = transpose(perm = var_1438, x = var_1436_cast_fp16)[name = string("transpose_166")]; tensor attn_out_33 = reshape(shape = var_1441, x = var_1439)[name = string("attn_out_33")]; tensor var_1443 = const()[name = string("op_1443"), val = tensor([0, 2, 1])]; tensor squeeze_5_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297367040))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297956928))))[name = string("squeeze_5_quantized")]; string var_1452_pad_type_0 = const()[name = string("op_1452_pad_type_0"), val = string("valid")]; int32 var_1452_groups_0 = const()[name = string("op_1452_groups_0"), val = int32(1)]; tensor var_1452_strides_0 = const()[name = string("op_1452_strides_0"), val = tensor([1])]; tensor var_1452_pad_0 = const()[name = string("op_1452_pad_0"), val = tensor([0, 0])]; tensor var_1452_dilations_0 = const()[name = string("op_1452_dilations_0"), val = tensor([1])]; tensor var_1444 = transpose(perm = var_1443, x = attn_out_33)[name = string("transpose_165")]; tensor var_1452 = conv(dilations = var_1452_dilations_0, groups = var_1452_groups_0, pad = var_1452_pad_0, pad_type = var_1452_pad_type_0, strides = var_1452_strides_0, weight = squeeze_5_quantized, x = var_1444)[name = string("op_1452")]; tensor var_1453 = const()[name = string("op_1453"), val = tensor([0, 2, 1])]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_89 = transpose(perm = var_1453, x = var_1452)[name = string("transpose_164")]; tensor var_1457_cast_fp16 = mul(x = x_89, y = const_78_promoted_to_fp16)[name = string("op_1457_cast_fp16")]; bool input_111_interleave_0 = const()[name = string("input_111_interleave_0"), val = bool(false)]; tensor input_111_cast_fp16 = concat(axis = var_22, interleave = input_111_interleave_0, values = (x_89, var_1457_cast_fp16))[name = string("input_111_cast_fp16")]; tensor normed_155_axes_0 = const()[name = string("normed_155_axes_0"), val = tensor([-1])]; tensor normed_155_cast_fp16 = layer_norm(axes = normed_155_axes_0, epsilon = var_8_to_fp16, x = input_111_cast_fp16)[name = string("normed_155_cast_fp16")]; tensor var_1462_split_sizes_0 = const()[name = string("op_1462_split_sizes_0"), val = tensor([768, 768])]; int32 var_1462_axis_0 = const()[name = string("op_1462_axis_0"), val = int32(-1)]; tensor var_1462_cast_fp16_0, tensor var_1462_cast_fp16_1 = split(axis = var_1462_axis_0, split_sizes = var_1462_split_sizes_0, x = normed_155_cast_fp16)[name = string("op_1462_cast_fp16")]; tensor var_1466_to_fp16 = const()[name = string("op_1466_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297958528)))]; tensor out_67_cast_fp16 = mul(x = var_1462_cast_fp16_0, y = var_1466_to_fp16)[name = string("out_67_cast_fp16")]; tensor x_91_cast_fp16 = add(x = x_81_cast_fp16, y = out_67_cast_fp16)[name = string("x_91_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1473_cast_fp16 = mul(x = x_91_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_1473_cast_fp16")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113_cast_fp16 = concat(axis = var_22, interleave = input_113_interleave_0, values = (x_91_cast_fp16, var_1473_cast_fp16))[name = string("input_113_cast_fp16")]; tensor normed_159_axes_0 = const()[name = string("normed_159_axes_0"), val = tensor([-1])]; tensor normed_159_cast_fp16 = layer_norm(axes = normed_159_axes_0, epsilon = var_8_to_fp16, x = input_113_cast_fp16)[name = string("normed_159_cast_fp16")]; tensor var_1478_split_sizes_0 = const()[name = string("op_1478_split_sizes_0"), val = tensor([768, 768])]; int32 var_1478_axis_0 = const()[name = string("op_1478_axis_0"), val = int32(-1)]; tensor var_1478_cast_fp16_0, tensor var_1478_cast_fp16_1 = split(axis = var_1478_axis_0, split_sizes = var_1478_split_sizes_0, x = normed_159_cast_fp16)[name = string("op_1478_cast_fp16")]; tensor var_1482_to_fp16 = const()[name = string("op_1482_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297960128)))]; tensor out_69_cast_fp16 = mul(x = var_1478_cast_fp16_0, y = var_1482_to_fp16)[name = string("out_69_cast_fp16")]; tensor var_1489 = const()[name = string("op_1489"), val = tensor([0, 2, 1])]; tensor input_115_axes_0 = const()[name = string("input_115_axes_0"), val = tensor([2])]; tensor var_1490 = transpose(perm = var_1489, x = out_69_cast_fp16)[name = string("transpose_163")]; tensor input_115 = expand_dims(axes = input_115_axes_0, x = var_1490)[name = string("input_115")]; string gate_21_pad_type_0 = const()[name = string("gate_21_pad_type_0"), val = string("valid")]; tensor gate_21_strides_0 = const()[name = string("gate_21_strides_0"), val = tensor([1, 1])]; tensor gate_21_pad_0 = const()[name = string("gate_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_21_dilations_0 = const()[name = string("gate_21_dilations_0"), val = tensor([1, 1])]; int32 gate_21_groups_0 = const()[name = string("gate_21_groups_0"), val = int32(1)]; tensor gate_21 = conv(dilations = gate_21_dilations_0, groups = gate_21_groups_0, pad = gate_21_pad_0, pad_type = gate_21_pad_type_0, strides = gate_21_strides_0, weight = encoder_layers_5_mlp_gate_proj_weight_quantized, x = input_115)[name = string("gate_21")]; string up_11_pad_type_0 = const()[name = string("up_11_pad_type_0"), val = string("valid")]; tensor up_11_strides_0 = const()[name = string("up_11_strides_0"), val = tensor([1, 1])]; tensor up_11_pad_0 = const()[name = string("up_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_11_dilations_0 = const()[name = string("up_11_dilations_0"), val = tensor([1, 1])]; int32 up_11_groups_0 = const()[name = string("up_11_groups_0"), val = int32(1)]; tensor up_11 = conv(dilations = up_11_dilations_0, groups = up_11_groups_0, pad = up_11_pad_0, pad_type = up_11_pad_type_0, strides = up_11_strides_0, weight = encoder_layers_5_mlp_up_proj_weight_quantized, x = input_115)[name = string("up_11")]; string gate_23_mode_0 = const()[name = string("gate_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_23 = gelu(mode = gate_23_mode_0, x = gate_21)[name = string("gate_23")]; tensor input_117 = mul(x = gate_23, y = up_11)[name = string("input_117")]; string var_1511_pad_type_0 = const()[name = string("op_1511_pad_type_0"), val = string("valid")]; tensor var_1511_strides_0 = const()[name = string("op_1511_strides_0"), val = tensor([1, 1])]; tensor var_1511_pad_0 = const()[name = string("op_1511_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1511_dilations_0 = const()[name = string("op_1511_dilations_0"), val = tensor([1, 1])]; int32 var_1511_groups_0 = const()[name = string("op_1511_groups_0"), val = int32(1)]; tensor var_1511 = conv(dilations = var_1511_dilations_0, groups = var_1511_groups_0, pad = var_1511_pad_0, pad_type = var_1511_pad_type_0, strides = var_1511_strides_0, weight = encoder_layers_5_mlp_down_proj_weight_quantized, x = input_117)[name = string("op_1511")]; tensor var_1512_axes_0 = const()[name = string("op_1512_axes_0"), val = tensor([2])]; tensor var_1512 = squeeze(axes = var_1512_axes_0, x = var_1511)[name = string("op_1512")]; tensor var_1513 = const()[name = string("op_1513"), val = tensor([0, 2, 1])]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_95 = transpose(perm = var_1513, x = var_1512)[name = string("transpose_162")]; tensor var_1517_cast_fp16 = mul(x = x_95, y = const_82_promoted_to_fp16)[name = string("op_1517_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_22, interleave = input_119_interleave_0, values = (x_95, var_1517_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_8_to_fp16, x = input_119_cast_fp16)[name = string("normed_165_cast_fp16")]; tensor var_1522_split_sizes_0 = const()[name = string("op_1522_split_sizes_0"), val = tensor([768, 768])]; int32 var_1522_axis_0 = const()[name = string("op_1522_axis_0"), val = int32(-1)]; tensor var_1522_cast_fp16_0, tensor var_1522_cast_fp16_1 = split(axis = var_1522_axis_0, split_sizes = var_1522_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_1522_cast_fp16")]; tensor var_1526_to_fp16 = const()[name = string("op_1526_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297961728)))]; tensor out_71_cast_fp16 = mul(x = var_1522_cast_fp16_0, y = var_1526_to_fp16)[name = string("out_71_cast_fp16")]; tensor x_97_cast_fp16 = add(x = x_91_cast_fp16, y = out_71_cast_fp16)[name = string("x_97_cast_fp16")]; fp16 const_84_promoted_to_fp16 = const()[name = string("const_84_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1555_cast_fp16 = mul(x = x_97_cast_fp16, y = const_84_promoted_to_fp16)[name = string("op_1555_cast_fp16")]; bool input_121_interleave_0 = const()[name = string("input_121_interleave_0"), val = bool(false)]; tensor input_121_cast_fp16 = concat(axis = var_22, interleave = input_121_interleave_0, values = (x_97_cast_fp16, var_1555_cast_fp16))[name = string("input_121_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_8_to_fp16, x = input_121_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_1560_split_sizes_0 = const()[name = string("op_1560_split_sizes_0"), val = tensor([768, 768])]; int32 var_1560_axis_0 = const()[name = string("op_1560_axis_0"), val = int32(-1)]; tensor var_1560_cast_fp16_0, tensor var_1560_cast_fp16_1 = split(axis = var_1560_axis_0, split_sizes = var_1560_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_1560_cast_fp16")]; tensor var_1564_to_fp16 = const()[name = string("op_1564_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297963328)))]; tensor out_73_cast_fp16 = mul(x = var_1560_cast_fp16_0, y = var_1564_to_fp16)[name = string("out_73_cast_fp16")]; tensor var_1570 = const()[name = string("op_1570"), val = tensor([0, 2, 1])]; tensor var_1572_axes_0 = const()[name = string("op_1572_axes_0"), val = tensor([2])]; tensor var_1571_cast_fp16 = transpose(perm = var_1570, x = out_73_cast_fp16)[name = string("transpose_161")]; tensor var_1572_cast_fp16 = expand_dims(axes = var_1572_axes_0, x = var_1571_cast_fp16)[name = string("op_1572_cast_fp16")]; string var_1579_pad_type_0 = const()[name = string("op_1579_pad_type_0"), val = string("valid")]; tensor var_1579_strides_0 = const()[name = string("op_1579_strides_0"), val = tensor([1, 1])]; tensor var_1579_pad_0 = const()[name = string("op_1579_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1579_dilations_0 = const()[name = string("op_1579_dilations_0"), val = tensor([1, 1])]; int32 var_1579_groups_0 = const()[name = string("op_1579_groups_0"), val = int32(1)]; tensor var_1579 = conv(dilations = var_1579_dilations_0, groups = var_1579_groups_0, pad = var_1579_pad_0, pad_type = var_1579_pad_type_0, strides = var_1579_strides_0, weight = encoder_layers_6_self_attn_q_proj_weight_quantized, x = var_1572_cast_fp16)[name = string("op_1579")]; tensor var_1580 = const()[name = string("op_1580"), val = tensor([1, 3, 256, 256])]; tensor var_1581 = reshape(shape = var_1580, x = var_1579)[name = string("op_1581")]; tensor var_1582 = const()[name = string("op_1582"), val = tensor([0, 1, 3, 2])]; string var_1589_pad_type_0 = const()[name = string("op_1589_pad_type_0"), val = string("valid")]; tensor var_1589_strides_0 = const()[name = string("op_1589_strides_0"), val = tensor([1, 1])]; tensor var_1589_pad_0 = const()[name = string("op_1589_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1589_dilations_0 = const()[name = string("op_1589_dilations_0"), val = tensor([1, 1])]; int32 var_1589_groups_0 = const()[name = string("op_1589_groups_0"), val = int32(1)]; tensor var_1589 = conv(dilations = var_1589_dilations_0, groups = var_1589_groups_0, pad = var_1589_pad_0, pad_type = var_1589_pad_type_0, strides = var_1589_strides_0, weight = encoder_layers_6_self_attn_k_proj_weight_quantized, x = var_1572_cast_fp16)[name = string("op_1589")]; tensor var_1590 = const()[name = string("op_1590"), val = tensor([1, 1, 256, 256])]; tensor var_1591 = reshape(shape = var_1590, x = var_1589)[name = string("op_1591")]; tensor var_1592 = const()[name = string("op_1592"), val = tensor([0, 1, 3, 2])]; string var_1599_pad_type_0 = const()[name = string("op_1599_pad_type_0"), val = string("valid")]; tensor var_1599_strides_0 = const()[name = string("op_1599_strides_0"), val = tensor([1, 1])]; tensor var_1599_pad_0 = const()[name = string("op_1599_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1599_dilations_0 = const()[name = string("op_1599_dilations_0"), val = tensor([1, 1])]; int32 var_1599_groups_0 = const()[name = string("op_1599_groups_0"), val = int32(1)]; tensor var_1599 = conv(dilations = var_1599_dilations_0, groups = var_1599_groups_0, pad = var_1599_pad_0, pad_type = var_1599_pad_type_0, strides = var_1599_strides_0, weight = encoder_layers_6_self_attn_v_proj_weight_quantized, x = var_1572_cast_fp16)[name = string("op_1599")]; tensor var_1600 = const()[name = string("op_1600"), val = tensor([1, 1, 256, 256])]; tensor var_1601 = reshape(shape = var_1600, x = var_1599)[name = string("op_1601")]; tensor var_1602 = const()[name = string("op_1602"), val = tensor([0, 1, 3, 2])]; fp16 const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_37 = transpose(perm = var_1582, x = var_1581)[name = string("transpose_160")]; tensor var_1608_cast_fp16 = mul(x = q_37, y = const_86_promoted_to_fp16)[name = string("op_1608_cast_fp16")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125_cast_fp16 = concat(axis = var_22, interleave = input_125_interleave_0, values = (q_37, var_1608_cast_fp16))[name = string("input_125_cast_fp16")]; tensor normed_175_axes_0 = const()[name = string("normed_175_axes_0"), val = tensor([-1])]; tensor normed_175_cast_fp16 = layer_norm(axes = normed_175_axes_0, epsilon = var_8_to_fp16, x = input_125_cast_fp16)[name = string("normed_175_cast_fp16")]; tensor var_1613_split_sizes_0 = const()[name = string("op_1613_split_sizes_0"), val = tensor([256, 256])]; int32 var_1613_axis_0 = const()[name = string("op_1613_axis_0"), val = int32(-1)]; tensor var_1613_cast_fp16_0, tensor var_1613_cast_fp16_1 = split(axis = var_1613_axis_0, split_sizes = var_1613_split_sizes_0, x = normed_175_cast_fp16)[name = string("op_1613_cast_fp16")]; tensor var_1617_to_fp16 = const()[name = string("op_1617_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297964928)))]; tensor out_75_cast_fp16 = mul(x = var_1613_cast_fp16_0, y = var_1617_to_fp16)[name = string("out_75_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_37 = transpose(perm = var_1592, x = var_1591)[name = string("transpose_159")]; tensor var_1624_cast_fp16 = mul(x = k_37, y = const_88_promoted_to_fp16)[name = string("op_1624_cast_fp16")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127_cast_fp16 = concat(axis = var_22, interleave = input_127_interleave_0, values = (k_37, var_1624_cast_fp16))[name = string("input_127_cast_fp16")]; tensor normed_179_axes_0 = const()[name = string("normed_179_axes_0"), val = tensor([-1])]; tensor normed_179_cast_fp16 = layer_norm(axes = normed_179_axes_0, epsilon = var_8_to_fp16, x = input_127_cast_fp16)[name = string("normed_179_cast_fp16")]; tensor var_1629_split_sizes_0 = const()[name = string("op_1629_split_sizes_0"), val = tensor([256, 256])]; int32 var_1629_axis_0 = const()[name = string("op_1629_axis_0"), val = int32(-1)]; tensor var_1629_cast_fp16_0, tensor var_1629_cast_fp16_1 = split(axis = var_1629_axis_0, split_sizes = var_1629_split_sizes_0, x = normed_179_cast_fp16)[name = string("op_1629_cast_fp16")]; tensor var_1633_to_fp16 = const()[name = string("op_1633_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297965504)))]; tensor out_77_cast_fp16 = mul(x = var_1629_cast_fp16_0, y = var_1633_to_fp16)[name = string("out_77_cast_fp16")]; tensor var_1636 = mul(x = out_75_cast_fp16, y = cos_1_quantized)[name = string("op_1636")]; tensor var_1637_split_sizes_0 = const()[name = string("op_1637_split_sizes_0"), val = tensor([128, 128])]; int32 var_1637_axis_0 = const()[name = string("op_1637_axis_0"), val = int32(-1)]; tensor var_1637_0, tensor var_1637_1 = split(axis = var_1637_axis_0, split_sizes = var_1637_split_sizes_0, x = out_75_cast_fp16)[name = string("op_1637")]; fp16 const_90_promoted = const()[name = string("const_90_promoted"), val = fp16(-0x1p+0)]; tensor var_1639 = mul(x = var_1637_1, y = const_90_promoted)[name = string("op_1639")]; bool var_1641_interleave_0 = const()[name = string("op_1641_interleave_0"), val = bool(false)]; tensor var_1641 = concat(axis = var_22, interleave = var_1641_interleave_0, values = (var_1639, var_1637_0))[name = string("op_1641")]; tensor var_1642 = mul(x = var_1641, y = sin_1_quantized)[name = string("op_1642")]; tensor q_41 = add(x = var_1636, y = var_1642)[name = string("q_41")]; tensor var_1644 = mul(x = out_77_cast_fp16, y = cos_1_quantized)[name = string("op_1644")]; tensor var_1645_split_sizes_0 = const()[name = string("op_1645_split_sizes_0"), val = tensor([128, 128])]; int32 var_1645_axis_0 = const()[name = string("op_1645_axis_0"), val = int32(-1)]; tensor var_1645_0, tensor var_1645_1 = split(axis = var_1645_axis_0, split_sizes = var_1645_split_sizes_0, x = out_77_cast_fp16)[name = string("op_1645")]; fp16 const_91_promoted = const()[name = string("const_91_promoted"), val = fp16(-0x1p+0)]; tensor var_1647 = mul(x = var_1645_1, y = const_91_promoted)[name = string("op_1647")]; bool var_1649_interleave_0 = const()[name = string("op_1649_interleave_0"), val = bool(false)]; tensor var_1649 = concat(axis = var_22, interleave = var_1649_interleave_0, values = (var_1647, var_1645_0))[name = string("op_1649")]; tensor var_1650 = mul(x = var_1649, y = sin_1_quantized)[name = string("op_1650")]; tensor hidden_states_73 = add(x = var_1644, y = var_1650)[name = string("hidden_states_73")]; tensor hidden_states_75_axes_0 = const()[name = string("hidden_states_75_axes_0"), val = tensor([2])]; tensor hidden_states_75 = expand_dims(axes = hidden_states_75_axes_0, x = hidden_states_73)[name = string("hidden_states_75")]; tensor var_1653 = const()[name = string("op_1653"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_77 = tile(reps = var_1653, x = hidden_states_75)[name = string("hidden_states_77")]; tensor var_1655 = const()[name = string("op_1655"), val = tensor([1, 3, 256, 256])]; tensor k_41 = reshape(shape = var_1655, x = hidden_states_77)[name = string("k_41")]; tensor hidden_states_81_axes_0 = const()[name = string("hidden_states_81_axes_0"), val = tensor([2])]; tensor hidden_states_79 = transpose(perm = var_1602, x = var_1601)[name = string("transpose_158")]; tensor hidden_states_81 = expand_dims(axes = hidden_states_81_axes_0, x = hidden_states_79)[name = string("hidden_states_81")]; tensor var_1658 = const()[name = string("op_1658"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_83 = tile(reps = var_1658, x = hidden_states_81)[name = string("hidden_states_83")]; tensor var_1660 = const()[name = string("op_1660"), val = tensor([1, 3, 256, 256])]; tensor v_13 = reshape(shape = var_1660, x = hidden_states_83)[name = string("v_13")]; bool var_1665_transpose_x_1 = const()[name = string("op_1665_transpose_x_1"), val = bool(false)]; bool var_1665_transpose_y_1 = const()[name = string("op_1665_transpose_y_1"), val = bool(true)]; tensor var_1665_cast_fp16 = matmul(transpose_x = var_1665_transpose_x_1, transpose_y = var_1665_transpose_y_1, x = q_41, y = k_41)[name = string("op_1665_cast_fp16")]; fp16 var_1666_to_fp16 = const()[name = string("op_1666_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_37_cast_fp16 = mul(x = var_1665_cast_fp16, y = var_1666_to_fp16)[name = string("attn_weights_37_cast_fp16")]; tensor attn_weights_39_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; tensor var_1670_cast_fp16 = softmax(axis = var_22, x = attn_weights_39_cast_fp16)[name = string("op_1670_cast_fp16")]; bool var_1674_transpose_x_0 = const()[name = string("op_1674_transpose_x_0"), val = bool(false)]; bool var_1674_transpose_y_0 = const()[name = string("op_1674_transpose_y_0"), val = bool(false)]; tensor var_1674_cast_fp16 = matmul(transpose_x = var_1674_transpose_x_0, transpose_y = var_1674_transpose_y_0, x = var_1670_cast_fp16, y = v_13)[name = string("op_1674_cast_fp16")]; tensor var_1676 = const()[name = string("op_1676"), val = tensor([0, 2, 1, 3])]; tensor var_1679 = const()[name = string("op_1679"), val = tensor([1, 256, 768])]; tensor var_1677 = transpose(perm = var_1676, x = var_1674_cast_fp16)[name = string("transpose_157")]; tensor attn_out_39 = reshape(shape = var_1679, x = var_1677)[name = string("attn_out_39")]; tensor var_1681 = const()[name = string("op_1681"), val = tensor([0, 2, 1])]; tensor squeeze_6_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297966080))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298555968))))[name = string("squeeze_6_quantized")]; string var_1690_pad_type_0 = const()[name = string("op_1690_pad_type_0"), val = string("valid")]; int32 var_1690_groups_0 = const()[name = string("op_1690_groups_0"), val = int32(1)]; tensor var_1690_strides_0 = const()[name = string("op_1690_strides_0"), val = tensor([1])]; tensor var_1690_pad_0 = const()[name = string("op_1690_pad_0"), val = tensor([0, 0])]; tensor var_1690_dilations_0 = const()[name = string("op_1690_dilations_0"), val = tensor([1])]; tensor var_1682 = transpose(perm = var_1681, x = attn_out_39)[name = string("transpose_156")]; tensor var_1690 = conv(dilations = var_1690_dilations_0, groups = var_1690_groups_0, pad = var_1690_pad_0, pad_type = var_1690_pad_type_0, strides = var_1690_strides_0, weight = squeeze_6_quantized, x = var_1682)[name = string("op_1690")]; tensor var_1691 = const()[name = string("op_1691"), val = tensor([0, 2, 1])]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_105 = transpose(perm = var_1691, x = var_1690)[name = string("transpose_155")]; tensor var_1695_cast_fp16 = mul(x = x_105, y = const_92_promoted_to_fp16)[name = string("op_1695_cast_fp16")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131_cast_fp16 = concat(axis = var_22, interleave = input_131_interleave_0, values = (x_105, var_1695_cast_fp16))[name = string("input_131_cast_fp16")]; tensor normed_183_axes_0 = const()[name = string("normed_183_axes_0"), val = tensor([-1])]; tensor normed_183_cast_fp16 = layer_norm(axes = normed_183_axes_0, epsilon = var_8_to_fp16, x = input_131_cast_fp16)[name = string("normed_183_cast_fp16")]; tensor var_1700_split_sizes_0 = const()[name = string("op_1700_split_sizes_0"), val = tensor([768, 768])]; int32 var_1700_axis_0 = const()[name = string("op_1700_axis_0"), val = int32(-1)]; tensor var_1700_cast_fp16_0, tensor var_1700_cast_fp16_1 = split(axis = var_1700_axis_0, split_sizes = var_1700_split_sizes_0, x = normed_183_cast_fp16)[name = string("op_1700_cast_fp16")]; tensor var_1704_to_fp16 = const()[name = string("op_1704_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298557568)))]; tensor out_79_cast_fp16 = mul(x = var_1700_cast_fp16_0, y = var_1704_to_fp16)[name = string("out_79_cast_fp16")]; tensor x_107_cast_fp16 = add(x = x_97_cast_fp16, y = out_79_cast_fp16)[name = string("x_107_cast_fp16")]; fp16 const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1711_cast_fp16 = mul(x = x_107_cast_fp16, y = const_94_promoted_to_fp16)[name = string("op_1711_cast_fp16")]; bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; tensor input_133_cast_fp16 = concat(axis = var_22, interleave = input_133_interleave_0, values = (x_107_cast_fp16, var_1711_cast_fp16))[name = string("input_133_cast_fp16")]; tensor normed_187_axes_0 = const()[name = string("normed_187_axes_0"), val = tensor([-1])]; tensor normed_187_cast_fp16 = layer_norm(axes = normed_187_axes_0, epsilon = var_8_to_fp16, x = input_133_cast_fp16)[name = string("normed_187_cast_fp16")]; tensor var_1716_split_sizes_0 = const()[name = string("op_1716_split_sizes_0"), val = tensor([768, 768])]; int32 var_1716_axis_0 = const()[name = string("op_1716_axis_0"), val = int32(-1)]; tensor var_1716_cast_fp16_0, tensor var_1716_cast_fp16_1 = split(axis = var_1716_axis_0, split_sizes = var_1716_split_sizes_0, x = normed_187_cast_fp16)[name = string("op_1716_cast_fp16")]; tensor var_1720_to_fp16 = const()[name = string("op_1720_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298559168)))]; tensor out_81_cast_fp16 = mul(x = var_1716_cast_fp16_0, y = var_1720_to_fp16)[name = string("out_81_cast_fp16")]; tensor var_1727 = const()[name = string("op_1727"), val = tensor([0, 2, 1])]; tensor input_135_axes_0 = const()[name = string("input_135_axes_0"), val = tensor([2])]; tensor var_1728 = transpose(perm = var_1727, x = out_81_cast_fp16)[name = string("transpose_154")]; tensor input_135 = expand_dims(axes = input_135_axes_0, x = var_1728)[name = string("input_135")]; string gate_25_pad_type_0 = const()[name = string("gate_25_pad_type_0"), val = string("valid")]; tensor gate_25_strides_0 = const()[name = string("gate_25_strides_0"), val = tensor([1, 1])]; tensor gate_25_pad_0 = const()[name = string("gate_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_25_dilations_0 = const()[name = string("gate_25_dilations_0"), val = tensor([1, 1])]; int32 gate_25_groups_0 = const()[name = string("gate_25_groups_0"), val = int32(1)]; tensor gate_25 = conv(dilations = gate_25_dilations_0, groups = gate_25_groups_0, pad = gate_25_pad_0, pad_type = gate_25_pad_type_0, strides = gate_25_strides_0, weight = encoder_layers_6_mlp_gate_proj_weight_quantized, x = input_135)[name = string("gate_25")]; string up_13_pad_type_0 = const()[name = string("up_13_pad_type_0"), val = string("valid")]; tensor up_13_strides_0 = const()[name = string("up_13_strides_0"), val = tensor([1, 1])]; tensor up_13_pad_0 = const()[name = string("up_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_13_dilations_0 = const()[name = string("up_13_dilations_0"), val = tensor([1, 1])]; int32 up_13_groups_0 = const()[name = string("up_13_groups_0"), val = int32(1)]; tensor up_13 = conv(dilations = up_13_dilations_0, groups = up_13_groups_0, pad = up_13_pad_0, pad_type = up_13_pad_type_0, strides = up_13_strides_0, weight = encoder_layers_6_mlp_up_proj_weight_quantized, x = input_135)[name = string("up_13")]; string gate_27_mode_0 = const()[name = string("gate_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_27 = gelu(mode = gate_27_mode_0, x = gate_25)[name = string("gate_27")]; tensor input_137 = mul(x = gate_27, y = up_13)[name = string("input_137")]; string var_1749_pad_type_0 = const()[name = string("op_1749_pad_type_0"), val = string("valid")]; tensor var_1749_strides_0 = const()[name = string("op_1749_strides_0"), val = tensor([1, 1])]; tensor var_1749_pad_0 = const()[name = string("op_1749_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1749_dilations_0 = const()[name = string("op_1749_dilations_0"), val = tensor([1, 1])]; int32 var_1749_groups_0 = const()[name = string("op_1749_groups_0"), val = int32(1)]; tensor var_1749 = conv(dilations = var_1749_dilations_0, groups = var_1749_groups_0, pad = var_1749_pad_0, pad_type = var_1749_pad_type_0, strides = var_1749_strides_0, weight = encoder_layers_6_mlp_down_proj_weight_quantized, x = input_137)[name = string("op_1749")]; tensor var_1750_axes_0 = const()[name = string("op_1750_axes_0"), val = tensor([2])]; tensor var_1750 = squeeze(axes = var_1750_axes_0, x = var_1749)[name = string("op_1750")]; tensor var_1751 = const()[name = string("op_1751"), val = tensor([0, 2, 1])]; fp16 const_96_promoted_to_fp16 = const()[name = string("const_96_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_111 = transpose(perm = var_1751, x = var_1750)[name = string("transpose_153")]; tensor var_1755_cast_fp16 = mul(x = x_111, y = const_96_promoted_to_fp16)[name = string("op_1755_cast_fp16")]; bool input_139_interleave_0 = const()[name = string("input_139_interleave_0"), val = bool(false)]; tensor input_139_cast_fp16 = concat(axis = var_22, interleave = input_139_interleave_0, values = (x_111, var_1755_cast_fp16))[name = string("input_139_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_8_to_fp16, x = input_139_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor var_1760_split_sizes_0 = const()[name = string("op_1760_split_sizes_0"), val = tensor([768, 768])]; int32 var_1760_axis_0 = const()[name = string("op_1760_axis_0"), val = int32(-1)]; tensor var_1760_cast_fp16_0, tensor var_1760_cast_fp16_1 = split(axis = var_1760_axis_0, split_sizes = var_1760_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_1760_cast_fp16")]; tensor var_1764_to_fp16 = const()[name = string("op_1764_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298560768)))]; tensor out_83_cast_fp16 = mul(x = var_1760_cast_fp16_0, y = var_1764_to_fp16)[name = string("out_83_cast_fp16")]; tensor x_113_cast_fp16 = add(x = x_107_cast_fp16, y = out_83_cast_fp16)[name = string("x_113_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1793_cast_fp16 = mul(x = x_113_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_1793_cast_fp16")]; bool input_141_interleave_0 = const()[name = string("input_141_interleave_0"), val = bool(false)]; tensor input_141_cast_fp16 = concat(axis = var_22, interleave = input_141_interleave_0, values = (x_113_cast_fp16, var_1793_cast_fp16))[name = string("input_141_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_8_to_fp16, x = input_141_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_1798_split_sizes_0 = const()[name = string("op_1798_split_sizes_0"), val = tensor([768, 768])]; int32 var_1798_axis_0 = const()[name = string("op_1798_axis_0"), val = int32(-1)]; tensor var_1798_cast_fp16_0, tensor var_1798_cast_fp16_1 = split(axis = var_1798_axis_0, split_sizes = var_1798_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_1798_cast_fp16")]; tensor var_1802_to_fp16 = const()[name = string("op_1802_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298562368)))]; tensor out_85_cast_fp16 = mul(x = var_1798_cast_fp16_0, y = var_1802_to_fp16)[name = string("out_85_cast_fp16")]; tensor var_1808 = const()[name = string("op_1808"), val = tensor([0, 2, 1])]; tensor var_1810_axes_0 = const()[name = string("op_1810_axes_0"), val = tensor([2])]; tensor var_1809_cast_fp16 = transpose(perm = var_1808, x = out_85_cast_fp16)[name = string("transpose_152")]; tensor var_1810_cast_fp16 = expand_dims(axes = var_1810_axes_0, x = var_1809_cast_fp16)[name = string("op_1810_cast_fp16")]; string var_1817_pad_type_0 = const()[name = string("op_1817_pad_type_0"), val = string("valid")]; tensor var_1817_strides_0 = const()[name = string("op_1817_strides_0"), val = tensor([1, 1])]; tensor var_1817_pad_0 = const()[name = string("op_1817_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1817_dilations_0 = const()[name = string("op_1817_dilations_0"), val = tensor([1, 1])]; int32 var_1817_groups_0 = const()[name = string("op_1817_groups_0"), val = int32(1)]; tensor var_1817 = conv(dilations = var_1817_dilations_0, groups = var_1817_groups_0, pad = var_1817_pad_0, pad_type = var_1817_pad_type_0, strides = var_1817_strides_0, weight = encoder_layers_7_self_attn_q_proj_weight_quantized, x = var_1810_cast_fp16)[name = string("op_1817")]; tensor var_1818 = const()[name = string("op_1818"), val = tensor([1, 3, 256, 256])]; tensor var_1819 = reshape(shape = var_1818, x = var_1817)[name = string("op_1819")]; tensor var_1820 = const()[name = string("op_1820"), val = tensor([0, 1, 3, 2])]; string var_1827_pad_type_0 = const()[name = string("op_1827_pad_type_0"), val = string("valid")]; tensor var_1827_strides_0 = const()[name = string("op_1827_strides_0"), val = tensor([1, 1])]; tensor var_1827_pad_0 = const()[name = string("op_1827_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1827_dilations_0 = const()[name = string("op_1827_dilations_0"), val = tensor([1, 1])]; int32 var_1827_groups_0 = const()[name = string("op_1827_groups_0"), val = int32(1)]; tensor var_1827 = conv(dilations = var_1827_dilations_0, groups = var_1827_groups_0, pad = var_1827_pad_0, pad_type = var_1827_pad_type_0, strides = var_1827_strides_0, weight = encoder_layers_7_self_attn_k_proj_weight_quantized, x = var_1810_cast_fp16)[name = string("op_1827")]; tensor var_1828 = const()[name = string("op_1828"), val = tensor([1, 1, 256, 256])]; tensor var_1829 = reshape(shape = var_1828, x = var_1827)[name = string("op_1829")]; tensor var_1830 = const()[name = string("op_1830"), val = tensor([0, 1, 3, 2])]; string var_1837_pad_type_0 = const()[name = string("op_1837_pad_type_0"), val = string("valid")]; tensor var_1837_strides_0 = const()[name = string("op_1837_strides_0"), val = tensor([1, 1])]; tensor var_1837_pad_0 = const()[name = string("op_1837_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1837_dilations_0 = const()[name = string("op_1837_dilations_0"), val = tensor([1, 1])]; int32 var_1837_groups_0 = const()[name = string("op_1837_groups_0"), val = int32(1)]; tensor var_1837 = conv(dilations = var_1837_dilations_0, groups = var_1837_groups_0, pad = var_1837_pad_0, pad_type = var_1837_pad_type_0, strides = var_1837_strides_0, weight = encoder_layers_7_self_attn_v_proj_weight_quantized, x = var_1810_cast_fp16)[name = string("op_1837")]; tensor var_1838 = const()[name = string("op_1838"), val = tensor([1, 1, 256, 256])]; tensor var_1839 = reshape(shape = var_1838, x = var_1837)[name = string("op_1839")]; tensor var_1840 = const()[name = string("op_1840"), val = tensor([0, 1, 3, 2])]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_43 = transpose(perm = var_1820, x = var_1819)[name = string("transpose_151")]; tensor var_1846_cast_fp16 = mul(x = q_43, y = const_100_promoted_to_fp16)[name = string("op_1846_cast_fp16")]; bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; tensor input_145_cast_fp16 = concat(axis = var_22, interleave = input_145_interleave_0, values = (q_43, var_1846_cast_fp16))[name = string("input_145_cast_fp16")]; tensor normed_203_axes_0 = const()[name = string("normed_203_axes_0"), val = tensor([-1])]; tensor normed_203_cast_fp16 = layer_norm(axes = normed_203_axes_0, epsilon = var_8_to_fp16, x = input_145_cast_fp16)[name = string("normed_203_cast_fp16")]; tensor var_1851_split_sizes_0 = const()[name = string("op_1851_split_sizes_0"), val = tensor([256, 256])]; int32 var_1851_axis_0 = const()[name = string("op_1851_axis_0"), val = int32(-1)]; tensor var_1851_cast_fp16_0, tensor var_1851_cast_fp16_1 = split(axis = var_1851_axis_0, split_sizes = var_1851_split_sizes_0, x = normed_203_cast_fp16)[name = string("op_1851_cast_fp16")]; tensor var_1855_to_fp16 = const()[name = string("op_1855_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298563968)))]; tensor out_87_cast_fp16 = mul(x = var_1851_cast_fp16_0, y = var_1855_to_fp16)[name = string("out_87_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_43 = transpose(perm = var_1830, x = var_1829)[name = string("transpose_150")]; tensor var_1862_cast_fp16 = mul(x = k_43, y = const_102_promoted_to_fp16)[name = string("op_1862_cast_fp16")]; bool input_147_interleave_0 = const()[name = string("input_147_interleave_0"), val = bool(false)]; tensor input_147_cast_fp16 = concat(axis = var_22, interleave = input_147_interleave_0, values = (k_43, var_1862_cast_fp16))[name = string("input_147_cast_fp16")]; tensor normed_207_axes_0 = const()[name = string("normed_207_axes_0"), val = tensor([-1])]; tensor normed_207_cast_fp16 = layer_norm(axes = normed_207_axes_0, epsilon = var_8_to_fp16, x = input_147_cast_fp16)[name = string("normed_207_cast_fp16")]; tensor var_1867_split_sizes_0 = const()[name = string("op_1867_split_sizes_0"), val = tensor([256, 256])]; int32 var_1867_axis_0 = const()[name = string("op_1867_axis_0"), val = int32(-1)]; tensor var_1867_cast_fp16_0, tensor var_1867_cast_fp16_1 = split(axis = var_1867_axis_0, split_sizes = var_1867_split_sizes_0, x = normed_207_cast_fp16)[name = string("op_1867_cast_fp16")]; tensor var_1871_to_fp16 = const()[name = string("op_1871_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298564544)))]; tensor out_89_cast_fp16 = mul(x = var_1867_cast_fp16_0, y = var_1871_to_fp16)[name = string("out_89_cast_fp16")]; tensor var_1874 = mul(x = out_87_cast_fp16, y = cos_1_quantized)[name = string("op_1874")]; tensor var_1875_split_sizes_0 = const()[name = string("op_1875_split_sizes_0"), val = tensor([128, 128])]; int32 var_1875_axis_0 = const()[name = string("op_1875_axis_0"), val = int32(-1)]; tensor var_1875_0, tensor var_1875_1 = split(axis = var_1875_axis_0, split_sizes = var_1875_split_sizes_0, x = out_87_cast_fp16)[name = string("op_1875")]; fp16 const_104_promoted = const()[name = string("const_104_promoted"), val = fp16(-0x1p+0)]; tensor var_1877 = mul(x = var_1875_1, y = const_104_promoted)[name = string("op_1877")]; bool var_1879_interleave_0 = const()[name = string("op_1879_interleave_0"), val = bool(false)]; tensor var_1879 = concat(axis = var_22, interleave = var_1879_interleave_0, values = (var_1877, var_1875_0))[name = string("op_1879")]; tensor var_1880 = mul(x = var_1879, y = sin_1_quantized)[name = string("op_1880")]; tensor q_47 = add(x = var_1874, y = var_1880)[name = string("q_47")]; tensor var_1882 = mul(x = out_89_cast_fp16, y = cos_1_quantized)[name = string("op_1882")]; tensor var_1883_split_sizes_0 = const()[name = string("op_1883_split_sizes_0"), val = tensor([128, 128])]; int32 var_1883_axis_0 = const()[name = string("op_1883_axis_0"), val = int32(-1)]; tensor var_1883_0, tensor var_1883_1 = split(axis = var_1883_axis_0, split_sizes = var_1883_split_sizes_0, x = out_89_cast_fp16)[name = string("op_1883")]; fp16 const_105_promoted = const()[name = string("const_105_promoted"), val = fp16(-0x1p+0)]; tensor var_1885 = mul(x = var_1883_1, y = const_105_promoted)[name = string("op_1885")]; bool var_1887_interleave_0 = const()[name = string("op_1887_interleave_0"), val = bool(false)]; tensor var_1887 = concat(axis = var_22, interleave = var_1887_interleave_0, values = (var_1885, var_1883_0))[name = string("op_1887")]; tensor var_1888 = mul(x = var_1887, y = sin_1_quantized)[name = string("op_1888")]; tensor hidden_states_85 = add(x = var_1882, y = var_1888)[name = string("hidden_states_85")]; tensor hidden_states_87_axes_0 = const()[name = string("hidden_states_87_axes_0"), val = tensor([2])]; tensor hidden_states_87 = expand_dims(axes = hidden_states_87_axes_0, x = hidden_states_85)[name = string("hidden_states_87")]; tensor var_1891 = const()[name = string("op_1891"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_89 = tile(reps = var_1891, x = hidden_states_87)[name = string("hidden_states_89")]; tensor var_1893 = const()[name = string("op_1893"), val = tensor([1, 3, 256, 256])]; tensor k_47 = reshape(shape = var_1893, x = hidden_states_89)[name = string("k_47")]; tensor hidden_states_93_axes_0 = const()[name = string("hidden_states_93_axes_0"), val = tensor([2])]; tensor hidden_states_91 = transpose(perm = var_1840, x = var_1839)[name = string("transpose_149")]; tensor hidden_states_93 = expand_dims(axes = hidden_states_93_axes_0, x = hidden_states_91)[name = string("hidden_states_93")]; tensor var_1896 = const()[name = string("op_1896"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_95 = tile(reps = var_1896, x = hidden_states_93)[name = string("hidden_states_95")]; tensor var_1898 = const()[name = string("op_1898"), val = tensor([1, 3, 256, 256])]; tensor v_15 = reshape(shape = var_1898, x = hidden_states_95)[name = string("v_15")]; bool var_1903_transpose_x_1 = const()[name = string("op_1903_transpose_x_1"), val = bool(false)]; bool var_1903_transpose_y_1 = const()[name = string("op_1903_transpose_y_1"), val = bool(true)]; tensor var_1903_cast_fp16 = matmul(transpose_x = var_1903_transpose_x_1, transpose_y = var_1903_transpose_y_1, x = q_47, y = k_47)[name = string("op_1903_cast_fp16")]; fp16 var_1904_to_fp16 = const()[name = string("op_1904_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_43_cast_fp16 = mul(x = var_1903_cast_fp16, y = var_1904_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_45_cast_fp16")]; tensor var_1908_cast_fp16 = softmax(axis = var_22, x = attn_weights_45_cast_fp16)[name = string("op_1908_cast_fp16")]; bool var_1912_transpose_x_0 = const()[name = string("op_1912_transpose_x_0"), val = bool(false)]; bool var_1912_transpose_y_0 = const()[name = string("op_1912_transpose_y_0"), val = bool(false)]; tensor var_1912_cast_fp16 = matmul(transpose_x = var_1912_transpose_x_0, transpose_y = var_1912_transpose_y_0, x = var_1908_cast_fp16, y = v_15)[name = string("op_1912_cast_fp16")]; tensor var_1914 = const()[name = string("op_1914"), val = tensor([0, 2, 1, 3])]; tensor var_1917 = const()[name = string("op_1917"), val = tensor([1, 256, 768])]; tensor var_1915 = transpose(perm = var_1914, x = var_1912_cast_fp16)[name = string("transpose_148")]; tensor attn_out_45 = reshape(shape = var_1917, x = var_1915)[name = string("attn_out_45")]; tensor var_1919 = const()[name = string("op_1919"), val = tensor([0, 2, 1])]; tensor squeeze_7_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(298565120))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299155008))))[name = string("squeeze_7_quantized")]; string var_1928_pad_type_0 = const()[name = string("op_1928_pad_type_0"), val = string("valid")]; int32 var_1928_groups_0 = const()[name = string("op_1928_groups_0"), val = int32(1)]; tensor var_1928_strides_0 = const()[name = string("op_1928_strides_0"), val = tensor([1])]; tensor var_1928_pad_0 = const()[name = string("op_1928_pad_0"), val = tensor([0, 0])]; tensor var_1928_dilations_0 = const()[name = string("op_1928_dilations_0"), val = tensor([1])]; tensor var_1920 = transpose(perm = var_1919, x = attn_out_45)[name = string("transpose_147")]; tensor var_1928 = conv(dilations = var_1928_dilations_0, groups = var_1928_groups_0, pad = var_1928_pad_0, pad_type = var_1928_pad_type_0, strides = var_1928_strides_0, weight = squeeze_7_quantized, x = var_1920)[name = string("op_1928")]; tensor var_1929 = const()[name = string("op_1929"), val = tensor([0, 2, 1])]; fp16 const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_121 = transpose(perm = var_1929, x = var_1928)[name = string("transpose_146")]; tensor var_1933_cast_fp16 = mul(x = x_121, y = const_106_promoted_to_fp16)[name = string("op_1933_cast_fp16")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151_cast_fp16 = concat(axis = var_22, interleave = input_151_interleave_0, values = (x_121, var_1933_cast_fp16))[name = string("input_151_cast_fp16")]; tensor normed_211_axes_0 = const()[name = string("normed_211_axes_0"), val = tensor([-1])]; tensor normed_211_cast_fp16 = layer_norm(axes = normed_211_axes_0, epsilon = var_8_to_fp16, x = input_151_cast_fp16)[name = string("normed_211_cast_fp16")]; tensor var_1938_split_sizes_0 = const()[name = string("op_1938_split_sizes_0"), val = tensor([768, 768])]; int32 var_1938_axis_0 = const()[name = string("op_1938_axis_0"), val = int32(-1)]; tensor var_1938_cast_fp16_0, tensor var_1938_cast_fp16_1 = split(axis = var_1938_axis_0, split_sizes = var_1938_split_sizes_0, x = normed_211_cast_fp16)[name = string("op_1938_cast_fp16")]; tensor var_1942_to_fp16 = const()[name = string("op_1942_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299156608)))]; tensor out_91_cast_fp16 = mul(x = var_1938_cast_fp16_0, y = var_1942_to_fp16)[name = string("out_91_cast_fp16")]; tensor x_123_cast_fp16 = add(x = x_113_cast_fp16, y = out_91_cast_fp16)[name = string("x_123_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1949_cast_fp16 = mul(x = x_123_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_1949_cast_fp16")]; bool input_153_interleave_0 = const()[name = string("input_153_interleave_0"), val = bool(false)]; tensor input_153_cast_fp16 = concat(axis = var_22, interleave = input_153_interleave_0, values = (x_123_cast_fp16, var_1949_cast_fp16))[name = string("input_153_cast_fp16")]; tensor normed_215_axes_0 = const()[name = string("normed_215_axes_0"), val = tensor([-1])]; tensor normed_215_cast_fp16 = layer_norm(axes = normed_215_axes_0, epsilon = var_8_to_fp16, x = input_153_cast_fp16)[name = string("normed_215_cast_fp16")]; tensor var_1954_split_sizes_0 = const()[name = string("op_1954_split_sizes_0"), val = tensor([768, 768])]; int32 var_1954_axis_0 = const()[name = string("op_1954_axis_0"), val = int32(-1)]; tensor var_1954_cast_fp16_0, tensor var_1954_cast_fp16_1 = split(axis = var_1954_axis_0, split_sizes = var_1954_split_sizes_0, x = normed_215_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_to_fp16 = const()[name = string("op_1958_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299158208)))]; tensor out_93_cast_fp16 = mul(x = var_1954_cast_fp16_0, y = var_1958_to_fp16)[name = string("out_93_cast_fp16")]; tensor var_1965 = const()[name = string("op_1965"), val = tensor([0, 2, 1])]; tensor input_155_axes_0 = const()[name = string("input_155_axes_0"), val = tensor([2])]; tensor var_1966 = transpose(perm = var_1965, x = out_93_cast_fp16)[name = string("transpose_145")]; tensor input_155 = expand_dims(axes = input_155_axes_0, x = var_1966)[name = string("input_155")]; string gate_29_pad_type_0 = const()[name = string("gate_29_pad_type_0"), val = string("valid")]; tensor gate_29_strides_0 = const()[name = string("gate_29_strides_0"), val = tensor([1, 1])]; tensor gate_29_pad_0 = const()[name = string("gate_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_29_dilations_0 = const()[name = string("gate_29_dilations_0"), val = tensor([1, 1])]; int32 gate_29_groups_0 = const()[name = string("gate_29_groups_0"), val = int32(1)]; tensor gate_29 = conv(dilations = gate_29_dilations_0, groups = gate_29_groups_0, pad = gate_29_pad_0, pad_type = gate_29_pad_type_0, strides = gate_29_strides_0, weight = encoder_layers_7_mlp_gate_proj_weight_quantized, x = input_155)[name = string("gate_29")]; string up_15_pad_type_0 = const()[name = string("up_15_pad_type_0"), val = string("valid")]; tensor up_15_strides_0 = const()[name = string("up_15_strides_0"), val = tensor([1, 1])]; tensor up_15_pad_0 = const()[name = string("up_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_15_dilations_0 = const()[name = string("up_15_dilations_0"), val = tensor([1, 1])]; int32 up_15_groups_0 = const()[name = string("up_15_groups_0"), val = int32(1)]; tensor up_15 = conv(dilations = up_15_dilations_0, groups = up_15_groups_0, pad = up_15_pad_0, pad_type = up_15_pad_type_0, strides = up_15_strides_0, weight = encoder_layers_7_mlp_up_proj_weight_quantized, x = input_155)[name = string("up_15")]; string gate_31_mode_0 = const()[name = string("gate_31_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_31 = gelu(mode = gate_31_mode_0, x = gate_29)[name = string("gate_31")]; tensor input_157 = mul(x = gate_31, y = up_15)[name = string("input_157")]; string var_1987_pad_type_0 = const()[name = string("op_1987_pad_type_0"), val = string("valid")]; tensor var_1987_strides_0 = const()[name = string("op_1987_strides_0"), val = tensor([1, 1])]; tensor var_1987_pad_0 = const()[name = string("op_1987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1987_dilations_0 = const()[name = string("op_1987_dilations_0"), val = tensor([1, 1])]; int32 var_1987_groups_0 = const()[name = string("op_1987_groups_0"), val = int32(1)]; tensor var_1987 = conv(dilations = var_1987_dilations_0, groups = var_1987_groups_0, pad = var_1987_pad_0, pad_type = var_1987_pad_type_0, strides = var_1987_strides_0, weight = encoder_layers_7_mlp_down_proj_weight_quantized, x = input_157)[name = string("op_1987")]; tensor var_1988_axes_0 = const()[name = string("op_1988_axes_0"), val = tensor([2])]; tensor var_1988 = squeeze(axes = var_1988_axes_0, x = var_1987)[name = string("op_1988")]; tensor var_1989 = const()[name = string("op_1989"), val = tensor([0, 2, 1])]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_127 = transpose(perm = var_1989, x = var_1988)[name = string("transpose_144")]; tensor var_1993_cast_fp16 = mul(x = x_127, y = const_110_promoted_to_fp16)[name = string("op_1993_cast_fp16")]; bool input_159_interleave_0 = const()[name = string("input_159_interleave_0"), val = bool(false)]; tensor input_159_cast_fp16 = concat(axis = var_22, interleave = input_159_interleave_0, values = (x_127, var_1993_cast_fp16))[name = string("input_159_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_8_to_fp16, x = input_159_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor var_1998_split_sizes_0 = const()[name = string("op_1998_split_sizes_0"), val = tensor([768, 768])]; int32 var_1998_axis_0 = const()[name = string("op_1998_axis_0"), val = int32(-1)]; tensor var_1998_cast_fp16_0, tensor var_1998_cast_fp16_1 = split(axis = var_1998_axis_0, split_sizes = var_1998_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_1998_cast_fp16")]; tensor var_2002_to_fp16 = const()[name = string("op_2002_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299159808)))]; tensor out_95_cast_fp16 = mul(x = var_1998_cast_fp16_0, y = var_2002_to_fp16)[name = string("out_95_cast_fp16")]; tensor x_129_cast_fp16 = add(x = x_123_cast_fp16, y = out_95_cast_fp16)[name = string("x_129_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2031_cast_fp16 = mul(x = x_129_cast_fp16, y = const_112_promoted_to_fp16)[name = string("op_2031_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_22, interleave = input_161_interleave_0, values = (x_129_cast_fp16, var_2031_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_8_to_fp16, x = input_161_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_2036_split_sizes_0 = const()[name = string("op_2036_split_sizes_0"), val = tensor([768, 768])]; int32 var_2036_axis_0 = const()[name = string("op_2036_axis_0"), val = int32(-1)]; tensor var_2036_cast_fp16_0, tensor var_2036_cast_fp16_1 = split(axis = var_2036_axis_0, split_sizes = var_2036_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_2036_cast_fp16")]; tensor var_2040_to_fp16 = const()[name = string("op_2040_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299161408)))]; tensor out_97_cast_fp16 = mul(x = var_2036_cast_fp16_0, y = var_2040_to_fp16)[name = string("out_97_cast_fp16")]; tensor var_2046 = const()[name = string("op_2046"), val = tensor([0, 2, 1])]; tensor var_2048_axes_0 = const()[name = string("op_2048_axes_0"), val = tensor([2])]; tensor var_2047_cast_fp16 = transpose(perm = var_2046, x = out_97_cast_fp16)[name = string("transpose_143")]; tensor var_2048_cast_fp16 = expand_dims(axes = var_2048_axes_0, x = var_2047_cast_fp16)[name = string("op_2048_cast_fp16")]; string var_2055_pad_type_0 = const()[name = string("op_2055_pad_type_0"), val = string("valid")]; tensor var_2055_strides_0 = const()[name = string("op_2055_strides_0"), val = tensor([1, 1])]; tensor var_2055_pad_0 = const()[name = string("op_2055_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2055_dilations_0 = const()[name = string("op_2055_dilations_0"), val = tensor([1, 1])]; int32 var_2055_groups_0 = const()[name = string("op_2055_groups_0"), val = int32(1)]; tensor var_2055 = conv(dilations = var_2055_dilations_0, groups = var_2055_groups_0, pad = var_2055_pad_0, pad_type = var_2055_pad_type_0, strides = var_2055_strides_0, weight = encoder_layers_8_self_attn_q_proj_weight_quantized, x = var_2048_cast_fp16)[name = string("op_2055")]; tensor var_2056 = const()[name = string("op_2056"), val = tensor([1, 3, 256, 256])]; tensor var_2057 = reshape(shape = var_2056, x = var_2055)[name = string("op_2057")]; tensor var_2058 = const()[name = string("op_2058"), val = tensor([0, 1, 3, 2])]; string var_2065_pad_type_0 = const()[name = string("op_2065_pad_type_0"), val = string("valid")]; tensor var_2065_strides_0 = const()[name = string("op_2065_strides_0"), val = tensor([1, 1])]; tensor var_2065_pad_0 = const()[name = string("op_2065_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2065_dilations_0 = const()[name = string("op_2065_dilations_0"), val = tensor([1, 1])]; int32 var_2065_groups_0 = const()[name = string("op_2065_groups_0"), val = int32(1)]; tensor var_2065 = conv(dilations = var_2065_dilations_0, groups = var_2065_groups_0, pad = var_2065_pad_0, pad_type = var_2065_pad_type_0, strides = var_2065_strides_0, weight = encoder_layers_8_self_attn_k_proj_weight_quantized, x = var_2048_cast_fp16)[name = string("op_2065")]; tensor var_2066 = const()[name = string("op_2066"), val = tensor([1, 1, 256, 256])]; tensor var_2067 = reshape(shape = var_2066, x = var_2065)[name = string("op_2067")]; tensor var_2068 = const()[name = string("op_2068"), val = tensor([0, 1, 3, 2])]; string var_2075_pad_type_0 = const()[name = string("op_2075_pad_type_0"), val = string("valid")]; tensor var_2075_strides_0 = const()[name = string("op_2075_strides_0"), val = tensor([1, 1])]; tensor var_2075_pad_0 = const()[name = string("op_2075_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2075_dilations_0 = const()[name = string("op_2075_dilations_0"), val = tensor([1, 1])]; int32 var_2075_groups_0 = const()[name = string("op_2075_groups_0"), val = int32(1)]; tensor var_2075 = conv(dilations = var_2075_dilations_0, groups = var_2075_groups_0, pad = var_2075_pad_0, pad_type = var_2075_pad_type_0, strides = var_2075_strides_0, weight = encoder_layers_8_self_attn_v_proj_weight_quantized, x = var_2048_cast_fp16)[name = string("op_2075")]; tensor var_2076 = const()[name = string("op_2076"), val = tensor([1, 1, 256, 256])]; tensor var_2077 = reshape(shape = var_2076, x = var_2075)[name = string("op_2077")]; tensor var_2078 = const()[name = string("op_2078"), val = tensor([0, 1, 3, 2])]; fp16 const_114_promoted_to_fp16 = const()[name = string("const_114_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_49 = transpose(perm = var_2058, x = var_2057)[name = string("transpose_142")]; tensor var_2084_cast_fp16 = mul(x = q_49, y = const_114_promoted_to_fp16)[name = string("op_2084_cast_fp16")]; bool input_165_interleave_0 = const()[name = string("input_165_interleave_0"), val = bool(false)]; tensor input_165_cast_fp16 = concat(axis = var_22, interleave = input_165_interleave_0, values = (q_49, var_2084_cast_fp16))[name = string("input_165_cast_fp16")]; tensor normed_231_axes_0 = const()[name = string("normed_231_axes_0"), val = tensor([-1])]; tensor normed_231_cast_fp16 = layer_norm(axes = normed_231_axes_0, epsilon = var_8_to_fp16, x = input_165_cast_fp16)[name = string("normed_231_cast_fp16")]; tensor var_2089_split_sizes_0 = const()[name = string("op_2089_split_sizes_0"), val = tensor([256, 256])]; int32 var_2089_axis_0 = const()[name = string("op_2089_axis_0"), val = int32(-1)]; tensor var_2089_cast_fp16_0, tensor var_2089_cast_fp16_1 = split(axis = var_2089_axis_0, split_sizes = var_2089_split_sizes_0, x = normed_231_cast_fp16)[name = string("op_2089_cast_fp16")]; tensor var_2093_to_fp16 = const()[name = string("op_2093_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299163008)))]; tensor out_99_cast_fp16 = mul(x = var_2089_cast_fp16_0, y = var_2093_to_fp16)[name = string("out_99_cast_fp16")]; fp16 const_116_promoted_to_fp16 = const()[name = string("const_116_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_49 = transpose(perm = var_2068, x = var_2067)[name = string("transpose_141")]; tensor var_2100_cast_fp16 = mul(x = k_49, y = const_116_promoted_to_fp16)[name = string("op_2100_cast_fp16")]; bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; tensor input_167_cast_fp16 = concat(axis = var_22, interleave = input_167_interleave_0, values = (k_49, var_2100_cast_fp16))[name = string("input_167_cast_fp16")]; tensor normed_235_axes_0 = const()[name = string("normed_235_axes_0"), val = tensor([-1])]; tensor normed_235_cast_fp16 = layer_norm(axes = normed_235_axes_0, epsilon = var_8_to_fp16, x = input_167_cast_fp16)[name = string("normed_235_cast_fp16")]; tensor var_2105_split_sizes_0 = const()[name = string("op_2105_split_sizes_0"), val = tensor([256, 256])]; int32 var_2105_axis_0 = const()[name = string("op_2105_axis_0"), val = int32(-1)]; tensor var_2105_cast_fp16_0, tensor var_2105_cast_fp16_1 = split(axis = var_2105_axis_0, split_sizes = var_2105_split_sizes_0, x = normed_235_cast_fp16)[name = string("op_2105_cast_fp16")]; tensor var_2109_to_fp16 = const()[name = string("op_2109_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299163584)))]; tensor out_101_cast_fp16 = mul(x = var_2105_cast_fp16_0, y = var_2109_to_fp16)[name = string("out_101_cast_fp16")]; tensor var_2112 = mul(x = out_99_cast_fp16, y = cos_1_quantized)[name = string("op_2112")]; tensor var_2113_split_sizes_0 = const()[name = string("op_2113_split_sizes_0"), val = tensor([128, 128])]; int32 var_2113_axis_0 = const()[name = string("op_2113_axis_0"), val = int32(-1)]; tensor var_2113_0, tensor var_2113_1 = split(axis = var_2113_axis_0, split_sizes = var_2113_split_sizes_0, x = out_99_cast_fp16)[name = string("op_2113")]; fp16 const_118_promoted = const()[name = string("const_118_promoted"), val = fp16(-0x1p+0)]; tensor var_2115 = mul(x = var_2113_1, y = const_118_promoted)[name = string("op_2115")]; bool var_2117_interleave_0 = const()[name = string("op_2117_interleave_0"), val = bool(false)]; tensor var_2117 = concat(axis = var_22, interleave = var_2117_interleave_0, values = (var_2115, var_2113_0))[name = string("op_2117")]; tensor var_2118 = mul(x = var_2117, y = sin_1_quantized)[name = string("op_2118")]; tensor q_53 = add(x = var_2112, y = var_2118)[name = string("q_53")]; tensor var_2120 = mul(x = out_101_cast_fp16, y = cos_1_quantized)[name = string("op_2120")]; tensor var_2121_split_sizes_0 = const()[name = string("op_2121_split_sizes_0"), val = tensor([128, 128])]; int32 var_2121_axis_0 = const()[name = string("op_2121_axis_0"), val = int32(-1)]; tensor var_2121_0, tensor var_2121_1 = split(axis = var_2121_axis_0, split_sizes = var_2121_split_sizes_0, x = out_101_cast_fp16)[name = string("op_2121")]; fp16 const_119_promoted = const()[name = string("const_119_promoted"), val = fp16(-0x1p+0)]; tensor var_2123 = mul(x = var_2121_1, y = const_119_promoted)[name = string("op_2123")]; bool var_2125_interleave_0 = const()[name = string("op_2125_interleave_0"), val = bool(false)]; tensor var_2125 = concat(axis = var_22, interleave = var_2125_interleave_0, values = (var_2123, var_2121_0))[name = string("op_2125")]; tensor var_2126 = mul(x = var_2125, y = sin_1_quantized)[name = string("op_2126")]; tensor hidden_states_97 = add(x = var_2120, y = var_2126)[name = string("hidden_states_97")]; tensor hidden_states_99_axes_0 = const()[name = string("hidden_states_99_axes_0"), val = tensor([2])]; tensor hidden_states_99 = expand_dims(axes = hidden_states_99_axes_0, x = hidden_states_97)[name = string("hidden_states_99")]; tensor var_2129 = const()[name = string("op_2129"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_101 = tile(reps = var_2129, x = hidden_states_99)[name = string("hidden_states_101")]; tensor var_2131 = const()[name = string("op_2131"), val = tensor([1, 3, 256, 256])]; tensor k_53 = reshape(shape = var_2131, x = hidden_states_101)[name = string("k_53")]; tensor hidden_states_105_axes_0 = const()[name = string("hidden_states_105_axes_0"), val = tensor([2])]; tensor hidden_states_103 = transpose(perm = var_2078, x = var_2077)[name = string("transpose_140")]; tensor hidden_states_105 = expand_dims(axes = hidden_states_105_axes_0, x = hidden_states_103)[name = string("hidden_states_105")]; tensor var_2134 = const()[name = string("op_2134"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_107 = tile(reps = var_2134, x = hidden_states_105)[name = string("hidden_states_107")]; tensor var_2136 = const()[name = string("op_2136"), val = tensor([1, 3, 256, 256])]; tensor v_17 = reshape(shape = var_2136, x = hidden_states_107)[name = string("v_17")]; bool var_2141_transpose_x_1 = const()[name = string("op_2141_transpose_x_1"), val = bool(false)]; bool var_2141_transpose_y_1 = const()[name = string("op_2141_transpose_y_1"), val = bool(true)]; tensor var_2141_cast_fp16 = matmul(transpose_x = var_2141_transpose_x_1, transpose_y = var_2141_transpose_y_1, x = q_53, y = k_53)[name = string("op_2141_cast_fp16")]; fp16 var_2142_to_fp16 = const()[name = string("op_2142_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_49_cast_fp16 = mul(x = var_2141_cast_fp16, y = var_2142_to_fp16)[name = string("attn_weights_49_cast_fp16")]; tensor attn_weights_51_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor var_2146_cast_fp16 = softmax(axis = var_22, x = attn_weights_51_cast_fp16)[name = string("op_2146_cast_fp16")]; bool var_2150_transpose_x_0 = const()[name = string("op_2150_transpose_x_0"), val = bool(false)]; bool var_2150_transpose_y_0 = const()[name = string("op_2150_transpose_y_0"), val = bool(false)]; tensor var_2150_cast_fp16 = matmul(transpose_x = var_2150_transpose_x_0, transpose_y = var_2150_transpose_y_0, x = var_2146_cast_fp16, y = v_17)[name = string("op_2150_cast_fp16")]; tensor var_2152 = const()[name = string("op_2152"), val = tensor([0, 2, 1, 3])]; tensor var_2155 = const()[name = string("op_2155"), val = tensor([1, 256, 768])]; tensor var_2153 = transpose(perm = var_2152, x = var_2150_cast_fp16)[name = string("transpose_139")]; tensor attn_out_51 = reshape(shape = var_2155, x = var_2153)[name = string("attn_out_51")]; tensor var_2157 = const()[name = string("op_2157"), val = tensor([0, 2, 1])]; tensor squeeze_8_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299164160))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299754048))))[name = string("squeeze_8_quantized")]; string var_2166_pad_type_0 = const()[name = string("op_2166_pad_type_0"), val = string("valid")]; int32 var_2166_groups_0 = const()[name = string("op_2166_groups_0"), val = int32(1)]; tensor var_2166_strides_0 = const()[name = string("op_2166_strides_0"), val = tensor([1])]; tensor var_2166_pad_0 = const()[name = string("op_2166_pad_0"), val = tensor([0, 0])]; tensor var_2166_dilations_0 = const()[name = string("op_2166_dilations_0"), val = tensor([1])]; tensor var_2158 = transpose(perm = var_2157, x = attn_out_51)[name = string("transpose_138")]; tensor var_2166 = conv(dilations = var_2166_dilations_0, groups = var_2166_groups_0, pad = var_2166_pad_0, pad_type = var_2166_pad_type_0, strides = var_2166_strides_0, weight = squeeze_8_quantized, x = var_2158)[name = string("op_2166")]; tensor var_2167 = const()[name = string("op_2167"), val = tensor([0, 2, 1])]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_137 = transpose(perm = var_2167, x = var_2166)[name = string("transpose_137")]; tensor var_2171_cast_fp16 = mul(x = x_137, y = const_120_promoted_to_fp16)[name = string("op_2171_cast_fp16")]; bool input_171_interleave_0 = const()[name = string("input_171_interleave_0"), val = bool(false)]; tensor input_171_cast_fp16 = concat(axis = var_22, interleave = input_171_interleave_0, values = (x_137, var_2171_cast_fp16))[name = string("input_171_cast_fp16")]; tensor normed_239_axes_0 = const()[name = string("normed_239_axes_0"), val = tensor([-1])]; tensor normed_239_cast_fp16 = layer_norm(axes = normed_239_axes_0, epsilon = var_8_to_fp16, x = input_171_cast_fp16)[name = string("normed_239_cast_fp16")]; tensor var_2176_split_sizes_0 = const()[name = string("op_2176_split_sizes_0"), val = tensor([768, 768])]; int32 var_2176_axis_0 = const()[name = string("op_2176_axis_0"), val = int32(-1)]; tensor var_2176_cast_fp16_0, tensor var_2176_cast_fp16_1 = split(axis = var_2176_axis_0, split_sizes = var_2176_split_sizes_0, x = normed_239_cast_fp16)[name = string("op_2176_cast_fp16")]; tensor var_2180_to_fp16 = const()[name = string("op_2180_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299755648)))]; tensor out_103_cast_fp16 = mul(x = var_2176_cast_fp16_0, y = var_2180_to_fp16)[name = string("out_103_cast_fp16")]; tensor x_139_cast_fp16 = add(x = x_129_cast_fp16, y = out_103_cast_fp16)[name = string("x_139_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2187_cast_fp16 = mul(x = x_139_cast_fp16, y = const_122_promoted_to_fp16)[name = string("op_2187_cast_fp16")]; bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; tensor input_173_cast_fp16 = concat(axis = var_22, interleave = input_173_interleave_0, values = (x_139_cast_fp16, var_2187_cast_fp16))[name = string("input_173_cast_fp16")]; tensor normed_243_axes_0 = const()[name = string("normed_243_axes_0"), val = tensor([-1])]; tensor normed_243_cast_fp16 = layer_norm(axes = normed_243_axes_0, epsilon = var_8_to_fp16, x = input_173_cast_fp16)[name = string("normed_243_cast_fp16")]; tensor var_2192_split_sizes_0 = const()[name = string("op_2192_split_sizes_0"), val = tensor([768, 768])]; int32 var_2192_axis_0 = const()[name = string("op_2192_axis_0"), val = int32(-1)]; tensor var_2192_cast_fp16_0, tensor var_2192_cast_fp16_1 = split(axis = var_2192_axis_0, split_sizes = var_2192_split_sizes_0, x = normed_243_cast_fp16)[name = string("op_2192_cast_fp16")]; tensor var_2196_to_fp16 = const()[name = string("op_2196_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299757248)))]; tensor out_105_cast_fp16 = mul(x = var_2192_cast_fp16_0, y = var_2196_to_fp16)[name = string("out_105_cast_fp16")]; tensor var_2203 = const()[name = string("op_2203"), val = tensor([0, 2, 1])]; tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; tensor var_2204 = transpose(perm = var_2203, x = out_105_cast_fp16)[name = string("transpose_136")]; tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_2204)[name = string("input_175")]; string gate_33_pad_type_0 = const()[name = string("gate_33_pad_type_0"), val = string("valid")]; tensor gate_33_strides_0 = const()[name = string("gate_33_strides_0"), val = tensor([1, 1])]; tensor gate_33_pad_0 = const()[name = string("gate_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_33_dilations_0 = const()[name = string("gate_33_dilations_0"), val = tensor([1, 1])]; int32 gate_33_groups_0 = const()[name = string("gate_33_groups_0"), val = int32(1)]; tensor gate_33 = conv(dilations = gate_33_dilations_0, groups = gate_33_groups_0, pad = gate_33_pad_0, pad_type = gate_33_pad_type_0, strides = gate_33_strides_0, weight = encoder_layers_8_mlp_gate_proj_weight_quantized, x = input_175)[name = string("gate_33")]; string up_17_pad_type_0 = const()[name = string("up_17_pad_type_0"), val = string("valid")]; tensor up_17_strides_0 = const()[name = string("up_17_strides_0"), val = tensor([1, 1])]; tensor up_17_pad_0 = const()[name = string("up_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_17_dilations_0 = const()[name = string("up_17_dilations_0"), val = tensor([1, 1])]; int32 up_17_groups_0 = const()[name = string("up_17_groups_0"), val = int32(1)]; tensor up_17 = conv(dilations = up_17_dilations_0, groups = up_17_groups_0, pad = up_17_pad_0, pad_type = up_17_pad_type_0, strides = up_17_strides_0, weight = encoder_layers_8_mlp_up_proj_weight_quantized, x = input_175)[name = string("up_17")]; string gate_35_mode_0 = const()[name = string("gate_35_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_35 = gelu(mode = gate_35_mode_0, x = gate_33)[name = string("gate_35")]; tensor input_177 = mul(x = gate_35, y = up_17)[name = string("input_177")]; string var_2225_pad_type_0 = const()[name = string("op_2225_pad_type_0"), val = string("valid")]; tensor var_2225_strides_0 = const()[name = string("op_2225_strides_0"), val = tensor([1, 1])]; tensor var_2225_pad_0 = const()[name = string("op_2225_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2225_dilations_0 = const()[name = string("op_2225_dilations_0"), val = tensor([1, 1])]; int32 var_2225_groups_0 = const()[name = string("op_2225_groups_0"), val = int32(1)]; tensor var_2225 = conv(dilations = var_2225_dilations_0, groups = var_2225_groups_0, pad = var_2225_pad_0, pad_type = var_2225_pad_type_0, strides = var_2225_strides_0, weight = encoder_layers_8_mlp_down_proj_weight_quantized, x = input_177)[name = string("op_2225")]; tensor var_2226_axes_0 = const()[name = string("op_2226_axes_0"), val = tensor([2])]; tensor var_2226 = squeeze(axes = var_2226_axes_0, x = var_2225)[name = string("op_2226")]; tensor var_2227 = const()[name = string("op_2227"), val = tensor([0, 2, 1])]; fp16 const_124_promoted_to_fp16 = const()[name = string("const_124_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_143 = transpose(perm = var_2227, x = var_2226)[name = string("transpose_135")]; tensor var_2231_cast_fp16 = mul(x = x_143, y = const_124_promoted_to_fp16)[name = string("op_2231_cast_fp16")]; bool input_179_interleave_0 = const()[name = string("input_179_interleave_0"), val = bool(false)]; tensor input_179_cast_fp16 = concat(axis = var_22, interleave = input_179_interleave_0, values = (x_143, var_2231_cast_fp16))[name = string("input_179_cast_fp16")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_8_to_fp16, x = input_179_cast_fp16)[name = string("normed_249_cast_fp16")]; tensor var_2236_split_sizes_0 = const()[name = string("op_2236_split_sizes_0"), val = tensor([768, 768])]; int32 var_2236_axis_0 = const()[name = string("op_2236_axis_0"), val = int32(-1)]; tensor var_2236_cast_fp16_0, tensor var_2236_cast_fp16_1 = split(axis = var_2236_axis_0, split_sizes = var_2236_split_sizes_0, x = normed_249_cast_fp16)[name = string("op_2236_cast_fp16")]; tensor var_2240_to_fp16 = const()[name = string("op_2240_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299758848)))]; tensor out_107_cast_fp16 = mul(x = var_2236_cast_fp16_0, y = var_2240_to_fp16)[name = string("out_107_cast_fp16")]; tensor x_145_cast_fp16 = add(x = x_139_cast_fp16, y = out_107_cast_fp16)[name = string("x_145_cast_fp16")]; fp16 const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2269_cast_fp16 = mul(x = x_145_cast_fp16, y = const_126_promoted_to_fp16)[name = string("op_2269_cast_fp16")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181_cast_fp16 = concat(axis = var_22, interleave = input_181_interleave_0, values = (x_145_cast_fp16, var_2269_cast_fp16))[name = string("input_181_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_8_to_fp16, x = input_181_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor var_2274_split_sizes_0 = const()[name = string("op_2274_split_sizes_0"), val = tensor([768, 768])]; int32 var_2274_axis_0 = const()[name = string("op_2274_axis_0"), val = int32(-1)]; tensor var_2274_cast_fp16_0, tensor var_2274_cast_fp16_1 = split(axis = var_2274_axis_0, split_sizes = var_2274_split_sizes_0, x = normed_253_cast_fp16)[name = string("op_2274_cast_fp16")]; tensor var_2278_to_fp16 = const()[name = string("op_2278_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299760448)))]; tensor out_109_cast_fp16 = mul(x = var_2274_cast_fp16_0, y = var_2278_to_fp16)[name = string("out_109_cast_fp16")]; tensor var_2284 = const()[name = string("op_2284"), val = tensor([0, 2, 1])]; tensor var_2286_axes_0 = const()[name = string("op_2286_axes_0"), val = tensor([2])]; tensor var_2285_cast_fp16 = transpose(perm = var_2284, x = out_109_cast_fp16)[name = string("transpose_134")]; tensor var_2286_cast_fp16 = expand_dims(axes = var_2286_axes_0, x = var_2285_cast_fp16)[name = string("op_2286_cast_fp16")]; string var_2293_pad_type_0 = const()[name = string("op_2293_pad_type_0"), val = string("valid")]; tensor var_2293_strides_0 = const()[name = string("op_2293_strides_0"), val = tensor([1, 1])]; tensor var_2293_pad_0 = const()[name = string("op_2293_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2293_dilations_0 = const()[name = string("op_2293_dilations_0"), val = tensor([1, 1])]; int32 var_2293_groups_0 = const()[name = string("op_2293_groups_0"), val = int32(1)]; tensor var_2293 = conv(dilations = var_2293_dilations_0, groups = var_2293_groups_0, pad = var_2293_pad_0, pad_type = var_2293_pad_type_0, strides = var_2293_strides_0, weight = encoder_layers_9_self_attn_q_proj_weight_quantized, x = var_2286_cast_fp16)[name = string("op_2293")]; tensor var_2294 = const()[name = string("op_2294"), val = tensor([1, 3, 256, 256])]; tensor var_2295 = reshape(shape = var_2294, x = var_2293)[name = string("op_2295")]; tensor var_2296 = const()[name = string("op_2296"), val = tensor([0, 1, 3, 2])]; string var_2303_pad_type_0 = const()[name = string("op_2303_pad_type_0"), val = string("valid")]; tensor var_2303_strides_0 = const()[name = string("op_2303_strides_0"), val = tensor([1, 1])]; tensor var_2303_pad_0 = const()[name = string("op_2303_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2303_dilations_0 = const()[name = string("op_2303_dilations_0"), val = tensor([1, 1])]; int32 var_2303_groups_0 = const()[name = string("op_2303_groups_0"), val = int32(1)]; tensor var_2303 = conv(dilations = var_2303_dilations_0, groups = var_2303_groups_0, pad = var_2303_pad_0, pad_type = var_2303_pad_type_0, strides = var_2303_strides_0, weight = encoder_layers_9_self_attn_k_proj_weight_quantized, x = var_2286_cast_fp16)[name = string("op_2303")]; tensor var_2304 = const()[name = string("op_2304"), val = tensor([1, 1, 256, 256])]; tensor var_2305 = reshape(shape = var_2304, x = var_2303)[name = string("op_2305")]; tensor var_2306 = const()[name = string("op_2306"), val = tensor([0, 1, 3, 2])]; string var_2313_pad_type_0 = const()[name = string("op_2313_pad_type_0"), val = string("valid")]; tensor var_2313_strides_0 = const()[name = string("op_2313_strides_0"), val = tensor([1, 1])]; tensor var_2313_pad_0 = const()[name = string("op_2313_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2313_dilations_0 = const()[name = string("op_2313_dilations_0"), val = tensor([1, 1])]; int32 var_2313_groups_0 = const()[name = string("op_2313_groups_0"), val = int32(1)]; tensor var_2313 = conv(dilations = var_2313_dilations_0, groups = var_2313_groups_0, pad = var_2313_pad_0, pad_type = var_2313_pad_type_0, strides = var_2313_strides_0, weight = encoder_layers_9_self_attn_v_proj_weight_quantized, x = var_2286_cast_fp16)[name = string("op_2313")]; tensor var_2314 = const()[name = string("op_2314"), val = tensor([1, 1, 256, 256])]; tensor var_2315 = reshape(shape = var_2314, x = var_2313)[name = string("op_2315")]; tensor var_2316 = const()[name = string("op_2316"), val = tensor([0, 1, 3, 2])]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_55 = transpose(perm = var_2296, x = var_2295)[name = string("transpose_133")]; tensor var_2322_cast_fp16 = mul(x = q_55, y = const_128_promoted_to_fp16)[name = string("op_2322_cast_fp16")]; bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; tensor input_185_cast_fp16 = concat(axis = var_22, interleave = input_185_interleave_0, values = (q_55, var_2322_cast_fp16))[name = string("input_185_cast_fp16")]; tensor normed_259_axes_0 = const()[name = string("normed_259_axes_0"), val = tensor([-1])]; tensor normed_259_cast_fp16 = layer_norm(axes = normed_259_axes_0, epsilon = var_8_to_fp16, x = input_185_cast_fp16)[name = string("normed_259_cast_fp16")]; tensor var_2327_split_sizes_0 = const()[name = string("op_2327_split_sizes_0"), val = tensor([256, 256])]; int32 var_2327_axis_0 = const()[name = string("op_2327_axis_0"), val = int32(-1)]; tensor var_2327_cast_fp16_0, tensor var_2327_cast_fp16_1 = split(axis = var_2327_axis_0, split_sizes = var_2327_split_sizes_0, x = normed_259_cast_fp16)[name = string("op_2327_cast_fp16")]; tensor var_2331_to_fp16 = const()[name = string("op_2331_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299762048)))]; tensor out_111_cast_fp16 = mul(x = var_2327_cast_fp16_0, y = var_2331_to_fp16)[name = string("out_111_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_55 = transpose(perm = var_2306, x = var_2305)[name = string("transpose_132")]; tensor var_2338_cast_fp16 = mul(x = k_55, y = const_130_promoted_to_fp16)[name = string("op_2338_cast_fp16")]; bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; tensor input_187_cast_fp16 = concat(axis = var_22, interleave = input_187_interleave_0, values = (k_55, var_2338_cast_fp16))[name = string("input_187_cast_fp16")]; tensor normed_263_axes_0 = const()[name = string("normed_263_axes_0"), val = tensor([-1])]; tensor normed_263_cast_fp16 = layer_norm(axes = normed_263_axes_0, epsilon = var_8_to_fp16, x = input_187_cast_fp16)[name = string("normed_263_cast_fp16")]; tensor var_2343_split_sizes_0 = const()[name = string("op_2343_split_sizes_0"), val = tensor([256, 256])]; int32 var_2343_axis_0 = const()[name = string("op_2343_axis_0"), val = int32(-1)]; tensor var_2343_cast_fp16_0, tensor var_2343_cast_fp16_1 = split(axis = var_2343_axis_0, split_sizes = var_2343_split_sizes_0, x = normed_263_cast_fp16)[name = string("op_2343_cast_fp16")]; tensor var_2347_to_fp16 = const()[name = string("op_2347_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299762624)))]; tensor out_113_cast_fp16 = mul(x = var_2343_cast_fp16_0, y = var_2347_to_fp16)[name = string("out_113_cast_fp16")]; tensor var_2350 = mul(x = out_111_cast_fp16, y = cos_1_quantized)[name = string("op_2350")]; tensor var_2351_split_sizes_0 = const()[name = string("op_2351_split_sizes_0"), val = tensor([128, 128])]; int32 var_2351_axis_0 = const()[name = string("op_2351_axis_0"), val = int32(-1)]; tensor var_2351_0, tensor var_2351_1 = split(axis = var_2351_axis_0, split_sizes = var_2351_split_sizes_0, x = out_111_cast_fp16)[name = string("op_2351")]; fp16 const_132_promoted = const()[name = string("const_132_promoted"), val = fp16(-0x1p+0)]; tensor var_2353 = mul(x = var_2351_1, y = const_132_promoted)[name = string("op_2353")]; bool var_2355_interleave_0 = const()[name = string("op_2355_interleave_0"), val = bool(false)]; tensor var_2355 = concat(axis = var_22, interleave = var_2355_interleave_0, values = (var_2353, var_2351_0))[name = string("op_2355")]; tensor var_2356 = mul(x = var_2355, y = sin_1_quantized)[name = string("op_2356")]; tensor q_59 = add(x = var_2350, y = var_2356)[name = string("q_59")]; tensor var_2358 = mul(x = out_113_cast_fp16, y = cos_1_quantized)[name = string("op_2358")]; tensor var_2359_split_sizes_0 = const()[name = string("op_2359_split_sizes_0"), val = tensor([128, 128])]; int32 var_2359_axis_0 = const()[name = string("op_2359_axis_0"), val = int32(-1)]; tensor var_2359_0, tensor var_2359_1 = split(axis = var_2359_axis_0, split_sizes = var_2359_split_sizes_0, x = out_113_cast_fp16)[name = string("op_2359")]; fp16 const_133_promoted = const()[name = string("const_133_promoted"), val = fp16(-0x1p+0)]; tensor var_2361 = mul(x = var_2359_1, y = const_133_promoted)[name = string("op_2361")]; bool var_2363_interleave_0 = const()[name = string("op_2363_interleave_0"), val = bool(false)]; tensor var_2363 = concat(axis = var_22, interleave = var_2363_interleave_0, values = (var_2361, var_2359_0))[name = string("op_2363")]; tensor var_2364 = mul(x = var_2363, y = sin_1_quantized)[name = string("op_2364")]; tensor hidden_states_109 = add(x = var_2358, y = var_2364)[name = string("hidden_states_109")]; tensor hidden_states_111_axes_0 = const()[name = string("hidden_states_111_axes_0"), val = tensor([2])]; tensor hidden_states_111 = expand_dims(axes = hidden_states_111_axes_0, x = hidden_states_109)[name = string("hidden_states_111")]; tensor var_2367 = const()[name = string("op_2367"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_113 = tile(reps = var_2367, x = hidden_states_111)[name = string("hidden_states_113")]; tensor var_2369 = const()[name = string("op_2369"), val = tensor([1, 3, 256, 256])]; tensor k_59 = reshape(shape = var_2369, x = hidden_states_113)[name = string("k_59")]; tensor hidden_states_117_axes_0 = const()[name = string("hidden_states_117_axes_0"), val = tensor([2])]; tensor hidden_states_115 = transpose(perm = var_2316, x = var_2315)[name = string("transpose_131")]; tensor hidden_states_117 = expand_dims(axes = hidden_states_117_axes_0, x = hidden_states_115)[name = string("hidden_states_117")]; tensor var_2372 = const()[name = string("op_2372"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_119 = tile(reps = var_2372, x = hidden_states_117)[name = string("hidden_states_119")]; tensor var_2374 = const()[name = string("op_2374"), val = tensor([1, 3, 256, 256])]; tensor v_19 = reshape(shape = var_2374, x = hidden_states_119)[name = string("v_19")]; bool var_2379_transpose_x_1 = const()[name = string("op_2379_transpose_x_1"), val = bool(false)]; bool var_2379_transpose_y_1 = const()[name = string("op_2379_transpose_y_1"), val = bool(true)]; tensor var_2379_cast_fp16 = matmul(transpose_x = var_2379_transpose_x_1, transpose_y = var_2379_transpose_y_1, x = q_59, y = k_59)[name = string("op_2379_cast_fp16")]; fp16 var_2380_to_fp16 = const()[name = string("op_2380_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_55_cast_fp16 = mul(x = var_2379_cast_fp16, y = var_2380_to_fp16)[name = string("attn_weights_55_cast_fp16")]; tensor attn_weights_57_cast_fp16 = add(x = attn_weights_55_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_57_cast_fp16")]; tensor var_2384_cast_fp16 = softmax(axis = var_22, x = attn_weights_57_cast_fp16)[name = string("op_2384_cast_fp16")]; bool var_2388_transpose_x_0 = const()[name = string("op_2388_transpose_x_0"), val = bool(false)]; bool var_2388_transpose_y_0 = const()[name = string("op_2388_transpose_y_0"), val = bool(false)]; tensor var_2388_cast_fp16 = matmul(transpose_x = var_2388_transpose_x_0, transpose_y = var_2388_transpose_y_0, x = var_2384_cast_fp16, y = v_19)[name = string("op_2388_cast_fp16")]; tensor var_2390 = const()[name = string("op_2390"), val = tensor([0, 2, 1, 3])]; tensor var_2393 = const()[name = string("op_2393"), val = tensor([1, 256, 768])]; tensor var_2391 = transpose(perm = var_2390, x = var_2388_cast_fp16)[name = string("transpose_130")]; tensor attn_out_57 = reshape(shape = var_2393, x = var_2391)[name = string("attn_out_57")]; tensor var_2395 = const()[name = string("op_2395"), val = tensor([0, 2, 1])]; tensor squeeze_9_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299763200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300353088))))[name = string("squeeze_9_quantized")]; string var_2404_pad_type_0 = const()[name = string("op_2404_pad_type_0"), val = string("valid")]; int32 var_2404_groups_0 = const()[name = string("op_2404_groups_0"), val = int32(1)]; tensor var_2404_strides_0 = const()[name = string("op_2404_strides_0"), val = tensor([1])]; tensor var_2404_pad_0 = const()[name = string("op_2404_pad_0"), val = tensor([0, 0])]; tensor var_2404_dilations_0 = const()[name = string("op_2404_dilations_0"), val = tensor([1])]; tensor var_2396 = transpose(perm = var_2395, x = attn_out_57)[name = string("transpose_129")]; tensor var_2404 = conv(dilations = var_2404_dilations_0, groups = var_2404_groups_0, pad = var_2404_pad_0, pad_type = var_2404_pad_type_0, strides = var_2404_strides_0, weight = squeeze_9_quantized, x = var_2396)[name = string("op_2404")]; tensor var_2405 = const()[name = string("op_2405"), val = tensor([0, 2, 1])]; fp16 const_134_promoted_to_fp16 = const()[name = string("const_134_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_153 = transpose(perm = var_2405, x = var_2404)[name = string("transpose_128")]; tensor var_2409_cast_fp16 = mul(x = x_153, y = const_134_promoted_to_fp16)[name = string("op_2409_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_22, interleave = input_191_interleave_0, values = (x_153, var_2409_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_267_axes_0 = const()[name = string("normed_267_axes_0"), val = tensor([-1])]; tensor normed_267_cast_fp16 = layer_norm(axes = normed_267_axes_0, epsilon = var_8_to_fp16, x = input_191_cast_fp16)[name = string("normed_267_cast_fp16")]; tensor var_2414_split_sizes_0 = const()[name = string("op_2414_split_sizes_0"), val = tensor([768, 768])]; int32 var_2414_axis_0 = const()[name = string("op_2414_axis_0"), val = int32(-1)]; tensor var_2414_cast_fp16_0, tensor var_2414_cast_fp16_1 = split(axis = var_2414_axis_0, split_sizes = var_2414_split_sizes_0, x = normed_267_cast_fp16)[name = string("op_2414_cast_fp16")]; tensor var_2418_to_fp16 = const()[name = string("op_2418_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300354688)))]; tensor out_115_cast_fp16 = mul(x = var_2414_cast_fp16_0, y = var_2418_to_fp16)[name = string("out_115_cast_fp16")]; tensor x_155_cast_fp16 = add(x = x_145_cast_fp16, y = out_115_cast_fp16)[name = string("x_155_cast_fp16")]; fp16 const_136_promoted_to_fp16 = const()[name = string("const_136_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2425_cast_fp16 = mul(x = x_155_cast_fp16, y = const_136_promoted_to_fp16)[name = string("op_2425_cast_fp16")]; bool input_193_interleave_0 = const()[name = string("input_193_interleave_0"), val = bool(false)]; tensor input_193_cast_fp16 = concat(axis = var_22, interleave = input_193_interleave_0, values = (x_155_cast_fp16, var_2425_cast_fp16))[name = string("input_193_cast_fp16")]; tensor normed_271_axes_0 = const()[name = string("normed_271_axes_0"), val = tensor([-1])]; tensor normed_271_cast_fp16 = layer_norm(axes = normed_271_axes_0, epsilon = var_8_to_fp16, x = input_193_cast_fp16)[name = string("normed_271_cast_fp16")]; tensor var_2430_split_sizes_0 = const()[name = string("op_2430_split_sizes_0"), val = tensor([768, 768])]; int32 var_2430_axis_0 = const()[name = string("op_2430_axis_0"), val = int32(-1)]; tensor var_2430_cast_fp16_0, tensor var_2430_cast_fp16_1 = split(axis = var_2430_axis_0, split_sizes = var_2430_split_sizes_0, x = normed_271_cast_fp16)[name = string("op_2430_cast_fp16")]; tensor var_2434_to_fp16 = const()[name = string("op_2434_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300356288)))]; tensor out_117_cast_fp16 = mul(x = var_2430_cast_fp16_0, y = var_2434_to_fp16)[name = string("out_117_cast_fp16")]; tensor var_2441 = const()[name = string("op_2441"), val = tensor([0, 2, 1])]; tensor input_195_axes_0 = const()[name = string("input_195_axes_0"), val = tensor([2])]; tensor var_2442 = transpose(perm = var_2441, x = out_117_cast_fp16)[name = string("transpose_127")]; tensor input_195 = expand_dims(axes = input_195_axes_0, x = var_2442)[name = string("input_195")]; string gate_37_pad_type_0 = const()[name = string("gate_37_pad_type_0"), val = string("valid")]; tensor gate_37_strides_0 = const()[name = string("gate_37_strides_0"), val = tensor([1, 1])]; tensor gate_37_pad_0 = const()[name = string("gate_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_37_dilations_0 = const()[name = string("gate_37_dilations_0"), val = tensor([1, 1])]; int32 gate_37_groups_0 = const()[name = string("gate_37_groups_0"), val = int32(1)]; tensor gate_37 = conv(dilations = gate_37_dilations_0, groups = gate_37_groups_0, pad = gate_37_pad_0, pad_type = gate_37_pad_type_0, strides = gate_37_strides_0, weight = encoder_layers_9_mlp_gate_proj_weight_quantized, x = input_195)[name = string("gate_37")]; string up_19_pad_type_0 = const()[name = string("up_19_pad_type_0"), val = string("valid")]; tensor up_19_strides_0 = const()[name = string("up_19_strides_0"), val = tensor([1, 1])]; tensor up_19_pad_0 = const()[name = string("up_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_19_dilations_0 = const()[name = string("up_19_dilations_0"), val = tensor([1, 1])]; int32 up_19_groups_0 = const()[name = string("up_19_groups_0"), val = int32(1)]; tensor up_19 = conv(dilations = up_19_dilations_0, groups = up_19_groups_0, pad = up_19_pad_0, pad_type = up_19_pad_type_0, strides = up_19_strides_0, weight = encoder_layers_9_mlp_up_proj_weight_quantized, x = input_195)[name = string("up_19")]; string gate_39_mode_0 = const()[name = string("gate_39_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_39 = gelu(mode = gate_39_mode_0, x = gate_37)[name = string("gate_39")]; tensor input_197 = mul(x = gate_39, y = up_19)[name = string("input_197")]; string var_2463_pad_type_0 = const()[name = string("op_2463_pad_type_0"), val = string("valid")]; tensor var_2463_strides_0 = const()[name = string("op_2463_strides_0"), val = tensor([1, 1])]; tensor var_2463_pad_0 = const()[name = string("op_2463_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2463_dilations_0 = const()[name = string("op_2463_dilations_0"), val = tensor([1, 1])]; int32 var_2463_groups_0 = const()[name = string("op_2463_groups_0"), val = int32(1)]; tensor var_2463 = conv(dilations = var_2463_dilations_0, groups = var_2463_groups_0, pad = var_2463_pad_0, pad_type = var_2463_pad_type_0, strides = var_2463_strides_0, weight = encoder_layers_9_mlp_down_proj_weight_quantized, x = input_197)[name = string("op_2463")]; tensor var_2464_axes_0 = const()[name = string("op_2464_axes_0"), val = tensor([2])]; tensor var_2464 = squeeze(axes = var_2464_axes_0, x = var_2463)[name = string("op_2464")]; tensor var_2465 = const()[name = string("op_2465"), val = tensor([0, 2, 1])]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_159 = transpose(perm = var_2465, x = var_2464)[name = string("transpose_126")]; tensor var_2469_cast_fp16 = mul(x = x_159, y = const_138_promoted_to_fp16)[name = string("op_2469_cast_fp16")]; bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; tensor input_199_cast_fp16 = concat(axis = var_22, interleave = input_199_interleave_0, values = (x_159, var_2469_cast_fp16))[name = string("input_199_cast_fp16")]; tensor normed_277_axes_0 = const()[name = string("normed_277_axes_0"), val = tensor([-1])]; tensor normed_277_cast_fp16 = layer_norm(axes = normed_277_axes_0, epsilon = var_8_to_fp16, x = input_199_cast_fp16)[name = string("normed_277_cast_fp16")]; tensor var_2474_split_sizes_0 = const()[name = string("op_2474_split_sizes_0"), val = tensor([768, 768])]; int32 var_2474_axis_0 = const()[name = string("op_2474_axis_0"), val = int32(-1)]; tensor var_2474_cast_fp16_0, tensor var_2474_cast_fp16_1 = split(axis = var_2474_axis_0, split_sizes = var_2474_split_sizes_0, x = normed_277_cast_fp16)[name = string("op_2474_cast_fp16")]; tensor var_2478_to_fp16 = const()[name = string("op_2478_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300357888)))]; tensor out_119_cast_fp16 = mul(x = var_2474_cast_fp16_0, y = var_2478_to_fp16)[name = string("out_119_cast_fp16")]; tensor x_161_cast_fp16 = add(x = x_155_cast_fp16, y = out_119_cast_fp16)[name = string("x_161_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2507_cast_fp16 = mul(x = x_161_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_2507_cast_fp16")]; bool input_201_interleave_0 = const()[name = string("input_201_interleave_0"), val = bool(false)]; tensor input_201_cast_fp16 = concat(axis = var_22, interleave = input_201_interleave_0, values = (x_161_cast_fp16, var_2507_cast_fp16))[name = string("input_201_cast_fp16")]; tensor normed_281_axes_0 = const()[name = string("normed_281_axes_0"), val = tensor([-1])]; tensor normed_281_cast_fp16 = layer_norm(axes = normed_281_axes_0, epsilon = var_8_to_fp16, x = input_201_cast_fp16)[name = string("normed_281_cast_fp16")]; tensor var_2512_split_sizes_0 = const()[name = string("op_2512_split_sizes_0"), val = tensor([768, 768])]; int32 var_2512_axis_0 = const()[name = string("op_2512_axis_0"), val = int32(-1)]; tensor var_2512_cast_fp16_0, tensor var_2512_cast_fp16_1 = split(axis = var_2512_axis_0, split_sizes = var_2512_split_sizes_0, x = normed_281_cast_fp16)[name = string("op_2512_cast_fp16")]; tensor var_2516_to_fp16 = const()[name = string("op_2516_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300359488)))]; tensor out_121_cast_fp16 = mul(x = var_2512_cast_fp16_0, y = var_2516_to_fp16)[name = string("out_121_cast_fp16")]; tensor var_2522 = const()[name = string("op_2522"), val = tensor([0, 2, 1])]; tensor var_2524_axes_0 = const()[name = string("op_2524_axes_0"), val = tensor([2])]; tensor var_2523_cast_fp16 = transpose(perm = var_2522, x = out_121_cast_fp16)[name = string("transpose_125")]; tensor var_2524_cast_fp16 = expand_dims(axes = var_2524_axes_0, x = var_2523_cast_fp16)[name = string("op_2524_cast_fp16")]; string var_2531_pad_type_0 = const()[name = string("op_2531_pad_type_0"), val = string("valid")]; tensor var_2531_strides_0 = const()[name = string("op_2531_strides_0"), val = tensor([1, 1])]; tensor var_2531_pad_0 = const()[name = string("op_2531_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2531_dilations_0 = const()[name = string("op_2531_dilations_0"), val = tensor([1, 1])]; int32 var_2531_groups_0 = const()[name = string("op_2531_groups_0"), val = int32(1)]; tensor var_2531 = conv(dilations = var_2531_dilations_0, groups = var_2531_groups_0, pad = var_2531_pad_0, pad_type = var_2531_pad_type_0, strides = var_2531_strides_0, weight = encoder_layers_10_self_attn_q_proj_weight_quantized, x = var_2524_cast_fp16)[name = string("op_2531")]; tensor var_2532 = const()[name = string("op_2532"), val = tensor([1, 3, 256, 256])]; tensor var_2533 = reshape(shape = var_2532, x = var_2531)[name = string("op_2533")]; tensor var_2534 = const()[name = string("op_2534"), val = tensor([0, 1, 3, 2])]; string var_2541_pad_type_0 = const()[name = string("op_2541_pad_type_0"), val = string("valid")]; tensor var_2541_strides_0 = const()[name = string("op_2541_strides_0"), val = tensor([1, 1])]; tensor var_2541_pad_0 = const()[name = string("op_2541_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2541_dilations_0 = const()[name = string("op_2541_dilations_0"), val = tensor([1, 1])]; int32 var_2541_groups_0 = const()[name = string("op_2541_groups_0"), val = int32(1)]; tensor var_2541 = conv(dilations = var_2541_dilations_0, groups = var_2541_groups_0, pad = var_2541_pad_0, pad_type = var_2541_pad_type_0, strides = var_2541_strides_0, weight = encoder_layers_10_self_attn_k_proj_weight_quantized, x = var_2524_cast_fp16)[name = string("op_2541")]; tensor var_2542 = const()[name = string("op_2542"), val = tensor([1, 1, 256, 256])]; tensor var_2543 = reshape(shape = var_2542, x = var_2541)[name = string("op_2543")]; tensor var_2544 = const()[name = string("op_2544"), val = tensor([0, 1, 3, 2])]; string var_2551_pad_type_0 = const()[name = string("op_2551_pad_type_0"), val = string("valid")]; tensor var_2551_strides_0 = const()[name = string("op_2551_strides_0"), val = tensor([1, 1])]; tensor var_2551_pad_0 = const()[name = string("op_2551_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2551_dilations_0 = const()[name = string("op_2551_dilations_0"), val = tensor([1, 1])]; int32 var_2551_groups_0 = const()[name = string("op_2551_groups_0"), val = int32(1)]; tensor var_2551 = conv(dilations = var_2551_dilations_0, groups = var_2551_groups_0, pad = var_2551_pad_0, pad_type = var_2551_pad_type_0, strides = var_2551_strides_0, weight = encoder_layers_10_self_attn_v_proj_weight_quantized, x = var_2524_cast_fp16)[name = string("op_2551")]; tensor var_2552 = const()[name = string("op_2552"), val = tensor([1, 1, 256, 256])]; tensor var_2553 = reshape(shape = var_2552, x = var_2551)[name = string("op_2553")]; tensor var_2554 = const()[name = string("op_2554"), val = tensor([0, 1, 3, 2])]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_61 = transpose(perm = var_2534, x = var_2533)[name = string("transpose_124")]; tensor var_2560_cast_fp16 = mul(x = q_61, y = const_142_promoted_to_fp16)[name = string("op_2560_cast_fp16")]; bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; tensor input_205_cast_fp16 = concat(axis = var_22, interleave = input_205_interleave_0, values = (q_61, var_2560_cast_fp16))[name = string("input_205_cast_fp16")]; tensor normed_287_axes_0 = const()[name = string("normed_287_axes_0"), val = tensor([-1])]; tensor normed_287_cast_fp16 = layer_norm(axes = normed_287_axes_0, epsilon = var_8_to_fp16, x = input_205_cast_fp16)[name = string("normed_287_cast_fp16")]; tensor var_2565_split_sizes_0 = const()[name = string("op_2565_split_sizes_0"), val = tensor([256, 256])]; int32 var_2565_axis_0 = const()[name = string("op_2565_axis_0"), val = int32(-1)]; tensor var_2565_cast_fp16_0, tensor var_2565_cast_fp16_1 = split(axis = var_2565_axis_0, split_sizes = var_2565_split_sizes_0, x = normed_287_cast_fp16)[name = string("op_2565_cast_fp16")]; tensor var_2569_to_fp16 = const()[name = string("op_2569_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300361088)))]; tensor out_123_cast_fp16 = mul(x = var_2565_cast_fp16_0, y = var_2569_to_fp16)[name = string("out_123_cast_fp16")]; fp16 const_144_promoted_to_fp16 = const()[name = string("const_144_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_61 = transpose(perm = var_2544, x = var_2543)[name = string("transpose_123")]; tensor var_2576_cast_fp16 = mul(x = k_61, y = const_144_promoted_to_fp16)[name = string("op_2576_cast_fp16")]; bool input_207_interleave_0 = const()[name = string("input_207_interleave_0"), val = bool(false)]; tensor input_207_cast_fp16 = concat(axis = var_22, interleave = input_207_interleave_0, values = (k_61, var_2576_cast_fp16))[name = string("input_207_cast_fp16")]; tensor normed_291_axes_0 = const()[name = string("normed_291_axes_0"), val = tensor([-1])]; tensor normed_291_cast_fp16 = layer_norm(axes = normed_291_axes_0, epsilon = var_8_to_fp16, x = input_207_cast_fp16)[name = string("normed_291_cast_fp16")]; tensor var_2581_split_sizes_0 = const()[name = string("op_2581_split_sizes_0"), val = tensor([256, 256])]; int32 var_2581_axis_0 = const()[name = string("op_2581_axis_0"), val = int32(-1)]; tensor var_2581_cast_fp16_0, tensor var_2581_cast_fp16_1 = split(axis = var_2581_axis_0, split_sizes = var_2581_split_sizes_0, x = normed_291_cast_fp16)[name = string("op_2581_cast_fp16")]; tensor var_2585_to_fp16 = const()[name = string("op_2585_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300361664)))]; tensor out_125_cast_fp16 = mul(x = var_2581_cast_fp16_0, y = var_2585_to_fp16)[name = string("out_125_cast_fp16")]; tensor var_2588 = mul(x = out_123_cast_fp16, y = cos_1_quantized)[name = string("op_2588")]; tensor var_2589_split_sizes_0 = const()[name = string("op_2589_split_sizes_0"), val = tensor([128, 128])]; int32 var_2589_axis_0 = const()[name = string("op_2589_axis_0"), val = int32(-1)]; tensor var_2589_0, tensor var_2589_1 = split(axis = var_2589_axis_0, split_sizes = var_2589_split_sizes_0, x = out_123_cast_fp16)[name = string("op_2589")]; fp16 const_146_promoted = const()[name = string("const_146_promoted"), val = fp16(-0x1p+0)]; tensor var_2591 = mul(x = var_2589_1, y = const_146_promoted)[name = string("op_2591")]; bool var_2593_interleave_0 = const()[name = string("op_2593_interleave_0"), val = bool(false)]; tensor var_2593 = concat(axis = var_22, interleave = var_2593_interleave_0, values = (var_2591, var_2589_0))[name = string("op_2593")]; tensor var_2594 = mul(x = var_2593, y = sin_1_quantized)[name = string("op_2594")]; tensor q_65 = add(x = var_2588, y = var_2594)[name = string("q_65")]; tensor var_2596 = mul(x = out_125_cast_fp16, y = cos_1_quantized)[name = string("op_2596")]; tensor var_2597_split_sizes_0 = const()[name = string("op_2597_split_sizes_0"), val = tensor([128, 128])]; int32 var_2597_axis_0 = const()[name = string("op_2597_axis_0"), val = int32(-1)]; tensor var_2597_0, tensor var_2597_1 = split(axis = var_2597_axis_0, split_sizes = var_2597_split_sizes_0, x = out_125_cast_fp16)[name = string("op_2597")]; fp16 const_147_promoted = const()[name = string("const_147_promoted"), val = fp16(-0x1p+0)]; tensor var_2599 = mul(x = var_2597_1, y = const_147_promoted)[name = string("op_2599")]; bool var_2601_interleave_0 = const()[name = string("op_2601_interleave_0"), val = bool(false)]; tensor var_2601 = concat(axis = var_22, interleave = var_2601_interleave_0, values = (var_2599, var_2597_0))[name = string("op_2601")]; tensor var_2602 = mul(x = var_2601, y = sin_1_quantized)[name = string("op_2602")]; tensor hidden_states_121 = add(x = var_2596, y = var_2602)[name = string("hidden_states_121")]; tensor hidden_states_123_axes_0 = const()[name = string("hidden_states_123_axes_0"), val = tensor([2])]; tensor hidden_states_123 = expand_dims(axes = hidden_states_123_axes_0, x = hidden_states_121)[name = string("hidden_states_123")]; tensor var_2605 = const()[name = string("op_2605"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_125 = tile(reps = var_2605, x = hidden_states_123)[name = string("hidden_states_125")]; tensor var_2607 = const()[name = string("op_2607"), val = tensor([1, 3, 256, 256])]; tensor k_65 = reshape(shape = var_2607, x = hidden_states_125)[name = string("k_65")]; tensor hidden_states_129_axes_0 = const()[name = string("hidden_states_129_axes_0"), val = tensor([2])]; tensor hidden_states_127 = transpose(perm = var_2554, x = var_2553)[name = string("transpose_122")]; tensor hidden_states_129 = expand_dims(axes = hidden_states_129_axes_0, x = hidden_states_127)[name = string("hidden_states_129")]; tensor var_2610 = const()[name = string("op_2610"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_131 = tile(reps = var_2610, x = hidden_states_129)[name = string("hidden_states_131")]; tensor var_2612 = const()[name = string("op_2612"), val = tensor([1, 3, 256, 256])]; tensor v_21 = reshape(shape = var_2612, x = hidden_states_131)[name = string("v_21")]; bool var_2617_transpose_x_1 = const()[name = string("op_2617_transpose_x_1"), val = bool(false)]; bool var_2617_transpose_y_1 = const()[name = string("op_2617_transpose_y_1"), val = bool(true)]; tensor var_2617_cast_fp16 = matmul(transpose_x = var_2617_transpose_x_1, transpose_y = var_2617_transpose_y_1, x = q_65, y = k_65)[name = string("op_2617_cast_fp16")]; fp16 var_2618_to_fp16 = const()[name = string("op_2618_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_61_cast_fp16 = mul(x = var_2617_cast_fp16, y = var_2618_to_fp16)[name = string("attn_weights_61_cast_fp16")]; tensor attn_weights_63_cast_fp16 = add(x = attn_weights_61_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; tensor var_2622_cast_fp16 = softmax(axis = var_22, x = attn_weights_63_cast_fp16)[name = string("op_2622_cast_fp16")]; bool var_2626_transpose_x_0 = const()[name = string("op_2626_transpose_x_0"), val = bool(false)]; bool var_2626_transpose_y_0 = const()[name = string("op_2626_transpose_y_0"), val = bool(false)]; tensor var_2626_cast_fp16 = matmul(transpose_x = var_2626_transpose_x_0, transpose_y = var_2626_transpose_y_0, x = var_2622_cast_fp16, y = v_21)[name = string("op_2626_cast_fp16")]; tensor var_2628 = const()[name = string("op_2628"), val = tensor([0, 2, 1, 3])]; tensor var_2631 = const()[name = string("op_2631"), val = tensor([1, 256, 768])]; tensor var_2629 = transpose(perm = var_2628, x = var_2626_cast_fp16)[name = string("transpose_121")]; tensor attn_out_63 = reshape(shape = var_2631, x = var_2629)[name = string("attn_out_63")]; tensor var_2633 = const()[name = string("op_2633"), val = tensor([0, 2, 1])]; tensor squeeze_10_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300362240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300952128))))[name = string("squeeze_10_quantized")]; string var_2642_pad_type_0 = const()[name = string("op_2642_pad_type_0"), val = string("valid")]; int32 var_2642_groups_0 = const()[name = string("op_2642_groups_0"), val = int32(1)]; tensor var_2642_strides_0 = const()[name = string("op_2642_strides_0"), val = tensor([1])]; tensor var_2642_pad_0 = const()[name = string("op_2642_pad_0"), val = tensor([0, 0])]; tensor var_2642_dilations_0 = const()[name = string("op_2642_dilations_0"), val = tensor([1])]; tensor var_2634 = transpose(perm = var_2633, x = attn_out_63)[name = string("transpose_120")]; tensor var_2642 = conv(dilations = var_2642_dilations_0, groups = var_2642_groups_0, pad = var_2642_pad_0, pad_type = var_2642_pad_type_0, strides = var_2642_strides_0, weight = squeeze_10_quantized, x = var_2634)[name = string("op_2642")]; tensor var_2643 = const()[name = string("op_2643"), val = tensor([0, 2, 1])]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_169 = transpose(perm = var_2643, x = var_2642)[name = string("transpose_119")]; tensor var_2647_cast_fp16 = mul(x = x_169, y = const_148_promoted_to_fp16)[name = string("op_2647_cast_fp16")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211_cast_fp16 = concat(axis = var_22, interleave = input_211_interleave_0, values = (x_169, var_2647_cast_fp16))[name = string("input_211_cast_fp16")]; tensor normed_295_axes_0 = const()[name = string("normed_295_axes_0"), val = tensor([-1])]; tensor normed_295_cast_fp16 = layer_norm(axes = normed_295_axes_0, epsilon = var_8_to_fp16, x = input_211_cast_fp16)[name = string("normed_295_cast_fp16")]; tensor var_2652_split_sizes_0 = const()[name = string("op_2652_split_sizes_0"), val = tensor([768, 768])]; int32 var_2652_axis_0 = const()[name = string("op_2652_axis_0"), val = int32(-1)]; tensor var_2652_cast_fp16_0, tensor var_2652_cast_fp16_1 = split(axis = var_2652_axis_0, split_sizes = var_2652_split_sizes_0, x = normed_295_cast_fp16)[name = string("op_2652_cast_fp16")]; tensor var_2656_to_fp16 = const()[name = string("op_2656_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300953728)))]; tensor out_127_cast_fp16 = mul(x = var_2652_cast_fp16_0, y = var_2656_to_fp16)[name = string("out_127_cast_fp16")]; tensor x_171_cast_fp16 = add(x = x_161_cast_fp16, y = out_127_cast_fp16)[name = string("x_171_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2663_cast_fp16 = mul(x = x_171_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_2663_cast_fp16")]; bool input_213_interleave_0 = const()[name = string("input_213_interleave_0"), val = bool(false)]; tensor input_213_cast_fp16 = concat(axis = var_22, interleave = input_213_interleave_0, values = (x_171_cast_fp16, var_2663_cast_fp16))[name = string("input_213_cast_fp16")]; tensor normed_299_axes_0 = const()[name = string("normed_299_axes_0"), val = tensor([-1])]; tensor normed_299_cast_fp16 = layer_norm(axes = normed_299_axes_0, epsilon = var_8_to_fp16, x = input_213_cast_fp16)[name = string("normed_299_cast_fp16")]; tensor var_2668_split_sizes_0 = const()[name = string("op_2668_split_sizes_0"), val = tensor([768, 768])]; int32 var_2668_axis_0 = const()[name = string("op_2668_axis_0"), val = int32(-1)]; tensor var_2668_cast_fp16_0, tensor var_2668_cast_fp16_1 = split(axis = var_2668_axis_0, split_sizes = var_2668_split_sizes_0, x = normed_299_cast_fp16)[name = string("op_2668_cast_fp16")]; tensor var_2672_to_fp16 = const()[name = string("op_2672_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300955328)))]; tensor out_129_cast_fp16 = mul(x = var_2668_cast_fp16_0, y = var_2672_to_fp16)[name = string("out_129_cast_fp16")]; tensor var_2679 = const()[name = string("op_2679"), val = tensor([0, 2, 1])]; tensor input_215_axes_0 = const()[name = string("input_215_axes_0"), val = tensor([2])]; tensor var_2680 = transpose(perm = var_2679, x = out_129_cast_fp16)[name = string("transpose_118")]; tensor input_215 = expand_dims(axes = input_215_axes_0, x = var_2680)[name = string("input_215")]; string gate_41_pad_type_0 = const()[name = string("gate_41_pad_type_0"), val = string("valid")]; tensor gate_41_strides_0 = const()[name = string("gate_41_strides_0"), val = tensor([1, 1])]; tensor gate_41_pad_0 = const()[name = string("gate_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_41_dilations_0 = const()[name = string("gate_41_dilations_0"), val = tensor([1, 1])]; int32 gate_41_groups_0 = const()[name = string("gate_41_groups_0"), val = int32(1)]; tensor gate_41 = conv(dilations = gate_41_dilations_0, groups = gate_41_groups_0, pad = gate_41_pad_0, pad_type = gate_41_pad_type_0, strides = gate_41_strides_0, weight = encoder_layers_10_mlp_gate_proj_weight_quantized, x = input_215)[name = string("gate_41")]; string up_21_pad_type_0 = const()[name = string("up_21_pad_type_0"), val = string("valid")]; tensor up_21_strides_0 = const()[name = string("up_21_strides_0"), val = tensor([1, 1])]; tensor up_21_pad_0 = const()[name = string("up_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_21_dilations_0 = const()[name = string("up_21_dilations_0"), val = tensor([1, 1])]; int32 up_21_groups_0 = const()[name = string("up_21_groups_0"), val = int32(1)]; tensor up_21 = conv(dilations = up_21_dilations_0, groups = up_21_groups_0, pad = up_21_pad_0, pad_type = up_21_pad_type_0, strides = up_21_strides_0, weight = encoder_layers_10_mlp_up_proj_weight_quantized, x = input_215)[name = string("up_21")]; string gate_43_mode_0 = const()[name = string("gate_43_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_43 = gelu(mode = gate_43_mode_0, x = gate_41)[name = string("gate_43")]; tensor input_217 = mul(x = gate_43, y = up_21)[name = string("input_217")]; string var_2701_pad_type_0 = const()[name = string("op_2701_pad_type_0"), val = string("valid")]; tensor var_2701_strides_0 = const()[name = string("op_2701_strides_0"), val = tensor([1, 1])]; tensor var_2701_pad_0 = const()[name = string("op_2701_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2701_dilations_0 = const()[name = string("op_2701_dilations_0"), val = tensor([1, 1])]; int32 var_2701_groups_0 = const()[name = string("op_2701_groups_0"), val = int32(1)]; tensor var_2701 = conv(dilations = var_2701_dilations_0, groups = var_2701_groups_0, pad = var_2701_pad_0, pad_type = var_2701_pad_type_0, strides = var_2701_strides_0, weight = encoder_layers_10_mlp_down_proj_weight_quantized, x = input_217)[name = string("op_2701")]; tensor var_2702_axes_0 = const()[name = string("op_2702_axes_0"), val = tensor([2])]; tensor var_2702 = squeeze(axes = var_2702_axes_0, x = var_2701)[name = string("op_2702")]; tensor var_2703 = const()[name = string("op_2703"), val = tensor([0, 2, 1])]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_175 = transpose(perm = var_2703, x = var_2702)[name = string("transpose_117")]; tensor var_2707_cast_fp16 = mul(x = x_175, y = const_152_promoted_to_fp16)[name = string("op_2707_cast_fp16")]; bool input_219_interleave_0 = const()[name = string("input_219_interleave_0"), val = bool(false)]; tensor input_219_cast_fp16 = concat(axis = var_22, interleave = input_219_interleave_0, values = (x_175, var_2707_cast_fp16))[name = string("input_219_cast_fp16")]; tensor normed_305_axes_0 = const()[name = string("normed_305_axes_0"), val = tensor([-1])]; tensor normed_305_cast_fp16 = layer_norm(axes = normed_305_axes_0, epsilon = var_8_to_fp16, x = input_219_cast_fp16)[name = string("normed_305_cast_fp16")]; tensor var_2712_split_sizes_0 = const()[name = string("op_2712_split_sizes_0"), val = tensor([768, 768])]; int32 var_2712_axis_0 = const()[name = string("op_2712_axis_0"), val = int32(-1)]; tensor var_2712_cast_fp16_0, tensor var_2712_cast_fp16_1 = split(axis = var_2712_axis_0, split_sizes = var_2712_split_sizes_0, x = normed_305_cast_fp16)[name = string("op_2712_cast_fp16")]; tensor var_2716_to_fp16 = const()[name = string("op_2716_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300956928)))]; tensor out_131_cast_fp16 = mul(x = var_2712_cast_fp16_0, y = var_2716_to_fp16)[name = string("out_131_cast_fp16")]; tensor x_177_cast_fp16 = add(x = x_171_cast_fp16, y = out_131_cast_fp16)[name = string("x_177_cast_fp16")]; fp16 const_154_promoted_to_fp16 = const()[name = string("const_154_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2745_cast_fp16 = mul(x = x_177_cast_fp16, y = const_154_promoted_to_fp16)[name = string("op_2745_cast_fp16")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221_cast_fp16 = concat(axis = var_22, interleave = input_221_interleave_0, values = (x_177_cast_fp16, var_2745_cast_fp16))[name = string("input_221_cast_fp16")]; tensor normed_309_axes_0 = const()[name = string("normed_309_axes_0"), val = tensor([-1])]; tensor normed_309_cast_fp16 = layer_norm(axes = normed_309_axes_0, epsilon = var_8_to_fp16, x = input_221_cast_fp16)[name = string("normed_309_cast_fp16")]; tensor var_2750_split_sizes_0 = const()[name = string("op_2750_split_sizes_0"), val = tensor([768, 768])]; int32 var_2750_axis_0 = const()[name = string("op_2750_axis_0"), val = int32(-1)]; tensor var_2750_cast_fp16_0, tensor var_2750_cast_fp16_1 = split(axis = var_2750_axis_0, split_sizes = var_2750_split_sizes_0, x = normed_309_cast_fp16)[name = string("op_2750_cast_fp16")]; tensor var_2754_to_fp16 = const()[name = string("op_2754_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300958528)))]; tensor out_133_cast_fp16 = mul(x = var_2750_cast_fp16_0, y = var_2754_to_fp16)[name = string("out_133_cast_fp16")]; tensor var_2760 = const()[name = string("op_2760"), val = tensor([0, 2, 1])]; tensor var_2762_axes_0 = const()[name = string("op_2762_axes_0"), val = tensor([2])]; tensor var_2761_cast_fp16 = transpose(perm = var_2760, x = out_133_cast_fp16)[name = string("transpose_116")]; tensor var_2762_cast_fp16 = expand_dims(axes = var_2762_axes_0, x = var_2761_cast_fp16)[name = string("op_2762_cast_fp16")]; string var_2769_pad_type_0 = const()[name = string("op_2769_pad_type_0"), val = string("valid")]; tensor var_2769_strides_0 = const()[name = string("op_2769_strides_0"), val = tensor([1, 1])]; tensor var_2769_pad_0 = const()[name = string("op_2769_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2769_dilations_0 = const()[name = string("op_2769_dilations_0"), val = tensor([1, 1])]; int32 var_2769_groups_0 = const()[name = string("op_2769_groups_0"), val = int32(1)]; tensor var_2769 = conv(dilations = var_2769_dilations_0, groups = var_2769_groups_0, pad = var_2769_pad_0, pad_type = var_2769_pad_type_0, strides = var_2769_strides_0, weight = encoder_layers_11_self_attn_q_proj_weight_quantized, x = var_2762_cast_fp16)[name = string("op_2769")]; tensor var_2770 = const()[name = string("op_2770"), val = tensor([1, 3, 256, 256])]; tensor var_2771 = reshape(shape = var_2770, x = var_2769)[name = string("op_2771")]; tensor var_2772 = const()[name = string("op_2772"), val = tensor([0, 1, 3, 2])]; string var_2779_pad_type_0 = const()[name = string("op_2779_pad_type_0"), val = string("valid")]; tensor var_2779_strides_0 = const()[name = string("op_2779_strides_0"), val = tensor([1, 1])]; tensor var_2779_pad_0 = const()[name = string("op_2779_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2779_dilations_0 = const()[name = string("op_2779_dilations_0"), val = tensor([1, 1])]; int32 var_2779_groups_0 = const()[name = string("op_2779_groups_0"), val = int32(1)]; tensor var_2779 = conv(dilations = var_2779_dilations_0, groups = var_2779_groups_0, pad = var_2779_pad_0, pad_type = var_2779_pad_type_0, strides = var_2779_strides_0, weight = encoder_layers_11_self_attn_k_proj_weight_quantized, x = var_2762_cast_fp16)[name = string("op_2779")]; tensor var_2780 = const()[name = string("op_2780"), val = tensor([1, 1, 256, 256])]; tensor var_2781 = reshape(shape = var_2780, x = var_2779)[name = string("op_2781")]; tensor var_2782 = const()[name = string("op_2782"), val = tensor([0, 1, 3, 2])]; string var_2789_pad_type_0 = const()[name = string("op_2789_pad_type_0"), val = string("valid")]; tensor var_2789_strides_0 = const()[name = string("op_2789_strides_0"), val = tensor([1, 1])]; tensor var_2789_pad_0 = const()[name = string("op_2789_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2789_dilations_0 = const()[name = string("op_2789_dilations_0"), val = tensor([1, 1])]; int32 var_2789_groups_0 = const()[name = string("op_2789_groups_0"), val = int32(1)]; tensor var_2789 = conv(dilations = var_2789_dilations_0, groups = var_2789_groups_0, pad = var_2789_pad_0, pad_type = var_2789_pad_type_0, strides = var_2789_strides_0, weight = encoder_layers_11_self_attn_v_proj_weight_quantized, x = var_2762_cast_fp16)[name = string("op_2789")]; tensor var_2790 = const()[name = string("op_2790"), val = tensor([1, 1, 256, 256])]; tensor var_2791 = reshape(shape = var_2790, x = var_2789)[name = string("op_2791")]; tensor var_2792 = const()[name = string("op_2792"), val = tensor([0, 1, 3, 2])]; fp16 const_156_promoted_to_fp16 = const()[name = string("const_156_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_67 = transpose(perm = var_2772, x = var_2771)[name = string("transpose_115")]; tensor var_2798_cast_fp16 = mul(x = q_67, y = const_156_promoted_to_fp16)[name = string("op_2798_cast_fp16")]; bool input_225_interleave_0 = const()[name = string("input_225_interleave_0"), val = bool(false)]; tensor input_225_cast_fp16 = concat(axis = var_22, interleave = input_225_interleave_0, values = (q_67, var_2798_cast_fp16))[name = string("input_225_cast_fp16")]; tensor normed_315_axes_0 = const()[name = string("normed_315_axes_0"), val = tensor([-1])]; tensor normed_315_cast_fp16 = layer_norm(axes = normed_315_axes_0, epsilon = var_8_to_fp16, x = input_225_cast_fp16)[name = string("normed_315_cast_fp16")]; tensor var_2803_split_sizes_0 = const()[name = string("op_2803_split_sizes_0"), val = tensor([256, 256])]; int32 var_2803_axis_0 = const()[name = string("op_2803_axis_0"), val = int32(-1)]; tensor var_2803_cast_fp16_0, tensor var_2803_cast_fp16_1 = split(axis = var_2803_axis_0, split_sizes = var_2803_split_sizes_0, x = normed_315_cast_fp16)[name = string("op_2803_cast_fp16")]; tensor var_2807_to_fp16 = const()[name = string("op_2807_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300960128)))]; tensor out_135_cast_fp16 = mul(x = var_2803_cast_fp16_0, y = var_2807_to_fp16)[name = string("out_135_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_67 = transpose(perm = var_2782, x = var_2781)[name = string("transpose_114")]; tensor var_2814_cast_fp16 = mul(x = k_67, y = const_158_promoted_to_fp16)[name = string("op_2814_cast_fp16")]; bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; tensor input_227_cast_fp16 = concat(axis = var_22, interleave = input_227_interleave_0, values = (k_67, var_2814_cast_fp16))[name = string("input_227_cast_fp16")]; tensor normed_319_axes_0 = const()[name = string("normed_319_axes_0"), val = tensor([-1])]; tensor normed_319_cast_fp16 = layer_norm(axes = normed_319_axes_0, epsilon = var_8_to_fp16, x = input_227_cast_fp16)[name = string("normed_319_cast_fp16")]; tensor var_2819_split_sizes_0 = const()[name = string("op_2819_split_sizes_0"), val = tensor([256, 256])]; int32 var_2819_axis_0 = const()[name = string("op_2819_axis_0"), val = int32(-1)]; tensor var_2819_cast_fp16_0, tensor var_2819_cast_fp16_1 = split(axis = var_2819_axis_0, split_sizes = var_2819_split_sizes_0, x = normed_319_cast_fp16)[name = string("op_2819_cast_fp16")]; tensor var_2823_to_fp16 = const()[name = string("op_2823_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300960704)))]; tensor out_137_cast_fp16 = mul(x = var_2819_cast_fp16_0, y = var_2823_to_fp16)[name = string("out_137_cast_fp16")]; tensor var_2826 = mul(x = out_135_cast_fp16, y = cos_quantized)[name = string("op_2826")]; tensor var_2827_split_sizes_0 = const()[name = string("op_2827_split_sizes_0"), val = tensor([128, 128])]; int32 var_2827_axis_0 = const()[name = string("op_2827_axis_0"), val = int32(-1)]; tensor var_2827_0, tensor var_2827_1 = split(axis = var_2827_axis_0, split_sizes = var_2827_split_sizes_0, x = out_135_cast_fp16)[name = string("op_2827")]; fp16 const_160_promoted = const()[name = string("const_160_promoted"), val = fp16(-0x1p+0)]; tensor var_2829 = mul(x = var_2827_1, y = const_160_promoted)[name = string("op_2829")]; bool var_2831_interleave_0 = const()[name = string("op_2831_interleave_0"), val = bool(false)]; tensor var_2831 = concat(axis = var_22, interleave = var_2831_interleave_0, values = (var_2829, var_2827_0))[name = string("op_2831")]; tensor var_2832 = mul(x = var_2831, y = sin_quantized)[name = string("op_2832")]; tensor q_71 = add(x = var_2826, y = var_2832)[name = string("q_71")]; tensor var_2834 = mul(x = out_137_cast_fp16, y = cos_quantized)[name = string("op_2834")]; tensor var_2835_split_sizes_0 = const()[name = string("op_2835_split_sizes_0"), val = tensor([128, 128])]; int32 var_2835_axis_0 = const()[name = string("op_2835_axis_0"), val = int32(-1)]; tensor var_2835_0, tensor var_2835_1 = split(axis = var_2835_axis_0, split_sizes = var_2835_split_sizes_0, x = out_137_cast_fp16)[name = string("op_2835")]; fp16 const_161_promoted = const()[name = string("const_161_promoted"), val = fp16(-0x1p+0)]; tensor var_2837 = mul(x = var_2835_1, y = const_161_promoted)[name = string("op_2837")]; bool var_2839_interleave_0 = const()[name = string("op_2839_interleave_0"), val = bool(false)]; tensor var_2839 = concat(axis = var_22, interleave = var_2839_interleave_0, values = (var_2837, var_2835_0))[name = string("op_2839")]; tensor var_2840 = mul(x = var_2839, y = sin_quantized)[name = string("op_2840")]; tensor hidden_states_133 = add(x = var_2834, y = var_2840)[name = string("hidden_states_133")]; tensor hidden_states_135_axes_0 = const()[name = string("hidden_states_135_axes_0"), val = tensor([2])]; tensor hidden_states_135 = expand_dims(axes = hidden_states_135_axes_0, x = hidden_states_133)[name = string("hidden_states_135")]; tensor var_2843 = const()[name = string("op_2843"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_137 = tile(reps = var_2843, x = hidden_states_135)[name = string("hidden_states_137")]; tensor var_2845 = const()[name = string("op_2845"), val = tensor([1, 3, 256, 256])]; tensor k_71 = reshape(shape = var_2845, x = hidden_states_137)[name = string("k_71")]; tensor hidden_states_141_axes_0 = const()[name = string("hidden_states_141_axes_0"), val = tensor([2])]; tensor hidden_states_139 = transpose(perm = var_2792, x = var_2791)[name = string("transpose_113")]; tensor hidden_states_141 = expand_dims(axes = hidden_states_141_axes_0, x = hidden_states_139)[name = string("hidden_states_141")]; tensor var_2848 = const()[name = string("op_2848"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_143 = tile(reps = var_2848, x = hidden_states_141)[name = string("hidden_states_143")]; tensor var_2850 = const()[name = string("op_2850"), val = tensor([1, 3, 256, 256])]; tensor v_23 = reshape(shape = var_2850, x = hidden_states_143)[name = string("v_23")]; bool var_2855_transpose_x_1 = const()[name = string("op_2855_transpose_x_1"), val = bool(false)]; bool var_2855_transpose_y_1 = const()[name = string("op_2855_transpose_y_1"), val = bool(true)]; tensor var_2855_cast_fp16 = matmul(transpose_x = var_2855_transpose_x_1, transpose_y = var_2855_transpose_y_1, x = q_71, y = k_71)[name = string("op_2855_cast_fp16")]; fp16 var_2856_to_fp16 = const()[name = string("op_2856_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_67_cast_fp16 = mul(x = var_2855_cast_fp16, y = var_2856_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_69_cast_fp16")]; tensor var_2860_cast_fp16 = softmax(axis = var_22, x = attn_weights_69_cast_fp16)[name = string("op_2860_cast_fp16")]; bool var_2864_transpose_x_0 = const()[name = string("op_2864_transpose_x_0"), val = bool(false)]; bool var_2864_transpose_y_0 = const()[name = string("op_2864_transpose_y_0"), val = bool(false)]; tensor var_2864_cast_fp16 = matmul(transpose_x = var_2864_transpose_x_0, transpose_y = var_2864_transpose_y_0, x = var_2860_cast_fp16, y = v_23)[name = string("op_2864_cast_fp16")]; tensor var_2866 = const()[name = string("op_2866"), val = tensor([0, 2, 1, 3])]; tensor var_2869 = const()[name = string("op_2869"), val = tensor([1, 256, 768])]; tensor var_2867 = transpose(perm = var_2866, x = var_2864_cast_fp16)[name = string("transpose_112")]; tensor attn_out_69 = reshape(shape = var_2869, x = var_2867)[name = string("attn_out_69")]; tensor var_2871 = const()[name = string("op_2871"), val = tensor([0, 2, 1])]; tensor squeeze_11_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300961280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301551168))))[name = string("squeeze_11_quantized")]; string var_2880_pad_type_0 = const()[name = string("op_2880_pad_type_0"), val = string("valid")]; int32 var_2880_groups_0 = const()[name = string("op_2880_groups_0"), val = int32(1)]; tensor var_2880_strides_0 = const()[name = string("op_2880_strides_0"), val = tensor([1])]; tensor var_2880_pad_0 = const()[name = string("op_2880_pad_0"), val = tensor([0, 0])]; tensor var_2880_dilations_0 = const()[name = string("op_2880_dilations_0"), val = tensor([1])]; tensor var_2872 = transpose(perm = var_2871, x = attn_out_69)[name = string("transpose_111")]; tensor var_2880 = conv(dilations = var_2880_dilations_0, groups = var_2880_groups_0, pad = var_2880_pad_0, pad_type = var_2880_pad_type_0, strides = var_2880_strides_0, weight = squeeze_11_quantized, x = var_2872)[name = string("op_2880")]; tensor var_2881 = const()[name = string("op_2881"), val = tensor([0, 2, 1])]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_185 = transpose(perm = var_2881, x = var_2880)[name = string("transpose_110")]; tensor var_2885_cast_fp16 = mul(x = x_185, y = const_162_promoted_to_fp16)[name = string("op_2885_cast_fp16")]; bool input_231_interleave_0 = const()[name = string("input_231_interleave_0"), val = bool(false)]; tensor input_231_cast_fp16 = concat(axis = var_22, interleave = input_231_interleave_0, values = (x_185, var_2885_cast_fp16))[name = string("input_231_cast_fp16")]; tensor normed_323_axes_0 = const()[name = string("normed_323_axes_0"), val = tensor([-1])]; tensor normed_323_cast_fp16 = layer_norm(axes = normed_323_axes_0, epsilon = var_8_to_fp16, x = input_231_cast_fp16)[name = string("normed_323_cast_fp16")]; tensor var_2890_split_sizes_0 = const()[name = string("op_2890_split_sizes_0"), val = tensor([768, 768])]; int32 var_2890_axis_0 = const()[name = string("op_2890_axis_0"), val = int32(-1)]; tensor var_2890_cast_fp16_0, tensor var_2890_cast_fp16_1 = split(axis = var_2890_axis_0, split_sizes = var_2890_split_sizes_0, x = normed_323_cast_fp16)[name = string("op_2890_cast_fp16")]; tensor var_2894_to_fp16 = const()[name = string("op_2894_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301552768)))]; tensor out_139_cast_fp16 = mul(x = var_2890_cast_fp16_0, y = var_2894_to_fp16)[name = string("out_139_cast_fp16")]; tensor x_187_cast_fp16 = add(x = x_177_cast_fp16, y = out_139_cast_fp16)[name = string("x_187_cast_fp16")]; fp16 const_164_promoted_to_fp16 = const()[name = string("const_164_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2901_cast_fp16 = mul(x = x_187_cast_fp16, y = const_164_promoted_to_fp16)[name = string("op_2901_cast_fp16")]; bool input_233_interleave_0 = const()[name = string("input_233_interleave_0"), val = bool(false)]; tensor input_233_cast_fp16 = concat(axis = var_22, interleave = input_233_interleave_0, values = (x_187_cast_fp16, var_2901_cast_fp16))[name = string("input_233_cast_fp16")]; tensor normed_327_axes_0 = const()[name = string("normed_327_axes_0"), val = tensor([-1])]; tensor normed_327_cast_fp16 = layer_norm(axes = normed_327_axes_0, epsilon = var_8_to_fp16, x = input_233_cast_fp16)[name = string("normed_327_cast_fp16")]; tensor var_2906_split_sizes_0 = const()[name = string("op_2906_split_sizes_0"), val = tensor([768, 768])]; int32 var_2906_axis_0 = const()[name = string("op_2906_axis_0"), val = int32(-1)]; tensor var_2906_cast_fp16_0, tensor var_2906_cast_fp16_1 = split(axis = var_2906_axis_0, split_sizes = var_2906_split_sizes_0, x = normed_327_cast_fp16)[name = string("op_2906_cast_fp16")]; tensor var_2910_to_fp16 = const()[name = string("op_2910_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301554368)))]; tensor out_141_cast_fp16 = mul(x = var_2906_cast_fp16_0, y = var_2910_to_fp16)[name = string("out_141_cast_fp16")]; tensor var_2917 = const()[name = string("op_2917"), val = tensor([0, 2, 1])]; tensor input_235_axes_0 = const()[name = string("input_235_axes_0"), val = tensor([2])]; tensor var_2918 = transpose(perm = var_2917, x = out_141_cast_fp16)[name = string("transpose_109")]; tensor input_235 = expand_dims(axes = input_235_axes_0, x = var_2918)[name = string("input_235")]; string gate_45_pad_type_0 = const()[name = string("gate_45_pad_type_0"), val = string("valid")]; tensor gate_45_strides_0 = const()[name = string("gate_45_strides_0"), val = tensor([1, 1])]; tensor gate_45_pad_0 = const()[name = string("gate_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_45_dilations_0 = const()[name = string("gate_45_dilations_0"), val = tensor([1, 1])]; int32 gate_45_groups_0 = const()[name = string("gate_45_groups_0"), val = int32(1)]; tensor gate_45 = conv(dilations = gate_45_dilations_0, groups = gate_45_groups_0, pad = gate_45_pad_0, pad_type = gate_45_pad_type_0, strides = gate_45_strides_0, weight = encoder_layers_11_mlp_gate_proj_weight_quantized, x = input_235)[name = string("gate_45")]; string up_23_pad_type_0 = const()[name = string("up_23_pad_type_0"), val = string("valid")]; tensor up_23_strides_0 = const()[name = string("up_23_strides_0"), val = tensor([1, 1])]; tensor up_23_pad_0 = const()[name = string("up_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_23_dilations_0 = const()[name = string("up_23_dilations_0"), val = tensor([1, 1])]; int32 up_23_groups_0 = const()[name = string("up_23_groups_0"), val = int32(1)]; tensor up_23 = conv(dilations = up_23_dilations_0, groups = up_23_groups_0, pad = up_23_pad_0, pad_type = up_23_pad_type_0, strides = up_23_strides_0, weight = encoder_layers_11_mlp_up_proj_weight_quantized, x = input_235)[name = string("up_23")]; string gate_47_mode_0 = const()[name = string("gate_47_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_47 = gelu(mode = gate_47_mode_0, x = gate_45)[name = string("gate_47")]; tensor input_237 = mul(x = gate_47, y = up_23)[name = string("input_237")]; string var_2939_pad_type_0 = const()[name = string("op_2939_pad_type_0"), val = string("valid")]; tensor var_2939_strides_0 = const()[name = string("op_2939_strides_0"), val = tensor([1, 1])]; tensor var_2939_pad_0 = const()[name = string("op_2939_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2939_dilations_0 = const()[name = string("op_2939_dilations_0"), val = tensor([1, 1])]; int32 var_2939_groups_0 = const()[name = string("op_2939_groups_0"), val = int32(1)]; tensor var_2939 = conv(dilations = var_2939_dilations_0, groups = var_2939_groups_0, pad = var_2939_pad_0, pad_type = var_2939_pad_type_0, strides = var_2939_strides_0, weight = encoder_layers_11_mlp_down_proj_weight_quantized, x = input_237)[name = string("op_2939")]; tensor var_2940_axes_0 = const()[name = string("op_2940_axes_0"), val = tensor([2])]; tensor var_2940 = squeeze(axes = var_2940_axes_0, x = var_2939)[name = string("op_2940")]; tensor var_2941 = const()[name = string("op_2941"), val = tensor([0, 2, 1])]; fp16 const_166_promoted_to_fp16 = const()[name = string("const_166_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_191 = transpose(perm = var_2941, x = var_2940)[name = string("transpose_108")]; tensor var_2945_cast_fp16 = mul(x = x_191, y = const_166_promoted_to_fp16)[name = string("op_2945_cast_fp16")]; bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; tensor input_239_cast_fp16 = concat(axis = var_22, interleave = input_239_interleave_0, values = (x_191, var_2945_cast_fp16))[name = string("input_239_cast_fp16")]; tensor normed_333_axes_0 = const()[name = string("normed_333_axes_0"), val = tensor([-1])]; tensor normed_333_cast_fp16 = layer_norm(axes = normed_333_axes_0, epsilon = var_8_to_fp16, x = input_239_cast_fp16)[name = string("normed_333_cast_fp16")]; tensor var_2950_split_sizes_0 = const()[name = string("op_2950_split_sizes_0"), val = tensor([768, 768])]; int32 var_2950_axis_0 = const()[name = string("op_2950_axis_0"), val = int32(-1)]; tensor var_2950_cast_fp16_0, tensor var_2950_cast_fp16_1 = split(axis = var_2950_axis_0, split_sizes = var_2950_split_sizes_0, x = normed_333_cast_fp16)[name = string("op_2950_cast_fp16")]; tensor var_2954_to_fp16 = const()[name = string("op_2954_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301555968)))]; tensor out_143_cast_fp16 = mul(x = var_2950_cast_fp16_0, y = var_2954_to_fp16)[name = string("out_143_cast_fp16")]; tensor x_193_cast_fp16 = add(x = x_187_cast_fp16, y = out_143_cast_fp16)[name = string("x_193_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2983_cast_fp16 = mul(x = x_193_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_2983_cast_fp16")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241_cast_fp16 = concat(axis = var_22, interleave = input_241_interleave_0, values = (x_193_cast_fp16, var_2983_cast_fp16))[name = string("input_241_cast_fp16")]; tensor normed_337_axes_0 = const()[name = string("normed_337_axes_0"), val = tensor([-1])]; tensor normed_337_cast_fp16 = layer_norm(axes = normed_337_axes_0, epsilon = var_8_to_fp16, x = input_241_cast_fp16)[name = string("normed_337_cast_fp16")]; tensor var_2988_split_sizes_0 = const()[name = string("op_2988_split_sizes_0"), val = tensor([768, 768])]; int32 var_2988_axis_0 = const()[name = string("op_2988_axis_0"), val = int32(-1)]; tensor var_2988_cast_fp16_0, tensor var_2988_cast_fp16_1 = split(axis = var_2988_axis_0, split_sizes = var_2988_split_sizes_0, x = normed_337_cast_fp16)[name = string("op_2988_cast_fp16")]; tensor var_2992_to_fp16 = const()[name = string("op_2992_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301557568)))]; tensor out_145_cast_fp16 = mul(x = var_2988_cast_fp16_0, y = var_2992_to_fp16)[name = string("out_145_cast_fp16")]; tensor var_2998 = const()[name = string("op_2998"), val = tensor([0, 2, 1])]; tensor var_3000_axes_0 = const()[name = string("op_3000_axes_0"), val = tensor([2])]; tensor var_2999_cast_fp16 = transpose(perm = var_2998, x = out_145_cast_fp16)[name = string("transpose_107")]; tensor var_3000_cast_fp16 = expand_dims(axes = var_3000_axes_0, x = var_2999_cast_fp16)[name = string("op_3000_cast_fp16")]; string var_3007_pad_type_0 = const()[name = string("op_3007_pad_type_0"), val = string("valid")]; tensor var_3007_strides_0 = const()[name = string("op_3007_strides_0"), val = tensor([1, 1])]; tensor var_3007_pad_0 = const()[name = string("op_3007_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3007_dilations_0 = const()[name = string("op_3007_dilations_0"), val = tensor([1, 1])]; int32 var_3007_groups_0 = const()[name = string("op_3007_groups_0"), val = int32(1)]; tensor var_3007 = conv(dilations = var_3007_dilations_0, groups = var_3007_groups_0, pad = var_3007_pad_0, pad_type = var_3007_pad_type_0, strides = var_3007_strides_0, weight = encoder_layers_12_self_attn_q_proj_weight_quantized, x = var_3000_cast_fp16)[name = string("op_3007")]; tensor var_3008 = const()[name = string("op_3008"), val = tensor([1, 3, 256, 256])]; tensor var_3009 = reshape(shape = var_3008, x = var_3007)[name = string("op_3009")]; tensor var_3010 = const()[name = string("op_3010"), val = tensor([0, 1, 3, 2])]; string var_3017_pad_type_0 = const()[name = string("op_3017_pad_type_0"), val = string("valid")]; tensor var_3017_strides_0 = const()[name = string("op_3017_strides_0"), val = tensor([1, 1])]; tensor var_3017_pad_0 = const()[name = string("op_3017_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3017_dilations_0 = const()[name = string("op_3017_dilations_0"), val = tensor([1, 1])]; int32 var_3017_groups_0 = const()[name = string("op_3017_groups_0"), val = int32(1)]; tensor var_3017 = conv(dilations = var_3017_dilations_0, groups = var_3017_groups_0, pad = var_3017_pad_0, pad_type = var_3017_pad_type_0, strides = var_3017_strides_0, weight = encoder_layers_12_self_attn_k_proj_weight_quantized, x = var_3000_cast_fp16)[name = string("op_3017")]; tensor var_3018 = const()[name = string("op_3018"), val = tensor([1, 1, 256, 256])]; tensor var_3019 = reshape(shape = var_3018, x = var_3017)[name = string("op_3019")]; tensor var_3020 = const()[name = string("op_3020"), val = tensor([0, 1, 3, 2])]; string var_3027_pad_type_0 = const()[name = string("op_3027_pad_type_0"), val = string("valid")]; tensor var_3027_strides_0 = const()[name = string("op_3027_strides_0"), val = tensor([1, 1])]; tensor var_3027_pad_0 = const()[name = string("op_3027_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3027_dilations_0 = const()[name = string("op_3027_dilations_0"), val = tensor([1, 1])]; int32 var_3027_groups_0 = const()[name = string("op_3027_groups_0"), val = int32(1)]; tensor var_3027 = conv(dilations = var_3027_dilations_0, groups = var_3027_groups_0, pad = var_3027_pad_0, pad_type = var_3027_pad_type_0, strides = var_3027_strides_0, weight = encoder_layers_12_self_attn_v_proj_weight_quantized, x = var_3000_cast_fp16)[name = string("op_3027")]; tensor var_3028 = const()[name = string("op_3028"), val = tensor([1, 1, 256, 256])]; tensor var_3029 = reshape(shape = var_3028, x = var_3027)[name = string("op_3029")]; tensor var_3030 = const()[name = string("op_3030"), val = tensor([0, 1, 3, 2])]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_73 = transpose(perm = var_3010, x = var_3009)[name = string("transpose_106")]; tensor var_3036_cast_fp16 = mul(x = q_73, y = const_170_promoted_to_fp16)[name = string("op_3036_cast_fp16")]; bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; tensor input_245_cast_fp16 = concat(axis = var_22, interleave = input_245_interleave_0, values = (q_73, var_3036_cast_fp16))[name = string("input_245_cast_fp16")]; tensor normed_343_axes_0 = const()[name = string("normed_343_axes_0"), val = tensor([-1])]; tensor normed_343_cast_fp16 = layer_norm(axes = normed_343_axes_0, epsilon = var_8_to_fp16, x = input_245_cast_fp16)[name = string("normed_343_cast_fp16")]; tensor var_3041_split_sizes_0 = const()[name = string("op_3041_split_sizes_0"), val = tensor([256, 256])]; int32 var_3041_axis_0 = const()[name = string("op_3041_axis_0"), val = int32(-1)]; tensor var_3041_cast_fp16_0, tensor var_3041_cast_fp16_1 = split(axis = var_3041_axis_0, split_sizes = var_3041_split_sizes_0, x = normed_343_cast_fp16)[name = string("op_3041_cast_fp16")]; tensor var_3045_to_fp16 = const()[name = string("op_3045_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301559168)))]; tensor out_147_cast_fp16 = mul(x = var_3041_cast_fp16_0, y = var_3045_to_fp16)[name = string("out_147_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_73 = transpose(perm = var_3020, x = var_3019)[name = string("transpose_105")]; tensor var_3052_cast_fp16 = mul(x = k_73, y = const_172_promoted_to_fp16)[name = string("op_3052_cast_fp16")]; bool input_247_interleave_0 = const()[name = string("input_247_interleave_0"), val = bool(false)]; tensor input_247_cast_fp16 = concat(axis = var_22, interleave = input_247_interleave_0, values = (k_73, var_3052_cast_fp16))[name = string("input_247_cast_fp16")]; tensor normed_347_axes_0 = const()[name = string("normed_347_axes_0"), val = tensor([-1])]; tensor normed_347_cast_fp16 = layer_norm(axes = normed_347_axes_0, epsilon = var_8_to_fp16, x = input_247_cast_fp16)[name = string("normed_347_cast_fp16")]; tensor var_3057_split_sizes_0 = const()[name = string("op_3057_split_sizes_0"), val = tensor([256, 256])]; int32 var_3057_axis_0 = const()[name = string("op_3057_axis_0"), val = int32(-1)]; tensor var_3057_cast_fp16_0, tensor var_3057_cast_fp16_1 = split(axis = var_3057_axis_0, split_sizes = var_3057_split_sizes_0, x = normed_347_cast_fp16)[name = string("op_3057_cast_fp16")]; tensor var_3061_to_fp16 = const()[name = string("op_3061_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301559744)))]; tensor out_149_cast_fp16 = mul(x = var_3057_cast_fp16_0, y = var_3061_to_fp16)[name = string("out_149_cast_fp16")]; tensor var_3064 = mul(x = out_147_cast_fp16, y = cos_1_quantized)[name = string("op_3064")]; tensor var_3065_split_sizes_0 = const()[name = string("op_3065_split_sizes_0"), val = tensor([128, 128])]; int32 var_3065_axis_0 = const()[name = string("op_3065_axis_0"), val = int32(-1)]; tensor var_3065_0, tensor var_3065_1 = split(axis = var_3065_axis_0, split_sizes = var_3065_split_sizes_0, x = out_147_cast_fp16)[name = string("op_3065")]; fp16 const_174_promoted = const()[name = string("const_174_promoted"), val = fp16(-0x1p+0)]; tensor var_3067 = mul(x = var_3065_1, y = const_174_promoted)[name = string("op_3067")]; bool var_3069_interleave_0 = const()[name = string("op_3069_interleave_0"), val = bool(false)]; tensor var_3069 = concat(axis = var_22, interleave = var_3069_interleave_0, values = (var_3067, var_3065_0))[name = string("op_3069")]; tensor var_3070 = mul(x = var_3069, y = sin_1_quantized)[name = string("op_3070")]; tensor q_77 = add(x = var_3064, y = var_3070)[name = string("q_77")]; tensor var_3072 = mul(x = out_149_cast_fp16, y = cos_1_quantized)[name = string("op_3072")]; tensor var_3073_split_sizes_0 = const()[name = string("op_3073_split_sizes_0"), val = tensor([128, 128])]; int32 var_3073_axis_0 = const()[name = string("op_3073_axis_0"), val = int32(-1)]; tensor var_3073_0, tensor var_3073_1 = split(axis = var_3073_axis_0, split_sizes = var_3073_split_sizes_0, x = out_149_cast_fp16)[name = string("op_3073")]; fp16 const_175_promoted = const()[name = string("const_175_promoted"), val = fp16(-0x1p+0)]; tensor var_3075 = mul(x = var_3073_1, y = const_175_promoted)[name = string("op_3075")]; bool var_3077_interleave_0 = const()[name = string("op_3077_interleave_0"), val = bool(false)]; tensor var_3077 = concat(axis = var_22, interleave = var_3077_interleave_0, values = (var_3075, var_3073_0))[name = string("op_3077")]; tensor var_3078 = mul(x = var_3077, y = sin_1_quantized)[name = string("op_3078")]; tensor hidden_states_145 = add(x = var_3072, y = var_3078)[name = string("hidden_states_145")]; tensor hidden_states_147_axes_0 = const()[name = string("hidden_states_147_axes_0"), val = tensor([2])]; tensor hidden_states_147 = expand_dims(axes = hidden_states_147_axes_0, x = hidden_states_145)[name = string("hidden_states_147")]; tensor var_3081 = const()[name = string("op_3081"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_149 = tile(reps = var_3081, x = hidden_states_147)[name = string("hidden_states_149")]; tensor var_3083 = const()[name = string("op_3083"), val = tensor([1, 3, 256, 256])]; tensor k_77 = reshape(shape = var_3083, x = hidden_states_149)[name = string("k_77")]; tensor hidden_states_153_axes_0 = const()[name = string("hidden_states_153_axes_0"), val = tensor([2])]; tensor hidden_states_151 = transpose(perm = var_3030, x = var_3029)[name = string("transpose_104")]; tensor hidden_states_153 = expand_dims(axes = hidden_states_153_axes_0, x = hidden_states_151)[name = string("hidden_states_153")]; tensor var_3086 = const()[name = string("op_3086"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_155 = tile(reps = var_3086, x = hidden_states_153)[name = string("hidden_states_155")]; tensor var_3088 = const()[name = string("op_3088"), val = tensor([1, 3, 256, 256])]; tensor v_25 = reshape(shape = var_3088, x = hidden_states_155)[name = string("v_25")]; bool var_3093_transpose_x_1 = const()[name = string("op_3093_transpose_x_1"), val = bool(false)]; bool var_3093_transpose_y_1 = const()[name = string("op_3093_transpose_y_1"), val = bool(true)]; tensor var_3093_cast_fp16 = matmul(transpose_x = var_3093_transpose_x_1, transpose_y = var_3093_transpose_y_1, x = q_77, y = k_77)[name = string("op_3093_cast_fp16")]; fp16 var_3094_to_fp16 = const()[name = string("op_3094_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_73_cast_fp16 = mul(x = var_3093_cast_fp16, y = var_3094_to_fp16)[name = string("attn_weights_73_cast_fp16")]; tensor attn_weights_75_cast_fp16 = add(x = attn_weights_73_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor var_3098_cast_fp16 = softmax(axis = var_22, x = attn_weights_75_cast_fp16)[name = string("op_3098_cast_fp16")]; bool var_3102_transpose_x_0 = const()[name = string("op_3102_transpose_x_0"), val = bool(false)]; bool var_3102_transpose_y_0 = const()[name = string("op_3102_transpose_y_0"), val = bool(false)]; tensor var_3102_cast_fp16 = matmul(transpose_x = var_3102_transpose_x_0, transpose_y = var_3102_transpose_y_0, x = var_3098_cast_fp16, y = v_25)[name = string("op_3102_cast_fp16")]; tensor var_3104 = const()[name = string("op_3104"), val = tensor([0, 2, 1, 3])]; tensor var_3107 = const()[name = string("op_3107"), val = tensor([1, 256, 768])]; tensor var_3105 = transpose(perm = var_3104, x = var_3102_cast_fp16)[name = string("transpose_103")]; tensor attn_out_75 = reshape(shape = var_3107, x = var_3105)[name = string("attn_out_75")]; tensor var_3109 = const()[name = string("op_3109"), val = tensor([0, 2, 1])]; tensor squeeze_12_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(301560320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302150208))))[name = string("squeeze_12_quantized")]; string var_3118_pad_type_0 = const()[name = string("op_3118_pad_type_0"), val = string("valid")]; int32 var_3118_groups_0 = const()[name = string("op_3118_groups_0"), val = int32(1)]; tensor var_3118_strides_0 = const()[name = string("op_3118_strides_0"), val = tensor([1])]; tensor var_3118_pad_0 = const()[name = string("op_3118_pad_0"), val = tensor([0, 0])]; tensor var_3118_dilations_0 = const()[name = string("op_3118_dilations_0"), val = tensor([1])]; tensor var_3110 = transpose(perm = var_3109, x = attn_out_75)[name = string("transpose_102")]; tensor var_3118 = conv(dilations = var_3118_dilations_0, groups = var_3118_groups_0, pad = var_3118_pad_0, pad_type = var_3118_pad_type_0, strides = var_3118_strides_0, weight = squeeze_12_quantized, x = var_3110)[name = string("op_3118")]; tensor var_3119 = const()[name = string("op_3119"), val = tensor([0, 2, 1])]; fp16 const_176_promoted_to_fp16 = const()[name = string("const_176_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_201 = transpose(perm = var_3119, x = var_3118)[name = string("transpose_101")]; tensor var_3123_cast_fp16 = mul(x = x_201, y = const_176_promoted_to_fp16)[name = string("op_3123_cast_fp16")]; bool input_251_interleave_0 = const()[name = string("input_251_interleave_0"), val = bool(false)]; tensor input_251_cast_fp16 = concat(axis = var_22, interleave = input_251_interleave_0, values = (x_201, var_3123_cast_fp16))[name = string("input_251_cast_fp16")]; tensor normed_351_axes_0 = const()[name = string("normed_351_axes_0"), val = tensor([-1])]; tensor normed_351_cast_fp16 = layer_norm(axes = normed_351_axes_0, epsilon = var_8_to_fp16, x = input_251_cast_fp16)[name = string("normed_351_cast_fp16")]; tensor var_3128_split_sizes_0 = const()[name = string("op_3128_split_sizes_0"), val = tensor([768, 768])]; int32 var_3128_axis_0 = const()[name = string("op_3128_axis_0"), val = int32(-1)]; tensor var_3128_cast_fp16_0, tensor var_3128_cast_fp16_1 = split(axis = var_3128_axis_0, split_sizes = var_3128_split_sizes_0, x = normed_351_cast_fp16)[name = string("op_3128_cast_fp16")]; tensor var_3132_to_fp16 = const()[name = string("op_3132_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302151808)))]; tensor out_151_cast_fp16 = mul(x = var_3128_cast_fp16_0, y = var_3132_to_fp16)[name = string("out_151_cast_fp16")]; tensor x_203_cast_fp16 = add(x = x_193_cast_fp16, y = out_151_cast_fp16)[name = string("x_203_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3139_cast_fp16 = mul(x = x_203_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_3139_cast_fp16")]; bool input_253_interleave_0 = const()[name = string("input_253_interleave_0"), val = bool(false)]; tensor input_253_cast_fp16 = concat(axis = var_22, interleave = input_253_interleave_0, values = (x_203_cast_fp16, var_3139_cast_fp16))[name = string("input_253_cast_fp16")]; tensor normed_355_axes_0 = const()[name = string("normed_355_axes_0"), val = tensor([-1])]; tensor normed_355_cast_fp16 = layer_norm(axes = normed_355_axes_0, epsilon = var_8_to_fp16, x = input_253_cast_fp16)[name = string("normed_355_cast_fp16")]; tensor var_3144_split_sizes_0 = const()[name = string("op_3144_split_sizes_0"), val = tensor([768, 768])]; int32 var_3144_axis_0 = const()[name = string("op_3144_axis_0"), val = int32(-1)]; tensor var_3144_cast_fp16_0, tensor var_3144_cast_fp16_1 = split(axis = var_3144_axis_0, split_sizes = var_3144_split_sizes_0, x = normed_355_cast_fp16)[name = string("op_3144_cast_fp16")]; tensor var_3148_to_fp16 = const()[name = string("op_3148_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302153408)))]; tensor out_153_cast_fp16 = mul(x = var_3144_cast_fp16_0, y = var_3148_to_fp16)[name = string("out_153_cast_fp16")]; tensor var_3155 = const()[name = string("op_3155"), val = tensor([0, 2, 1])]; tensor input_255_axes_0 = const()[name = string("input_255_axes_0"), val = tensor([2])]; tensor var_3156 = transpose(perm = var_3155, x = out_153_cast_fp16)[name = string("transpose_100")]; tensor input_255 = expand_dims(axes = input_255_axes_0, x = var_3156)[name = string("input_255")]; string gate_49_pad_type_0 = const()[name = string("gate_49_pad_type_0"), val = string("valid")]; tensor gate_49_strides_0 = const()[name = string("gate_49_strides_0"), val = tensor([1, 1])]; tensor gate_49_pad_0 = const()[name = string("gate_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_49_dilations_0 = const()[name = string("gate_49_dilations_0"), val = tensor([1, 1])]; int32 gate_49_groups_0 = const()[name = string("gate_49_groups_0"), val = int32(1)]; tensor gate_49 = conv(dilations = gate_49_dilations_0, groups = gate_49_groups_0, pad = gate_49_pad_0, pad_type = gate_49_pad_type_0, strides = gate_49_strides_0, weight = encoder_layers_12_mlp_gate_proj_weight_quantized, x = input_255)[name = string("gate_49")]; string up_25_pad_type_0 = const()[name = string("up_25_pad_type_0"), val = string("valid")]; tensor up_25_strides_0 = const()[name = string("up_25_strides_0"), val = tensor([1, 1])]; tensor up_25_pad_0 = const()[name = string("up_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_25_dilations_0 = const()[name = string("up_25_dilations_0"), val = tensor([1, 1])]; int32 up_25_groups_0 = const()[name = string("up_25_groups_0"), val = int32(1)]; tensor up_25 = conv(dilations = up_25_dilations_0, groups = up_25_groups_0, pad = up_25_pad_0, pad_type = up_25_pad_type_0, strides = up_25_strides_0, weight = encoder_layers_12_mlp_up_proj_weight_quantized, x = input_255)[name = string("up_25")]; string gate_51_mode_0 = const()[name = string("gate_51_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_51 = gelu(mode = gate_51_mode_0, x = gate_49)[name = string("gate_51")]; tensor input_257 = mul(x = gate_51, y = up_25)[name = string("input_257")]; string var_3177_pad_type_0 = const()[name = string("op_3177_pad_type_0"), val = string("valid")]; tensor var_3177_strides_0 = const()[name = string("op_3177_strides_0"), val = tensor([1, 1])]; tensor var_3177_pad_0 = const()[name = string("op_3177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3177_dilations_0 = const()[name = string("op_3177_dilations_0"), val = tensor([1, 1])]; int32 var_3177_groups_0 = const()[name = string("op_3177_groups_0"), val = int32(1)]; tensor var_3177 = conv(dilations = var_3177_dilations_0, groups = var_3177_groups_0, pad = var_3177_pad_0, pad_type = var_3177_pad_type_0, strides = var_3177_strides_0, weight = encoder_layers_12_mlp_down_proj_weight_quantized, x = input_257)[name = string("op_3177")]; tensor var_3178_axes_0 = const()[name = string("op_3178_axes_0"), val = tensor([2])]; tensor var_3178 = squeeze(axes = var_3178_axes_0, x = var_3177)[name = string("op_3178")]; tensor var_3179 = const()[name = string("op_3179"), val = tensor([0, 2, 1])]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_207 = transpose(perm = var_3179, x = var_3178)[name = string("transpose_99")]; tensor var_3183_cast_fp16 = mul(x = x_207, y = const_180_promoted_to_fp16)[name = string("op_3183_cast_fp16")]; bool input_259_interleave_0 = const()[name = string("input_259_interleave_0"), val = bool(false)]; tensor input_259_cast_fp16 = concat(axis = var_22, interleave = input_259_interleave_0, values = (x_207, var_3183_cast_fp16))[name = string("input_259_cast_fp16")]; tensor normed_361_axes_0 = const()[name = string("normed_361_axes_0"), val = tensor([-1])]; tensor normed_361_cast_fp16 = layer_norm(axes = normed_361_axes_0, epsilon = var_8_to_fp16, x = input_259_cast_fp16)[name = string("normed_361_cast_fp16")]; tensor var_3188_split_sizes_0 = const()[name = string("op_3188_split_sizes_0"), val = tensor([768, 768])]; int32 var_3188_axis_0 = const()[name = string("op_3188_axis_0"), val = int32(-1)]; tensor var_3188_cast_fp16_0, tensor var_3188_cast_fp16_1 = split(axis = var_3188_axis_0, split_sizes = var_3188_split_sizes_0, x = normed_361_cast_fp16)[name = string("op_3188_cast_fp16")]; tensor var_3192_to_fp16 = const()[name = string("op_3192_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302155008)))]; tensor out_155_cast_fp16 = mul(x = var_3188_cast_fp16_0, y = var_3192_to_fp16)[name = string("out_155_cast_fp16")]; tensor x_209_cast_fp16 = add(x = x_203_cast_fp16, y = out_155_cast_fp16)[name = string("x_209_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3221_cast_fp16 = mul(x = x_209_cast_fp16, y = const_182_promoted_to_fp16)[name = string("op_3221_cast_fp16")]; bool input_261_interleave_0 = const()[name = string("input_261_interleave_0"), val = bool(false)]; tensor input_261_cast_fp16 = concat(axis = var_22, interleave = input_261_interleave_0, values = (x_209_cast_fp16, var_3221_cast_fp16))[name = string("input_261_cast_fp16")]; tensor normed_365_axes_0 = const()[name = string("normed_365_axes_0"), val = tensor([-1])]; tensor normed_365_cast_fp16 = layer_norm(axes = normed_365_axes_0, epsilon = var_8_to_fp16, x = input_261_cast_fp16)[name = string("normed_365_cast_fp16")]; tensor var_3226_split_sizes_0 = const()[name = string("op_3226_split_sizes_0"), val = tensor([768, 768])]; int32 var_3226_axis_0 = const()[name = string("op_3226_axis_0"), val = int32(-1)]; tensor var_3226_cast_fp16_0, tensor var_3226_cast_fp16_1 = split(axis = var_3226_axis_0, split_sizes = var_3226_split_sizes_0, x = normed_365_cast_fp16)[name = string("op_3226_cast_fp16")]; tensor var_3230_to_fp16 = const()[name = string("op_3230_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302156608)))]; tensor out_157_cast_fp16 = mul(x = var_3226_cast_fp16_0, y = var_3230_to_fp16)[name = string("out_157_cast_fp16")]; tensor var_3236 = const()[name = string("op_3236"), val = tensor([0, 2, 1])]; tensor var_3238_axes_0 = const()[name = string("op_3238_axes_0"), val = tensor([2])]; tensor var_3237_cast_fp16 = transpose(perm = var_3236, x = out_157_cast_fp16)[name = string("transpose_98")]; tensor var_3238_cast_fp16 = expand_dims(axes = var_3238_axes_0, x = var_3237_cast_fp16)[name = string("op_3238_cast_fp16")]; string var_3245_pad_type_0 = const()[name = string("op_3245_pad_type_0"), val = string("valid")]; tensor var_3245_strides_0 = const()[name = string("op_3245_strides_0"), val = tensor([1, 1])]; tensor var_3245_pad_0 = const()[name = string("op_3245_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3245_dilations_0 = const()[name = string("op_3245_dilations_0"), val = tensor([1, 1])]; int32 var_3245_groups_0 = const()[name = string("op_3245_groups_0"), val = int32(1)]; tensor var_3245 = conv(dilations = var_3245_dilations_0, groups = var_3245_groups_0, pad = var_3245_pad_0, pad_type = var_3245_pad_type_0, strides = var_3245_strides_0, weight = encoder_layers_13_self_attn_q_proj_weight_quantized, x = var_3238_cast_fp16)[name = string("op_3245")]; tensor var_3246 = const()[name = string("op_3246"), val = tensor([1, 3, 256, 256])]; tensor var_3247 = reshape(shape = var_3246, x = var_3245)[name = string("op_3247")]; tensor var_3248 = const()[name = string("op_3248"), val = tensor([0, 1, 3, 2])]; string var_3255_pad_type_0 = const()[name = string("op_3255_pad_type_0"), val = string("valid")]; tensor var_3255_strides_0 = const()[name = string("op_3255_strides_0"), val = tensor([1, 1])]; tensor var_3255_pad_0 = const()[name = string("op_3255_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3255_dilations_0 = const()[name = string("op_3255_dilations_0"), val = tensor([1, 1])]; int32 var_3255_groups_0 = const()[name = string("op_3255_groups_0"), val = int32(1)]; tensor var_3255 = conv(dilations = var_3255_dilations_0, groups = var_3255_groups_0, pad = var_3255_pad_0, pad_type = var_3255_pad_type_0, strides = var_3255_strides_0, weight = encoder_layers_13_self_attn_k_proj_weight_quantized, x = var_3238_cast_fp16)[name = string("op_3255")]; tensor var_3256 = const()[name = string("op_3256"), val = tensor([1, 1, 256, 256])]; tensor var_3257 = reshape(shape = var_3256, x = var_3255)[name = string("op_3257")]; tensor var_3258 = const()[name = string("op_3258"), val = tensor([0, 1, 3, 2])]; string var_3265_pad_type_0 = const()[name = string("op_3265_pad_type_0"), val = string("valid")]; tensor var_3265_strides_0 = const()[name = string("op_3265_strides_0"), val = tensor([1, 1])]; tensor var_3265_pad_0 = const()[name = string("op_3265_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3265_dilations_0 = const()[name = string("op_3265_dilations_0"), val = tensor([1, 1])]; int32 var_3265_groups_0 = const()[name = string("op_3265_groups_0"), val = int32(1)]; tensor var_3265 = conv(dilations = var_3265_dilations_0, groups = var_3265_groups_0, pad = var_3265_pad_0, pad_type = var_3265_pad_type_0, strides = var_3265_strides_0, weight = encoder_layers_13_self_attn_v_proj_weight_quantized, x = var_3238_cast_fp16)[name = string("op_3265")]; tensor var_3266 = const()[name = string("op_3266"), val = tensor([1, 1, 256, 256])]; tensor var_3267 = reshape(shape = var_3266, x = var_3265)[name = string("op_3267")]; tensor var_3268 = const()[name = string("op_3268"), val = tensor([0, 1, 3, 2])]; fp16 const_184_promoted_to_fp16 = const()[name = string("const_184_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_79 = transpose(perm = var_3248, x = var_3247)[name = string("transpose_97")]; tensor var_3274_cast_fp16 = mul(x = q_79, y = const_184_promoted_to_fp16)[name = string("op_3274_cast_fp16")]; bool input_265_interleave_0 = const()[name = string("input_265_interleave_0"), val = bool(false)]; tensor input_265_cast_fp16 = concat(axis = var_22, interleave = input_265_interleave_0, values = (q_79, var_3274_cast_fp16))[name = string("input_265_cast_fp16")]; tensor normed_371_axes_0 = const()[name = string("normed_371_axes_0"), val = tensor([-1])]; tensor normed_371_cast_fp16 = layer_norm(axes = normed_371_axes_0, epsilon = var_8_to_fp16, x = input_265_cast_fp16)[name = string("normed_371_cast_fp16")]; tensor var_3279_split_sizes_0 = const()[name = string("op_3279_split_sizes_0"), val = tensor([256, 256])]; int32 var_3279_axis_0 = const()[name = string("op_3279_axis_0"), val = int32(-1)]; tensor var_3279_cast_fp16_0, tensor var_3279_cast_fp16_1 = split(axis = var_3279_axis_0, split_sizes = var_3279_split_sizes_0, x = normed_371_cast_fp16)[name = string("op_3279_cast_fp16")]; tensor var_3283_to_fp16 = const()[name = string("op_3283_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302158208)))]; tensor out_159_cast_fp16 = mul(x = var_3279_cast_fp16_0, y = var_3283_to_fp16)[name = string("out_159_cast_fp16")]; fp16 const_186_promoted_to_fp16 = const()[name = string("const_186_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_79 = transpose(perm = var_3258, x = var_3257)[name = string("transpose_96")]; tensor var_3290_cast_fp16 = mul(x = k_79, y = const_186_promoted_to_fp16)[name = string("op_3290_cast_fp16")]; bool input_267_interleave_0 = const()[name = string("input_267_interleave_0"), val = bool(false)]; tensor input_267_cast_fp16 = concat(axis = var_22, interleave = input_267_interleave_0, values = (k_79, var_3290_cast_fp16))[name = string("input_267_cast_fp16")]; tensor normed_375_axes_0 = const()[name = string("normed_375_axes_0"), val = tensor([-1])]; tensor normed_375_cast_fp16 = layer_norm(axes = normed_375_axes_0, epsilon = var_8_to_fp16, x = input_267_cast_fp16)[name = string("normed_375_cast_fp16")]; tensor var_3295_split_sizes_0 = const()[name = string("op_3295_split_sizes_0"), val = tensor([256, 256])]; int32 var_3295_axis_0 = const()[name = string("op_3295_axis_0"), val = int32(-1)]; tensor var_3295_cast_fp16_0, tensor var_3295_cast_fp16_1 = split(axis = var_3295_axis_0, split_sizes = var_3295_split_sizes_0, x = normed_375_cast_fp16)[name = string("op_3295_cast_fp16")]; tensor var_3299_to_fp16 = const()[name = string("op_3299_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302158784)))]; tensor out_161_cast_fp16 = mul(x = var_3295_cast_fp16_0, y = var_3299_to_fp16)[name = string("out_161_cast_fp16")]; tensor var_3302 = mul(x = out_159_cast_fp16, y = cos_1_quantized)[name = string("op_3302")]; tensor var_3303_split_sizes_0 = const()[name = string("op_3303_split_sizes_0"), val = tensor([128, 128])]; int32 var_3303_axis_0 = const()[name = string("op_3303_axis_0"), val = int32(-1)]; tensor var_3303_0, tensor var_3303_1 = split(axis = var_3303_axis_0, split_sizes = var_3303_split_sizes_0, x = out_159_cast_fp16)[name = string("op_3303")]; fp16 const_188_promoted = const()[name = string("const_188_promoted"), val = fp16(-0x1p+0)]; tensor var_3305 = mul(x = var_3303_1, y = const_188_promoted)[name = string("op_3305")]; bool var_3307_interleave_0 = const()[name = string("op_3307_interleave_0"), val = bool(false)]; tensor var_3307 = concat(axis = var_22, interleave = var_3307_interleave_0, values = (var_3305, var_3303_0))[name = string("op_3307")]; tensor var_3308 = mul(x = var_3307, y = sin_1_quantized)[name = string("op_3308")]; tensor q_83 = add(x = var_3302, y = var_3308)[name = string("q_83")]; tensor var_3310 = mul(x = out_161_cast_fp16, y = cos_1_quantized)[name = string("op_3310")]; tensor var_3311_split_sizes_0 = const()[name = string("op_3311_split_sizes_0"), val = tensor([128, 128])]; int32 var_3311_axis_0 = const()[name = string("op_3311_axis_0"), val = int32(-1)]; tensor var_3311_0, tensor var_3311_1 = split(axis = var_3311_axis_0, split_sizes = var_3311_split_sizes_0, x = out_161_cast_fp16)[name = string("op_3311")]; fp16 const_189_promoted = const()[name = string("const_189_promoted"), val = fp16(-0x1p+0)]; tensor var_3313 = mul(x = var_3311_1, y = const_189_promoted)[name = string("op_3313")]; bool var_3315_interleave_0 = const()[name = string("op_3315_interleave_0"), val = bool(false)]; tensor var_3315 = concat(axis = var_22, interleave = var_3315_interleave_0, values = (var_3313, var_3311_0))[name = string("op_3315")]; tensor var_3316 = mul(x = var_3315, y = sin_1_quantized)[name = string("op_3316")]; tensor hidden_states_157 = add(x = var_3310, y = var_3316)[name = string("hidden_states_157")]; tensor hidden_states_159_axes_0 = const()[name = string("hidden_states_159_axes_0"), val = tensor([2])]; tensor hidden_states_159 = expand_dims(axes = hidden_states_159_axes_0, x = hidden_states_157)[name = string("hidden_states_159")]; tensor var_3319 = const()[name = string("op_3319"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_161 = tile(reps = var_3319, x = hidden_states_159)[name = string("hidden_states_161")]; tensor var_3321 = const()[name = string("op_3321"), val = tensor([1, 3, 256, 256])]; tensor k_83 = reshape(shape = var_3321, x = hidden_states_161)[name = string("k_83")]; tensor hidden_states_165_axes_0 = const()[name = string("hidden_states_165_axes_0"), val = tensor([2])]; tensor hidden_states_163 = transpose(perm = var_3268, x = var_3267)[name = string("transpose_95")]; tensor hidden_states_165 = expand_dims(axes = hidden_states_165_axes_0, x = hidden_states_163)[name = string("hidden_states_165")]; tensor var_3324 = const()[name = string("op_3324"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_167 = tile(reps = var_3324, x = hidden_states_165)[name = string("hidden_states_167")]; tensor var_3326 = const()[name = string("op_3326"), val = tensor([1, 3, 256, 256])]; tensor v_27 = reshape(shape = var_3326, x = hidden_states_167)[name = string("v_27")]; bool var_3331_transpose_x_1 = const()[name = string("op_3331_transpose_x_1"), val = bool(false)]; bool var_3331_transpose_y_1 = const()[name = string("op_3331_transpose_y_1"), val = bool(true)]; tensor var_3331_cast_fp16 = matmul(transpose_x = var_3331_transpose_x_1, transpose_y = var_3331_transpose_y_1, x = q_83, y = k_83)[name = string("op_3331_cast_fp16")]; fp16 var_3332_to_fp16 = const()[name = string("op_3332_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_79_cast_fp16 = mul(x = var_3331_cast_fp16, y = var_3332_to_fp16)[name = string("attn_weights_79_cast_fp16")]; tensor attn_weights_81_cast_fp16 = add(x = attn_weights_79_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_81_cast_fp16")]; tensor var_3336_cast_fp16 = softmax(axis = var_22, x = attn_weights_81_cast_fp16)[name = string("op_3336_cast_fp16")]; bool var_3340_transpose_x_0 = const()[name = string("op_3340_transpose_x_0"), val = bool(false)]; bool var_3340_transpose_y_0 = const()[name = string("op_3340_transpose_y_0"), val = bool(false)]; tensor var_3340_cast_fp16 = matmul(transpose_x = var_3340_transpose_x_0, transpose_y = var_3340_transpose_y_0, x = var_3336_cast_fp16, y = v_27)[name = string("op_3340_cast_fp16")]; tensor var_3342 = const()[name = string("op_3342"), val = tensor([0, 2, 1, 3])]; tensor var_3345 = const()[name = string("op_3345"), val = tensor([1, 256, 768])]; tensor var_3343 = transpose(perm = var_3342, x = var_3340_cast_fp16)[name = string("transpose_94")]; tensor attn_out_81 = reshape(shape = var_3345, x = var_3343)[name = string("attn_out_81")]; tensor var_3347 = const()[name = string("op_3347"), val = tensor([0, 2, 1])]; tensor squeeze_13_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302159360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302749248))))[name = string("squeeze_13_quantized")]; string var_3356_pad_type_0 = const()[name = string("op_3356_pad_type_0"), val = string("valid")]; int32 var_3356_groups_0 = const()[name = string("op_3356_groups_0"), val = int32(1)]; tensor var_3356_strides_0 = const()[name = string("op_3356_strides_0"), val = tensor([1])]; tensor var_3356_pad_0 = const()[name = string("op_3356_pad_0"), val = tensor([0, 0])]; tensor var_3356_dilations_0 = const()[name = string("op_3356_dilations_0"), val = tensor([1])]; tensor var_3348 = transpose(perm = var_3347, x = attn_out_81)[name = string("transpose_93")]; tensor var_3356 = conv(dilations = var_3356_dilations_0, groups = var_3356_groups_0, pad = var_3356_pad_0, pad_type = var_3356_pad_type_0, strides = var_3356_strides_0, weight = squeeze_13_quantized, x = var_3348)[name = string("op_3356")]; tensor var_3357 = const()[name = string("op_3357"), val = tensor([0, 2, 1])]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_217 = transpose(perm = var_3357, x = var_3356)[name = string("transpose_92")]; tensor var_3361_cast_fp16 = mul(x = x_217, y = const_190_promoted_to_fp16)[name = string("op_3361_cast_fp16")]; bool input_271_interleave_0 = const()[name = string("input_271_interleave_0"), val = bool(false)]; tensor input_271_cast_fp16 = concat(axis = var_22, interleave = input_271_interleave_0, values = (x_217, var_3361_cast_fp16))[name = string("input_271_cast_fp16")]; tensor normed_379_axes_0 = const()[name = string("normed_379_axes_0"), val = tensor([-1])]; tensor normed_379_cast_fp16 = layer_norm(axes = normed_379_axes_0, epsilon = var_8_to_fp16, x = input_271_cast_fp16)[name = string("normed_379_cast_fp16")]; tensor var_3366_split_sizes_0 = const()[name = string("op_3366_split_sizes_0"), val = tensor([768, 768])]; int32 var_3366_axis_0 = const()[name = string("op_3366_axis_0"), val = int32(-1)]; tensor var_3366_cast_fp16_0, tensor var_3366_cast_fp16_1 = split(axis = var_3366_axis_0, split_sizes = var_3366_split_sizes_0, x = normed_379_cast_fp16)[name = string("op_3366_cast_fp16")]; tensor var_3370_to_fp16 = const()[name = string("op_3370_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302750848)))]; tensor out_163_cast_fp16 = mul(x = var_3366_cast_fp16_0, y = var_3370_to_fp16)[name = string("out_163_cast_fp16")]; tensor x_219_cast_fp16 = add(x = x_209_cast_fp16, y = out_163_cast_fp16)[name = string("x_219_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3377_cast_fp16 = mul(x = x_219_cast_fp16, y = const_192_promoted_to_fp16)[name = string("op_3377_cast_fp16")]; bool input_273_interleave_0 = const()[name = string("input_273_interleave_0"), val = bool(false)]; tensor input_273_cast_fp16 = concat(axis = var_22, interleave = input_273_interleave_0, values = (x_219_cast_fp16, var_3377_cast_fp16))[name = string("input_273_cast_fp16")]; tensor normed_383_axes_0 = const()[name = string("normed_383_axes_0"), val = tensor([-1])]; tensor normed_383_cast_fp16 = layer_norm(axes = normed_383_axes_0, epsilon = var_8_to_fp16, x = input_273_cast_fp16)[name = string("normed_383_cast_fp16")]; tensor var_3382_split_sizes_0 = const()[name = string("op_3382_split_sizes_0"), val = tensor([768, 768])]; int32 var_3382_axis_0 = const()[name = string("op_3382_axis_0"), val = int32(-1)]; tensor var_3382_cast_fp16_0, tensor var_3382_cast_fp16_1 = split(axis = var_3382_axis_0, split_sizes = var_3382_split_sizes_0, x = normed_383_cast_fp16)[name = string("op_3382_cast_fp16")]; tensor var_3386_to_fp16 = const()[name = string("op_3386_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302752448)))]; tensor out_165_cast_fp16 = mul(x = var_3382_cast_fp16_0, y = var_3386_to_fp16)[name = string("out_165_cast_fp16")]; tensor var_3393 = const()[name = string("op_3393"), val = tensor([0, 2, 1])]; tensor input_275_axes_0 = const()[name = string("input_275_axes_0"), val = tensor([2])]; tensor var_3394 = transpose(perm = var_3393, x = out_165_cast_fp16)[name = string("transpose_91")]; tensor input_275 = expand_dims(axes = input_275_axes_0, x = var_3394)[name = string("input_275")]; string gate_53_pad_type_0 = const()[name = string("gate_53_pad_type_0"), val = string("valid")]; tensor gate_53_strides_0 = const()[name = string("gate_53_strides_0"), val = tensor([1, 1])]; tensor gate_53_pad_0 = const()[name = string("gate_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_53_dilations_0 = const()[name = string("gate_53_dilations_0"), val = tensor([1, 1])]; int32 gate_53_groups_0 = const()[name = string("gate_53_groups_0"), val = int32(1)]; tensor gate_53 = conv(dilations = gate_53_dilations_0, groups = gate_53_groups_0, pad = gate_53_pad_0, pad_type = gate_53_pad_type_0, strides = gate_53_strides_0, weight = encoder_layers_13_mlp_gate_proj_weight_quantized, x = input_275)[name = string("gate_53")]; string up_27_pad_type_0 = const()[name = string("up_27_pad_type_0"), val = string("valid")]; tensor up_27_strides_0 = const()[name = string("up_27_strides_0"), val = tensor([1, 1])]; tensor up_27_pad_0 = const()[name = string("up_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_27_dilations_0 = const()[name = string("up_27_dilations_0"), val = tensor([1, 1])]; int32 up_27_groups_0 = const()[name = string("up_27_groups_0"), val = int32(1)]; tensor up_27 = conv(dilations = up_27_dilations_0, groups = up_27_groups_0, pad = up_27_pad_0, pad_type = up_27_pad_type_0, strides = up_27_strides_0, weight = encoder_layers_13_mlp_up_proj_weight_quantized, x = input_275)[name = string("up_27")]; string gate_55_mode_0 = const()[name = string("gate_55_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_55 = gelu(mode = gate_55_mode_0, x = gate_53)[name = string("gate_55")]; tensor input_277 = mul(x = gate_55, y = up_27)[name = string("input_277")]; string var_3415_pad_type_0 = const()[name = string("op_3415_pad_type_0"), val = string("valid")]; tensor var_3415_strides_0 = const()[name = string("op_3415_strides_0"), val = tensor([1, 1])]; tensor var_3415_pad_0 = const()[name = string("op_3415_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3415_dilations_0 = const()[name = string("op_3415_dilations_0"), val = tensor([1, 1])]; int32 var_3415_groups_0 = const()[name = string("op_3415_groups_0"), val = int32(1)]; tensor var_3415 = conv(dilations = var_3415_dilations_0, groups = var_3415_groups_0, pad = var_3415_pad_0, pad_type = var_3415_pad_type_0, strides = var_3415_strides_0, weight = encoder_layers_13_mlp_down_proj_weight_quantized, x = input_277)[name = string("op_3415")]; tensor var_3416_axes_0 = const()[name = string("op_3416_axes_0"), val = tensor([2])]; tensor var_3416 = squeeze(axes = var_3416_axes_0, x = var_3415)[name = string("op_3416")]; tensor var_3417 = const()[name = string("op_3417"), val = tensor([0, 2, 1])]; fp16 const_194_promoted_to_fp16 = const()[name = string("const_194_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_223 = transpose(perm = var_3417, x = var_3416)[name = string("transpose_90")]; tensor var_3421_cast_fp16 = mul(x = x_223, y = const_194_promoted_to_fp16)[name = string("op_3421_cast_fp16")]; bool input_279_interleave_0 = const()[name = string("input_279_interleave_0"), val = bool(false)]; tensor input_279_cast_fp16 = concat(axis = var_22, interleave = input_279_interleave_0, values = (x_223, var_3421_cast_fp16))[name = string("input_279_cast_fp16")]; tensor normed_389_axes_0 = const()[name = string("normed_389_axes_0"), val = tensor([-1])]; tensor normed_389_cast_fp16 = layer_norm(axes = normed_389_axes_0, epsilon = var_8_to_fp16, x = input_279_cast_fp16)[name = string("normed_389_cast_fp16")]; tensor var_3426_split_sizes_0 = const()[name = string("op_3426_split_sizes_0"), val = tensor([768, 768])]; int32 var_3426_axis_0 = const()[name = string("op_3426_axis_0"), val = int32(-1)]; tensor var_3426_cast_fp16_0, tensor var_3426_cast_fp16_1 = split(axis = var_3426_axis_0, split_sizes = var_3426_split_sizes_0, x = normed_389_cast_fp16)[name = string("op_3426_cast_fp16")]; tensor var_3430_to_fp16 = const()[name = string("op_3430_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302754048)))]; tensor out_167_cast_fp16 = mul(x = var_3426_cast_fp16_0, y = var_3430_to_fp16)[name = string("out_167_cast_fp16")]; tensor x_225_cast_fp16 = add(x = x_219_cast_fp16, y = out_167_cast_fp16)[name = string("x_225_cast_fp16")]; fp16 const_196_promoted_to_fp16 = const()[name = string("const_196_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3459_cast_fp16 = mul(x = x_225_cast_fp16, y = const_196_promoted_to_fp16)[name = string("op_3459_cast_fp16")]; bool input_281_interleave_0 = const()[name = string("input_281_interleave_0"), val = bool(false)]; tensor input_281_cast_fp16 = concat(axis = var_22, interleave = input_281_interleave_0, values = (x_225_cast_fp16, var_3459_cast_fp16))[name = string("input_281_cast_fp16")]; tensor normed_393_axes_0 = const()[name = string("normed_393_axes_0"), val = tensor([-1])]; tensor normed_393_cast_fp16 = layer_norm(axes = normed_393_axes_0, epsilon = var_8_to_fp16, x = input_281_cast_fp16)[name = string("normed_393_cast_fp16")]; tensor var_3464_split_sizes_0 = const()[name = string("op_3464_split_sizes_0"), val = tensor([768, 768])]; int32 var_3464_axis_0 = const()[name = string("op_3464_axis_0"), val = int32(-1)]; tensor var_3464_cast_fp16_0, tensor var_3464_cast_fp16_1 = split(axis = var_3464_axis_0, split_sizes = var_3464_split_sizes_0, x = normed_393_cast_fp16)[name = string("op_3464_cast_fp16")]; tensor var_3468_to_fp16 = const()[name = string("op_3468_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302755648)))]; tensor out_169_cast_fp16 = mul(x = var_3464_cast_fp16_0, y = var_3468_to_fp16)[name = string("out_169_cast_fp16")]; tensor var_3474 = const()[name = string("op_3474"), val = tensor([0, 2, 1])]; tensor var_3476_axes_0 = const()[name = string("op_3476_axes_0"), val = tensor([2])]; tensor var_3475_cast_fp16 = transpose(perm = var_3474, x = out_169_cast_fp16)[name = string("transpose_89")]; tensor var_3476_cast_fp16 = expand_dims(axes = var_3476_axes_0, x = var_3475_cast_fp16)[name = string("op_3476_cast_fp16")]; string var_3483_pad_type_0 = const()[name = string("op_3483_pad_type_0"), val = string("valid")]; tensor var_3483_strides_0 = const()[name = string("op_3483_strides_0"), val = tensor([1, 1])]; tensor var_3483_pad_0 = const()[name = string("op_3483_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3483_dilations_0 = const()[name = string("op_3483_dilations_0"), val = tensor([1, 1])]; int32 var_3483_groups_0 = const()[name = string("op_3483_groups_0"), val = int32(1)]; tensor var_3483 = conv(dilations = var_3483_dilations_0, groups = var_3483_groups_0, pad = var_3483_pad_0, pad_type = var_3483_pad_type_0, strides = var_3483_strides_0, weight = encoder_layers_14_self_attn_q_proj_weight_quantized, x = var_3476_cast_fp16)[name = string("op_3483")]; tensor var_3484 = const()[name = string("op_3484"), val = tensor([1, 3, 256, 256])]; tensor var_3485 = reshape(shape = var_3484, x = var_3483)[name = string("op_3485")]; tensor var_3486 = const()[name = string("op_3486"), val = tensor([0, 1, 3, 2])]; string var_3493_pad_type_0 = const()[name = string("op_3493_pad_type_0"), val = string("valid")]; tensor var_3493_strides_0 = const()[name = string("op_3493_strides_0"), val = tensor([1, 1])]; tensor var_3493_pad_0 = const()[name = string("op_3493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3493_dilations_0 = const()[name = string("op_3493_dilations_0"), val = tensor([1, 1])]; int32 var_3493_groups_0 = const()[name = string("op_3493_groups_0"), val = int32(1)]; tensor var_3493 = conv(dilations = var_3493_dilations_0, groups = var_3493_groups_0, pad = var_3493_pad_0, pad_type = var_3493_pad_type_0, strides = var_3493_strides_0, weight = encoder_layers_14_self_attn_k_proj_weight_quantized, x = var_3476_cast_fp16)[name = string("op_3493")]; tensor var_3494 = const()[name = string("op_3494"), val = tensor([1, 1, 256, 256])]; tensor var_3495 = reshape(shape = var_3494, x = var_3493)[name = string("op_3495")]; tensor var_3496 = const()[name = string("op_3496"), val = tensor([0, 1, 3, 2])]; string var_3503_pad_type_0 = const()[name = string("op_3503_pad_type_0"), val = string("valid")]; tensor var_3503_strides_0 = const()[name = string("op_3503_strides_0"), val = tensor([1, 1])]; tensor var_3503_pad_0 = const()[name = string("op_3503_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3503_dilations_0 = const()[name = string("op_3503_dilations_0"), val = tensor([1, 1])]; int32 var_3503_groups_0 = const()[name = string("op_3503_groups_0"), val = int32(1)]; tensor var_3503 = conv(dilations = var_3503_dilations_0, groups = var_3503_groups_0, pad = var_3503_pad_0, pad_type = var_3503_pad_type_0, strides = var_3503_strides_0, weight = encoder_layers_14_self_attn_v_proj_weight_quantized, x = var_3476_cast_fp16)[name = string("op_3503")]; tensor var_3504 = const()[name = string("op_3504"), val = tensor([1, 1, 256, 256])]; tensor var_3505 = reshape(shape = var_3504, x = var_3503)[name = string("op_3505")]; tensor var_3506 = const()[name = string("op_3506"), val = tensor([0, 1, 3, 2])]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_85 = transpose(perm = var_3486, x = var_3485)[name = string("transpose_88")]; tensor var_3512_cast_fp16 = mul(x = q_85, y = const_198_promoted_to_fp16)[name = string("op_3512_cast_fp16")]; bool input_285_interleave_0 = const()[name = string("input_285_interleave_0"), val = bool(false)]; tensor input_285_cast_fp16 = concat(axis = var_22, interleave = input_285_interleave_0, values = (q_85, var_3512_cast_fp16))[name = string("input_285_cast_fp16")]; tensor normed_399_axes_0 = const()[name = string("normed_399_axes_0"), val = tensor([-1])]; tensor normed_399_cast_fp16 = layer_norm(axes = normed_399_axes_0, epsilon = var_8_to_fp16, x = input_285_cast_fp16)[name = string("normed_399_cast_fp16")]; tensor var_3517_split_sizes_0 = const()[name = string("op_3517_split_sizes_0"), val = tensor([256, 256])]; int32 var_3517_axis_0 = const()[name = string("op_3517_axis_0"), val = int32(-1)]; tensor var_3517_cast_fp16_0, tensor var_3517_cast_fp16_1 = split(axis = var_3517_axis_0, split_sizes = var_3517_split_sizes_0, x = normed_399_cast_fp16)[name = string("op_3517_cast_fp16")]; tensor var_3521_to_fp16 = const()[name = string("op_3521_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302757248)))]; tensor out_171_cast_fp16 = mul(x = var_3517_cast_fp16_0, y = var_3521_to_fp16)[name = string("out_171_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_85 = transpose(perm = var_3496, x = var_3495)[name = string("transpose_87")]; tensor var_3528_cast_fp16 = mul(x = k_85, y = const_200_promoted_to_fp16)[name = string("op_3528_cast_fp16")]; bool input_287_interleave_0 = const()[name = string("input_287_interleave_0"), val = bool(false)]; tensor input_287_cast_fp16 = concat(axis = var_22, interleave = input_287_interleave_0, values = (k_85, var_3528_cast_fp16))[name = string("input_287_cast_fp16")]; tensor normed_403_axes_0 = const()[name = string("normed_403_axes_0"), val = tensor([-1])]; tensor normed_403_cast_fp16 = layer_norm(axes = normed_403_axes_0, epsilon = var_8_to_fp16, x = input_287_cast_fp16)[name = string("normed_403_cast_fp16")]; tensor var_3533_split_sizes_0 = const()[name = string("op_3533_split_sizes_0"), val = tensor([256, 256])]; int32 var_3533_axis_0 = const()[name = string("op_3533_axis_0"), val = int32(-1)]; tensor var_3533_cast_fp16_0, tensor var_3533_cast_fp16_1 = split(axis = var_3533_axis_0, split_sizes = var_3533_split_sizes_0, x = normed_403_cast_fp16)[name = string("op_3533_cast_fp16")]; tensor var_3537_to_fp16 = const()[name = string("op_3537_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302757824)))]; tensor out_173_cast_fp16 = mul(x = var_3533_cast_fp16_0, y = var_3537_to_fp16)[name = string("out_173_cast_fp16")]; tensor var_3540 = mul(x = out_171_cast_fp16, y = cos_1_quantized)[name = string("op_3540")]; tensor var_3541_split_sizes_0 = const()[name = string("op_3541_split_sizes_0"), val = tensor([128, 128])]; int32 var_3541_axis_0 = const()[name = string("op_3541_axis_0"), val = int32(-1)]; tensor var_3541_0, tensor var_3541_1 = split(axis = var_3541_axis_0, split_sizes = var_3541_split_sizes_0, x = out_171_cast_fp16)[name = string("op_3541")]; fp16 const_202_promoted = const()[name = string("const_202_promoted"), val = fp16(-0x1p+0)]; tensor var_3543 = mul(x = var_3541_1, y = const_202_promoted)[name = string("op_3543")]; bool var_3545_interleave_0 = const()[name = string("op_3545_interleave_0"), val = bool(false)]; tensor var_3545 = concat(axis = var_22, interleave = var_3545_interleave_0, values = (var_3543, var_3541_0))[name = string("op_3545")]; tensor var_3546 = mul(x = var_3545, y = sin_1_quantized)[name = string("op_3546")]; tensor q_89 = add(x = var_3540, y = var_3546)[name = string("q_89")]; tensor var_3548 = mul(x = out_173_cast_fp16, y = cos_1_quantized)[name = string("op_3548")]; tensor var_3549_split_sizes_0 = const()[name = string("op_3549_split_sizes_0"), val = tensor([128, 128])]; int32 var_3549_axis_0 = const()[name = string("op_3549_axis_0"), val = int32(-1)]; tensor var_3549_0, tensor var_3549_1 = split(axis = var_3549_axis_0, split_sizes = var_3549_split_sizes_0, x = out_173_cast_fp16)[name = string("op_3549")]; fp16 const_203_promoted = const()[name = string("const_203_promoted"), val = fp16(-0x1p+0)]; tensor var_3551 = mul(x = var_3549_1, y = const_203_promoted)[name = string("op_3551")]; bool var_3553_interleave_0 = const()[name = string("op_3553_interleave_0"), val = bool(false)]; tensor var_3553 = concat(axis = var_22, interleave = var_3553_interleave_0, values = (var_3551, var_3549_0))[name = string("op_3553")]; tensor var_3554 = mul(x = var_3553, y = sin_1_quantized)[name = string("op_3554")]; tensor hidden_states_169 = add(x = var_3548, y = var_3554)[name = string("hidden_states_169")]; tensor hidden_states_171_axes_0 = const()[name = string("hidden_states_171_axes_0"), val = tensor([2])]; tensor hidden_states_171 = expand_dims(axes = hidden_states_171_axes_0, x = hidden_states_169)[name = string("hidden_states_171")]; tensor var_3557 = const()[name = string("op_3557"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_173 = tile(reps = var_3557, x = hidden_states_171)[name = string("hidden_states_173")]; tensor var_3559 = const()[name = string("op_3559"), val = tensor([1, 3, 256, 256])]; tensor k_89 = reshape(shape = var_3559, x = hidden_states_173)[name = string("k_89")]; tensor hidden_states_177_axes_0 = const()[name = string("hidden_states_177_axes_0"), val = tensor([2])]; tensor hidden_states_175 = transpose(perm = var_3506, x = var_3505)[name = string("transpose_86")]; tensor hidden_states_177 = expand_dims(axes = hidden_states_177_axes_0, x = hidden_states_175)[name = string("hidden_states_177")]; tensor var_3562 = const()[name = string("op_3562"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_179 = tile(reps = var_3562, x = hidden_states_177)[name = string("hidden_states_179")]; tensor var_3564 = const()[name = string("op_3564"), val = tensor([1, 3, 256, 256])]; tensor v_29 = reshape(shape = var_3564, x = hidden_states_179)[name = string("v_29")]; bool var_3569_transpose_x_1 = const()[name = string("op_3569_transpose_x_1"), val = bool(false)]; bool var_3569_transpose_y_1 = const()[name = string("op_3569_transpose_y_1"), val = bool(true)]; tensor var_3569_cast_fp16 = matmul(transpose_x = var_3569_transpose_x_1, transpose_y = var_3569_transpose_y_1, x = q_89, y = k_89)[name = string("op_3569_cast_fp16")]; fp16 var_3570_to_fp16 = const()[name = string("op_3570_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_85_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_3570_to_fp16)[name = string("attn_weights_85_cast_fp16")]; tensor attn_weights_87_cast_fp16 = add(x = attn_weights_85_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; tensor var_3574_cast_fp16 = softmax(axis = var_22, x = attn_weights_87_cast_fp16)[name = string("op_3574_cast_fp16")]; bool var_3578_transpose_x_0 = const()[name = string("op_3578_transpose_x_0"), val = bool(false)]; bool var_3578_transpose_y_0 = const()[name = string("op_3578_transpose_y_0"), val = bool(false)]; tensor var_3578_cast_fp16 = matmul(transpose_x = var_3578_transpose_x_0, transpose_y = var_3578_transpose_y_0, x = var_3574_cast_fp16, y = v_29)[name = string("op_3578_cast_fp16")]; tensor var_3580 = const()[name = string("op_3580"), val = tensor([0, 2, 1, 3])]; tensor var_3583 = const()[name = string("op_3583"), val = tensor([1, 256, 768])]; tensor var_3581 = transpose(perm = var_3580, x = var_3578_cast_fp16)[name = string("transpose_85")]; tensor attn_out_87 = reshape(shape = var_3583, x = var_3581)[name = string("attn_out_87")]; tensor var_3585 = const()[name = string("op_3585"), val = tensor([0, 2, 1])]; tensor squeeze_14_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302758400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303348288))))[name = string("squeeze_14_quantized")]; string var_3594_pad_type_0 = const()[name = string("op_3594_pad_type_0"), val = string("valid")]; int32 var_3594_groups_0 = const()[name = string("op_3594_groups_0"), val = int32(1)]; tensor var_3594_strides_0 = const()[name = string("op_3594_strides_0"), val = tensor([1])]; tensor var_3594_pad_0 = const()[name = string("op_3594_pad_0"), val = tensor([0, 0])]; tensor var_3594_dilations_0 = const()[name = string("op_3594_dilations_0"), val = tensor([1])]; tensor var_3586 = transpose(perm = var_3585, x = attn_out_87)[name = string("transpose_84")]; tensor var_3594 = conv(dilations = var_3594_dilations_0, groups = var_3594_groups_0, pad = var_3594_pad_0, pad_type = var_3594_pad_type_0, strides = var_3594_strides_0, weight = squeeze_14_quantized, x = var_3586)[name = string("op_3594")]; tensor var_3595 = const()[name = string("op_3595"), val = tensor([0, 2, 1])]; fp16 const_204_promoted_to_fp16 = const()[name = string("const_204_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_233 = transpose(perm = var_3595, x = var_3594)[name = string("transpose_83")]; tensor var_3599_cast_fp16 = mul(x = x_233, y = const_204_promoted_to_fp16)[name = string("op_3599_cast_fp16")]; bool input_291_interleave_0 = const()[name = string("input_291_interleave_0"), val = bool(false)]; tensor input_291_cast_fp16 = concat(axis = var_22, interleave = input_291_interleave_0, values = (x_233, var_3599_cast_fp16))[name = string("input_291_cast_fp16")]; tensor normed_407_axes_0 = const()[name = string("normed_407_axes_0"), val = tensor([-1])]; tensor normed_407_cast_fp16 = layer_norm(axes = normed_407_axes_0, epsilon = var_8_to_fp16, x = input_291_cast_fp16)[name = string("normed_407_cast_fp16")]; tensor var_3604_split_sizes_0 = const()[name = string("op_3604_split_sizes_0"), val = tensor([768, 768])]; int32 var_3604_axis_0 = const()[name = string("op_3604_axis_0"), val = int32(-1)]; tensor var_3604_cast_fp16_0, tensor var_3604_cast_fp16_1 = split(axis = var_3604_axis_0, split_sizes = var_3604_split_sizes_0, x = normed_407_cast_fp16)[name = string("op_3604_cast_fp16")]; tensor var_3608_to_fp16 = const()[name = string("op_3608_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303349888)))]; tensor out_175_cast_fp16 = mul(x = var_3604_cast_fp16_0, y = var_3608_to_fp16)[name = string("out_175_cast_fp16")]; tensor x_235_cast_fp16 = add(x = x_225_cast_fp16, y = out_175_cast_fp16)[name = string("x_235_cast_fp16")]; fp16 const_206_promoted_to_fp16 = const()[name = string("const_206_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3615_cast_fp16 = mul(x = x_235_cast_fp16, y = const_206_promoted_to_fp16)[name = string("op_3615_cast_fp16")]; bool input_293_interleave_0 = const()[name = string("input_293_interleave_0"), val = bool(false)]; tensor input_293_cast_fp16 = concat(axis = var_22, interleave = input_293_interleave_0, values = (x_235_cast_fp16, var_3615_cast_fp16))[name = string("input_293_cast_fp16")]; tensor normed_411_axes_0 = const()[name = string("normed_411_axes_0"), val = tensor([-1])]; tensor normed_411_cast_fp16 = layer_norm(axes = normed_411_axes_0, epsilon = var_8_to_fp16, x = input_293_cast_fp16)[name = string("normed_411_cast_fp16")]; tensor var_3620_split_sizes_0 = const()[name = string("op_3620_split_sizes_0"), val = tensor([768, 768])]; int32 var_3620_axis_0 = const()[name = string("op_3620_axis_0"), val = int32(-1)]; tensor var_3620_cast_fp16_0, tensor var_3620_cast_fp16_1 = split(axis = var_3620_axis_0, split_sizes = var_3620_split_sizes_0, x = normed_411_cast_fp16)[name = string("op_3620_cast_fp16")]; tensor var_3624_to_fp16 = const()[name = string("op_3624_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303351488)))]; tensor out_177_cast_fp16 = mul(x = var_3620_cast_fp16_0, y = var_3624_to_fp16)[name = string("out_177_cast_fp16")]; tensor var_3631 = const()[name = string("op_3631"), val = tensor([0, 2, 1])]; tensor input_295_axes_0 = const()[name = string("input_295_axes_0"), val = tensor([2])]; tensor var_3632 = transpose(perm = var_3631, x = out_177_cast_fp16)[name = string("transpose_82")]; tensor input_295 = expand_dims(axes = input_295_axes_0, x = var_3632)[name = string("input_295")]; string gate_57_pad_type_0 = const()[name = string("gate_57_pad_type_0"), val = string("valid")]; tensor gate_57_strides_0 = const()[name = string("gate_57_strides_0"), val = tensor([1, 1])]; tensor gate_57_pad_0 = const()[name = string("gate_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_57_dilations_0 = const()[name = string("gate_57_dilations_0"), val = tensor([1, 1])]; int32 gate_57_groups_0 = const()[name = string("gate_57_groups_0"), val = int32(1)]; tensor gate_57 = conv(dilations = gate_57_dilations_0, groups = gate_57_groups_0, pad = gate_57_pad_0, pad_type = gate_57_pad_type_0, strides = gate_57_strides_0, weight = encoder_layers_14_mlp_gate_proj_weight_quantized, x = input_295)[name = string("gate_57")]; string up_29_pad_type_0 = const()[name = string("up_29_pad_type_0"), val = string("valid")]; tensor up_29_strides_0 = const()[name = string("up_29_strides_0"), val = tensor([1, 1])]; tensor up_29_pad_0 = const()[name = string("up_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_29_dilations_0 = const()[name = string("up_29_dilations_0"), val = tensor([1, 1])]; int32 up_29_groups_0 = const()[name = string("up_29_groups_0"), val = int32(1)]; tensor up_29 = conv(dilations = up_29_dilations_0, groups = up_29_groups_0, pad = up_29_pad_0, pad_type = up_29_pad_type_0, strides = up_29_strides_0, weight = encoder_layers_14_mlp_up_proj_weight_quantized, x = input_295)[name = string("up_29")]; string gate_59_mode_0 = const()[name = string("gate_59_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_59 = gelu(mode = gate_59_mode_0, x = gate_57)[name = string("gate_59")]; tensor input_297 = mul(x = gate_59, y = up_29)[name = string("input_297")]; string var_3653_pad_type_0 = const()[name = string("op_3653_pad_type_0"), val = string("valid")]; tensor var_3653_strides_0 = const()[name = string("op_3653_strides_0"), val = tensor([1, 1])]; tensor var_3653_pad_0 = const()[name = string("op_3653_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3653_dilations_0 = const()[name = string("op_3653_dilations_0"), val = tensor([1, 1])]; int32 var_3653_groups_0 = const()[name = string("op_3653_groups_0"), val = int32(1)]; tensor var_3653 = conv(dilations = var_3653_dilations_0, groups = var_3653_groups_0, pad = var_3653_pad_0, pad_type = var_3653_pad_type_0, strides = var_3653_strides_0, weight = encoder_layers_14_mlp_down_proj_weight_quantized, x = input_297)[name = string("op_3653")]; tensor var_3654_axes_0 = const()[name = string("op_3654_axes_0"), val = tensor([2])]; tensor var_3654 = squeeze(axes = var_3654_axes_0, x = var_3653)[name = string("op_3654")]; tensor var_3655 = const()[name = string("op_3655"), val = tensor([0, 2, 1])]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_239 = transpose(perm = var_3655, x = var_3654)[name = string("transpose_81")]; tensor var_3659_cast_fp16 = mul(x = x_239, y = const_208_promoted_to_fp16)[name = string("op_3659_cast_fp16")]; bool input_299_interleave_0 = const()[name = string("input_299_interleave_0"), val = bool(false)]; tensor input_299_cast_fp16 = concat(axis = var_22, interleave = input_299_interleave_0, values = (x_239, var_3659_cast_fp16))[name = string("input_299_cast_fp16")]; tensor normed_417_axes_0 = const()[name = string("normed_417_axes_0"), val = tensor([-1])]; tensor normed_417_cast_fp16 = layer_norm(axes = normed_417_axes_0, epsilon = var_8_to_fp16, x = input_299_cast_fp16)[name = string("normed_417_cast_fp16")]; tensor var_3664_split_sizes_0 = const()[name = string("op_3664_split_sizes_0"), val = tensor([768, 768])]; int32 var_3664_axis_0 = const()[name = string("op_3664_axis_0"), val = int32(-1)]; tensor var_3664_cast_fp16_0, tensor var_3664_cast_fp16_1 = split(axis = var_3664_axis_0, split_sizes = var_3664_split_sizes_0, x = normed_417_cast_fp16)[name = string("op_3664_cast_fp16")]; tensor var_3668_to_fp16 = const()[name = string("op_3668_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303353088)))]; tensor out_179_cast_fp16 = mul(x = var_3664_cast_fp16_0, y = var_3668_to_fp16)[name = string("out_179_cast_fp16")]; tensor x_241_cast_fp16 = add(x = x_235_cast_fp16, y = out_179_cast_fp16)[name = string("x_241_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3697_cast_fp16 = mul(x = x_241_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_3697_cast_fp16")]; bool input_301_interleave_0 = const()[name = string("input_301_interleave_0"), val = bool(false)]; tensor input_301_cast_fp16 = concat(axis = var_22, interleave = input_301_interleave_0, values = (x_241_cast_fp16, var_3697_cast_fp16))[name = string("input_301_cast_fp16")]; tensor normed_421_axes_0 = const()[name = string("normed_421_axes_0"), val = tensor([-1])]; tensor normed_421_cast_fp16 = layer_norm(axes = normed_421_axes_0, epsilon = var_8_to_fp16, x = input_301_cast_fp16)[name = string("normed_421_cast_fp16")]; tensor var_3702_split_sizes_0 = const()[name = string("op_3702_split_sizes_0"), val = tensor([768, 768])]; int32 var_3702_axis_0 = const()[name = string("op_3702_axis_0"), val = int32(-1)]; tensor var_3702_cast_fp16_0, tensor var_3702_cast_fp16_1 = split(axis = var_3702_axis_0, split_sizes = var_3702_split_sizes_0, x = normed_421_cast_fp16)[name = string("op_3702_cast_fp16")]; tensor var_3706_to_fp16 = const()[name = string("op_3706_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303354688)))]; tensor out_181_cast_fp16 = mul(x = var_3702_cast_fp16_0, y = var_3706_to_fp16)[name = string("out_181_cast_fp16")]; tensor var_3712 = const()[name = string("op_3712"), val = tensor([0, 2, 1])]; tensor var_3714_axes_0 = const()[name = string("op_3714_axes_0"), val = tensor([2])]; tensor var_3713_cast_fp16 = transpose(perm = var_3712, x = out_181_cast_fp16)[name = string("transpose_80")]; tensor var_3714_cast_fp16 = expand_dims(axes = var_3714_axes_0, x = var_3713_cast_fp16)[name = string("op_3714_cast_fp16")]; string var_3721_pad_type_0 = const()[name = string("op_3721_pad_type_0"), val = string("valid")]; tensor var_3721_strides_0 = const()[name = string("op_3721_strides_0"), val = tensor([1, 1])]; tensor var_3721_pad_0 = const()[name = string("op_3721_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3721_dilations_0 = const()[name = string("op_3721_dilations_0"), val = tensor([1, 1])]; int32 var_3721_groups_0 = const()[name = string("op_3721_groups_0"), val = int32(1)]; tensor var_3721 = conv(dilations = var_3721_dilations_0, groups = var_3721_groups_0, pad = var_3721_pad_0, pad_type = var_3721_pad_type_0, strides = var_3721_strides_0, weight = encoder_layers_15_self_attn_q_proj_weight_quantized, x = var_3714_cast_fp16)[name = string("op_3721")]; tensor var_3722 = const()[name = string("op_3722"), val = tensor([1, 3, 256, 256])]; tensor var_3723 = reshape(shape = var_3722, x = var_3721)[name = string("op_3723")]; tensor var_3724 = const()[name = string("op_3724"), val = tensor([0, 1, 3, 2])]; string var_3731_pad_type_0 = const()[name = string("op_3731_pad_type_0"), val = string("valid")]; tensor var_3731_strides_0 = const()[name = string("op_3731_strides_0"), val = tensor([1, 1])]; tensor var_3731_pad_0 = const()[name = string("op_3731_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3731_dilations_0 = const()[name = string("op_3731_dilations_0"), val = tensor([1, 1])]; int32 var_3731_groups_0 = const()[name = string("op_3731_groups_0"), val = int32(1)]; tensor var_3731 = conv(dilations = var_3731_dilations_0, groups = var_3731_groups_0, pad = var_3731_pad_0, pad_type = var_3731_pad_type_0, strides = var_3731_strides_0, weight = encoder_layers_15_self_attn_k_proj_weight_quantized, x = var_3714_cast_fp16)[name = string("op_3731")]; tensor var_3732 = const()[name = string("op_3732"), val = tensor([1, 1, 256, 256])]; tensor var_3733 = reshape(shape = var_3732, x = var_3731)[name = string("op_3733")]; tensor var_3734 = const()[name = string("op_3734"), val = tensor([0, 1, 3, 2])]; string var_3741_pad_type_0 = const()[name = string("op_3741_pad_type_0"), val = string("valid")]; tensor var_3741_strides_0 = const()[name = string("op_3741_strides_0"), val = tensor([1, 1])]; tensor var_3741_pad_0 = const()[name = string("op_3741_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3741_dilations_0 = const()[name = string("op_3741_dilations_0"), val = tensor([1, 1])]; int32 var_3741_groups_0 = const()[name = string("op_3741_groups_0"), val = int32(1)]; tensor var_3741 = conv(dilations = var_3741_dilations_0, groups = var_3741_groups_0, pad = var_3741_pad_0, pad_type = var_3741_pad_type_0, strides = var_3741_strides_0, weight = encoder_layers_15_self_attn_v_proj_weight_quantized, x = var_3714_cast_fp16)[name = string("op_3741")]; tensor var_3742 = const()[name = string("op_3742"), val = tensor([1, 1, 256, 256])]; tensor var_3743 = reshape(shape = var_3742, x = var_3741)[name = string("op_3743")]; tensor var_3744 = const()[name = string("op_3744"), val = tensor([0, 1, 3, 2])]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_91 = transpose(perm = var_3724, x = var_3723)[name = string("transpose_79")]; tensor var_3750_cast_fp16 = mul(x = q_91, y = const_212_promoted_to_fp16)[name = string("op_3750_cast_fp16")]; bool input_305_interleave_0 = const()[name = string("input_305_interleave_0"), val = bool(false)]; tensor input_305_cast_fp16 = concat(axis = var_22, interleave = input_305_interleave_0, values = (q_91, var_3750_cast_fp16))[name = string("input_305_cast_fp16")]; tensor normed_427_axes_0 = const()[name = string("normed_427_axes_0"), val = tensor([-1])]; tensor normed_427_cast_fp16 = layer_norm(axes = normed_427_axes_0, epsilon = var_8_to_fp16, x = input_305_cast_fp16)[name = string("normed_427_cast_fp16")]; tensor var_3755_split_sizes_0 = const()[name = string("op_3755_split_sizes_0"), val = tensor([256, 256])]; int32 var_3755_axis_0 = const()[name = string("op_3755_axis_0"), val = int32(-1)]; tensor var_3755_cast_fp16_0, tensor var_3755_cast_fp16_1 = split(axis = var_3755_axis_0, split_sizes = var_3755_split_sizes_0, x = normed_427_cast_fp16)[name = string("op_3755_cast_fp16")]; tensor var_3759_to_fp16 = const()[name = string("op_3759_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303356288)))]; tensor out_183_cast_fp16 = mul(x = var_3755_cast_fp16_0, y = var_3759_to_fp16)[name = string("out_183_cast_fp16")]; fp16 const_214_promoted_to_fp16 = const()[name = string("const_214_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_91 = transpose(perm = var_3734, x = var_3733)[name = string("transpose_78")]; tensor var_3766_cast_fp16 = mul(x = k_91, y = const_214_promoted_to_fp16)[name = string("op_3766_cast_fp16")]; bool input_307_interleave_0 = const()[name = string("input_307_interleave_0"), val = bool(false)]; tensor input_307_cast_fp16 = concat(axis = var_22, interleave = input_307_interleave_0, values = (k_91, var_3766_cast_fp16))[name = string("input_307_cast_fp16")]; tensor normed_431_axes_0 = const()[name = string("normed_431_axes_0"), val = tensor([-1])]; tensor normed_431_cast_fp16 = layer_norm(axes = normed_431_axes_0, epsilon = var_8_to_fp16, x = input_307_cast_fp16)[name = string("normed_431_cast_fp16")]; tensor var_3771_split_sizes_0 = const()[name = string("op_3771_split_sizes_0"), val = tensor([256, 256])]; int32 var_3771_axis_0 = const()[name = string("op_3771_axis_0"), val = int32(-1)]; tensor var_3771_cast_fp16_0, tensor var_3771_cast_fp16_1 = split(axis = var_3771_axis_0, split_sizes = var_3771_split_sizes_0, x = normed_431_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3775_to_fp16 = const()[name = string("op_3775_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303356864)))]; tensor out_185_cast_fp16 = mul(x = var_3771_cast_fp16_0, y = var_3775_to_fp16)[name = string("out_185_cast_fp16")]; tensor var_3778 = mul(x = out_183_cast_fp16, y = cos_1_quantized)[name = string("op_3778")]; tensor var_3779_split_sizes_0 = const()[name = string("op_3779_split_sizes_0"), val = tensor([128, 128])]; int32 var_3779_axis_0 = const()[name = string("op_3779_axis_0"), val = int32(-1)]; tensor var_3779_0, tensor var_3779_1 = split(axis = var_3779_axis_0, split_sizes = var_3779_split_sizes_0, x = out_183_cast_fp16)[name = string("op_3779")]; fp16 const_216_promoted = const()[name = string("const_216_promoted"), val = fp16(-0x1p+0)]; tensor var_3781 = mul(x = var_3779_1, y = const_216_promoted)[name = string("op_3781")]; bool var_3783_interleave_0 = const()[name = string("op_3783_interleave_0"), val = bool(false)]; tensor var_3783 = concat(axis = var_22, interleave = var_3783_interleave_0, values = (var_3781, var_3779_0))[name = string("op_3783")]; tensor var_3784 = mul(x = var_3783, y = sin_1_quantized)[name = string("op_3784")]; tensor q_95 = add(x = var_3778, y = var_3784)[name = string("q_95")]; tensor var_3786 = mul(x = out_185_cast_fp16, y = cos_1_quantized)[name = string("op_3786")]; tensor var_3787_split_sizes_0 = const()[name = string("op_3787_split_sizes_0"), val = tensor([128, 128])]; int32 var_3787_axis_0 = const()[name = string("op_3787_axis_0"), val = int32(-1)]; tensor var_3787_0, tensor var_3787_1 = split(axis = var_3787_axis_0, split_sizes = var_3787_split_sizes_0, x = out_185_cast_fp16)[name = string("op_3787")]; fp16 const_217_promoted = const()[name = string("const_217_promoted"), val = fp16(-0x1p+0)]; tensor var_3789 = mul(x = var_3787_1, y = const_217_promoted)[name = string("op_3789")]; bool var_3791_interleave_0 = const()[name = string("op_3791_interleave_0"), val = bool(false)]; tensor var_3791 = concat(axis = var_22, interleave = var_3791_interleave_0, values = (var_3789, var_3787_0))[name = string("op_3791")]; tensor var_3792 = mul(x = var_3791, y = sin_1_quantized)[name = string("op_3792")]; tensor hidden_states_181 = add(x = var_3786, y = var_3792)[name = string("hidden_states_181")]; tensor hidden_states_183_axes_0 = const()[name = string("hidden_states_183_axes_0"), val = tensor([2])]; tensor hidden_states_183 = expand_dims(axes = hidden_states_183_axes_0, x = hidden_states_181)[name = string("hidden_states_183")]; tensor var_3795 = const()[name = string("op_3795"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_185 = tile(reps = var_3795, x = hidden_states_183)[name = string("hidden_states_185")]; tensor var_3797 = const()[name = string("op_3797"), val = tensor([1, 3, 256, 256])]; tensor k_95 = reshape(shape = var_3797, x = hidden_states_185)[name = string("k_95")]; tensor hidden_states_189_axes_0 = const()[name = string("hidden_states_189_axes_0"), val = tensor([2])]; tensor hidden_states_187 = transpose(perm = var_3744, x = var_3743)[name = string("transpose_77")]; tensor hidden_states_189 = expand_dims(axes = hidden_states_189_axes_0, x = hidden_states_187)[name = string("hidden_states_189")]; tensor var_3800 = const()[name = string("op_3800"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_191 = tile(reps = var_3800, x = hidden_states_189)[name = string("hidden_states_191")]; tensor var_3802 = const()[name = string("op_3802"), val = tensor([1, 3, 256, 256])]; tensor v_31 = reshape(shape = var_3802, x = hidden_states_191)[name = string("v_31")]; bool var_3807_transpose_x_1 = const()[name = string("op_3807_transpose_x_1"), val = bool(false)]; bool var_3807_transpose_y_1 = const()[name = string("op_3807_transpose_y_1"), val = bool(true)]; tensor var_3807_cast_fp16 = matmul(transpose_x = var_3807_transpose_x_1, transpose_y = var_3807_transpose_y_1, x = q_95, y = k_95)[name = string("op_3807_cast_fp16")]; fp16 var_3808_to_fp16 = const()[name = string("op_3808_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_91_cast_fp16 = mul(x = var_3807_cast_fp16, y = var_3808_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_93_cast_fp16")]; tensor var_3812_cast_fp16 = softmax(axis = var_22, x = attn_weights_93_cast_fp16)[name = string("op_3812_cast_fp16")]; bool var_3816_transpose_x_0 = const()[name = string("op_3816_transpose_x_0"), val = bool(false)]; bool var_3816_transpose_y_0 = const()[name = string("op_3816_transpose_y_0"), val = bool(false)]; tensor var_3816_cast_fp16 = matmul(transpose_x = var_3816_transpose_x_0, transpose_y = var_3816_transpose_y_0, x = var_3812_cast_fp16, y = v_31)[name = string("op_3816_cast_fp16")]; tensor var_3818 = const()[name = string("op_3818"), val = tensor([0, 2, 1, 3])]; tensor var_3821 = const()[name = string("op_3821"), val = tensor([1, 256, 768])]; tensor var_3819 = transpose(perm = var_3818, x = var_3816_cast_fp16)[name = string("transpose_76")]; tensor attn_out_93 = reshape(shape = var_3821, x = var_3819)[name = string("attn_out_93")]; tensor var_3823 = const()[name = string("op_3823"), val = tensor([0, 2, 1])]; tensor squeeze_15_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303357440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303947328))))[name = string("squeeze_15_quantized")]; string var_3832_pad_type_0 = const()[name = string("op_3832_pad_type_0"), val = string("valid")]; int32 var_3832_groups_0 = const()[name = string("op_3832_groups_0"), val = int32(1)]; tensor var_3832_strides_0 = const()[name = string("op_3832_strides_0"), val = tensor([1])]; tensor var_3832_pad_0 = const()[name = string("op_3832_pad_0"), val = tensor([0, 0])]; tensor var_3832_dilations_0 = const()[name = string("op_3832_dilations_0"), val = tensor([1])]; tensor var_3824 = transpose(perm = var_3823, x = attn_out_93)[name = string("transpose_75")]; tensor var_3832 = conv(dilations = var_3832_dilations_0, groups = var_3832_groups_0, pad = var_3832_pad_0, pad_type = var_3832_pad_type_0, strides = var_3832_strides_0, weight = squeeze_15_quantized, x = var_3824)[name = string("op_3832")]; tensor var_3833 = const()[name = string("op_3833"), val = tensor([0, 2, 1])]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_249 = transpose(perm = var_3833, x = var_3832)[name = string("transpose_74")]; tensor var_3837_cast_fp16 = mul(x = x_249, y = const_218_promoted_to_fp16)[name = string("op_3837_cast_fp16")]; bool input_311_interleave_0 = const()[name = string("input_311_interleave_0"), val = bool(false)]; tensor input_311_cast_fp16 = concat(axis = var_22, interleave = input_311_interleave_0, values = (x_249, var_3837_cast_fp16))[name = string("input_311_cast_fp16")]; tensor normed_435_axes_0 = const()[name = string("normed_435_axes_0"), val = tensor([-1])]; tensor normed_435_cast_fp16 = layer_norm(axes = normed_435_axes_0, epsilon = var_8_to_fp16, x = input_311_cast_fp16)[name = string("normed_435_cast_fp16")]; tensor var_3842_split_sizes_0 = const()[name = string("op_3842_split_sizes_0"), val = tensor([768, 768])]; int32 var_3842_axis_0 = const()[name = string("op_3842_axis_0"), val = int32(-1)]; tensor var_3842_cast_fp16_0, tensor var_3842_cast_fp16_1 = split(axis = var_3842_axis_0, split_sizes = var_3842_split_sizes_0, x = normed_435_cast_fp16)[name = string("op_3842_cast_fp16")]; tensor var_3846_to_fp16 = const()[name = string("op_3846_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303948928)))]; tensor out_187_cast_fp16 = mul(x = var_3842_cast_fp16_0, y = var_3846_to_fp16)[name = string("out_187_cast_fp16")]; tensor x_251_cast_fp16 = add(x = x_241_cast_fp16, y = out_187_cast_fp16)[name = string("x_251_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3853_cast_fp16 = mul(x = x_251_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_3853_cast_fp16")]; bool input_313_interleave_0 = const()[name = string("input_313_interleave_0"), val = bool(false)]; tensor input_313_cast_fp16 = concat(axis = var_22, interleave = input_313_interleave_0, values = (x_251_cast_fp16, var_3853_cast_fp16))[name = string("input_313_cast_fp16")]; tensor normed_439_axes_0 = const()[name = string("normed_439_axes_0"), val = tensor([-1])]; tensor normed_439_cast_fp16 = layer_norm(axes = normed_439_axes_0, epsilon = var_8_to_fp16, x = input_313_cast_fp16)[name = string("normed_439_cast_fp16")]; tensor var_3858_split_sizes_0 = const()[name = string("op_3858_split_sizes_0"), val = tensor([768, 768])]; int32 var_3858_axis_0 = const()[name = string("op_3858_axis_0"), val = int32(-1)]; tensor var_3858_cast_fp16_0, tensor var_3858_cast_fp16_1 = split(axis = var_3858_axis_0, split_sizes = var_3858_split_sizes_0, x = normed_439_cast_fp16)[name = string("op_3858_cast_fp16")]; tensor var_3862_to_fp16 = const()[name = string("op_3862_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303950528)))]; tensor out_189_cast_fp16 = mul(x = var_3858_cast_fp16_0, y = var_3862_to_fp16)[name = string("out_189_cast_fp16")]; tensor var_3869 = const()[name = string("op_3869"), val = tensor([0, 2, 1])]; tensor input_315_axes_0 = const()[name = string("input_315_axes_0"), val = tensor([2])]; tensor var_3870 = transpose(perm = var_3869, x = out_189_cast_fp16)[name = string("transpose_73")]; tensor input_315 = expand_dims(axes = input_315_axes_0, x = var_3870)[name = string("input_315")]; string gate_61_pad_type_0 = const()[name = string("gate_61_pad_type_0"), val = string("valid")]; tensor gate_61_strides_0 = const()[name = string("gate_61_strides_0"), val = tensor([1, 1])]; tensor gate_61_pad_0 = const()[name = string("gate_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_61_dilations_0 = const()[name = string("gate_61_dilations_0"), val = tensor([1, 1])]; int32 gate_61_groups_0 = const()[name = string("gate_61_groups_0"), val = int32(1)]; tensor gate_61 = conv(dilations = gate_61_dilations_0, groups = gate_61_groups_0, pad = gate_61_pad_0, pad_type = gate_61_pad_type_0, strides = gate_61_strides_0, weight = encoder_layers_15_mlp_gate_proj_weight_quantized, x = input_315)[name = string("gate_61")]; string up_31_pad_type_0 = const()[name = string("up_31_pad_type_0"), val = string("valid")]; tensor up_31_strides_0 = const()[name = string("up_31_strides_0"), val = tensor([1, 1])]; tensor up_31_pad_0 = const()[name = string("up_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_31_dilations_0 = const()[name = string("up_31_dilations_0"), val = tensor([1, 1])]; int32 up_31_groups_0 = const()[name = string("up_31_groups_0"), val = int32(1)]; tensor up_31 = conv(dilations = up_31_dilations_0, groups = up_31_groups_0, pad = up_31_pad_0, pad_type = up_31_pad_type_0, strides = up_31_strides_0, weight = encoder_layers_15_mlp_up_proj_weight_quantized, x = input_315)[name = string("up_31")]; string gate_63_mode_0 = const()[name = string("gate_63_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_63 = gelu(mode = gate_63_mode_0, x = gate_61)[name = string("gate_63")]; tensor input_317 = mul(x = gate_63, y = up_31)[name = string("input_317")]; string var_3891_pad_type_0 = const()[name = string("op_3891_pad_type_0"), val = string("valid")]; tensor var_3891_strides_0 = const()[name = string("op_3891_strides_0"), val = tensor([1, 1])]; tensor var_3891_pad_0 = const()[name = string("op_3891_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3891_dilations_0 = const()[name = string("op_3891_dilations_0"), val = tensor([1, 1])]; int32 var_3891_groups_0 = const()[name = string("op_3891_groups_0"), val = int32(1)]; tensor var_3891 = conv(dilations = var_3891_dilations_0, groups = var_3891_groups_0, pad = var_3891_pad_0, pad_type = var_3891_pad_type_0, strides = var_3891_strides_0, weight = encoder_layers_15_mlp_down_proj_weight_quantized, x = input_317)[name = string("op_3891")]; tensor var_3892_axes_0 = const()[name = string("op_3892_axes_0"), val = tensor([2])]; tensor var_3892 = squeeze(axes = var_3892_axes_0, x = var_3891)[name = string("op_3892")]; tensor var_3893 = const()[name = string("op_3893"), val = tensor([0, 2, 1])]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_255 = transpose(perm = var_3893, x = var_3892)[name = string("transpose_72")]; tensor var_3897_cast_fp16 = mul(x = x_255, y = const_222_promoted_to_fp16)[name = string("op_3897_cast_fp16")]; bool input_319_interleave_0 = const()[name = string("input_319_interleave_0"), val = bool(false)]; tensor input_319_cast_fp16 = concat(axis = var_22, interleave = input_319_interleave_0, values = (x_255, var_3897_cast_fp16))[name = string("input_319_cast_fp16")]; tensor normed_445_axes_0 = const()[name = string("normed_445_axes_0"), val = tensor([-1])]; tensor normed_445_cast_fp16 = layer_norm(axes = normed_445_axes_0, epsilon = var_8_to_fp16, x = input_319_cast_fp16)[name = string("normed_445_cast_fp16")]; tensor var_3902_split_sizes_0 = const()[name = string("op_3902_split_sizes_0"), val = tensor([768, 768])]; int32 var_3902_axis_0 = const()[name = string("op_3902_axis_0"), val = int32(-1)]; tensor var_3902_cast_fp16_0, tensor var_3902_cast_fp16_1 = split(axis = var_3902_axis_0, split_sizes = var_3902_split_sizes_0, x = normed_445_cast_fp16)[name = string("op_3902_cast_fp16")]; tensor var_3906_to_fp16 = const()[name = string("op_3906_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303952128)))]; tensor out_191_cast_fp16 = mul(x = var_3902_cast_fp16_0, y = var_3906_to_fp16)[name = string("out_191_cast_fp16")]; tensor x_257_cast_fp16 = add(x = x_251_cast_fp16, y = out_191_cast_fp16)[name = string("x_257_cast_fp16")]; fp16 const_224_promoted_to_fp16 = const()[name = string("const_224_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3935_cast_fp16 = mul(x = x_257_cast_fp16, y = const_224_promoted_to_fp16)[name = string("op_3935_cast_fp16")]; bool input_321_interleave_0 = const()[name = string("input_321_interleave_0"), val = bool(false)]; tensor input_321_cast_fp16 = concat(axis = var_22, interleave = input_321_interleave_0, values = (x_257_cast_fp16, var_3935_cast_fp16))[name = string("input_321_cast_fp16")]; tensor normed_449_axes_0 = const()[name = string("normed_449_axes_0"), val = tensor([-1])]; tensor normed_449_cast_fp16 = layer_norm(axes = normed_449_axes_0, epsilon = var_8_to_fp16, x = input_321_cast_fp16)[name = string("normed_449_cast_fp16")]; tensor var_3940_split_sizes_0 = const()[name = string("op_3940_split_sizes_0"), val = tensor([768, 768])]; int32 var_3940_axis_0 = const()[name = string("op_3940_axis_0"), val = int32(-1)]; tensor var_3940_cast_fp16_0, tensor var_3940_cast_fp16_1 = split(axis = var_3940_axis_0, split_sizes = var_3940_split_sizes_0, x = normed_449_cast_fp16)[name = string("op_3940_cast_fp16")]; tensor var_3944_to_fp16 = const()[name = string("op_3944_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303953728)))]; tensor out_193_cast_fp16 = mul(x = var_3940_cast_fp16_0, y = var_3944_to_fp16)[name = string("out_193_cast_fp16")]; tensor var_3950 = const()[name = string("op_3950"), val = tensor([0, 2, 1])]; tensor var_3952_axes_0 = const()[name = string("op_3952_axes_0"), val = tensor([2])]; tensor var_3951_cast_fp16 = transpose(perm = var_3950, x = out_193_cast_fp16)[name = string("transpose_71")]; tensor var_3952_cast_fp16 = expand_dims(axes = var_3952_axes_0, x = var_3951_cast_fp16)[name = string("op_3952_cast_fp16")]; string var_3959_pad_type_0 = const()[name = string("op_3959_pad_type_0"), val = string("valid")]; tensor var_3959_strides_0 = const()[name = string("op_3959_strides_0"), val = tensor([1, 1])]; tensor var_3959_pad_0 = const()[name = string("op_3959_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3959_dilations_0 = const()[name = string("op_3959_dilations_0"), val = tensor([1, 1])]; int32 var_3959_groups_0 = const()[name = string("op_3959_groups_0"), val = int32(1)]; tensor var_3959 = conv(dilations = var_3959_dilations_0, groups = var_3959_groups_0, pad = var_3959_pad_0, pad_type = var_3959_pad_type_0, strides = var_3959_strides_0, weight = encoder_layers_16_self_attn_q_proj_weight_quantized, x = var_3952_cast_fp16)[name = string("op_3959")]; tensor var_3960 = const()[name = string("op_3960"), val = tensor([1, 3, 256, 256])]; tensor var_3961 = reshape(shape = var_3960, x = var_3959)[name = string("op_3961")]; tensor var_3962 = const()[name = string("op_3962"), val = tensor([0, 1, 3, 2])]; string var_3969_pad_type_0 = const()[name = string("op_3969_pad_type_0"), val = string("valid")]; tensor var_3969_strides_0 = const()[name = string("op_3969_strides_0"), val = tensor([1, 1])]; tensor var_3969_pad_0 = const()[name = string("op_3969_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3969_dilations_0 = const()[name = string("op_3969_dilations_0"), val = tensor([1, 1])]; int32 var_3969_groups_0 = const()[name = string("op_3969_groups_0"), val = int32(1)]; tensor var_3969 = conv(dilations = var_3969_dilations_0, groups = var_3969_groups_0, pad = var_3969_pad_0, pad_type = var_3969_pad_type_0, strides = var_3969_strides_0, weight = encoder_layers_16_self_attn_k_proj_weight_quantized, x = var_3952_cast_fp16)[name = string("op_3969")]; tensor var_3970 = const()[name = string("op_3970"), val = tensor([1, 1, 256, 256])]; tensor var_3971 = reshape(shape = var_3970, x = var_3969)[name = string("op_3971")]; tensor var_3972 = const()[name = string("op_3972"), val = tensor([0, 1, 3, 2])]; string var_3979_pad_type_0 = const()[name = string("op_3979_pad_type_0"), val = string("valid")]; tensor var_3979_strides_0 = const()[name = string("op_3979_strides_0"), val = tensor([1, 1])]; tensor var_3979_pad_0 = const()[name = string("op_3979_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3979_dilations_0 = const()[name = string("op_3979_dilations_0"), val = tensor([1, 1])]; int32 var_3979_groups_0 = const()[name = string("op_3979_groups_0"), val = int32(1)]; tensor var_3979 = conv(dilations = var_3979_dilations_0, groups = var_3979_groups_0, pad = var_3979_pad_0, pad_type = var_3979_pad_type_0, strides = var_3979_strides_0, weight = encoder_layers_16_self_attn_v_proj_weight_quantized, x = var_3952_cast_fp16)[name = string("op_3979")]; tensor var_3980 = const()[name = string("op_3980"), val = tensor([1, 1, 256, 256])]; tensor var_3981 = reshape(shape = var_3980, x = var_3979)[name = string("op_3981")]; tensor var_3982 = const()[name = string("op_3982"), val = tensor([0, 1, 3, 2])]; fp16 const_226_promoted_to_fp16 = const()[name = string("const_226_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_97 = transpose(perm = var_3962, x = var_3961)[name = string("transpose_70")]; tensor var_3988_cast_fp16 = mul(x = q_97, y = const_226_promoted_to_fp16)[name = string("op_3988_cast_fp16")]; bool input_325_interleave_0 = const()[name = string("input_325_interleave_0"), val = bool(false)]; tensor input_325_cast_fp16 = concat(axis = var_22, interleave = input_325_interleave_0, values = (q_97, var_3988_cast_fp16))[name = string("input_325_cast_fp16")]; tensor normed_455_axes_0 = const()[name = string("normed_455_axes_0"), val = tensor([-1])]; tensor normed_455_cast_fp16 = layer_norm(axes = normed_455_axes_0, epsilon = var_8_to_fp16, x = input_325_cast_fp16)[name = string("normed_455_cast_fp16")]; tensor var_3993_split_sizes_0 = const()[name = string("op_3993_split_sizes_0"), val = tensor([256, 256])]; int32 var_3993_axis_0 = const()[name = string("op_3993_axis_0"), val = int32(-1)]; tensor var_3993_cast_fp16_0, tensor var_3993_cast_fp16_1 = split(axis = var_3993_axis_0, split_sizes = var_3993_split_sizes_0, x = normed_455_cast_fp16)[name = string("op_3993_cast_fp16")]; tensor var_3997_to_fp16 = const()[name = string("op_3997_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303955328)))]; tensor out_195_cast_fp16 = mul(x = var_3993_cast_fp16_0, y = var_3997_to_fp16)[name = string("out_195_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_97 = transpose(perm = var_3972, x = var_3971)[name = string("transpose_69")]; tensor var_4004_cast_fp16 = mul(x = k_97, y = const_228_promoted_to_fp16)[name = string("op_4004_cast_fp16")]; bool input_327_interleave_0 = const()[name = string("input_327_interleave_0"), val = bool(false)]; tensor input_327_cast_fp16 = concat(axis = var_22, interleave = input_327_interleave_0, values = (k_97, var_4004_cast_fp16))[name = string("input_327_cast_fp16")]; tensor normed_459_axes_0 = const()[name = string("normed_459_axes_0"), val = tensor([-1])]; tensor normed_459_cast_fp16 = layer_norm(axes = normed_459_axes_0, epsilon = var_8_to_fp16, x = input_327_cast_fp16)[name = string("normed_459_cast_fp16")]; tensor var_4009_split_sizes_0 = const()[name = string("op_4009_split_sizes_0"), val = tensor([256, 256])]; int32 var_4009_axis_0 = const()[name = string("op_4009_axis_0"), val = int32(-1)]; tensor var_4009_cast_fp16_0, tensor var_4009_cast_fp16_1 = split(axis = var_4009_axis_0, split_sizes = var_4009_split_sizes_0, x = normed_459_cast_fp16)[name = string("op_4009_cast_fp16")]; tensor var_4013_to_fp16 = const()[name = string("op_4013_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303955904)))]; tensor out_197_cast_fp16 = mul(x = var_4009_cast_fp16_0, y = var_4013_to_fp16)[name = string("out_197_cast_fp16")]; tensor var_4016 = mul(x = out_195_cast_fp16, y = cos_1_quantized)[name = string("op_4016")]; tensor var_4017_split_sizes_0 = const()[name = string("op_4017_split_sizes_0"), val = tensor([128, 128])]; int32 var_4017_axis_0 = const()[name = string("op_4017_axis_0"), val = int32(-1)]; tensor var_4017_0, tensor var_4017_1 = split(axis = var_4017_axis_0, split_sizes = var_4017_split_sizes_0, x = out_195_cast_fp16)[name = string("op_4017")]; fp16 const_230_promoted = const()[name = string("const_230_promoted"), val = fp16(-0x1p+0)]; tensor var_4019 = mul(x = var_4017_1, y = const_230_promoted)[name = string("op_4019")]; bool var_4021_interleave_0 = const()[name = string("op_4021_interleave_0"), val = bool(false)]; tensor var_4021 = concat(axis = var_22, interleave = var_4021_interleave_0, values = (var_4019, var_4017_0))[name = string("op_4021")]; tensor var_4022 = mul(x = var_4021, y = sin_1_quantized)[name = string("op_4022")]; tensor q_101 = add(x = var_4016, y = var_4022)[name = string("q_101")]; tensor var_4024 = mul(x = out_197_cast_fp16, y = cos_1_quantized)[name = string("op_4024")]; tensor var_4025_split_sizes_0 = const()[name = string("op_4025_split_sizes_0"), val = tensor([128, 128])]; int32 var_4025_axis_0 = const()[name = string("op_4025_axis_0"), val = int32(-1)]; tensor var_4025_0, tensor var_4025_1 = split(axis = var_4025_axis_0, split_sizes = var_4025_split_sizes_0, x = out_197_cast_fp16)[name = string("op_4025")]; fp16 const_231_promoted = const()[name = string("const_231_promoted"), val = fp16(-0x1p+0)]; tensor var_4027 = mul(x = var_4025_1, y = const_231_promoted)[name = string("op_4027")]; bool var_4029_interleave_0 = const()[name = string("op_4029_interleave_0"), val = bool(false)]; tensor var_4029 = concat(axis = var_22, interleave = var_4029_interleave_0, values = (var_4027, var_4025_0))[name = string("op_4029")]; tensor var_4030 = mul(x = var_4029, y = sin_1_quantized)[name = string("op_4030")]; tensor hidden_states_193 = add(x = var_4024, y = var_4030)[name = string("hidden_states_193")]; tensor hidden_states_195_axes_0 = const()[name = string("hidden_states_195_axes_0"), val = tensor([2])]; tensor hidden_states_195 = expand_dims(axes = hidden_states_195_axes_0, x = hidden_states_193)[name = string("hidden_states_195")]; tensor var_4033 = const()[name = string("op_4033"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_197 = tile(reps = var_4033, x = hidden_states_195)[name = string("hidden_states_197")]; tensor var_4035 = const()[name = string("op_4035"), val = tensor([1, 3, 256, 256])]; tensor k_101 = reshape(shape = var_4035, x = hidden_states_197)[name = string("k_101")]; tensor hidden_states_201_axes_0 = const()[name = string("hidden_states_201_axes_0"), val = tensor([2])]; tensor hidden_states_199 = transpose(perm = var_3982, x = var_3981)[name = string("transpose_68")]; tensor hidden_states_201 = expand_dims(axes = hidden_states_201_axes_0, x = hidden_states_199)[name = string("hidden_states_201")]; tensor var_4038 = const()[name = string("op_4038"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_203 = tile(reps = var_4038, x = hidden_states_201)[name = string("hidden_states_203")]; tensor var_4040 = const()[name = string("op_4040"), val = tensor([1, 3, 256, 256])]; tensor v_33 = reshape(shape = var_4040, x = hidden_states_203)[name = string("v_33")]; bool var_4045_transpose_x_1 = const()[name = string("op_4045_transpose_x_1"), val = bool(false)]; bool var_4045_transpose_y_1 = const()[name = string("op_4045_transpose_y_1"), val = bool(true)]; tensor var_4045_cast_fp16 = matmul(transpose_x = var_4045_transpose_x_1, transpose_y = var_4045_transpose_y_1, x = q_101, y = k_101)[name = string("op_4045_cast_fp16")]; fp16 var_4046_to_fp16 = const()[name = string("op_4046_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_97_cast_fp16 = mul(x = var_4045_cast_fp16, y = var_4046_to_fp16)[name = string("attn_weights_97_cast_fp16")]; tensor attn_weights_99_cast_fp16 = add(x = attn_weights_97_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor var_4050_cast_fp16 = softmax(axis = var_22, x = attn_weights_99_cast_fp16)[name = string("op_4050_cast_fp16")]; bool var_4054_transpose_x_0 = const()[name = string("op_4054_transpose_x_0"), val = bool(false)]; bool var_4054_transpose_y_0 = const()[name = string("op_4054_transpose_y_0"), val = bool(false)]; tensor var_4054_cast_fp16 = matmul(transpose_x = var_4054_transpose_x_0, transpose_y = var_4054_transpose_y_0, x = var_4050_cast_fp16, y = v_33)[name = string("op_4054_cast_fp16")]; tensor var_4056 = const()[name = string("op_4056"), val = tensor([0, 2, 1, 3])]; tensor var_4059 = const()[name = string("op_4059"), val = tensor([1, 256, 768])]; tensor var_4057 = transpose(perm = var_4056, x = var_4054_cast_fp16)[name = string("transpose_67")]; tensor attn_out_99 = reshape(shape = var_4059, x = var_4057)[name = string("attn_out_99")]; tensor var_4061 = const()[name = string("op_4061"), val = tensor([0, 2, 1])]; tensor squeeze_16_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303956480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304546368))))[name = string("squeeze_16_quantized")]; string var_4070_pad_type_0 = const()[name = string("op_4070_pad_type_0"), val = string("valid")]; int32 var_4070_groups_0 = const()[name = string("op_4070_groups_0"), val = int32(1)]; tensor var_4070_strides_0 = const()[name = string("op_4070_strides_0"), val = tensor([1])]; tensor var_4070_pad_0 = const()[name = string("op_4070_pad_0"), val = tensor([0, 0])]; tensor var_4070_dilations_0 = const()[name = string("op_4070_dilations_0"), val = tensor([1])]; tensor var_4062 = transpose(perm = var_4061, x = attn_out_99)[name = string("transpose_66")]; tensor var_4070 = conv(dilations = var_4070_dilations_0, groups = var_4070_groups_0, pad = var_4070_pad_0, pad_type = var_4070_pad_type_0, strides = var_4070_strides_0, weight = squeeze_16_quantized, x = var_4062)[name = string("op_4070")]; tensor var_4071 = const()[name = string("op_4071"), val = tensor([0, 2, 1])]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_265 = transpose(perm = var_4071, x = var_4070)[name = string("transpose_65")]; tensor var_4075_cast_fp16 = mul(x = x_265, y = const_232_promoted_to_fp16)[name = string("op_4075_cast_fp16")]; bool input_331_interleave_0 = const()[name = string("input_331_interleave_0"), val = bool(false)]; tensor input_331_cast_fp16 = concat(axis = var_22, interleave = input_331_interleave_0, values = (x_265, var_4075_cast_fp16))[name = string("input_331_cast_fp16")]; tensor normed_463_axes_0 = const()[name = string("normed_463_axes_0"), val = tensor([-1])]; tensor normed_463_cast_fp16 = layer_norm(axes = normed_463_axes_0, epsilon = var_8_to_fp16, x = input_331_cast_fp16)[name = string("normed_463_cast_fp16")]; tensor var_4080_split_sizes_0 = const()[name = string("op_4080_split_sizes_0"), val = tensor([768, 768])]; int32 var_4080_axis_0 = const()[name = string("op_4080_axis_0"), val = int32(-1)]; tensor var_4080_cast_fp16_0, tensor var_4080_cast_fp16_1 = split(axis = var_4080_axis_0, split_sizes = var_4080_split_sizes_0, x = normed_463_cast_fp16)[name = string("op_4080_cast_fp16")]; tensor var_4084_to_fp16 = const()[name = string("op_4084_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304547968)))]; tensor out_199_cast_fp16 = mul(x = var_4080_cast_fp16_0, y = var_4084_to_fp16)[name = string("out_199_cast_fp16")]; tensor x_267_cast_fp16 = add(x = x_257_cast_fp16, y = out_199_cast_fp16)[name = string("x_267_cast_fp16")]; fp16 const_234_promoted_to_fp16 = const()[name = string("const_234_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4091_cast_fp16 = mul(x = x_267_cast_fp16, y = const_234_promoted_to_fp16)[name = string("op_4091_cast_fp16")]; bool input_333_interleave_0 = const()[name = string("input_333_interleave_0"), val = bool(false)]; tensor input_333_cast_fp16 = concat(axis = var_22, interleave = input_333_interleave_0, values = (x_267_cast_fp16, var_4091_cast_fp16))[name = string("input_333_cast_fp16")]; tensor normed_467_axes_0 = const()[name = string("normed_467_axes_0"), val = tensor([-1])]; tensor normed_467_cast_fp16 = layer_norm(axes = normed_467_axes_0, epsilon = var_8_to_fp16, x = input_333_cast_fp16)[name = string("normed_467_cast_fp16")]; tensor var_4096_split_sizes_0 = const()[name = string("op_4096_split_sizes_0"), val = tensor([768, 768])]; int32 var_4096_axis_0 = const()[name = string("op_4096_axis_0"), val = int32(-1)]; tensor var_4096_cast_fp16_0, tensor var_4096_cast_fp16_1 = split(axis = var_4096_axis_0, split_sizes = var_4096_split_sizes_0, x = normed_467_cast_fp16)[name = string("op_4096_cast_fp16")]; tensor var_4100_to_fp16 = const()[name = string("op_4100_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304549568)))]; tensor out_201_cast_fp16 = mul(x = var_4096_cast_fp16_0, y = var_4100_to_fp16)[name = string("out_201_cast_fp16")]; tensor var_4107 = const()[name = string("op_4107"), val = tensor([0, 2, 1])]; tensor input_335_axes_0 = const()[name = string("input_335_axes_0"), val = tensor([2])]; tensor var_4108 = transpose(perm = var_4107, x = out_201_cast_fp16)[name = string("transpose_64")]; tensor input_335 = expand_dims(axes = input_335_axes_0, x = var_4108)[name = string("input_335")]; string gate_65_pad_type_0 = const()[name = string("gate_65_pad_type_0"), val = string("valid")]; tensor gate_65_strides_0 = const()[name = string("gate_65_strides_0"), val = tensor([1, 1])]; tensor gate_65_pad_0 = const()[name = string("gate_65_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_65_dilations_0 = const()[name = string("gate_65_dilations_0"), val = tensor([1, 1])]; int32 gate_65_groups_0 = const()[name = string("gate_65_groups_0"), val = int32(1)]; tensor gate_65 = conv(dilations = gate_65_dilations_0, groups = gate_65_groups_0, pad = gate_65_pad_0, pad_type = gate_65_pad_type_0, strides = gate_65_strides_0, weight = encoder_layers_16_mlp_gate_proj_weight_quantized, x = input_335)[name = string("gate_65")]; string up_33_pad_type_0 = const()[name = string("up_33_pad_type_0"), val = string("valid")]; tensor up_33_strides_0 = const()[name = string("up_33_strides_0"), val = tensor([1, 1])]; tensor up_33_pad_0 = const()[name = string("up_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_33_dilations_0 = const()[name = string("up_33_dilations_0"), val = tensor([1, 1])]; int32 up_33_groups_0 = const()[name = string("up_33_groups_0"), val = int32(1)]; tensor up_33 = conv(dilations = up_33_dilations_0, groups = up_33_groups_0, pad = up_33_pad_0, pad_type = up_33_pad_type_0, strides = up_33_strides_0, weight = encoder_layers_16_mlp_up_proj_weight_quantized, x = input_335)[name = string("up_33")]; string gate_67_mode_0 = const()[name = string("gate_67_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_67 = gelu(mode = gate_67_mode_0, x = gate_65)[name = string("gate_67")]; tensor input_337 = mul(x = gate_67, y = up_33)[name = string("input_337")]; string var_4129_pad_type_0 = const()[name = string("op_4129_pad_type_0"), val = string("valid")]; tensor var_4129_strides_0 = const()[name = string("op_4129_strides_0"), val = tensor([1, 1])]; tensor var_4129_pad_0 = const()[name = string("op_4129_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4129_dilations_0 = const()[name = string("op_4129_dilations_0"), val = tensor([1, 1])]; int32 var_4129_groups_0 = const()[name = string("op_4129_groups_0"), val = int32(1)]; tensor var_4129 = conv(dilations = var_4129_dilations_0, groups = var_4129_groups_0, pad = var_4129_pad_0, pad_type = var_4129_pad_type_0, strides = var_4129_strides_0, weight = encoder_layers_16_mlp_down_proj_weight_quantized, x = input_337)[name = string("op_4129")]; tensor var_4130_axes_0 = const()[name = string("op_4130_axes_0"), val = tensor([2])]; tensor var_4130 = squeeze(axes = var_4130_axes_0, x = var_4129)[name = string("op_4130")]; tensor var_4131 = const()[name = string("op_4131"), val = tensor([0, 2, 1])]; fp16 const_236_promoted_to_fp16 = const()[name = string("const_236_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_271 = transpose(perm = var_4131, x = var_4130)[name = string("transpose_63")]; tensor var_4135_cast_fp16 = mul(x = x_271, y = const_236_promoted_to_fp16)[name = string("op_4135_cast_fp16")]; bool input_339_interleave_0 = const()[name = string("input_339_interleave_0"), val = bool(false)]; tensor input_339_cast_fp16 = concat(axis = var_22, interleave = input_339_interleave_0, values = (x_271, var_4135_cast_fp16))[name = string("input_339_cast_fp16")]; tensor normed_473_axes_0 = const()[name = string("normed_473_axes_0"), val = tensor([-1])]; tensor normed_473_cast_fp16 = layer_norm(axes = normed_473_axes_0, epsilon = var_8_to_fp16, x = input_339_cast_fp16)[name = string("normed_473_cast_fp16")]; tensor var_4140_split_sizes_0 = const()[name = string("op_4140_split_sizes_0"), val = tensor([768, 768])]; int32 var_4140_axis_0 = const()[name = string("op_4140_axis_0"), val = int32(-1)]; tensor var_4140_cast_fp16_0, tensor var_4140_cast_fp16_1 = split(axis = var_4140_axis_0, split_sizes = var_4140_split_sizes_0, x = normed_473_cast_fp16)[name = string("op_4140_cast_fp16")]; tensor var_4144_to_fp16 = const()[name = string("op_4144_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304551168)))]; tensor out_203_cast_fp16 = mul(x = var_4140_cast_fp16_0, y = var_4144_to_fp16)[name = string("out_203_cast_fp16")]; tensor x_273_cast_fp16 = add(x = x_267_cast_fp16, y = out_203_cast_fp16)[name = string("x_273_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4173_cast_fp16 = mul(x = x_273_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_4173_cast_fp16")]; bool input_341_interleave_0 = const()[name = string("input_341_interleave_0"), val = bool(false)]; tensor input_341_cast_fp16 = concat(axis = var_22, interleave = input_341_interleave_0, values = (x_273_cast_fp16, var_4173_cast_fp16))[name = string("input_341_cast_fp16")]; tensor normed_477_axes_0 = const()[name = string("normed_477_axes_0"), val = tensor([-1])]; tensor normed_477_cast_fp16 = layer_norm(axes = normed_477_axes_0, epsilon = var_8_to_fp16, x = input_341_cast_fp16)[name = string("normed_477_cast_fp16")]; tensor var_4178_split_sizes_0 = const()[name = string("op_4178_split_sizes_0"), val = tensor([768, 768])]; int32 var_4178_axis_0 = const()[name = string("op_4178_axis_0"), val = int32(-1)]; tensor var_4178_cast_fp16_0, tensor var_4178_cast_fp16_1 = split(axis = var_4178_axis_0, split_sizes = var_4178_split_sizes_0, x = normed_477_cast_fp16)[name = string("op_4178_cast_fp16")]; tensor var_4182_to_fp16 = const()[name = string("op_4182_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304552768)))]; tensor out_205_cast_fp16 = mul(x = var_4178_cast_fp16_0, y = var_4182_to_fp16)[name = string("out_205_cast_fp16")]; tensor var_4188 = const()[name = string("op_4188"), val = tensor([0, 2, 1])]; tensor var_4190_axes_0 = const()[name = string("op_4190_axes_0"), val = tensor([2])]; tensor var_4189_cast_fp16 = transpose(perm = var_4188, x = out_205_cast_fp16)[name = string("transpose_62")]; tensor var_4190_cast_fp16 = expand_dims(axes = var_4190_axes_0, x = var_4189_cast_fp16)[name = string("op_4190_cast_fp16")]; string var_4197_pad_type_0 = const()[name = string("op_4197_pad_type_0"), val = string("valid")]; tensor var_4197_strides_0 = const()[name = string("op_4197_strides_0"), val = tensor([1, 1])]; tensor var_4197_pad_0 = const()[name = string("op_4197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4197_dilations_0 = const()[name = string("op_4197_dilations_0"), val = tensor([1, 1])]; int32 var_4197_groups_0 = const()[name = string("op_4197_groups_0"), val = int32(1)]; tensor var_4197 = conv(dilations = var_4197_dilations_0, groups = var_4197_groups_0, pad = var_4197_pad_0, pad_type = var_4197_pad_type_0, strides = var_4197_strides_0, weight = encoder_layers_17_self_attn_q_proj_weight_quantized, x = var_4190_cast_fp16)[name = string("op_4197")]; tensor var_4198 = const()[name = string("op_4198"), val = tensor([1, 3, 256, 256])]; tensor var_4199 = reshape(shape = var_4198, x = var_4197)[name = string("op_4199")]; tensor var_4200 = const()[name = string("op_4200"), val = tensor([0, 1, 3, 2])]; string var_4207_pad_type_0 = const()[name = string("op_4207_pad_type_0"), val = string("valid")]; tensor var_4207_strides_0 = const()[name = string("op_4207_strides_0"), val = tensor([1, 1])]; tensor var_4207_pad_0 = const()[name = string("op_4207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4207_dilations_0 = const()[name = string("op_4207_dilations_0"), val = tensor([1, 1])]; int32 var_4207_groups_0 = const()[name = string("op_4207_groups_0"), val = int32(1)]; tensor var_4207 = conv(dilations = var_4207_dilations_0, groups = var_4207_groups_0, pad = var_4207_pad_0, pad_type = var_4207_pad_type_0, strides = var_4207_strides_0, weight = encoder_layers_17_self_attn_k_proj_weight_quantized, x = var_4190_cast_fp16)[name = string("op_4207")]; tensor var_4208 = const()[name = string("op_4208"), val = tensor([1, 1, 256, 256])]; tensor var_4209 = reshape(shape = var_4208, x = var_4207)[name = string("op_4209")]; tensor var_4210 = const()[name = string("op_4210"), val = tensor([0, 1, 3, 2])]; string var_4217_pad_type_0 = const()[name = string("op_4217_pad_type_0"), val = string("valid")]; tensor var_4217_strides_0 = const()[name = string("op_4217_strides_0"), val = tensor([1, 1])]; tensor var_4217_pad_0 = const()[name = string("op_4217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4217_dilations_0 = const()[name = string("op_4217_dilations_0"), val = tensor([1, 1])]; int32 var_4217_groups_0 = const()[name = string("op_4217_groups_0"), val = int32(1)]; tensor var_4217 = conv(dilations = var_4217_dilations_0, groups = var_4217_groups_0, pad = var_4217_pad_0, pad_type = var_4217_pad_type_0, strides = var_4217_strides_0, weight = encoder_layers_17_self_attn_v_proj_weight_quantized, x = var_4190_cast_fp16)[name = string("op_4217")]; tensor var_4218 = const()[name = string("op_4218"), val = tensor([1, 1, 256, 256])]; tensor var_4219 = reshape(shape = var_4218, x = var_4217)[name = string("op_4219")]; tensor var_4220 = const()[name = string("op_4220"), val = tensor([0, 1, 3, 2])]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_103 = transpose(perm = var_4200, x = var_4199)[name = string("transpose_61")]; tensor var_4226_cast_fp16 = mul(x = q_103, y = const_240_promoted_to_fp16)[name = string("op_4226_cast_fp16")]; bool input_345_interleave_0 = const()[name = string("input_345_interleave_0"), val = bool(false)]; tensor input_345_cast_fp16 = concat(axis = var_22, interleave = input_345_interleave_0, values = (q_103, var_4226_cast_fp16))[name = string("input_345_cast_fp16")]; tensor normed_483_axes_0 = const()[name = string("normed_483_axes_0"), val = tensor([-1])]; tensor normed_483_cast_fp16 = layer_norm(axes = normed_483_axes_0, epsilon = var_8_to_fp16, x = input_345_cast_fp16)[name = string("normed_483_cast_fp16")]; tensor var_4231_split_sizes_0 = const()[name = string("op_4231_split_sizes_0"), val = tensor([256, 256])]; int32 var_4231_axis_0 = const()[name = string("op_4231_axis_0"), val = int32(-1)]; tensor var_4231_cast_fp16_0, tensor var_4231_cast_fp16_1 = split(axis = var_4231_axis_0, split_sizes = var_4231_split_sizes_0, x = normed_483_cast_fp16)[name = string("op_4231_cast_fp16")]; tensor var_4235_to_fp16 = const()[name = string("op_4235_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304554368)))]; tensor out_207_cast_fp16 = mul(x = var_4231_cast_fp16_0, y = var_4235_to_fp16)[name = string("out_207_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_103 = transpose(perm = var_4210, x = var_4209)[name = string("transpose_60")]; tensor var_4242_cast_fp16 = mul(x = k_103, y = const_242_promoted_to_fp16)[name = string("op_4242_cast_fp16")]; bool input_347_interleave_0 = const()[name = string("input_347_interleave_0"), val = bool(false)]; tensor input_347_cast_fp16 = concat(axis = var_22, interleave = input_347_interleave_0, values = (k_103, var_4242_cast_fp16))[name = string("input_347_cast_fp16")]; tensor normed_487_axes_0 = const()[name = string("normed_487_axes_0"), val = tensor([-1])]; tensor normed_487_cast_fp16 = layer_norm(axes = normed_487_axes_0, epsilon = var_8_to_fp16, x = input_347_cast_fp16)[name = string("normed_487_cast_fp16")]; tensor var_4247_split_sizes_0 = const()[name = string("op_4247_split_sizes_0"), val = tensor([256, 256])]; int32 var_4247_axis_0 = const()[name = string("op_4247_axis_0"), val = int32(-1)]; tensor var_4247_cast_fp16_0, tensor var_4247_cast_fp16_1 = split(axis = var_4247_axis_0, split_sizes = var_4247_split_sizes_0, x = normed_487_cast_fp16)[name = string("op_4247_cast_fp16")]; tensor var_4251_to_fp16 = const()[name = string("op_4251_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304554944)))]; tensor out_209_cast_fp16 = mul(x = var_4247_cast_fp16_0, y = var_4251_to_fp16)[name = string("out_209_cast_fp16")]; tensor var_4254 = mul(x = out_207_cast_fp16, y = cos_quantized)[name = string("op_4254")]; tensor var_4255_split_sizes_0 = const()[name = string("op_4255_split_sizes_0"), val = tensor([128, 128])]; int32 var_4255_axis_0 = const()[name = string("op_4255_axis_0"), val = int32(-1)]; tensor var_4255_0, tensor var_4255_1 = split(axis = var_4255_axis_0, split_sizes = var_4255_split_sizes_0, x = out_207_cast_fp16)[name = string("op_4255")]; fp16 const_244_promoted = const()[name = string("const_244_promoted"), val = fp16(-0x1p+0)]; tensor var_4257 = mul(x = var_4255_1, y = const_244_promoted)[name = string("op_4257")]; bool var_4259_interleave_0 = const()[name = string("op_4259_interleave_0"), val = bool(false)]; tensor var_4259 = concat(axis = var_22, interleave = var_4259_interleave_0, values = (var_4257, var_4255_0))[name = string("op_4259")]; tensor var_4260 = mul(x = var_4259, y = sin_quantized)[name = string("op_4260")]; tensor q_107 = add(x = var_4254, y = var_4260)[name = string("q_107")]; tensor var_4262 = mul(x = out_209_cast_fp16, y = cos_quantized)[name = string("op_4262")]; tensor var_4263_split_sizes_0 = const()[name = string("op_4263_split_sizes_0"), val = tensor([128, 128])]; int32 var_4263_axis_0 = const()[name = string("op_4263_axis_0"), val = int32(-1)]; tensor var_4263_0, tensor var_4263_1 = split(axis = var_4263_axis_0, split_sizes = var_4263_split_sizes_0, x = out_209_cast_fp16)[name = string("op_4263")]; fp16 const_245_promoted = const()[name = string("const_245_promoted"), val = fp16(-0x1p+0)]; tensor var_4265 = mul(x = var_4263_1, y = const_245_promoted)[name = string("op_4265")]; bool var_4267_interleave_0 = const()[name = string("op_4267_interleave_0"), val = bool(false)]; tensor var_4267 = concat(axis = var_22, interleave = var_4267_interleave_0, values = (var_4265, var_4263_0))[name = string("op_4267")]; tensor var_4268 = mul(x = var_4267, y = sin_quantized)[name = string("op_4268")]; tensor hidden_states_205 = add(x = var_4262, y = var_4268)[name = string("hidden_states_205")]; tensor hidden_states_207_axes_0 = const()[name = string("hidden_states_207_axes_0"), val = tensor([2])]; tensor hidden_states_207 = expand_dims(axes = hidden_states_207_axes_0, x = hidden_states_205)[name = string("hidden_states_207")]; tensor var_4271 = const()[name = string("op_4271"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_209 = tile(reps = var_4271, x = hidden_states_207)[name = string("hidden_states_209")]; tensor var_4273 = const()[name = string("op_4273"), val = tensor([1, 3, 256, 256])]; tensor k_107 = reshape(shape = var_4273, x = hidden_states_209)[name = string("k_107")]; tensor hidden_states_213_axes_0 = const()[name = string("hidden_states_213_axes_0"), val = tensor([2])]; tensor hidden_states_211 = transpose(perm = var_4220, x = var_4219)[name = string("transpose_59")]; tensor hidden_states_213 = expand_dims(axes = hidden_states_213_axes_0, x = hidden_states_211)[name = string("hidden_states_213")]; tensor var_4276 = const()[name = string("op_4276"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_215 = tile(reps = var_4276, x = hidden_states_213)[name = string("hidden_states_215")]; tensor var_4278 = const()[name = string("op_4278"), val = tensor([1, 3, 256, 256])]; tensor v_35 = reshape(shape = var_4278, x = hidden_states_215)[name = string("v_35")]; bool var_4283_transpose_x_1 = const()[name = string("op_4283_transpose_x_1"), val = bool(false)]; bool var_4283_transpose_y_1 = const()[name = string("op_4283_transpose_y_1"), val = bool(true)]; tensor var_4283_cast_fp16 = matmul(transpose_x = var_4283_transpose_x_1, transpose_y = var_4283_transpose_y_1, x = q_107, y = k_107)[name = string("op_4283_cast_fp16")]; fp16 var_4284_to_fp16 = const()[name = string("op_4284_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_103_cast_fp16 = mul(x = var_4283_cast_fp16, y = var_4284_to_fp16)[name = string("attn_weights_103_cast_fp16")]; tensor attn_weights_105_cast_fp16 = add(x = attn_weights_103_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_105_cast_fp16")]; tensor var_4288_cast_fp16 = softmax(axis = var_22, x = attn_weights_105_cast_fp16)[name = string("op_4288_cast_fp16")]; bool var_4292_transpose_x_0 = const()[name = string("op_4292_transpose_x_0"), val = bool(false)]; bool var_4292_transpose_y_0 = const()[name = string("op_4292_transpose_y_0"), val = bool(false)]; tensor var_4292_cast_fp16 = matmul(transpose_x = var_4292_transpose_x_0, transpose_y = var_4292_transpose_y_0, x = var_4288_cast_fp16, y = v_35)[name = string("op_4292_cast_fp16")]; tensor var_4294 = const()[name = string("op_4294"), val = tensor([0, 2, 1, 3])]; tensor var_4297 = const()[name = string("op_4297"), val = tensor([1, 256, 768])]; tensor var_4295 = transpose(perm = var_4294, x = var_4292_cast_fp16)[name = string("transpose_58")]; tensor attn_out_105 = reshape(shape = var_4297, x = var_4295)[name = string("attn_out_105")]; tensor var_4299 = const()[name = string("op_4299"), val = tensor([0, 2, 1])]; tensor squeeze_17_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304555520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305145408))))[name = string("squeeze_17_quantized")]; string var_4308_pad_type_0 = const()[name = string("op_4308_pad_type_0"), val = string("valid")]; int32 var_4308_groups_0 = const()[name = string("op_4308_groups_0"), val = int32(1)]; tensor var_4308_strides_0 = const()[name = string("op_4308_strides_0"), val = tensor([1])]; tensor var_4308_pad_0 = const()[name = string("op_4308_pad_0"), val = tensor([0, 0])]; tensor var_4308_dilations_0 = const()[name = string("op_4308_dilations_0"), val = tensor([1])]; tensor var_4300 = transpose(perm = var_4299, x = attn_out_105)[name = string("transpose_57")]; tensor var_4308 = conv(dilations = var_4308_dilations_0, groups = var_4308_groups_0, pad = var_4308_pad_0, pad_type = var_4308_pad_type_0, strides = var_4308_strides_0, weight = squeeze_17_quantized, x = var_4300)[name = string("op_4308")]; tensor var_4309 = const()[name = string("op_4309"), val = tensor([0, 2, 1])]; fp16 const_246_promoted_to_fp16 = const()[name = string("const_246_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_281 = transpose(perm = var_4309, x = var_4308)[name = string("transpose_56")]; tensor var_4313_cast_fp16 = mul(x = x_281, y = const_246_promoted_to_fp16)[name = string("op_4313_cast_fp16")]; bool input_351_interleave_0 = const()[name = string("input_351_interleave_0"), val = bool(false)]; tensor input_351_cast_fp16 = concat(axis = var_22, interleave = input_351_interleave_0, values = (x_281, var_4313_cast_fp16))[name = string("input_351_cast_fp16")]; tensor normed_491_axes_0 = const()[name = string("normed_491_axes_0"), val = tensor([-1])]; tensor normed_491_cast_fp16 = layer_norm(axes = normed_491_axes_0, epsilon = var_8_to_fp16, x = input_351_cast_fp16)[name = string("normed_491_cast_fp16")]; tensor var_4318_split_sizes_0 = const()[name = string("op_4318_split_sizes_0"), val = tensor([768, 768])]; int32 var_4318_axis_0 = const()[name = string("op_4318_axis_0"), val = int32(-1)]; tensor var_4318_cast_fp16_0, tensor var_4318_cast_fp16_1 = split(axis = var_4318_axis_0, split_sizes = var_4318_split_sizes_0, x = normed_491_cast_fp16)[name = string("op_4318_cast_fp16")]; tensor var_4322_to_fp16 = const()[name = string("op_4322_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305147008)))]; tensor out_211_cast_fp16 = mul(x = var_4318_cast_fp16_0, y = var_4322_to_fp16)[name = string("out_211_cast_fp16")]; tensor x_283_cast_fp16 = add(x = x_273_cast_fp16, y = out_211_cast_fp16)[name = string("x_283_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4329_cast_fp16 = mul(x = x_283_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_4329_cast_fp16")]; bool input_353_interleave_0 = const()[name = string("input_353_interleave_0"), val = bool(false)]; tensor input_353_cast_fp16 = concat(axis = var_22, interleave = input_353_interleave_0, values = (x_283_cast_fp16, var_4329_cast_fp16))[name = string("input_353_cast_fp16")]; tensor normed_495_axes_0 = const()[name = string("normed_495_axes_0"), val = tensor([-1])]; tensor normed_495_cast_fp16 = layer_norm(axes = normed_495_axes_0, epsilon = var_8_to_fp16, x = input_353_cast_fp16)[name = string("normed_495_cast_fp16")]; tensor var_4334_split_sizes_0 = const()[name = string("op_4334_split_sizes_0"), val = tensor([768, 768])]; int32 var_4334_axis_0 = const()[name = string("op_4334_axis_0"), val = int32(-1)]; tensor var_4334_cast_fp16_0, tensor var_4334_cast_fp16_1 = split(axis = var_4334_axis_0, split_sizes = var_4334_split_sizes_0, x = normed_495_cast_fp16)[name = string("op_4334_cast_fp16")]; tensor var_4338_to_fp16 = const()[name = string("op_4338_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305148608)))]; tensor out_213_cast_fp16 = mul(x = var_4334_cast_fp16_0, y = var_4338_to_fp16)[name = string("out_213_cast_fp16")]; tensor var_4345 = const()[name = string("op_4345"), val = tensor([0, 2, 1])]; tensor input_355_axes_0 = const()[name = string("input_355_axes_0"), val = tensor([2])]; tensor var_4346 = transpose(perm = var_4345, x = out_213_cast_fp16)[name = string("transpose_55")]; tensor input_355 = expand_dims(axes = input_355_axes_0, x = var_4346)[name = string("input_355")]; string gate_69_pad_type_0 = const()[name = string("gate_69_pad_type_0"), val = string("valid")]; tensor gate_69_strides_0 = const()[name = string("gate_69_strides_0"), val = tensor([1, 1])]; tensor gate_69_pad_0 = const()[name = string("gate_69_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_69_dilations_0 = const()[name = string("gate_69_dilations_0"), val = tensor([1, 1])]; int32 gate_69_groups_0 = const()[name = string("gate_69_groups_0"), val = int32(1)]; tensor gate_69 = conv(dilations = gate_69_dilations_0, groups = gate_69_groups_0, pad = gate_69_pad_0, pad_type = gate_69_pad_type_0, strides = gate_69_strides_0, weight = encoder_layers_17_mlp_gate_proj_weight_quantized, x = input_355)[name = string("gate_69")]; string up_35_pad_type_0 = const()[name = string("up_35_pad_type_0"), val = string("valid")]; tensor up_35_strides_0 = const()[name = string("up_35_strides_0"), val = tensor([1, 1])]; tensor up_35_pad_0 = const()[name = string("up_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_35_dilations_0 = const()[name = string("up_35_dilations_0"), val = tensor([1, 1])]; int32 up_35_groups_0 = const()[name = string("up_35_groups_0"), val = int32(1)]; tensor up_35 = conv(dilations = up_35_dilations_0, groups = up_35_groups_0, pad = up_35_pad_0, pad_type = up_35_pad_type_0, strides = up_35_strides_0, weight = encoder_layers_17_mlp_up_proj_weight_quantized, x = input_355)[name = string("up_35")]; string gate_71_mode_0 = const()[name = string("gate_71_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_71 = gelu(mode = gate_71_mode_0, x = gate_69)[name = string("gate_71")]; tensor input_357 = mul(x = gate_71, y = up_35)[name = string("input_357")]; string var_4367_pad_type_0 = const()[name = string("op_4367_pad_type_0"), val = string("valid")]; tensor var_4367_strides_0 = const()[name = string("op_4367_strides_0"), val = tensor([1, 1])]; tensor var_4367_pad_0 = const()[name = string("op_4367_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4367_dilations_0 = const()[name = string("op_4367_dilations_0"), val = tensor([1, 1])]; int32 var_4367_groups_0 = const()[name = string("op_4367_groups_0"), val = int32(1)]; tensor var_4367 = conv(dilations = var_4367_dilations_0, groups = var_4367_groups_0, pad = var_4367_pad_0, pad_type = var_4367_pad_type_0, strides = var_4367_strides_0, weight = encoder_layers_17_mlp_down_proj_weight_quantized, x = input_357)[name = string("op_4367")]; tensor var_4368_axes_0 = const()[name = string("op_4368_axes_0"), val = tensor([2])]; tensor var_4368 = squeeze(axes = var_4368_axes_0, x = var_4367)[name = string("op_4368")]; tensor var_4369 = const()[name = string("op_4369"), val = tensor([0, 2, 1])]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_287 = transpose(perm = var_4369, x = var_4368)[name = string("transpose_54")]; tensor var_4373_cast_fp16 = mul(x = x_287, y = const_250_promoted_to_fp16)[name = string("op_4373_cast_fp16")]; bool input_359_interleave_0 = const()[name = string("input_359_interleave_0"), val = bool(false)]; tensor input_359_cast_fp16 = concat(axis = var_22, interleave = input_359_interleave_0, values = (x_287, var_4373_cast_fp16))[name = string("input_359_cast_fp16")]; tensor normed_501_axes_0 = const()[name = string("normed_501_axes_0"), val = tensor([-1])]; tensor normed_501_cast_fp16 = layer_norm(axes = normed_501_axes_0, epsilon = var_8_to_fp16, x = input_359_cast_fp16)[name = string("normed_501_cast_fp16")]; tensor var_4378_split_sizes_0 = const()[name = string("op_4378_split_sizes_0"), val = tensor([768, 768])]; int32 var_4378_axis_0 = const()[name = string("op_4378_axis_0"), val = int32(-1)]; tensor var_4378_cast_fp16_0, tensor var_4378_cast_fp16_1 = split(axis = var_4378_axis_0, split_sizes = var_4378_split_sizes_0, x = normed_501_cast_fp16)[name = string("op_4378_cast_fp16")]; tensor var_4382_to_fp16 = const()[name = string("op_4382_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305150208)))]; tensor out_215_cast_fp16 = mul(x = var_4378_cast_fp16_0, y = var_4382_to_fp16)[name = string("out_215_cast_fp16")]; tensor x_289_cast_fp16 = add(x = x_283_cast_fp16, y = out_215_cast_fp16)[name = string("x_289_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4411_cast_fp16 = mul(x = x_289_cast_fp16, y = const_252_promoted_to_fp16)[name = string("op_4411_cast_fp16")]; bool input_361_interleave_0 = const()[name = string("input_361_interleave_0"), val = bool(false)]; tensor input_361_cast_fp16 = concat(axis = var_22, interleave = input_361_interleave_0, values = (x_289_cast_fp16, var_4411_cast_fp16))[name = string("input_361_cast_fp16")]; tensor normed_505_axes_0 = const()[name = string("normed_505_axes_0"), val = tensor([-1])]; tensor normed_505_cast_fp16 = layer_norm(axes = normed_505_axes_0, epsilon = var_8_to_fp16, x = input_361_cast_fp16)[name = string("normed_505_cast_fp16")]; tensor var_4416_split_sizes_0 = const()[name = string("op_4416_split_sizes_0"), val = tensor([768, 768])]; int32 var_4416_axis_0 = const()[name = string("op_4416_axis_0"), val = int32(-1)]; tensor var_4416_cast_fp16_0, tensor var_4416_cast_fp16_1 = split(axis = var_4416_axis_0, split_sizes = var_4416_split_sizes_0, x = normed_505_cast_fp16)[name = string("op_4416_cast_fp16")]; tensor var_4420_to_fp16 = const()[name = string("op_4420_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305151808)))]; tensor out_217_cast_fp16 = mul(x = var_4416_cast_fp16_0, y = var_4420_to_fp16)[name = string("out_217_cast_fp16")]; tensor var_4426 = const()[name = string("op_4426"), val = tensor([0, 2, 1])]; tensor var_4428_axes_0 = const()[name = string("op_4428_axes_0"), val = tensor([2])]; tensor var_4427_cast_fp16 = transpose(perm = var_4426, x = out_217_cast_fp16)[name = string("transpose_53")]; tensor var_4428_cast_fp16 = expand_dims(axes = var_4428_axes_0, x = var_4427_cast_fp16)[name = string("op_4428_cast_fp16")]; string var_4435_pad_type_0 = const()[name = string("op_4435_pad_type_0"), val = string("valid")]; tensor var_4435_strides_0 = const()[name = string("op_4435_strides_0"), val = tensor([1, 1])]; tensor var_4435_pad_0 = const()[name = string("op_4435_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4435_dilations_0 = const()[name = string("op_4435_dilations_0"), val = tensor([1, 1])]; int32 var_4435_groups_0 = const()[name = string("op_4435_groups_0"), val = int32(1)]; tensor var_4435 = conv(dilations = var_4435_dilations_0, groups = var_4435_groups_0, pad = var_4435_pad_0, pad_type = var_4435_pad_type_0, strides = var_4435_strides_0, weight = encoder_layers_18_self_attn_q_proj_weight_quantized, x = var_4428_cast_fp16)[name = string("op_4435")]; tensor var_4436 = const()[name = string("op_4436"), val = tensor([1, 3, 256, 256])]; tensor var_4437 = reshape(shape = var_4436, x = var_4435)[name = string("op_4437")]; tensor var_4438 = const()[name = string("op_4438"), val = tensor([0, 1, 3, 2])]; string var_4445_pad_type_0 = const()[name = string("op_4445_pad_type_0"), val = string("valid")]; tensor var_4445_strides_0 = const()[name = string("op_4445_strides_0"), val = tensor([1, 1])]; tensor var_4445_pad_0 = const()[name = string("op_4445_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4445_dilations_0 = const()[name = string("op_4445_dilations_0"), val = tensor([1, 1])]; int32 var_4445_groups_0 = const()[name = string("op_4445_groups_0"), val = int32(1)]; tensor var_4445 = conv(dilations = var_4445_dilations_0, groups = var_4445_groups_0, pad = var_4445_pad_0, pad_type = var_4445_pad_type_0, strides = var_4445_strides_0, weight = encoder_layers_18_self_attn_k_proj_weight_quantized, x = var_4428_cast_fp16)[name = string("op_4445")]; tensor var_4446 = const()[name = string("op_4446"), val = tensor([1, 1, 256, 256])]; tensor var_4447 = reshape(shape = var_4446, x = var_4445)[name = string("op_4447")]; tensor var_4448 = const()[name = string("op_4448"), val = tensor([0, 1, 3, 2])]; string var_4455_pad_type_0 = const()[name = string("op_4455_pad_type_0"), val = string("valid")]; tensor var_4455_strides_0 = const()[name = string("op_4455_strides_0"), val = tensor([1, 1])]; tensor var_4455_pad_0 = const()[name = string("op_4455_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4455_dilations_0 = const()[name = string("op_4455_dilations_0"), val = tensor([1, 1])]; int32 var_4455_groups_0 = const()[name = string("op_4455_groups_0"), val = int32(1)]; tensor var_4455 = conv(dilations = var_4455_dilations_0, groups = var_4455_groups_0, pad = var_4455_pad_0, pad_type = var_4455_pad_type_0, strides = var_4455_strides_0, weight = encoder_layers_18_self_attn_v_proj_weight_quantized, x = var_4428_cast_fp16)[name = string("op_4455")]; tensor var_4456 = const()[name = string("op_4456"), val = tensor([1, 1, 256, 256])]; tensor var_4457 = reshape(shape = var_4456, x = var_4455)[name = string("op_4457")]; tensor var_4458 = const()[name = string("op_4458"), val = tensor([0, 1, 3, 2])]; fp16 const_254_promoted_to_fp16 = const()[name = string("const_254_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_109 = transpose(perm = var_4438, x = var_4437)[name = string("transpose_52")]; tensor var_4464_cast_fp16 = mul(x = q_109, y = const_254_promoted_to_fp16)[name = string("op_4464_cast_fp16")]; bool input_365_interleave_0 = const()[name = string("input_365_interleave_0"), val = bool(false)]; tensor input_365_cast_fp16 = concat(axis = var_22, interleave = input_365_interleave_0, values = (q_109, var_4464_cast_fp16))[name = string("input_365_cast_fp16")]; tensor normed_511_axes_0 = const()[name = string("normed_511_axes_0"), val = tensor([-1])]; tensor normed_511_cast_fp16 = layer_norm(axes = normed_511_axes_0, epsilon = var_8_to_fp16, x = input_365_cast_fp16)[name = string("normed_511_cast_fp16")]; tensor var_4469_split_sizes_0 = const()[name = string("op_4469_split_sizes_0"), val = tensor([256, 256])]; int32 var_4469_axis_0 = const()[name = string("op_4469_axis_0"), val = int32(-1)]; tensor var_4469_cast_fp16_0, tensor var_4469_cast_fp16_1 = split(axis = var_4469_axis_0, split_sizes = var_4469_split_sizes_0, x = normed_511_cast_fp16)[name = string("op_4469_cast_fp16")]; tensor var_4473_to_fp16 = const()[name = string("op_4473_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305153408)))]; tensor out_219_cast_fp16 = mul(x = var_4469_cast_fp16_0, y = var_4473_to_fp16)[name = string("out_219_cast_fp16")]; fp16 const_256_promoted_to_fp16 = const()[name = string("const_256_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_109 = transpose(perm = var_4448, x = var_4447)[name = string("transpose_51")]; tensor var_4480_cast_fp16 = mul(x = k_109, y = const_256_promoted_to_fp16)[name = string("op_4480_cast_fp16")]; bool input_367_interleave_0 = const()[name = string("input_367_interleave_0"), val = bool(false)]; tensor input_367_cast_fp16 = concat(axis = var_22, interleave = input_367_interleave_0, values = (k_109, var_4480_cast_fp16))[name = string("input_367_cast_fp16")]; tensor normed_515_axes_0 = const()[name = string("normed_515_axes_0"), val = tensor([-1])]; tensor normed_515_cast_fp16 = layer_norm(axes = normed_515_axes_0, epsilon = var_8_to_fp16, x = input_367_cast_fp16)[name = string("normed_515_cast_fp16")]; tensor var_4485_split_sizes_0 = const()[name = string("op_4485_split_sizes_0"), val = tensor([256, 256])]; int32 var_4485_axis_0 = const()[name = string("op_4485_axis_0"), val = int32(-1)]; tensor var_4485_cast_fp16_0, tensor var_4485_cast_fp16_1 = split(axis = var_4485_axis_0, split_sizes = var_4485_split_sizes_0, x = normed_515_cast_fp16)[name = string("op_4485_cast_fp16")]; tensor var_4489_to_fp16 = const()[name = string("op_4489_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305153984)))]; tensor out_221_cast_fp16 = mul(x = var_4485_cast_fp16_0, y = var_4489_to_fp16)[name = string("out_221_cast_fp16")]; tensor var_4492 = mul(x = out_219_cast_fp16, y = cos_1_quantized)[name = string("op_4492")]; tensor var_4493_split_sizes_0 = const()[name = string("op_4493_split_sizes_0"), val = tensor([128, 128])]; int32 var_4493_axis_0 = const()[name = string("op_4493_axis_0"), val = int32(-1)]; tensor var_4493_0, tensor var_4493_1 = split(axis = var_4493_axis_0, split_sizes = var_4493_split_sizes_0, x = out_219_cast_fp16)[name = string("op_4493")]; fp16 const_258_promoted = const()[name = string("const_258_promoted"), val = fp16(-0x1p+0)]; tensor var_4495 = mul(x = var_4493_1, y = const_258_promoted)[name = string("op_4495")]; bool var_4497_interleave_0 = const()[name = string("op_4497_interleave_0"), val = bool(false)]; tensor var_4497 = concat(axis = var_22, interleave = var_4497_interleave_0, values = (var_4495, var_4493_0))[name = string("op_4497")]; tensor var_4498 = mul(x = var_4497, y = sin_1_quantized)[name = string("op_4498")]; tensor q_113 = add(x = var_4492, y = var_4498)[name = string("q_113")]; tensor var_4500 = mul(x = out_221_cast_fp16, y = cos_1_quantized)[name = string("op_4500")]; tensor var_4501_split_sizes_0 = const()[name = string("op_4501_split_sizes_0"), val = tensor([128, 128])]; int32 var_4501_axis_0 = const()[name = string("op_4501_axis_0"), val = int32(-1)]; tensor var_4501_0, tensor var_4501_1 = split(axis = var_4501_axis_0, split_sizes = var_4501_split_sizes_0, x = out_221_cast_fp16)[name = string("op_4501")]; fp16 const_259_promoted = const()[name = string("const_259_promoted"), val = fp16(-0x1p+0)]; tensor var_4503 = mul(x = var_4501_1, y = const_259_promoted)[name = string("op_4503")]; bool var_4505_interleave_0 = const()[name = string("op_4505_interleave_0"), val = bool(false)]; tensor var_4505 = concat(axis = var_22, interleave = var_4505_interleave_0, values = (var_4503, var_4501_0))[name = string("op_4505")]; tensor var_4506 = mul(x = var_4505, y = sin_1_quantized)[name = string("op_4506")]; tensor hidden_states_217 = add(x = var_4500, y = var_4506)[name = string("hidden_states_217")]; tensor hidden_states_219_axes_0 = const()[name = string("hidden_states_219_axes_0"), val = tensor([2])]; tensor hidden_states_219 = expand_dims(axes = hidden_states_219_axes_0, x = hidden_states_217)[name = string("hidden_states_219")]; tensor var_4509 = const()[name = string("op_4509"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_221 = tile(reps = var_4509, x = hidden_states_219)[name = string("hidden_states_221")]; tensor var_4511 = const()[name = string("op_4511"), val = tensor([1, 3, 256, 256])]; tensor k_113 = reshape(shape = var_4511, x = hidden_states_221)[name = string("k_113")]; tensor hidden_states_225_axes_0 = const()[name = string("hidden_states_225_axes_0"), val = tensor([2])]; tensor hidden_states_223 = transpose(perm = var_4458, x = var_4457)[name = string("transpose_50")]; tensor hidden_states_225 = expand_dims(axes = hidden_states_225_axes_0, x = hidden_states_223)[name = string("hidden_states_225")]; tensor var_4514 = const()[name = string("op_4514"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_227 = tile(reps = var_4514, x = hidden_states_225)[name = string("hidden_states_227")]; tensor var_4516 = const()[name = string("op_4516"), val = tensor([1, 3, 256, 256])]; tensor v_37 = reshape(shape = var_4516, x = hidden_states_227)[name = string("v_37")]; bool var_4521_transpose_x_1 = const()[name = string("op_4521_transpose_x_1"), val = bool(false)]; bool var_4521_transpose_y_1 = const()[name = string("op_4521_transpose_y_1"), val = bool(true)]; tensor var_4521_cast_fp16 = matmul(transpose_x = var_4521_transpose_x_1, transpose_y = var_4521_transpose_y_1, x = q_113, y = k_113)[name = string("op_4521_cast_fp16")]; fp16 var_4522_to_fp16 = const()[name = string("op_4522_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_109_cast_fp16 = mul(x = var_4521_cast_fp16, y = var_4522_to_fp16)[name = string("attn_weights_109_cast_fp16")]; tensor attn_weights_111_cast_fp16 = add(x = attn_weights_109_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; tensor var_4526_cast_fp16 = softmax(axis = var_22, x = attn_weights_111_cast_fp16)[name = string("op_4526_cast_fp16")]; bool var_4530_transpose_x_0 = const()[name = string("op_4530_transpose_x_0"), val = bool(false)]; bool var_4530_transpose_y_0 = const()[name = string("op_4530_transpose_y_0"), val = bool(false)]; tensor var_4530_cast_fp16 = matmul(transpose_x = var_4530_transpose_x_0, transpose_y = var_4530_transpose_y_0, x = var_4526_cast_fp16, y = v_37)[name = string("op_4530_cast_fp16")]; tensor var_4532 = const()[name = string("op_4532"), val = tensor([0, 2, 1, 3])]; tensor var_4535 = const()[name = string("op_4535"), val = tensor([1, 256, 768])]; tensor var_4533 = transpose(perm = var_4532, x = var_4530_cast_fp16)[name = string("transpose_49")]; tensor attn_out_111 = reshape(shape = var_4535, x = var_4533)[name = string("attn_out_111")]; tensor var_4537 = const()[name = string("op_4537"), val = tensor([0, 2, 1])]; tensor squeeze_18_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305154560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305744448))))[name = string("squeeze_18_quantized")]; string var_4546_pad_type_0 = const()[name = string("op_4546_pad_type_0"), val = string("valid")]; int32 var_4546_groups_0 = const()[name = string("op_4546_groups_0"), val = int32(1)]; tensor var_4546_strides_0 = const()[name = string("op_4546_strides_0"), val = tensor([1])]; tensor var_4546_pad_0 = const()[name = string("op_4546_pad_0"), val = tensor([0, 0])]; tensor var_4546_dilations_0 = const()[name = string("op_4546_dilations_0"), val = tensor([1])]; tensor var_4538 = transpose(perm = var_4537, x = attn_out_111)[name = string("transpose_48")]; tensor var_4546 = conv(dilations = var_4546_dilations_0, groups = var_4546_groups_0, pad = var_4546_pad_0, pad_type = var_4546_pad_type_0, strides = var_4546_strides_0, weight = squeeze_18_quantized, x = var_4538)[name = string("op_4546")]; tensor var_4547 = const()[name = string("op_4547"), val = tensor([0, 2, 1])]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_297 = transpose(perm = var_4547, x = var_4546)[name = string("transpose_47")]; tensor var_4551_cast_fp16 = mul(x = x_297, y = const_260_promoted_to_fp16)[name = string("op_4551_cast_fp16")]; bool input_371_interleave_0 = const()[name = string("input_371_interleave_0"), val = bool(false)]; tensor input_371_cast_fp16 = concat(axis = var_22, interleave = input_371_interleave_0, values = (x_297, var_4551_cast_fp16))[name = string("input_371_cast_fp16")]; tensor normed_519_axes_0 = const()[name = string("normed_519_axes_0"), val = tensor([-1])]; tensor normed_519_cast_fp16 = layer_norm(axes = normed_519_axes_0, epsilon = var_8_to_fp16, x = input_371_cast_fp16)[name = string("normed_519_cast_fp16")]; tensor var_4556_split_sizes_0 = const()[name = string("op_4556_split_sizes_0"), val = tensor([768, 768])]; int32 var_4556_axis_0 = const()[name = string("op_4556_axis_0"), val = int32(-1)]; tensor var_4556_cast_fp16_0, tensor var_4556_cast_fp16_1 = split(axis = var_4556_axis_0, split_sizes = var_4556_split_sizes_0, x = normed_519_cast_fp16)[name = string("op_4556_cast_fp16")]; tensor var_4560_to_fp16 = const()[name = string("op_4560_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305746048)))]; tensor out_223_cast_fp16 = mul(x = var_4556_cast_fp16_0, y = var_4560_to_fp16)[name = string("out_223_cast_fp16")]; tensor x_299_cast_fp16 = add(x = x_289_cast_fp16, y = out_223_cast_fp16)[name = string("x_299_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4567_cast_fp16 = mul(x = x_299_cast_fp16, y = const_262_promoted_to_fp16)[name = string("op_4567_cast_fp16")]; bool input_373_interleave_0 = const()[name = string("input_373_interleave_0"), val = bool(false)]; tensor input_373_cast_fp16 = concat(axis = var_22, interleave = input_373_interleave_0, values = (x_299_cast_fp16, var_4567_cast_fp16))[name = string("input_373_cast_fp16")]; tensor normed_523_axes_0 = const()[name = string("normed_523_axes_0"), val = tensor([-1])]; tensor normed_523_cast_fp16 = layer_norm(axes = normed_523_axes_0, epsilon = var_8_to_fp16, x = input_373_cast_fp16)[name = string("normed_523_cast_fp16")]; tensor var_4572_split_sizes_0 = const()[name = string("op_4572_split_sizes_0"), val = tensor([768, 768])]; int32 var_4572_axis_0 = const()[name = string("op_4572_axis_0"), val = int32(-1)]; tensor var_4572_cast_fp16_0, tensor var_4572_cast_fp16_1 = split(axis = var_4572_axis_0, split_sizes = var_4572_split_sizes_0, x = normed_523_cast_fp16)[name = string("op_4572_cast_fp16")]; tensor var_4576_to_fp16 = const()[name = string("op_4576_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305747648)))]; tensor out_225_cast_fp16 = mul(x = var_4572_cast_fp16_0, y = var_4576_to_fp16)[name = string("out_225_cast_fp16")]; tensor var_4583 = const()[name = string("op_4583"), val = tensor([0, 2, 1])]; tensor input_375_axes_0 = const()[name = string("input_375_axes_0"), val = tensor([2])]; tensor var_4584 = transpose(perm = var_4583, x = out_225_cast_fp16)[name = string("transpose_46")]; tensor input_375 = expand_dims(axes = input_375_axes_0, x = var_4584)[name = string("input_375")]; string gate_73_pad_type_0 = const()[name = string("gate_73_pad_type_0"), val = string("valid")]; tensor gate_73_strides_0 = const()[name = string("gate_73_strides_0"), val = tensor([1, 1])]; tensor gate_73_pad_0 = const()[name = string("gate_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_73_dilations_0 = const()[name = string("gate_73_dilations_0"), val = tensor([1, 1])]; int32 gate_73_groups_0 = const()[name = string("gate_73_groups_0"), val = int32(1)]; tensor gate_73 = conv(dilations = gate_73_dilations_0, groups = gate_73_groups_0, pad = gate_73_pad_0, pad_type = gate_73_pad_type_0, strides = gate_73_strides_0, weight = encoder_layers_18_mlp_gate_proj_weight_quantized, x = input_375)[name = string("gate_73")]; string up_37_pad_type_0 = const()[name = string("up_37_pad_type_0"), val = string("valid")]; tensor up_37_strides_0 = const()[name = string("up_37_strides_0"), val = tensor([1, 1])]; tensor up_37_pad_0 = const()[name = string("up_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_37_dilations_0 = const()[name = string("up_37_dilations_0"), val = tensor([1, 1])]; int32 up_37_groups_0 = const()[name = string("up_37_groups_0"), val = int32(1)]; tensor up_37 = conv(dilations = up_37_dilations_0, groups = up_37_groups_0, pad = up_37_pad_0, pad_type = up_37_pad_type_0, strides = up_37_strides_0, weight = encoder_layers_18_mlp_up_proj_weight_quantized, x = input_375)[name = string("up_37")]; string gate_75_mode_0 = const()[name = string("gate_75_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_75 = gelu(mode = gate_75_mode_0, x = gate_73)[name = string("gate_75")]; tensor input_377 = mul(x = gate_75, y = up_37)[name = string("input_377")]; string var_4605_pad_type_0 = const()[name = string("op_4605_pad_type_0"), val = string("valid")]; tensor var_4605_strides_0 = const()[name = string("op_4605_strides_0"), val = tensor([1, 1])]; tensor var_4605_pad_0 = const()[name = string("op_4605_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4605_dilations_0 = const()[name = string("op_4605_dilations_0"), val = tensor([1, 1])]; int32 var_4605_groups_0 = const()[name = string("op_4605_groups_0"), val = int32(1)]; tensor var_4605 = conv(dilations = var_4605_dilations_0, groups = var_4605_groups_0, pad = var_4605_pad_0, pad_type = var_4605_pad_type_0, strides = var_4605_strides_0, weight = encoder_layers_18_mlp_down_proj_weight_quantized, x = input_377)[name = string("op_4605")]; tensor var_4606_axes_0 = const()[name = string("op_4606_axes_0"), val = tensor([2])]; tensor var_4606 = squeeze(axes = var_4606_axes_0, x = var_4605)[name = string("op_4606")]; tensor var_4607 = const()[name = string("op_4607"), val = tensor([0, 2, 1])]; fp16 const_264_promoted_to_fp16 = const()[name = string("const_264_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_303 = transpose(perm = var_4607, x = var_4606)[name = string("transpose_45")]; tensor var_4611_cast_fp16 = mul(x = x_303, y = const_264_promoted_to_fp16)[name = string("op_4611_cast_fp16")]; bool input_379_interleave_0 = const()[name = string("input_379_interleave_0"), val = bool(false)]; tensor input_379_cast_fp16 = concat(axis = var_22, interleave = input_379_interleave_0, values = (x_303, var_4611_cast_fp16))[name = string("input_379_cast_fp16")]; tensor normed_529_axes_0 = const()[name = string("normed_529_axes_0"), val = tensor([-1])]; tensor normed_529_cast_fp16 = layer_norm(axes = normed_529_axes_0, epsilon = var_8_to_fp16, x = input_379_cast_fp16)[name = string("normed_529_cast_fp16")]; tensor var_4616_split_sizes_0 = const()[name = string("op_4616_split_sizes_0"), val = tensor([768, 768])]; int32 var_4616_axis_0 = const()[name = string("op_4616_axis_0"), val = int32(-1)]; tensor var_4616_cast_fp16_0, tensor var_4616_cast_fp16_1 = split(axis = var_4616_axis_0, split_sizes = var_4616_split_sizes_0, x = normed_529_cast_fp16)[name = string("op_4616_cast_fp16")]; tensor var_4620_to_fp16 = const()[name = string("op_4620_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305749248)))]; tensor out_227_cast_fp16 = mul(x = var_4616_cast_fp16_0, y = var_4620_to_fp16)[name = string("out_227_cast_fp16")]; tensor x_305_cast_fp16 = add(x = x_299_cast_fp16, y = out_227_cast_fp16)[name = string("x_305_cast_fp16")]; fp16 const_266_promoted_to_fp16 = const()[name = string("const_266_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4649_cast_fp16 = mul(x = x_305_cast_fp16, y = const_266_promoted_to_fp16)[name = string("op_4649_cast_fp16")]; bool input_381_interleave_0 = const()[name = string("input_381_interleave_0"), val = bool(false)]; tensor input_381_cast_fp16 = concat(axis = var_22, interleave = input_381_interleave_0, values = (x_305_cast_fp16, var_4649_cast_fp16))[name = string("input_381_cast_fp16")]; tensor normed_533_axes_0 = const()[name = string("normed_533_axes_0"), val = tensor([-1])]; tensor normed_533_cast_fp16 = layer_norm(axes = normed_533_axes_0, epsilon = var_8_to_fp16, x = input_381_cast_fp16)[name = string("normed_533_cast_fp16")]; tensor var_4654_split_sizes_0 = const()[name = string("op_4654_split_sizes_0"), val = tensor([768, 768])]; int32 var_4654_axis_0 = const()[name = string("op_4654_axis_0"), val = int32(-1)]; tensor var_4654_cast_fp16_0, tensor var_4654_cast_fp16_1 = split(axis = var_4654_axis_0, split_sizes = var_4654_split_sizes_0, x = normed_533_cast_fp16)[name = string("op_4654_cast_fp16")]; tensor var_4658_to_fp16 = const()[name = string("op_4658_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305750848)))]; tensor out_229_cast_fp16 = mul(x = var_4654_cast_fp16_0, y = var_4658_to_fp16)[name = string("out_229_cast_fp16")]; tensor var_4664 = const()[name = string("op_4664"), val = tensor([0, 2, 1])]; tensor var_4666_axes_0 = const()[name = string("op_4666_axes_0"), val = tensor([2])]; tensor var_4665_cast_fp16 = transpose(perm = var_4664, x = out_229_cast_fp16)[name = string("transpose_44")]; tensor var_4666_cast_fp16 = expand_dims(axes = var_4666_axes_0, x = var_4665_cast_fp16)[name = string("op_4666_cast_fp16")]; string var_4673_pad_type_0 = const()[name = string("op_4673_pad_type_0"), val = string("valid")]; tensor var_4673_strides_0 = const()[name = string("op_4673_strides_0"), val = tensor([1, 1])]; tensor var_4673_pad_0 = const()[name = string("op_4673_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4673_dilations_0 = const()[name = string("op_4673_dilations_0"), val = tensor([1, 1])]; int32 var_4673_groups_0 = const()[name = string("op_4673_groups_0"), val = int32(1)]; tensor var_4673 = conv(dilations = var_4673_dilations_0, groups = var_4673_groups_0, pad = var_4673_pad_0, pad_type = var_4673_pad_type_0, strides = var_4673_strides_0, weight = encoder_layers_19_self_attn_q_proj_weight_quantized, x = var_4666_cast_fp16)[name = string("op_4673")]; tensor var_4674 = const()[name = string("op_4674"), val = tensor([1, 3, 256, 256])]; tensor var_4675 = reshape(shape = var_4674, x = var_4673)[name = string("op_4675")]; tensor var_4676 = const()[name = string("op_4676"), val = tensor([0, 1, 3, 2])]; string var_4683_pad_type_0 = const()[name = string("op_4683_pad_type_0"), val = string("valid")]; tensor var_4683_strides_0 = const()[name = string("op_4683_strides_0"), val = tensor([1, 1])]; tensor var_4683_pad_0 = const()[name = string("op_4683_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4683_dilations_0 = const()[name = string("op_4683_dilations_0"), val = tensor([1, 1])]; int32 var_4683_groups_0 = const()[name = string("op_4683_groups_0"), val = int32(1)]; tensor var_4683 = conv(dilations = var_4683_dilations_0, groups = var_4683_groups_0, pad = var_4683_pad_0, pad_type = var_4683_pad_type_0, strides = var_4683_strides_0, weight = encoder_layers_19_self_attn_k_proj_weight_quantized, x = var_4666_cast_fp16)[name = string("op_4683")]; tensor var_4684 = const()[name = string("op_4684"), val = tensor([1, 1, 256, 256])]; tensor var_4685 = reshape(shape = var_4684, x = var_4683)[name = string("op_4685")]; tensor var_4686 = const()[name = string("op_4686"), val = tensor([0, 1, 3, 2])]; string var_4693_pad_type_0 = const()[name = string("op_4693_pad_type_0"), val = string("valid")]; tensor var_4693_strides_0 = const()[name = string("op_4693_strides_0"), val = tensor([1, 1])]; tensor var_4693_pad_0 = const()[name = string("op_4693_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4693_dilations_0 = const()[name = string("op_4693_dilations_0"), val = tensor([1, 1])]; int32 var_4693_groups_0 = const()[name = string("op_4693_groups_0"), val = int32(1)]; tensor var_4693 = conv(dilations = var_4693_dilations_0, groups = var_4693_groups_0, pad = var_4693_pad_0, pad_type = var_4693_pad_type_0, strides = var_4693_strides_0, weight = encoder_layers_19_self_attn_v_proj_weight_quantized, x = var_4666_cast_fp16)[name = string("op_4693")]; tensor var_4694 = const()[name = string("op_4694"), val = tensor([1, 1, 256, 256])]; tensor var_4695 = reshape(shape = var_4694, x = var_4693)[name = string("op_4695")]; tensor var_4696 = const()[name = string("op_4696"), val = tensor([0, 1, 3, 2])]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_115 = transpose(perm = var_4676, x = var_4675)[name = string("transpose_43")]; tensor var_4702_cast_fp16 = mul(x = q_115, y = const_268_promoted_to_fp16)[name = string("op_4702_cast_fp16")]; bool input_385_interleave_0 = const()[name = string("input_385_interleave_0"), val = bool(false)]; tensor input_385_cast_fp16 = concat(axis = var_22, interleave = input_385_interleave_0, values = (q_115, var_4702_cast_fp16))[name = string("input_385_cast_fp16")]; tensor normed_539_axes_0 = const()[name = string("normed_539_axes_0"), val = tensor([-1])]; tensor normed_539_cast_fp16 = layer_norm(axes = normed_539_axes_0, epsilon = var_8_to_fp16, x = input_385_cast_fp16)[name = string("normed_539_cast_fp16")]; tensor var_4707_split_sizes_0 = const()[name = string("op_4707_split_sizes_0"), val = tensor([256, 256])]; int32 var_4707_axis_0 = const()[name = string("op_4707_axis_0"), val = int32(-1)]; tensor var_4707_cast_fp16_0, tensor var_4707_cast_fp16_1 = split(axis = var_4707_axis_0, split_sizes = var_4707_split_sizes_0, x = normed_539_cast_fp16)[name = string("op_4707_cast_fp16")]; tensor var_4711_to_fp16 = const()[name = string("op_4711_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305752448)))]; tensor out_231_cast_fp16 = mul(x = var_4707_cast_fp16_0, y = var_4711_to_fp16)[name = string("out_231_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_115 = transpose(perm = var_4686, x = var_4685)[name = string("transpose_42")]; tensor var_4718_cast_fp16 = mul(x = k_115, y = const_270_promoted_to_fp16)[name = string("op_4718_cast_fp16")]; bool input_387_interleave_0 = const()[name = string("input_387_interleave_0"), val = bool(false)]; tensor input_387_cast_fp16 = concat(axis = var_22, interleave = input_387_interleave_0, values = (k_115, var_4718_cast_fp16))[name = string("input_387_cast_fp16")]; tensor normed_543_axes_0 = const()[name = string("normed_543_axes_0"), val = tensor([-1])]; tensor normed_543_cast_fp16 = layer_norm(axes = normed_543_axes_0, epsilon = var_8_to_fp16, x = input_387_cast_fp16)[name = string("normed_543_cast_fp16")]; tensor var_4723_split_sizes_0 = const()[name = string("op_4723_split_sizes_0"), val = tensor([256, 256])]; int32 var_4723_axis_0 = const()[name = string("op_4723_axis_0"), val = int32(-1)]; tensor var_4723_cast_fp16_0, tensor var_4723_cast_fp16_1 = split(axis = var_4723_axis_0, split_sizes = var_4723_split_sizes_0, x = normed_543_cast_fp16)[name = string("op_4723_cast_fp16")]; tensor var_4727_to_fp16 = const()[name = string("op_4727_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305753024)))]; tensor out_233_cast_fp16 = mul(x = var_4723_cast_fp16_0, y = var_4727_to_fp16)[name = string("out_233_cast_fp16")]; tensor var_4730 = mul(x = out_231_cast_fp16, y = cos_1_quantized)[name = string("op_4730")]; tensor var_4731_split_sizes_0 = const()[name = string("op_4731_split_sizes_0"), val = tensor([128, 128])]; int32 var_4731_axis_0 = const()[name = string("op_4731_axis_0"), val = int32(-1)]; tensor var_4731_0, tensor var_4731_1 = split(axis = var_4731_axis_0, split_sizes = var_4731_split_sizes_0, x = out_231_cast_fp16)[name = string("op_4731")]; fp16 const_272_promoted = const()[name = string("const_272_promoted"), val = fp16(-0x1p+0)]; tensor var_4733 = mul(x = var_4731_1, y = const_272_promoted)[name = string("op_4733")]; bool var_4735_interleave_0 = const()[name = string("op_4735_interleave_0"), val = bool(false)]; tensor var_4735 = concat(axis = var_22, interleave = var_4735_interleave_0, values = (var_4733, var_4731_0))[name = string("op_4735")]; tensor var_4736 = mul(x = var_4735, y = sin_1_quantized)[name = string("op_4736")]; tensor q_119 = add(x = var_4730, y = var_4736)[name = string("q_119")]; tensor var_4738 = mul(x = out_233_cast_fp16, y = cos_1_quantized)[name = string("op_4738")]; tensor var_4739_split_sizes_0 = const()[name = string("op_4739_split_sizes_0"), val = tensor([128, 128])]; int32 var_4739_axis_0 = const()[name = string("op_4739_axis_0"), val = int32(-1)]; tensor var_4739_0, tensor var_4739_1 = split(axis = var_4739_axis_0, split_sizes = var_4739_split_sizes_0, x = out_233_cast_fp16)[name = string("op_4739")]; fp16 const_273_promoted = const()[name = string("const_273_promoted"), val = fp16(-0x1p+0)]; tensor var_4741 = mul(x = var_4739_1, y = const_273_promoted)[name = string("op_4741")]; bool var_4743_interleave_0 = const()[name = string("op_4743_interleave_0"), val = bool(false)]; tensor var_4743 = concat(axis = var_22, interleave = var_4743_interleave_0, values = (var_4741, var_4739_0))[name = string("op_4743")]; tensor var_4744 = mul(x = var_4743, y = sin_1_quantized)[name = string("op_4744")]; tensor hidden_states_229 = add(x = var_4738, y = var_4744)[name = string("hidden_states_229")]; tensor hidden_states_231_axes_0 = const()[name = string("hidden_states_231_axes_0"), val = tensor([2])]; tensor hidden_states_231 = expand_dims(axes = hidden_states_231_axes_0, x = hidden_states_229)[name = string("hidden_states_231")]; tensor var_4747 = const()[name = string("op_4747"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_233 = tile(reps = var_4747, x = hidden_states_231)[name = string("hidden_states_233")]; tensor var_4749 = const()[name = string("op_4749"), val = tensor([1, 3, 256, 256])]; tensor k_119 = reshape(shape = var_4749, x = hidden_states_233)[name = string("k_119")]; tensor hidden_states_237_axes_0 = const()[name = string("hidden_states_237_axes_0"), val = tensor([2])]; tensor hidden_states_235 = transpose(perm = var_4696, x = var_4695)[name = string("transpose_41")]; tensor hidden_states_237 = expand_dims(axes = hidden_states_237_axes_0, x = hidden_states_235)[name = string("hidden_states_237")]; tensor var_4752 = const()[name = string("op_4752"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_239 = tile(reps = var_4752, x = hidden_states_237)[name = string("hidden_states_239")]; tensor var_4754 = const()[name = string("op_4754"), val = tensor([1, 3, 256, 256])]; tensor v_39 = reshape(shape = var_4754, x = hidden_states_239)[name = string("v_39")]; bool var_4759_transpose_x_1 = const()[name = string("op_4759_transpose_x_1"), val = bool(false)]; bool var_4759_transpose_y_1 = const()[name = string("op_4759_transpose_y_1"), val = bool(true)]; tensor var_4759_cast_fp16 = matmul(transpose_x = var_4759_transpose_x_1, transpose_y = var_4759_transpose_y_1, x = q_119, y = k_119)[name = string("op_4759_cast_fp16")]; fp16 var_4760_to_fp16 = const()[name = string("op_4760_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_115_cast_fp16 = mul(x = var_4759_cast_fp16, y = var_4760_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_117_cast_fp16")]; tensor var_4764_cast_fp16 = softmax(axis = var_22, x = attn_weights_117_cast_fp16)[name = string("op_4764_cast_fp16")]; bool var_4768_transpose_x_0 = const()[name = string("op_4768_transpose_x_0"), val = bool(false)]; bool var_4768_transpose_y_0 = const()[name = string("op_4768_transpose_y_0"), val = bool(false)]; tensor var_4768_cast_fp16 = matmul(transpose_x = var_4768_transpose_x_0, transpose_y = var_4768_transpose_y_0, x = var_4764_cast_fp16, y = v_39)[name = string("op_4768_cast_fp16")]; tensor var_4770 = const()[name = string("op_4770"), val = tensor([0, 2, 1, 3])]; tensor var_4773 = const()[name = string("op_4773"), val = tensor([1, 256, 768])]; tensor var_4771 = transpose(perm = var_4770, x = var_4768_cast_fp16)[name = string("transpose_40")]; tensor attn_out_117 = reshape(shape = var_4773, x = var_4771)[name = string("attn_out_117")]; tensor var_4775 = const()[name = string("op_4775"), val = tensor([0, 2, 1])]; tensor squeeze_19_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305753600))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306343488))))[name = string("squeeze_19_quantized")]; string var_4784_pad_type_0 = const()[name = string("op_4784_pad_type_0"), val = string("valid")]; int32 var_4784_groups_0 = const()[name = string("op_4784_groups_0"), val = int32(1)]; tensor var_4784_strides_0 = const()[name = string("op_4784_strides_0"), val = tensor([1])]; tensor var_4784_pad_0 = const()[name = string("op_4784_pad_0"), val = tensor([0, 0])]; tensor var_4784_dilations_0 = const()[name = string("op_4784_dilations_0"), val = tensor([1])]; tensor var_4776 = transpose(perm = var_4775, x = attn_out_117)[name = string("transpose_39")]; tensor var_4784 = conv(dilations = var_4784_dilations_0, groups = var_4784_groups_0, pad = var_4784_pad_0, pad_type = var_4784_pad_type_0, strides = var_4784_strides_0, weight = squeeze_19_quantized, x = var_4776)[name = string("op_4784")]; tensor var_4785 = const()[name = string("op_4785"), val = tensor([0, 2, 1])]; fp16 const_274_promoted_to_fp16 = const()[name = string("const_274_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_313 = transpose(perm = var_4785, x = var_4784)[name = string("transpose_38")]; tensor var_4789_cast_fp16 = mul(x = x_313, y = const_274_promoted_to_fp16)[name = string("op_4789_cast_fp16")]; bool input_391_interleave_0 = const()[name = string("input_391_interleave_0"), val = bool(false)]; tensor input_391_cast_fp16 = concat(axis = var_22, interleave = input_391_interleave_0, values = (x_313, var_4789_cast_fp16))[name = string("input_391_cast_fp16")]; tensor normed_547_axes_0 = const()[name = string("normed_547_axes_0"), val = tensor([-1])]; tensor normed_547_cast_fp16 = layer_norm(axes = normed_547_axes_0, epsilon = var_8_to_fp16, x = input_391_cast_fp16)[name = string("normed_547_cast_fp16")]; tensor var_4794_split_sizes_0 = const()[name = string("op_4794_split_sizes_0"), val = tensor([768, 768])]; int32 var_4794_axis_0 = const()[name = string("op_4794_axis_0"), val = int32(-1)]; tensor var_4794_cast_fp16_0, tensor var_4794_cast_fp16_1 = split(axis = var_4794_axis_0, split_sizes = var_4794_split_sizes_0, x = normed_547_cast_fp16)[name = string("op_4794_cast_fp16")]; tensor var_4798_to_fp16 = const()[name = string("op_4798_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306345088)))]; tensor out_235_cast_fp16 = mul(x = var_4794_cast_fp16_0, y = var_4798_to_fp16)[name = string("out_235_cast_fp16")]; tensor x_315_cast_fp16 = add(x = x_305_cast_fp16, y = out_235_cast_fp16)[name = string("x_315_cast_fp16")]; fp16 const_276_promoted_to_fp16 = const()[name = string("const_276_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4805_cast_fp16 = mul(x = x_315_cast_fp16, y = const_276_promoted_to_fp16)[name = string("op_4805_cast_fp16")]; bool input_393_interleave_0 = const()[name = string("input_393_interleave_0"), val = bool(false)]; tensor input_393_cast_fp16 = concat(axis = var_22, interleave = input_393_interleave_0, values = (x_315_cast_fp16, var_4805_cast_fp16))[name = string("input_393_cast_fp16")]; tensor normed_551_axes_0 = const()[name = string("normed_551_axes_0"), val = tensor([-1])]; tensor normed_551_cast_fp16 = layer_norm(axes = normed_551_axes_0, epsilon = var_8_to_fp16, x = input_393_cast_fp16)[name = string("normed_551_cast_fp16")]; tensor var_4810_split_sizes_0 = const()[name = string("op_4810_split_sizes_0"), val = tensor([768, 768])]; int32 var_4810_axis_0 = const()[name = string("op_4810_axis_0"), val = int32(-1)]; tensor var_4810_cast_fp16_0, tensor var_4810_cast_fp16_1 = split(axis = var_4810_axis_0, split_sizes = var_4810_split_sizes_0, x = normed_551_cast_fp16)[name = string("op_4810_cast_fp16")]; tensor var_4814_to_fp16 = const()[name = string("op_4814_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306346688)))]; tensor out_237_cast_fp16 = mul(x = var_4810_cast_fp16_0, y = var_4814_to_fp16)[name = string("out_237_cast_fp16")]; tensor var_4821 = const()[name = string("op_4821"), val = tensor([0, 2, 1])]; tensor input_395_axes_0 = const()[name = string("input_395_axes_0"), val = tensor([2])]; tensor var_4822 = transpose(perm = var_4821, x = out_237_cast_fp16)[name = string("transpose_37")]; tensor input_395 = expand_dims(axes = input_395_axes_0, x = var_4822)[name = string("input_395")]; string gate_77_pad_type_0 = const()[name = string("gate_77_pad_type_0"), val = string("valid")]; tensor gate_77_strides_0 = const()[name = string("gate_77_strides_0"), val = tensor([1, 1])]; tensor gate_77_pad_0 = const()[name = string("gate_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_77_dilations_0 = const()[name = string("gate_77_dilations_0"), val = tensor([1, 1])]; int32 gate_77_groups_0 = const()[name = string("gate_77_groups_0"), val = int32(1)]; tensor gate_77 = conv(dilations = gate_77_dilations_0, groups = gate_77_groups_0, pad = gate_77_pad_0, pad_type = gate_77_pad_type_0, strides = gate_77_strides_0, weight = encoder_layers_19_mlp_gate_proj_weight_quantized, x = input_395)[name = string("gate_77")]; string up_39_pad_type_0 = const()[name = string("up_39_pad_type_0"), val = string("valid")]; tensor up_39_strides_0 = const()[name = string("up_39_strides_0"), val = tensor([1, 1])]; tensor up_39_pad_0 = const()[name = string("up_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_39_dilations_0 = const()[name = string("up_39_dilations_0"), val = tensor([1, 1])]; int32 up_39_groups_0 = const()[name = string("up_39_groups_0"), val = int32(1)]; tensor up_39 = conv(dilations = up_39_dilations_0, groups = up_39_groups_0, pad = up_39_pad_0, pad_type = up_39_pad_type_0, strides = up_39_strides_0, weight = encoder_layers_19_mlp_up_proj_weight_quantized, x = input_395)[name = string("up_39")]; string gate_79_mode_0 = const()[name = string("gate_79_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_79 = gelu(mode = gate_79_mode_0, x = gate_77)[name = string("gate_79")]; tensor input_397 = mul(x = gate_79, y = up_39)[name = string("input_397")]; string var_4843_pad_type_0 = const()[name = string("op_4843_pad_type_0"), val = string("valid")]; tensor var_4843_strides_0 = const()[name = string("op_4843_strides_0"), val = tensor([1, 1])]; tensor var_4843_pad_0 = const()[name = string("op_4843_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4843_dilations_0 = const()[name = string("op_4843_dilations_0"), val = tensor([1, 1])]; int32 var_4843_groups_0 = const()[name = string("op_4843_groups_0"), val = int32(1)]; tensor var_4843 = conv(dilations = var_4843_dilations_0, groups = var_4843_groups_0, pad = var_4843_pad_0, pad_type = var_4843_pad_type_0, strides = var_4843_strides_0, weight = encoder_layers_19_mlp_down_proj_weight_quantized, x = input_397)[name = string("op_4843")]; tensor var_4844_axes_0 = const()[name = string("op_4844_axes_0"), val = tensor([2])]; tensor var_4844 = squeeze(axes = var_4844_axes_0, x = var_4843)[name = string("op_4844")]; tensor var_4845 = const()[name = string("op_4845"), val = tensor([0, 2, 1])]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_319 = transpose(perm = var_4845, x = var_4844)[name = string("transpose_36")]; tensor var_4849_cast_fp16 = mul(x = x_319, y = const_278_promoted_to_fp16)[name = string("op_4849_cast_fp16")]; bool input_399_interleave_0 = const()[name = string("input_399_interleave_0"), val = bool(false)]; tensor input_399_cast_fp16 = concat(axis = var_22, interleave = input_399_interleave_0, values = (x_319, var_4849_cast_fp16))[name = string("input_399_cast_fp16")]; tensor normed_557_axes_0 = const()[name = string("normed_557_axes_0"), val = tensor([-1])]; tensor normed_557_cast_fp16 = layer_norm(axes = normed_557_axes_0, epsilon = var_8_to_fp16, x = input_399_cast_fp16)[name = string("normed_557_cast_fp16")]; tensor var_4854_split_sizes_0 = const()[name = string("op_4854_split_sizes_0"), val = tensor([768, 768])]; int32 var_4854_axis_0 = const()[name = string("op_4854_axis_0"), val = int32(-1)]; tensor var_4854_cast_fp16_0, tensor var_4854_cast_fp16_1 = split(axis = var_4854_axis_0, split_sizes = var_4854_split_sizes_0, x = normed_557_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor var_4858_to_fp16 = const()[name = string("op_4858_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306348288)))]; tensor out_239_cast_fp16 = mul(x = var_4854_cast_fp16_0, y = var_4858_to_fp16)[name = string("out_239_cast_fp16")]; tensor x_321_cast_fp16 = add(x = x_315_cast_fp16, y = out_239_cast_fp16)[name = string("x_321_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4887_cast_fp16 = mul(x = x_321_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_4887_cast_fp16")]; bool input_401_interleave_0 = const()[name = string("input_401_interleave_0"), val = bool(false)]; tensor input_401_cast_fp16 = concat(axis = var_22, interleave = input_401_interleave_0, values = (x_321_cast_fp16, var_4887_cast_fp16))[name = string("input_401_cast_fp16")]; tensor normed_561_axes_0 = const()[name = string("normed_561_axes_0"), val = tensor([-1])]; tensor normed_561_cast_fp16 = layer_norm(axes = normed_561_axes_0, epsilon = var_8_to_fp16, x = input_401_cast_fp16)[name = string("normed_561_cast_fp16")]; tensor var_4892_split_sizes_0 = const()[name = string("op_4892_split_sizes_0"), val = tensor([768, 768])]; int32 var_4892_axis_0 = const()[name = string("op_4892_axis_0"), val = int32(-1)]; tensor var_4892_cast_fp16_0, tensor var_4892_cast_fp16_1 = split(axis = var_4892_axis_0, split_sizes = var_4892_split_sizes_0, x = normed_561_cast_fp16)[name = string("op_4892_cast_fp16")]; tensor var_4896_to_fp16 = const()[name = string("op_4896_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306349888)))]; tensor out_241_cast_fp16 = mul(x = var_4892_cast_fp16_0, y = var_4896_to_fp16)[name = string("out_241_cast_fp16")]; tensor var_4902 = const()[name = string("op_4902"), val = tensor([0, 2, 1])]; tensor var_4904_axes_0 = const()[name = string("op_4904_axes_0"), val = tensor([2])]; tensor var_4903_cast_fp16 = transpose(perm = var_4902, x = out_241_cast_fp16)[name = string("transpose_35")]; tensor var_4904_cast_fp16 = expand_dims(axes = var_4904_axes_0, x = var_4903_cast_fp16)[name = string("op_4904_cast_fp16")]; string var_4911_pad_type_0 = const()[name = string("op_4911_pad_type_0"), val = string("valid")]; tensor var_4911_strides_0 = const()[name = string("op_4911_strides_0"), val = tensor([1, 1])]; tensor var_4911_pad_0 = const()[name = string("op_4911_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4911_dilations_0 = const()[name = string("op_4911_dilations_0"), val = tensor([1, 1])]; int32 var_4911_groups_0 = const()[name = string("op_4911_groups_0"), val = int32(1)]; tensor var_4911 = conv(dilations = var_4911_dilations_0, groups = var_4911_groups_0, pad = var_4911_pad_0, pad_type = var_4911_pad_type_0, strides = var_4911_strides_0, weight = encoder_layers_20_self_attn_q_proj_weight_quantized, x = var_4904_cast_fp16)[name = string("op_4911")]; tensor var_4912 = const()[name = string("op_4912"), val = tensor([1, 3, 256, 256])]; tensor var_4913 = reshape(shape = var_4912, x = var_4911)[name = string("op_4913")]; tensor var_4914 = const()[name = string("op_4914"), val = tensor([0, 1, 3, 2])]; string var_4921_pad_type_0 = const()[name = string("op_4921_pad_type_0"), val = string("valid")]; tensor var_4921_strides_0 = const()[name = string("op_4921_strides_0"), val = tensor([1, 1])]; tensor var_4921_pad_0 = const()[name = string("op_4921_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4921_dilations_0 = const()[name = string("op_4921_dilations_0"), val = tensor([1, 1])]; int32 var_4921_groups_0 = const()[name = string("op_4921_groups_0"), val = int32(1)]; tensor var_4921 = conv(dilations = var_4921_dilations_0, groups = var_4921_groups_0, pad = var_4921_pad_0, pad_type = var_4921_pad_type_0, strides = var_4921_strides_0, weight = encoder_layers_20_self_attn_k_proj_weight_quantized, x = var_4904_cast_fp16)[name = string("op_4921")]; tensor var_4922 = const()[name = string("op_4922"), val = tensor([1, 1, 256, 256])]; tensor var_4923 = reshape(shape = var_4922, x = var_4921)[name = string("op_4923")]; tensor var_4924 = const()[name = string("op_4924"), val = tensor([0, 1, 3, 2])]; string var_4931_pad_type_0 = const()[name = string("op_4931_pad_type_0"), val = string("valid")]; tensor var_4931_strides_0 = const()[name = string("op_4931_strides_0"), val = tensor([1, 1])]; tensor var_4931_pad_0 = const()[name = string("op_4931_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4931_dilations_0 = const()[name = string("op_4931_dilations_0"), val = tensor([1, 1])]; int32 var_4931_groups_0 = const()[name = string("op_4931_groups_0"), val = int32(1)]; tensor var_4931 = conv(dilations = var_4931_dilations_0, groups = var_4931_groups_0, pad = var_4931_pad_0, pad_type = var_4931_pad_type_0, strides = var_4931_strides_0, weight = encoder_layers_20_self_attn_v_proj_weight_quantized, x = var_4904_cast_fp16)[name = string("op_4931")]; tensor var_4932 = const()[name = string("op_4932"), val = tensor([1, 1, 256, 256])]; tensor var_4933 = reshape(shape = var_4932, x = var_4931)[name = string("op_4933")]; tensor var_4934 = const()[name = string("op_4934"), val = tensor([0, 1, 3, 2])]; fp16 const_282_promoted_to_fp16 = const()[name = string("const_282_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_121 = transpose(perm = var_4914, x = var_4913)[name = string("transpose_34")]; tensor var_4940_cast_fp16 = mul(x = q_121, y = const_282_promoted_to_fp16)[name = string("op_4940_cast_fp16")]; bool input_405_interleave_0 = const()[name = string("input_405_interleave_0"), val = bool(false)]; tensor input_405_cast_fp16 = concat(axis = var_22, interleave = input_405_interleave_0, values = (q_121, var_4940_cast_fp16))[name = string("input_405_cast_fp16")]; tensor normed_567_axes_0 = const()[name = string("normed_567_axes_0"), val = tensor([-1])]; tensor normed_567_cast_fp16 = layer_norm(axes = normed_567_axes_0, epsilon = var_8_to_fp16, x = input_405_cast_fp16)[name = string("normed_567_cast_fp16")]; tensor var_4945_split_sizes_0 = const()[name = string("op_4945_split_sizes_0"), val = tensor([256, 256])]; int32 var_4945_axis_0 = const()[name = string("op_4945_axis_0"), val = int32(-1)]; tensor var_4945_cast_fp16_0, tensor var_4945_cast_fp16_1 = split(axis = var_4945_axis_0, split_sizes = var_4945_split_sizes_0, x = normed_567_cast_fp16)[name = string("op_4945_cast_fp16")]; tensor var_4949_to_fp16 = const()[name = string("op_4949_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306351488)))]; tensor out_243_cast_fp16 = mul(x = var_4945_cast_fp16_0, y = var_4949_to_fp16)[name = string("out_243_cast_fp16")]; fp16 const_284_promoted_to_fp16 = const()[name = string("const_284_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_121 = transpose(perm = var_4924, x = var_4923)[name = string("transpose_33")]; tensor var_4956_cast_fp16 = mul(x = k_121, y = const_284_promoted_to_fp16)[name = string("op_4956_cast_fp16")]; bool input_407_interleave_0 = const()[name = string("input_407_interleave_0"), val = bool(false)]; tensor input_407_cast_fp16 = concat(axis = var_22, interleave = input_407_interleave_0, values = (k_121, var_4956_cast_fp16))[name = string("input_407_cast_fp16")]; tensor normed_571_axes_0 = const()[name = string("normed_571_axes_0"), val = tensor([-1])]; tensor normed_571_cast_fp16 = layer_norm(axes = normed_571_axes_0, epsilon = var_8_to_fp16, x = input_407_cast_fp16)[name = string("normed_571_cast_fp16")]; tensor var_4961_split_sizes_0 = const()[name = string("op_4961_split_sizes_0"), val = tensor([256, 256])]; int32 var_4961_axis_0 = const()[name = string("op_4961_axis_0"), val = int32(-1)]; tensor var_4961_cast_fp16_0, tensor var_4961_cast_fp16_1 = split(axis = var_4961_axis_0, split_sizes = var_4961_split_sizes_0, x = normed_571_cast_fp16)[name = string("op_4961_cast_fp16")]; tensor var_4965_to_fp16 = const()[name = string("op_4965_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306352064)))]; tensor out_245_cast_fp16 = mul(x = var_4961_cast_fp16_0, y = var_4965_to_fp16)[name = string("out_245_cast_fp16")]; tensor var_4968 = mul(x = out_243_cast_fp16, y = cos_1_quantized)[name = string("op_4968")]; tensor var_4969_split_sizes_0 = const()[name = string("op_4969_split_sizes_0"), val = tensor([128, 128])]; int32 var_4969_axis_0 = const()[name = string("op_4969_axis_0"), val = int32(-1)]; tensor var_4969_0, tensor var_4969_1 = split(axis = var_4969_axis_0, split_sizes = var_4969_split_sizes_0, x = out_243_cast_fp16)[name = string("op_4969")]; fp16 const_286_promoted = const()[name = string("const_286_promoted"), val = fp16(-0x1p+0)]; tensor var_4971 = mul(x = var_4969_1, y = const_286_promoted)[name = string("op_4971")]; bool var_4973_interleave_0 = const()[name = string("op_4973_interleave_0"), val = bool(false)]; tensor var_4973 = concat(axis = var_22, interleave = var_4973_interleave_0, values = (var_4971, var_4969_0))[name = string("op_4973")]; tensor var_4974 = mul(x = var_4973, y = sin_1_quantized)[name = string("op_4974")]; tensor q_125 = add(x = var_4968, y = var_4974)[name = string("q_125")]; tensor var_4976 = mul(x = out_245_cast_fp16, y = cos_1_quantized)[name = string("op_4976")]; tensor var_4977_split_sizes_0 = const()[name = string("op_4977_split_sizes_0"), val = tensor([128, 128])]; int32 var_4977_axis_0 = const()[name = string("op_4977_axis_0"), val = int32(-1)]; tensor var_4977_0, tensor var_4977_1 = split(axis = var_4977_axis_0, split_sizes = var_4977_split_sizes_0, x = out_245_cast_fp16)[name = string("op_4977")]; fp16 const_287_promoted = const()[name = string("const_287_promoted"), val = fp16(-0x1p+0)]; tensor var_4979 = mul(x = var_4977_1, y = const_287_promoted)[name = string("op_4979")]; bool var_4981_interleave_0 = const()[name = string("op_4981_interleave_0"), val = bool(false)]; tensor var_4981 = concat(axis = var_22, interleave = var_4981_interleave_0, values = (var_4979, var_4977_0))[name = string("op_4981")]; tensor var_4982 = mul(x = var_4981, y = sin_1_quantized)[name = string("op_4982")]; tensor hidden_states_241 = add(x = var_4976, y = var_4982)[name = string("hidden_states_241")]; tensor hidden_states_243_axes_0 = const()[name = string("hidden_states_243_axes_0"), val = tensor([2])]; tensor hidden_states_243 = expand_dims(axes = hidden_states_243_axes_0, x = hidden_states_241)[name = string("hidden_states_243")]; tensor var_4985 = const()[name = string("op_4985"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_245 = tile(reps = var_4985, x = hidden_states_243)[name = string("hidden_states_245")]; tensor var_4987 = const()[name = string("op_4987"), val = tensor([1, 3, 256, 256])]; tensor k_125 = reshape(shape = var_4987, x = hidden_states_245)[name = string("k_125")]; tensor hidden_states_249_axes_0 = const()[name = string("hidden_states_249_axes_0"), val = tensor([2])]; tensor hidden_states_247 = transpose(perm = var_4934, x = var_4933)[name = string("transpose_32")]; tensor hidden_states_249 = expand_dims(axes = hidden_states_249_axes_0, x = hidden_states_247)[name = string("hidden_states_249")]; tensor var_4990 = const()[name = string("op_4990"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_251 = tile(reps = var_4990, x = hidden_states_249)[name = string("hidden_states_251")]; tensor var_4992 = const()[name = string("op_4992"), val = tensor([1, 3, 256, 256])]; tensor v_41 = reshape(shape = var_4992, x = hidden_states_251)[name = string("v_41")]; bool var_4997_transpose_x_1 = const()[name = string("op_4997_transpose_x_1"), val = bool(false)]; bool var_4997_transpose_y_1 = const()[name = string("op_4997_transpose_y_1"), val = bool(true)]; tensor var_4997_cast_fp16 = matmul(transpose_x = var_4997_transpose_x_1, transpose_y = var_4997_transpose_y_1, x = q_125, y = k_125)[name = string("op_4997_cast_fp16")]; fp16 var_4998_to_fp16 = const()[name = string("op_4998_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_121_cast_fp16 = mul(x = var_4997_cast_fp16, y = var_4998_to_fp16)[name = string("attn_weights_121_cast_fp16")]; tensor attn_weights_123_cast_fp16 = add(x = attn_weights_121_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor var_5002_cast_fp16 = softmax(axis = var_22, x = attn_weights_123_cast_fp16)[name = string("op_5002_cast_fp16")]; bool var_5006_transpose_x_0 = const()[name = string("op_5006_transpose_x_0"), val = bool(false)]; bool var_5006_transpose_y_0 = const()[name = string("op_5006_transpose_y_0"), val = bool(false)]; tensor var_5006_cast_fp16 = matmul(transpose_x = var_5006_transpose_x_0, transpose_y = var_5006_transpose_y_0, x = var_5002_cast_fp16, y = v_41)[name = string("op_5006_cast_fp16")]; tensor var_5008 = const()[name = string("op_5008"), val = tensor([0, 2, 1, 3])]; tensor var_5011 = const()[name = string("op_5011"), val = tensor([1, 256, 768])]; tensor var_5009 = transpose(perm = var_5008, x = var_5006_cast_fp16)[name = string("transpose_31")]; tensor attn_out_123 = reshape(shape = var_5011, x = var_5009)[name = string("attn_out_123")]; tensor var_5013 = const()[name = string("op_5013"), val = tensor([0, 2, 1])]; tensor squeeze_20_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306352640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306942528))))[name = string("squeeze_20_quantized")]; string var_5022_pad_type_0 = const()[name = string("op_5022_pad_type_0"), val = string("valid")]; int32 var_5022_groups_0 = const()[name = string("op_5022_groups_0"), val = int32(1)]; tensor var_5022_strides_0 = const()[name = string("op_5022_strides_0"), val = tensor([1])]; tensor var_5022_pad_0 = const()[name = string("op_5022_pad_0"), val = tensor([0, 0])]; tensor var_5022_dilations_0 = const()[name = string("op_5022_dilations_0"), val = tensor([1])]; tensor var_5014 = transpose(perm = var_5013, x = attn_out_123)[name = string("transpose_30")]; tensor var_5022 = conv(dilations = var_5022_dilations_0, groups = var_5022_groups_0, pad = var_5022_pad_0, pad_type = var_5022_pad_type_0, strides = var_5022_strides_0, weight = squeeze_20_quantized, x = var_5014)[name = string("op_5022")]; tensor var_5023 = const()[name = string("op_5023"), val = tensor([0, 2, 1])]; fp16 const_288_promoted_to_fp16 = const()[name = string("const_288_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_329 = transpose(perm = var_5023, x = var_5022)[name = string("transpose_29")]; tensor var_5027_cast_fp16 = mul(x = x_329, y = const_288_promoted_to_fp16)[name = string("op_5027_cast_fp16")]; bool input_411_interleave_0 = const()[name = string("input_411_interleave_0"), val = bool(false)]; tensor input_411_cast_fp16 = concat(axis = var_22, interleave = input_411_interleave_0, values = (x_329, var_5027_cast_fp16))[name = string("input_411_cast_fp16")]; tensor normed_575_axes_0 = const()[name = string("normed_575_axes_0"), val = tensor([-1])]; tensor normed_575_cast_fp16 = layer_norm(axes = normed_575_axes_0, epsilon = var_8_to_fp16, x = input_411_cast_fp16)[name = string("normed_575_cast_fp16")]; tensor var_5032_split_sizes_0 = const()[name = string("op_5032_split_sizes_0"), val = tensor([768, 768])]; int32 var_5032_axis_0 = const()[name = string("op_5032_axis_0"), val = int32(-1)]; tensor var_5032_cast_fp16_0, tensor var_5032_cast_fp16_1 = split(axis = var_5032_axis_0, split_sizes = var_5032_split_sizes_0, x = normed_575_cast_fp16)[name = string("op_5032_cast_fp16")]; tensor var_5036_to_fp16 = const()[name = string("op_5036_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306944128)))]; tensor out_247_cast_fp16 = mul(x = var_5032_cast_fp16_0, y = var_5036_to_fp16)[name = string("out_247_cast_fp16")]; tensor x_331_cast_fp16 = add(x = x_321_cast_fp16, y = out_247_cast_fp16)[name = string("x_331_cast_fp16")]; fp16 const_290_promoted_to_fp16 = const()[name = string("const_290_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5043_cast_fp16 = mul(x = x_331_cast_fp16, y = const_290_promoted_to_fp16)[name = string("op_5043_cast_fp16")]; bool input_413_interleave_0 = const()[name = string("input_413_interleave_0"), val = bool(false)]; tensor input_413_cast_fp16 = concat(axis = var_22, interleave = input_413_interleave_0, values = (x_331_cast_fp16, var_5043_cast_fp16))[name = string("input_413_cast_fp16")]; tensor normed_579_axes_0 = const()[name = string("normed_579_axes_0"), val = tensor([-1])]; tensor normed_579_cast_fp16 = layer_norm(axes = normed_579_axes_0, epsilon = var_8_to_fp16, x = input_413_cast_fp16)[name = string("normed_579_cast_fp16")]; tensor var_5048_split_sizes_0 = const()[name = string("op_5048_split_sizes_0"), val = tensor([768, 768])]; int32 var_5048_axis_0 = const()[name = string("op_5048_axis_0"), val = int32(-1)]; tensor var_5048_cast_fp16_0, tensor var_5048_cast_fp16_1 = split(axis = var_5048_axis_0, split_sizes = var_5048_split_sizes_0, x = normed_579_cast_fp16)[name = string("op_5048_cast_fp16")]; tensor var_5052_to_fp16 = const()[name = string("op_5052_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306945728)))]; tensor out_249_cast_fp16 = mul(x = var_5048_cast_fp16_0, y = var_5052_to_fp16)[name = string("out_249_cast_fp16")]; tensor var_5059 = const()[name = string("op_5059"), val = tensor([0, 2, 1])]; tensor input_415_axes_0 = const()[name = string("input_415_axes_0"), val = tensor([2])]; tensor var_5060 = transpose(perm = var_5059, x = out_249_cast_fp16)[name = string("transpose_28")]; tensor input_415 = expand_dims(axes = input_415_axes_0, x = var_5060)[name = string("input_415")]; string gate_81_pad_type_0 = const()[name = string("gate_81_pad_type_0"), val = string("valid")]; tensor gate_81_strides_0 = const()[name = string("gate_81_strides_0"), val = tensor([1, 1])]; tensor gate_81_pad_0 = const()[name = string("gate_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_81_dilations_0 = const()[name = string("gate_81_dilations_0"), val = tensor([1, 1])]; int32 gate_81_groups_0 = const()[name = string("gate_81_groups_0"), val = int32(1)]; tensor gate_81 = conv(dilations = gate_81_dilations_0, groups = gate_81_groups_0, pad = gate_81_pad_0, pad_type = gate_81_pad_type_0, strides = gate_81_strides_0, weight = encoder_layers_20_mlp_gate_proj_weight_quantized, x = input_415)[name = string("gate_81")]; string up_41_pad_type_0 = const()[name = string("up_41_pad_type_0"), val = string("valid")]; tensor up_41_strides_0 = const()[name = string("up_41_strides_0"), val = tensor([1, 1])]; tensor up_41_pad_0 = const()[name = string("up_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_41_dilations_0 = const()[name = string("up_41_dilations_0"), val = tensor([1, 1])]; int32 up_41_groups_0 = const()[name = string("up_41_groups_0"), val = int32(1)]; tensor up_41 = conv(dilations = up_41_dilations_0, groups = up_41_groups_0, pad = up_41_pad_0, pad_type = up_41_pad_type_0, strides = up_41_strides_0, weight = encoder_layers_20_mlp_up_proj_weight_quantized, x = input_415)[name = string("up_41")]; string gate_83_mode_0 = const()[name = string("gate_83_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_83 = gelu(mode = gate_83_mode_0, x = gate_81)[name = string("gate_83")]; tensor input_417 = mul(x = gate_83, y = up_41)[name = string("input_417")]; string var_5081_pad_type_0 = const()[name = string("op_5081_pad_type_0"), val = string("valid")]; tensor var_5081_strides_0 = const()[name = string("op_5081_strides_0"), val = tensor([1, 1])]; tensor var_5081_pad_0 = const()[name = string("op_5081_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5081_dilations_0 = const()[name = string("op_5081_dilations_0"), val = tensor([1, 1])]; int32 var_5081_groups_0 = const()[name = string("op_5081_groups_0"), val = int32(1)]; tensor var_5081 = conv(dilations = var_5081_dilations_0, groups = var_5081_groups_0, pad = var_5081_pad_0, pad_type = var_5081_pad_type_0, strides = var_5081_strides_0, weight = encoder_layers_20_mlp_down_proj_weight_quantized, x = input_417)[name = string("op_5081")]; tensor var_5082_axes_0 = const()[name = string("op_5082_axes_0"), val = tensor([2])]; tensor var_5082 = squeeze(axes = var_5082_axes_0, x = var_5081)[name = string("op_5082")]; tensor var_5083 = const()[name = string("op_5083"), val = tensor([0, 2, 1])]; fp16 const_292_promoted_to_fp16 = const()[name = string("const_292_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_335 = transpose(perm = var_5083, x = var_5082)[name = string("transpose_27")]; tensor var_5087_cast_fp16 = mul(x = x_335, y = const_292_promoted_to_fp16)[name = string("op_5087_cast_fp16")]; bool input_419_interleave_0 = const()[name = string("input_419_interleave_0"), val = bool(false)]; tensor input_419_cast_fp16 = concat(axis = var_22, interleave = input_419_interleave_0, values = (x_335, var_5087_cast_fp16))[name = string("input_419_cast_fp16")]; tensor normed_585_axes_0 = const()[name = string("normed_585_axes_0"), val = tensor([-1])]; tensor normed_585_cast_fp16 = layer_norm(axes = normed_585_axes_0, epsilon = var_8_to_fp16, x = input_419_cast_fp16)[name = string("normed_585_cast_fp16")]; tensor var_5092_split_sizes_0 = const()[name = string("op_5092_split_sizes_0"), val = tensor([768, 768])]; int32 var_5092_axis_0 = const()[name = string("op_5092_axis_0"), val = int32(-1)]; tensor var_5092_cast_fp16_0, tensor var_5092_cast_fp16_1 = split(axis = var_5092_axis_0, split_sizes = var_5092_split_sizes_0, x = normed_585_cast_fp16)[name = string("op_5092_cast_fp16")]; tensor var_5096_to_fp16 = const()[name = string("op_5096_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306947328)))]; tensor out_251_cast_fp16 = mul(x = var_5092_cast_fp16_0, y = var_5096_to_fp16)[name = string("out_251_cast_fp16")]; tensor x_337_cast_fp16 = add(x = x_331_cast_fp16, y = out_251_cast_fp16)[name = string("x_337_cast_fp16")]; fp16 const_294_promoted_to_fp16 = const()[name = string("const_294_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5125_cast_fp16 = mul(x = x_337_cast_fp16, y = const_294_promoted_to_fp16)[name = string("op_5125_cast_fp16")]; bool input_421_interleave_0 = const()[name = string("input_421_interleave_0"), val = bool(false)]; tensor input_421_cast_fp16 = concat(axis = var_22, interleave = input_421_interleave_0, values = (x_337_cast_fp16, var_5125_cast_fp16))[name = string("input_421_cast_fp16")]; tensor normed_589_axes_0 = const()[name = string("normed_589_axes_0"), val = tensor([-1])]; tensor normed_589_cast_fp16 = layer_norm(axes = normed_589_axes_0, epsilon = var_8_to_fp16, x = input_421_cast_fp16)[name = string("normed_589_cast_fp16")]; tensor var_5130_split_sizes_0 = const()[name = string("op_5130_split_sizes_0"), val = tensor([768, 768])]; int32 var_5130_axis_0 = const()[name = string("op_5130_axis_0"), val = int32(-1)]; tensor var_5130_cast_fp16_0, tensor var_5130_cast_fp16_1 = split(axis = var_5130_axis_0, split_sizes = var_5130_split_sizes_0, x = normed_589_cast_fp16)[name = string("op_5130_cast_fp16")]; tensor var_5134_to_fp16 = const()[name = string("op_5134_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306948928)))]; tensor out_253_cast_fp16 = mul(x = var_5130_cast_fp16_0, y = var_5134_to_fp16)[name = string("out_253_cast_fp16")]; tensor var_5140 = const()[name = string("op_5140"), val = tensor([0, 2, 1])]; tensor var_5142_axes_0 = const()[name = string("op_5142_axes_0"), val = tensor([2])]; tensor var_5141_cast_fp16 = transpose(perm = var_5140, x = out_253_cast_fp16)[name = string("transpose_26")]; tensor var_5142_cast_fp16 = expand_dims(axes = var_5142_axes_0, x = var_5141_cast_fp16)[name = string("op_5142_cast_fp16")]; string var_5149_pad_type_0 = const()[name = string("op_5149_pad_type_0"), val = string("valid")]; tensor var_5149_strides_0 = const()[name = string("op_5149_strides_0"), val = tensor([1, 1])]; tensor var_5149_pad_0 = const()[name = string("op_5149_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5149_dilations_0 = const()[name = string("op_5149_dilations_0"), val = tensor([1, 1])]; int32 var_5149_groups_0 = const()[name = string("op_5149_groups_0"), val = int32(1)]; tensor var_5149 = conv(dilations = var_5149_dilations_0, groups = var_5149_groups_0, pad = var_5149_pad_0, pad_type = var_5149_pad_type_0, strides = var_5149_strides_0, weight = encoder_layers_21_self_attn_q_proj_weight_quantized, x = var_5142_cast_fp16)[name = string("op_5149")]; tensor var_5150 = const()[name = string("op_5150"), val = tensor([1, 3, 256, 256])]; tensor var_5151 = reshape(shape = var_5150, x = var_5149)[name = string("op_5151")]; tensor var_5152 = const()[name = string("op_5152"), val = tensor([0, 1, 3, 2])]; string var_5159_pad_type_0 = const()[name = string("op_5159_pad_type_0"), val = string("valid")]; tensor var_5159_strides_0 = const()[name = string("op_5159_strides_0"), val = tensor([1, 1])]; tensor var_5159_pad_0 = const()[name = string("op_5159_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5159_dilations_0 = const()[name = string("op_5159_dilations_0"), val = tensor([1, 1])]; int32 var_5159_groups_0 = const()[name = string("op_5159_groups_0"), val = int32(1)]; tensor var_5159 = conv(dilations = var_5159_dilations_0, groups = var_5159_groups_0, pad = var_5159_pad_0, pad_type = var_5159_pad_type_0, strides = var_5159_strides_0, weight = encoder_layers_21_self_attn_k_proj_weight_quantized, x = var_5142_cast_fp16)[name = string("op_5159")]; tensor var_5160 = const()[name = string("op_5160"), val = tensor([1, 1, 256, 256])]; tensor var_5161 = reshape(shape = var_5160, x = var_5159)[name = string("op_5161")]; tensor var_5162 = const()[name = string("op_5162"), val = tensor([0, 1, 3, 2])]; string var_5169_pad_type_0 = const()[name = string("op_5169_pad_type_0"), val = string("valid")]; tensor var_5169_strides_0 = const()[name = string("op_5169_strides_0"), val = tensor([1, 1])]; tensor var_5169_pad_0 = const()[name = string("op_5169_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5169_dilations_0 = const()[name = string("op_5169_dilations_0"), val = tensor([1, 1])]; int32 var_5169_groups_0 = const()[name = string("op_5169_groups_0"), val = int32(1)]; tensor var_5169 = conv(dilations = var_5169_dilations_0, groups = var_5169_groups_0, pad = var_5169_pad_0, pad_type = var_5169_pad_type_0, strides = var_5169_strides_0, weight = encoder_layers_21_self_attn_v_proj_weight_quantized, x = var_5142_cast_fp16)[name = string("op_5169")]; tensor var_5170 = const()[name = string("op_5170"), val = tensor([1, 1, 256, 256])]; tensor var_5171 = reshape(shape = var_5170, x = var_5169)[name = string("op_5171")]; tensor var_5172 = const()[name = string("op_5172"), val = tensor([0, 1, 3, 2])]; fp16 const_296_promoted_to_fp16 = const()[name = string("const_296_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_127 = transpose(perm = var_5152, x = var_5151)[name = string("transpose_25")]; tensor var_5178_cast_fp16 = mul(x = q_127, y = const_296_promoted_to_fp16)[name = string("op_5178_cast_fp16")]; bool input_425_interleave_0 = const()[name = string("input_425_interleave_0"), val = bool(false)]; tensor input_425_cast_fp16 = concat(axis = var_22, interleave = input_425_interleave_0, values = (q_127, var_5178_cast_fp16))[name = string("input_425_cast_fp16")]; tensor normed_595_axes_0 = const()[name = string("normed_595_axes_0"), val = tensor([-1])]; tensor normed_595_cast_fp16 = layer_norm(axes = normed_595_axes_0, epsilon = var_8_to_fp16, x = input_425_cast_fp16)[name = string("normed_595_cast_fp16")]; tensor var_5183_split_sizes_0 = const()[name = string("op_5183_split_sizes_0"), val = tensor([256, 256])]; int32 var_5183_axis_0 = const()[name = string("op_5183_axis_0"), val = int32(-1)]; tensor var_5183_cast_fp16_0, tensor var_5183_cast_fp16_1 = split(axis = var_5183_axis_0, split_sizes = var_5183_split_sizes_0, x = normed_595_cast_fp16)[name = string("op_5183_cast_fp16")]; tensor var_5187_to_fp16 = const()[name = string("op_5187_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306950528)))]; tensor out_255_cast_fp16 = mul(x = var_5183_cast_fp16_0, y = var_5187_to_fp16)[name = string("out_255_cast_fp16")]; fp16 const_298_promoted_to_fp16 = const()[name = string("const_298_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_127 = transpose(perm = var_5162, x = var_5161)[name = string("transpose_24")]; tensor var_5194_cast_fp16 = mul(x = k_127, y = const_298_promoted_to_fp16)[name = string("op_5194_cast_fp16")]; bool input_427_interleave_0 = const()[name = string("input_427_interleave_0"), val = bool(false)]; tensor input_427_cast_fp16 = concat(axis = var_22, interleave = input_427_interleave_0, values = (k_127, var_5194_cast_fp16))[name = string("input_427_cast_fp16")]; tensor normed_599_axes_0 = const()[name = string("normed_599_axes_0"), val = tensor([-1])]; tensor normed_599_cast_fp16 = layer_norm(axes = normed_599_axes_0, epsilon = var_8_to_fp16, x = input_427_cast_fp16)[name = string("normed_599_cast_fp16")]; tensor var_5199_split_sizes_0 = const()[name = string("op_5199_split_sizes_0"), val = tensor([256, 256])]; int32 var_5199_axis_0 = const()[name = string("op_5199_axis_0"), val = int32(-1)]; tensor var_5199_cast_fp16_0, tensor var_5199_cast_fp16_1 = split(axis = var_5199_axis_0, split_sizes = var_5199_split_sizes_0, x = normed_599_cast_fp16)[name = string("op_5199_cast_fp16")]; tensor var_5203_to_fp16 = const()[name = string("op_5203_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306951104)))]; tensor out_257_cast_fp16 = mul(x = var_5199_cast_fp16_0, y = var_5203_to_fp16)[name = string("out_257_cast_fp16")]; tensor var_5206 = mul(x = out_255_cast_fp16, y = cos_1_quantized)[name = string("op_5206")]; tensor var_5207_split_sizes_0 = const()[name = string("op_5207_split_sizes_0"), val = tensor([128, 128])]; int32 var_5207_axis_0 = const()[name = string("op_5207_axis_0"), val = int32(-1)]; tensor var_5207_0, tensor var_5207_1 = split(axis = var_5207_axis_0, split_sizes = var_5207_split_sizes_0, x = out_255_cast_fp16)[name = string("op_5207")]; fp16 const_300_promoted = const()[name = string("const_300_promoted"), val = fp16(-0x1p+0)]; tensor var_5209 = mul(x = var_5207_1, y = const_300_promoted)[name = string("op_5209")]; bool var_5211_interleave_0 = const()[name = string("op_5211_interleave_0"), val = bool(false)]; tensor var_5211 = concat(axis = var_22, interleave = var_5211_interleave_0, values = (var_5209, var_5207_0))[name = string("op_5211")]; tensor var_5212 = mul(x = var_5211, y = sin_1_quantized)[name = string("op_5212")]; tensor q_131 = add(x = var_5206, y = var_5212)[name = string("q_131")]; tensor var_5214 = mul(x = out_257_cast_fp16, y = cos_1_quantized)[name = string("op_5214")]; tensor var_5215_split_sizes_0 = const()[name = string("op_5215_split_sizes_0"), val = tensor([128, 128])]; int32 var_5215_axis_0 = const()[name = string("op_5215_axis_0"), val = int32(-1)]; tensor var_5215_0, tensor var_5215_1 = split(axis = var_5215_axis_0, split_sizes = var_5215_split_sizes_0, x = out_257_cast_fp16)[name = string("op_5215")]; fp16 const_301_promoted = const()[name = string("const_301_promoted"), val = fp16(-0x1p+0)]; tensor var_5217 = mul(x = var_5215_1, y = const_301_promoted)[name = string("op_5217")]; bool var_5219_interleave_0 = const()[name = string("op_5219_interleave_0"), val = bool(false)]; tensor var_5219 = concat(axis = var_22, interleave = var_5219_interleave_0, values = (var_5217, var_5215_0))[name = string("op_5219")]; tensor var_5220 = mul(x = var_5219, y = sin_1_quantized)[name = string("op_5220")]; tensor hidden_states_253 = add(x = var_5214, y = var_5220)[name = string("hidden_states_253")]; tensor hidden_states_255_axes_0 = const()[name = string("hidden_states_255_axes_0"), val = tensor([2])]; tensor hidden_states_255 = expand_dims(axes = hidden_states_255_axes_0, x = hidden_states_253)[name = string("hidden_states_255")]; tensor var_5223 = const()[name = string("op_5223"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_257 = tile(reps = var_5223, x = hidden_states_255)[name = string("hidden_states_257")]; tensor var_5225 = const()[name = string("op_5225"), val = tensor([1, 3, 256, 256])]; tensor k_131 = reshape(shape = var_5225, x = hidden_states_257)[name = string("k_131")]; tensor hidden_states_261_axes_0 = const()[name = string("hidden_states_261_axes_0"), val = tensor([2])]; tensor hidden_states_259 = transpose(perm = var_5172, x = var_5171)[name = string("transpose_23")]; tensor hidden_states_261 = expand_dims(axes = hidden_states_261_axes_0, x = hidden_states_259)[name = string("hidden_states_261")]; tensor var_5228 = const()[name = string("op_5228"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_263 = tile(reps = var_5228, x = hidden_states_261)[name = string("hidden_states_263")]; tensor var_5230 = const()[name = string("op_5230"), val = tensor([1, 3, 256, 256])]; tensor v_43 = reshape(shape = var_5230, x = hidden_states_263)[name = string("v_43")]; bool var_5235_transpose_x_1 = const()[name = string("op_5235_transpose_x_1"), val = bool(false)]; bool var_5235_transpose_y_1 = const()[name = string("op_5235_transpose_y_1"), val = bool(true)]; tensor var_5235_cast_fp16 = matmul(transpose_x = var_5235_transpose_x_1, transpose_y = var_5235_transpose_y_1, x = q_131, y = k_131)[name = string("op_5235_cast_fp16")]; fp16 var_5236_to_fp16 = const()[name = string("op_5236_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_127_cast_fp16 = mul(x = var_5235_cast_fp16, y = var_5236_to_fp16)[name = string("attn_weights_127_cast_fp16")]; tensor attn_weights_129_cast_fp16 = add(x = attn_weights_127_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_129_cast_fp16")]; tensor var_5240_cast_fp16 = softmax(axis = var_22, x = attn_weights_129_cast_fp16)[name = string("op_5240_cast_fp16")]; bool var_5244_transpose_x_0 = const()[name = string("op_5244_transpose_x_0"), val = bool(false)]; bool var_5244_transpose_y_0 = const()[name = string("op_5244_transpose_y_0"), val = bool(false)]; tensor var_5244_cast_fp16 = matmul(transpose_x = var_5244_transpose_x_0, transpose_y = var_5244_transpose_y_0, x = var_5240_cast_fp16, y = v_43)[name = string("op_5244_cast_fp16")]; tensor var_5246 = const()[name = string("op_5246"), val = tensor([0, 2, 1, 3])]; tensor var_5249 = const()[name = string("op_5249"), val = tensor([1, 256, 768])]; tensor var_5247 = transpose(perm = var_5246, x = var_5244_cast_fp16)[name = string("transpose_22")]; tensor attn_out_129 = reshape(shape = var_5249, x = var_5247)[name = string("attn_out_129")]; tensor var_5251 = const()[name = string("op_5251"), val = tensor([0, 2, 1])]; tensor squeeze_21_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306951680))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307541568))))[name = string("squeeze_21_quantized")]; string var_5260_pad_type_0 = const()[name = string("op_5260_pad_type_0"), val = string("valid")]; int32 var_5260_groups_0 = const()[name = string("op_5260_groups_0"), val = int32(1)]; tensor var_5260_strides_0 = const()[name = string("op_5260_strides_0"), val = tensor([1])]; tensor var_5260_pad_0 = const()[name = string("op_5260_pad_0"), val = tensor([0, 0])]; tensor var_5260_dilations_0 = const()[name = string("op_5260_dilations_0"), val = tensor([1])]; tensor var_5252 = transpose(perm = var_5251, x = attn_out_129)[name = string("transpose_21")]; tensor var_5260 = conv(dilations = var_5260_dilations_0, groups = var_5260_groups_0, pad = var_5260_pad_0, pad_type = var_5260_pad_type_0, strides = var_5260_strides_0, weight = squeeze_21_quantized, x = var_5252)[name = string("op_5260")]; tensor var_5261 = const()[name = string("op_5261"), val = tensor([0, 2, 1])]; fp16 const_302_promoted_to_fp16 = const()[name = string("const_302_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_345 = transpose(perm = var_5261, x = var_5260)[name = string("transpose_20")]; tensor var_5265_cast_fp16 = mul(x = x_345, y = const_302_promoted_to_fp16)[name = string("op_5265_cast_fp16")]; bool input_431_interleave_0 = const()[name = string("input_431_interleave_0"), val = bool(false)]; tensor input_431_cast_fp16 = concat(axis = var_22, interleave = input_431_interleave_0, values = (x_345, var_5265_cast_fp16))[name = string("input_431_cast_fp16")]; tensor normed_603_axes_0 = const()[name = string("normed_603_axes_0"), val = tensor([-1])]; tensor normed_603_cast_fp16 = layer_norm(axes = normed_603_axes_0, epsilon = var_8_to_fp16, x = input_431_cast_fp16)[name = string("normed_603_cast_fp16")]; tensor var_5270_split_sizes_0 = const()[name = string("op_5270_split_sizes_0"), val = tensor([768, 768])]; int32 var_5270_axis_0 = const()[name = string("op_5270_axis_0"), val = int32(-1)]; tensor var_5270_cast_fp16_0, tensor var_5270_cast_fp16_1 = split(axis = var_5270_axis_0, split_sizes = var_5270_split_sizes_0, x = normed_603_cast_fp16)[name = string("op_5270_cast_fp16")]; tensor var_5274_to_fp16 = const()[name = string("op_5274_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307543168)))]; tensor out_259_cast_fp16 = mul(x = var_5270_cast_fp16_0, y = var_5274_to_fp16)[name = string("out_259_cast_fp16")]; tensor x_347_cast_fp16 = add(x = x_337_cast_fp16, y = out_259_cast_fp16)[name = string("x_347_cast_fp16")]; fp16 const_304_promoted_to_fp16 = const()[name = string("const_304_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5281_cast_fp16 = mul(x = x_347_cast_fp16, y = const_304_promoted_to_fp16)[name = string("op_5281_cast_fp16")]; bool input_433_interleave_0 = const()[name = string("input_433_interleave_0"), val = bool(false)]; tensor input_433_cast_fp16 = concat(axis = var_22, interleave = input_433_interleave_0, values = (x_347_cast_fp16, var_5281_cast_fp16))[name = string("input_433_cast_fp16")]; tensor normed_607_axes_0 = const()[name = string("normed_607_axes_0"), val = tensor([-1])]; tensor normed_607_cast_fp16 = layer_norm(axes = normed_607_axes_0, epsilon = var_8_to_fp16, x = input_433_cast_fp16)[name = string("normed_607_cast_fp16")]; tensor var_5286_split_sizes_0 = const()[name = string("op_5286_split_sizes_0"), val = tensor([768, 768])]; int32 var_5286_axis_0 = const()[name = string("op_5286_axis_0"), val = int32(-1)]; tensor var_5286_cast_fp16_0, tensor var_5286_cast_fp16_1 = split(axis = var_5286_axis_0, split_sizes = var_5286_split_sizes_0, x = normed_607_cast_fp16)[name = string("op_5286_cast_fp16")]; tensor var_5290_to_fp16 = const()[name = string("op_5290_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307544768)))]; tensor out_261_cast_fp16 = mul(x = var_5286_cast_fp16_0, y = var_5290_to_fp16)[name = string("out_261_cast_fp16")]; tensor var_5297 = const()[name = string("op_5297"), val = tensor([0, 2, 1])]; tensor input_435_axes_0 = const()[name = string("input_435_axes_0"), val = tensor([2])]; tensor var_5298 = transpose(perm = var_5297, x = out_261_cast_fp16)[name = string("transpose_19")]; tensor input_435 = expand_dims(axes = input_435_axes_0, x = var_5298)[name = string("input_435")]; string gate_85_pad_type_0 = const()[name = string("gate_85_pad_type_0"), val = string("valid")]; tensor gate_85_strides_0 = const()[name = string("gate_85_strides_0"), val = tensor([1, 1])]; tensor gate_85_pad_0 = const()[name = string("gate_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_85_dilations_0 = const()[name = string("gate_85_dilations_0"), val = tensor([1, 1])]; int32 gate_85_groups_0 = const()[name = string("gate_85_groups_0"), val = int32(1)]; tensor gate_85 = conv(dilations = gate_85_dilations_0, groups = gate_85_groups_0, pad = gate_85_pad_0, pad_type = gate_85_pad_type_0, strides = gate_85_strides_0, weight = encoder_layers_21_mlp_gate_proj_weight_quantized, x = input_435)[name = string("gate_85")]; string up_43_pad_type_0 = const()[name = string("up_43_pad_type_0"), val = string("valid")]; tensor up_43_strides_0 = const()[name = string("up_43_strides_0"), val = tensor([1, 1])]; tensor up_43_pad_0 = const()[name = string("up_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_43_dilations_0 = const()[name = string("up_43_dilations_0"), val = tensor([1, 1])]; int32 up_43_groups_0 = const()[name = string("up_43_groups_0"), val = int32(1)]; tensor up_43 = conv(dilations = up_43_dilations_0, groups = up_43_groups_0, pad = up_43_pad_0, pad_type = up_43_pad_type_0, strides = up_43_strides_0, weight = encoder_layers_21_mlp_up_proj_weight_quantized, x = input_435)[name = string("up_43")]; string gate_87_mode_0 = const()[name = string("gate_87_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_87 = gelu(mode = gate_87_mode_0, x = gate_85)[name = string("gate_87")]; tensor input_437 = mul(x = gate_87, y = up_43)[name = string("input_437")]; string var_5319_pad_type_0 = const()[name = string("op_5319_pad_type_0"), val = string("valid")]; tensor var_5319_strides_0 = const()[name = string("op_5319_strides_0"), val = tensor([1, 1])]; tensor var_5319_pad_0 = const()[name = string("op_5319_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5319_dilations_0 = const()[name = string("op_5319_dilations_0"), val = tensor([1, 1])]; int32 var_5319_groups_0 = const()[name = string("op_5319_groups_0"), val = int32(1)]; tensor var_5319 = conv(dilations = var_5319_dilations_0, groups = var_5319_groups_0, pad = var_5319_pad_0, pad_type = var_5319_pad_type_0, strides = var_5319_strides_0, weight = encoder_layers_21_mlp_down_proj_weight_quantized, x = input_437)[name = string("op_5319")]; tensor var_5320_axes_0 = const()[name = string("op_5320_axes_0"), val = tensor([2])]; tensor var_5320 = squeeze(axes = var_5320_axes_0, x = var_5319)[name = string("op_5320")]; tensor var_5321 = const()[name = string("op_5321"), val = tensor([0, 2, 1])]; fp16 const_306_promoted_to_fp16 = const()[name = string("const_306_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_351 = transpose(perm = var_5321, x = var_5320)[name = string("transpose_18")]; tensor var_5325_cast_fp16 = mul(x = x_351, y = const_306_promoted_to_fp16)[name = string("op_5325_cast_fp16")]; bool input_439_interleave_0 = const()[name = string("input_439_interleave_0"), val = bool(false)]; tensor input_439_cast_fp16 = concat(axis = var_22, interleave = input_439_interleave_0, values = (x_351, var_5325_cast_fp16))[name = string("input_439_cast_fp16")]; tensor normed_613_axes_0 = const()[name = string("normed_613_axes_0"), val = tensor([-1])]; tensor normed_613_cast_fp16 = layer_norm(axes = normed_613_axes_0, epsilon = var_8_to_fp16, x = input_439_cast_fp16)[name = string("normed_613_cast_fp16")]; tensor var_5330_split_sizes_0 = const()[name = string("op_5330_split_sizes_0"), val = tensor([768, 768])]; int32 var_5330_axis_0 = const()[name = string("op_5330_axis_0"), val = int32(-1)]; tensor var_5330_cast_fp16_0, tensor var_5330_cast_fp16_1 = split(axis = var_5330_axis_0, split_sizes = var_5330_split_sizes_0, x = normed_613_cast_fp16)[name = string("op_5330_cast_fp16")]; tensor var_5334_to_fp16 = const()[name = string("op_5334_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307546368)))]; tensor out_263_cast_fp16 = mul(x = var_5330_cast_fp16_0, y = var_5334_to_fp16)[name = string("out_263_cast_fp16")]; tensor x_353_cast_fp16 = add(x = x_347_cast_fp16, y = out_263_cast_fp16)[name = string("x_353_cast_fp16")]; fp16 const_308_promoted_to_fp16 = const()[name = string("const_308_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5363_cast_fp16 = mul(x = x_353_cast_fp16, y = const_308_promoted_to_fp16)[name = string("op_5363_cast_fp16")]; bool input_441_interleave_0 = const()[name = string("input_441_interleave_0"), val = bool(false)]; tensor input_441_cast_fp16 = concat(axis = var_22, interleave = input_441_interleave_0, values = (x_353_cast_fp16, var_5363_cast_fp16))[name = string("input_441_cast_fp16")]; tensor normed_617_axes_0 = const()[name = string("normed_617_axes_0"), val = tensor([-1])]; tensor normed_617_cast_fp16 = layer_norm(axes = normed_617_axes_0, epsilon = var_8_to_fp16, x = input_441_cast_fp16)[name = string("normed_617_cast_fp16")]; tensor var_5368_split_sizes_0 = const()[name = string("op_5368_split_sizes_0"), val = tensor([768, 768])]; int32 var_5368_axis_0 = const()[name = string("op_5368_axis_0"), val = int32(-1)]; tensor var_5368_cast_fp16_0, tensor var_5368_cast_fp16_1 = split(axis = var_5368_axis_0, split_sizes = var_5368_split_sizes_0, x = normed_617_cast_fp16)[name = string("op_5368_cast_fp16")]; tensor var_5372_to_fp16 = const()[name = string("op_5372_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307547968)))]; tensor out_265_cast_fp16 = mul(x = var_5368_cast_fp16_0, y = var_5372_to_fp16)[name = string("out_265_cast_fp16")]; tensor var_5378 = const()[name = string("op_5378"), val = tensor([0, 2, 1])]; tensor var_5380_axes_0 = const()[name = string("op_5380_axes_0"), val = tensor([2])]; tensor var_5379_cast_fp16 = transpose(perm = var_5378, x = out_265_cast_fp16)[name = string("transpose_17")]; tensor var_5380_cast_fp16 = expand_dims(axes = var_5380_axes_0, x = var_5379_cast_fp16)[name = string("op_5380_cast_fp16")]; string var_5387_pad_type_0 = const()[name = string("op_5387_pad_type_0"), val = string("valid")]; tensor var_5387_strides_0 = const()[name = string("op_5387_strides_0"), val = tensor([1, 1])]; tensor var_5387_pad_0 = const()[name = string("op_5387_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5387_dilations_0 = const()[name = string("op_5387_dilations_0"), val = tensor([1, 1])]; int32 var_5387_groups_0 = const()[name = string("op_5387_groups_0"), val = int32(1)]; tensor var_5387 = conv(dilations = var_5387_dilations_0, groups = var_5387_groups_0, pad = var_5387_pad_0, pad_type = var_5387_pad_type_0, strides = var_5387_strides_0, weight = encoder_layers_22_self_attn_q_proj_weight_quantized, x = var_5380_cast_fp16)[name = string("op_5387")]; tensor var_5388 = const()[name = string("op_5388"), val = tensor([1, 3, 256, 256])]; tensor var_5389 = reshape(shape = var_5388, x = var_5387)[name = string("op_5389")]; tensor var_5390 = const()[name = string("op_5390"), val = tensor([0, 1, 3, 2])]; string var_5397_pad_type_0 = const()[name = string("op_5397_pad_type_0"), val = string("valid")]; tensor var_5397_strides_0 = const()[name = string("op_5397_strides_0"), val = tensor([1, 1])]; tensor var_5397_pad_0 = const()[name = string("op_5397_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5397_dilations_0 = const()[name = string("op_5397_dilations_0"), val = tensor([1, 1])]; int32 var_5397_groups_0 = const()[name = string("op_5397_groups_0"), val = int32(1)]; tensor var_5397 = conv(dilations = var_5397_dilations_0, groups = var_5397_groups_0, pad = var_5397_pad_0, pad_type = var_5397_pad_type_0, strides = var_5397_strides_0, weight = encoder_layers_22_self_attn_k_proj_weight_quantized, x = var_5380_cast_fp16)[name = string("op_5397")]; tensor var_5398 = const()[name = string("op_5398"), val = tensor([1, 1, 256, 256])]; tensor var_5399 = reshape(shape = var_5398, x = var_5397)[name = string("op_5399")]; tensor var_5400 = const()[name = string("op_5400"), val = tensor([0, 1, 3, 2])]; string var_5407_pad_type_0 = const()[name = string("op_5407_pad_type_0"), val = string("valid")]; tensor var_5407_strides_0 = const()[name = string("op_5407_strides_0"), val = tensor([1, 1])]; tensor var_5407_pad_0 = const()[name = string("op_5407_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5407_dilations_0 = const()[name = string("op_5407_dilations_0"), val = tensor([1, 1])]; int32 var_5407_groups_0 = const()[name = string("op_5407_groups_0"), val = int32(1)]; tensor var_5407 = conv(dilations = var_5407_dilations_0, groups = var_5407_groups_0, pad = var_5407_pad_0, pad_type = var_5407_pad_type_0, strides = var_5407_strides_0, weight = encoder_layers_22_self_attn_v_proj_weight_quantized, x = var_5380_cast_fp16)[name = string("op_5407")]; tensor var_5408 = const()[name = string("op_5408"), val = tensor([1, 1, 256, 256])]; tensor var_5409 = reshape(shape = var_5408, x = var_5407)[name = string("op_5409")]; tensor var_5410 = const()[name = string("op_5410"), val = tensor([0, 1, 3, 2])]; fp16 const_310_promoted_to_fp16 = const()[name = string("const_310_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_133 = transpose(perm = var_5390, x = var_5389)[name = string("transpose_16")]; tensor var_5416_cast_fp16 = mul(x = q_133, y = const_310_promoted_to_fp16)[name = string("op_5416_cast_fp16")]; bool input_445_interleave_0 = const()[name = string("input_445_interleave_0"), val = bool(false)]; tensor input_445_cast_fp16 = concat(axis = var_22, interleave = input_445_interleave_0, values = (q_133, var_5416_cast_fp16))[name = string("input_445_cast_fp16")]; tensor normed_623_axes_0 = const()[name = string("normed_623_axes_0"), val = tensor([-1])]; tensor normed_623_cast_fp16 = layer_norm(axes = normed_623_axes_0, epsilon = var_8_to_fp16, x = input_445_cast_fp16)[name = string("normed_623_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([256, 256])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(-1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = normed_623_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor var_5425_to_fp16 = const()[name = string("op_5425_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307549568)))]; tensor out_267_cast_fp16 = mul(x = var_5421_cast_fp16_0, y = var_5425_to_fp16)[name = string("out_267_cast_fp16")]; fp16 const_312_promoted_to_fp16 = const()[name = string("const_312_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_133 = transpose(perm = var_5400, x = var_5399)[name = string("transpose_15")]; tensor var_5432_cast_fp16 = mul(x = k_133, y = const_312_promoted_to_fp16)[name = string("op_5432_cast_fp16")]; bool input_447_interleave_0 = const()[name = string("input_447_interleave_0"), val = bool(false)]; tensor input_447_cast_fp16 = concat(axis = var_22, interleave = input_447_interleave_0, values = (k_133, var_5432_cast_fp16))[name = string("input_447_cast_fp16")]; tensor normed_627_axes_0 = const()[name = string("normed_627_axes_0"), val = tensor([-1])]; tensor normed_627_cast_fp16 = layer_norm(axes = normed_627_axes_0, epsilon = var_8_to_fp16, x = input_447_cast_fp16)[name = string("normed_627_cast_fp16")]; tensor var_5437_split_sizes_0 = const()[name = string("op_5437_split_sizes_0"), val = tensor([256, 256])]; int32 var_5437_axis_0 = const()[name = string("op_5437_axis_0"), val = int32(-1)]; tensor var_5437_cast_fp16_0, tensor var_5437_cast_fp16_1 = split(axis = var_5437_axis_0, split_sizes = var_5437_split_sizes_0, x = normed_627_cast_fp16)[name = string("op_5437_cast_fp16")]; tensor var_5441_to_fp16 = const()[name = string("op_5441_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307550144)))]; tensor out_269_cast_fp16 = mul(x = var_5437_cast_fp16_0, y = var_5441_to_fp16)[name = string("out_269_cast_fp16")]; tensor var_5444 = mul(x = out_267_cast_fp16, y = cos_1_quantized)[name = string("op_5444")]; tensor var_5445_split_sizes_0 = const()[name = string("op_5445_split_sizes_0"), val = tensor([128, 128])]; int32 var_5445_axis_0 = const()[name = string("op_5445_axis_0"), val = int32(-1)]; tensor var_5445_0, tensor var_5445_1 = split(axis = var_5445_axis_0, split_sizes = var_5445_split_sizes_0, x = out_267_cast_fp16)[name = string("op_5445")]; fp16 const_314_promoted = const()[name = string("const_314_promoted"), val = fp16(-0x1p+0)]; tensor var_5447 = mul(x = var_5445_1, y = const_314_promoted)[name = string("op_5447")]; bool var_5449_interleave_0 = const()[name = string("op_5449_interleave_0"), val = bool(false)]; tensor var_5449 = concat(axis = var_22, interleave = var_5449_interleave_0, values = (var_5447, var_5445_0))[name = string("op_5449")]; tensor var_5450 = mul(x = var_5449, y = sin_1_quantized)[name = string("op_5450")]; tensor q_137 = add(x = var_5444, y = var_5450)[name = string("q_137")]; tensor var_5452 = mul(x = out_269_cast_fp16, y = cos_1_quantized)[name = string("op_5452")]; tensor var_5453_split_sizes_0 = const()[name = string("op_5453_split_sizes_0"), val = tensor([128, 128])]; int32 var_5453_axis_0 = const()[name = string("op_5453_axis_0"), val = int32(-1)]; tensor var_5453_0, tensor var_5453_1 = split(axis = var_5453_axis_0, split_sizes = var_5453_split_sizes_0, x = out_269_cast_fp16)[name = string("op_5453")]; fp16 const_315_promoted = const()[name = string("const_315_promoted"), val = fp16(-0x1p+0)]; tensor var_5455 = mul(x = var_5453_1, y = const_315_promoted)[name = string("op_5455")]; bool var_5457_interleave_0 = const()[name = string("op_5457_interleave_0"), val = bool(false)]; tensor var_5457 = concat(axis = var_22, interleave = var_5457_interleave_0, values = (var_5455, var_5453_0))[name = string("op_5457")]; tensor var_5458 = mul(x = var_5457, y = sin_1_quantized)[name = string("op_5458")]; tensor hidden_states_265 = add(x = var_5452, y = var_5458)[name = string("hidden_states_265")]; tensor hidden_states_267_axes_0 = const()[name = string("hidden_states_267_axes_0"), val = tensor([2])]; tensor hidden_states_267 = expand_dims(axes = hidden_states_267_axes_0, x = hidden_states_265)[name = string("hidden_states_267")]; tensor var_5461 = const()[name = string("op_5461"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_269 = tile(reps = var_5461, x = hidden_states_267)[name = string("hidden_states_269")]; tensor var_5463 = const()[name = string("op_5463"), val = tensor([1, 3, 256, 256])]; tensor k_137 = reshape(shape = var_5463, x = hidden_states_269)[name = string("k_137")]; tensor hidden_states_273_axes_0 = const()[name = string("hidden_states_273_axes_0"), val = tensor([2])]; tensor hidden_states_271 = transpose(perm = var_5410, x = var_5409)[name = string("transpose_14")]; tensor hidden_states_273 = expand_dims(axes = hidden_states_273_axes_0, x = hidden_states_271)[name = string("hidden_states_273")]; tensor var_5466 = const()[name = string("op_5466"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_275 = tile(reps = var_5466, x = hidden_states_273)[name = string("hidden_states_275")]; tensor var_5468 = const()[name = string("op_5468"), val = tensor([1, 3, 256, 256])]; tensor v_45 = reshape(shape = var_5468, x = hidden_states_275)[name = string("v_45")]; bool var_5473_transpose_x_1 = const()[name = string("op_5473_transpose_x_1"), val = bool(false)]; bool var_5473_transpose_y_1 = const()[name = string("op_5473_transpose_y_1"), val = bool(true)]; tensor var_5473_cast_fp16 = matmul(transpose_x = var_5473_transpose_x_1, transpose_y = var_5473_transpose_y_1, x = q_137, y = k_137)[name = string("op_5473_cast_fp16")]; fp16 var_5474_to_fp16 = const()[name = string("op_5474_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_133_cast_fp16 = mul(x = var_5473_cast_fp16, y = var_5474_to_fp16)[name = string("attn_weights_133_cast_fp16")]; tensor attn_weights_135_cast_fp16 = add(x = attn_weights_133_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; tensor var_5478_cast_fp16 = softmax(axis = var_22, x = attn_weights_135_cast_fp16)[name = string("op_5478_cast_fp16")]; bool var_5482_transpose_x_0 = const()[name = string("op_5482_transpose_x_0"), val = bool(false)]; bool var_5482_transpose_y_0 = const()[name = string("op_5482_transpose_y_0"), val = bool(false)]; tensor var_5482_cast_fp16 = matmul(transpose_x = var_5482_transpose_x_0, transpose_y = var_5482_transpose_y_0, x = var_5478_cast_fp16, y = v_45)[name = string("op_5482_cast_fp16")]; tensor var_5484 = const()[name = string("op_5484"), val = tensor([0, 2, 1, 3])]; tensor var_5487 = const()[name = string("op_5487"), val = tensor([1, 256, 768])]; tensor var_5485 = transpose(perm = var_5484, x = var_5482_cast_fp16)[name = string("transpose_13")]; tensor attn_out_135 = reshape(shape = var_5487, x = var_5485)[name = string("attn_out_135")]; tensor var_5489 = const()[name = string("op_5489"), val = tensor([0, 2, 1])]; tensor squeeze_22_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307550720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308140608))))[name = string("squeeze_22_quantized")]; string var_5498_pad_type_0 = const()[name = string("op_5498_pad_type_0"), val = string("valid")]; int32 var_5498_groups_0 = const()[name = string("op_5498_groups_0"), val = int32(1)]; tensor var_5498_strides_0 = const()[name = string("op_5498_strides_0"), val = tensor([1])]; tensor var_5498_pad_0 = const()[name = string("op_5498_pad_0"), val = tensor([0, 0])]; tensor var_5498_dilations_0 = const()[name = string("op_5498_dilations_0"), val = tensor([1])]; tensor var_5490 = transpose(perm = var_5489, x = attn_out_135)[name = string("transpose_12")]; tensor var_5498 = conv(dilations = var_5498_dilations_0, groups = var_5498_groups_0, pad = var_5498_pad_0, pad_type = var_5498_pad_type_0, strides = var_5498_strides_0, weight = squeeze_22_quantized, x = var_5490)[name = string("op_5498")]; tensor var_5499 = const()[name = string("op_5499"), val = tensor([0, 2, 1])]; fp16 const_316_promoted_to_fp16 = const()[name = string("const_316_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_361 = transpose(perm = var_5499, x = var_5498)[name = string("transpose_11")]; tensor var_5503_cast_fp16 = mul(x = x_361, y = const_316_promoted_to_fp16)[name = string("op_5503_cast_fp16")]; bool input_451_interleave_0 = const()[name = string("input_451_interleave_0"), val = bool(false)]; tensor input_451_cast_fp16 = concat(axis = var_22, interleave = input_451_interleave_0, values = (x_361, var_5503_cast_fp16))[name = string("input_451_cast_fp16")]; tensor normed_631_axes_0 = const()[name = string("normed_631_axes_0"), val = tensor([-1])]; tensor normed_631_cast_fp16 = layer_norm(axes = normed_631_axes_0, epsilon = var_8_to_fp16, x = input_451_cast_fp16)[name = string("normed_631_cast_fp16")]; tensor var_5508_split_sizes_0 = const()[name = string("op_5508_split_sizes_0"), val = tensor([768, 768])]; int32 var_5508_axis_0 = const()[name = string("op_5508_axis_0"), val = int32(-1)]; tensor var_5508_cast_fp16_0, tensor var_5508_cast_fp16_1 = split(axis = var_5508_axis_0, split_sizes = var_5508_split_sizes_0, x = normed_631_cast_fp16)[name = string("op_5508_cast_fp16")]; tensor var_5512_to_fp16 = const()[name = string("op_5512_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308142208)))]; tensor out_271_cast_fp16 = mul(x = var_5508_cast_fp16_0, y = var_5512_to_fp16)[name = string("out_271_cast_fp16")]; tensor x_363_cast_fp16 = add(x = x_353_cast_fp16, y = out_271_cast_fp16)[name = string("x_363_cast_fp16")]; fp16 const_318_promoted_to_fp16 = const()[name = string("const_318_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5519_cast_fp16 = mul(x = x_363_cast_fp16, y = const_318_promoted_to_fp16)[name = string("op_5519_cast_fp16")]; bool input_453_interleave_0 = const()[name = string("input_453_interleave_0"), val = bool(false)]; tensor input_453_cast_fp16 = concat(axis = var_22, interleave = input_453_interleave_0, values = (x_363_cast_fp16, var_5519_cast_fp16))[name = string("input_453_cast_fp16")]; tensor normed_635_axes_0 = const()[name = string("normed_635_axes_0"), val = tensor([-1])]; tensor normed_635_cast_fp16 = layer_norm(axes = normed_635_axes_0, epsilon = var_8_to_fp16, x = input_453_cast_fp16)[name = string("normed_635_cast_fp16")]; tensor var_5524_split_sizes_0 = const()[name = string("op_5524_split_sizes_0"), val = tensor([768, 768])]; int32 var_5524_axis_0 = const()[name = string("op_5524_axis_0"), val = int32(-1)]; tensor var_5524_cast_fp16_0, tensor var_5524_cast_fp16_1 = split(axis = var_5524_axis_0, split_sizes = var_5524_split_sizes_0, x = normed_635_cast_fp16)[name = string("op_5524_cast_fp16")]; tensor var_5528_to_fp16 = const()[name = string("op_5528_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308143808)))]; tensor out_273_cast_fp16 = mul(x = var_5524_cast_fp16_0, y = var_5528_to_fp16)[name = string("out_273_cast_fp16")]; tensor var_5535 = const()[name = string("op_5535"), val = tensor([0, 2, 1])]; tensor input_455_axes_0 = const()[name = string("input_455_axes_0"), val = tensor([2])]; tensor var_5536 = transpose(perm = var_5535, x = out_273_cast_fp16)[name = string("transpose_10")]; tensor input_455 = expand_dims(axes = input_455_axes_0, x = var_5536)[name = string("input_455")]; string gate_89_pad_type_0 = const()[name = string("gate_89_pad_type_0"), val = string("valid")]; tensor gate_89_strides_0 = const()[name = string("gate_89_strides_0"), val = tensor([1, 1])]; tensor gate_89_pad_0 = const()[name = string("gate_89_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_89_dilations_0 = const()[name = string("gate_89_dilations_0"), val = tensor([1, 1])]; int32 gate_89_groups_0 = const()[name = string("gate_89_groups_0"), val = int32(1)]; tensor gate_89 = conv(dilations = gate_89_dilations_0, groups = gate_89_groups_0, pad = gate_89_pad_0, pad_type = gate_89_pad_type_0, strides = gate_89_strides_0, weight = encoder_layers_22_mlp_gate_proj_weight_quantized, x = input_455)[name = string("gate_89")]; string up_45_pad_type_0 = const()[name = string("up_45_pad_type_0"), val = string("valid")]; tensor up_45_strides_0 = const()[name = string("up_45_strides_0"), val = tensor([1, 1])]; tensor up_45_pad_0 = const()[name = string("up_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_45_dilations_0 = const()[name = string("up_45_dilations_0"), val = tensor([1, 1])]; int32 up_45_groups_0 = const()[name = string("up_45_groups_0"), val = int32(1)]; tensor up_45 = conv(dilations = up_45_dilations_0, groups = up_45_groups_0, pad = up_45_pad_0, pad_type = up_45_pad_type_0, strides = up_45_strides_0, weight = encoder_layers_22_mlp_up_proj_weight_quantized, x = input_455)[name = string("up_45")]; string gate_91_mode_0 = const()[name = string("gate_91_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_91 = gelu(mode = gate_91_mode_0, x = gate_89)[name = string("gate_91")]; tensor input_457 = mul(x = gate_91, y = up_45)[name = string("input_457")]; string var_5557_pad_type_0 = const()[name = string("op_5557_pad_type_0"), val = string("valid")]; tensor var_5557_strides_0 = const()[name = string("op_5557_strides_0"), val = tensor([1, 1])]; tensor var_5557_pad_0 = const()[name = string("op_5557_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5557_dilations_0 = const()[name = string("op_5557_dilations_0"), val = tensor([1, 1])]; int32 var_5557_groups_0 = const()[name = string("op_5557_groups_0"), val = int32(1)]; tensor var_5557 = conv(dilations = var_5557_dilations_0, groups = var_5557_groups_0, pad = var_5557_pad_0, pad_type = var_5557_pad_type_0, strides = var_5557_strides_0, weight = encoder_layers_22_mlp_down_proj_weight_quantized, x = input_457)[name = string("op_5557")]; tensor var_5558_axes_0 = const()[name = string("op_5558_axes_0"), val = tensor([2])]; tensor var_5558 = squeeze(axes = var_5558_axes_0, x = var_5557)[name = string("op_5558")]; tensor var_5559 = const()[name = string("op_5559"), val = tensor([0, 2, 1])]; fp16 const_320_promoted_to_fp16 = const()[name = string("const_320_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_367 = transpose(perm = var_5559, x = var_5558)[name = string("transpose_9")]; tensor var_5563_cast_fp16 = mul(x = x_367, y = const_320_promoted_to_fp16)[name = string("op_5563_cast_fp16")]; bool input_459_interleave_0 = const()[name = string("input_459_interleave_0"), val = bool(false)]; tensor input_459_cast_fp16 = concat(axis = var_22, interleave = input_459_interleave_0, values = (x_367, var_5563_cast_fp16))[name = string("input_459_cast_fp16")]; tensor normed_641_axes_0 = const()[name = string("normed_641_axes_0"), val = tensor([-1])]; tensor normed_641_cast_fp16 = layer_norm(axes = normed_641_axes_0, epsilon = var_8_to_fp16, x = input_459_cast_fp16)[name = string("normed_641_cast_fp16")]; tensor var_5568_split_sizes_0 = const()[name = string("op_5568_split_sizes_0"), val = tensor([768, 768])]; int32 var_5568_axis_0 = const()[name = string("op_5568_axis_0"), val = int32(-1)]; tensor var_5568_cast_fp16_0, tensor var_5568_cast_fp16_1 = split(axis = var_5568_axis_0, split_sizes = var_5568_split_sizes_0, x = normed_641_cast_fp16)[name = string("op_5568_cast_fp16")]; tensor var_5572_to_fp16 = const()[name = string("op_5572_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308145408)))]; tensor out_275_cast_fp16 = mul(x = var_5568_cast_fp16_0, y = var_5572_to_fp16)[name = string("out_275_cast_fp16")]; tensor x_369_cast_fp16 = add(x = x_363_cast_fp16, y = out_275_cast_fp16)[name = string("x_369_cast_fp16")]; fp16 const_322_promoted_to_fp16 = const()[name = string("const_322_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5601_cast_fp16 = mul(x = x_369_cast_fp16, y = const_322_promoted_to_fp16)[name = string("op_5601_cast_fp16")]; bool input_461_interleave_0 = const()[name = string("input_461_interleave_0"), val = bool(false)]; tensor input_461_cast_fp16 = concat(axis = var_22, interleave = input_461_interleave_0, values = (x_369_cast_fp16, var_5601_cast_fp16))[name = string("input_461_cast_fp16")]; tensor normed_645_axes_0 = const()[name = string("normed_645_axes_0"), val = tensor([-1])]; tensor normed_645_cast_fp16 = layer_norm(axes = normed_645_axes_0, epsilon = var_8_to_fp16, x = input_461_cast_fp16)[name = string("normed_645_cast_fp16")]; tensor var_5606_split_sizes_0 = const()[name = string("op_5606_split_sizes_0"), val = tensor([768, 768])]; int32 var_5606_axis_0 = const()[name = string("op_5606_axis_0"), val = int32(-1)]; tensor var_5606_cast_fp16_0, tensor var_5606_cast_fp16_1 = split(axis = var_5606_axis_0, split_sizes = var_5606_split_sizes_0, x = normed_645_cast_fp16)[name = string("op_5606_cast_fp16")]; tensor var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308147008)))]; tensor out_277_cast_fp16 = mul(x = var_5606_cast_fp16_0, y = var_5610_to_fp16)[name = string("out_277_cast_fp16")]; tensor var_5616 = const()[name = string("op_5616"), val = tensor([0, 2, 1])]; tensor var_5618_axes_0 = const()[name = string("op_5618_axes_0"), val = tensor([2])]; tensor var_5617_cast_fp16 = transpose(perm = var_5616, x = out_277_cast_fp16)[name = string("transpose_8")]; tensor var_5618_cast_fp16 = expand_dims(axes = var_5618_axes_0, x = var_5617_cast_fp16)[name = string("op_5618_cast_fp16")]; string var_5625_pad_type_0 = const()[name = string("op_5625_pad_type_0"), val = string("valid")]; tensor var_5625_strides_0 = const()[name = string("op_5625_strides_0"), val = tensor([1, 1])]; tensor var_5625_pad_0 = const()[name = string("op_5625_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5625_dilations_0 = const()[name = string("op_5625_dilations_0"), val = tensor([1, 1])]; int32 var_5625_groups_0 = const()[name = string("op_5625_groups_0"), val = int32(1)]; tensor var_5625 = conv(dilations = var_5625_dilations_0, groups = var_5625_groups_0, pad = var_5625_pad_0, pad_type = var_5625_pad_type_0, strides = var_5625_strides_0, weight = encoder_layers_23_self_attn_q_proj_weight_quantized, x = var_5618_cast_fp16)[name = string("op_5625")]; tensor var_5626 = const()[name = string("op_5626"), val = tensor([1, 3, 256, 256])]; tensor var_5627 = reshape(shape = var_5626, x = var_5625)[name = string("op_5627")]; tensor var_5628 = const()[name = string("op_5628"), val = tensor([0, 1, 3, 2])]; string var_5635_pad_type_0 = const()[name = string("op_5635_pad_type_0"), val = string("valid")]; tensor var_5635_strides_0 = const()[name = string("op_5635_strides_0"), val = tensor([1, 1])]; tensor var_5635_pad_0 = const()[name = string("op_5635_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5635_dilations_0 = const()[name = string("op_5635_dilations_0"), val = tensor([1, 1])]; int32 var_5635_groups_0 = const()[name = string("op_5635_groups_0"), val = int32(1)]; tensor var_5635 = conv(dilations = var_5635_dilations_0, groups = var_5635_groups_0, pad = var_5635_pad_0, pad_type = var_5635_pad_type_0, strides = var_5635_strides_0, weight = encoder_layers_23_self_attn_k_proj_weight_quantized, x = var_5618_cast_fp16)[name = string("op_5635")]; tensor var_5636 = const()[name = string("op_5636"), val = tensor([1, 1, 256, 256])]; tensor var_5637 = reshape(shape = var_5636, x = var_5635)[name = string("op_5637")]; tensor var_5638 = const()[name = string("op_5638"), val = tensor([0, 1, 3, 2])]; string var_5645_pad_type_0 = const()[name = string("op_5645_pad_type_0"), val = string("valid")]; tensor var_5645_strides_0 = const()[name = string("op_5645_strides_0"), val = tensor([1, 1])]; tensor var_5645_pad_0 = const()[name = string("op_5645_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5645_dilations_0 = const()[name = string("op_5645_dilations_0"), val = tensor([1, 1])]; int32 var_5645_groups_0 = const()[name = string("op_5645_groups_0"), val = int32(1)]; tensor var_5645 = conv(dilations = var_5645_dilations_0, groups = var_5645_groups_0, pad = var_5645_pad_0, pad_type = var_5645_pad_type_0, strides = var_5645_strides_0, weight = encoder_layers_23_self_attn_v_proj_weight_quantized, x = var_5618_cast_fp16)[name = string("op_5645")]; tensor var_5646 = const()[name = string("op_5646"), val = tensor([1, 1, 256, 256])]; tensor var_5647 = reshape(shape = var_5646, x = var_5645)[name = string("op_5647")]; tensor var_5648 = const()[name = string("op_5648"), val = tensor([0, 1, 3, 2])]; fp16 const_324_promoted_to_fp16 = const()[name = string("const_324_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor q_139 = transpose(perm = var_5628, x = var_5627)[name = string("transpose_7")]; tensor var_5654_cast_fp16 = mul(x = q_139, y = const_324_promoted_to_fp16)[name = string("op_5654_cast_fp16")]; bool input_465_interleave_0 = const()[name = string("input_465_interleave_0"), val = bool(false)]; tensor input_465_cast_fp16 = concat(axis = var_22, interleave = input_465_interleave_0, values = (q_139, var_5654_cast_fp16))[name = string("input_465_cast_fp16")]; tensor normed_651_axes_0 = const()[name = string("normed_651_axes_0"), val = tensor([-1])]; tensor normed_651_cast_fp16 = layer_norm(axes = normed_651_axes_0, epsilon = var_8_to_fp16, x = input_465_cast_fp16)[name = string("normed_651_cast_fp16")]; tensor var_5659_split_sizes_0 = const()[name = string("op_5659_split_sizes_0"), val = tensor([256, 256])]; int32 var_5659_axis_0 = const()[name = string("op_5659_axis_0"), val = int32(-1)]; tensor var_5659_cast_fp16_0, tensor var_5659_cast_fp16_1 = split(axis = var_5659_axis_0, split_sizes = var_5659_split_sizes_0, x = normed_651_cast_fp16)[name = string("op_5659_cast_fp16")]; tensor var_5663_to_fp16 = const()[name = string("op_5663_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308148608)))]; tensor out_279_cast_fp16 = mul(x = var_5659_cast_fp16_0, y = var_5663_to_fp16)[name = string("out_279_cast_fp16")]; fp16 const_326_promoted_to_fp16 = const()[name = string("const_326_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor k_139 = transpose(perm = var_5638, x = var_5637)[name = string("transpose_6")]; tensor var_5670_cast_fp16 = mul(x = k_139, y = const_326_promoted_to_fp16)[name = string("op_5670_cast_fp16")]; bool input_467_interleave_0 = const()[name = string("input_467_interleave_0"), val = bool(false)]; tensor input_467_cast_fp16 = concat(axis = var_22, interleave = input_467_interleave_0, values = (k_139, var_5670_cast_fp16))[name = string("input_467_cast_fp16")]; tensor normed_655_axes_0 = const()[name = string("normed_655_axes_0"), val = tensor([-1])]; tensor normed_655_cast_fp16 = layer_norm(axes = normed_655_axes_0, epsilon = var_8_to_fp16, x = input_467_cast_fp16)[name = string("normed_655_cast_fp16")]; tensor var_5675_split_sizes_0 = const()[name = string("op_5675_split_sizes_0"), val = tensor([256, 256])]; int32 var_5675_axis_0 = const()[name = string("op_5675_axis_0"), val = int32(-1)]; tensor var_5675_cast_fp16_0, tensor var_5675_cast_fp16_1 = split(axis = var_5675_axis_0, split_sizes = var_5675_split_sizes_0, x = normed_655_cast_fp16)[name = string("op_5675_cast_fp16")]; tensor var_5679_to_fp16 = const()[name = string("op_5679_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308149184)))]; tensor out_281_cast_fp16 = mul(x = var_5675_cast_fp16_0, y = var_5679_to_fp16)[name = string("out_281_cast_fp16")]; tensor var_5682 = mul(x = out_279_cast_fp16, y = cos_quantized)[name = string("op_5682")]; tensor var_5683_split_sizes_0 = const()[name = string("op_5683_split_sizes_0"), val = tensor([128, 128])]; int32 var_5683_axis_0 = const()[name = string("op_5683_axis_0"), val = int32(-1)]; tensor var_5683_0, tensor var_5683_1 = split(axis = var_5683_axis_0, split_sizes = var_5683_split_sizes_0, x = out_279_cast_fp16)[name = string("op_5683")]; fp16 const_328_promoted = const()[name = string("const_328_promoted"), val = fp16(-0x1p+0)]; tensor var_5685 = mul(x = var_5683_1, y = const_328_promoted)[name = string("op_5685")]; bool var_5687_interleave_0 = const()[name = string("op_5687_interleave_0"), val = bool(false)]; tensor var_5687 = concat(axis = var_22, interleave = var_5687_interleave_0, values = (var_5685, var_5683_0))[name = string("op_5687")]; tensor var_5688 = mul(x = var_5687, y = sin_quantized)[name = string("op_5688")]; tensor q = add(x = var_5682, y = var_5688)[name = string("q")]; tensor var_5690 = mul(x = out_281_cast_fp16, y = cos_quantized)[name = string("op_5690")]; tensor var_5691_split_sizes_0 = const()[name = string("op_5691_split_sizes_0"), val = tensor([128, 128])]; int32 var_5691_axis_0 = const()[name = string("op_5691_axis_0"), val = int32(-1)]; tensor var_5691_0, tensor var_5691_1 = split(axis = var_5691_axis_0, split_sizes = var_5691_split_sizes_0, x = out_281_cast_fp16)[name = string("op_5691")]; fp16 const_329_promoted = const()[name = string("const_329_promoted"), val = fp16(-0x1p+0)]; tensor var_5693 = mul(x = var_5691_1, y = const_329_promoted)[name = string("op_5693")]; bool var_5695_interleave_0 = const()[name = string("op_5695_interleave_0"), val = bool(false)]; tensor var_5695 = concat(axis = var_22, interleave = var_5695_interleave_0, values = (var_5693, var_5691_0))[name = string("op_5695")]; tensor var_5696 = mul(x = var_5695, y = sin_quantized)[name = string("op_5696")]; tensor hidden_states_277 = add(x = var_5690, y = var_5696)[name = string("hidden_states_277")]; tensor hidden_states_279_axes_0 = const()[name = string("hidden_states_279_axes_0"), val = tensor([2])]; tensor hidden_states_279 = expand_dims(axes = hidden_states_279_axes_0, x = hidden_states_277)[name = string("hidden_states_279")]; tensor var_5699 = const()[name = string("op_5699"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_281 = tile(reps = var_5699, x = hidden_states_279)[name = string("hidden_states_281")]; tensor var_5701 = const()[name = string("op_5701"), val = tensor([1, 3, 256, 256])]; tensor k = reshape(shape = var_5701, x = hidden_states_281)[name = string("k")]; tensor hidden_states_285_axes_0 = const()[name = string("hidden_states_285_axes_0"), val = tensor([2])]; tensor hidden_states_283 = transpose(perm = var_5648, x = var_5647)[name = string("transpose_5")]; tensor hidden_states_285 = expand_dims(axes = hidden_states_285_axes_0, x = hidden_states_283)[name = string("hidden_states_285")]; tensor var_5704 = const()[name = string("op_5704"), val = tensor([1, 1, 3, 1, 1])]; tensor hidden_states_287 = tile(reps = var_5704, x = hidden_states_285)[name = string("hidden_states_287")]; tensor var_5706 = const()[name = string("op_5706"), val = tensor([1, 3, 256, 256])]; tensor v = reshape(shape = var_5706, x = hidden_states_287)[name = string("v")]; bool var_5711_transpose_x_1 = const()[name = string("op_5711_transpose_x_1"), val = bool(false)]; bool var_5711_transpose_y_1 = const()[name = string("op_5711_transpose_y_1"), val = bool(true)]; tensor var_5711_cast_fp16 = matmul(transpose_x = var_5711_transpose_x_1, transpose_y = var_5711_transpose_y_1, x = q, y = k)[name = string("op_5711_cast_fp16")]; fp16 var_5712_to_fp16 = const()[name = string("op_5712_to_fp16"), val = fp16(0x1p-4)]; tensor attn_weights_139_cast_fp16 = mul(x = var_5711_cast_fp16, y = var_5712_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = full_mask_cast_fp16)[name = string("attn_weights_141_cast_fp16")]; tensor var_5716_cast_fp16 = softmax(axis = var_22, x = attn_weights_141_cast_fp16)[name = string("op_5716_cast_fp16")]; bool var_5720_transpose_x_0 = const()[name = string("op_5720_transpose_x_0"), val = bool(false)]; bool var_5720_transpose_y_0 = const()[name = string("op_5720_transpose_y_0"), val = bool(false)]; tensor var_5720_cast_fp16 = matmul(transpose_x = var_5720_transpose_x_0, transpose_y = var_5720_transpose_y_0, x = var_5716_cast_fp16, y = v)[name = string("op_5720_cast_fp16")]; tensor var_5722 = const()[name = string("op_5722"), val = tensor([0, 2, 1, 3])]; tensor var_5725 = const()[name = string("op_5725"), val = tensor([1, 256, 768])]; tensor var_5723 = transpose(perm = var_5722, x = var_5720_cast_fp16)[name = string("transpose_4")]; tensor attn_out_141 = reshape(shape = var_5725, x = var_5723)[name = string("attn_out_141")]; tensor var_5727 = const()[name = string("op_5727"), val = tensor([0, 2, 1])]; tensor squeeze_23_quantized = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308149760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308739648))))[name = string("squeeze_23_quantized")]; string var_5736_pad_type_0 = const()[name = string("op_5736_pad_type_0"), val = string("valid")]; int32 var_5736_groups_0 = const()[name = string("op_5736_groups_0"), val = int32(1)]; tensor var_5736_strides_0 = const()[name = string("op_5736_strides_0"), val = tensor([1])]; tensor var_5736_pad_0 = const()[name = string("op_5736_pad_0"), val = tensor([0, 0])]; tensor var_5736_dilations_0 = const()[name = string("op_5736_dilations_0"), val = tensor([1])]; tensor var_5728 = transpose(perm = var_5727, x = attn_out_141)[name = string("transpose_3")]; tensor var_5736 = conv(dilations = var_5736_dilations_0, groups = var_5736_groups_0, pad = var_5736_pad_0, pad_type = var_5736_pad_type_0, strides = var_5736_strides_0, weight = squeeze_23_quantized, x = var_5728)[name = string("op_5736")]; tensor var_5737 = const()[name = string("op_5737"), val = tensor([0, 2, 1])]; fp16 const_330_promoted_to_fp16 = const()[name = string("const_330_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_377 = transpose(perm = var_5737, x = var_5736)[name = string("transpose_2")]; tensor var_5741_cast_fp16 = mul(x = x_377, y = const_330_promoted_to_fp16)[name = string("op_5741_cast_fp16")]; bool input_471_interleave_0 = const()[name = string("input_471_interleave_0"), val = bool(false)]; tensor input_471_cast_fp16 = concat(axis = var_22, interleave = input_471_interleave_0, values = (x_377, var_5741_cast_fp16))[name = string("input_471_cast_fp16")]; tensor normed_659_axes_0 = const()[name = string("normed_659_axes_0"), val = tensor([-1])]; tensor normed_659_cast_fp16 = layer_norm(axes = normed_659_axes_0, epsilon = var_8_to_fp16, x = input_471_cast_fp16)[name = string("normed_659_cast_fp16")]; tensor var_5746_split_sizes_0 = const()[name = string("op_5746_split_sizes_0"), val = tensor([768, 768])]; int32 var_5746_axis_0 = const()[name = string("op_5746_axis_0"), val = int32(-1)]; tensor var_5746_cast_fp16_0, tensor var_5746_cast_fp16_1 = split(axis = var_5746_axis_0, split_sizes = var_5746_split_sizes_0, x = normed_659_cast_fp16)[name = string("op_5746_cast_fp16")]; tensor var_5750_to_fp16 = const()[name = string("op_5750_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308741248)))]; tensor out_283_cast_fp16 = mul(x = var_5746_cast_fp16_0, y = var_5750_to_fp16)[name = string("out_283_cast_fp16")]; tensor x_379_cast_fp16 = add(x = x_369_cast_fp16, y = out_283_cast_fp16)[name = string("x_379_cast_fp16")]; fp16 const_332_promoted_to_fp16 = const()[name = string("const_332_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5757_cast_fp16 = mul(x = x_379_cast_fp16, y = const_332_promoted_to_fp16)[name = string("op_5757_cast_fp16")]; bool input_473_interleave_0 = const()[name = string("input_473_interleave_0"), val = bool(false)]; tensor input_473_cast_fp16 = concat(axis = var_22, interleave = input_473_interleave_0, values = (x_379_cast_fp16, var_5757_cast_fp16))[name = string("input_473_cast_fp16")]; tensor normed_663_axes_0 = const()[name = string("normed_663_axes_0"), val = tensor([-1])]; tensor normed_663_cast_fp16 = layer_norm(axes = normed_663_axes_0, epsilon = var_8_to_fp16, x = input_473_cast_fp16)[name = string("normed_663_cast_fp16")]; tensor var_5762_split_sizes_0 = const()[name = string("op_5762_split_sizes_0"), val = tensor([768, 768])]; int32 var_5762_axis_0 = const()[name = string("op_5762_axis_0"), val = int32(-1)]; tensor var_5762_cast_fp16_0, tensor var_5762_cast_fp16_1 = split(axis = var_5762_axis_0, split_sizes = var_5762_split_sizes_0, x = normed_663_cast_fp16)[name = string("op_5762_cast_fp16")]; tensor var_5766_to_fp16 = const()[name = string("op_5766_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308742848)))]; tensor out_285_cast_fp16 = mul(x = var_5762_cast_fp16_0, y = var_5766_to_fp16)[name = string("out_285_cast_fp16")]; tensor var_5773 = const()[name = string("op_5773"), val = tensor([0, 2, 1])]; tensor input_475_axes_0 = const()[name = string("input_475_axes_0"), val = tensor([2])]; tensor var_5774 = transpose(perm = var_5773, x = out_285_cast_fp16)[name = string("transpose_1")]; tensor input_475 = expand_dims(axes = input_475_axes_0, x = var_5774)[name = string("input_475")]; string gate_93_pad_type_0 = const()[name = string("gate_93_pad_type_0"), val = string("valid")]; tensor gate_93_strides_0 = const()[name = string("gate_93_strides_0"), val = tensor([1, 1])]; tensor gate_93_pad_0 = const()[name = string("gate_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor gate_93_dilations_0 = const()[name = string("gate_93_dilations_0"), val = tensor([1, 1])]; int32 gate_93_groups_0 = const()[name = string("gate_93_groups_0"), val = int32(1)]; tensor gate_93 = conv(dilations = gate_93_dilations_0, groups = gate_93_groups_0, pad = gate_93_pad_0, pad_type = gate_93_pad_type_0, strides = gate_93_strides_0, weight = encoder_layers_23_mlp_gate_proj_weight_quantized, x = input_475)[name = string("gate_93")]; string up_pad_type_0 = const()[name = string("up_pad_type_0"), val = string("valid")]; tensor up_strides_0 = const()[name = string("up_strides_0"), val = tensor([1, 1])]; tensor up_pad_0 = const()[name = string("up_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_dilations_0 = const()[name = string("up_dilations_0"), val = tensor([1, 1])]; int32 up_groups_0 = const()[name = string("up_groups_0"), val = int32(1)]; tensor up = conv(dilations = up_dilations_0, groups = up_groups_0, pad = up_pad_0, pad_type = up_pad_type_0, strides = up_strides_0, weight = encoder_layers_23_mlp_up_proj_weight_quantized, x = input_475)[name = string("up")]; string gate_mode_0 = const()[name = string("gate_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate = gelu(mode = gate_mode_0, x = gate_93)[name = string("gate")]; tensor input_477 = mul(x = gate, y = up)[name = string("input_477")]; string var_5795_pad_type_0 = const()[name = string("op_5795_pad_type_0"), val = string("valid")]; tensor var_5795_strides_0 = const()[name = string("op_5795_strides_0"), val = tensor([1, 1])]; tensor var_5795_pad_0 = const()[name = string("op_5795_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5795_dilations_0 = const()[name = string("op_5795_dilations_0"), val = tensor([1, 1])]; int32 var_5795_groups_0 = const()[name = string("op_5795_groups_0"), val = int32(1)]; tensor var_5795 = conv(dilations = var_5795_dilations_0, groups = var_5795_groups_0, pad = var_5795_pad_0, pad_type = var_5795_pad_type_0, strides = var_5795_strides_0, weight = encoder_layers_23_mlp_down_proj_weight_quantized, x = input_477)[name = string("op_5795")]; tensor var_5796_axes_0 = const()[name = string("op_5796_axes_0"), val = tensor([2])]; tensor var_5796 = squeeze(axes = var_5796_axes_0, x = var_5795)[name = string("op_5796")]; tensor var_5797 = const()[name = string("op_5797"), val = tensor([0, 2, 1])]; fp16 const_334_promoted_to_fp16 = const()[name = string("const_334_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_383 = transpose(perm = var_5797, x = var_5796)[name = string("transpose_0")]; tensor var_5801_cast_fp16 = mul(x = x_383, y = const_334_promoted_to_fp16)[name = string("op_5801_cast_fp16")]; bool input_479_interleave_0 = const()[name = string("input_479_interleave_0"), val = bool(false)]; tensor input_479_cast_fp16 = concat(axis = var_22, interleave = input_479_interleave_0, values = (x_383, var_5801_cast_fp16))[name = string("input_479_cast_fp16")]; tensor normed_669_axes_0 = const()[name = string("normed_669_axes_0"), val = tensor([-1])]; tensor normed_669_cast_fp16 = layer_norm(axes = normed_669_axes_0, epsilon = var_8_to_fp16, x = input_479_cast_fp16)[name = string("normed_669_cast_fp16")]; tensor var_5806_split_sizes_0 = const()[name = string("op_5806_split_sizes_0"), val = tensor([768, 768])]; int32 var_5806_axis_0 = const()[name = string("op_5806_axis_0"), val = int32(-1)]; tensor var_5806_cast_fp16_0, tensor var_5806_cast_fp16_1 = split(axis = var_5806_axis_0, split_sizes = var_5806_split_sizes_0, x = normed_669_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_to_fp16 = const()[name = string("op_5810_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308744448)))]; tensor out_287_cast_fp16 = mul(x = var_5806_cast_fp16_0, y = var_5810_to_fp16)[name = string("out_287_cast_fp16")]; tensor x_385_cast_fp16 = add(x = x_379_cast_fp16, y = out_287_cast_fp16)[name = string("x_385_cast_fp16")]; fp16 const_336_promoted_to_fp16 = const()[name = string("const_336_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5817_cast_fp16 = mul(x = x_385_cast_fp16, y = const_336_promoted_to_fp16)[name = string("op_5817_cast_fp16")]; bool input_481_interleave_0 = const()[name = string("input_481_interleave_0"), val = bool(false)]; tensor input_481_cast_fp16 = concat(axis = var_22, interleave = input_481_interleave_0, values = (x_385_cast_fp16, var_5817_cast_fp16))[name = string("input_481_cast_fp16")]; tensor normed_673_axes_0 = const()[name = string("normed_673_axes_0"), val = tensor([-1])]; tensor normed_673_cast_fp16 = layer_norm(axes = normed_673_axes_0, epsilon = var_8_to_fp16, x = input_481_cast_fp16)[name = string("normed_673_cast_fp16")]; tensor var_5822_split_sizes_0 = const()[name = string("op_5822_split_sizes_0"), val = tensor([768, 768])]; int32 var_5822_axis_0 = const()[name = string("op_5822_axis_0"), val = int32(-1)]; tensor var_5822_cast_fp16_0, tensor var_5822_cast_fp16_1 = split(axis = var_5822_axis_0, split_sizes = var_5822_split_sizes_0, x = normed_673_cast_fp16)[name = string("op_5822_cast_fp16")]; tensor var_5826_to_fp16 = const()[name = string("op_5826_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308746048)))]; tensor out_cast_fp16 = mul(x = var_5822_cast_fp16_0, y = var_5826_to_fp16)[name = string("out_cast_fp16")]; tensor mask_axes_0 = const()[name = string("mask_axes_0"), val = tensor([-1])]; tensor mask_cast_fp16 = expand_dims(axes = mask_axes_0, x = attention_mask)[name = string("mask_cast_fp16")]; tensor masked_cast_fp16 = mul(x = out_cast_fp16, y = mask_cast_fp16)[name = string("masked_cast_fp16")]; tensor summed_axes_0 = const()[name = string("summed_axes_0"), val = tensor([1])]; bool summed_keep_dims_0 = const()[name = string("summed_keep_dims_0"), val = bool(false)]; tensor summed_cast_fp16 = reduce_sum(axes = summed_axes_0, keep_dims = summed_keep_dims_0, x = masked_cast_fp16)[name = string("summed_cast_fp16")]; tensor var_5846_axes_0 = const()[name = string("op_5846_axes_0"), val = tensor([1])]; bool var_5846_keep_dims_0 = const()[name = string("op_5846_keep_dims_0"), val = bool(false)]; tensor var_5846_cast_fp16 = reduce_sum(axes = var_5846_axes_0, keep_dims = var_5846_keep_dims_0, x = mask_cast_fp16)[name = string("op_5846_cast_fp16")]; fp16 var_5847_to_fp16 = const()[name = string("op_5847_to_fp16"), val = fp16(0x1p+0)]; tensor denom_cast_fp16 = maximum(x = var_5846_cast_fp16, y = var_5847_to_fp16)[name = string("denom_cast_fp16")]; tensor pooled_1_cast_fp16 = real_div(x = summed_cast_fp16, y = denom_cast_fp16)[name = string("pooled_1_cast_fp16")]; tensor var_5856_axes_0 = const()[name = string("op_5856_axes_0"), val = tensor([-1])]; tensor var_5856 = expand_dims(axes = var_5856_axes_0, x = pooled_1_cast_fp16)[name = string("op_5856")]; tensor input_483_axes_0 = const()[name = string("input_483_axes_0"), val = tensor([-1])]; tensor input_483 = expand_dims(axes = input_483_axes_0, x = var_5856)[name = string("input_483")]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input = conv(bias = dense1_bias, dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = dense1_weight_quantized, x = input_483)[name = string("input")]; string x_pad_type_0 = const()[name = string("x_pad_type_0"), val = string("valid")]; tensor x_strides_0 = const()[name = string("x_strides_0"), val = tensor([1, 1])]; tensor x_pad_0 = const()[name = string("x_pad_0"), val = tensor([0, 0, 0, 0])]; tensor x_dilations_0 = const()[name = string("x_dilations_0"), val = tensor([1, 1])]; int32 x_groups_0 = const()[name = string("x_groups_0"), val = int32(1)]; tensor x = conv(bias = dense2_bias, dilations = x_dilations_0, groups = x_groups_0, pad = x_pad_0, pad_type = x_pad_type_0, strides = x_strides_0, weight = dense2_weight_quantized, x = input)[name = string("x")]; tensor var_5883 = const()[name = string("op_5883"), val = tensor([1, 768])]; tensor pooled = reshape(shape = var_5883, x = x)[name = string("pooled")]; fp16 const_338 = const()[name = string("const_338"), val = fp16(0x1.1p-20)]; tensor var_5893 = abs(x = pooled)[name = string("op_5893")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = var_5893)[name = string("reduce_max_0")]; tensor abs_max = maximum(x = reduce_max_0, y = const_338)[name = string("abs_max")]; tensor scaled_cast_fp16 = real_div(x = pooled, y = abs_max)[name = string("scaled_cast_fp16")]; tensor var_5900_cast_fp16 = mul(x = scaled_cast_fp16, y = scaled_cast_fp16)[name = string("op_5900_cast_fp16")]; tensor sumsq_axes_0 = const()[name = string("sumsq_axes_0"), val = tensor([-1])]; bool sumsq_keep_dims_0 = const()[name = string("sumsq_keep_dims_0"), val = bool(true)]; tensor sumsq_cast_fp16 = reduce_sum(axes = sumsq_axes_0, keep_dims = sumsq_keep_dims_0, x = var_5900_cast_fp16)[name = string("sumsq_cast_fp16")]; fp16 var_5907_to_fp16 = const()[name = string("op_5907_to_fp16"), val = fp16(0x1p-24)]; tensor var_5908_cast_fp16 = add(x = sumsq_cast_fp16, y = var_5907_to_fp16)[name = string("op_5908_cast_fp16")]; fp32 inv_norm_epsilon_0 = const()[name = string("inv_norm_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor inv_norm_cast_fp16 = rsqrt(epsilon = inv_norm_epsilon_0, x = var_5908_cast_fp16)[name = string("inv_norm_cast_fp16")]; tensor embedding = mul(x = scaled_cast_fp16, y = inv_norm_cast_fp16)[name = string("normalized_cast_fp16")]; } -> (embedding); }