program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3510.2.1"}, {"coremlc-version", "3500.32.1"}})] { func infer(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { tensor model_model_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1572992))))[name = string("model_model_layers_0_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1605824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2392320))))[name = string("model_model_layers_0_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2408768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3195264))))[name = string("model_model_layers_0_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3211712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5571072))))[name = string("model_model_layers_0_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5620288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7979648))))[name = string("model_model_layers_0_mlp_up_proj_weight_palettized")]; tensor model_model_layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8028864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10388224))))[name = string("model_model_layers_0_mlp_down_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10404672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11977600))))[name = string("model_model_layers_1_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12010432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12796928))))[name = string("model_model_layers_1_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12813376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13599872))))[name = string("model_model_layers_1_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13616320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15975680))))[name = string("model_model_layers_1_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16024896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18384256))))[name = string("model_model_layers_1_mlp_up_proj_weight_palettized")]; tensor model_model_layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18433472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20792832))))[name = string("model_model_layers_1_mlp_down_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20809280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22382208))))[name = string("model_model_layers_2_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22415040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23201536))))[name = string("model_model_layers_2_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23217984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24004480))))[name = string("model_model_layers_2_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24020928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26380288))))[name = string("model_model_layers_2_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26429504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28788864))))[name = string("model_model_layers_2_mlp_up_proj_weight_palettized")]; tensor model_model_layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28838080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31197440))))[name = string("model_model_layers_2_mlp_down_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31213888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(32786816))))[name = string("model_model_layers_3_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(32819648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33606144))))[name = string("model_model_layers_3_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33622592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34409088))))[name = string("model_model_layers_3_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34425536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36784896))))[name = string("model_model_layers_3_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36834112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39193472))))[name = string("model_model_layers_3_mlp_up_proj_weight_palettized")]; tensor model_model_layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39242688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41602048))))[name = string("model_model_layers_3_mlp_down_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41618496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43191424))))[name = string("model_model_layers_4_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43224256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44010752))))[name = string("model_model_layers_4_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44027200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44813696))))[name = string("model_model_layers_4_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44830144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47189504))))[name = string("model_model_layers_4_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47238720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49598080))))[name = string("model_model_layers_4_mlp_up_proj_weight_palettized")]; tensor model_model_layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49647296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52006656))))[name = string("model_model_layers_4_mlp_down_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52023104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53596032))))[name = string("model_model_layers_5_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53628864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(54415360))))[name = string("model_model_layers_5_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(54431808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55218304))))[name = string("model_model_layers_5_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55234752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57594112))))[name = string("model_model_layers_5_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57643328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60002688))))[name = string("model_model_layers_5_mlp_up_proj_weight_palettized")]; tensor model_model_layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60051904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62411264))))[name = string("model_model_layers_5_mlp_down_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62427712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64000640))))[name = string("model_model_layers_6_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64033472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64819968))))[name = string("model_model_layers_6_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64836416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65622912))))[name = string("model_model_layers_6_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65639360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67998720))))[name = string("model_model_layers_6_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68047936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70407296))))[name = string("model_model_layers_6_mlp_up_proj_weight_palettized")]; tensor model_model_layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70456512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72815872))))[name = string("model_model_layers_6_mlp_down_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72832320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74405248))))[name = string("model_model_layers_7_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74438080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75224576))))[name = string("model_model_layers_7_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75241024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76027520))))[name = string("model_model_layers_7_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76043968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78403328))))[name = string("model_model_layers_7_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78452544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80811904))))[name = string("model_model_layers_7_mlp_up_proj_weight_palettized")]; tensor model_model_layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80861120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(83220480))))[name = string("model_model_layers_7_mlp_down_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(83236928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84809856))))[name = string("model_model_layers_8_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84842688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85629184))))[name = string("model_model_layers_8_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85645632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86432128))))[name = string("model_model_layers_8_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86448576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88807936))))[name = string("model_model_layers_8_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88857152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91216512))))[name = string("model_model_layers_8_mlp_up_proj_weight_palettized")]; tensor model_model_layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91265728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93625088))))[name = string("model_model_layers_8_mlp_down_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93641536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95214464))))[name = string("model_model_layers_9_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95247296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96033792))))[name = string("model_model_layers_9_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96050240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96836736))))[name = string("model_model_layers_9_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96853184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99212544))))[name = string("model_model_layers_9_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99261760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101621120))))[name = string("model_model_layers_9_mlp_up_proj_weight_palettized")]; tensor model_model_layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101670336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104029696))))[name = string("model_model_layers_9_mlp_down_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104046144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105619072))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105651904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106438400))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106454848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(107241344))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(107257792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109617152))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109666368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112025728))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112074944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(114434304))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(114450752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116023680))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116056512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116843008))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116859456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117645952))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117662400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120021760))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120070976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(122430336))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(122479552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124838912))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124855360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(126428288))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(126461120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(127247616))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(127264064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128050560))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128067008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130426368))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130475584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(132834944))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(132884160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135243520))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135259968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136832896))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136865728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137652224))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137668672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138455168))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138471616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(140830976))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(140880192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143239552))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143288768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145648128))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145664576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147237504))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147270336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148056832))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148073280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148859776))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148876224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151235584))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151284800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153644160))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153693376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(156052736))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(156069184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(157642112))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(157674944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158461440))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158477888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159264384))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159280832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161640192))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161689408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(164048768))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(164097984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(166457344))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; tensor model_model_layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(166473792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168046720))))[name = string("model_model_layers_16_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_16_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168079552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168866048))))[name = string("model_model_layers_16_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_16_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168882496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169668992))))[name = string("model_model_layers_16_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169685440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172044800))))[name = string("model_model_layers_16_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172094016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174453376))))[name = string("model_model_layers_16_mlp_up_proj_weight_palettized")]; tensor model_model_layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174502592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176861952))))[name = string("model_model_layers_16_mlp_down_proj_weight_palettized")]; tensor model_model_layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176878400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178451328))))[name = string("model_model_layers_17_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_17_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178484160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(179270656))))[name = string("model_model_layers_17_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_17_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(179287104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180073600))))[name = string("model_model_layers_17_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180090048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(182449408))))[name = string("model_model_layers_17_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(182498624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184857984))))[name = string("model_model_layers_17_mlp_up_proj_weight_palettized")]; tensor model_model_layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184907200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(187266560))))[name = string("model_model_layers_17_mlp_down_proj_weight_palettized")]; tensor model_model_layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(187283008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(188855936))))[name = string("model_model_layers_18_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_18_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(188888768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189675264))))[name = string("model_model_layers_18_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_18_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189691712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190478208))))[name = string("model_model_layers_18_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190494656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(192854016))))[name = string("model_model_layers_18_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(192903232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195262592))))[name = string("model_model_layers_18_mlp_up_proj_weight_palettized")]; tensor model_model_layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195311808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197671168))))[name = string("model_model_layers_18_mlp_down_proj_weight_palettized")]; tensor model_model_layers_19_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197687616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(199260544))))[name = string("model_model_layers_19_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_19_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(199293376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200079872))))[name = string("model_model_layers_19_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_19_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200096320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200882816))))[name = string("model_model_layers_19_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_19_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200899264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203258624))))[name = string("model_model_layers_19_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_19_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203307840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205667200))))[name = string("model_model_layers_19_mlp_up_proj_weight_palettized")]; tensor model_model_layers_19_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205716416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(208075776))))[name = string("model_model_layers_19_mlp_down_proj_weight_palettized")]; tensor model_model_layers_20_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(208092224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(209665152))))[name = string("model_model_layers_20_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_20_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(209697984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210484480))))[name = string("model_model_layers_20_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_20_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210500928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211287424))))[name = string("model_model_layers_20_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_20_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211303872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213663232))))[name = string("model_model_layers_20_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_20_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213712448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216071808))))[name = string("model_model_layers_20_mlp_up_proj_weight_palettized")]; tensor model_model_layers_20_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216121024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218480384))))[name = string("model_model_layers_20_mlp_down_proj_weight_palettized")]; tensor model_model_layers_21_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218496832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220069760))))[name = string("model_model_layers_21_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_21_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220102592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220889088))))[name = string("model_model_layers_21_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_21_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220905536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(221692032))))[name = string("model_model_layers_21_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_21_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(221708480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224067840))))[name = string("model_model_layers_21_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_21_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224117056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226476416))))[name = string("model_model_layers_21_mlp_up_proj_weight_palettized")]; tensor model_model_layers_21_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226525632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228884992))))[name = string("model_model_layers_21_mlp_down_proj_weight_palettized")]; tensor model_model_layers_22_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228901440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230474368))))[name = string("model_model_layers_22_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_22_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230507200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(231293696))))[name = string("model_model_layers_22_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_22_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(231310144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232096640))))[name = string("model_model_layers_22_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_22_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232113088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234472448))))[name = string("model_model_layers_22_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_22_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234521664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236881024))))[name = string("model_model_layers_22_mlp_up_proj_weight_palettized")]; tensor model_model_layers_22_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236930240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239289600))))[name = string("model_model_layers_22_mlp_down_proj_weight_palettized")]; tensor model_model_layers_23_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239306048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240878976))))[name = string("model_model_layers_23_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_23_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240911808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241698304))))[name = string("model_model_layers_23_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_23_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241714752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242501248))))[name = string("model_model_layers_23_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_23_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242517696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244877056))))[name = string("model_model_layers_23_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_23_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244926272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247285632))))[name = string("model_model_layers_23_mlp_up_proj_weight_palettized")]; tensor model_model_layers_23_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247334848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249694208))))[name = string("model_model_layers_23_mlp_down_proj_weight_palettized")]; tensor model_model_layers_24_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249710656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251283584))))[name = string("model_model_layers_24_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_24_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251316416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252102912))))[name = string("model_model_layers_24_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_24_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252119360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252905856))))[name = string("model_model_layers_24_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_24_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252922304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255281664))))[name = string("model_model_layers_24_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_24_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255330880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257690240))))[name = string("model_model_layers_24_mlp_up_proj_weight_palettized")]; tensor model_model_layers_24_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257739456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260098816))))[name = string("model_model_layers_24_mlp_down_proj_weight_palettized")]; tensor model_model_layers_25_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260115264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261688192))))[name = string("model_model_layers_25_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_25_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261721024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262507520))))[name = string("model_model_layers_25_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_25_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262523968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263310464))))[name = string("model_model_layers_25_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_25_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263326912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265686272))))[name = string("model_model_layers_25_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_25_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265735488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268094848))))[name = string("model_model_layers_25_mlp_up_proj_weight_palettized")]; tensor model_model_layers_25_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268144064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270503424))))[name = string("model_model_layers_25_mlp_down_proj_weight_palettized")]; tensor model_model_layers_26_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270519872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272092800))))[name = string("model_model_layers_26_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_26_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272125632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272912128))))[name = string("model_model_layers_26_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_26_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272928576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273715072))))[name = string("model_model_layers_26_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_26_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273731520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276090880))))[name = string("model_model_layers_26_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_26_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276140096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278499456))))[name = string("model_model_layers_26_mlp_up_proj_weight_palettized")]; tensor model_model_layers_26_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278548672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280908032))))[name = string("model_model_layers_26_mlp_down_proj_weight_palettized")]; tensor model_model_layers_27_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280924480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282497408))))[name = string("model_model_layers_27_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_27_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282530240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283316736))))[name = string("model_model_layers_27_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_27_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283333184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284119680))))[name = string("model_model_layers_27_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_27_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284136128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286495488))))[name = string("model_model_layers_27_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_27_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286544704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288904064))))[name = string("model_model_layers_27_mlp_up_proj_weight_palettized")]; tensor model_model_layers_27_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288953280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291312640))))[name = string("model_model_layers_27_mlp_down_proj_weight_palettized")]; int32 var_1503_batch_dims_0 = const()[name = string("op_1503_batch_dims_0"), val = int32(0)]; bool var_1503_validate_indices_0 = const()[name = string("op_1503_validate_indices_0"), val = bool(false)]; tensor var_1495_to_fp16 = const()[name = string("op_1495_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291329088)))]; string current_pos_to_int16_dtype_0 = const()[name = string("current_pos_to_int16_dtype_0"), val = string("int16")]; string cast_230_dtype_0 = const()[name = string("cast_230_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor current_pos_to_int16 = cast(dtype = current_pos_to_int16_dtype_0, x = current_pos)[name = string("cast_5")]; tensor cast_230 = cast(dtype = cast_230_dtype_0, x = current_pos_to_int16)[name = string("cast_4")]; tensor greater_equal_0 = greater_equal(x = cast_230, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(3072)]; tensor add_0 = add(x = cast_230, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = cast_230, b = add_0, cond = greater_equal_0)[name = string("select_0")]; string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(3072)]; tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_1503_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_1503_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; tensor op_1503_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_1503_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_1503_batch_dims_0, indices = select_0_1, validate_indices = var_1503_validate_indices_0, x = var_1495_to_fp16)[name = string("op_1503_cast_fp16_cast_uint16_cast_uint16")]; tensor var_1508 = const()[name = string("op_1508"), val = tensor([1, 1, 1, -1])]; tensor sin_1_cast_fp16 = reshape(shape = var_1508, x = op_1503_cast_fp16_cast_uint16_cast_uint16)[name = string("sin_1_cast_fp16")]; int32 var_1518_axis_0 = const()[name = string("op_1518_axis_0"), val = int32(1)]; int32 var_1518_batch_dims_0 = const()[name = string("op_1518_batch_dims_0"), val = int32(0)]; bool var_1518_validate_indices_0 = const()[name = string("op_1518_validate_indices_0"), val = bool(false)]; tensor var_1510_to_fp16 = const()[name = string("op_1510_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292115584)))]; string current_pos_to_uint16_dtype_0 = const()[name = string("current_pos_to_uint16_dtype_0"), val = string("uint16")]; tensor current_pos_to_uint16 = cast(dtype = current_pos_to_uint16_dtype_0, x = current_pos)[name = string("cast_1")]; tensor var_1518_cast_fp16_cast_uint16 = gather(axis = var_1518_axis_0, batch_dims = var_1518_batch_dims_0, indices = current_pos_to_uint16, validate_indices = var_1518_validate_indices_0, x = var_1510_to_fp16)[name = string("op_1518_cast_fp16_cast_uint16")]; tensor var_1523 = const()[name = string("op_1523"), val = tensor([1, 1, 1, -1])]; tensor cos_1_cast_fp16 = reshape(shape = var_1523, x = var_1518_cast_fp16_cast_uint16)[name = string("cos_1_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(-1)]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_1546, interleave = input_1_interleave_0, values = (hidden_states, var_1548_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_1543_to_fp16 = const()[name = string("op_1543_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_1543_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 1, 1024])]; tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; tensor const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292902080)))]; tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_2_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor var_1560 = const()[name = string("op_1560"), val = tensor([0, 2, 1])]; tensor var_1563_axes_0 = const()[name = string("op_1563_axes_0"), val = tensor([2])]; tensor var_1561_cast_fp16 = transpose(perm = var_1560, x = hidden_states_3_cast_fp16)[name = string("transpose_167")]; tensor var_1563_cast_fp16 = expand_dims(axes = var_1563_axes_0, x = var_1561_cast_fp16)[name = string("op_1563_cast_fp16")]; string var_1579_pad_type_0 = const()[name = string("op_1579_pad_type_0"), val = string("valid")]; tensor var_1579_strides_0 = const()[name = string("op_1579_strides_0"), val = tensor([1, 1])]; tensor var_1579_pad_0 = const()[name = string("op_1579_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1579_dilations_0 = const()[name = string("op_1579_dilations_0"), val = tensor([1, 1])]; int32 var_1579_groups_0 = const()[name = string("op_1579_groups_0"), val = int32(1)]; tensor var_1579 = conv(dilations = var_1579_dilations_0, groups = var_1579_groups_0, pad = var_1579_pad_0, pad_type = var_1579_pad_type_0, strides = var_1579_strides_0, weight = model_model_layers_0_self_attn_q_proj_weight_palettized, x = var_1563_cast_fp16)[name = string("op_1579")]; tensor var_1584 = const()[name = string("op_1584"), val = tensor([1, 16, 1, 128])]; tensor var_1585 = reshape(shape = var_1584, x = var_1579)[name = string("op_1585")]; string var_1601_pad_type_0 = const()[name = string("op_1601_pad_type_0"), val = string("valid")]; tensor var_1601_strides_0 = const()[name = string("op_1601_strides_0"), val = tensor([1, 1])]; tensor var_1601_pad_0 = const()[name = string("op_1601_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1601_dilations_0 = const()[name = string("op_1601_dilations_0"), val = tensor([1, 1])]; int32 var_1601_groups_0 = const()[name = string("op_1601_groups_0"), val = int32(1)]; tensor var_1601 = conv(dilations = var_1601_dilations_0, groups = var_1601_groups_0, pad = var_1601_pad_0, pad_type = var_1601_pad_type_0, strides = var_1601_strides_0, weight = model_model_layers_0_self_attn_k_proj_weight_palettized, x = var_1563_cast_fp16)[name = string("op_1601")]; tensor var_1606 = const()[name = string("op_1606"), val = tensor([1, 8, 1, 128])]; tensor var_1607 = reshape(shape = var_1606, x = var_1601)[name = string("op_1607")]; string var_1623_pad_type_0 = const()[name = string("op_1623_pad_type_0"), val = string("valid")]; tensor var_1623_strides_0 = const()[name = string("op_1623_strides_0"), val = tensor([1, 1])]; tensor var_1623_pad_0 = const()[name = string("op_1623_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1623_dilations_0 = const()[name = string("op_1623_dilations_0"), val = tensor([1, 1])]; int32 var_1623_groups_0 = const()[name = string("op_1623_groups_0"), val = int32(1)]; tensor var_1623 = conv(dilations = var_1623_dilations_0, groups = var_1623_groups_0, pad = var_1623_pad_0, pad_type = var_1623_pad_type_0, strides = var_1623_strides_0, weight = model_model_layers_0_self_attn_v_proj_weight_palettized, x = var_1563_cast_fp16)[name = string("op_1623")]; tensor var_1628 = const()[name = string("op_1628"), val = tensor([1, 8, 1, 128])]; tensor var_1629 = reshape(shape = var_1628, x = var_1623)[name = string("op_1629")]; int32 var_1646 = const()[name = string("op_1646"), val = int32(-1)]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_1648 = mul(x = var_1585, y = const_3_promoted)[name = string("op_1648")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_1646, interleave = input_5_interleave_0, values = (var_1585, var_1648))[name = string("input_5")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_1643_to_fp16 = const()[name = string("op_1643_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_1643_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_7 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7")]; tensor const_5 = const()[name = string("const_5"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292904192)))]; tensor q_1 = mul(x = normed_7, y = const_5)[name = string("q_1")]; int32 var_1668 = const()[name = string("op_1668"), val = int32(-1)]; fp16 const_6_promoted = const()[name = string("const_6_promoted"), val = fp16(-0x1p+0)]; tensor var_1670 = mul(x = var_1607, y = const_6_promoted)[name = string("op_1670")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7 = concat(axis = var_1668, interleave = input_7_interleave_0, values = (var_1607, var_1670))[name = string("input_7")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_1665_to_fp16 = const()[name = string("op_1665_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_1665_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_11 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11")]; tensor const_8 = const()[name = string("const_8"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292904512)))]; tensor k_1 = mul(x = normed_11, y = const_8)[name = string("k_1")]; tensor var_1679 = mul(x = q_1, y = cos_1_cast_fp16)[name = string("op_1679")]; tensor var_1684_begin_0 = const()[name = string("op_1684_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_1684_end_0 = const()[name = string("op_1684_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_1684_end_mask_0 = const()[name = string("op_1684_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1684 = slice_by_index(begin = var_1684_begin_0, end = var_1684_end_0, end_mask = var_1684_end_mask_0, x = q_1)[name = string("op_1684")]; fp16 const_9_promoted = const()[name = string("const_9_promoted"), val = fp16(-0x1p+0)]; tensor var_1685 = mul(x = var_1684, y = const_9_promoted)[name = string("op_1685")]; tensor var_1690_begin_0 = const()[name = string("op_1690_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1690_end_0 = const()[name = string("op_1690_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_1690_end_mask_0 = const()[name = string("op_1690_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_1690 = slice_by_index(begin = var_1690_begin_0, end = var_1690_end_0, end_mask = var_1690_end_mask_0, x = q_1)[name = string("op_1690")]; int32 var_1692 = const()[name = string("op_1692"), val = int32(-1)]; bool var_1693_interleave_0 = const()[name = string("op_1693_interleave_0"), val = bool(false)]; tensor var_1693 = concat(axis = var_1692, interleave = var_1693_interleave_0, values = (var_1685, var_1690))[name = string("op_1693")]; tensor var_1694 = mul(x = var_1693, y = sin_1_cast_fp16)[name = string("op_1694")]; tensor query_states_1 = add(x = var_1679, y = var_1694)[name = string("query_states_1")]; tensor var_1697 = mul(x = k_1, y = cos_1_cast_fp16)[name = string("op_1697")]; tensor var_1702_begin_0 = const()[name = string("op_1702_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_1702_end_0 = const()[name = string("op_1702_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_1702_end_mask_0 = const()[name = string("op_1702_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1702 = slice_by_index(begin = var_1702_begin_0, end = var_1702_end_0, end_mask = var_1702_end_mask_0, x = k_1)[name = string("op_1702")]; fp16 const_10_promoted = const()[name = string("const_10_promoted"), val = fp16(-0x1p+0)]; tensor var_1703 = mul(x = var_1702, y = const_10_promoted)[name = string("op_1703")]; tensor var_1708_begin_0 = const()[name = string("op_1708_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1708_end_0 = const()[name = string("op_1708_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_1708_end_mask_0 = const()[name = string("op_1708_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_1708 = slice_by_index(begin = var_1708_begin_0, end = var_1708_end_0, end_mask = var_1708_end_mask_0, x = k_1)[name = string("op_1708")]; int32 var_1710 = const()[name = string("op_1710"), val = int32(-1)]; bool var_1711_interleave_0 = const()[name = string("op_1711_interleave_0"), val = bool(false)]; tensor var_1711 = concat(axis = var_1710, interleave = var_1711_interleave_0, values = (var_1703, var_1708))[name = string("op_1711")]; tensor var_1712 = mul(x = var_1711, y = sin_1_cast_fp16)[name = string("op_1712")]; tensor key_states_1 = add(x = var_1697, y = var_1712)[name = string("key_states_1")]; int32 var_1716 = const()[name = string("op_1716"), val = int32(1)]; tensor var_1717 = add(x = current_pos, y = var_1716)[name = string("op_1717")]; tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_1717, concat_3_values3_0))[name = string("concat_3")]; tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_1, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_112_write_state")]; tensor coreml_update_state_56 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_112")]; tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([28])]; tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([29])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_1717, concat_7_values3_0))[name = string("concat_7")]; tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = var_1629, x = coreml_update_state_56)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_113_write_state")]; tensor coreml_update_state_57 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_113")]; tensor var_1767_begin_0 = const()[name = string("op_1767_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1767_end_0 = const()[name = string("op_1767_end_0"), val = tensor([1, 8, 1536, 128])]; tensor var_1767_end_mask_0 = const()[name = string("op_1767_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1767_cast_fp16 = slice_by_index(begin = var_1767_begin_0, end = var_1767_end_0, end_mask = var_1767_end_mask_0, x = coreml_update_state_57)[name = string("op_1767_cast_fp16")]; tensor key_cache_1_axes_0 = const()[name = string("key_cache_1_axes_0"), val = tensor([0])]; tensor key_cache_1_cast_fp16 = squeeze(axes = key_cache_1_axes_0, x = var_1767_cast_fp16)[name = string("key_cache_1_cast_fp16")]; tensor var_1774_begin_0 = const()[name = string("op_1774_begin_0"), val = tensor([28, 0, 0, 0])]; tensor var_1774_end_0 = const()[name = string("op_1774_end_0"), val = tensor([29, 8, 1536, 128])]; tensor var_1774_end_mask_0 = const()[name = string("op_1774_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1774_cast_fp16 = slice_by_index(begin = var_1774_begin_0, end = var_1774_end_0, end_mask = var_1774_end_mask_0, x = coreml_update_state_57)[name = string("op_1774_cast_fp16")]; tensor value_cache_1_axes_0 = const()[name = string("value_cache_1_axes_0"), val = tensor([0])]; tensor value_cache_1_cast_fp16 = squeeze(axes = value_cache_1_axes_0, x = var_1774_cast_fp16)[name = string("value_cache_1_cast_fp16")]; tensor var_1798_axes_0 = const()[name = string("op_1798_axes_0"), val = tensor([1])]; tensor var_1798_cast_fp16 = expand_dims(axes = var_1798_axes_0, x = key_cache_1_cast_fp16)[name = string("op_1798_cast_fp16")]; tensor var_1803 = const()[name = string("op_1803"), val = tensor([1, 2, 1, 1])]; tensor value_3_cast_fp16 = tile(reps = var_1803, x = var_1798_cast_fp16)[name = string("value_3_cast_fp16")]; tensor var_1809 = const()[name = string("op_1809"), val = tensor([1, 16, 1536, 128])]; tensor key_states_3_cast_fp16 = reshape(shape = var_1809, x = value_3_cast_fp16)[name = string("key_states_3_cast_fp16")]; tensor var_1812_axes_0 = const()[name = string("op_1812_axes_0"), val = tensor([1])]; tensor var_1812_cast_fp16 = expand_dims(axes = var_1812_axes_0, x = value_cache_1_cast_fp16)[name = string("op_1812_cast_fp16")]; tensor var_1817 = const()[name = string("op_1817"), val = tensor([1, 2, 1, 1])]; tensor value_7_cast_fp16 = tile(reps = var_1817, x = var_1812_cast_fp16)[name = string("value_7_cast_fp16")]; tensor var_1823 = const()[name = string("op_1823"), val = tensor([1, 16, 1536, 128])]; tensor value_states_3_cast_fp16 = reshape(shape = var_1823, x = value_7_cast_fp16)[name = string("value_states_3_cast_fp16")]; bool var_1838_transpose_x_1 = const()[name = string("op_1838_transpose_x_1"), val = bool(false)]; bool var_1838_transpose_y_1 = const()[name = string("op_1838_transpose_y_1"), val = bool(true)]; tensor var_1838 = matmul(transpose_x = var_1838_transpose_x_1, transpose_y = var_1838_transpose_y_1, x = query_states_1, y = key_states_3_cast_fp16)[name = string("op_1838")]; fp16 var_1839_to_fp16 = const()[name = string("op_1839_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_1_cast_fp16 = mul(x = var_1838, y = var_1839_to_fp16)[name = string("attention_1_cast_fp16")]; tensor attention_3_cast_fp16 = add(x = attention_1_cast_fp16, y = causal_mask)[name = string("attention_3_cast_fp16")]; int32 var_1848 = const()[name = string("op_1848"), val = int32(-1)]; tensor probabilities_1_cast_fp16 = softmax(axis = var_1848, x = attention_3_cast_fp16)[name = string("probabilities_1_cast_fp16")]; bool output_1_transpose_x_0 = const()[name = string("output_1_transpose_x_0"), val = bool(false)]; bool output_1_transpose_y_0 = const()[name = string("output_1_transpose_y_0"), val = bool(false)]; tensor output_1_cast_fp16 = matmul(transpose_x = output_1_transpose_x_0, transpose_y = output_1_transpose_y_0, x = probabilities_1_cast_fp16, y = value_states_3_cast_fp16)[name = string("output_1_cast_fp16")]; tensor var_1859_perm_0 = const()[name = string("op_1859_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1865 = const()[name = string("op_1865"), val = tensor([1, 1, 2048])]; tensor var_1859_cast_fp16 = transpose(perm = var_1859_perm_0, x = output_1_cast_fp16)[name = string("transpose_166")]; tensor output_3_cast_fp16 = reshape(shape = var_1865, x = var_1859_cast_fp16)[name = string("output_3_cast_fp16")]; tensor var_1870 = const()[name = string("op_1870"), val = tensor([0, 2, 1])]; string var_1886_pad_type_0 = const()[name = string("op_1886_pad_type_0"), val = string("valid")]; int32 var_1886_groups_0 = const()[name = string("op_1886_groups_0"), val = int32(1)]; tensor var_1886_strides_0 = const()[name = string("op_1886_strides_0"), val = tensor([1])]; tensor var_1886_pad_0 = const()[name = string("op_1886_pad_0"), val = tensor([0, 0])]; tensor var_1886_dilations_0 = const()[name = string("op_1886_dilations_0"), val = tensor([1])]; tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292904832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294477760))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_1871_cast_fp16 = transpose(perm = var_1870, x = output_3_cast_fp16)[name = string("transpose_165")]; tensor var_1886_cast_fp16 = conv(dilations = var_1886_dilations_0, groups = var_1886_groups_0, pad = var_1886_pad_0, pad_type = var_1886_pad_type_0, strides = var_1886_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1871_cast_fp16)[name = string("op_1886_cast_fp16")]; tensor var_1890 = const()[name = string("op_1890"), val = tensor([0, 2, 1])]; tensor attn_output_1_cast_fp16 = transpose(perm = var_1890, x = var_1886_cast_fp16)[name = string("transpose_164")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states, y = attn_output_1_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; int32 var_1905 = const()[name = string("op_1905"), val = int32(-1)]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1907_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_1907_cast_fp16")]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11_cast_fp16 = concat(axis = var_1905, interleave = input_11_interleave_0, values = (hidden_states_9_cast_fp16, var_1907_cast_fp16))[name = string("input_11_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_1902_to_fp16 = const()[name = string("op_1902_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1902_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 1, 1024])]; tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; tensor const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294494208)))]; tensor x_1_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_13_promoted_to_fp16)[name = string("x_1_cast_fp16")]; tensor var_1927 = const()[name = string("op_1927"), val = tensor([0, 2, 1])]; tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; tensor var_1928 = transpose(perm = var_1927, x = x_1_cast_fp16)[name = string("transpose_163")]; tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_1928)[name = string("input_13")]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = model_model_layers_0_mlp_gate_proj_weight_palettized, x = input_13)[name = string("input_15")]; string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_0_mlp_up_proj_weight_palettized, x = input_13)[name = string("b_1")]; tensor c_1 = silu(x = input_15)[name = string("c_1")]; tensor input_17 = mul(x = c_1, y = b_1)[name = string("input_17")]; string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_0_mlp_down_proj_weight_palettized, x = input_17)[name = string("e_1")]; tensor var_1950_axes_0 = const()[name = string("op_1950_axes_0"), val = tensor([2])]; tensor var_1950 = squeeze(axes = var_1950_axes_0, x = e_1)[name = string("op_1950")]; tensor var_1951 = const()[name = string("op_1951"), val = tensor([0, 2, 1])]; tensor var_1952 = transpose(perm = var_1951, x = var_1950)[name = string("transpose_162")]; tensor hidden_states_11_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = var_1952)[name = string("hidden_states_11_cast_fp16")]; int32 var_1966 = const()[name = string("op_1966"), val = int32(-1)]; fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1968_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_1968_cast_fp16")]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_1966, interleave = input_19_interleave_0, values = (hidden_states_11_cast_fp16, var_1968_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_1963_to_fp16 = const()[name = string("op_1963_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1963_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 1, 1024])]; tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; tensor const_16_promoted_to_fp16 = const()[name = string("const_16_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294496320)))]; tensor hidden_states_13_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_16_promoted_to_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor var_1980 = const()[name = string("op_1980"), val = tensor([0, 2, 1])]; tensor var_1983_axes_0 = const()[name = string("op_1983_axes_0"), val = tensor([2])]; tensor var_1981_cast_fp16 = transpose(perm = var_1980, x = hidden_states_13_cast_fp16)[name = string("transpose_161")]; tensor var_1983_cast_fp16 = expand_dims(axes = var_1983_axes_0, x = var_1981_cast_fp16)[name = string("op_1983_cast_fp16")]; string var_1999_pad_type_0 = const()[name = string("op_1999_pad_type_0"), val = string("valid")]; tensor var_1999_strides_0 = const()[name = string("op_1999_strides_0"), val = tensor([1, 1])]; tensor var_1999_pad_0 = const()[name = string("op_1999_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1999_dilations_0 = const()[name = string("op_1999_dilations_0"), val = tensor([1, 1])]; int32 var_1999_groups_0 = const()[name = string("op_1999_groups_0"), val = int32(1)]; tensor var_1999 = conv(dilations = var_1999_dilations_0, groups = var_1999_groups_0, pad = var_1999_pad_0, pad_type = var_1999_pad_type_0, strides = var_1999_strides_0, weight = model_model_layers_1_self_attn_q_proj_weight_palettized, x = var_1983_cast_fp16)[name = string("op_1999")]; tensor var_2004 = const()[name = string("op_2004"), val = tensor([1, 16, 1, 128])]; tensor var_2005 = reshape(shape = var_2004, x = var_1999)[name = string("op_2005")]; string var_2021_pad_type_0 = const()[name = string("op_2021_pad_type_0"), val = string("valid")]; tensor var_2021_strides_0 = const()[name = string("op_2021_strides_0"), val = tensor([1, 1])]; tensor var_2021_pad_0 = const()[name = string("op_2021_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2021_dilations_0 = const()[name = string("op_2021_dilations_0"), val = tensor([1, 1])]; int32 var_2021_groups_0 = const()[name = string("op_2021_groups_0"), val = int32(1)]; tensor var_2021 = conv(dilations = var_2021_dilations_0, groups = var_2021_groups_0, pad = var_2021_pad_0, pad_type = var_2021_pad_type_0, strides = var_2021_strides_0, weight = model_model_layers_1_self_attn_k_proj_weight_palettized, x = var_1983_cast_fp16)[name = string("op_2021")]; tensor var_2026 = const()[name = string("op_2026"), val = tensor([1, 8, 1, 128])]; tensor var_2027 = reshape(shape = var_2026, x = var_2021)[name = string("op_2027")]; string var_2043_pad_type_0 = const()[name = string("op_2043_pad_type_0"), val = string("valid")]; tensor var_2043_strides_0 = const()[name = string("op_2043_strides_0"), val = tensor([1, 1])]; tensor var_2043_pad_0 = const()[name = string("op_2043_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2043_dilations_0 = const()[name = string("op_2043_dilations_0"), val = tensor([1, 1])]; int32 var_2043_groups_0 = const()[name = string("op_2043_groups_0"), val = int32(1)]; tensor var_2043 = conv(dilations = var_2043_dilations_0, groups = var_2043_groups_0, pad = var_2043_pad_0, pad_type = var_2043_pad_type_0, strides = var_2043_strides_0, weight = model_model_layers_1_self_attn_v_proj_weight_palettized, x = var_1983_cast_fp16)[name = string("op_2043")]; tensor var_2048 = const()[name = string("op_2048"), val = tensor([1, 8, 1, 128])]; tensor var_2049 = reshape(shape = var_2048, x = var_2043)[name = string("op_2049")]; int32 var_2066 = const()[name = string("op_2066"), val = int32(-1)]; fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; tensor var_2068 = mul(x = var_2005, y = const_17_promoted)[name = string("op_2068")]; bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; tensor input_23 = concat(axis = var_2066, interleave = input_23_interleave_0, values = (var_2005, var_2068))[name = string("input_23")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_2063_to_fp16 = const()[name = string("op_2063_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_2063_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_23 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23")]; tensor const_19 = const()[name = string("const_19"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294498432)))]; tensor q_3 = mul(x = normed_23, y = const_19)[name = string("q_3")]; int32 var_2088 = const()[name = string("op_2088"), val = int32(-1)]; fp16 const_20_promoted = const()[name = string("const_20_promoted"), val = fp16(-0x1p+0)]; tensor var_2090 = mul(x = var_2027, y = const_20_promoted)[name = string("op_2090")]; bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; tensor input_25 = concat(axis = var_2088, interleave = input_25_interleave_0, values = (var_2027, var_2090))[name = string("input_25")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_2085_to_fp16 = const()[name = string("op_2085_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_2085_to_fp16, x = input_25)[name = string("normed_25_cast_fp16")]; tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_27 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27")]; tensor const_22 = const()[name = string("const_22"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294498752)))]; tensor k_3 = mul(x = normed_27, y = const_22)[name = string("k_3")]; tensor var_2099 = mul(x = q_3, y = cos_1_cast_fp16)[name = string("op_2099")]; tensor var_2104_begin_0 = const()[name = string("op_2104_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2104_end_0 = const()[name = string("op_2104_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_2104_end_mask_0 = const()[name = string("op_2104_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2104 = slice_by_index(begin = var_2104_begin_0, end = var_2104_end_0, end_mask = var_2104_end_mask_0, x = q_3)[name = string("op_2104")]; fp16 const_23_promoted = const()[name = string("const_23_promoted"), val = fp16(-0x1p+0)]; tensor var_2105 = mul(x = var_2104, y = const_23_promoted)[name = string("op_2105")]; tensor var_2110_begin_0 = const()[name = string("op_2110_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2110_end_0 = const()[name = string("op_2110_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_2110_end_mask_0 = const()[name = string("op_2110_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2110 = slice_by_index(begin = var_2110_begin_0, end = var_2110_end_0, end_mask = var_2110_end_mask_0, x = q_3)[name = string("op_2110")]; int32 var_2112 = const()[name = string("op_2112"), val = int32(-1)]; bool var_2113_interleave_0 = const()[name = string("op_2113_interleave_0"), val = bool(false)]; tensor var_2113 = concat(axis = var_2112, interleave = var_2113_interleave_0, values = (var_2105, var_2110))[name = string("op_2113")]; tensor var_2114 = mul(x = var_2113, y = sin_1_cast_fp16)[name = string("op_2114")]; tensor query_states_5 = add(x = var_2099, y = var_2114)[name = string("query_states_5")]; tensor var_2117 = mul(x = k_3, y = cos_1_cast_fp16)[name = string("op_2117")]; tensor var_2122_begin_0 = const()[name = string("op_2122_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2122_end_0 = const()[name = string("op_2122_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_2122_end_mask_0 = const()[name = string("op_2122_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2122 = slice_by_index(begin = var_2122_begin_0, end = var_2122_end_0, end_mask = var_2122_end_mask_0, x = k_3)[name = string("op_2122")]; fp16 const_24_promoted = const()[name = string("const_24_promoted"), val = fp16(-0x1p+0)]; tensor var_2123 = mul(x = var_2122, y = const_24_promoted)[name = string("op_2123")]; tensor var_2128_begin_0 = const()[name = string("op_2128_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2128_end_0 = const()[name = string("op_2128_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_2128_end_mask_0 = const()[name = string("op_2128_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2128 = slice_by_index(begin = var_2128_begin_0, end = var_2128_end_0, end_mask = var_2128_end_mask_0, x = k_3)[name = string("op_2128")]; int32 var_2130 = const()[name = string("op_2130"), val = int32(-1)]; bool var_2131_interleave_0 = const()[name = string("op_2131_interleave_0"), val = bool(false)]; tensor var_2131 = concat(axis = var_2130, interleave = var_2131_interleave_0, values = (var_2123, var_2128))[name = string("op_2131")]; tensor var_2132 = mul(x = var_2131, y = sin_1_cast_fp16)[name = string("op_2132")]; tensor key_states_5 = add(x = var_2117, y = var_2132)[name = string("key_states_5")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; int32 concat_10_axis_0 = const()[name = string("concat_10_axis_0"), val = int32(0)]; bool concat_10_interleave_0 = const()[name = string("concat_10_interleave_0"), val = bool(false)]; tensor concat_10 = concat(axis = concat_10_axis_0, interleave = concat_10_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_10")]; tensor concat_11_values1_0 = const()[name = string("concat_11_values1_0"), val = tensor([0])]; tensor concat_11_values3_0 = const()[name = string("concat_11_values3_0"), val = tensor([0])]; int32 concat_11_axis_0 = const()[name = string("concat_11_axis_0"), val = int32(0)]; bool concat_11_interleave_0 = const()[name = string("concat_11_interleave_0"), val = bool(false)]; tensor concat_11 = concat(axis = concat_11_axis_0, interleave = concat_11_interleave_0, values = (expand_dims_16, concat_11_values1_0, var_1717, concat_11_values3_0))[name = string("concat_11")]; tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_10, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_11, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_5, x = coreml_update_state_57)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_114_write_state")]; tensor coreml_update_state_58 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_114")]; tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([29])]; tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([30])]; int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_14")]; tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_22, concat_15_values1_0, var_1717, concat_15_values3_0))[name = string("concat_15")]; tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_14, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_15, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = var_2049, x = coreml_update_state_58)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_115_write_state")]; tensor coreml_update_state_59 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_115")]; tensor var_2187_begin_0 = const()[name = string("op_2187_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_2187_end_0 = const()[name = string("op_2187_end_0"), val = tensor([2, 8, 1536, 128])]; tensor var_2187_end_mask_0 = const()[name = string("op_2187_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2187_cast_fp16 = slice_by_index(begin = var_2187_begin_0, end = var_2187_end_0, end_mask = var_2187_end_mask_0, x = coreml_update_state_59)[name = string("op_2187_cast_fp16")]; tensor key_cache_3_axes_0 = const()[name = string("key_cache_3_axes_0"), val = tensor([0])]; tensor key_cache_3_cast_fp16 = squeeze(axes = key_cache_3_axes_0, x = var_2187_cast_fp16)[name = string("key_cache_3_cast_fp16")]; tensor var_2194_begin_0 = const()[name = string("op_2194_begin_0"), val = tensor([29, 0, 0, 0])]; tensor var_2194_end_0 = const()[name = string("op_2194_end_0"), val = tensor([30, 8, 1536, 128])]; tensor var_2194_end_mask_0 = const()[name = string("op_2194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2194_cast_fp16 = slice_by_index(begin = var_2194_begin_0, end = var_2194_end_0, end_mask = var_2194_end_mask_0, x = coreml_update_state_59)[name = string("op_2194_cast_fp16")]; tensor value_cache_3_axes_0 = const()[name = string("value_cache_3_axes_0"), val = tensor([0])]; tensor value_cache_3_cast_fp16 = squeeze(axes = value_cache_3_axes_0, x = var_2194_cast_fp16)[name = string("value_cache_3_cast_fp16")]; tensor var_2218_axes_0 = const()[name = string("op_2218_axes_0"), val = tensor([1])]; tensor var_2218_cast_fp16 = expand_dims(axes = var_2218_axes_0, x = key_cache_3_cast_fp16)[name = string("op_2218_cast_fp16")]; tensor var_2223 = const()[name = string("op_2223"), val = tensor([1, 2, 1, 1])]; tensor value_11_cast_fp16 = tile(reps = var_2223, x = var_2218_cast_fp16)[name = string("value_11_cast_fp16")]; tensor var_2229 = const()[name = string("op_2229"), val = tensor([1, 16, 1536, 128])]; tensor key_states_7_cast_fp16 = reshape(shape = var_2229, x = value_11_cast_fp16)[name = string("key_states_7_cast_fp16")]; tensor var_2232_axes_0 = const()[name = string("op_2232_axes_0"), val = tensor([1])]; tensor var_2232_cast_fp16 = expand_dims(axes = var_2232_axes_0, x = value_cache_3_cast_fp16)[name = string("op_2232_cast_fp16")]; tensor var_2237 = const()[name = string("op_2237"), val = tensor([1, 2, 1, 1])]; tensor value_15_cast_fp16 = tile(reps = var_2237, x = var_2232_cast_fp16)[name = string("value_15_cast_fp16")]; tensor var_2243 = const()[name = string("op_2243"), val = tensor([1, 16, 1536, 128])]; tensor value_states_9_cast_fp16 = reshape(shape = var_2243, x = value_15_cast_fp16)[name = string("value_states_9_cast_fp16")]; bool var_2258_transpose_x_1 = const()[name = string("op_2258_transpose_x_1"), val = bool(false)]; bool var_2258_transpose_y_1 = const()[name = string("op_2258_transpose_y_1"), val = bool(true)]; tensor var_2258 = matmul(transpose_x = var_2258_transpose_x_1, transpose_y = var_2258_transpose_y_1, x = query_states_5, y = key_states_7_cast_fp16)[name = string("op_2258")]; fp16 var_2259_to_fp16 = const()[name = string("op_2259_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_5_cast_fp16 = mul(x = var_2258, y = var_2259_to_fp16)[name = string("attention_5_cast_fp16")]; tensor attention_7_cast_fp16 = add(x = attention_5_cast_fp16, y = causal_mask)[name = string("attention_7_cast_fp16")]; int32 var_2268 = const()[name = string("op_2268"), val = int32(-1)]; tensor probabilities_3_cast_fp16 = softmax(axis = var_2268, x = attention_7_cast_fp16)[name = string("probabilities_3_cast_fp16")]; bool output_7_transpose_x_0 = const()[name = string("output_7_transpose_x_0"), val = bool(false)]; bool output_7_transpose_y_0 = const()[name = string("output_7_transpose_y_0"), val = bool(false)]; tensor output_7_cast_fp16 = matmul(transpose_x = output_7_transpose_x_0, transpose_y = output_7_transpose_y_0, x = probabilities_3_cast_fp16, y = value_states_9_cast_fp16)[name = string("output_7_cast_fp16")]; tensor var_2279_perm_0 = const()[name = string("op_2279_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2285 = const()[name = string("op_2285"), val = tensor([1, 1, 2048])]; tensor var_2279_cast_fp16 = transpose(perm = var_2279_perm_0, x = output_7_cast_fp16)[name = string("transpose_160")]; tensor output_9_cast_fp16 = reshape(shape = var_2285, x = var_2279_cast_fp16)[name = string("output_9_cast_fp16")]; tensor var_2290 = const()[name = string("op_2290"), val = tensor([0, 2, 1])]; string var_2306_pad_type_0 = const()[name = string("op_2306_pad_type_0"), val = string("valid")]; int32 var_2306_groups_0 = const()[name = string("op_2306_groups_0"), val = int32(1)]; tensor var_2306_strides_0 = const()[name = string("op_2306_strides_0"), val = tensor([1])]; tensor var_2306_pad_0 = const()[name = string("op_2306_pad_0"), val = tensor([0, 0])]; tensor var_2306_dilations_0 = const()[name = string("op_2306_dilations_0"), val = tensor([1])]; tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294499072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296072000))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2291_cast_fp16 = transpose(perm = var_2290, x = output_9_cast_fp16)[name = string("transpose_159")]; tensor var_2306_cast_fp16 = conv(dilations = var_2306_dilations_0, groups = var_2306_groups_0, pad = var_2306_pad_0, pad_type = var_2306_pad_type_0, strides = var_2306_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_2291_cast_fp16)[name = string("op_2306_cast_fp16")]; tensor var_2310 = const()[name = string("op_2310"), val = tensor([0, 2, 1])]; tensor attn_output_3_cast_fp16 = transpose(perm = var_2310, x = var_2306_cast_fp16)[name = string("transpose_158")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = attn_output_3_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; int32 var_2325 = const()[name = string("op_2325"), val = int32(-1)]; fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2327_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_2327_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_2325, interleave = input_29_interleave_0, values = (hidden_states_19_cast_fp16, var_2327_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_2322_to_fp16 = const()[name = string("op_2322_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_2322_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 1, 1024])]; tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; tensor const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296088448)))]; tensor x_5_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_27_promoted_to_fp16)[name = string("x_5_cast_fp16")]; tensor var_2347 = const()[name = string("op_2347"), val = tensor([0, 2, 1])]; tensor input_31_axes_0 = const()[name = string("input_31_axes_0"), val = tensor([2])]; tensor var_2348 = transpose(perm = var_2347, x = x_5_cast_fp16)[name = string("transpose_157")]; tensor input_31 = expand_dims(axes = input_31_axes_0, x = var_2348)[name = string("input_31")]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = model_model_layers_1_mlp_gate_proj_weight_palettized, x = input_31)[name = string("input_33")]; string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_1_mlp_up_proj_weight_palettized, x = input_31)[name = string("b_3")]; tensor c_3 = silu(x = input_33)[name = string("c_3")]; tensor input_35 = mul(x = c_3, y = b_3)[name = string("input_35")]; string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_1_mlp_down_proj_weight_palettized, x = input_35)[name = string("e_3")]; tensor var_2370_axes_0 = const()[name = string("op_2370_axes_0"), val = tensor([2])]; tensor var_2370 = squeeze(axes = var_2370_axes_0, x = e_3)[name = string("op_2370")]; tensor var_2371 = const()[name = string("op_2371"), val = tensor([0, 2, 1])]; tensor var_2372 = transpose(perm = var_2371, x = var_2370)[name = string("transpose_156")]; tensor hidden_states_21_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = var_2372)[name = string("hidden_states_21_cast_fp16")]; int32 var_2386 = const()[name = string("op_2386"), val = int32(-1)]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2388_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_2388_cast_fp16")]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37_cast_fp16 = concat(axis = var_2386, interleave = input_37_interleave_0, values = (hidden_states_21_cast_fp16, var_2388_cast_fp16))[name = string("input_37_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_2383_to_fp16 = const()[name = string("op_2383_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_2383_to_fp16, x = input_37_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 1, 1024])]; tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; tensor const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296090560)))]; tensor hidden_states_23_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_30_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor var_2400 = const()[name = string("op_2400"), val = tensor([0, 2, 1])]; tensor var_2403_axes_0 = const()[name = string("op_2403_axes_0"), val = tensor([2])]; tensor var_2401_cast_fp16 = transpose(perm = var_2400, x = hidden_states_23_cast_fp16)[name = string("transpose_155")]; tensor var_2403_cast_fp16 = expand_dims(axes = var_2403_axes_0, x = var_2401_cast_fp16)[name = string("op_2403_cast_fp16")]; string var_2419_pad_type_0 = const()[name = string("op_2419_pad_type_0"), val = string("valid")]; tensor var_2419_strides_0 = const()[name = string("op_2419_strides_0"), val = tensor([1, 1])]; tensor var_2419_pad_0 = const()[name = string("op_2419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2419_dilations_0 = const()[name = string("op_2419_dilations_0"), val = tensor([1, 1])]; int32 var_2419_groups_0 = const()[name = string("op_2419_groups_0"), val = int32(1)]; tensor var_2419 = conv(dilations = var_2419_dilations_0, groups = var_2419_groups_0, pad = var_2419_pad_0, pad_type = var_2419_pad_type_0, strides = var_2419_strides_0, weight = model_model_layers_2_self_attn_q_proj_weight_palettized, x = var_2403_cast_fp16)[name = string("op_2419")]; tensor var_2424 = const()[name = string("op_2424"), val = tensor([1, 16, 1, 128])]; tensor var_2425 = reshape(shape = var_2424, x = var_2419)[name = string("op_2425")]; string var_2441_pad_type_0 = const()[name = string("op_2441_pad_type_0"), val = string("valid")]; tensor var_2441_strides_0 = const()[name = string("op_2441_strides_0"), val = tensor([1, 1])]; tensor var_2441_pad_0 = const()[name = string("op_2441_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2441_dilations_0 = const()[name = string("op_2441_dilations_0"), val = tensor([1, 1])]; int32 var_2441_groups_0 = const()[name = string("op_2441_groups_0"), val = int32(1)]; tensor var_2441 = conv(dilations = var_2441_dilations_0, groups = var_2441_groups_0, pad = var_2441_pad_0, pad_type = var_2441_pad_type_0, strides = var_2441_strides_0, weight = model_model_layers_2_self_attn_k_proj_weight_palettized, x = var_2403_cast_fp16)[name = string("op_2441")]; tensor var_2446 = const()[name = string("op_2446"), val = tensor([1, 8, 1, 128])]; tensor var_2447 = reshape(shape = var_2446, x = var_2441)[name = string("op_2447")]; string var_2463_pad_type_0 = const()[name = string("op_2463_pad_type_0"), val = string("valid")]; tensor var_2463_strides_0 = const()[name = string("op_2463_strides_0"), val = tensor([1, 1])]; tensor var_2463_pad_0 = const()[name = string("op_2463_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2463_dilations_0 = const()[name = string("op_2463_dilations_0"), val = tensor([1, 1])]; int32 var_2463_groups_0 = const()[name = string("op_2463_groups_0"), val = int32(1)]; tensor var_2463 = conv(dilations = var_2463_dilations_0, groups = var_2463_groups_0, pad = var_2463_pad_0, pad_type = var_2463_pad_type_0, strides = var_2463_strides_0, weight = model_model_layers_2_self_attn_v_proj_weight_palettized, x = var_2403_cast_fp16)[name = string("op_2463")]; tensor var_2468 = const()[name = string("op_2468"), val = tensor([1, 8, 1, 128])]; tensor var_2469 = reshape(shape = var_2468, x = var_2463)[name = string("op_2469")]; int32 var_2486 = const()[name = string("op_2486"), val = int32(-1)]; fp16 const_31_promoted = const()[name = string("const_31_promoted"), val = fp16(-0x1p+0)]; tensor var_2488 = mul(x = var_2425, y = const_31_promoted)[name = string("op_2488")]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41 = concat(axis = var_2486, interleave = input_41_interleave_0, values = (var_2425, var_2488))[name = string("input_41")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_2483_to_fp16 = const()[name = string("op_2483_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_2483_to_fp16, x = input_41)[name = string("normed_37_cast_fp16")]; tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_39 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39")]; tensor const_33 = const()[name = string("const_33"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296092672)))]; tensor q_5 = mul(x = normed_39, y = const_33)[name = string("q_5")]; int32 var_2508 = const()[name = string("op_2508"), val = int32(-1)]; fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; tensor var_2510 = mul(x = var_2447, y = const_34_promoted)[name = string("op_2510")]; bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; tensor input_43 = concat(axis = var_2508, interleave = input_43_interleave_0, values = (var_2447, var_2510))[name = string("input_43")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_2505_to_fp16 = const()[name = string("op_2505_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2505_to_fp16, x = input_43)[name = string("normed_41_cast_fp16")]; tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_43 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43")]; tensor const_36 = const()[name = string("const_36"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296092992)))]; tensor k_5 = mul(x = normed_43, y = const_36)[name = string("k_5")]; tensor var_2519 = mul(x = q_5, y = cos_1_cast_fp16)[name = string("op_2519")]; tensor var_2524_begin_0 = const()[name = string("op_2524_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2524_end_0 = const()[name = string("op_2524_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_2524_end_mask_0 = const()[name = string("op_2524_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2524 = slice_by_index(begin = var_2524_begin_0, end = var_2524_end_0, end_mask = var_2524_end_mask_0, x = q_5)[name = string("op_2524")]; fp16 const_37_promoted = const()[name = string("const_37_promoted"), val = fp16(-0x1p+0)]; tensor var_2525 = mul(x = var_2524, y = const_37_promoted)[name = string("op_2525")]; tensor var_2530_begin_0 = const()[name = string("op_2530_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2530_end_0 = const()[name = string("op_2530_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_2530_end_mask_0 = const()[name = string("op_2530_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2530 = slice_by_index(begin = var_2530_begin_0, end = var_2530_end_0, end_mask = var_2530_end_mask_0, x = q_5)[name = string("op_2530")]; int32 var_2532 = const()[name = string("op_2532"), val = int32(-1)]; bool var_2533_interleave_0 = const()[name = string("op_2533_interleave_0"), val = bool(false)]; tensor var_2533 = concat(axis = var_2532, interleave = var_2533_interleave_0, values = (var_2525, var_2530))[name = string("op_2533")]; tensor var_2534 = mul(x = var_2533, y = sin_1_cast_fp16)[name = string("op_2534")]; tensor query_states_9 = add(x = var_2519, y = var_2534)[name = string("query_states_9")]; tensor var_2537 = mul(x = k_5, y = cos_1_cast_fp16)[name = string("op_2537")]; tensor var_2542_begin_0 = const()[name = string("op_2542_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2542_end_0 = const()[name = string("op_2542_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_2542_end_mask_0 = const()[name = string("op_2542_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2542 = slice_by_index(begin = var_2542_begin_0, end = var_2542_end_0, end_mask = var_2542_end_mask_0, x = k_5)[name = string("op_2542")]; fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; tensor var_2543 = mul(x = var_2542, y = const_38_promoted)[name = string("op_2543")]; tensor var_2548_begin_0 = const()[name = string("op_2548_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2548_end_0 = const()[name = string("op_2548_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_2548_end_mask_0 = const()[name = string("op_2548_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2548 = slice_by_index(begin = var_2548_begin_0, end = var_2548_end_0, end_mask = var_2548_end_mask_0, x = k_5)[name = string("op_2548")]; int32 var_2550 = const()[name = string("op_2550"), val = int32(-1)]; bool var_2551_interleave_0 = const()[name = string("op_2551_interleave_0"), val = bool(false)]; tensor var_2551 = concat(axis = var_2550, interleave = var_2551_interleave_0, values = (var_2543, var_2548))[name = string("op_2551")]; tensor var_2552 = mul(x = var_2551, y = sin_1_cast_fp16)[name = string("op_2552")]; tensor key_states_9 = add(x = var_2537, y = var_2552)[name = string("key_states_9")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_18")]; tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_28, concat_19_values1_0, var_1717, concat_19_values3_0))[name = string("concat_19")]; tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_18, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_19, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_9, x = coreml_update_state_59)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_116_write_state")]; tensor coreml_update_state_60 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_116")]; tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([30])]; tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([31])]; int32 concat_22_axis_0 = const()[name = string("concat_22_axis_0"), val = int32(0)]; bool concat_22_interleave_0 = const()[name = string("concat_22_interleave_0"), val = bool(false)]; tensor concat_22 = concat(axis = concat_22_axis_0, interleave = concat_22_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_22")]; tensor concat_23_values1_0 = const()[name = string("concat_23_values1_0"), val = tensor([0])]; tensor concat_23_values3_0 = const()[name = string("concat_23_values3_0"), val = tensor([0])]; int32 concat_23_axis_0 = const()[name = string("concat_23_axis_0"), val = int32(0)]; bool concat_23_interleave_0 = const()[name = string("concat_23_interleave_0"), val = bool(false)]; tensor concat_23 = concat(axis = concat_23_axis_0, interleave = concat_23_interleave_0, values = (expand_dims_34, concat_23_values1_0, var_1717, concat_23_values3_0))[name = string("concat_23")]; tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_22, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_23, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = var_2469, x = coreml_update_state_60)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_117_write_state")]; tensor coreml_update_state_61 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_117")]; tensor var_2607_begin_0 = const()[name = string("op_2607_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_2607_end_0 = const()[name = string("op_2607_end_0"), val = tensor([3, 8, 1536, 128])]; tensor var_2607_end_mask_0 = const()[name = string("op_2607_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2607_cast_fp16 = slice_by_index(begin = var_2607_begin_0, end = var_2607_end_0, end_mask = var_2607_end_mask_0, x = coreml_update_state_61)[name = string("op_2607_cast_fp16")]; tensor key_cache_5_axes_0 = const()[name = string("key_cache_5_axes_0"), val = tensor([0])]; tensor key_cache_5_cast_fp16 = squeeze(axes = key_cache_5_axes_0, x = var_2607_cast_fp16)[name = string("key_cache_5_cast_fp16")]; tensor var_2614_begin_0 = const()[name = string("op_2614_begin_0"), val = tensor([30, 0, 0, 0])]; tensor var_2614_end_0 = const()[name = string("op_2614_end_0"), val = tensor([31, 8, 1536, 128])]; tensor var_2614_end_mask_0 = const()[name = string("op_2614_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2614_cast_fp16 = slice_by_index(begin = var_2614_begin_0, end = var_2614_end_0, end_mask = var_2614_end_mask_0, x = coreml_update_state_61)[name = string("op_2614_cast_fp16")]; tensor value_cache_5_axes_0 = const()[name = string("value_cache_5_axes_0"), val = tensor([0])]; tensor value_cache_5_cast_fp16 = squeeze(axes = value_cache_5_axes_0, x = var_2614_cast_fp16)[name = string("value_cache_5_cast_fp16")]; tensor var_2638_axes_0 = const()[name = string("op_2638_axes_0"), val = tensor([1])]; tensor var_2638_cast_fp16 = expand_dims(axes = var_2638_axes_0, x = key_cache_5_cast_fp16)[name = string("op_2638_cast_fp16")]; tensor var_2643 = const()[name = string("op_2643"), val = tensor([1, 2, 1, 1])]; tensor value_19_cast_fp16 = tile(reps = var_2643, x = var_2638_cast_fp16)[name = string("value_19_cast_fp16")]; tensor var_2649 = const()[name = string("op_2649"), val = tensor([1, 16, 1536, 128])]; tensor key_states_11_cast_fp16 = reshape(shape = var_2649, x = value_19_cast_fp16)[name = string("key_states_11_cast_fp16")]; tensor var_2652_axes_0 = const()[name = string("op_2652_axes_0"), val = tensor([1])]; tensor var_2652_cast_fp16 = expand_dims(axes = var_2652_axes_0, x = value_cache_5_cast_fp16)[name = string("op_2652_cast_fp16")]; tensor var_2657 = const()[name = string("op_2657"), val = tensor([1, 2, 1, 1])]; tensor value_23_cast_fp16 = tile(reps = var_2657, x = var_2652_cast_fp16)[name = string("value_23_cast_fp16")]; tensor var_2663 = const()[name = string("op_2663"), val = tensor([1, 16, 1536, 128])]; tensor value_states_15_cast_fp16 = reshape(shape = var_2663, x = value_23_cast_fp16)[name = string("value_states_15_cast_fp16")]; bool var_2678_transpose_x_1 = const()[name = string("op_2678_transpose_x_1"), val = bool(false)]; bool var_2678_transpose_y_1 = const()[name = string("op_2678_transpose_y_1"), val = bool(true)]; tensor var_2678 = matmul(transpose_x = var_2678_transpose_x_1, transpose_y = var_2678_transpose_y_1, x = query_states_9, y = key_states_11_cast_fp16)[name = string("op_2678")]; fp16 var_2679_to_fp16 = const()[name = string("op_2679_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_9_cast_fp16 = mul(x = var_2678, y = var_2679_to_fp16)[name = string("attention_9_cast_fp16")]; tensor attention_11_cast_fp16 = add(x = attention_9_cast_fp16, y = causal_mask)[name = string("attention_11_cast_fp16")]; int32 var_2688 = const()[name = string("op_2688"), val = int32(-1)]; tensor probabilities_5_cast_fp16 = softmax(axis = var_2688, x = attention_11_cast_fp16)[name = string("probabilities_5_cast_fp16")]; bool output_13_transpose_x_0 = const()[name = string("output_13_transpose_x_0"), val = bool(false)]; bool output_13_transpose_y_0 = const()[name = string("output_13_transpose_y_0"), val = bool(false)]; tensor output_13_cast_fp16 = matmul(transpose_x = output_13_transpose_x_0, transpose_y = output_13_transpose_y_0, x = probabilities_5_cast_fp16, y = value_states_15_cast_fp16)[name = string("output_13_cast_fp16")]; tensor var_2699_perm_0 = const()[name = string("op_2699_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2705 = const()[name = string("op_2705"), val = tensor([1, 1, 2048])]; tensor var_2699_cast_fp16 = transpose(perm = var_2699_perm_0, x = output_13_cast_fp16)[name = string("transpose_154")]; tensor output_15_cast_fp16 = reshape(shape = var_2705, x = var_2699_cast_fp16)[name = string("output_15_cast_fp16")]; tensor var_2710 = const()[name = string("op_2710"), val = tensor([0, 2, 1])]; string var_2726_pad_type_0 = const()[name = string("op_2726_pad_type_0"), val = string("valid")]; int32 var_2726_groups_0 = const()[name = string("op_2726_groups_0"), val = int32(1)]; tensor var_2726_strides_0 = const()[name = string("op_2726_strides_0"), val = tensor([1])]; tensor var_2726_pad_0 = const()[name = string("op_2726_pad_0"), val = tensor([0, 0])]; tensor var_2726_dilations_0 = const()[name = string("op_2726_dilations_0"), val = tensor([1])]; tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296093312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297666240))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2711_cast_fp16 = transpose(perm = var_2710, x = output_15_cast_fp16)[name = string("transpose_153")]; tensor var_2726_cast_fp16 = conv(dilations = var_2726_dilations_0, groups = var_2726_groups_0, pad = var_2726_pad_0, pad_type = var_2726_pad_type_0, strides = var_2726_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2711_cast_fp16)[name = string("op_2726_cast_fp16")]; tensor var_2730 = const()[name = string("op_2730"), val = tensor([0, 2, 1])]; tensor attn_output_5_cast_fp16 = transpose(perm = var_2730, x = var_2726_cast_fp16)[name = string("transpose_152")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = attn_output_5_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; int32 var_2745 = const()[name = string("op_2745"), val = int32(-1)]; fp16 const_39_promoted_to_fp16 = const()[name = string("const_39_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2747_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_39_promoted_to_fp16)[name = string("op_2747_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_2745, interleave = input_47_interleave_0, values = (hidden_states_29_cast_fp16, var_2747_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_2742_to_fp16 = const()[name = string("op_2742_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2742_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 1, 1024])]; tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; tensor const_41_promoted_to_fp16 = const()[name = string("const_41_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297682688)))]; tensor x_9_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_41_promoted_to_fp16)[name = string("x_9_cast_fp16")]; tensor var_2767 = const()[name = string("op_2767"), val = tensor([0, 2, 1])]; tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; tensor var_2768 = transpose(perm = var_2767, x = x_9_cast_fp16)[name = string("transpose_151")]; tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2768)[name = string("input_49")]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = model_model_layers_2_mlp_gate_proj_weight_palettized, x = input_49)[name = string("input_51")]; string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_2_mlp_up_proj_weight_palettized, x = input_49)[name = string("b_5")]; tensor c_5 = silu(x = input_51)[name = string("c_5")]; tensor input_53 = mul(x = c_5, y = b_5)[name = string("input_53")]; string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_2_mlp_down_proj_weight_palettized, x = input_53)[name = string("e_5")]; tensor var_2790_axes_0 = const()[name = string("op_2790_axes_0"), val = tensor([2])]; tensor var_2790 = squeeze(axes = var_2790_axes_0, x = e_5)[name = string("op_2790")]; tensor var_2791 = const()[name = string("op_2791"), val = tensor([0, 2, 1])]; tensor var_2792 = transpose(perm = var_2791, x = var_2790)[name = string("transpose_150")]; tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2792)[name = string("hidden_states_31_cast_fp16")]; int32 var_2806 = const()[name = string("op_2806"), val = int32(-1)]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2808_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_42_promoted_to_fp16)[name = string("op_2808_cast_fp16")]; bool input_55_interleave_0 = const()[name = string("input_55_interleave_0"), val = bool(false)]; tensor input_55_cast_fp16 = concat(axis = var_2806, interleave = input_55_interleave_0, values = (hidden_states_31_cast_fp16, var_2808_cast_fp16))[name = string("input_55_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_2803_to_fp16 = const()[name = string("op_2803_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2803_to_fp16, x = input_55_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 1, 1024])]; tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; tensor const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297684800)))]; tensor hidden_states_33_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_44_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor var_2820 = const()[name = string("op_2820"), val = tensor([0, 2, 1])]; tensor var_2823_axes_0 = const()[name = string("op_2823_axes_0"), val = tensor([2])]; tensor var_2821_cast_fp16 = transpose(perm = var_2820, x = hidden_states_33_cast_fp16)[name = string("transpose_149")]; tensor var_2823_cast_fp16 = expand_dims(axes = var_2823_axes_0, x = var_2821_cast_fp16)[name = string("op_2823_cast_fp16")]; string var_2839_pad_type_0 = const()[name = string("op_2839_pad_type_0"), val = string("valid")]; tensor var_2839_strides_0 = const()[name = string("op_2839_strides_0"), val = tensor([1, 1])]; tensor var_2839_pad_0 = const()[name = string("op_2839_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2839_dilations_0 = const()[name = string("op_2839_dilations_0"), val = tensor([1, 1])]; int32 var_2839_groups_0 = const()[name = string("op_2839_groups_0"), val = int32(1)]; tensor var_2839 = conv(dilations = var_2839_dilations_0, groups = var_2839_groups_0, pad = var_2839_pad_0, pad_type = var_2839_pad_type_0, strides = var_2839_strides_0, weight = model_model_layers_3_self_attn_q_proj_weight_palettized, x = var_2823_cast_fp16)[name = string("op_2839")]; tensor var_2844 = const()[name = string("op_2844"), val = tensor([1, 16, 1, 128])]; tensor var_2845 = reshape(shape = var_2844, x = var_2839)[name = string("op_2845")]; string var_2861_pad_type_0 = const()[name = string("op_2861_pad_type_0"), val = string("valid")]; tensor var_2861_strides_0 = const()[name = string("op_2861_strides_0"), val = tensor([1, 1])]; tensor var_2861_pad_0 = const()[name = string("op_2861_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2861_dilations_0 = const()[name = string("op_2861_dilations_0"), val = tensor([1, 1])]; int32 var_2861_groups_0 = const()[name = string("op_2861_groups_0"), val = int32(1)]; tensor var_2861 = conv(dilations = var_2861_dilations_0, groups = var_2861_groups_0, pad = var_2861_pad_0, pad_type = var_2861_pad_type_0, strides = var_2861_strides_0, weight = model_model_layers_3_self_attn_k_proj_weight_palettized, x = var_2823_cast_fp16)[name = string("op_2861")]; tensor var_2866 = const()[name = string("op_2866"), val = tensor([1, 8, 1, 128])]; tensor var_2867 = reshape(shape = var_2866, x = var_2861)[name = string("op_2867")]; string var_2883_pad_type_0 = const()[name = string("op_2883_pad_type_0"), val = string("valid")]; tensor var_2883_strides_0 = const()[name = string("op_2883_strides_0"), val = tensor([1, 1])]; tensor var_2883_pad_0 = const()[name = string("op_2883_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2883_dilations_0 = const()[name = string("op_2883_dilations_0"), val = tensor([1, 1])]; int32 var_2883_groups_0 = const()[name = string("op_2883_groups_0"), val = int32(1)]; tensor var_2883 = conv(dilations = var_2883_dilations_0, groups = var_2883_groups_0, pad = var_2883_pad_0, pad_type = var_2883_pad_type_0, strides = var_2883_strides_0, weight = model_model_layers_3_self_attn_v_proj_weight_palettized, x = var_2823_cast_fp16)[name = string("op_2883")]; tensor var_2888 = const()[name = string("op_2888"), val = tensor([1, 8, 1, 128])]; tensor var_2889 = reshape(shape = var_2888, x = var_2883)[name = string("op_2889")]; int32 var_2906 = const()[name = string("op_2906"), val = int32(-1)]; fp16 const_45_promoted = const()[name = string("const_45_promoted"), val = fp16(-0x1p+0)]; tensor var_2908 = mul(x = var_2845, y = const_45_promoted)[name = string("op_2908")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59 = concat(axis = var_2906, interleave = input_59_interleave_0, values = (var_2845, var_2908))[name = string("input_59")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_2903_to_fp16 = const()[name = string("op_2903_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_2903_to_fp16, x = input_59)[name = string("normed_53_cast_fp16")]; tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_55 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55")]; tensor const_47 = const()[name = string("const_47"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297686912)))]; tensor q_7 = mul(x = normed_55, y = const_47)[name = string("q_7")]; int32 var_2928 = const()[name = string("op_2928"), val = int32(-1)]; fp16 const_48_promoted = const()[name = string("const_48_promoted"), val = fp16(-0x1p+0)]; tensor var_2930 = mul(x = var_2867, y = const_48_promoted)[name = string("op_2930")]; bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; tensor input_61 = concat(axis = var_2928, interleave = input_61_interleave_0, values = (var_2867, var_2930))[name = string("input_61")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_2925_to_fp16 = const()[name = string("op_2925_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_2925_to_fp16, x = input_61)[name = string("normed_57_cast_fp16")]; tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_59 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59")]; tensor const_50 = const()[name = string("const_50"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297687232)))]; tensor k_7 = mul(x = normed_59, y = const_50)[name = string("k_7")]; tensor var_2939 = mul(x = q_7, y = cos_1_cast_fp16)[name = string("op_2939")]; tensor var_2944_begin_0 = const()[name = string("op_2944_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2944_end_0 = const()[name = string("op_2944_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_2944_end_mask_0 = const()[name = string("op_2944_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2944 = slice_by_index(begin = var_2944_begin_0, end = var_2944_end_0, end_mask = var_2944_end_mask_0, x = q_7)[name = string("op_2944")]; fp16 const_51_promoted = const()[name = string("const_51_promoted"), val = fp16(-0x1p+0)]; tensor var_2945 = mul(x = var_2944, y = const_51_promoted)[name = string("op_2945")]; tensor var_2950_begin_0 = const()[name = string("op_2950_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2950_end_0 = const()[name = string("op_2950_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_2950_end_mask_0 = const()[name = string("op_2950_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2950 = slice_by_index(begin = var_2950_begin_0, end = var_2950_end_0, end_mask = var_2950_end_mask_0, x = q_7)[name = string("op_2950")]; int32 var_2952 = const()[name = string("op_2952"), val = int32(-1)]; bool var_2953_interleave_0 = const()[name = string("op_2953_interleave_0"), val = bool(false)]; tensor var_2953 = concat(axis = var_2952, interleave = var_2953_interleave_0, values = (var_2945, var_2950))[name = string("op_2953")]; tensor var_2954 = mul(x = var_2953, y = sin_1_cast_fp16)[name = string("op_2954")]; tensor query_states_13 = add(x = var_2939, y = var_2954)[name = string("query_states_13")]; tensor var_2957 = mul(x = k_7, y = cos_1_cast_fp16)[name = string("op_2957")]; tensor var_2962_begin_0 = const()[name = string("op_2962_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2962_end_0 = const()[name = string("op_2962_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_2962_end_mask_0 = const()[name = string("op_2962_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2962 = slice_by_index(begin = var_2962_begin_0, end = var_2962_end_0, end_mask = var_2962_end_mask_0, x = k_7)[name = string("op_2962")]; fp16 const_52_promoted = const()[name = string("const_52_promoted"), val = fp16(-0x1p+0)]; tensor var_2963 = mul(x = var_2962, y = const_52_promoted)[name = string("op_2963")]; tensor var_2968_begin_0 = const()[name = string("op_2968_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2968_end_0 = const()[name = string("op_2968_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_2968_end_mask_0 = const()[name = string("op_2968_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2968 = slice_by_index(begin = var_2968_begin_0, end = var_2968_end_0, end_mask = var_2968_end_mask_0, x = k_7)[name = string("op_2968")]; int32 var_2970 = const()[name = string("op_2970"), val = int32(-1)]; bool var_2971_interleave_0 = const()[name = string("op_2971_interleave_0"), val = bool(false)]; tensor var_2971 = concat(axis = var_2970, interleave = var_2971_interleave_0, values = (var_2963, var_2968))[name = string("op_2971")]; tensor var_2972 = mul(x = var_2971, y = sin_1_cast_fp16)[name = string("op_2972")]; tensor key_states_13 = add(x = var_2957, y = var_2972)[name = string("key_states_13")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_26")]; tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_40, concat_27_values1_0, var_1717, concat_27_values3_0))[name = string("concat_27")]; tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_26, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_27, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_13, x = coreml_update_state_61)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_118_write_state")]; tensor coreml_update_state_62 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_118")]; tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([31])]; tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([32])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_30")]; tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_46, concat_31_values1_0, var_1717, concat_31_values3_0))[name = string("concat_31")]; tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_30, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_31, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = var_2889, x = coreml_update_state_62)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_119_write_state")]; tensor coreml_update_state_63 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_119")]; tensor var_3027_begin_0 = const()[name = string("op_3027_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_3027_end_0 = const()[name = string("op_3027_end_0"), val = tensor([4, 8, 1536, 128])]; tensor var_3027_end_mask_0 = const()[name = string("op_3027_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3027_cast_fp16 = slice_by_index(begin = var_3027_begin_0, end = var_3027_end_0, end_mask = var_3027_end_mask_0, x = coreml_update_state_63)[name = string("op_3027_cast_fp16")]; tensor key_cache_7_axes_0 = const()[name = string("key_cache_7_axes_0"), val = tensor([0])]; tensor key_cache_7_cast_fp16 = squeeze(axes = key_cache_7_axes_0, x = var_3027_cast_fp16)[name = string("key_cache_7_cast_fp16")]; tensor var_3034_begin_0 = const()[name = string("op_3034_begin_0"), val = tensor([31, 0, 0, 0])]; tensor var_3034_end_0 = const()[name = string("op_3034_end_0"), val = tensor([32, 8, 1536, 128])]; tensor var_3034_end_mask_0 = const()[name = string("op_3034_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3034_cast_fp16 = slice_by_index(begin = var_3034_begin_0, end = var_3034_end_0, end_mask = var_3034_end_mask_0, x = coreml_update_state_63)[name = string("op_3034_cast_fp16")]; tensor value_cache_7_axes_0 = const()[name = string("value_cache_7_axes_0"), val = tensor([0])]; tensor value_cache_7_cast_fp16 = squeeze(axes = value_cache_7_axes_0, x = var_3034_cast_fp16)[name = string("value_cache_7_cast_fp16")]; tensor var_3058_axes_0 = const()[name = string("op_3058_axes_0"), val = tensor([1])]; tensor var_3058_cast_fp16 = expand_dims(axes = var_3058_axes_0, x = key_cache_7_cast_fp16)[name = string("op_3058_cast_fp16")]; tensor var_3063 = const()[name = string("op_3063"), val = tensor([1, 2, 1, 1])]; tensor value_27_cast_fp16 = tile(reps = var_3063, x = var_3058_cast_fp16)[name = string("value_27_cast_fp16")]; tensor var_3069 = const()[name = string("op_3069"), val = tensor([1, 16, 1536, 128])]; tensor key_states_15_cast_fp16 = reshape(shape = var_3069, x = value_27_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor var_3072_axes_0 = const()[name = string("op_3072_axes_0"), val = tensor([1])]; tensor var_3072_cast_fp16 = expand_dims(axes = var_3072_axes_0, x = value_cache_7_cast_fp16)[name = string("op_3072_cast_fp16")]; tensor var_3077 = const()[name = string("op_3077"), val = tensor([1, 2, 1, 1])]; tensor value_31_cast_fp16 = tile(reps = var_3077, x = var_3072_cast_fp16)[name = string("value_31_cast_fp16")]; tensor var_3083 = const()[name = string("op_3083"), val = tensor([1, 16, 1536, 128])]; tensor value_states_21_cast_fp16 = reshape(shape = var_3083, x = value_31_cast_fp16)[name = string("value_states_21_cast_fp16")]; bool var_3098_transpose_x_1 = const()[name = string("op_3098_transpose_x_1"), val = bool(false)]; bool var_3098_transpose_y_1 = const()[name = string("op_3098_transpose_y_1"), val = bool(true)]; tensor var_3098 = matmul(transpose_x = var_3098_transpose_x_1, transpose_y = var_3098_transpose_y_1, x = query_states_13, y = key_states_15_cast_fp16)[name = string("op_3098")]; fp16 var_3099_to_fp16 = const()[name = string("op_3099_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_13_cast_fp16 = mul(x = var_3098, y = var_3099_to_fp16)[name = string("attention_13_cast_fp16")]; tensor attention_15_cast_fp16 = add(x = attention_13_cast_fp16, y = causal_mask)[name = string("attention_15_cast_fp16")]; int32 var_3108 = const()[name = string("op_3108"), val = int32(-1)]; tensor probabilities_7_cast_fp16 = softmax(axis = var_3108, x = attention_15_cast_fp16)[name = string("probabilities_7_cast_fp16")]; bool output_19_transpose_x_0 = const()[name = string("output_19_transpose_x_0"), val = bool(false)]; bool output_19_transpose_y_0 = const()[name = string("output_19_transpose_y_0"), val = bool(false)]; tensor output_19_cast_fp16 = matmul(transpose_x = output_19_transpose_x_0, transpose_y = output_19_transpose_y_0, x = probabilities_7_cast_fp16, y = value_states_21_cast_fp16)[name = string("output_19_cast_fp16")]; tensor var_3119_perm_0 = const()[name = string("op_3119_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3125 = const()[name = string("op_3125"), val = tensor([1, 1, 2048])]; tensor var_3119_cast_fp16 = transpose(perm = var_3119_perm_0, x = output_19_cast_fp16)[name = string("transpose_148")]; tensor output_21_cast_fp16 = reshape(shape = var_3125, x = var_3119_cast_fp16)[name = string("output_21_cast_fp16")]; tensor var_3130 = const()[name = string("op_3130"), val = tensor([0, 2, 1])]; string var_3146_pad_type_0 = const()[name = string("op_3146_pad_type_0"), val = string("valid")]; int32 var_3146_groups_0 = const()[name = string("op_3146_groups_0"), val = int32(1)]; tensor var_3146_strides_0 = const()[name = string("op_3146_strides_0"), val = tensor([1])]; tensor var_3146_pad_0 = const()[name = string("op_3146_pad_0"), val = tensor([0, 0])]; tensor var_3146_dilations_0 = const()[name = string("op_3146_dilations_0"), val = tensor([1])]; tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297687552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299260480))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3131_cast_fp16 = transpose(perm = var_3130, x = output_21_cast_fp16)[name = string("transpose_147")]; tensor var_3146_cast_fp16 = conv(dilations = var_3146_dilations_0, groups = var_3146_groups_0, pad = var_3146_pad_0, pad_type = var_3146_pad_type_0, strides = var_3146_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_3131_cast_fp16)[name = string("op_3146_cast_fp16")]; tensor var_3150 = const()[name = string("op_3150"), val = tensor([0, 2, 1])]; tensor attn_output_7_cast_fp16 = transpose(perm = var_3150, x = var_3146_cast_fp16)[name = string("transpose_146")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_7_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; int32 var_3165 = const()[name = string("op_3165"), val = int32(-1)]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3167_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_53_promoted_to_fp16)[name = string("op_3167_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_3165, interleave = input_65_interleave_0, values = (hidden_states_39_cast_fp16, var_3167_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_3162_to_fp16 = const()[name = string("op_3162_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_3162_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 1, 1024])]; tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; tensor const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299276928)))]; tensor x_13_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_55_promoted_to_fp16)[name = string("x_13_cast_fp16")]; tensor var_3187 = const()[name = string("op_3187"), val = tensor([0, 2, 1])]; tensor input_67_axes_0 = const()[name = string("input_67_axes_0"), val = tensor([2])]; tensor var_3188 = transpose(perm = var_3187, x = x_13_cast_fp16)[name = string("transpose_145")]; tensor input_67 = expand_dims(axes = input_67_axes_0, x = var_3188)[name = string("input_67")]; string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; tensor input_69 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = model_model_layers_3_mlp_gate_proj_weight_palettized, x = input_67)[name = string("input_69")]; string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_3_mlp_up_proj_weight_palettized, x = input_67)[name = string("b_7")]; tensor c_7 = silu(x = input_69)[name = string("c_7")]; tensor input_71 = mul(x = c_7, y = b_7)[name = string("input_71")]; string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_3_mlp_down_proj_weight_palettized, x = input_71)[name = string("e_7")]; tensor var_3210_axes_0 = const()[name = string("op_3210_axes_0"), val = tensor([2])]; tensor var_3210 = squeeze(axes = var_3210_axes_0, x = e_7)[name = string("op_3210")]; tensor var_3211 = const()[name = string("op_3211"), val = tensor([0, 2, 1])]; tensor var_3212 = transpose(perm = var_3211, x = var_3210)[name = string("transpose_144")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = var_3212)[name = string("hidden_states_41_cast_fp16")]; int32 var_3226 = const()[name = string("op_3226"), val = int32(-1)]; fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3228_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_3228_cast_fp16")]; bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; tensor input_73_cast_fp16 = concat(axis = var_3226, interleave = input_73_interleave_0, values = (hidden_states_41_cast_fp16, var_3228_cast_fp16))[name = string("input_73_cast_fp16")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_3223_to_fp16 = const()[name = string("op_3223_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_3223_to_fp16, x = input_73_cast_fp16)[name = string("normed_65_cast_fp16")]; tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 1, 1024])]; tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; tensor const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299279040)))]; tensor hidden_states_43_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_58_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor var_3240 = const()[name = string("op_3240"), val = tensor([0, 2, 1])]; tensor var_3243_axes_0 = const()[name = string("op_3243_axes_0"), val = tensor([2])]; tensor var_3241_cast_fp16 = transpose(perm = var_3240, x = hidden_states_43_cast_fp16)[name = string("transpose_143")]; tensor var_3243_cast_fp16 = expand_dims(axes = var_3243_axes_0, x = var_3241_cast_fp16)[name = string("op_3243_cast_fp16")]; string var_3259_pad_type_0 = const()[name = string("op_3259_pad_type_0"), val = string("valid")]; tensor var_3259_strides_0 = const()[name = string("op_3259_strides_0"), val = tensor([1, 1])]; tensor var_3259_pad_0 = const()[name = string("op_3259_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3259_dilations_0 = const()[name = string("op_3259_dilations_0"), val = tensor([1, 1])]; int32 var_3259_groups_0 = const()[name = string("op_3259_groups_0"), val = int32(1)]; tensor var_3259 = conv(dilations = var_3259_dilations_0, groups = var_3259_groups_0, pad = var_3259_pad_0, pad_type = var_3259_pad_type_0, strides = var_3259_strides_0, weight = model_model_layers_4_self_attn_q_proj_weight_palettized, x = var_3243_cast_fp16)[name = string("op_3259")]; tensor var_3264 = const()[name = string("op_3264"), val = tensor([1, 16, 1, 128])]; tensor var_3265 = reshape(shape = var_3264, x = var_3259)[name = string("op_3265")]; string var_3281_pad_type_0 = const()[name = string("op_3281_pad_type_0"), val = string("valid")]; tensor var_3281_strides_0 = const()[name = string("op_3281_strides_0"), val = tensor([1, 1])]; tensor var_3281_pad_0 = const()[name = string("op_3281_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3281_dilations_0 = const()[name = string("op_3281_dilations_0"), val = tensor([1, 1])]; int32 var_3281_groups_0 = const()[name = string("op_3281_groups_0"), val = int32(1)]; tensor var_3281 = conv(dilations = var_3281_dilations_0, groups = var_3281_groups_0, pad = var_3281_pad_0, pad_type = var_3281_pad_type_0, strides = var_3281_strides_0, weight = model_model_layers_4_self_attn_k_proj_weight_palettized, x = var_3243_cast_fp16)[name = string("op_3281")]; tensor var_3286 = const()[name = string("op_3286"), val = tensor([1, 8, 1, 128])]; tensor var_3287 = reshape(shape = var_3286, x = var_3281)[name = string("op_3287")]; string var_3303_pad_type_0 = const()[name = string("op_3303_pad_type_0"), val = string("valid")]; tensor var_3303_strides_0 = const()[name = string("op_3303_strides_0"), val = tensor([1, 1])]; tensor var_3303_pad_0 = const()[name = string("op_3303_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3303_dilations_0 = const()[name = string("op_3303_dilations_0"), val = tensor([1, 1])]; int32 var_3303_groups_0 = const()[name = string("op_3303_groups_0"), val = int32(1)]; tensor var_3303 = conv(dilations = var_3303_dilations_0, groups = var_3303_groups_0, pad = var_3303_pad_0, pad_type = var_3303_pad_type_0, strides = var_3303_strides_0, weight = model_model_layers_4_self_attn_v_proj_weight_palettized, x = var_3243_cast_fp16)[name = string("op_3303")]; tensor var_3308 = const()[name = string("op_3308"), val = tensor([1, 8, 1, 128])]; tensor var_3309 = reshape(shape = var_3308, x = var_3303)[name = string("op_3309")]; int32 var_3326 = const()[name = string("op_3326"), val = int32(-1)]; fp16 const_59_promoted = const()[name = string("const_59_promoted"), val = fp16(-0x1p+0)]; tensor var_3328 = mul(x = var_3265, y = const_59_promoted)[name = string("op_3328")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77 = concat(axis = var_3326, interleave = input_77_interleave_0, values = (var_3265, var_3328))[name = string("input_77")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_3323_to_fp16 = const()[name = string("op_3323_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_3323_to_fp16, x = input_77)[name = string("normed_69_cast_fp16")]; tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_71 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71")]; tensor const_61 = const()[name = string("const_61"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299281152)))]; tensor q_9 = mul(x = normed_71, y = const_61)[name = string("q_9")]; int32 var_3348 = const()[name = string("op_3348"), val = int32(-1)]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_3350 = mul(x = var_3287, y = const_62_promoted)[name = string("op_3350")]; bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; tensor input_79 = concat(axis = var_3348, interleave = input_79_interleave_0, values = (var_3287, var_3350))[name = string("input_79")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_3345_to_fp16 = const()[name = string("op_3345_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_3345_to_fp16, x = input_79)[name = string("normed_73_cast_fp16")]; tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_75 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75")]; tensor const_64 = const()[name = string("const_64"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299281472)))]; tensor k_9 = mul(x = normed_75, y = const_64)[name = string("k_9")]; tensor var_3359 = mul(x = q_9, y = cos_1_cast_fp16)[name = string("op_3359")]; tensor var_3364_begin_0 = const()[name = string("op_3364_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3364_end_0 = const()[name = string("op_3364_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_3364_end_mask_0 = const()[name = string("op_3364_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3364 = slice_by_index(begin = var_3364_begin_0, end = var_3364_end_0, end_mask = var_3364_end_mask_0, x = q_9)[name = string("op_3364")]; fp16 const_65_promoted = const()[name = string("const_65_promoted"), val = fp16(-0x1p+0)]; tensor var_3365 = mul(x = var_3364, y = const_65_promoted)[name = string("op_3365")]; tensor var_3370_begin_0 = const()[name = string("op_3370_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3370_end_0 = const()[name = string("op_3370_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_3370_end_mask_0 = const()[name = string("op_3370_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3370 = slice_by_index(begin = var_3370_begin_0, end = var_3370_end_0, end_mask = var_3370_end_mask_0, x = q_9)[name = string("op_3370")]; int32 var_3372 = const()[name = string("op_3372"), val = int32(-1)]; bool var_3373_interleave_0 = const()[name = string("op_3373_interleave_0"), val = bool(false)]; tensor var_3373 = concat(axis = var_3372, interleave = var_3373_interleave_0, values = (var_3365, var_3370))[name = string("op_3373")]; tensor var_3374 = mul(x = var_3373, y = sin_1_cast_fp16)[name = string("op_3374")]; tensor query_states_17 = add(x = var_3359, y = var_3374)[name = string("query_states_17")]; tensor var_3377 = mul(x = k_9, y = cos_1_cast_fp16)[name = string("op_3377")]; tensor var_3382_begin_0 = const()[name = string("op_3382_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3382_end_0 = const()[name = string("op_3382_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_3382_end_mask_0 = const()[name = string("op_3382_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3382 = slice_by_index(begin = var_3382_begin_0, end = var_3382_end_0, end_mask = var_3382_end_mask_0, x = k_9)[name = string("op_3382")]; fp16 const_66_promoted = const()[name = string("const_66_promoted"), val = fp16(-0x1p+0)]; tensor var_3383 = mul(x = var_3382, y = const_66_promoted)[name = string("op_3383")]; tensor var_3388_begin_0 = const()[name = string("op_3388_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3388_end_0 = const()[name = string("op_3388_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_3388_end_mask_0 = const()[name = string("op_3388_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3388 = slice_by_index(begin = var_3388_begin_0, end = var_3388_end_0, end_mask = var_3388_end_mask_0, x = k_9)[name = string("op_3388")]; int32 var_3390 = const()[name = string("op_3390"), val = int32(-1)]; bool var_3391_interleave_0 = const()[name = string("op_3391_interleave_0"), val = bool(false)]; tensor var_3391 = concat(axis = var_3390, interleave = var_3391_interleave_0, values = (var_3383, var_3388))[name = string("op_3391")]; tensor var_3392 = mul(x = var_3391, y = sin_1_cast_fp16)[name = string("op_3392")]; tensor key_states_17 = add(x = var_3377, y = var_3392)[name = string("key_states_17")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; int32 concat_34_axis_0 = const()[name = string("concat_34_axis_0"), val = int32(0)]; bool concat_34_interleave_0 = const()[name = string("concat_34_interleave_0"), val = bool(false)]; tensor concat_34 = concat(axis = concat_34_axis_0, interleave = concat_34_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_34")]; tensor concat_35_values1_0 = const()[name = string("concat_35_values1_0"), val = tensor([0])]; tensor concat_35_values3_0 = const()[name = string("concat_35_values3_0"), val = tensor([0])]; int32 concat_35_axis_0 = const()[name = string("concat_35_axis_0"), val = int32(0)]; bool concat_35_interleave_0 = const()[name = string("concat_35_interleave_0"), val = bool(false)]; tensor concat_35 = concat(axis = concat_35_axis_0, interleave = concat_35_interleave_0, values = (expand_dims_52, concat_35_values1_0, var_1717, concat_35_values3_0))[name = string("concat_35")]; tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_34, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_35, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_17, x = coreml_update_state_63)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_120_write_state")]; tensor coreml_update_state_64 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_120")]; tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([32])]; tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([33])]; int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_38")]; tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_58, concat_39_values1_0, var_1717, concat_39_values3_0))[name = string("concat_39")]; tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = var_3309, x = coreml_update_state_64)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_121_write_state")]; tensor coreml_update_state_65 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_121")]; tensor var_3447_begin_0 = const()[name = string("op_3447_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_3447_end_0 = const()[name = string("op_3447_end_0"), val = tensor([5, 8, 1536, 128])]; tensor var_3447_end_mask_0 = const()[name = string("op_3447_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3447_cast_fp16 = slice_by_index(begin = var_3447_begin_0, end = var_3447_end_0, end_mask = var_3447_end_mask_0, x = coreml_update_state_65)[name = string("op_3447_cast_fp16")]; tensor key_cache_9_axes_0 = const()[name = string("key_cache_9_axes_0"), val = tensor([0])]; tensor key_cache_9_cast_fp16 = squeeze(axes = key_cache_9_axes_0, x = var_3447_cast_fp16)[name = string("key_cache_9_cast_fp16")]; tensor var_3454_begin_0 = const()[name = string("op_3454_begin_0"), val = tensor([32, 0, 0, 0])]; tensor var_3454_end_0 = const()[name = string("op_3454_end_0"), val = tensor([33, 8, 1536, 128])]; tensor var_3454_end_mask_0 = const()[name = string("op_3454_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3454_cast_fp16 = slice_by_index(begin = var_3454_begin_0, end = var_3454_end_0, end_mask = var_3454_end_mask_0, x = coreml_update_state_65)[name = string("op_3454_cast_fp16")]; tensor value_cache_9_axes_0 = const()[name = string("value_cache_9_axes_0"), val = tensor([0])]; tensor value_cache_9_cast_fp16 = squeeze(axes = value_cache_9_axes_0, x = var_3454_cast_fp16)[name = string("value_cache_9_cast_fp16")]; tensor var_3478_axes_0 = const()[name = string("op_3478_axes_0"), val = tensor([1])]; tensor var_3478_cast_fp16 = expand_dims(axes = var_3478_axes_0, x = key_cache_9_cast_fp16)[name = string("op_3478_cast_fp16")]; tensor var_3483 = const()[name = string("op_3483"), val = tensor([1, 2, 1, 1])]; tensor value_35_cast_fp16 = tile(reps = var_3483, x = var_3478_cast_fp16)[name = string("value_35_cast_fp16")]; tensor var_3489 = const()[name = string("op_3489"), val = tensor([1, 16, 1536, 128])]; tensor key_states_19_cast_fp16 = reshape(shape = var_3489, x = value_35_cast_fp16)[name = string("key_states_19_cast_fp16")]; tensor var_3492_axes_0 = const()[name = string("op_3492_axes_0"), val = tensor([1])]; tensor var_3492_cast_fp16 = expand_dims(axes = var_3492_axes_0, x = value_cache_9_cast_fp16)[name = string("op_3492_cast_fp16")]; tensor var_3497 = const()[name = string("op_3497"), val = tensor([1, 2, 1, 1])]; tensor value_39_cast_fp16 = tile(reps = var_3497, x = var_3492_cast_fp16)[name = string("value_39_cast_fp16")]; tensor var_3503 = const()[name = string("op_3503"), val = tensor([1, 16, 1536, 128])]; tensor value_states_27_cast_fp16 = reshape(shape = var_3503, x = value_39_cast_fp16)[name = string("value_states_27_cast_fp16")]; bool var_3518_transpose_x_1 = const()[name = string("op_3518_transpose_x_1"), val = bool(false)]; bool var_3518_transpose_y_1 = const()[name = string("op_3518_transpose_y_1"), val = bool(true)]; tensor var_3518 = matmul(transpose_x = var_3518_transpose_x_1, transpose_y = var_3518_transpose_y_1, x = query_states_17, y = key_states_19_cast_fp16)[name = string("op_3518")]; fp16 var_3519_to_fp16 = const()[name = string("op_3519_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_17_cast_fp16 = mul(x = var_3518, y = var_3519_to_fp16)[name = string("attention_17_cast_fp16")]; tensor attention_19_cast_fp16 = add(x = attention_17_cast_fp16, y = causal_mask)[name = string("attention_19_cast_fp16")]; int32 var_3528 = const()[name = string("op_3528"), val = int32(-1)]; tensor probabilities_9_cast_fp16 = softmax(axis = var_3528, x = attention_19_cast_fp16)[name = string("probabilities_9_cast_fp16")]; bool output_25_transpose_x_0 = const()[name = string("output_25_transpose_x_0"), val = bool(false)]; bool output_25_transpose_y_0 = const()[name = string("output_25_transpose_y_0"), val = bool(false)]; tensor output_25_cast_fp16 = matmul(transpose_x = output_25_transpose_x_0, transpose_y = output_25_transpose_y_0, x = probabilities_9_cast_fp16, y = value_states_27_cast_fp16)[name = string("output_25_cast_fp16")]; tensor var_3539_perm_0 = const()[name = string("op_3539_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3545 = const()[name = string("op_3545"), val = tensor([1, 1, 2048])]; tensor var_3539_cast_fp16 = transpose(perm = var_3539_perm_0, x = output_25_cast_fp16)[name = string("transpose_142")]; tensor output_27_cast_fp16 = reshape(shape = var_3545, x = var_3539_cast_fp16)[name = string("output_27_cast_fp16")]; tensor var_3550 = const()[name = string("op_3550"), val = tensor([0, 2, 1])]; string var_3566_pad_type_0 = const()[name = string("op_3566_pad_type_0"), val = string("valid")]; int32 var_3566_groups_0 = const()[name = string("op_3566_groups_0"), val = int32(1)]; tensor var_3566_strides_0 = const()[name = string("op_3566_strides_0"), val = tensor([1])]; tensor var_3566_pad_0 = const()[name = string("op_3566_pad_0"), val = tensor([0, 0])]; tensor var_3566_dilations_0 = const()[name = string("op_3566_dilations_0"), val = tensor([1])]; tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299281792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300854720))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3551_cast_fp16 = transpose(perm = var_3550, x = output_27_cast_fp16)[name = string("transpose_141")]; tensor var_3566_cast_fp16 = conv(dilations = var_3566_dilations_0, groups = var_3566_groups_0, pad = var_3566_pad_0, pad_type = var_3566_pad_type_0, strides = var_3566_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3551_cast_fp16)[name = string("op_3566_cast_fp16")]; tensor var_3570 = const()[name = string("op_3570"), val = tensor([0, 2, 1])]; tensor attn_output_9_cast_fp16 = transpose(perm = var_3570, x = var_3566_cast_fp16)[name = string("transpose_140")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = attn_output_9_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; int32 var_3585 = const()[name = string("op_3585"), val = int32(-1)]; fp16 const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3587_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_67_promoted_to_fp16)[name = string("op_3587_cast_fp16")]; bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; tensor input_83_cast_fp16 = concat(axis = var_3585, interleave = input_83_interleave_0, values = (hidden_states_49_cast_fp16, var_3587_cast_fp16))[name = string("input_83_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_3582_to_fp16 = const()[name = string("op_3582_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3582_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 1, 1024])]; tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; tensor const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300871168)))]; tensor x_17_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_69_promoted_to_fp16)[name = string("x_17_cast_fp16")]; tensor var_3607 = const()[name = string("op_3607"), val = tensor([0, 2, 1])]; tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; tensor var_3608 = transpose(perm = var_3607, x = x_17_cast_fp16)[name = string("transpose_139")]; tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_3608)[name = string("input_85")]; string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; tensor input_87 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = model_model_layers_4_mlp_gate_proj_weight_palettized, x = input_85)[name = string("input_87")]; string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_4_mlp_up_proj_weight_palettized, x = input_85)[name = string("b_9")]; tensor c_9 = silu(x = input_87)[name = string("c_9")]; tensor input_89 = mul(x = c_9, y = b_9)[name = string("input_89")]; string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_4_mlp_down_proj_weight_palettized, x = input_89)[name = string("e_9")]; tensor var_3630_axes_0 = const()[name = string("op_3630_axes_0"), val = tensor([2])]; tensor var_3630 = squeeze(axes = var_3630_axes_0, x = e_9)[name = string("op_3630")]; tensor var_3631 = const()[name = string("op_3631"), val = tensor([0, 2, 1])]; tensor var_3632 = transpose(perm = var_3631, x = var_3630)[name = string("transpose_138")]; tensor hidden_states_51_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = var_3632)[name = string("hidden_states_51_cast_fp16")]; int32 var_3646 = const()[name = string("op_3646"), val = int32(-1)]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3648_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3648_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_3646, interleave = input_91_interleave_0, values = (hidden_states_51_cast_fp16, var_3648_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_3643_to_fp16 = const()[name = string("op_3643_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3643_to_fp16, x = input_91_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 1, 1024])]; tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; tensor const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300873280)))]; tensor hidden_states_53_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_72_promoted_to_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor var_3660 = const()[name = string("op_3660"), val = tensor([0, 2, 1])]; tensor var_3663_axes_0 = const()[name = string("op_3663_axes_0"), val = tensor([2])]; tensor var_3661_cast_fp16 = transpose(perm = var_3660, x = hidden_states_53_cast_fp16)[name = string("transpose_137")]; tensor var_3663_cast_fp16 = expand_dims(axes = var_3663_axes_0, x = var_3661_cast_fp16)[name = string("op_3663_cast_fp16")]; string var_3679_pad_type_0 = const()[name = string("op_3679_pad_type_0"), val = string("valid")]; tensor var_3679_strides_0 = const()[name = string("op_3679_strides_0"), val = tensor([1, 1])]; tensor var_3679_pad_0 = const()[name = string("op_3679_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3679_dilations_0 = const()[name = string("op_3679_dilations_0"), val = tensor([1, 1])]; int32 var_3679_groups_0 = const()[name = string("op_3679_groups_0"), val = int32(1)]; tensor var_3679 = conv(dilations = var_3679_dilations_0, groups = var_3679_groups_0, pad = var_3679_pad_0, pad_type = var_3679_pad_type_0, strides = var_3679_strides_0, weight = model_model_layers_5_self_attn_q_proj_weight_palettized, x = var_3663_cast_fp16)[name = string("op_3679")]; tensor var_3684 = const()[name = string("op_3684"), val = tensor([1, 16, 1, 128])]; tensor var_3685 = reshape(shape = var_3684, x = var_3679)[name = string("op_3685")]; string var_3701_pad_type_0 = const()[name = string("op_3701_pad_type_0"), val = string("valid")]; tensor var_3701_strides_0 = const()[name = string("op_3701_strides_0"), val = tensor([1, 1])]; tensor var_3701_pad_0 = const()[name = string("op_3701_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3701_dilations_0 = const()[name = string("op_3701_dilations_0"), val = tensor([1, 1])]; int32 var_3701_groups_0 = const()[name = string("op_3701_groups_0"), val = int32(1)]; tensor var_3701 = conv(dilations = var_3701_dilations_0, groups = var_3701_groups_0, pad = var_3701_pad_0, pad_type = var_3701_pad_type_0, strides = var_3701_strides_0, weight = model_model_layers_5_self_attn_k_proj_weight_palettized, x = var_3663_cast_fp16)[name = string("op_3701")]; tensor var_3706 = const()[name = string("op_3706"), val = tensor([1, 8, 1, 128])]; tensor var_3707 = reshape(shape = var_3706, x = var_3701)[name = string("op_3707")]; string var_3723_pad_type_0 = const()[name = string("op_3723_pad_type_0"), val = string("valid")]; tensor var_3723_strides_0 = const()[name = string("op_3723_strides_0"), val = tensor([1, 1])]; tensor var_3723_pad_0 = const()[name = string("op_3723_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3723_dilations_0 = const()[name = string("op_3723_dilations_0"), val = tensor([1, 1])]; int32 var_3723_groups_0 = const()[name = string("op_3723_groups_0"), val = int32(1)]; tensor var_3723 = conv(dilations = var_3723_dilations_0, groups = var_3723_groups_0, pad = var_3723_pad_0, pad_type = var_3723_pad_type_0, strides = var_3723_strides_0, weight = model_model_layers_5_self_attn_v_proj_weight_palettized, x = var_3663_cast_fp16)[name = string("op_3723")]; tensor var_3728 = const()[name = string("op_3728"), val = tensor([1, 8, 1, 128])]; tensor var_3729 = reshape(shape = var_3728, x = var_3723)[name = string("op_3729")]; int32 var_3746 = const()[name = string("op_3746"), val = int32(-1)]; fp16 const_73_promoted = const()[name = string("const_73_promoted"), val = fp16(-0x1p+0)]; tensor var_3748 = mul(x = var_3685, y = const_73_promoted)[name = string("op_3748")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95 = concat(axis = var_3746, interleave = input_95_interleave_0, values = (var_3685, var_3748))[name = string("input_95")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3743_to_fp16, x = input_95)[name = string("normed_85_cast_fp16")]; tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_87 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87")]; tensor const_75 = const()[name = string("const_75"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300875392)))]; tensor q_11 = mul(x = normed_87, y = const_75)[name = string("q_11")]; int32 var_3768 = const()[name = string("op_3768"), val = int32(-1)]; fp16 const_76_promoted = const()[name = string("const_76_promoted"), val = fp16(-0x1p+0)]; tensor var_3770 = mul(x = var_3707, y = const_76_promoted)[name = string("op_3770")]; bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; tensor input_97 = concat(axis = var_3768, interleave = input_97_interleave_0, values = (var_3707, var_3770))[name = string("input_97")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_3765_to_fp16 = const()[name = string("op_3765_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3765_to_fp16, x = input_97)[name = string("normed_89_cast_fp16")]; tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_91 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91")]; tensor const_78 = const()[name = string("const_78"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300875712)))]; tensor k_11 = mul(x = normed_91, y = const_78)[name = string("k_11")]; tensor var_3779 = mul(x = q_11, y = cos_1_cast_fp16)[name = string("op_3779")]; tensor var_3784_begin_0 = const()[name = string("op_3784_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3784_end_0 = const()[name = string("op_3784_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_3784_end_mask_0 = const()[name = string("op_3784_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3784 = slice_by_index(begin = var_3784_begin_0, end = var_3784_end_0, end_mask = var_3784_end_mask_0, x = q_11)[name = string("op_3784")]; fp16 const_79_promoted = const()[name = string("const_79_promoted"), val = fp16(-0x1p+0)]; tensor var_3785 = mul(x = var_3784, y = const_79_promoted)[name = string("op_3785")]; tensor var_3790_begin_0 = const()[name = string("op_3790_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3790_end_0 = const()[name = string("op_3790_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_3790_end_mask_0 = const()[name = string("op_3790_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3790 = slice_by_index(begin = var_3790_begin_0, end = var_3790_end_0, end_mask = var_3790_end_mask_0, x = q_11)[name = string("op_3790")]; int32 var_3792 = const()[name = string("op_3792"), val = int32(-1)]; bool var_3793_interleave_0 = const()[name = string("op_3793_interleave_0"), val = bool(false)]; tensor var_3793 = concat(axis = var_3792, interleave = var_3793_interleave_0, values = (var_3785, var_3790))[name = string("op_3793")]; tensor var_3794 = mul(x = var_3793, y = sin_1_cast_fp16)[name = string("op_3794")]; tensor query_states_21 = add(x = var_3779, y = var_3794)[name = string("query_states_21")]; tensor var_3797 = mul(x = k_11, y = cos_1_cast_fp16)[name = string("op_3797")]; tensor var_3802_begin_0 = const()[name = string("op_3802_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3802_end_0 = const()[name = string("op_3802_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_3802_end_mask_0 = const()[name = string("op_3802_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3802 = slice_by_index(begin = var_3802_begin_0, end = var_3802_end_0, end_mask = var_3802_end_mask_0, x = k_11)[name = string("op_3802")]; fp16 const_80_promoted = const()[name = string("const_80_promoted"), val = fp16(-0x1p+0)]; tensor var_3803 = mul(x = var_3802, y = const_80_promoted)[name = string("op_3803")]; tensor var_3808_begin_0 = const()[name = string("op_3808_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3808_end_0 = const()[name = string("op_3808_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_3808_end_mask_0 = const()[name = string("op_3808_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3808 = slice_by_index(begin = var_3808_begin_0, end = var_3808_end_0, end_mask = var_3808_end_mask_0, x = k_11)[name = string("op_3808")]; int32 var_3810 = const()[name = string("op_3810"), val = int32(-1)]; bool var_3811_interleave_0 = const()[name = string("op_3811_interleave_0"), val = bool(false)]; tensor var_3811 = concat(axis = var_3810, interleave = var_3811_interleave_0, values = (var_3803, var_3808))[name = string("op_3811")]; tensor var_3812 = mul(x = var_3811, y = sin_1_cast_fp16)[name = string("op_3812")]; tensor key_states_21 = add(x = var_3797, y = var_3812)[name = string("key_states_21")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_42")]; tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_64, concat_43_values1_0, var_1717, concat_43_values3_0))[name = string("concat_43")]; tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_21, x = coreml_update_state_65)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_122_write_state")]; tensor coreml_update_state_66 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_122")]; tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([33])]; tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([34])]; int32 concat_46_axis_0 = const()[name = string("concat_46_axis_0"), val = int32(0)]; bool concat_46_interleave_0 = const()[name = string("concat_46_interleave_0"), val = bool(false)]; tensor concat_46 = concat(axis = concat_46_axis_0, interleave = concat_46_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_46")]; tensor concat_47_values1_0 = const()[name = string("concat_47_values1_0"), val = tensor([0])]; tensor concat_47_values3_0 = const()[name = string("concat_47_values3_0"), val = tensor([0])]; int32 concat_47_axis_0 = const()[name = string("concat_47_axis_0"), val = int32(0)]; bool concat_47_interleave_0 = const()[name = string("concat_47_interleave_0"), val = bool(false)]; tensor concat_47 = concat(axis = concat_47_axis_0, interleave = concat_47_interleave_0, values = (expand_dims_70, concat_47_values1_0, var_1717, concat_47_values3_0))[name = string("concat_47")]; tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_46, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_47, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = var_3729, x = coreml_update_state_66)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_123_write_state")]; tensor coreml_update_state_67 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_123")]; tensor var_3867_begin_0 = const()[name = string("op_3867_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_3867_end_0 = const()[name = string("op_3867_end_0"), val = tensor([6, 8, 1536, 128])]; tensor var_3867_end_mask_0 = const()[name = string("op_3867_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3867_cast_fp16 = slice_by_index(begin = var_3867_begin_0, end = var_3867_end_0, end_mask = var_3867_end_mask_0, x = coreml_update_state_67)[name = string("op_3867_cast_fp16")]; tensor key_cache_11_axes_0 = const()[name = string("key_cache_11_axes_0"), val = tensor([0])]; tensor key_cache_11_cast_fp16 = squeeze(axes = key_cache_11_axes_0, x = var_3867_cast_fp16)[name = string("key_cache_11_cast_fp16")]; tensor var_3874_begin_0 = const()[name = string("op_3874_begin_0"), val = tensor([33, 0, 0, 0])]; tensor var_3874_end_0 = const()[name = string("op_3874_end_0"), val = tensor([34, 8, 1536, 128])]; tensor var_3874_end_mask_0 = const()[name = string("op_3874_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3874_cast_fp16 = slice_by_index(begin = var_3874_begin_0, end = var_3874_end_0, end_mask = var_3874_end_mask_0, x = coreml_update_state_67)[name = string("op_3874_cast_fp16")]; tensor value_cache_11_axes_0 = const()[name = string("value_cache_11_axes_0"), val = tensor([0])]; tensor value_cache_11_cast_fp16 = squeeze(axes = value_cache_11_axes_0, x = var_3874_cast_fp16)[name = string("value_cache_11_cast_fp16")]; tensor var_3898_axes_0 = const()[name = string("op_3898_axes_0"), val = tensor([1])]; tensor var_3898_cast_fp16 = expand_dims(axes = var_3898_axes_0, x = key_cache_11_cast_fp16)[name = string("op_3898_cast_fp16")]; tensor var_3903 = const()[name = string("op_3903"), val = tensor([1, 2, 1, 1])]; tensor value_43_cast_fp16 = tile(reps = var_3903, x = var_3898_cast_fp16)[name = string("value_43_cast_fp16")]; tensor var_3909 = const()[name = string("op_3909"), val = tensor([1, 16, 1536, 128])]; tensor key_states_23_cast_fp16 = reshape(shape = var_3909, x = value_43_cast_fp16)[name = string("key_states_23_cast_fp16")]; tensor var_3912_axes_0 = const()[name = string("op_3912_axes_0"), val = tensor([1])]; tensor var_3912_cast_fp16 = expand_dims(axes = var_3912_axes_0, x = value_cache_11_cast_fp16)[name = string("op_3912_cast_fp16")]; tensor var_3917 = const()[name = string("op_3917"), val = tensor([1, 2, 1, 1])]; tensor value_47_cast_fp16 = tile(reps = var_3917, x = var_3912_cast_fp16)[name = string("value_47_cast_fp16")]; tensor var_3923 = const()[name = string("op_3923"), val = tensor([1, 16, 1536, 128])]; tensor value_states_33_cast_fp16 = reshape(shape = var_3923, x = value_47_cast_fp16)[name = string("value_states_33_cast_fp16")]; bool var_3938_transpose_x_1 = const()[name = string("op_3938_transpose_x_1"), val = bool(false)]; bool var_3938_transpose_y_1 = const()[name = string("op_3938_transpose_y_1"), val = bool(true)]; tensor var_3938 = matmul(transpose_x = var_3938_transpose_x_1, transpose_y = var_3938_transpose_y_1, x = query_states_21, y = key_states_23_cast_fp16)[name = string("op_3938")]; fp16 var_3939_to_fp16 = const()[name = string("op_3939_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_21_cast_fp16 = mul(x = var_3938, y = var_3939_to_fp16)[name = string("attention_21_cast_fp16")]; tensor attention_23_cast_fp16 = add(x = attention_21_cast_fp16, y = causal_mask)[name = string("attention_23_cast_fp16")]; int32 var_3948 = const()[name = string("op_3948"), val = int32(-1)]; tensor probabilities_11_cast_fp16 = softmax(axis = var_3948, x = attention_23_cast_fp16)[name = string("probabilities_11_cast_fp16")]; bool output_31_transpose_x_0 = const()[name = string("output_31_transpose_x_0"), val = bool(false)]; bool output_31_transpose_y_0 = const()[name = string("output_31_transpose_y_0"), val = bool(false)]; tensor output_31_cast_fp16 = matmul(transpose_x = output_31_transpose_x_0, transpose_y = output_31_transpose_y_0, x = probabilities_11_cast_fp16, y = value_states_33_cast_fp16)[name = string("output_31_cast_fp16")]; tensor var_3959_perm_0 = const()[name = string("op_3959_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3965 = const()[name = string("op_3965"), val = tensor([1, 1, 2048])]; tensor var_3959_cast_fp16 = transpose(perm = var_3959_perm_0, x = output_31_cast_fp16)[name = string("transpose_136")]; tensor output_33_cast_fp16 = reshape(shape = var_3965, x = var_3959_cast_fp16)[name = string("output_33_cast_fp16")]; tensor var_3970 = const()[name = string("op_3970"), val = tensor([0, 2, 1])]; string var_3986_pad_type_0 = const()[name = string("op_3986_pad_type_0"), val = string("valid")]; int32 var_3986_groups_0 = const()[name = string("op_3986_groups_0"), val = int32(1)]; tensor var_3986_strides_0 = const()[name = string("op_3986_strides_0"), val = tensor([1])]; tensor var_3986_pad_0 = const()[name = string("op_3986_pad_0"), val = tensor([0, 0])]; tensor var_3986_dilations_0 = const()[name = string("op_3986_dilations_0"), val = tensor([1])]; tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300876032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302448960))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3971_cast_fp16 = transpose(perm = var_3970, x = output_33_cast_fp16)[name = string("transpose_135")]; tensor var_3986_cast_fp16 = conv(dilations = var_3986_dilations_0, groups = var_3986_groups_0, pad = var_3986_pad_0, pad_type = var_3986_pad_type_0, strides = var_3986_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_3971_cast_fp16)[name = string("op_3986_cast_fp16")]; tensor var_3990 = const()[name = string("op_3990"), val = tensor([0, 2, 1])]; tensor attn_output_11_cast_fp16 = transpose(perm = var_3990, x = var_3986_cast_fp16)[name = string("transpose_134")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_51_cast_fp16, y = attn_output_11_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; int32 var_4005 = const()[name = string("op_4005"), val = int32(-1)]; fp16 const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4007_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_81_promoted_to_fp16)[name = string("op_4007_cast_fp16")]; bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; tensor input_101_cast_fp16 = concat(axis = var_4005, interleave = input_101_interleave_0, values = (hidden_states_59_cast_fp16, var_4007_cast_fp16))[name = string("input_101_cast_fp16")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_4002_to_fp16 = const()[name = string("op_4002_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_4002_to_fp16, x = input_101_cast_fp16)[name = string("normed_93_cast_fp16")]; tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 1, 1024])]; tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; tensor const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302465408)))]; tensor x_21_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_83_promoted_to_fp16)[name = string("x_21_cast_fp16")]; tensor var_4027 = const()[name = string("op_4027"), val = tensor([0, 2, 1])]; tensor input_103_axes_0 = const()[name = string("input_103_axes_0"), val = tensor([2])]; tensor var_4028 = transpose(perm = var_4027, x = x_21_cast_fp16)[name = string("transpose_133")]; tensor input_103 = expand_dims(axes = input_103_axes_0, x = var_4028)[name = string("input_103")]; string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; tensor input_105 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = model_model_layers_5_mlp_gate_proj_weight_palettized, x = input_103)[name = string("input_105")]; string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_5_mlp_up_proj_weight_palettized, x = input_103)[name = string("b_11")]; tensor c_11 = silu(x = input_105)[name = string("c_11")]; tensor input_107 = mul(x = c_11, y = b_11)[name = string("input_107")]; string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_5_mlp_down_proj_weight_palettized, x = input_107)[name = string("e_11")]; tensor var_4050_axes_0 = const()[name = string("op_4050_axes_0"), val = tensor([2])]; tensor var_4050 = squeeze(axes = var_4050_axes_0, x = e_11)[name = string("op_4050")]; tensor var_4051 = const()[name = string("op_4051"), val = tensor([0, 2, 1])]; tensor var_4052 = transpose(perm = var_4051, x = var_4050)[name = string("transpose_132")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = var_4052)[name = string("hidden_states_61_cast_fp16")]; int32 var_4066 = const()[name = string("op_4066"), val = int32(-1)]; fp16 const_84_promoted_to_fp16 = const()[name = string("const_84_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4068_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_84_promoted_to_fp16)[name = string("op_4068_cast_fp16")]; bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; tensor input_109_cast_fp16 = concat(axis = var_4066, interleave = input_109_interleave_0, values = (hidden_states_61_cast_fp16, var_4068_cast_fp16))[name = string("input_109_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_4063_to_fp16 = const()[name = string("op_4063_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_4063_to_fp16, x = input_109_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 1, 1024])]; tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; tensor const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302467520)))]; tensor hidden_states_63_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_86_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor var_4080 = const()[name = string("op_4080"), val = tensor([0, 2, 1])]; tensor var_4083_axes_0 = const()[name = string("op_4083_axes_0"), val = tensor([2])]; tensor var_4081_cast_fp16 = transpose(perm = var_4080, x = hidden_states_63_cast_fp16)[name = string("transpose_131")]; tensor var_4083_cast_fp16 = expand_dims(axes = var_4083_axes_0, x = var_4081_cast_fp16)[name = string("op_4083_cast_fp16")]; string var_4099_pad_type_0 = const()[name = string("op_4099_pad_type_0"), val = string("valid")]; tensor var_4099_strides_0 = const()[name = string("op_4099_strides_0"), val = tensor([1, 1])]; tensor var_4099_pad_0 = const()[name = string("op_4099_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4099_dilations_0 = const()[name = string("op_4099_dilations_0"), val = tensor([1, 1])]; int32 var_4099_groups_0 = const()[name = string("op_4099_groups_0"), val = int32(1)]; tensor var_4099 = conv(dilations = var_4099_dilations_0, groups = var_4099_groups_0, pad = var_4099_pad_0, pad_type = var_4099_pad_type_0, strides = var_4099_strides_0, weight = model_model_layers_6_self_attn_q_proj_weight_palettized, x = var_4083_cast_fp16)[name = string("op_4099")]; tensor var_4104 = const()[name = string("op_4104"), val = tensor([1, 16, 1, 128])]; tensor var_4105 = reshape(shape = var_4104, x = var_4099)[name = string("op_4105")]; string var_4121_pad_type_0 = const()[name = string("op_4121_pad_type_0"), val = string("valid")]; tensor var_4121_strides_0 = const()[name = string("op_4121_strides_0"), val = tensor([1, 1])]; tensor var_4121_pad_0 = const()[name = string("op_4121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4121_dilations_0 = const()[name = string("op_4121_dilations_0"), val = tensor([1, 1])]; int32 var_4121_groups_0 = const()[name = string("op_4121_groups_0"), val = int32(1)]; tensor var_4121 = conv(dilations = var_4121_dilations_0, groups = var_4121_groups_0, pad = var_4121_pad_0, pad_type = var_4121_pad_type_0, strides = var_4121_strides_0, weight = model_model_layers_6_self_attn_k_proj_weight_palettized, x = var_4083_cast_fp16)[name = string("op_4121")]; tensor var_4126 = const()[name = string("op_4126"), val = tensor([1, 8, 1, 128])]; tensor var_4127 = reshape(shape = var_4126, x = var_4121)[name = string("op_4127")]; string var_4143_pad_type_0 = const()[name = string("op_4143_pad_type_0"), val = string("valid")]; tensor var_4143_strides_0 = const()[name = string("op_4143_strides_0"), val = tensor([1, 1])]; tensor var_4143_pad_0 = const()[name = string("op_4143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4143_dilations_0 = const()[name = string("op_4143_dilations_0"), val = tensor([1, 1])]; int32 var_4143_groups_0 = const()[name = string("op_4143_groups_0"), val = int32(1)]; tensor var_4143 = conv(dilations = var_4143_dilations_0, groups = var_4143_groups_0, pad = var_4143_pad_0, pad_type = var_4143_pad_type_0, strides = var_4143_strides_0, weight = model_model_layers_6_self_attn_v_proj_weight_palettized, x = var_4083_cast_fp16)[name = string("op_4143")]; tensor var_4148 = const()[name = string("op_4148"), val = tensor([1, 8, 1, 128])]; tensor var_4149 = reshape(shape = var_4148, x = var_4143)[name = string("op_4149")]; int32 var_4166 = const()[name = string("op_4166"), val = int32(-1)]; fp16 const_87_promoted = const()[name = string("const_87_promoted"), val = fp16(-0x1p+0)]; tensor var_4168 = mul(x = var_4105, y = const_87_promoted)[name = string("op_4168")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113 = concat(axis = var_4166, interleave = input_113_interleave_0, values = (var_4105, var_4168))[name = string("input_113")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_4163_to_fp16 = const()[name = string("op_4163_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_4163_to_fp16, x = input_113)[name = string("normed_101_cast_fp16")]; tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_103 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103")]; tensor const_89 = const()[name = string("const_89"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302469632)))]; tensor q_13 = mul(x = normed_103, y = const_89)[name = string("q_13")]; int32 var_4188 = const()[name = string("op_4188"), val = int32(-1)]; fp16 const_90_promoted = const()[name = string("const_90_promoted"), val = fp16(-0x1p+0)]; tensor var_4190 = mul(x = var_4127, y = const_90_promoted)[name = string("op_4190")]; bool input_115_interleave_0 = const()[name = string("input_115_interleave_0"), val = bool(false)]; tensor input_115 = concat(axis = var_4188, interleave = input_115_interleave_0, values = (var_4127, var_4190))[name = string("input_115")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_4185_to_fp16 = const()[name = string("op_4185_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_4185_to_fp16, x = input_115)[name = string("normed_105_cast_fp16")]; tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_107 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107")]; tensor const_92 = const()[name = string("const_92"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302469952)))]; tensor k_13 = mul(x = normed_107, y = const_92)[name = string("k_13")]; tensor var_4199 = mul(x = q_13, y = cos_1_cast_fp16)[name = string("op_4199")]; tensor var_4204_begin_0 = const()[name = string("op_4204_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4204_end_0 = const()[name = string("op_4204_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_4204_end_mask_0 = const()[name = string("op_4204_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4204 = slice_by_index(begin = var_4204_begin_0, end = var_4204_end_0, end_mask = var_4204_end_mask_0, x = q_13)[name = string("op_4204")]; fp16 const_93_promoted = const()[name = string("const_93_promoted"), val = fp16(-0x1p+0)]; tensor var_4205 = mul(x = var_4204, y = const_93_promoted)[name = string("op_4205")]; tensor var_4210_begin_0 = const()[name = string("op_4210_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4210_end_0 = const()[name = string("op_4210_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_4210_end_mask_0 = const()[name = string("op_4210_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4210 = slice_by_index(begin = var_4210_begin_0, end = var_4210_end_0, end_mask = var_4210_end_mask_0, x = q_13)[name = string("op_4210")]; int32 var_4212 = const()[name = string("op_4212"), val = int32(-1)]; bool var_4213_interleave_0 = const()[name = string("op_4213_interleave_0"), val = bool(false)]; tensor var_4213 = concat(axis = var_4212, interleave = var_4213_interleave_0, values = (var_4205, var_4210))[name = string("op_4213")]; tensor var_4214 = mul(x = var_4213, y = sin_1_cast_fp16)[name = string("op_4214")]; tensor query_states_25 = add(x = var_4199, y = var_4214)[name = string("query_states_25")]; tensor var_4217 = mul(x = k_13, y = cos_1_cast_fp16)[name = string("op_4217")]; tensor var_4222_begin_0 = const()[name = string("op_4222_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4222_end_0 = const()[name = string("op_4222_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_4222_end_mask_0 = const()[name = string("op_4222_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4222 = slice_by_index(begin = var_4222_begin_0, end = var_4222_end_0, end_mask = var_4222_end_mask_0, x = k_13)[name = string("op_4222")]; fp16 const_94_promoted = const()[name = string("const_94_promoted"), val = fp16(-0x1p+0)]; tensor var_4223 = mul(x = var_4222, y = const_94_promoted)[name = string("op_4223")]; tensor var_4228_begin_0 = const()[name = string("op_4228_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4228_end_0 = const()[name = string("op_4228_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_4228_end_mask_0 = const()[name = string("op_4228_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4228 = slice_by_index(begin = var_4228_begin_0, end = var_4228_end_0, end_mask = var_4228_end_mask_0, x = k_13)[name = string("op_4228")]; int32 var_4230 = const()[name = string("op_4230"), val = int32(-1)]; bool var_4231_interleave_0 = const()[name = string("op_4231_interleave_0"), val = bool(false)]; tensor var_4231 = concat(axis = var_4230, interleave = var_4231_interleave_0, values = (var_4223, var_4228))[name = string("op_4231")]; tensor var_4232 = mul(x = var_4231, y = sin_1_cast_fp16)[name = string("op_4232")]; tensor key_states_25 = add(x = var_4217, y = var_4232)[name = string("key_states_25")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_50")]; tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_76, concat_51_values1_0, var_1717, concat_51_values3_0))[name = string("concat_51")]; tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_50, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_51, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_25, x = coreml_update_state_67)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_124_write_state")]; tensor coreml_update_state_68 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_124")]; tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([34])]; tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([35])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_54")]; tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_82, concat_55_values1_0, var_1717, concat_55_values3_0))[name = string("concat_55")]; tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_54, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_55, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = var_4149, x = coreml_update_state_68)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_125_write_state")]; tensor coreml_update_state_69 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_125")]; tensor var_4287_begin_0 = const()[name = string("op_4287_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_4287_end_0 = const()[name = string("op_4287_end_0"), val = tensor([7, 8, 1536, 128])]; tensor var_4287_end_mask_0 = const()[name = string("op_4287_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4287_cast_fp16 = slice_by_index(begin = var_4287_begin_0, end = var_4287_end_0, end_mask = var_4287_end_mask_0, x = coreml_update_state_69)[name = string("op_4287_cast_fp16")]; tensor key_cache_13_axes_0 = const()[name = string("key_cache_13_axes_0"), val = tensor([0])]; tensor key_cache_13_cast_fp16 = squeeze(axes = key_cache_13_axes_0, x = var_4287_cast_fp16)[name = string("key_cache_13_cast_fp16")]; tensor var_4294_begin_0 = const()[name = string("op_4294_begin_0"), val = tensor([34, 0, 0, 0])]; tensor var_4294_end_0 = const()[name = string("op_4294_end_0"), val = tensor([35, 8, 1536, 128])]; tensor var_4294_end_mask_0 = const()[name = string("op_4294_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4294_cast_fp16 = slice_by_index(begin = var_4294_begin_0, end = var_4294_end_0, end_mask = var_4294_end_mask_0, x = coreml_update_state_69)[name = string("op_4294_cast_fp16")]; tensor value_cache_13_axes_0 = const()[name = string("value_cache_13_axes_0"), val = tensor([0])]; tensor value_cache_13_cast_fp16 = squeeze(axes = value_cache_13_axes_0, x = var_4294_cast_fp16)[name = string("value_cache_13_cast_fp16")]; tensor var_4318_axes_0 = const()[name = string("op_4318_axes_0"), val = tensor([1])]; tensor var_4318_cast_fp16 = expand_dims(axes = var_4318_axes_0, x = key_cache_13_cast_fp16)[name = string("op_4318_cast_fp16")]; tensor var_4323 = const()[name = string("op_4323"), val = tensor([1, 2, 1, 1])]; tensor value_51_cast_fp16 = tile(reps = var_4323, x = var_4318_cast_fp16)[name = string("value_51_cast_fp16")]; tensor var_4329 = const()[name = string("op_4329"), val = tensor([1, 16, 1536, 128])]; tensor key_states_27_cast_fp16 = reshape(shape = var_4329, x = value_51_cast_fp16)[name = string("key_states_27_cast_fp16")]; tensor var_4332_axes_0 = const()[name = string("op_4332_axes_0"), val = tensor([1])]; tensor var_4332_cast_fp16 = expand_dims(axes = var_4332_axes_0, x = value_cache_13_cast_fp16)[name = string("op_4332_cast_fp16")]; tensor var_4337 = const()[name = string("op_4337"), val = tensor([1, 2, 1, 1])]; tensor value_55_cast_fp16 = tile(reps = var_4337, x = var_4332_cast_fp16)[name = string("value_55_cast_fp16")]; tensor var_4343 = const()[name = string("op_4343"), val = tensor([1, 16, 1536, 128])]; tensor value_states_39_cast_fp16 = reshape(shape = var_4343, x = value_55_cast_fp16)[name = string("value_states_39_cast_fp16")]; bool var_4358_transpose_x_1 = const()[name = string("op_4358_transpose_x_1"), val = bool(false)]; bool var_4358_transpose_y_1 = const()[name = string("op_4358_transpose_y_1"), val = bool(true)]; tensor var_4358 = matmul(transpose_x = var_4358_transpose_x_1, transpose_y = var_4358_transpose_y_1, x = query_states_25, y = key_states_27_cast_fp16)[name = string("op_4358")]; fp16 var_4359_to_fp16 = const()[name = string("op_4359_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_25_cast_fp16 = mul(x = var_4358, y = var_4359_to_fp16)[name = string("attention_25_cast_fp16")]; tensor attention_27_cast_fp16 = add(x = attention_25_cast_fp16, y = causal_mask)[name = string("attention_27_cast_fp16")]; int32 var_4368 = const()[name = string("op_4368"), val = int32(-1)]; tensor probabilities_13_cast_fp16 = softmax(axis = var_4368, x = attention_27_cast_fp16)[name = string("probabilities_13_cast_fp16")]; bool output_37_transpose_x_0 = const()[name = string("output_37_transpose_x_0"), val = bool(false)]; bool output_37_transpose_y_0 = const()[name = string("output_37_transpose_y_0"), val = bool(false)]; tensor output_37_cast_fp16 = matmul(transpose_x = output_37_transpose_x_0, transpose_y = output_37_transpose_y_0, x = probabilities_13_cast_fp16, y = value_states_39_cast_fp16)[name = string("output_37_cast_fp16")]; tensor var_4379_perm_0 = const()[name = string("op_4379_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4385 = const()[name = string("op_4385"), val = tensor([1, 1, 2048])]; tensor var_4379_cast_fp16 = transpose(perm = var_4379_perm_0, x = output_37_cast_fp16)[name = string("transpose_130")]; tensor output_39_cast_fp16 = reshape(shape = var_4385, x = var_4379_cast_fp16)[name = string("output_39_cast_fp16")]; tensor var_4390 = const()[name = string("op_4390"), val = tensor([0, 2, 1])]; string var_4406_pad_type_0 = const()[name = string("op_4406_pad_type_0"), val = string("valid")]; int32 var_4406_groups_0 = const()[name = string("op_4406_groups_0"), val = int32(1)]; tensor var_4406_strides_0 = const()[name = string("op_4406_strides_0"), val = tensor([1])]; tensor var_4406_pad_0 = const()[name = string("op_4406_pad_0"), val = tensor([0, 0])]; tensor var_4406_dilations_0 = const()[name = string("op_4406_dilations_0"), val = tensor([1])]; tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302470272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304043200))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4391_cast_fp16 = transpose(perm = var_4390, x = output_39_cast_fp16)[name = string("transpose_129")]; tensor var_4406_cast_fp16 = conv(dilations = var_4406_dilations_0, groups = var_4406_groups_0, pad = var_4406_pad_0, pad_type = var_4406_pad_type_0, strides = var_4406_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4391_cast_fp16)[name = string("op_4406_cast_fp16")]; tensor var_4410 = const()[name = string("op_4410"), val = tensor([0, 2, 1])]; tensor attn_output_13_cast_fp16 = transpose(perm = var_4410, x = var_4406_cast_fp16)[name = string("transpose_128")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = attn_output_13_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; int32 var_4425 = const()[name = string("op_4425"), val = int32(-1)]; fp16 const_95_promoted_to_fp16 = const()[name = string("const_95_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4427_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_95_promoted_to_fp16)[name = string("op_4427_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_4425, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_4427_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_4422_to_fp16 = const()[name = string("op_4422_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_4422_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 1, 1024])]; tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; tensor const_97_promoted_to_fp16 = const()[name = string("const_97_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304059648)))]; tensor x_25_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_97_promoted_to_fp16)[name = string("x_25_cast_fp16")]; tensor var_4447 = const()[name = string("op_4447"), val = tensor([0, 2, 1])]; tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; tensor var_4448 = transpose(perm = var_4447, x = x_25_cast_fp16)[name = string("transpose_127")]; tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4448)[name = string("input_121")]; string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_6_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_6_mlp_up_proj_weight_palettized, x = input_121)[name = string("b_13")]; tensor c_13 = silu(x = input_123)[name = string("c_13")]; tensor input_125 = mul(x = c_13, y = b_13)[name = string("input_125")]; string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_6_mlp_down_proj_weight_palettized, x = input_125)[name = string("e_13")]; tensor var_4470_axes_0 = const()[name = string("op_4470_axes_0"), val = tensor([2])]; tensor var_4470 = squeeze(axes = var_4470_axes_0, x = e_13)[name = string("op_4470")]; tensor var_4471 = const()[name = string("op_4471"), val = tensor([0, 2, 1])]; tensor var_4472 = transpose(perm = var_4471, x = var_4470)[name = string("transpose_126")]; tensor hidden_states_71_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_4472)[name = string("hidden_states_71_cast_fp16")]; int32 var_4486 = const()[name = string("op_4486"), val = int32(-1)]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4488_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4488_cast_fp16")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127_cast_fp16 = concat(axis = var_4486, interleave = input_127_interleave_0, values = (hidden_states_71_cast_fp16, var_4488_cast_fp16))[name = string("input_127_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_4483_to_fp16 = const()[name = string("op_4483_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4483_to_fp16, x = input_127_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 1, 1024])]; tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; tensor const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304061760)))]; tensor hidden_states_73_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_100_promoted_to_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor var_4500 = const()[name = string("op_4500"), val = tensor([0, 2, 1])]; tensor var_4503_axes_0 = const()[name = string("op_4503_axes_0"), val = tensor([2])]; tensor var_4501_cast_fp16 = transpose(perm = var_4500, x = hidden_states_73_cast_fp16)[name = string("transpose_125")]; tensor var_4503_cast_fp16 = expand_dims(axes = var_4503_axes_0, x = var_4501_cast_fp16)[name = string("op_4503_cast_fp16")]; string var_4519_pad_type_0 = const()[name = string("op_4519_pad_type_0"), val = string("valid")]; tensor var_4519_strides_0 = const()[name = string("op_4519_strides_0"), val = tensor([1, 1])]; tensor var_4519_pad_0 = const()[name = string("op_4519_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4519_dilations_0 = const()[name = string("op_4519_dilations_0"), val = tensor([1, 1])]; int32 var_4519_groups_0 = const()[name = string("op_4519_groups_0"), val = int32(1)]; tensor var_4519 = conv(dilations = var_4519_dilations_0, groups = var_4519_groups_0, pad = var_4519_pad_0, pad_type = var_4519_pad_type_0, strides = var_4519_strides_0, weight = model_model_layers_7_self_attn_q_proj_weight_palettized, x = var_4503_cast_fp16)[name = string("op_4519")]; tensor var_4524 = const()[name = string("op_4524"), val = tensor([1, 16, 1, 128])]; tensor var_4525 = reshape(shape = var_4524, x = var_4519)[name = string("op_4525")]; string var_4541_pad_type_0 = const()[name = string("op_4541_pad_type_0"), val = string("valid")]; tensor var_4541_strides_0 = const()[name = string("op_4541_strides_0"), val = tensor([1, 1])]; tensor var_4541_pad_0 = const()[name = string("op_4541_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4541_dilations_0 = const()[name = string("op_4541_dilations_0"), val = tensor([1, 1])]; int32 var_4541_groups_0 = const()[name = string("op_4541_groups_0"), val = int32(1)]; tensor var_4541 = conv(dilations = var_4541_dilations_0, groups = var_4541_groups_0, pad = var_4541_pad_0, pad_type = var_4541_pad_type_0, strides = var_4541_strides_0, weight = model_model_layers_7_self_attn_k_proj_weight_palettized, x = var_4503_cast_fp16)[name = string("op_4541")]; tensor var_4546 = const()[name = string("op_4546"), val = tensor([1, 8, 1, 128])]; tensor var_4547 = reshape(shape = var_4546, x = var_4541)[name = string("op_4547")]; string var_4563_pad_type_0 = const()[name = string("op_4563_pad_type_0"), val = string("valid")]; tensor var_4563_strides_0 = const()[name = string("op_4563_strides_0"), val = tensor([1, 1])]; tensor var_4563_pad_0 = const()[name = string("op_4563_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4563_dilations_0 = const()[name = string("op_4563_dilations_0"), val = tensor([1, 1])]; int32 var_4563_groups_0 = const()[name = string("op_4563_groups_0"), val = int32(1)]; tensor var_4563 = conv(dilations = var_4563_dilations_0, groups = var_4563_groups_0, pad = var_4563_pad_0, pad_type = var_4563_pad_type_0, strides = var_4563_strides_0, weight = model_model_layers_7_self_attn_v_proj_weight_palettized, x = var_4503_cast_fp16)[name = string("op_4563")]; tensor var_4568 = const()[name = string("op_4568"), val = tensor([1, 8, 1, 128])]; tensor var_4569 = reshape(shape = var_4568, x = var_4563)[name = string("op_4569")]; int32 var_4586 = const()[name = string("op_4586"), val = int32(-1)]; fp16 const_101_promoted = const()[name = string("const_101_promoted"), val = fp16(-0x1p+0)]; tensor var_4588 = mul(x = var_4525, y = const_101_promoted)[name = string("op_4588")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131 = concat(axis = var_4586, interleave = input_131_interleave_0, values = (var_4525, var_4588))[name = string("input_131")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_4583_to_fp16 = const()[name = string("op_4583_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4583_to_fp16, x = input_131)[name = string("normed_117_cast_fp16")]; tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_119 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119")]; tensor const_103 = const()[name = string("const_103"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304063872)))]; tensor q_15 = mul(x = normed_119, y = const_103)[name = string("q_15")]; int32 var_4608 = const()[name = string("op_4608"), val = int32(-1)]; fp16 const_104_promoted = const()[name = string("const_104_promoted"), val = fp16(-0x1p+0)]; tensor var_4610 = mul(x = var_4547, y = const_104_promoted)[name = string("op_4610")]; bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; tensor input_133 = concat(axis = var_4608, interleave = input_133_interleave_0, values = (var_4547, var_4610))[name = string("input_133")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_4605_to_fp16 = const()[name = string("op_4605_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4605_to_fp16, x = input_133)[name = string("normed_121_cast_fp16")]; tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_123 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123")]; tensor const_106 = const()[name = string("const_106"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304064192)))]; tensor k_15 = mul(x = normed_123, y = const_106)[name = string("k_15")]; tensor var_4619 = mul(x = q_15, y = cos_1_cast_fp16)[name = string("op_4619")]; tensor var_4624_begin_0 = const()[name = string("op_4624_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4624_end_0 = const()[name = string("op_4624_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_4624_end_mask_0 = const()[name = string("op_4624_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4624 = slice_by_index(begin = var_4624_begin_0, end = var_4624_end_0, end_mask = var_4624_end_mask_0, x = q_15)[name = string("op_4624")]; fp16 const_107_promoted = const()[name = string("const_107_promoted"), val = fp16(-0x1p+0)]; tensor var_4625 = mul(x = var_4624, y = const_107_promoted)[name = string("op_4625")]; tensor var_4630_begin_0 = const()[name = string("op_4630_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4630_end_0 = const()[name = string("op_4630_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_4630_end_mask_0 = const()[name = string("op_4630_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4630 = slice_by_index(begin = var_4630_begin_0, end = var_4630_end_0, end_mask = var_4630_end_mask_0, x = q_15)[name = string("op_4630")]; int32 var_4632 = const()[name = string("op_4632"), val = int32(-1)]; bool var_4633_interleave_0 = const()[name = string("op_4633_interleave_0"), val = bool(false)]; tensor var_4633 = concat(axis = var_4632, interleave = var_4633_interleave_0, values = (var_4625, var_4630))[name = string("op_4633")]; tensor var_4634 = mul(x = var_4633, y = sin_1_cast_fp16)[name = string("op_4634")]; tensor query_states_29 = add(x = var_4619, y = var_4634)[name = string("query_states_29")]; tensor var_4637 = mul(x = k_15, y = cos_1_cast_fp16)[name = string("op_4637")]; tensor var_4642_begin_0 = const()[name = string("op_4642_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4642_end_0 = const()[name = string("op_4642_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_4642_end_mask_0 = const()[name = string("op_4642_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4642 = slice_by_index(begin = var_4642_begin_0, end = var_4642_end_0, end_mask = var_4642_end_mask_0, x = k_15)[name = string("op_4642")]; fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; tensor var_4643 = mul(x = var_4642, y = const_108_promoted)[name = string("op_4643")]; tensor var_4648_begin_0 = const()[name = string("op_4648_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4648_end_0 = const()[name = string("op_4648_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_4648_end_mask_0 = const()[name = string("op_4648_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4648 = slice_by_index(begin = var_4648_begin_0, end = var_4648_end_0, end_mask = var_4648_end_mask_0, x = k_15)[name = string("op_4648")]; int32 var_4650 = const()[name = string("op_4650"), val = int32(-1)]; bool var_4651_interleave_0 = const()[name = string("op_4651_interleave_0"), val = bool(false)]; tensor var_4651 = concat(axis = var_4650, interleave = var_4651_interleave_0, values = (var_4643, var_4648))[name = string("op_4651")]; tensor var_4652 = mul(x = var_4651, y = sin_1_cast_fp16)[name = string("op_4652")]; tensor key_states_29 = add(x = var_4637, y = var_4652)[name = string("key_states_29")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; int32 concat_58_axis_0 = const()[name = string("concat_58_axis_0"), val = int32(0)]; bool concat_58_interleave_0 = const()[name = string("concat_58_interleave_0"), val = bool(false)]; tensor concat_58 = concat(axis = concat_58_axis_0, interleave = concat_58_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_58")]; tensor concat_59_values1_0 = const()[name = string("concat_59_values1_0"), val = tensor([0])]; tensor concat_59_values3_0 = const()[name = string("concat_59_values3_0"), val = tensor([0])]; int32 concat_59_axis_0 = const()[name = string("concat_59_axis_0"), val = int32(0)]; bool concat_59_interleave_0 = const()[name = string("concat_59_interleave_0"), val = bool(false)]; tensor concat_59 = concat(axis = concat_59_axis_0, interleave = concat_59_interleave_0, values = (expand_dims_88, concat_59_values1_0, var_1717, concat_59_values3_0))[name = string("concat_59")]; tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_58, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_59, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_29, x = coreml_update_state_69)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_126_write_state")]; tensor coreml_update_state_70 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_126")]; tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([35])]; tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([36])]; int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_62")]; tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_94, concat_63_values1_0, var_1717, concat_63_values3_0))[name = string("concat_63")]; tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_62, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_63, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = var_4569, x = coreml_update_state_70)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_127_write_state")]; tensor coreml_update_state_71 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_127")]; tensor var_4707_begin_0 = const()[name = string("op_4707_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_4707_end_0 = const()[name = string("op_4707_end_0"), val = tensor([8, 8, 1536, 128])]; tensor var_4707_end_mask_0 = const()[name = string("op_4707_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4707_cast_fp16 = slice_by_index(begin = var_4707_begin_0, end = var_4707_end_0, end_mask = var_4707_end_mask_0, x = coreml_update_state_71)[name = string("op_4707_cast_fp16")]; tensor key_cache_15_axes_0 = const()[name = string("key_cache_15_axes_0"), val = tensor([0])]; tensor key_cache_15_cast_fp16 = squeeze(axes = key_cache_15_axes_0, x = var_4707_cast_fp16)[name = string("key_cache_15_cast_fp16")]; tensor var_4714_begin_0 = const()[name = string("op_4714_begin_0"), val = tensor([35, 0, 0, 0])]; tensor var_4714_end_0 = const()[name = string("op_4714_end_0"), val = tensor([36, 8, 1536, 128])]; tensor var_4714_end_mask_0 = const()[name = string("op_4714_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4714_cast_fp16 = slice_by_index(begin = var_4714_begin_0, end = var_4714_end_0, end_mask = var_4714_end_mask_0, x = coreml_update_state_71)[name = string("op_4714_cast_fp16")]; tensor value_cache_15_axes_0 = const()[name = string("value_cache_15_axes_0"), val = tensor([0])]; tensor value_cache_15_cast_fp16 = squeeze(axes = value_cache_15_axes_0, x = var_4714_cast_fp16)[name = string("value_cache_15_cast_fp16")]; tensor var_4738_axes_0 = const()[name = string("op_4738_axes_0"), val = tensor([1])]; tensor var_4738_cast_fp16 = expand_dims(axes = var_4738_axes_0, x = key_cache_15_cast_fp16)[name = string("op_4738_cast_fp16")]; tensor var_4743 = const()[name = string("op_4743"), val = tensor([1, 2, 1, 1])]; tensor value_59_cast_fp16 = tile(reps = var_4743, x = var_4738_cast_fp16)[name = string("value_59_cast_fp16")]; tensor var_4749 = const()[name = string("op_4749"), val = tensor([1, 16, 1536, 128])]; tensor key_states_31_cast_fp16 = reshape(shape = var_4749, x = value_59_cast_fp16)[name = string("key_states_31_cast_fp16")]; tensor var_4752_axes_0 = const()[name = string("op_4752_axes_0"), val = tensor([1])]; tensor var_4752_cast_fp16 = expand_dims(axes = var_4752_axes_0, x = value_cache_15_cast_fp16)[name = string("op_4752_cast_fp16")]; tensor var_4757 = const()[name = string("op_4757"), val = tensor([1, 2, 1, 1])]; tensor value_63_cast_fp16 = tile(reps = var_4757, x = var_4752_cast_fp16)[name = string("value_63_cast_fp16")]; tensor var_4763 = const()[name = string("op_4763"), val = tensor([1, 16, 1536, 128])]; tensor value_states_45_cast_fp16 = reshape(shape = var_4763, x = value_63_cast_fp16)[name = string("value_states_45_cast_fp16")]; bool var_4778_transpose_x_1 = const()[name = string("op_4778_transpose_x_1"), val = bool(false)]; bool var_4778_transpose_y_1 = const()[name = string("op_4778_transpose_y_1"), val = bool(true)]; tensor var_4778 = matmul(transpose_x = var_4778_transpose_x_1, transpose_y = var_4778_transpose_y_1, x = query_states_29, y = key_states_31_cast_fp16)[name = string("op_4778")]; fp16 var_4779_to_fp16 = const()[name = string("op_4779_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_29_cast_fp16 = mul(x = var_4778, y = var_4779_to_fp16)[name = string("attention_29_cast_fp16")]; tensor attention_31_cast_fp16 = add(x = attention_29_cast_fp16, y = causal_mask)[name = string("attention_31_cast_fp16")]; int32 var_4788 = const()[name = string("op_4788"), val = int32(-1)]; tensor probabilities_15_cast_fp16 = softmax(axis = var_4788, x = attention_31_cast_fp16)[name = string("probabilities_15_cast_fp16")]; bool output_43_transpose_x_0 = const()[name = string("output_43_transpose_x_0"), val = bool(false)]; bool output_43_transpose_y_0 = const()[name = string("output_43_transpose_y_0"), val = bool(false)]; tensor output_43_cast_fp16 = matmul(transpose_x = output_43_transpose_x_0, transpose_y = output_43_transpose_y_0, x = probabilities_15_cast_fp16, y = value_states_45_cast_fp16)[name = string("output_43_cast_fp16")]; tensor var_4799_perm_0 = const()[name = string("op_4799_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4805 = const()[name = string("op_4805"), val = tensor([1, 1, 2048])]; tensor var_4799_cast_fp16 = transpose(perm = var_4799_perm_0, x = output_43_cast_fp16)[name = string("transpose_124")]; tensor output_45_cast_fp16 = reshape(shape = var_4805, x = var_4799_cast_fp16)[name = string("output_45_cast_fp16")]; tensor var_4810 = const()[name = string("op_4810"), val = tensor([0, 2, 1])]; string var_4826_pad_type_0 = const()[name = string("op_4826_pad_type_0"), val = string("valid")]; int32 var_4826_groups_0 = const()[name = string("op_4826_groups_0"), val = int32(1)]; tensor var_4826_strides_0 = const()[name = string("op_4826_strides_0"), val = tensor([1])]; tensor var_4826_pad_0 = const()[name = string("op_4826_pad_0"), val = tensor([0, 0])]; tensor var_4826_dilations_0 = const()[name = string("op_4826_dilations_0"), val = tensor([1])]; tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304064512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305637440))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4811_cast_fp16 = transpose(perm = var_4810, x = output_45_cast_fp16)[name = string("transpose_123")]; tensor var_4826_cast_fp16 = conv(dilations = var_4826_dilations_0, groups = var_4826_groups_0, pad = var_4826_pad_0, pad_type = var_4826_pad_type_0, strides = var_4826_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_4811_cast_fp16)[name = string("op_4826_cast_fp16")]; tensor var_4830 = const()[name = string("op_4830"), val = tensor([0, 2, 1])]; tensor attn_output_15_cast_fp16 = transpose(perm = var_4830, x = var_4826_cast_fp16)[name = string("transpose_122")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_71_cast_fp16, y = attn_output_15_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; int32 var_4845 = const()[name = string("op_4845"), val = int32(-1)]; fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4847_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_109_promoted_to_fp16)[name = string("op_4847_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_4845, interleave = input_137_interleave_0, values = (hidden_states_79_cast_fp16, var_4847_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_4842_to_fp16 = const()[name = string("op_4842_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_4842_to_fp16, x = input_137_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 1, 1024])]; tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; tensor const_111_promoted_to_fp16 = const()[name = string("const_111_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305653888)))]; tensor x_29_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_111_promoted_to_fp16)[name = string("x_29_cast_fp16")]; tensor var_4867 = const()[name = string("op_4867"), val = tensor([0, 2, 1])]; tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; tensor var_4868 = transpose(perm = var_4867, x = x_29_cast_fp16)[name = string("transpose_121")]; tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_4868)[name = string("input_139")]; string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; tensor input_141 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = model_model_layers_7_mlp_gate_proj_weight_palettized, x = input_139)[name = string("input_141")]; string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_7_mlp_up_proj_weight_palettized, x = input_139)[name = string("b_15")]; tensor c_15 = silu(x = input_141)[name = string("c_15")]; tensor input_143 = mul(x = c_15, y = b_15)[name = string("input_143")]; string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_7_mlp_down_proj_weight_palettized, x = input_143)[name = string("e_15")]; tensor var_4890_axes_0 = const()[name = string("op_4890_axes_0"), val = tensor([2])]; tensor var_4890 = squeeze(axes = var_4890_axes_0, x = e_15)[name = string("op_4890")]; tensor var_4891 = const()[name = string("op_4891"), val = tensor([0, 2, 1])]; tensor var_4892 = transpose(perm = var_4891, x = var_4890)[name = string("transpose_120")]; tensor hidden_states_81_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = var_4892)[name = string("hidden_states_81_cast_fp16")]; int32 var_4906 = const()[name = string("op_4906"), val = int32(-1)]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4908_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_112_promoted_to_fp16)[name = string("op_4908_cast_fp16")]; bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; tensor input_145_cast_fp16 = concat(axis = var_4906, interleave = input_145_interleave_0, values = (hidden_states_81_cast_fp16, var_4908_cast_fp16))[name = string("input_145_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_4903_to_fp16 = const()[name = string("op_4903_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_4903_to_fp16, x = input_145_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor normed_131_begin_0 = const()[name = string("normed_131_begin_0"), val = tensor([0, 0, 0])]; tensor normed_131_end_0 = const()[name = string("normed_131_end_0"), val = tensor([1, 1, 1024])]; tensor normed_131_end_mask_0 = const()[name = string("normed_131_end_mask_0"), val = tensor([true, true, false])]; tensor normed_131_cast_fp16 = slice_by_index(begin = normed_131_begin_0, end = normed_131_end_0, end_mask = normed_131_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_131_cast_fp16")]; tensor const_114_promoted_to_fp16 = const()[name = string("const_114_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305656000)))]; tensor hidden_states_83_cast_fp16 = mul(x = normed_131_cast_fp16, y = const_114_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor var_4920 = const()[name = string("op_4920"), val = tensor([0, 2, 1])]; tensor var_4923_axes_0 = const()[name = string("op_4923_axes_0"), val = tensor([2])]; tensor var_4921_cast_fp16 = transpose(perm = var_4920, x = hidden_states_83_cast_fp16)[name = string("transpose_119")]; tensor var_4923_cast_fp16 = expand_dims(axes = var_4923_axes_0, x = var_4921_cast_fp16)[name = string("op_4923_cast_fp16")]; string var_4939_pad_type_0 = const()[name = string("op_4939_pad_type_0"), val = string("valid")]; tensor var_4939_strides_0 = const()[name = string("op_4939_strides_0"), val = tensor([1, 1])]; tensor var_4939_pad_0 = const()[name = string("op_4939_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4939_dilations_0 = const()[name = string("op_4939_dilations_0"), val = tensor([1, 1])]; int32 var_4939_groups_0 = const()[name = string("op_4939_groups_0"), val = int32(1)]; tensor var_4939 = conv(dilations = var_4939_dilations_0, groups = var_4939_groups_0, pad = var_4939_pad_0, pad_type = var_4939_pad_type_0, strides = var_4939_strides_0, weight = model_model_layers_8_self_attn_q_proj_weight_palettized, x = var_4923_cast_fp16)[name = string("op_4939")]; tensor var_4944 = const()[name = string("op_4944"), val = tensor([1, 16, 1, 128])]; tensor var_4945 = reshape(shape = var_4944, x = var_4939)[name = string("op_4945")]; string var_4961_pad_type_0 = const()[name = string("op_4961_pad_type_0"), val = string("valid")]; tensor var_4961_strides_0 = const()[name = string("op_4961_strides_0"), val = tensor([1, 1])]; tensor var_4961_pad_0 = const()[name = string("op_4961_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4961_dilations_0 = const()[name = string("op_4961_dilations_0"), val = tensor([1, 1])]; int32 var_4961_groups_0 = const()[name = string("op_4961_groups_0"), val = int32(1)]; tensor var_4961 = conv(dilations = var_4961_dilations_0, groups = var_4961_groups_0, pad = var_4961_pad_0, pad_type = var_4961_pad_type_0, strides = var_4961_strides_0, weight = model_model_layers_8_self_attn_k_proj_weight_palettized, x = var_4923_cast_fp16)[name = string("op_4961")]; tensor var_4966 = const()[name = string("op_4966"), val = tensor([1, 8, 1, 128])]; tensor var_4967 = reshape(shape = var_4966, x = var_4961)[name = string("op_4967")]; string var_4983_pad_type_0 = const()[name = string("op_4983_pad_type_0"), val = string("valid")]; tensor var_4983_strides_0 = const()[name = string("op_4983_strides_0"), val = tensor([1, 1])]; tensor var_4983_pad_0 = const()[name = string("op_4983_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4983_dilations_0 = const()[name = string("op_4983_dilations_0"), val = tensor([1, 1])]; int32 var_4983_groups_0 = const()[name = string("op_4983_groups_0"), val = int32(1)]; tensor var_4983 = conv(dilations = var_4983_dilations_0, groups = var_4983_groups_0, pad = var_4983_pad_0, pad_type = var_4983_pad_type_0, strides = var_4983_strides_0, weight = model_model_layers_8_self_attn_v_proj_weight_palettized, x = var_4923_cast_fp16)[name = string("op_4983")]; tensor var_4988 = const()[name = string("op_4988"), val = tensor([1, 8, 1, 128])]; tensor var_4989 = reshape(shape = var_4988, x = var_4983)[name = string("op_4989")]; int32 var_5006 = const()[name = string("op_5006"), val = int32(-1)]; fp16 const_115_promoted = const()[name = string("const_115_promoted"), val = fp16(-0x1p+0)]; tensor var_5008 = mul(x = var_4945, y = const_115_promoted)[name = string("op_5008")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149 = concat(axis = var_5006, interleave = input_149_interleave_0, values = (var_4945, var_5008))[name = string("input_149")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_5003_to_fp16 = const()[name = string("op_5003_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_5003_to_fp16, x = input_149)[name = string("normed_133_cast_fp16")]; tensor normed_135_begin_0 = const()[name = string("normed_135_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_135_end_0 = const()[name = string("normed_135_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_135_end_mask_0 = const()[name = string("normed_135_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_135 = slice_by_index(begin = normed_135_begin_0, end = normed_135_end_0, end_mask = normed_135_end_mask_0, x = normed_133_cast_fp16)[name = string("normed_135")]; tensor const_117 = const()[name = string("const_117"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305658112)))]; tensor q_17 = mul(x = normed_135, y = const_117)[name = string("q_17")]; int32 var_5028 = const()[name = string("op_5028"), val = int32(-1)]; fp16 const_118_promoted = const()[name = string("const_118_promoted"), val = fp16(-0x1p+0)]; tensor var_5030 = mul(x = var_4967, y = const_118_promoted)[name = string("op_5030")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151 = concat(axis = var_5028, interleave = input_151_interleave_0, values = (var_4967, var_5030))[name = string("input_151")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_5025_to_fp16 = const()[name = string("op_5025_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_5025_to_fp16, x = input_151)[name = string("normed_137_cast_fp16")]; tensor normed_139_begin_0 = const()[name = string("normed_139_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_139_end_0 = const()[name = string("normed_139_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_139_end_mask_0 = const()[name = string("normed_139_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_139 = slice_by_index(begin = normed_139_begin_0, end = normed_139_end_0, end_mask = normed_139_end_mask_0, x = normed_137_cast_fp16)[name = string("normed_139")]; tensor const_120 = const()[name = string("const_120"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305658432)))]; tensor k_17 = mul(x = normed_139, y = const_120)[name = string("k_17")]; tensor var_5039 = mul(x = q_17, y = cos_1_cast_fp16)[name = string("op_5039")]; tensor var_5044_begin_0 = const()[name = string("op_5044_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5044_end_0 = const()[name = string("op_5044_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_5044_end_mask_0 = const()[name = string("op_5044_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5044 = slice_by_index(begin = var_5044_begin_0, end = var_5044_end_0, end_mask = var_5044_end_mask_0, x = q_17)[name = string("op_5044")]; fp16 const_121_promoted = const()[name = string("const_121_promoted"), val = fp16(-0x1p+0)]; tensor var_5045 = mul(x = var_5044, y = const_121_promoted)[name = string("op_5045")]; tensor var_5050_begin_0 = const()[name = string("op_5050_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5050_end_0 = const()[name = string("op_5050_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_5050_end_mask_0 = const()[name = string("op_5050_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5050 = slice_by_index(begin = var_5050_begin_0, end = var_5050_end_0, end_mask = var_5050_end_mask_0, x = q_17)[name = string("op_5050")]; int32 var_5052 = const()[name = string("op_5052"), val = int32(-1)]; bool var_5053_interleave_0 = const()[name = string("op_5053_interleave_0"), val = bool(false)]; tensor var_5053 = concat(axis = var_5052, interleave = var_5053_interleave_0, values = (var_5045, var_5050))[name = string("op_5053")]; tensor var_5054 = mul(x = var_5053, y = sin_1_cast_fp16)[name = string("op_5054")]; tensor query_states_33 = add(x = var_5039, y = var_5054)[name = string("query_states_33")]; tensor var_5057 = mul(x = k_17, y = cos_1_cast_fp16)[name = string("op_5057")]; tensor var_5062_begin_0 = const()[name = string("op_5062_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5062_end_0 = const()[name = string("op_5062_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_5062_end_mask_0 = const()[name = string("op_5062_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5062 = slice_by_index(begin = var_5062_begin_0, end = var_5062_end_0, end_mask = var_5062_end_mask_0, x = k_17)[name = string("op_5062")]; fp16 const_122_promoted = const()[name = string("const_122_promoted"), val = fp16(-0x1p+0)]; tensor var_5063 = mul(x = var_5062, y = const_122_promoted)[name = string("op_5063")]; tensor var_5068_begin_0 = const()[name = string("op_5068_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5068_end_0 = const()[name = string("op_5068_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_5068_end_mask_0 = const()[name = string("op_5068_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5068 = slice_by_index(begin = var_5068_begin_0, end = var_5068_end_0, end_mask = var_5068_end_mask_0, x = k_17)[name = string("op_5068")]; int32 var_5070 = const()[name = string("op_5070"), val = int32(-1)]; bool var_5071_interleave_0 = const()[name = string("op_5071_interleave_0"), val = bool(false)]; tensor var_5071 = concat(axis = var_5070, interleave = var_5071_interleave_0, values = (var_5063, var_5068))[name = string("op_5071")]; tensor var_5072 = mul(x = var_5071, y = sin_1_cast_fp16)[name = string("op_5072")]; tensor key_states_33 = add(x = var_5057, y = var_5072)[name = string("key_states_33")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_66")]; tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_100, concat_67_values1_0, var_1717, concat_67_values3_0))[name = string("concat_67")]; tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_66, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_67, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_33, x = coreml_update_state_71)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_128_write_state")]; tensor coreml_update_state_72 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_128")]; tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([36])]; tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([37])]; int32 concat_70_axis_0 = const()[name = string("concat_70_axis_0"), val = int32(0)]; bool concat_70_interleave_0 = const()[name = string("concat_70_interleave_0"), val = bool(false)]; tensor concat_70 = concat(axis = concat_70_axis_0, interleave = concat_70_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_70")]; tensor concat_71_values1_0 = const()[name = string("concat_71_values1_0"), val = tensor([0])]; tensor concat_71_values3_0 = const()[name = string("concat_71_values3_0"), val = tensor([0])]; int32 concat_71_axis_0 = const()[name = string("concat_71_axis_0"), val = int32(0)]; bool concat_71_interleave_0 = const()[name = string("concat_71_interleave_0"), val = bool(false)]; tensor concat_71 = concat(axis = concat_71_axis_0, interleave = concat_71_interleave_0, values = (expand_dims_106, concat_71_values1_0, var_1717, concat_71_values3_0))[name = string("concat_71")]; tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_70, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_71, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = var_4989, x = coreml_update_state_72)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_129_write_state")]; tensor coreml_update_state_73 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_129")]; tensor var_5127_begin_0 = const()[name = string("op_5127_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_5127_end_0 = const()[name = string("op_5127_end_0"), val = tensor([9, 8, 1536, 128])]; tensor var_5127_end_mask_0 = const()[name = string("op_5127_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5127_cast_fp16 = slice_by_index(begin = var_5127_begin_0, end = var_5127_end_0, end_mask = var_5127_end_mask_0, x = coreml_update_state_73)[name = string("op_5127_cast_fp16")]; tensor key_cache_17_axes_0 = const()[name = string("key_cache_17_axes_0"), val = tensor([0])]; tensor key_cache_17_cast_fp16 = squeeze(axes = key_cache_17_axes_0, x = var_5127_cast_fp16)[name = string("key_cache_17_cast_fp16")]; tensor var_5134_begin_0 = const()[name = string("op_5134_begin_0"), val = tensor([36, 0, 0, 0])]; tensor var_5134_end_0 = const()[name = string("op_5134_end_0"), val = tensor([37, 8, 1536, 128])]; tensor var_5134_end_mask_0 = const()[name = string("op_5134_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5134_cast_fp16 = slice_by_index(begin = var_5134_begin_0, end = var_5134_end_0, end_mask = var_5134_end_mask_0, x = coreml_update_state_73)[name = string("op_5134_cast_fp16")]; tensor value_cache_17_axes_0 = const()[name = string("value_cache_17_axes_0"), val = tensor([0])]; tensor value_cache_17_cast_fp16 = squeeze(axes = value_cache_17_axes_0, x = var_5134_cast_fp16)[name = string("value_cache_17_cast_fp16")]; tensor var_5158_axes_0 = const()[name = string("op_5158_axes_0"), val = tensor([1])]; tensor var_5158_cast_fp16 = expand_dims(axes = var_5158_axes_0, x = key_cache_17_cast_fp16)[name = string("op_5158_cast_fp16")]; tensor var_5163 = const()[name = string("op_5163"), val = tensor([1, 2, 1, 1])]; tensor value_67_cast_fp16 = tile(reps = var_5163, x = var_5158_cast_fp16)[name = string("value_67_cast_fp16")]; tensor var_5169 = const()[name = string("op_5169"), val = tensor([1, 16, 1536, 128])]; tensor key_states_35_cast_fp16 = reshape(shape = var_5169, x = value_67_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor var_5172_axes_0 = const()[name = string("op_5172_axes_0"), val = tensor([1])]; tensor var_5172_cast_fp16 = expand_dims(axes = var_5172_axes_0, x = value_cache_17_cast_fp16)[name = string("op_5172_cast_fp16")]; tensor var_5177 = const()[name = string("op_5177"), val = tensor([1, 2, 1, 1])]; tensor value_71_cast_fp16 = tile(reps = var_5177, x = var_5172_cast_fp16)[name = string("value_71_cast_fp16")]; tensor var_5183 = const()[name = string("op_5183"), val = tensor([1, 16, 1536, 128])]; tensor value_states_51_cast_fp16 = reshape(shape = var_5183, x = value_71_cast_fp16)[name = string("value_states_51_cast_fp16")]; bool var_5198_transpose_x_1 = const()[name = string("op_5198_transpose_x_1"), val = bool(false)]; bool var_5198_transpose_y_1 = const()[name = string("op_5198_transpose_y_1"), val = bool(true)]; tensor var_5198 = matmul(transpose_x = var_5198_transpose_x_1, transpose_y = var_5198_transpose_y_1, x = query_states_33, y = key_states_35_cast_fp16)[name = string("op_5198")]; fp16 var_5199_to_fp16 = const()[name = string("op_5199_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_33_cast_fp16 = mul(x = var_5198, y = var_5199_to_fp16)[name = string("attention_33_cast_fp16")]; tensor attention_35_cast_fp16 = add(x = attention_33_cast_fp16, y = causal_mask)[name = string("attention_35_cast_fp16")]; int32 var_5208 = const()[name = string("op_5208"), val = int32(-1)]; tensor probabilities_17_cast_fp16 = softmax(axis = var_5208, x = attention_35_cast_fp16)[name = string("probabilities_17_cast_fp16")]; bool output_49_transpose_x_0 = const()[name = string("output_49_transpose_x_0"), val = bool(false)]; bool output_49_transpose_y_0 = const()[name = string("output_49_transpose_y_0"), val = bool(false)]; tensor output_49_cast_fp16 = matmul(transpose_x = output_49_transpose_x_0, transpose_y = output_49_transpose_y_0, x = probabilities_17_cast_fp16, y = value_states_51_cast_fp16)[name = string("output_49_cast_fp16")]; tensor var_5219_perm_0 = const()[name = string("op_5219_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_5225 = const()[name = string("op_5225"), val = tensor([1, 1, 2048])]; tensor var_5219_cast_fp16 = transpose(perm = var_5219_perm_0, x = output_49_cast_fp16)[name = string("transpose_118")]; tensor output_51_cast_fp16 = reshape(shape = var_5225, x = var_5219_cast_fp16)[name = string("output_51_cast_fp16")]; tensor var_5230 = const()[name = string("op_5230"), val = tensor([0, 2, 1])]; string var_5246_pad_type_0 = const()[name = string("op_5246_pad_type_0"), val = string("valid")]; int32 var_5246_groups_0 = const()[name = string("op_5246_groups_0"), val = int32(1)]; tensor var_5246_strides_0 = const()[name = string("op_5246_strides_0"), val = tensor([1])]; tensor var_5246_pad_0 = const()[name = string("op_5246_pad_0"), val = tensor([0, 0])]; tensor var_5246_dilations_0 = const()[name = string("op_5246_dilations_0"), val = tensor([1])]; tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305658752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307231680))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5231_cast_fp16 = transpose(perm = var_5230, x = output_51_cast_fp16)[name = string("transpose_117")]; tensor var_5246_cast_fp16 = conv(dilations = var_5246_dilations_0, groups = var_5246_groups_0, pad = var_5246_pad_0, pad_type = var_5246_pad_type_0, strides = var_5246_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5231_cast_fp16)[name = string("op_5246_cast_fp16")]; tensor var_5250 = const()[name = string("op_5250"), val = tensor([0, 2, 1])]; tensor attn_output_17_cast_fp16 = transpose(perm = var_5250, x = var_5246_cast_fp16)[name = string("transpose_116")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = attn_output_17_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; int32 var_5265 = const()[name = string("op_5265"), val = int32(-1)]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5267_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_123_promoted_to_fp16)[name = string("op_5267_cast_fp16")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155_cast_fp16 = concat(axis = var_5265, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_5267_cast_fp16))[name = string("input_155_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_5262_to_fp16 = const()[name = string("op_5262_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_5262_to_fp16, x = input_155_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor normed_143_begin_0 = const()[name = string("normed_143_begin_0"), val = tensor([0, 0, 0])]; tensor normed_143_end_0 = const()[name = string("normed_143_end_0"), val = tensor([1, 1, 1024])]; tensor normed_143_end_mask_0 = const()[name = string("normed_143_end_mask_0"), val = tensor([true, true, false])]; tensor normed_143_cast_fp16 = slice_by_index(begin = normed_143_begin_0, end = normed_143_end_0, end_mask = normed_143_end_mask_0, x = normed_141_cast_fp16)[name = string("normed_143_cast_fp16")]; tensor const_125_promoted_to_fp16 = const()[name = string("const_125_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307248128)))]; tensor x_33_cast_fp16 = mul(x = normed_143_cast_fp16, y = const_125_promoted_to_fp16)[name = string("x_33_cast_fp16")]; tensor var_5287 = const()[name = string("op_5287"), val = tensor([0, 2, 1])]; tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; tensor var_5288 = transpose(perm = var_5287, x = x_33_cast_fp16)[name = string("transpose_115")]; tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_5288)[name = string("input_157")]; string input_159_pad_type_0 = const()[name = string("input_159_pad_type_0"), val = string("valid")]; tensor input_159_strides_0 = const()[name = string("input_159_strides_0"), val = tensor([1, 1])]; tensor input_159_pad_0 = const()[name = string("input_159_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_159_dilations_0 = const()[name = string("input_159_dilations_0"), val = tensor([1, 1])]; int32 input_159_groups_0 = const()[name = string("input_159_groups_0"), val = int32(1)]; tensor input_159 = conv(dilations = input_159_dilations_0, groups = input_159_groups_0, pad = input_159_pad_0, pad_type = input_159_pad_type_0, strides = input_159_strides_0, weight = model_model_layers_8_mlp_gate_proj_weight_palettized, x = input_157)[name = string("input_159")]; string b_17_pad_type_0 = const()[name = string("b_17_pad_type_0"), val = string("valid")]; tensor b_17_strides_0 = const()[name = string("b_17_strides_0"), val = tensor([1, 1])]; tensor b_17_pad_0 = const()[name = string("b_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_17_dilations_0 = const()[name = string("b_17_dilations_0"), val = tensor([1, 1])]; int32 b_17_groups_0 = const()[name = string("b_17_groups_0"), val = int32(1)]; tensor b_17 = conv(dilations = b_17_dilations_0, groups = b_17_groups_0, pad = b_17_pad_0, pad_type = b_17_pad_type_0, strides = b_17_strides_0, weight = model_model_layers_8_mlp_up_proj_weight_palettized, x = input_157)[name = string("b_17")]; tensor c_17 = silu(x = input_159)[name = string("c_17")]; tensor input_161 = mul(x = c_17, y = b_17)[name = string("input_161")]; string e_17_pad_type_0 = const()[name = string("e_17_pad_type_0"), val = string("valid")]; tensor e_17_strides_0 = const()[name = string("e_17_strides_0"), val = tensor([1, 1])]; tensor e_17_pad_0 = const()[name = string("e_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_17_dilations_0 = const()[name = string("e_17_dilations_0"), val = tensor([1, 1])]; int32 e_17_groups_0 = const()[name = string("e_17_groups_0"), val = int32(1)]; tensor e_17 = conv(dilations = e_17_dilations_0, groups = e_17_groups_0, pad = e_17_pad_0, pad_type = e_17_pad_type_0, strides = e_17_strides_0, weight = model_model_layers_8_mlp_down_proj_weight_palettized, x = input_161)[name = string("e_17")]; tensor var_5310_axes_0 = const()[name = string("op_5310_axes_0"), val = tensor([2])]; tensor var_5310 = squeeze(axes = var_5310_axes_0, x = e_17)[name = string("op_5310")]; tensor var_5311 = const()[name = string("op_5311"), val = tensor([0, 2, 1])]; tensor var_5312 = transpose(perm = var_5311, x = var_5310)[name = string("transpose_114")]; tensor hidden_states_91_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = var_5312)[name = string("hidden_states_91_cast_fp16")]; int32 var_5326 = const()[name = string("op_5326"), val = int32(-1)]; fp16 const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5328_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_126_promoted_to_fp16)[name = string("op_5328_cast_fp16")]; bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; tensor input_163_cast_fp16 = concat(axis = var_5326, interleave = input_163_interleave_0, values = (hidden_states_91_cast_fp16, var_5328_cast_fp16))[name = string("input_163_cast_fp16")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_5323_to_fp16 = const()[name = string("op_5323_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_5323_to_fp16, x = input_163_cast_fp16)[name = string("normed_145_cast_fp16")]; tensor normed_147_begin_0 = const()[name = string("normed_147_begin_0"), val = tensor([0, 0, 0])]; tensor normed_147_end_0 = const()[name = string("normed_147_end_0"), val = tensor([1, 1, 1024])]; tensor normed_147_end_mask_0 = const()[name = string("normed_147_end_mask_0"), val = tensor([true, true, false])]; tensor normed_147_cast_fp16 = slice_by_index(begin = normed_147_begin_0, end = normed_147_end_0, end_mask = normed_147_end_mask_0, x = normed_145_cast_fp16)[name = string("normed_147_cast_fp16")]; tensor const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307250240)))]; tensor hidden_states_93_cast_fp16 = mul(x = normed_147_cast_fp16, y = const_128_promoted_to_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor var_5340 = const()[name = string("op_5340"), val = tensor([0, 2, 1])]; tensor var_5343_axes_0 = const()[name = string("op_5343_axes_0"), val = tensor([2])]; tensor var_5341_cast_fp16 = transpose(perm = var_5340, x = hidden_states_93_cast_fp16)[name = string("transpose_113")]; tensor var_5343_cast_fp16 = expand_dims(axes = var_5343_axes_0, x = var_5341_cast_fp16)[name = string("op_5343_cast_fp16")]; string var_5359_pad_type_0 = const()[name = string("op_5359_pad_type_0"), val = string("valid")]; tensor var_5359_strides_0 = const()[name = string("op_5359_strides_0"), val = tensor([1, 1])]; tensor var_5359_pad_0 = const()[name = string("op_5359_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5359_dilations_0 = const()[name = string("op_5359_dilations_0"), val = tensor([1, 1])]; int32 var_5359_groups_0 = const()[name = string("op_5359_groups_0"), val = int32(1)]; tensor var_5359 = conv(dilations = var_5359_dilations_0, groups = var_5359_groups_0, pad = var_5359_pad_0, pad_type = var_5359_pad_type_0, strides = var_5359_strides_0, weight = model_model_layers_9_self_attn_q_proj_weight_palettized, x = var_5343_cast_fp16)[name = string("op_5359")]; tensor var_5364 = const()[name = string("op_5364"), val = tensor([1, 16, 1, 128])]; tensor var_5365 = reshape(shape = var_5364, x = var_5359)[name = string("op_5365")]; string var_5381_pad_type_0 = const()[name = string("op_5381_pad_type_0"), val = string("valid")]; tensor var_5381_strides_0 = const()[name = string("op_5381_strides_0"), val = tensor([1, 1])]; tensor var_5381_pad_0 = const()[name = string("op_5381_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5381_dilations_0 = const()[name = string("op_5381_dilations_0"), val = tensor([1, 1])]; int32 var_5381_groups_0 = const()[name = string("op_5381_groups_0"), val = int32(1)]; tensor var_5381 = conv(dilations = var_5381_dilations_0, groups = var_5381_groups_0, pad = var_5381_pad_0, pad_type = var_5381_pad_type_0, strides = var_5381_strides_0, weight = model_model_layers_9_self_attn_k_proj_weight_palettized, x = var_5343_cast_fp16)[name = string("op_5381")]; tensor var_5386 = const()[name = string("op_5386"), val = tensor([1, 8, 1, 128])]; tensor var_5387 = reshape(shape = var_5386, x = var_5381)[name = string("op_5387")]; string var_5403_pad_type_0 = const()[name = string("op_5403_pad_type_0"), val = string("valid")]; tensor var_5403_strides_0 = const()[name = string("op_5403_strides_0"), val = tensor([1, 1])]; tensor var_5403_pad_0 = const()[name = string("op_5403_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5403_dilations_0 = const()[name = string("op_5403_dilations_0"), val = tensor([1, 1])]; int32 var_5403_groups_0 = const()[name = string("op_5403_groups_0"), val = int32(1)]; tensor var_5403 = conv(dilations = var_5403_dilations_0, groups = var_5403_groups_0, pad = var_5403_pad_0, pad_type = var_5403_pad_type_0, strides = var_5403_strides_0, weight = model_model_layers_9_self_attn_v_proj_weight_palettized, x = var_5343_cast_fp16)[name = string("op_5403")]; tensor var_5408 = const()[name = string("op_5408"), val = tensor([1, 8, 1, 128])]; tensor var_5409 = reshape(shape = var_5408, x = var_5403)[name = string("op_5409")]; int32 var_5426 = const()[name = string("op_5426"), val = int32(-1)]; fp16 const_129_promoted = const()[name = string("const_129_promoted"), val = fp16(-0x1p+0)]; tensor var_5428 = mul(x = var_5365, y = const_129_promoted)[name = string("op_5428")]; bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; tensor input_167 = concat(axis = var_5426, interleave = input_167_interleave_0, values = (var_5365, var_5428))[name = string("input_167")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_5423_to_fp16 = const()[name = string("op_5423_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_5423_to_fp16, x = input_167)[name = string("normed_149_cast_fp16")]; tensor normed_151_begin_0 = const()[name = string("normed_151_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_151_end_0 = const()[name = string("normed_151_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_151_end_mask_0 = const()[name = string("normed_151_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_151 = slice_by_index(begin = normed_151_begin_0, end = normed_151_end_0, end_mask = normed_151_end_mask_0, x = normed_149_cast_fp16)[name = string("normed_151")]; tensor const_131 = const()[name = string("const_131"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307252352)))]; tensor q_19 = mul(x = normed_151, y = const_131)[name = string("q_19")]; int32 var_5448 = const()[name = string("op_5448"), val = int32(-1)]; fp16 const_132_promoted = const()[name = string("const_132_promoted"), val = fp16(-0x1p+0)]; tensor var_5450 = mul(x = var_5387, y = const_132_promoted)[name = string("op_5450")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169 = concat(axis = var_5448, interleave = input_169_interleave_0, values = (var_5387, var_5450))[name = string("input_169")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_5445_to_fp16 = const()[name = string("op_5445_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_5445_to_fp16, x = input_169)[name = string("normed_153_cast_fp16")]; tensor normed_155_begin_0 = const()[name = string("normed_155_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_155_end_0 = const()[name = string("normed_155_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_155_end_mask_0 = const()[name = string("normed_155_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_155 = slice_by_index(begin = normed_155_begin_0, end = normed_155_end_0, end_mask = normed_155_end_mask_0, x = normed_153_cast_fp16)[name = string("normed_155")]; tensor const_134 = const()[name = string("const_134"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307252672)))]; tensor k_19 = mul(x = normed_155, y = const_134)[name = string("k_19")]; tensor var_5459 = mul(x = q_19, y = cos_1_cast_fp16)[name = string("op_5459")]; tensor var_5464_begin_0 = const()[name = string("op_5464_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5464_end_0 = const()[name = string("op_5464_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_5464_end_mask_0 = const()[name = string("op_5464_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5464 = slice_by_index(begin = var_5464_begin_0, end = var_5464_end_0, end_mask = var_5464_end_mask_0, x = q_19)[name = string("op_5464")]; fp16 const_135_promoted = const()[name = string("const_135_promoted"), val = fp16(-0x1p+0)]; tensor var_5465 = mul(x = var_5464, y = const_135_promoted)[name = string("op_5465")]; tensor var_5470_begin_0 = const()[name = string("op_5470_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5470_end_0 = const()[name = string("op_5470_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_5470_end_mask_0 = const()[name = string("op_5470_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5470 = slice_by_index(begin = var_5470_begin_0, end = var_5470_end_0, end_mask = var_5470_end_mask_0, x = q_19)[name = string("op_5470")]; int32 var_5472 = const()[name = string("op_5472"), val = int32(-1)]; bool var_5473_interleave_0 = const()[name = string("op_5473_interleave_0"), val = bool(false)]; tensor var_5473 = concat(axis = var_5472, interleave = var_5473_interleave_0, values = (var_5465, var_5470))[name = string("op_5473")]; tensor var_5474 = mul(x = var_5473, y = sin_1_cast_fp16)[name = string("op_5474")]; tensor query_states_37 = add(x = var_5459, y = var_5474)[name = string("query_states_37")]; tensor var_5477 = mul(x = k_19, y = cos_1_cast_fp16)[name = string("op_5477")]; tensor var_5482_begin_0 = const()[name = string("op_5482_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5482_end_0 = const()[name = string("op_5482_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_5482_end_mask_0 = const()[name = string("op_5482_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5482 = slice_by_index(begin = var_5482_begin_0, end = var_5482_end_0, end_mask = var_5482_end_mask_0, x = k_19)[name = string("op_5482")]; fp16 const_136_promoted = const()[name = string("const_136_promoted"), val = fp16(-0x1p+0)]; tensor var_5483 = mul(x = var_5482, y = const_136_promoted)[name = string("op_5483")]; tensor var_5488_begin_0 = const()[name = string("op_5488_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5488_end_0 = const()[name = string("op_5488_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_5488_end_mask_0 = const()[name = string("op_5488_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5488 = slice_by_index(begin = var_5488_begin_0, end = var_5488_end_0, end_mask = var_5488_end_mask_0, x = k_19)[name = string("op_5488")]; int32 var_5490 = const()[name = string("op_5490"), val = int32(-1)]; bool var_5491_interleave_0 = const()[name = string("op_5491_interleave_0"), val = bool(false)]; tensor var_5491 = concat(axis = var_5490, interleave = var_5491_interleave_0, values = (var_5483, var_5488))[name = string("op_5491")]; tensor var_5492 = mul(x = var_5491, y = sin_1_cast_fp16)[name = string("op_5492")]; tensor key_states_37 = add(x = var_5477, y = var_5492)[name = string("key_states_37")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_74")]; tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_112, concat_75_values1_0, var_1717, concat_75_values3_0))[name = string("concat_75")]; tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = key_states_37, x = coreml_update_state_73)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_130_write_state")]; tensor coreml_update_state_74 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_130")]; tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([37])]; tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([38])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_78")]; tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_118, concat_79_values1_0, var_1717, concat_79_values3_0))[name = string("concat_79")]; tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = var_5409, x = coreml_update_state_74)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_131_write_state")]; tensor coreml_update_state_75 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_131")]; tensor var_5547_begin_0 = const()[name = string("op_5547_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_5547_end_0 = const()[name = string("op_5547_end_0"), val = tensor([10, 8, 1536, 128])]; tensor var_5547_end_mask_0 = const()[name = string("op_5547_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5547_cast_fp16 = slice_by_index(begin = var_5547_begin_0, end = var_5547_end_0, end_mask = var_5547_end_mask_0, x = coreml_update_state_75)[name = string("op_5547_cast_fp16")]; tensor key_cache_19_axes_0 = const()[name = string("key_cache_19_axes_0"), val = tensor([0])]; tensor key_cache_19_cast_fp16 = squeeze(axes = key_cache_19_axes_0, x = var_5547_cast_fp16)[name = string("key_cache_19_cast_fp16")]; tensor var_5554_begin_0 = const()[name = string("op_5554_begin_0"), val = tensor([37, 0, 0, 0])]; tensor var_5554_end_0 = const()[name = string("op_5554_end_0"), val = tensor([38, 8, 1536, 128])]; tensor var_5554_end_mask_0 = const()[name = string("op_5554_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5554_cast_fp16 = slice_by_index(begin = var_5554_begin_0, end = var_5554_end_0, end_mask = var_5554_end_mask_0, x = coreml_update_state_75)[name = string("op_5554_cast_fp16")]; tensor value_cache_19_axes_0 = const()[name = string("value_cache_19_axes_0"), val = tensor([0])]; tensor value_cache_19_cast_fp16 = squeeze(axes = value_cache_19_axes_0, x = var_5554_cast_fp16)[name = string("value_cache_19_cast_fp16")]; tensor var_5578_axes_0 = const()[name = string("op_5578_axes_0"), val = tensor([1])]; tensor var_5578_cast_fp16 = expand_dims(axes = var_5578_axes_0, x = key_cache_19_cast_fp16)[name = string("op_5578_cast_fp16")]; tensor var_5583 = const()[name = string("op_5583"), val = tensor([1, 2, 1, 1])]; tensor value_75_cast_fp16 = tile(reps = var_5583, x = var_5578_cast_fp16)[name = string("value_75_cast_fp16")]; tensor var_5589 = const()[name = string("op_5589"), val = tensor([1, 16, 1536, 128])]; tensor key_states_39_cast_fp16 = reshape(shape = var_5589, x = value_75_cast_fp16)[name = string("key_states_39_cast_fp16")]; tensor var_5592_axes_0 = const()[name = string("op_5592_axes_0"), val = tensor([1])]; tensor var_5592_cast_fp16 = expand_dims(axes = var_5592_axes_0, x = value_cache_19_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5597 = const()[name = string("op_5597"), val = tensor([1, 2, 1, 1])]; tensor value_79_cast_fp16 = tile(reps = var_5597, x = var_5592_cast_fp16)[name = string("value_79_cast_fp16")]; tensor var_5603 = const()[name = string("op_5603"), val = tensor([1, 16, 1536, 128])]; tensor value_states_57_cast_fp16 = reshape(shape = var_5603, x = value_79_cast_fp16)[name = string("value_states_57_cast_fp16")]; bool var_5618_transpose_x_1 = const()[name = string("op_5618_transpose_x_1"), val = bool(false)]; bool var_5618_transpose_y_1 = const()[name = string("op_5618_transpose_y_1"), val = bool(true)]; tensor var_5618 = matmul(transpose_x = var_5618_transpose_x_1, transpose_y = var_5618_transpose_y_1, x = query_states_37, y = key_states_39_cast_fp16)[name = string("op_5618")]; fp16 var_5619_to_fp16 = const()[name = string("op_5619_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_37_cast_fp16 = mul(x = var_5618, y = var_5619_to_fp16)[name = string("attention_37_cast_fp16")]; tensor attention_39_cast_fp16 = add(x = attention_37_cast_fp16, y = causal_mask)[name = string("attention_39_cast_fp16")]; int32 var_5628 = const()[name = string("op_5628"), val = int32(-1)]; tensor probabilities_19_cast_fp16 = softmax(axis = var_5628, x = attention_39_cast_fp16)[name = string("probabilities_19_cast_fp16")]; bool output_55_transpose_x_0 = const()[name = string("output_55_transpose_x_0"), val = bool(false)]; bool output_55_transpose_y_0 = const()[name = string("output_55_transpose_y_0"), val = bool(false)]; tensor output_55_cast_fp16 = matmul(transpose_x = output_55_transpose_x_0, transpose_y = output_55_transpose_y_0, x = probabilities_19_cast_fp16, y = value_states_57_cast_fp16)[name = string("output_55_cast_fp16")]; tensor var_5639_perm_0 = const()[name = string("op_5639_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_5645 = const()[name = string("op_5645"), val = tensor([1, 1, 2048])]; tensor var_5639_cast_fp16 = transpose(perm = var_5639_perm_0, x = output_55_cast_fp16)[name = string("transpose_112")]; tensor output_57_cast_fp16 = reshape(shape = var_5645, x = var_5639_cast_fp16)[name = string("output_57_cast_fp16")]; tensor var_5650 = const()[name = string("op_5650"), val = tensor([0, 2, 1])]; string var_5666_pad_type_0 = const()[name = string("op_5666_pad_type_0"), val = string("valid")]; int32 var_5666_groups_0 = const()[name = string("op_5666_groups_0"), val = int32(1)]; tensor var_5666_strides_0 = const()[name = string("op_5666_strides_0"), val = tensor([1])]; tensor var_5666_pad_0 = const()[name = string("op_5666_pad_0"), val = tensor([0, 0])]; tensor var_5666_dilations_0 = const()[name = string("op_5666_dilations_0"), val = tensor([1])]; tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307252992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308825920))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5651_cast_fp16 = transpose(perm = var_5650, x = output_57_cast_fp16)[name = string("transpose_111")]; tensor var_5666_cast_fp16 = conv(dilations = var_5666_dilations_0, groups = var_5666_groups_0, pad = var_5666_pad_0, pad_type = var_5666_pad_type_0, strides = var_5666_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_5651_cast_fp16)[name = string("op_5666_cast_fp16")]; tensor var_5670 = const()[name = string("op_5670"), val = tensor([0, 2, 1])]; tensor attn_output_19_cast_fp16 = transpose(perm = var_5670, x = var_5666_cast_fp16)[name = string("transpose_110")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_91_cast_fp16, y = attn_output_19_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; int32 var_5685 = const()[name = string("op_5685"), val = int32(-1)]; fp16 const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5687_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_137_promoted_to_fp16)[name = string("op_5687_cast_fp16")]; bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; tensor input_173_cast_fp16 = concat(axis = var_5685, interleave = input_173_interleave_0, values = (hidden_states_99_cast_fp16, var_5687_cast_fp16))[name = string("input_173_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_5682_to_fp16 = const()[name = string("op_5682_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_5682_to_fp16, x = input_173_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor normed_159_begin_0 = const()[name = string("normed_159_begin_0"), val = tensor([0, 0, 0])]; tensor normed_159_end_0 = const()[name = string("normed_159_end_0"), val = tensor([1, 1, 1024])]; tensor normed_159_end_mask_0 = const()[name = string("normed_159_end_mask_0"), val = tensor([true, true, false])]; tensor normed_159_cast_fp16 = slice_by_index(begin = normed_159_begin_0, end = normed_159_end_0, end_mask = normed_159_end_mask_0, x = normed_157_cast_fp16)[name = string("normed_159_cast_fp16")]; tensor const_139_promoted_to_fp16 = const()[name = string("const_139_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308842368)))]; tensor x_37_cast_fp16 = mul(x = normed_159_cast_fp16, y = const_139_promoted_to_fp16)[name = string("x_37_cast_fp16")]; tensor var_5707 = const()[name = string("op_5707"), val = tensor([0, 2, 1])]; tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; tensor var_5708 = transpose(perm = var_5707, x = x_37_cast_fp16)[name = string("transpose_109")]; tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_5708)[name = string("input_175")]; string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; tensor input_177 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = model_model_layers_9_mlp_gate_proj_weight_palettized, x = input_175)[name = string("input_177")]; string b_19_pad_type_0 = const()[name = string("b_19_pad_type_0"), val = string("valid")]; tensor b_19_strides_0 = const()[name = string("b_19_strides_0"), val = tensor([1, 1])]; tensor b_19_pad_0 = const()[name = string("b_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_19_dilations_0 = const()[name = string("b_19_dilations_0"), val = tensor([1, 1])]; int32 b_19_groups_0 = const()[name = string("b_19_groups_0"), val = int32(1)]; tensor b_19 = conv(dilations = b_19_dilations_0, groups = b_19_groups_0, pad = b_19_pad_0, pad_type = b_19_pad_type_0, strides = b_19_strides_0, weight = model_model_layers_9_mlp_up_proj_weight_palettized, x = input_175)[name = string("b_19")]; tensor c_19 = silu(x = input_177)[name = string("c_19")]; tensor input_179 = mul(x = c_19, y = b_19)[name = string("input_179")]; string e_19_pad_type_0 = const()[name = string("e_19_pad_type_0"), val = string("valid")]; tensor e_19_strides_0 = const()[name = string("e_19_strides_0"), val = tensor([1, 1])]; tensor e_19_pad_0 = const()[name = string("e_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_19_dilations_0 = const()[name = string("e_19_dilations_0"), val = tensor([1, 1])]; int32 e_19_groups_0 = const()[name = string("e_19_groups_0"), val = int32(1)]; tensor e_19 = conv(dilations = e_19_dilations_0, groups = e_19_groups_0, pad = e_19_pad_0, pad_type = e_19_pad_type_0, strides = e_19_strides_0, weight = model_model_layers_9_mlp_down_proj_weight_palettized, x = input_179)[name = string("e_19")]; tensor var_5730_axes_0 = const()[name = string("op_5730_axes_0"), val = tensor([2])]; tensor var_5730 = squeeze(axes = var_5730_axes_0, x = e_19)[name = string("op_5730")]; tensor var_5731 = const()[name = string("op_5731"), val = tensor([0, 2, 1])]; tensor var_5732 = transpose(perm = var_5731, x = var_5730)[name = string("transpose_108")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = var_5732)[name = string("hidden_states_101_cast_fp16")]; int32 var_5746 = const()[name = string("op_5746"), val = int32(-1)]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5748_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5748_cast_fp16")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181_cast_fp16 = concat(axis = var_5746, interleave = input_181_interleave_0, values = (hidden_states_101_cast_fp16, var_5748_cast_fp16))[name = string("input_181_cast_fp16")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_5743_to_fp16 = const()[name = string("op_5743_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_5743_to_fp16, x = input_181_cast_fp16)[name = string("normed_161_cast_fp16")]; tensor normed_163_begin_0 = const()[name = string("normed_163_begin_0"), val = tensor([0, 0, 0])]; tensor normed_163_end_0 = const()[name = string("normed_163_end_0"), val = tensor([1, 1, 1024])]; tensor normed_163_end_mask_0 = const()[name = string("normed_163_end_mask_0"), val = tensor([true, true, false])]; tensor normed_163_cast_fp16 = slice_by_index(begin = normed_163_begin_0, end = normed_163_end_0, end_mask = normed_163_end_mask_0, x = normed_161_cast_fp16)[name = string("normed_163_cast_fp16")]; tensor const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308844480)))]; tensor hidden_states_103_cast_fp16 = mul(x = normed_163_cast_fp16, y = const_142_promoted_to_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor var_5760 = const()[name = string("op_5760"), val = tensor([0, 2, 1])]; tensor var_5763_axes_0 = const()[name = string("op_5763_axes_0"), val = tensor([2])]; tensor var_5761_cast_fp16 = transpose(perm = var_5760, x = hidden_states_103_cast_fp16)[name = string("transpose_107")]; tensor var_5763_cast_fp16 = expand_dims(axes = var_5763_axes_0, x = var_5761_cast_fp16)[name = string("op_5763_cast_fp16")]; string var_5779_pad_type_0 = const()[name = string("op_5779_pad_type_0"), val = string("valid")]; tensor var_5779_strides_0 = const()[name = string("op_5779_strides_0"), val = tensor([1, 1])]; tensor var_5779_pad_0 = const()[name = string("op_5779_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5779_dilations_0 = const()[name = string("op_5779_dilations_0"), val = tensor([1, 1])]; int32 var_5779_groups_0 = const()[name = string("op_5779_groups_0"), val = int32(1)]; tensor var_5779 = conv(dilations = var_5779_dilations_0, groups = var_5779_groups_0, pad = var_5779_pad_0, pad_type = var_5779_pad_type_0, strides = var_5779_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_5763_cast_fp16)[name = string("op_5779")]; tensor var_5784 = const()[name = string("op_5784"), val = tensor([1, 16, 1, 128])]; tensor var_5785 = reshape(shape = var_5784, x = var_5779)[name = string("op_5785")]; string var_5801_pad_type_0 = const()[name = string("op_5801_pad_type_0"), val = string("valid")]; tensor var_5801_strides_0 = const()[name = string("op_5801_strides_0"), val = tensor([1, 1])]; tensor var_5801_pad_0 = const()[name = string("op_5801_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5801_dilations_0 = const()[name = string("op_5801_dilations_0"), val = tensor([1, 1])]; int32 var_5801_groups_0 = const()[name = string("op_5801_groups_0"), val = int32(1)]; tensor var_5801 = conv(dilations = var_5801_dilations_0, groups = var_5801_groups_0, pad = var_5801_pad_0, pad_type = var_5801_pad_type_0, strides = var_5801_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_5763_cast_fp16)[name = string("op_5801")]; tensor var_5806 = const()[name = string("op_5806"), val = tensor([1, 8, 1, 128])]; tensor var_5807 = reshape(shape = var_5806, x = var_5801)[name = string("op_5807")]; string var_5823_pad_type_0 = const()[name = string("op_5823_pad_type_0"), val = string("valid")]; tensor var_5823_strides_0 = const()[name = string("op_5823_strides_0"), val = tensor([1, 1])]; tensor var_5823_pad_0 = const()[name = string("op_5823_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5823_dilations_0 = const()[name = string("op_5823_dilations_0"), val = tensor([1, 1])]; int32 var_5823_groups_0 = const()[name = string("op_5823_groups_0"), val = int32(1)]; tensor var_5823 = conv(dilations = var_5823_dilations_0, groups = var_5823_groups_0, pad = var_5823_pad_0, pad_type = var_5823_pad_type_0, strides = var_5823_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_5763_cast_fp16)[name = string("op_5823")]; tensor var_5828 = const()[name = string("op_5828"), val = tensor([1, 8, 1, 128])]; tensor var_5829 = reshape(shape = var_5828, x = var_5823)[name = string("op_5829")]; int32 var_5846 = const()[name = string("op_5846"), val = int32(-1)]; fp16 const_143_promoted = const()[name = string("const_143_promoted"), val = fp16(-0x1p+0)]; tensor var_5848 = mul(x = var_5785, y = const_143_promoted)[name = string("op_5848")]; bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; tensor input_185 = concat(axis = var_5846, interleave = input_185_interleave_0, values = (var_5785, var_5848))[name = string("input_185")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_5843_to_fp16 = const()[name = string("op_5843_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_5843_to_fp16, x = input_185)[name = string("normed_165_cast_fp16")]; tensor normed_167_begin_0 = const()[name = string("normed_167_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_167_end_0 = const()[name = string("normed_167_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_167_end_mask_0 = const()[name = string("normed_167_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_167 = slice_by_index(begin = normed_167_begin_0, end = normed_167_end_0, end_mask = normed_167_end_mask_0, x = normed_165_cast_fp16)[name = string("normed_167")]; tensor const_145 = const()[name = string("const_145"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308846592)))]; tensor q_21 = mul(x = normed_167, y = const_145)[name = string("q_21")]; int32 var_5868 = const()[name = string("op_5868"), val = int32(-1)]; fp16 const_146_promoted = const()[name = string("const_146_promoted"), val = fp16(-0x1p+0)]; tensor var_5870 = mul(x = var_5807, y = const_146_promoted)[name = string("op_5870")]; bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; tensor input_187 = concat(axis = var_5868, interleave = input_187_interleave_0, values = (var_5807, var_5870))[name = string("input_187")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_5865_to_fp16 = const()[name = string("op_5865_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_5865_to_fp16, x = input_187)[name = string("normed_169_cast_fp16")]; tensor normed_171_begin_0 = const()[name = string("normed_171_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_171_end_0 = const()[name = string("normed_171_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_171_end_mask_0 = const()[name = string("normed_171_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_171 = slice_by_index(begin = normed_171_begin_0, end = normed_171_end_0, end_mask = normed_171_end_mask_0, x = normed_169_cast_fp16)[name = string("normed_171")]; tensor const_148 = const()[name = string("const_148"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308846912)))]; tensor k_21 = mul(x = normed_171, y = const_148)[name = string("k_21")]; tensor var_5879 = mul(x = q_21, y = cos_1_cast_fp16)[name = string("op_5879")]; tensor var_5884_begin_0 = const()[name = string("op_5884_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5884_end_0 = const()[name = string("op_5884_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_5884_end_mask_0 = const()[name = string("op_5884_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5884 = slice_by_index(begin = var_5884_begin_0, end = var_5884_end_0, end_mask = var_5884_end_mask_0, x = q_21)[name = string("op_5884")]; fp16 const_149_promoted = const()[name = string("const_149_promoted"), val = fp16(-0x1p+0)]; tensor var_5885 = mul(x = var_5884, y = const_149_promoted)[name = string("op_5885")]; tensor var_5890_begin_0 = const()[name = string("op_5890_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5890_end_0 = const()[name = string("op_5890_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_5890_end_mask_0 = const()[name = string("op_5890_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5890 = slice_by_index(begin = var_5890_begin_0, end = var_5890_end_0, end_mask = var_5890_end_mask_0, x = q_21)[name = string("op_5890")]; int32 var_5892 = const()[name = string("op_5892"), val = int32(-1)]; bool var_5893_interleave_0 = const()[name = string("op_5893_interleave_0"), val = bool(false)]; tensor var_5893 = concat(axis = var_5892, interleave = var_5893_interleave_0, values = (var_5885, var_5890))[name = string("op_5893")]; tensor var_5894 = mul(x = var_5893, y = sin_1_cast_fp16)[name = string("op_5894")]; tensor query_states_41 = add(x = var_5879, y = var_5894)[name = string("query_states_41")]; tensor var_5897 = mul(x = k_21, y = cos_1_cast_fp16)[name = string("op_5897")]; tensor var_5902_begin_0 = const()[name = string("op_5902_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5902_end_0 = const()[name = string("op_5902_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_5902_end_mask_0 = const()[name = string("op_5902_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5902 = slice_by_index(begin = var_5902_begin_0, end = var_5902_end_0, end_mask = var_5902_end_mask_0, x = k_21)[name = string("op_5902")]; fp16 const_150_promoted = const()[name = string("const_150_promoted"), val = fp16(-0x1p+0)]; tensor var_5903 = mul(x = var_5902, y = const_150_promoted)[name = string("op_5903")]; tensor var_5908_begin_0 = const()[name = string("op_5908_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5908_end_0 = const()[name = string("op_5908_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_5908_end_mask_0 = const()[name = string("op_5908_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5908 = slice_by_index(begin = var_5908_begin_0, end = var_5908_end_0, end_mask = var_5908_end_mask_0, x = k_21)[name = string("op_5908")]; int32 var_5910 = const()[name = string("op_5910"), val = int32(-1)]; bool var_5911_interleave_0 = const()[name = string("op_5911_interleave_0"), val = bool(false)]; tensor var_5911 = concat(axis = var_5910, interleave = var_5911_interleave_0, values = (var_5903, var_5908))[name = string("op_5911")]; tensor var_5912 = mul(x = var_5911, y = sin_1_cast_fp16)[name = string("op_5912")]; tensor key_states_41 = add(x = var_5897, y = var_5912)[name = string("key_states_41")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; int32 concat_82_axis_0 = const()[name = string("concat_82_axis_0"), val = int32(0)]; bool concat_82_interleave_0 = const()[name = string("concat_82_interleave_0"), val = bool(false)]; tensor concat_82 = concat(axis = concat_82_axis_0, interleave = concat_82_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_82")]; tensor concat_83_values1_0 = const()[name = string("concat_83_values1_0"), val = tensor([0])]; tensor concat_83_values3_0 = const()[name = string("concat_83_values3_0"), val = tensor([0])]; int32 concat_83_axis_0 = const()[name = string("concat_83_axis_0"), val = int32(0)]; bool concat_83_interleave_0 = const()[name = string("concat_83_interleave_0"), val = bool(false)]; tensor concat_83 = concat(axis = concat_83_axis_0, interleave = concat_83_interleave_0, values = (expand_dims_124, concat_83_values1_0, var_1717, concat_83_values3_0))[name = string("concat_83")]; tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_82, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_83, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = key_states_41, x = coreml_update_state_75)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_132_write_state")]; tensor coreml_update_state_76 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_132")]; tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([38])]; tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([39])]; int32 concat_86_axis_0 = const()[name = string("concat_86_axis_0"), val = int32(0)]; bool concat_86_interleave_0 = const()[name = string("concat_86_interleave_0"), val = bool(false)]; tensor concat_86 = concat(axis = concat_86_axis_0, interleave = concat_86_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_86")]; tensor concat_87_values1_0 = const()[name = string("concat_87_values1_0"), val = tensor([0])]; tensor concat_87_values3_0 = const()[name = string("concat_87_values3_0"), val = tensor([0])]; int32 concat_87_axis_0 = const()[name = string("concat_87_axis_0"), val = int32(0)]; bool concat_87_interleave_0 = const()[name = string("concat_87_interleave_0"), val = bool(false)]; tensor concat_87 = concat(axis = concat_87_axis_0, interleave = concat_87_interleave_0, values = (expand_dims_130, concat_87_values1_0, var_1717, concat_87_values3_0))[name = string("concat_87")]; tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_86, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_87, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = var_5829, x = coreml_update_state_76)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_133_write_state")]; tensor coreml_update_state_77 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_133")]; tensor var_5967_begin_0 = const()[name = string("op_5967_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_5967_end_0 = const()[name = string("op_5967_end_0"), val = tensor([11, 8, 1536, 128])]; tensor var_5967_end_mask_0 = const()[name = string("op_5967_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5967_cast_fp16 = slice_by_index(begin = var_5967_begin_0, end = var_5967_end_0, end_mask = var_5967_end_mask_0, x = coreml_update_state_77)[name = string("op_5967_cast_fp16")]; tensor key_cache_21_axes_0 = const()[name = string("key_cache_21_axes_0"), val = tensor([0])]; tensor key_cache_21_cast_fp16 = squeeze(axes = key_cache_21_axes_0, x = var_5967_cast_fp16)[name = string("key_cache_21_cast_fp16")]; tensor var_5974_begin_0 = const()[name = string("op_5974_begin_0"), val = tensor([38, 0, 0, 0])]; tensor var_5974_end_0 = const()[name = string("op_5974_end_0"), val = tensor([39, 8, 1536, 128])]; tensor var_5974_end_mask_0 = const()[name = string("op_5974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5974_cast_fp16 = slice_by_index(begin = var_5974_begin_0, end = var_5974_end_0, end_mask = var_5974_end_mask_0, x = coreml_update_state_77)[name = string("op_5974_cast_fp16")]; tensor value_cache_21_axes_0 = const()[name = string("value_cache_21_axes_0"), val = tensor([0])]; tensor value_cache_21_cast_fp16 = squeeze(axes = value_cache_21_axes_0, x = var_5974_cast_fp16)[name = string("value_cache_21_cast_fp16")]; tensor var_5998_axes_0 = const()[name = string("op_5998_axes_0"), val = tensor([1])]; tensor var_5998_cast_fp16 = expand_dims(axes = var_5998_axes_0, x = key_cache_21_cast_fp16)[name = string("op_5998_cast_fp16")]; tensor var_6003 = const()[name = string("op_6003"), val = tensor([1, 2, 1, 1])]; tensor value_83_cast_fp16 = tile(reps = var_6003, x = var_5998_cast_fp16)[name = string("value_83_cast_fp16")]; tensor var_6009 = const()[name = string("op_6009"), val = tensor([1, 16, 1536, 128])]; tensor key_states_43_cast_fp16 = reshape(shape = var_6009, x = value_83_cast_fp16)[name = string("key_states_43_cast_fp16")]; tensor var_6012_axes_0 = const()[name = string("op_6012_axes_0"), val = tensor([1])]; tensor var_6012_cast_fp16 = expand_dims(axes = var_6012_axes_0, x = value_cache_21_cast_fp16)[name = string("op_6012_cast_fp16")]; tensor var_6017 = const()[name = string("op_6017"), val = tensor([1, 2, 1, 1])]; tensor value_87_cast_fp16 = tile(reps = var_6017, x = var_6012_cast_fp16)[name = string("value_87_cast_fp16")]; tensor var_6023 = const()[name = string("op_6023"), val = tensor([1, 16, 1536, 128])]; tensor value_states_63_cast_fp16 = reshape(shape = var_6023, x = value_87_cast_fp16)[name = string("value_states_63_cast_fp16")]; bool var_6038_transpose_x_1 = const()[name = string("op_6038_transpose_x_1"), val = bool(false)]; bool var_6038_transpose_y_1 = const()[name = string("op_6038_transpose_y_1"), val = bool(true)]; tensor var_6038 = matmul(transpose_x = var_6038_transpose_x_1, transpose_y = var_6038_transpose_y_1, x = query_states_41, y = key_states_43_cast_fp16)[name = string("op_6038")]; fp16 var_6039_to_fp16 = const()[name = string("op_6039_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_41_cast_fp16 = mul(x = var_6038, y = var_6039_to_fp16)[name = string("attention_41_cast_fp16")]; tensor attention_43_cast_fp16 = add(x = attention_41_cast_fp16, y = causal_mask)[name = string("attention_43_cast_fp16")]; int32 var_6048 = const()[name = string("op_6048"), val = int32(-1)]; tensor probabilities_21_cast_fp16 = softmax(axis = var_6048, x = attention_43_cast_fp16)[name = string("probabilities_21_cast_fp16")]; bool output_61_transpose_x_0 = const()[name = string("output_61_transpose_x_0"), val = bool(false)]; bool output_61_transpose_y_0 = const()[name = string("output_61_transpose_y_0"), val = bool(false)]; tensor output_61_cast_fp16 = matmul(transpose_x = output_61_transpose_x_0, transpose_y = output_61_transpose_y_0, x = probabilities_21_cast_fp16, y = value_states_63_cast_fp16)[name = string("output_61_cast_fp16")]; tensor var_6059_perm_0 = const()[name = string("op_6059_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_6065 = const()[name = string("op_6065"), val = tensor([1, 1, 2048])]; tensor var_6059_cast_fp16 = transpose(perm = var_6059_perm_0, x = output_61_cast_fp16)[name = string("transpose_106")]; tensor output_63_cast_fp16 = reshape(shape = var_6065, x = var_6059_cast_fp16)[name = string("output_63_cast_fp16")]; tensor var_6070 = const()[name = string("op_6070"), val = tensor([0, 2, 1])]; string var_6086_pad_type_0 = const()[name = string("op_6086_pad_type_0"), val = string("valid")]; int32 var_6086_groups_0 = const()[name = string("op_6086_groups_0"), val = int32(1)]; tensor var_6086_strides_0 = const()[name = string("op_6086_strides_0"), val = tensor([1])]; tensor var_6086_pad_0 = const()[name = string("op_6086_pad_0"), val = tensor([0, 0])]; tensor var_6086_dilations_0 = const()[name = string("op_6086_dilations_0"), val = tensor([1])]; tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308847232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310420160))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6071_cast_fp16 = transpose(perm = var_6070, x = output_63_cast_fp16)[name = string("transpose_105")]; tensor var_6086_cast_fp16 = conv(dilations = var_6086_dilations_0, groups = var_6086_groups_0, pad = var_6086_pad_0, pad_type = var_6086_pad_type_0, strides = var_6086_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6071_cast_fp16)[name = string("op_6086_cast_fp16")]; tensor var_6090 = const()[name = string("op_6090"), val = tensor([0, 2, 1])]; tensor attn_output_21_cast_fp16 = transpose(perm = var_6090, x = var_6086_cast_fp16)[name = string("transpose_104")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = attn_output_21_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; int32 var_6105 = const()[name = string("op_6105"), val = int32(-1)]; fp16 const_151_promoted_to_fp16 = const()[name = string("const_151_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6107_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_151_promoted_to_fp16)[name = string("op_6107_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_6105, interleave = input_191_interleave_0, values = (hidden_states_109_cast_fp16, var_6107_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_6102_to_fp16 = const()[name = string("op_6102_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_6102_to_fp16, x = input_191_cast_fp16)[name = string("normed_173_cast_fp16")]; tensor normed_175_begin_0 = const()[name = string("normed_175_begin_0"), val = tensor([0, 0, 0])]; tensor normed_175_end_0 = const()[name = string("normed_175_end_0"), val = tensor([1, 1, 1024])]; tensor normed_175_end_mask_0 = const()[name = string("normed_175_end_mask_0"), val = tensor([true, true, false])]; tensor normed_175_cast_fp16 = slice_by_index(begin = normed_175_begin_0, end = normed_175_end_0, end_mask = normed_175_end_mask_0, x = normed_173_cast_fp16)[name = string("normed_175_cast_fp16")]; tensor const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310436608)))]; tensor x_41_cast_fp16 = mul(x = normed_175_cast_fp16, y = const_153_promoted_to_fp16)[name = string("x_41_cast_fp16")]; tensor var_6127 = const()[name = string("op_6127"), val = tensor([0, 2, 1])]; tensor input_193_axes_0 = const()[name = string("input_193_axes_0"), val = tensor([2])]; tensor var_6128 = transpose(perm = var_6127, x = x_41_cast_fp16)[name = string("transpose_103")]; tensor input_193 = expand_dims(axes = input_193_axes_0, x = var_6128)[name = string("input_193")]; string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; tensor input_195 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_193)[name = string("input_195")]; string b_21_pad_type_0 = const()[name = string("b_21_pad_type_0"), val = string("valid")]; tensor b_21_strides_0 = const()[name = string("b_21_strides_0"), val = tensor([1, 1])]; tensor b_21_pad_0 = const()[name = string("b_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_21_dilations_0 = const()[name = string("b_21_dilations_0"), val = tensor([1, 1])]; int32 b_21_groups_0 = const()[name = string("b_21_groups_0"), val = int32(1)]; tensor b_21 = conv(dilations = b_21_dilations_0, groups = b_21_groups_0, pad = b_21_pad_0, pad_type = b_21_pad_type_0, strides = b_21_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_193)[name = string("b_21")]; tensor c_21 = silu(x = input_195)[name = string("c_21")]; tensor input_197 = mul(x = c_21, y = b_21)[name = string("input_197")]; string e_21_pad_type_0 = const()[name = string("e_21_pad_type_0"), val = string("valid")]; tensor e_21_strides_0 = const()[name = string("e_21_strides_0"), val = tensor([1, 1])]; tensor e_21_pad_0 = const()[name = string("e_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_21_dilations_0 = const()[name = string("e_21_dilations_0"), val = tensor([1, 1])]; int32 e_21_groups_0 = const()[name = string("e_21_groups_0"), val = int32(1)]; tensor e_21 = conv(dilations = e_21_dilations_0, groups = e_21_groups_0, pad = e_21_pad_0, pad_type = e_21_pad_type_0, strides = e_21_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_197)[name = string("e_21")]; tensor var_6150_axes_0 = const()[name = string("op_6150_axes_0"), val = tensor([2])]; tensor var_6150 = squeeze(axes = var_6150_axes_0, x = e_21)[name = string("op_6150")]; tensor var_6151 = const()[name = string("op_6151"), val = tensor([0, 2, 1])]; tensor var_6152 = transpose(perm = var_6151, x = var_6150)[name = string("transpose_102")]; tensor hidden_states_111_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_6152)[name = string("hidden_states_111_cast_fp16")]; int32 var_6166 = const()[name = string("op_6166"), val = int32(-1)]; fp16 const_154_promoted_to_fp16 = const()[name = string("const_154_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6168_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_154_promoted_to_fp16)[name = string("op_6168_cast_fp16")]; bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; tensor input_199_cast_fp16 = concat(axis = var_6166, interleave = input_199_interleave_0, values = (hidden_states_111_cast_fp16, var_6168_cast_fp16))[name = string("input_199_cast_fp16")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_6163_to_fp16 = const()[name = string("op_6163_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_6163_to_fp16, x = input_199_cast_fp16)[name = string("normed_177_cast_fp16")]; tensor normed_179_begin_0 = const()[name = string("normed_179_begin_0"), val = tensor([0, 0, 0])]; tensor normed_179_end_0 = const()[name = string("normed_179_end_0"), val = tensor([1, 1, 1024])]; tensor normed_179_end_mask_0 = const()[name = string("normed_179_end_mask_0"), val = tensor([true, true, false])]; tensor normed_179_cast_fp16 = slice_by_index(begin = normed_179_begin_0, end = normed_179_end_0, end_mask = normed_179_end_mask_0, x = normed_177_cast_fp16)[name = string("normed_179_cast_fp16")]; tensor const_156_promoted_to_fp16 = const()[name = string("const_156_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310438720)))]; tensor hidden_states_113_cast_fp16 = mul(x = normed_179_cast_fp16, y = const_156_promoted_to_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor var_6180 = const()[name = string("op_6180"), val = tensor([0, 2, 1])]; tensor var_6183_axes_0 = const()[name = string("op_6183_axes_0"), val = tensor([2])]; tensor var_6181_cast_fp16 = transpose(perm = var_6180, x = hidden_states_113_cast_fp16)[name = string("transpose_101")]; tensor var_6183_cast_fp16 = expand_dims(axes = var_6183_axes_0, x = var_6181_cast_fp16)[name = string("op_6183_cast_fp16")]; string var_6199_pad_type_0 = const()[name = string("op_6199_pad_type_0"), val = string("valid")]; tensor var_6199_strides_0 = const()[name = string("op_6199_strides_0"), val = tensor([1, 1])]; tensor var_6199_pad_0 = const()[name = string("op_6199_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6199_dilations_0 = const()[name = string("op_6199_dilations_0"), val = tensor([1, 1])]; int32 var_6199_groups_0 = const()[name = string("op_6199_groups_0"), val = int32(1)]; tensor var_6199 = conv(dilations = var_6199_dilations_0, groups = var_6199_groups_0, pad = var_6199_pad_0, pad_type = var_6199_pad_type_0, strides = var_6199_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_6183_cast_fp16)[name = string("op_6199")]; tensor var_6204 = const()[name = string("op_6204"), val = tensor([1, 16, 1, 128])]; tensor var_6205 = reshape(shape = var_6204, x = var_6199)[name = string("op_6205")]; string var_6221_pad_type_0 = const()[name = string("op_6221_pad_type_0"), val = string("valid")]; tensor var_6221_strides_0 = const()[name = string("op_6221_strides_0"), val = tensor([1, 1])]; tensor var_6221_pad_0 = const()[name = string("op_6221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6221_dilations_0 = const()[name = string("op_6221_dilations_0"), val = tensor([1, 1])]; int32 var_6221_groups_0 = const()[name = string("op_6221_groups_0"), val = int32(1)]; tensor var_6221 = conv(dilations = var_6221_dilations_0, groups = var_6221_groups_0, pad = var_6221_pad_0, pad_type = var_6221_pad_type_0, strides = var_6221_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_6183_cast_fp16)[name = string("op_6221")]; tensor var_6226 = const()[name = string("op_6226"), val = tensor([1, 8, 1, 128])]; tensor var_6227 = reshape(shape = var_6226, x = var_6221)[name = string("op_6227")]; string var_6243_pad_type_0 = const()[name = string("op_6243_pad_type_0"), val = string("valid")]; tensor var_6243_strides_0 = const()[name = string("op_6243_strides_0"), val = tensor([1, 1])]; tensor var_6243_pad_0 = const()[name = string("op_6243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6243_dilations_0 = const()[name = string("op_6243_dilations_0"), val = tensor([1, 1])]; int32 var_6243_groups_0 = const()[name = string("op_6243_groups_0"), val = int32(1)]; tensor var_6243 = conv(dilations = var_6243_dilations_0, groups = var_6243_groups_0, pad = var_6243_pad_0, pad_type = var_6243_pad_type_0, strides = var_6243_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_6183_cast_fp16)[name = string("op_6243")]; tensor var_6248 = const()[name = string("op_6248"), val = tensor([1, 8, 1, 128])]; tensor var_6249 = reshape(shape = var_6248, x = var_6243)[name = string("op_6249")]; int32 var_6266 = const()[name = string("op_6266"), val = int32(-1)]; fp16 const_157_promoted = const()[name = string("const_157_promoted"), val = fp16(-0x1p+0)]; tensor var_6268 = mul(x = var_6205, y = const_157_promoted)[name = string("op_6268")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203 = concat(axis = var_6266, interleave = input_203_interleave_0, values = (var_6205, var_6268))[name = string("input_203")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_6263_to_fp16 = const()[name = string("op_6263_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_6263_to_fp16, x = input_203)[name = string("normed_181_cast_fp16")]; tensor normed_183_begin_0 = const()[name = string("normed_183_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_183_end_0 = const()[name = string("normed_183_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_183_end_mask_0 = const()[name = string("normed_183_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_183 = slice_by_index(begin = normed_183_begin_0, end = normed_183_end_0, end_mask = normed_183_end_mask_0, x = normed_181_cast_fp16)[name = string("normed_183")]; tensor const_159 = const()[name = string("const_159"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310440832)))]; tensor q_23 = mul(x = normed_183, y = const_159)[name = string("q_23")]; int32 var_6288 = const()[name = string("op_6288"), val = int32(-1)]; fp16 const_160_promoted = const()[name = string("const_160_promoted"), val = fp16(-0x1p+0)]; tensor var_6290 = mul(x = var_6227, y = const_160_promoted)[name = string("op_6290")]; bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; tensor input_205 = concat(axis = var_6288, interleave = input_205_interleave_0, values = (var_6227, var_6290))[name = string("input_205")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_6285_to_fp16 = const()[name = string("op_6285_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_6285_to_fp16, x = input_205)[name = string("normed_185_cast_fp16")]; tensor normed_187_begin_0 = const()[name = string("normed_187_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_187_end_0 = const()[name = string("normed_187_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_187_end_mask_0 = const()[name = string("normed_187_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_187 = slice_by_index(begin = normed_187_begin_0, end = normed_187_end_0, end_mask = normed_187_end_mask_0, x = normed_185_cast_fp16)[name = string("normed_187")]; tensor const_162 = const()[name = string("const_162"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310441152)))]; tensor k_23 = mul(x = normed_187, y = const_162)[name = string("k_23")]; tensor var_6299 = mul(x = q_23, y = cos_1_cast_fp16)[name = string("op_6299")]; tensor var_6304_begin_0 = const()[name = string("op_6304_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6304_end_0 = const()[name = string("op_6304_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_6304_end_mask_0 = const()[name = string("op_6304_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6304 = slice_by_index(begin = var_6304_begin_0, end = var_6304_end_0, end_mask = var_6304_end_mask_0, x = q_23)[name = string("op_6304")]; fp16 const_163_promoted = const()[name = string("const_163_promoted"), val = fp16(-0x1p+0)]; tensor var_6305 = mul(x = var_6304, y = const_163_promoted)[name = string("op_6305")]; tensor var_6310_begin_0 = const()[name = string("op_6310_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6310_end_0 = const()[name = string("op_6310_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_6310_end_mask_0 = const()[name = string("op_6310_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6310 = slice_by_index(begin = var_6310_begin_0, end = var_6310_end_0, end_mask = var_6310_end_mask_0, x = q_23)[name = string("op_6310")]; int32 var_6312 = const()[name = string("op_6312"), val = int32(-1)]; bool var_6313_interleave_0 = const()[name = string("op_6313_interleave_0"), val = bool(false)]; tensor var_6313 = concat(axis = var_6312, interleave = var_6313_interleave_0, values = (var_6305, var_6310))[name = string("op_6313")]; tensor var_6314 = mul(x = var_6313, y = sin_1_cast_fp16)[name = string("op_6314")]; tensor query_states_45 = add(x = var_6299, y = var_6314)[name = string("query_states_45")]; tensor var_6317 = mul(x = k_23, y = cos_1_cast_fp16)[name = string("op_6317")]; tensor var_6322_begin_0 = const()[name = string("op_6322_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6322_end_0 = const()[name = string("op_6322_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_6322_end_mask_0 = const()[name = string("op_6322_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6322 = slice_by_index(begin = var_6322_begin_0, end = var_6322_end_0, end_mask = var_6322_end_mask_0, x = k_23)[name = string("op_6322")]; fp16 const_164_promoted = const()[name = string("const_164_promoted"), val = fp16(-0x1p+0)]; tensor var_6323 = mul(x = var_6322, y = const_164_promoted)[name = string("op_6323")]; tensor var_6328_begin_0 = const()[name = string("op_6328_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6328_end_0 = const()[name = string("op_6328_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_6328_end_mask_0 = const()[name = string("op_6328_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6328 = slice_by_index(begin = var_6328_begin_0, end = var_6328_end_0, end_mask = var_6328_end_mask_0, x = k_23)[name = string("op_6328")]; int32 var_6330 = const()[name = string("op_6330"), val = int32(-1)]; bool var_6331_interleave_0 = const()[name = string("op_6331_interleave_0"), val = bool(false)]; tensor var_6331 = concat(axis = var_6330, interleave = var_6331_interleave_0, values = (var_6323, var_6328))[name = string("op_6331")]; tensor var_6332 = mul(x = var_6331, y = sin_1_cast_fp16)[name = string("op_6332")]; tensor key_states_45 = add(x = var_6317, y = var_6332)[name = string("key_states_45")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_90")]; tensor concat_91_values1_0 = const()[name = string("concat_91_values1_0"), val = tensor([0])]; tensor concat_91_values3_0 = const()[name = string("concat_91_values3_0"), val = tensor([0])]; int32 concat_91_axis_0 = const()[name = string("concat_91_axis_0"), val = int32(0)]; bool concat_91_interleave_0 = const()[name = string("concat_91_interleave_0"), val = bool(false)]; tensor concat_91 = concat(axis = concat_91_axis_0, interleave = concat_91_interleave_0, values = (expand_dims_136, concat_91_values1_0, var_1717, concat_91_values3_0))[name = string("concat_91")]; tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_90, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_91, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = key_states_45, x = coreml_update_state_77)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_134_write_state")]; tensor coreml_update_state_78 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_134")]; tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([39])]; tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([40])]; int32 concat_94_axis_0 = const()[name = string("concat_94_axis_0"), val = int32(0)]; bool concat_94_interleave_0 = const()[name = string("concat_94_interleave_0"), val = bool(false)]; tensor concat_94 = concat(axis = concat_94_axis_0, interleave = concat_94_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_94")]; tensor concat_95_values1_0 = const()[name = string("concat_95_values1_0"), val = tensor([0])]; tensor concat_95_values3_0 = const()[name = string("concat_95_values3_0"), val = tensor([0])]; int32 concat_95_axis_0 = const()[name = string("concat_95_axis_0"), val = int32(0)]; bool concat_95_interleave_0 = const()[name = string("concat_95_interleave_0"), val = bool(false)]; tensor concat_95 = concat(axis = concat_95_axis_0, interleave = concat_95_interleave_0, values = (expand_dims_142, concat_95_values1_0, var_1717, concat_95_values3_0))[name = string("concat_95")]; tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_94, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_95, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = var_6249, x = coreml_update_state_78)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_135_write_state")]; tensor coreml_update_state_79 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_135")]; tensor var_6387_begin_0 = const()[name = string("op_6387_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_6387_end_0 = const()[name = string("op_6387_end_0"), val = tensor([12, 8, 1536, 128])]; tensor var_6387_end_mask_0 = const()[name = string("op_6387_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6387_cast_fp16 = slice_by_index(begin = var_6387_begin_0, end = var_6387_end_0, end_mask = var_6387_end_mask_0, x = coreml_update_state_79)[name = string("op_6387_cast_fp16")]; tensor key_cache_23_axes_0 = const()[name = string("key_cache_23_axes_0"), val = tensor([0])]; tensor key_cache_23_cast_fp16 = squeeze(axes = key_cache_23_axes_0, x = var_6387_cast_fp16)[name = string("key_cache_23_cast_fp16")]; tensor var_6394_begin_0 = const()[name = string("op_6394_begin_0"), val = tensor([39, 0, 0, 0])]; tensor var_6394_end_0 = const()[name = string("op_6394_end_0"), val = tensor([40, 8, 1536, 128])]; tensor var_6394_end_mask_0 = const()[name = string("op_6394_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6394_cast_fp16 = slice_by_index(begin = var_6394_begin_0, end = var_6394_end_0, end_mask = var_6394_end_mask_0, x = coreml_update_state_79)[name = string("op_6394_cast_fp16")]; tensor value_cache_23_axes_0 = const()[name = string("value_cache_23_axes_0"), val = tensor([0])]; tensor value_cache_23_cast_fp16 = squeeze(axes = value_cache_23_axes_0, x = var_6394_cast_fp16)[name = string("value_cache_23_cast_fp16")]; tensor var_6418_axes_0 = const()[name = string("op_6418_axes_0"), val = tensor([1])]; tensor var_6418_cast_fp16 = expand_dims(axes = var_6418_axes_0, x = key_cache_23_cast_fp16)[name = string("op_6418_cast_fp16")]; tensor var_6423 = const()[name = string("op_6423"), val = tensor([1, 2, 1, 1])]; tensor value_91_cast_fp16 = tile(reps = var_6423, x = var_6418_cast_fp16)[name = string("value_91_cast_fp16")]; tensor var_6429 = const()[name = string("op_6429"), val = tensor([1, 16, 1536, 128])]; tensor key_states_47_cast_fp16 = reshape(shape = var_6429, x = value_91_cast_fp16)[name = string("key_states_47_cast_fp16")]; tensor var_6432_axes_0 = const()[name = string("op_6432_axes_0"), val = tensor([1])]; tensor var_6432_cast_fp16 = expand_dims(axes = var_6432_axes_0, x = value_cache_23_cast_fp16)[name = string("op_6432_cast_fp16")]; tensor var_6437 = const()[name = string("op_6437"), val = tensor([1, 2, 1, 1])]; tensor value_95_cast_fp16 = tile(reps = var_6437, x = var_6432_cast_fp16)[name = string("value_95_cast_fp16")]; tensor var_6443 = const()[name = string("op_6443"), val = tensor([1, 16, 1536, 128])]; tensor value_states_69_cast_fp16 = reshape(shape = var_6443, x = value_95_cast_fp16)[name = string("value_states_69_cast_fp16")]; bool var_6458_transpose_x_1 = const()[name = string("op_6458_transpose_x_1"), val = bool(false)]; bool var_6458_transpose_y_1 = const()[name = string("op_6458_transpose_y_1"), val = bool(true)]; tensor var_6458 = matmul(transpose_x = var_6458_transpose_x_1, transpose_y = var_6458_transpose_y_1, x = query_states_45, y = key_states_47_cast_fp16)[name = string("op_6458")]; fp16 var_6459_to_fp16 = const()[name = string("op_6459_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_45_cast_fp16 = mul(x = var_6458, y = var_6459_to_fp16)[name = string("attention_45_cast_fp16")]; tensor attention_47_cast_fp16 = add(x = attention_45_cast_fp16, y = causal_mask)[name = string("attention_47_cast_fp16")]; int32 var_6468 = const()[name = string("op_6468"), val = int32(-1)]; tensor probabilities_23_cast_fp16 = softmax(axis = var_6468, x = attention_47_cast_fp16)[name = string("probabilities_23_cast_fp16")]; bool output_67_transpose_x_0 = const()[name = string("output_67_transpose_x_0"), val = bool(false)]; bool output_67_transpose_y_0 = const()[name = string("output_67_transpose_y_0"), val = bool(false)]; tensor output_67_cast_fp16 = matmul(transpose_x = output_67_transpose_x_0, transpose_y = output_67_transpose_y_0, x = probabilities_23_cast_fp16, y = value_states_69_cast_fp16)[name = string("output_67_cast_fp16")]; tensor var_6479_perm_0 = const()[name = string("op_6479_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_6485 = const()[name = string("op_6485"), val = tensor([1, 1, 2048])]; tensor var_6479_cast_fp16 = transpose(perm = var_6479_perm_0, x = output_67_cast_fp16)[name = string("transpose_100")]; tensor output_69_cast_fp16 = reshape(shape = var_6485, x = var_6479_cast_fp16)[name = string("output_69_cast_fp16")]; tensor var_6490 = const()[name = string("op_6490"), val = tensor([0, 2, 1])]; string var_6506_pad_type_0 = const()[name = string("op_6506_pad_type_0"), val = string("valid")]; int32 var_6506_groups_0 = const()[name = string("op_6506_groups_0"), val = int32(1)]; tensor var_6506_strides_0 = const()[name = string("op_6506_strides_0"), val = tensor([1])]; tensor var_6506_pad_0 = const()[name = string("op_6506_pad_0"), val = tensor([0, 0])]; tensor var_6506_dilations_0 = const()[name = string("op_6506_dilations_0"), val = tensor([1])]; tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310441472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312014400))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6491_cast_fp16 = transpose(perm = var_6490, x = output_69_cast_fp16)[name = string("transpose_99")]; tensor var_6506_cast_fp16 = conv(dilations = var_6506_dilations_0, groups = var_6506_groups_0, pad = var_6506_pad_0, pad_type = var_6506_pad_type_0, strides = var_6506_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_6491_cast_fp16)[name = string("op_6506_cast_fp16")]; tensor var_6510 = const()[name = string("op_6510"), val = tensor([0, 2, 1])]; tensor attn_output_23_cast_fp16 = transpose(perm = var_6510, x = var_6506_cast_fp16)[name = string("transpose_98")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = attn_output_23_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; int32 var_6525 = const()[name = string("op_6525"), val = int32(-1)]; fp16 const_165_promoted_to_fp16 = const()[name = string("const_165_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6527_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_165_promoted_to_fp16)[name = string("op_6527_cast_fp16")]; bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; tensor input_209_cast_fp16 = concat(axis = var_6525, interleave = input_209_interleave_0, values = (hidden_states_119_cast_fp16, var_6527_cast_fp16))[name = string("input_209_cast_fp16")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_6522_to_fp16 = const()[name = string("op_6522_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_6522_to_fp16, x = input_209_cast_fp16)[name = string("normed_189_cast_fp16")]; tensor normed_191_begin_0 = const()[name = string("normed_191_begin_0"), val = tensor([0, 0, 0])]; tensor normed_191_end_0 = const()[name = string("normed_191_end_0"), val = tensor([1, 1, 1024])]; tensor normed_191_end_mask_0 = const()[name = string("normed_191_end_mask_0"), val = tensor([true, true, false])]; tensor normed_191_cast_fp16 = slice_by_index(begin = normed_191_begin_0, end = normed_191_end_0, end_mask = normed_191_end_mask_0, x = normed_189_cast_fp16)[name = string("normed_191_cast_fp16")]; tensor const_167_promoted_to_fp16 = const()[name = string("const_167_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312030848)))]; tensor x_45_cast_fp16 = mul(x = normed_191_cast_fp16, y = const_167_promoted_to_fp16)[name = string("x_45_cast_fp16")]; tensor var_6547 = const()[name = string("op_6547"), val = tensor([0, 2, 1])]; tensor input_211_axes_0 = const()[name = string("input_211_axes_0"), val = tensor([2])]; tensor var_6548 = transpose(perm = var_6547, x = x_45_cast_fp16)[name = string("transpose_97")]; tensor input_211 = expand_dims(axes = input_211_axes_0, x = var_6548)[name = string("input_211")]; string input_213_pad_type_0 = const()[name = string("input_213_pad_type_0"), val = string("valid")]; tensor input_213_strides_0 = const()[name = string("input_213_strides_0"), val = tensor([1, 1])]; tensor input_213_pad_0 = const()[name = string("input_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_213_dilations_0 = const()[name = string("input_213_dilations_0"), val = tensor([1, 1])]; int32 input_213_groups_0 = const()[name = string("input_213_groups_0"), val = int32(1)]; tensor input_213 = conv(dilations = input_213_dilations_0, groups = input_213_groups_0, pad = input_213_pad_0, pad_type = input_213_pad_type_0, strides = input_213_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_211)[name = string("input_213")]; string b_23_pad_type_0 = const()[name = string("b_23_pad_type_0"), val = string("valid")]; tensor b_23_strides_0 = const()[name = string("b_23_strides_0"), val = tensor([1, 1])]; tensor b_23_pad_0 = const()[name = string("b_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_23_dilations_0 = const()[name = string("b_23_dilations_0"), val = tensor([1, 1])]; int32 b_23_groups_0 = const()[name = string("b_23_groups_0"), val = int32(1)]; tensor b_23 = conv(dilations = b_23_dilations_0, groups = b_23_groups_0, pad = b_23_pad_0, pad_type = b_23_pad_type_0, strides = b_23_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_211)[name = string("b_23")]; tensor c_23 = silu(x = input_213)[name = string("c_23")]; tensor input_215 = mul(x = c_23, y = b_23)[name = string("input_215")]; string e_23_pad_type_0 = const()[name = string("e_23_pad_type_0"), val = string("valid")]; tensor e_23_strides_0 = const()[name = string("e_23_strides_0"), val = tensor([1, 1])]; tensor e_23_pad_0 = const()[name = string("e_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_23_dilations_0 = const()[name = string("e_23_dilations_0"), val = tensor([1, 1])]; int32 e_23_groups_0 = const()[name = string("e_23_groups_0"), val = int32(1)]; tensor e_23 = conv(dilations = e_23_dilations_0, groups = e_23_groups_0, pad = e_23_pad_0, pad_type = e_23_pad_type_0, strides = e_23_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_215)[name = string("e_23")]; tensor var_6570_axes_0 = const()[name = string("op_6570_axes_0"), val = tensor([2])]; tensor var_6570 = squeeze(axes = var_6570_axes_0, x = e_23)[name = string("op_6570")]; tensor var_6571 = const()[name = string("op_6571"), val = tensor([0, 2, 1])]; tensor var_6572 = transpose(perm = var_6571, x = var_6570)[name = string("transpose_96")]; tensor hidden_states_121_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = var_6572)[name = string("hidden_states_121_cast_fp16")]; int32 var_6586 = const()[name = string("op_6586"), val = int32(-1)]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6588_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6588_cast_fp16")]; bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; tensor input_217_cast_fp16 = concat(axis = var_6586, interleave = input_217_interleave_0, values = (hidden_states_121_cast_fp16, var_6588_cast_fp16))[name = string("input_217_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_6583_to_fp16 = const()[name = string("op_6583_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_6583_to_fp16, x = input_217_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor normed_195_begin_0 = const()[name = string("normed_195_begin_0"), val = tensor([0, 0, 0])]; tensor normed_195_end_0 = const()[name = string("normed_195_end_0"), val = tensor([1, 1, 1024])]; tensor normed_195_end_mask_0 = const()[name = string("normed_195_end_mask_0"), val = tensor([true, true, false])]; tensor normed_195_cast_fp16 = slice_by_index(begin = normed_195_begin_0, end = normed_195_end_0, end_mask = normed_195_end_mask_0, x = normed_193_cast_fp16)[name = string("normed_195_cast_fp16")]; tensor const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312032960)))]; tensor hidden_states_123_cast_fp16 = mul(x = normed_195_cast_fp16, y = const_170_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor var_6600 = const()[name = string("op_6600"), val = tensor([0, 2, 1])]; tensor var_6603_axes_0 = const()[name = string("op_6603_axes_0"), val = tensor([2])]; tensor var_6601_cast_fp16 = transpose(perm = var_6600, x = hidden_states_123_cast_fp16)[name = string("transpose_95")]; tensor var_6603_cast_fp16 = expand_dims(axes = var_6603_axes_0, x = var_6601_cast_fp16)[name = string("op_6603_cast_fp16")]; string var_6619_pad_type_0 = const()[name = string("op_6619_pad_type_0"), val = string("valid")]; tensor var_6619_strides_0 = const()[name = string("op_6619_strides_0"), val = tensor([1, 1])]; tensor var_6619_pad_0 = const()[name = string("op_6619_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6619_dilations_0 = const()[name = string("op_6619_dilations_0"), val = tensor([1, 1])]; int32 var_6619_groups_0 = const()[name = string("op_6619_groups_0"), val = int32(1)]; tensor var_6619 = conv(dilations = var_6619_dilations_0, groups = var_6619_groups_0, pad = var_6619_pad_0, pad_type = var_6619_pad_type_0, strides = var_6619_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_6603_cast_fp16)[name = string("op_6619")]; tensor var_6624 = const()[name = string("op_6624"), val = tensor([1, 16, 1, 128])]; tensor var_6625 = reshape(shape = var_6624, x = var_6619)[name = string("op_6625")]; string var_6641_pad_type_0 = const()[name = string("op_6641_pad_type_0"), val = string("valid")]; tensor var_6641_strides_0 = const()[name = string("op_6641_strides_0"), val = tensor([1, 1])]; tensor var_6641_pad_0 = const()[name = string("op_6641_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6641_dilations_0 = const()[name = string("op_6641_dilations_0"), val = tensor([1, 1])]; int32 var_6641_groups_0 = const()[name = string("op_6641_groups_0"), val = int32(1)]; tensor var_6641 = conv(dilations = var_6641_dilations_0, groups = var_6641_groups_0, pad = var_6641_pad_0, pad_type = var_6641_pad_type_0, strides = var_6641_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_6603_cast_fp16)[name = string("op_6641")]; tensor var_6646 = const()[name = string("op_6646"), val = tensor([1, 8, 1, 128])]; tensor var_6647 = reshape(shape = var_6646, x = var_6641)[name = string("op_6647")]; string var_6663_pad_type_0 = const()[name = string("op_6663_pad_type_0"), val = string("valid")]; tensor var_6663_strides_0 = const()[name = string("op_6663_strides_0"), val = tensor([1, 1])]; tensor var_6663_pad_0 = const()[name = string("op_6663_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6663_dilations_0 = const()[name = string("op_6663_dilations_0"), val = tensor([1, 1])]; int32 var_6663_groups_0 = const()[name = string("op_6663_groups_0"), val = int32(1)]; tensor var_6663 = conv(dilations = var_6663_dilations_0, groups = var_6663_groups_0, pad = var_6663_pad_0, pad_type = var_6663_pad_type_0, strides = var_6663_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_6603_cast_fp16)[name = string("op_6663")]; tensor var_6668 = const()[name = string("op_6668"), val = tensor([1, 8, 1, 128])]; tensor var_6669 = reshape(shape = var_6668, x = var_6663)[name = string("op_6669")]; int32 var_6686 = const()[name = string("op_6686"), val = int32(-1)]; fp16 const_171_promoted = const()[name = string("const_171_promoted"), val = fp16(-0x1p+0)]; tensor var_6688 = mul(x = var_6625, y = const_171_promoted)[name = string("op_6688")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221 = concat(axis = var_6686, interleave = input_221_interleave_0, values = (var_6625, var_6688))[name = string("input_221")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_6683_to_fp16 = const()[name = string("op_6683_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_6683_to_fp16, x = input_221)[name = string("normed_197_cast_fp16")]; tensor normed_199_begin_0 = const()[name = string("normed_199_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_199_end_0 = const()[name = string("normed_199_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_199_end_mask_0 = const()[name = string("normed_199_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_199 = slice_by_index(begin = normed_199_begin_0, end = normed_199_end_0, end_mask = normed_199_end_mask_0, x = normed_197_cast_fp16)[name = string("normed_199")]; tensor const_173 = const()[name = string("const_173"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312035072)))]; tensor q_25 = mul(x = normed_199, y = const_173)[name = string("q_25")]; int32 var_6708 = const()[name = string("op_6708"), val = int32(-1)]; fp16 const_174_promoted = const()[name = string("const_174_promoted"), val = fp16(-0x1p+0)]; tensor var_6710 = mul(x = var_6647, y = const_174_promoted)[name = string("op_6710")]; bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; tensor input_223 = concat(axis = var_6708, interleave = input_223_interleave_0, values = (var_6647, var_6710))[name = string("input_223")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_6705_to_fp16 = const()[name = string("op_6705_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_6705_to_fp16, x = input_223)[name = string("normed_201_cast_fp16")]; tensor normed_203_begin_0 = const()[name = string("normed_203_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_203_end_0 = const()[name = string("normed_203_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_203_end_mask_0 = const()[name = string("normed_203_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_203 = slice_by_index(begin = normed_203_begin_0, end = normed_203_end_0, end_mask = normed_203_end_mask_0, x = normed_201_cast_fp16)[name = string("normed_203")]; tensor const_176 = const()[name = string("const_176"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312035392)))]; tensor k_25 = mul(x = normed_203, y = const_176)[name = string("k_25")]; tensor var_6719 = mul(x = q_25, y = cos_1_cast_fp16)[name = string("op_6719")]; tensor var_6724_begin_0 = const()[name = string("op_6724_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6724_end_0 = const()[name = string("op_6724_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_6724_end_mask_0 = const()[name = string("op_6724_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6724 = slice_by_index(begin = var_6724_begin_0, end = var_6724_end_0, end_mask = var_6724_end_mask_0, x = q_25)[name = string("op_6724")]; fp16 const_177_promoted = const()[name = string("const_177_promoted"), val = fp16(-0x1p+0)]; tensor var_6725 = mul(x = var_6724, y = const_177_promoted)[name = string("op_6725")]; tensor var_6730_begin_0 = const()[name = string("op_6730_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6730_end_0 = const()[name = string("op_6730_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_6730_end_mask_0 = const()[name = string("op_6730_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6730 = slice_by_index(begin = var_6730_begin_0, end = var_6730_end_0, end_mask = var_6730_end_mask_0, x = q_25)[name = string("op_6730")]; int32 var_6732 = const()[name = string("op_6732"), val = int32(-1)]; bool var_6733_interleave_0 = const()[name = string("op_6733_interleave_0"), val = bool(false)]; tensor var_6733 = concat(axis = var_6732, interleave = var_6733_interleave_0, values = (var_6725, var_6730))[name = string("op_6733")]; tensor var_6734 = mul(x = var_6733, y = sin_1_cast_fp16)[name = string("op_6734")]; tensor query_states_49 = add(x = var_6719, y = var_6734)[name = string("query_states_49")]; tensor var_6737 = mul(x = k_25, y = cos_1_cast_fp16)[name = string("op_6737")]; tensor var_6742_begin_0 = const()[name = string("op_6742_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6742_end_0 = const()[name = string("op_6742_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_6742_end_mask_0 = const()[name = string("op_6742_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6742 = slice_by_index(begin = var_6742_begin_0, end = var_6742_end_0, end_mask = var_6742_end_mask_0, x = k_25)[name = string("op_6742")]; fp16 const_178_promoted = const()[name = string("const_178_promoted"), val = fp16(-0x1p+0)]; tensor var_6743 = mul(x = var_6742, y = const_178_promoted)[name = string("op_6743")]; tensor var_6748_begin_0 = const()[name = string("op_6748_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6748_end_0 = const()[name = string("op_6748_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_6748_end_mask_0 = const()[name = string("op_6748_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6748 = slice_by_index(begin = var_6748_begin_0, end = var_6748_end_0, end_mask = var_6748_end_mask_0, x = k_25)[name = string("op_6748")]; int32 var_6750 = const()[name = string("op_6750"), val = int32(-1)]; bool var_6751_interleave_0 = const()[name = string("op_6751_interleave_0"), val = bool(false)]; tensor var_6751 = concat(axis = var_6750, interleave = var_6751_interleave_0, values = (var_6743, var_6748))[name = string("op_6751")]; tensor var_6752 = mul(x = var_6751, y = sin_1_cast_fp16)[name = string("op_6752")]; tensor key_states_49 = add(x = var_6737, y = var_6752)[name = string("key_states_49")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; int32 concat_98_axis_0 = const()[name = string("concat_98_axis_0"), val = int32(0)]; bool concat_98_interleave_0 = const()[name = string("concat_98_interleave_0"), val = bool(false)]; tensor concat_98 = concat(axis = concat_98_axis_0, interleave = concat_98_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_98")]; tensor concat_99_values1_0 = const()[name = string("concat_99_values1_0"), val = tensor([0])]; tensor concat_99_values3_0 = const()[name = string("concat_99_values3_0"), val = tensor([0])]; int32 concat_99_axis_0 = const()[name = string("concat_99_axis_0"), val = int32(0)]; bool concat_99_interleave_0 = const()[name = string("concat_99_interleave_0"), val = bool(false)]; tensor concat_99 = concat(axis = concat_99_axis_0, interleave = concat_99_interleave_0, values = (expand_dims_148, concat_99_values1_0, var_1717, concat_99_values3_0))[name = string("concat_99")]; tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_98, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_99, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = key_states_49, x = coreml_update_state_79)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_136_write_state")]; tensor coreml_update_state_80 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_136")]; tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([40])]; tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([41])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_102")]; tensor concat_103_values1_0 = const()[name = string("concat_103_values1_0"), val = tensor([0])]; tensor concat_103_values3_0 = const()[name = string("concat_103_values3_0"), val = tensor([0])]; int32 concat_103_axis_0 = const()[name = string("concat_103_axis_0"), val = int32(0)]; bool concat_103_interleave_0 = const()[name = string("concat_103_interleave_0"), val = bool(false)]; tensor concat_103 = concat(axis = concat_103_axis_0, interleave = concat_103_interleave_0, values = (expand_dims_154, concat_103_values1_0, var_1717, concat_103_values3_0))[name = string("concat_103")]; tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_102, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_103, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = var_6669, x = coreml_update_state_80)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_137_write_state")]; tensor coreml_update_state_81 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_137")]; tensor var_6807_begin_0 = const()[name = string("op_6807_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_6807_end_0 = const()[name = string("op_6807_end_0"), val = tensor([13, 8, 1536, 128])]; tensor var_6807_end_mask_0 = const()[name = string("op_6807_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6807_cast_fp16 = slice_by_index(begin = var_6807_begin_0, end = var_6807_end_0, end_mask = var_6807_end_mask_0, x = coreml_update_state_81)[name = string("op_6807_cast_fp16")]; tensor key_cache_25_axes_0 = const()[name = string("key_cache_25_axes_0"), val = tensor([0])]; tensor key_cache_25_cast_fp16 = squeeze(axes = key_cache_25_axes_0, x = var_6807_cast_fp16)[name = string("key_cache_25_cast_fp16")]; tensor var_6814_begin_0 = const()[name = string("op_6814_begin_0"), val = tensor([40, 0, 0, 0])]; tensor var_6814_end_0 = const()[name = string("op_6814_end_0"), val = tensor([41, 8, 1536, 128])]; tensor var_6814_end_mask_0 = const()[name = string("op_6814_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6814_cast_fp16 = slice_by_index(begin = var_6814_begin_0, end = var_6814_end_0, end_mask = var_6814_end_mask_0, x = coreml_update_state_81)[name = string("op_6814_cast_fp16")]; tensor value_cache_25_axes_0 = const()[name = string("value_cache_25_axes_0"), val = tensor([0])]; tensor value_cache_25_cast_fp16 = squeeze(axes = value_cache_25_axes_0, x = var_6814_cast_fp16)[name = string("value_cache_25_cast_fp16")]; tensor var_6838_axes_0 = const()[name = string("op_6838_axes_0"), val = tensor([1])]; tensor var_6838_cast_fp16 = expand_dims(axes = var_6838_axes_0, x = key_cache_25_cast_fp16)[name = string("op_6838_cast_fp16")]; tensor var_6843 = const()[name = string("op_6843"), val = tensor([1, 2, 1, 1])]; tensor value_99_cast_fp16 = tile(reps = var_6843, x = var_6838_cast_fp16)[name = string("value_99_cast_fp16")]; tensor var_6849 = const()[name = string("op_6849"), val = tensor([1, 16, 1536, 128])]; tensor key_states_51_cast_fp16 = reshape(shape = var_6849, x = value_99_cast_fp16)[name = string("key_states_51_cast_fp16")]; tensor var_6852_axes_0 = const()[name = string("op_6852_axes_0"), val = tensor([1])]; tensor var_6852_cast_fp16 = expand_dims(axes = var_6852_axes_0, x = value_cache_25_cast_fp16)[name = string("op_6852_cast_fp16")]; tensor var_6857 = const()[name = string("op_6857"), val = tensor([1, 2, 1, 1])]; tensor value_103_cast_fp16 = tile(reps = var_6857, x = var_6852_cast_fp16)[name = string("value_103_cast_fp16")]; tensor var_6863 = const()[name = string("op_6863"), val = tensor([1, 16, 1536, 128])]; tensor value_states_75_cast_fp16 = reshape(shape = var_6863, x = value_103_cast_fp16)[name = string("value_states_75_cast_fp16")]; bool var_6878_transpose_x_1 = const()[name = string("op_6878_transpose_x_1"), val = bool(false)]; bool var_6878_transpose_y_1 = const()[name = string("op_6878_transpose_y_1"), val = bool(true)]; tensor var_6878 = matmul(transpose_x = var_6878_transpose_x_1, transpose_y = var_6878_transpose_y_1, x = query_states_49, y = key_states_51_cast_fp16)[name = string("op_6878")]; fp16 var_6879_to_fp16 = const()[name = string("op_6879_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_49_cast_fp16 = mul(x = var_6878, y = var_6879_to_fp16)[name = string("attention_49_cast_fp16")]; tensor attention_51_cast_fp16 = add(x = attention_49_cast_fp16, y = causal_mask)[name = string("attention_51_cast_fp16")]; int32 var_6888 = const()[name = string("op_6888"), val = int32(-1)]; tensor probabilities_25_cast_fp16 = softmax(axis = var_6888, x = attention_51_cast_fp16)[name = string("probabilities_25_cast_fp16")]; bool output_73_transpose_x_0 = const()[name = string("output_73_transpose_x_0"), val = bool(false)]; bool output_73_transpose_y_0 = const()[name = string("output_73_transpose_y_0"), val = bool(false)]; tensor output_73_cast_fp16 = matmul(transpose_x = output_73_transpose_x_0, transpose_y = output_73_transpose_y_0, x = probabilities_25_cast_fp16, y = value_states_75_cast_fp16)[name = string("output_73_cast_fp16")]; tensor var_6899_perm_0 = const()[name = string("op_6899_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_6905 = const()[name = string("op_6905"), val = tensor([1, 1, 2048])]; tensor var_6899_cast_fp16 = transpose(perm = var_6899_perm_0, x = output_73_cast_fp16)[name = string("transpose_94")]; tensor output_75_cast_fp16 = reshape(shape = var_6905, x = var_6899_cast_fp16)[name = string("output_75_cast_fp16")]; tensor var_6910 = const()[name = string("op_6910"), val = tensor([0, 2, 1])]; string var_6926_pad_type_0 = const()[name = string("op_6926_pad_type_0"), val = string("valid")]; int32 var_6926_groups_0 = const()[name = string("op_6926_groups_0"), val = int32(1)]; tensor var_6926_strides_0 = const()[name = string("op_6926_strides_0"), val = tensor([1])]; tensor var_6926_pad_0 = const()[name = string("op_6926_pad_0"), val = tensor([0, 0])]; tensor var_6926_dilations_0 = const()[name = string("op_6926_dilations_0"), val = tensor([1])]; tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312035712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313608640))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6911_cast_fp16 = transpose(perm = var_6910, x = output_75_cast_fp16)[name = string("transpose_93")]; tensor var_6926_cast_fp16 = conv(dilations = var_6926_dilations_0, groups = var_6926_groups_0, pad = var_6926_pad_0, pad_type = var_6926_pad_type_0, strides = var_6926_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_6911_cast_fp16)[name = string("op_6926_cast_fp16")]; tensor var_6930 = const()[name = string("op_6930"), val = tensor([0, 2, 1])]; tensor attn_output_25_cast_fp16 = transpose(perm = var_6930, x = var_6926_cast_fp16)[name = string("transpose_92")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = attn_output_25_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; int32 var_6945 = const()[name = string("op_6945"), val = int32(-1)]; fp16 const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6947_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_179_promoted_to_fp16)[name = string("op_6947_cast_fp16")]; bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; tensor input_227_cast_fp16 = concat(axis = var_6945, interleave = input_227_interleave_0, values = (hidden_states_129_cast_fp16, var_6947_cast_fp16))[name = string("input_227_cast_fp16")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_6942_to_fp16 = const()[name = string("op_6942_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_6942_to_fp16, x = input_227_cast_fp16)[name = string("normed_205_cast_fp16")]; tensor normed_207_begin_0 = const()[name = string("normed_207_begin_0"), val = tensor([0, 0, 0])]; tensor normed_207_end_0 = const()[name = string("normed_207_end_0"), val = tensor([1, 1, 1024])]; tensor normed_207_end_mask_0 = const()[name = string("normed_207_end_mask_0"), val = tensor([true, true, false])]; tensor normed_207_cast_fp16 = slice_by_index(begin = normed_207_begin_0, end = normed_207_end_0, end_mask = normed_207_end_mask_0, x = normed_205_cast_fp16)[name = string("normed_207_cast_fp16")]; tensor const_181_promoted_to_fp16 = const()[name = string("const_181_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313625088)))]; tensor x_49_cast_fp16 = mul(x = normed_207_cast_fp16, y = const_181_promoted_to_fp16)[name = string("x_49_cast_fp16")]; tensor var_6967 = const()[name = string("op_6967"), val = tensor([0, 2, 1])]; tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; tensor var_6968 = transpose(perm = var_6967, x = x_49_cast_fp16)[name = string("transpose_91")]; tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_6968)[name = string("input_229")]; string input_231_pad_type_0 = const()[name = string("input_231_pad_type_0"), val = string("valid")]; tensor input_231_strides_0 = const()[name = string("input_231_strides_0"), val = tensor([1, 1])]; tensor input_231_pad_0 = const()[name = string("input_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_231_dilations_0 = const()[name = string("input_231_dilations_0"), val = tensor([1, 1])]; int32 input_231_groups_0 = const()[name = string("input_231_groups_0"), val = int32(1)]; tensor input_231 = conv(dilations = input_231_dilations_0, groups = input_231_groups_0, pad = input_231_pad_0, pad_type = input_231_pad_type_0, strides = input_231_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_229)[name = string("input_231")]; string b_25_pad_type_0 = const()[name = string("b_25_pad_type_0"), val = string("valid")]; tensor b_25_strides_0 = const()[name = string("b_25_strides_0"), val = tensor([1, 1])]; tensor b_25_pad_0 = const()[name = string("b_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_25_dilations_0 = const()[name = string("b_25_dilations_0"), val = tensor([1, 1])]; int32 b_25_groups_0 = const()[name = string("b_25_groups_0"), val = int32(1)]; tensor b_25 = conv(dilations = b_25_dilations_0, groups = b_25_groups_0, pad = b_25_pad_0, pad_type = b_25_pad_type_0, strides = b_25_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_229)[name = string("b_25")]; tensor c_25 = silu(x = input_231)[name = string("c_25")]; tensor input_233 = mul(x = c_25, y = b_25)[name = string("input_233")]; string e_25_pad_type_0 = const()[name = string("e_25_pad_type_0"), val = string("valid")]; tensor e_25_strides_0 = const()[name = string("e_25_strides_0"), val = tensor([1, 1])]; tensor e_25_pad_0 = const()[name = string("e_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_25_dilations_0 = const()[name = string("e_25_dilations_0"), val = tensor([1, 1])]; int32 e_25_groups_0 = const()[name = string("e_25_groups_0"), val = int32(1)]; tensor e_25 = conv(dilations = e_25_dilations_0, groups = e_25_groups_0, pad = e_25_pad_0, pad_type = e_25_pad_type_0, strides = e_25_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_233)[name = string("e_25")]; tensor var_6990_axes_0 = const()[name = string("op_6990_axes_0"), val = tensor([2])]; tensor var_6990 = squeeze(axes = var_6990_axes_0, x = e_25)[name = string("op_6990")]; tensor var_6991 = const()[name = string("op_6991"), val = tensor([0, 2, 1])]; tensor var_6992 = transpose(perm = var_6991, x = var_6990)[name = string("transpose_90")]; tensor hidden_states_131_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_6992)[name = string("hidden_states_131_cast_fp16")]; int32 var_7006 = const()[name = string("op_7006"), val = int32(-1)]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7008_cast_fp16 = mul(x = hidden_states_131_cast_fp16, y = const_182_promoted_to_fp16)[name = string("op_7008_cast_fp16")]; bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; tensor input_235_cast_fp16 = concat(axis = var_7006, interleave = input_235_interleave_0, values = (hidden_states_131_cast_fp16, var_7008_cast_fp16))[name = string("input_235_cast_fp16")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_7003_to_fp16 = const()[name = string("op_7003_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_7003_to_fp16, x = input_235_cast_fp16)[name = string("normed_209_cast_fp16")]; tensor normed_211_begin_0 = const()[name = string("normed_211_begin_0"), val = tensor([0, 0, 0])]; tensor normed_211_end_0 = const()[name = string("normed_211_end_0"), val = tensor([1, 1, 1024])]; tensor normed_211_end_mask_0 = const()[name = string("normed_211_end_mask_0"), val = tensor([true, true, false])]; tensor normed_211_cast_fp16 = slice_by_index(begin = normed_211_begin_0, end = normed_211_end_0, end_mask = normed_211_end_mask_0, x = normed_209_cast_fp16)[name = string("normed_211_cast_fp16")]; tensor const_184_promoted_to_fp16 = const()[name = string("const_184_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313627200)))]; tensor hidden_states_133_cast_fp16 = mul(x = normed_211_cast_fp16, y = const_184_promoted_to_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor var_7020 = const()[name = string("op_7020"), val = tensor([0, 2, 1])]; tensor var_7023_axes_0 = const()[name = string("op_7023_axes_0"), val = tensor([2])]; tensor var_7021_cast_fp16 = transpose(perm = var_7020, x = hidden_states_133_cast_fp16)[name = string("transpose_89")]; tensor var_7023_cast_fp16 = expand_dims(axes = var_7023_axes_0, x = var_7021_cast_fp16)[name = string("op_7023_cast_fp16")]; string var_7039_pad_type_0 = const()[name = string("op_7039_pad_type_0"), val = string("valid")]; tensor var_7039_strides_0 = const()[name = string("op_7039_strides_0"), val = tensor([1, 1])]; tensor var_7039_pad_0 = const()[name = string("op_7039_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7039_dilations_0 = const()[name = string("op_7039_dilations_0"), val = tensor([1, 1])]; int32 var_7039_groups_0 = const()[name = string("op_7039_groups_0"), val = int32(1)]; tensor var_7039 = conv(dilations = var_7039_dilations_0, groups = var_7039_groups_0, pad = var_7039_pad_0, pad_type = var_7039_pad_type_0, strides = var_7039_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_7023_cast_fp16)[name = string("op_7039")]; tensor var_7044 = const()[name = string("op_7044"), val = tensor([1, 16, 1, 128])]; tensor var_7045 = reshape(shape = var_7044, x = var_7039)[name = string("op_7045")]; string var_7061_pad_type_0 = const()[name = string("op_7061_pad_type_0"), val = string("valid")]; tensor var_7061_strides_0 = const()[name = string("op_7061_strides_0"), val = tensor([1, 1])]; tensor var_7061_pad_0 = const()[name = string("op_7061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7061_dilations_0 = const()[name = string("op_7061_dilations_0"), val = tensor([1, 1])]; int32 var_7061_groups_0 = const()[name = string("op_7061_groups_0"), val = int32(1)]; tensor var_7061 = conv(dilations = var_7061_dilations_0, groups = var_7061_groups_0, pad = var_7061_pad_0, pad_type = var_7061_pad_type_0, strides = var_7061_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_7023_cast_fp16)[name = string("op_7061")]; tensor var_7066 = const()[name = string("op_7066"), val = tensor([1, 8, 1, 128])]; tensor var_7067 = reshape(shape = var_7066, x = var_7061)[name = string("op_7067")]; string var_7083_pad_type_0 = const()[name = string("op_7083_pad_type_0"), val = string("valid")]; tensor var_7083_strides_0 = const()[name = string("op_7083_strides_0"), val = tensor([1, 1])]; tensor var_7083_pad_0 = const()[name = string("op_7083_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7083_dilations_0 = const()[name = string("op_7083_dilations_0"), val = tensor([1, 1])]; int32 var_7083_groups_0 = const()[name = string("op_7083_groups_0"), val = int32(1)]; tensor var_7083 = conv(dilations = var_7083_dilations_0, groups = var_7083_groups_0, pad = var_7083_pad_0, pad_type = var_7083_pad_type_0, strides = var_7083_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_7023_cast_fp16)[name = string("op_7083")]; tensor var_7088 = const()[name = string("op_7088"), val = tensor([1, 8, 1, 128])]; tensor var_7089 = reshape(shape = var_7088, x = var_7083)[name = string("op_7089")]; int32 var_7106 = const()[name = string("op_7106"), val = int32(-1)]; fp16 const_185_promoted = const()[name = string("const_185_promoted"), val = fp16(-0x1p+0)]; tensor var_7108 = mul(x = var_7045, y = const_185_promoted)[name = string("op_7108")]; bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; tensor input_239 = concat(axis = var_7106, interleave = input_239_interleave_0, values = (var_7045, var_7108))[name = string("input_239")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_7103_to_fp16 = const()[name = string("op_7103_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_7103_to_fp16, x = input_239)[name = string("normed_213_cast_fp16")]; tensor normed_215_begin_0 = const()[name = string("normed_215_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_215_end_0 = const()[name = string("normed_215_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_215_end_mask_0 = const()[name = string("normed_215_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_215 = slice_by_index(begin = normed_215_begin_0, end = normed_215_end_0, end_mask = normed_215_end_mask_0, x = normed_213_cast_fp16)[name = string("normed_215")]; tensor const_187 = const()[name = string("const_187"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313629312)))]; tensor q_27 = mul(x = normed_215, y = const_187)[name = string("q_27")]; int32 var_7128 = const()[name = string("op_7128"), val = int32(-1)]; fp16 const_188_promoted = const()[name = string("const_188_promoted"), val = fp16(-0x1p+0)]; tensor var_7130 = mul(x = var_7067, y = const_188_promoted)[name = string("op_7130")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241 = concat(axis = var_7128, interleave = input_241_interleave_0, values = (var_7067, var_7130))[name = string("input_241")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_7125_to_fp16 = const()[name = string("op_7125_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_7125_to_fp16, x = input_241)[name = string("normed_217_cast_fp16")]; tensor normed_219_begin_0 = const()[name = string("normed_219_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_219_end_0 = const()[name = string("normed_219_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_219_end_mask_0 = const()[name = string("normed_219_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_219 = slice_by_index(begin = normed_219_begin_0, end = normed_219_end_0, end_mask = normed_219_end_mask_0, x = normed_217_cast_fp16)[name = string("normed_219")]; tensor const_190 = const()[name = string("const_190"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313629632)))]; tensor k_27 = mul(x = normed_219, y = const_190)[name = string("k_27")]; tensor var_7139 = mul(x = q_27, y = cos_1_cast_fp16)[name = string("op_7139")]; tensor var_7144_begin_0 = const()[name = string("op_7144_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7144_end_0 = const()[name = string("op_7144_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_7144_end_mask_0 = const()[name = string("op_7144_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7144 = slice_by_index(begin = var_7144_begin_0, end = var_7144_end_0, end_mask = var_7144_end_mask_0, x = q_27)[name = string("op_7144")]; fp16 const_191_promoted = const()[name = string("const_191_promoted"), val = fp16(-0x1p+0)]; tensor var_7145 = mul(x = var_7144, y = const_191_promoted)[name = string("op_7145")]; tensor var_7150_begin_0 = const()[name = string("op_7150_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7150_end_0 = const()[name = string("op_7150_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_7150_end_mask_0 = const()[name = string("op_7150_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7150 = slice_by_index(begin = var_7150_begin_0, end = var_7150_end_0, end_mask = var_7150_end_mask_0, x = q_27)[name = string("op_7150")]; int32 var_7152 = const()[name = string("op_7152"), val = int32(-1)]; bool var_7153_interleave_0 = const()[name = string("op_7153_interleave_0"), val = bool(false)]; tensor var_7153 = concat(axis = var_7152, interleave = var_7153_interleave_0, values = (var_7145, var_7150))[name = string("op_7153")]; tensor var_7154 = mul(x = var_7153, y = sin_1_cast_fp16)[name = string("op_7154")]; tensor query_states_53 = add(x = var_7139, y = var_7154)[name = string("query_states_53")]; tensor var_7157 = mul(x = k_27, y = cos_1_cast_fp16)[name = string("op_7157")]; tensor var_7162_begin_0 = const()[name = string("op_7162_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7162_end_0 = const()[name = string("op_7162_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_7162_end_mask_0 = const()[name = string("op_7162_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7162 = slice_by_index(begin = var_7162_begin_0, end = var_7162_end_0, end_mask = var_7162_end_mask_0, x = k_27)[name = string("op_7162")]; fp16 const_192_promoted = const()[name = string("const_192_promoted"), val = fp16(-0x1p+0)]; tensor var_7163 = mul(x = var_7162, y = const_192_promoted)[name = string("op_7163")]; tensor var_7168_begin_0 = const()[name = string("op_7168_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7168_end_0 = const()[name = string("op_7168_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_7168_end_mask_0 = const()[name = string("op_7168_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7168 = slice_by_index(begin = var_7168_begin_0, end = var_7168_end_0, end_mask = var_7168_end_mask_0, x = k_27)[name = string("op_7168")]; int32 var_7170 = const()[name = string("op_7170"), val = int32(-1)]; bool var_7171_interleave_0 = const()[name = string("op_7171_interleave_0"), val = bool(false)]; tensor var_7171 = concat(axis = var_7170, interleave = var_7171_interleave_0, values = (var_7163, var_7168))[name = string("op_7171")]; tensor var_7172 = mul(x = var_7171, y = sin_1_cast_fp16)[name = string("op_7172")]; tensor key_states_53 = add(x = var_7157, y = var_7172)[name = string("key_states_53")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; int32 concat_106_axis_0 = const()[name = string("concat_106_axis_0"), val = int32(0)]; bool concat_106_interleave_0 = const()[name = string("concat_106_interleave_0"), val = bool(false)]; tensor concat_106 = concat(axis = concat_106_axis_0, interleave = concat_106_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_106")]; tensor concat_107_values1_0 = const()[name = string("concat_107_values1_0"), val = tensor([0])]; tensor concat_107_values3_0 = const()[name = string("concat_107_values3_0"), val = tensor([0])]; int32 concat_107_axis_0 = const()[name = string("concat_107_axis_0"), val = int32(0)]; bool concat_107_interleave_0 = const()[name = string("concat_107_interleave_0"), val = bool(false)]; tensor concat_107 = concat(axis = concat_107_axis_0, interleave = concat_107_interleave_0, values = (expand_dims_160, concat_107_values1_0, var_1717, concat_107_values3_0))[name = string("concat_107")]; tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_106, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_107, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = key_states_53, x = coreml_update_state_81)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_138_write_state")]; tensor coreml_update_state_82 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_138")]; tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([41])]; tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([42])]; int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_110")]; tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_166, concat_111_values1_0, var_1717, concat_111_values3_0))[name = string("concat_111")]; tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = var_7089, x = coreml_update_state_82)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_139_write_state")]; tensor coreml_update_state_83 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_139")]; tensor var_7227_begin_0 = const()[name = string("op_7227_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_7227_end_0 = const()[name = string("op_7227_end_0"), val = tensor([14, 8, 1536, 128])]; tensor var_7227_end_mask_0 = const()[name = string("op_7227_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7227_cast_fp16 = slice_by_index(begin = var_7227_begin_0, end = var_7227_end_0, end_mask = var_7227_end_mask_0, x = coreml_update_state_83)[name = string("op_7227_cast_fp16")]; tensor key_cache_27_axes_0 = const()[name = string("key_cache_27_axes_0"), val = tensor([0])]; tensor key_cache_27_cast_fp16 = squeeze(axes = key_cache_27_axes_0, x = var_7227_cast_fp16)[name = string("key_cache_27_cast_fp16")]; tensor var_7234_begin_0 = const()[name = string("op_7234_begin_0"), val = tensor([41, 0, 0, 0])]; tensor var_7234_end_0 = const()[name = string("op_7234_end_0"), val = tensor([42, 8, 1536, 128])]; tensor var_7234_end_mask_0 = const()[name = string("op_7234_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7234_cast_fp16 = slice_by_index(begin = var_7234_begin_0, end = var_7234_end_0, end_mask = var_7234_end_mask_0, x = coreml_update_state_83)[name = string("op_7234_cast_fp16")]; tensor value_cache_27_axes_0 = const()[name = string("value_cache_27_axes_0"), val = tensor([0])]; tensor value_cache_27_cast_fp16 = squeeze(axes = value_cache_27_axes_0, x = var_7234_cast_fp16)[name = string("value_cache_27_cast_fp16")]; tensor var_7258_axes_0 = const()[name = string("op_7258_axes_0"), val = tensor([1])]; tensor var_7258_cast_fp16 = expand_dims(axes = var_7258_axes_0, x = key_cache_27_cast_fp16)[name = string("op_7258_cast_fp16")]; tensor var_7263 = const()[name = string("op_7263"), val = tensor([1, 2, 1, 1])]; tensor value_107_cast_fp16 = tile(reps = var_7263, x = var_7258_cast_fp16)[name = string("value_107_cast_fp16")]; tensor var_7269 = const()[name = string("op_7269"), val = tensor([1, 16, 1536, 128])]; tensor key_states_55_cast_fp16 = reshape(shape = var_7269, x = value_107_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor var_7272_axes_0 = const()[name = string("op_7272_axes_0"), val = tensor([1])]; tensor var_7272_cast_fp16 = expand_dims(axes = var_7272_axes_0, x = value_cache_27_cast_fp16)[name = string("op_7272_cast_fp16")]; tensor var_7277 = const()[name = string("op_7277"), val = tensor([1, 2, 1, 1])]; tensor value_111_cast_fp16 = tile(reps = var_7277, x = var_7272_cast_fp16)[name = string("value_111_cast_fp16")]; tensor var_7283 = const()[name = string("op_7283"), val = tensor([1, 16, 1536, 128])]; tensor value_states_81_cast_fp16 = reshape(shape = var_7283, x = value_111_cast_fp16)[name = string("value_states_81_cast_fp16")]; bool var_7298_transpose_x_1 = const()[name = string("op_7298_transpose_x_1"), val = bool(false)]; bool var_7298_transpose_y_1 = const()[name = string("op_7298_transpose_y_1"), val = bool(true)]; tensor var_7298 = matmul(transpose_x = var_7298_transpose_x_1, transpose_y = var_7298_transpose_y_1, x = query_states_53, y = key_states_55_cast_fp16)[name = string("op_7298")]; fp16 var_7299_to_fp16 = const()[name = string("op_7299_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_53_cast_fp16 = mul(x = var_7298, y = var_7299_to_fp16)[name = string("attention_53_cast_fp16")]; tensor attention_55_cast_fp16 = add(x = attention_53_cast_fp16, y = causal_mask)[name = string("attention_55_cast_fp16")]; int32 var_7308 = const()[name = string("op_7308"), val = int32(-1)]; tensor probabilities_27_cast_fp16 = softmax(axis = var_7308, x = attention_55_cast_fp16)[name = string("probabilities_27_cast_fp16")]; bool output_79_transpose_x_0 = const()[name = string("output_79_transpose_x_0"), val = bool(false)]; bool output_79_transpose_y_0 = const()[name = string("output_79_transpose_y_0"), val = bool(false)]; tensor output_79_cast_fp16 = matmul(transpose_x = output_79_transpose_x_0, transpose_y = output_79_transpose_y_0, x = probabilities_27_cast_fp16, y = value_states_81_cast_fp16)[name = string("output_79_cast_fp16")]; tensor var_7319_perm_0 = const()[name = string("op_7319_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_7325 = const()[name = string("op_7325"), val = tensor([1, 1, 2048])]; tensor var_7319_cast_fp16 = transpose(perm = var_7319_perm_0, x = output_79_cast_fp16)[name = string("transpose_88")]; tensor output_81_cast_fp16 = reshape(shape = var_7325, x = var_7319_cast_fp16)[name = string("output_81_cast_fp16")]; tensor var_7330 = const()[name = string("op_7330"), val = tensor([0, 2, 1])]; string var_7346_pad_type_0 = const()[name = string("op_7346_pad_type_0"), val = string("valid")]; int32 var_7346_groups_0 = const()[name = string("op_7346_groups_0"), val = int32(1)]; tensor var_7346_strides_0 = const()[name = string("op_7346_strides_0"), val = tensor([1])]; tensor var_7346_pad_0 = const()[name = string("op_7346_pad_0"), val = tensor([0, 0])]; tensor var_7346_dilations_0 = const()[name = string("op_7346_dilations_0"), val = tensor([1])]; tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313629952))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315202880))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7331_cast_fp16 = transpose(perm = var_7330, x = output_81_cast_fp16)[name = string("transpose_87")]; tensor var_7346_cast_fp16 = conv(dilations = var_7346_dilations_0, groups = var_7346_groups_0, pad = var_7346_pad_0, pad_type = var_7346_pad_type_0, strides = var_7346_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_7331_cast_fp16)[name = string("op_7346_cast_fp16")]; tensor var_7350 = const()[name = string("op_7350"), val = tensor([0, 2, 1])]; tensor attn_output_27_cast_fp16 = transpose(perm = var_7350, x = var_7346_cast_fp16)[name = string("transpose_86")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_131_cast_fp16, y = attn_output_27_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; int32 var_7365 = const()[name = string("op_7365"), val = int32(-1)]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7367_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_193_promoted_to_fp16)[name = string("op_7367_cast_fp16")]; bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; tensor input_245_cast_fp16 = concat(axis = var_7365, interleave = input_245_interleave_0, values = (hidden_states_139_cast_fp16, var_7367_cast_fp16))[name = string("input_245_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_7362_to_fp16 = const()[name = string("op_7362_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_7362_to_fp16, x = input_245_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor normed_223_begin_0 = const()[name = string("normed_223_begin_0"), val = tensor([0, 0, 0])]; tensor normed_223_end_0 = const()[name = string("normed_223_end_0"), val = tensor([1, 1, 1024])]; tensor normed_223_end_mask_0 = const()[name = string("normed_223_end_mask_0"), val = tensor([true, true, false])]; tensor normed_223_cast_fp16 = slice_by_index(begin = normed_223_begin_0, end = normed_223_end_0, end_mask = normed_223_end_mask_0, x = normed_221_cast_fp16)[name = string("normed_223_cast_fp16")]; tensor const_195_promoted_to_fp16 = const()[name = string("const_195_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315219328)))]; tensor x_53_cast_fp16 = mul(x = normed_223_cast_fp16, y = const_195_promoted_to_fp16)[name = string("x_53_cast_fp16")]; tensor var_7387 = const()[name = string("op_7387"), val = tensor([0, 2, 1])]; tensor input_247_axes_0 = const()[name = string("input_247_axes_0"), val = tensor([2])]; tensor var_7388 = transpose(perm = var_7387, x = x_53_cast_fp16)[name = string("transpose_85")]; tensor input_247 = expand_dims(axes = input_247_axes_0, x = var_7388)[name = string("input_247")]; string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; tensor input_249 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_247)[name = string("input_249")]; string b_27_pad_type_0 = const()[name = string("b_27_pad_type_0"), val = string("valid")]; tensor b_27_strides_0 = const()[name = string("b_27_strides_0"), val = tensor([1, 1])]; tensor b_27_pad_0 = const()[name = string("b_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_27_dilations_0 = const()[name = string("b_27_dilations_0"), val = tensor([1, 1])]; int32 b_27_groups_0 = const()[name = string("b_27_groups_0"), val = int32(1)]; tensor b_27 = conv(dilations = b_27_dilations_0, groups = b_27_groups_0, pad = b_27_pad_0, pad_type = b_27_pad_type_0, strides = b_27_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_247)[name = string("b_27")]; tensor c_27 = silu(x = input_249)[name = string("c_27")]; tensor input_251 = mul(x = c_27, y = b_27)[name = string("input_251")]; string e_27_pad_type_0 = const()[name = string("e_27_pad_type_0"), val = string("valid")]; tensor e_27_strides_0 = const()[name = string("e_27_strides_0"), val = tensor([1, 1])]; tensor e_27_pad_0 = const()[name = string("e_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_27_dilations_0 = const()[name = string("e_27_dilations_0"), val = tensor([1, 1])]; int32 e_27_groups_0 = const()[name = string("e_27_groups_0"), val = int32(1)]; tensor e_27 = conv(dilations = e_27_dilations_0, groups = e_27_groups_0, pad = e_27_pad_0, pad_type = e_27_pad_type_0, strides = e_27_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input_251)[name = string("e_27")]; tensor var_7410_axes_0 = const()[name = string("op_7410_axes_0"), val = tensor([2])]; tensor var_7410 = squeeze(axes = var_7410_axes_0, x = e_27)[name = string("op_7410")]; tensor var_7411 = const()[name = string("op_7411"), val = tensor([0, 2, 1])]; tensor var_7412 = transpose(perm = var_7411, x = var_7410)[name = string("transpose_84")]; tensor hidden_states_141_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = var_7412)[name = string("hidden_states_141_cast_fp16")]; int32 var_7426 = const()[name = string("op_7426"), val = int32(-1)]; fp16 const_196_promoted_to_fp16 = const()[name = string("const_196_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7428_cast_fp16 = mul(x = hidden_states_141_cast_fp16, y = const_196_promoted_to_fp16)[name = string("op_7428_cast_fp16")]; bool input_253_interleave_0 = const()[name = string("input_253_interleave_0"), val = bool(false)]; tensor input_253_cast_fp16 = concat(axis = var_7426, interleave = input_253_interleave_0, values = (hidden_states_141_cast_fp16, var_7428_cast_fp16))[name = string("input_253_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_7423_to_fp16 = const()[name = string("op_7423_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_7423_to_fp16, x = input_253_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor normed_227_begin_0 = const()[name = string("normed_227_begin_0"), val = tensor([0, 0, 0])]; tensor normed_227_end_0 = const()[name = string("normed_227_end_0"), val = tensor([1, 1, 1024])]; tensor normed_227_end_mask_0 = const()[name = string("normed_227_end_mask_0"), val = tensor([true, true, false])]; tensor normed_227_cast_fp16 = slice_by_index(begin = normed_227_begin_0, end = normed_227_end_0, end_mask = normed_227_end_mask_0, x = normed_225_cast_fp16)[name = string("normed_227_cast_fp16")]; tensor const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315221440)))]; tensor hidden_states_143_cast_fp16 = mul(x = normed_227_cast_fp16, y = const_198_promoted_to_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor var_7440 = const()[name = string("op_7440"), val = tensor([0, 2, 1])]; tensor var_7443_axes_0 = const()[name = string("op_7443_axes_0"), val = tensor([2])]; tensor var_7441_cast_fp16 = transpose(perm = var_7440, x = hidden_states_143_cast_fp16)[name = string("transpose_83")]; tensor var_7443_cast_fp16 = expand_dims(axes = var_7443_axes_0, x = var_7441_cast_fp16)[name = string("op_7443_cast_fp16")]; string var_7459_pad_type_0 = const()[name = string("op_7459_pad_type_0"), val = string("valid")]; tensor var_7459_strides_0 = const()[name = string("op_7459_strides_0"), val = tensor([1, 1])]; tensor var_7459_pad_0 = const()[name = string("op_7459_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7459_dilations_0 = const()[name = string("op_7459_dilations_0"), val = tensor([1, 1])]; int32 var_7459_groups_0 = const()[name = string("op_7459_groups_0"), val = int32(1)]; tensor var_7459 = conv(dilations = var_7459_dilations_0, groups = var_7459_groups_0, pad = var_7459_pad_0, pad_type = var_7459_pad_type_0, strides = var_7459_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_7443_cast_fp16)[name = string("op_7459")]; tensor var_7464 = const()[name = string("op_7464"), val = tensor([1, 16, 1, 128])]; tensor var_7465 = reshape(shape = var_7464, x = var_7459)[name = string("op_7465")]; string var_7481_pad_type_0 = const()[name = string("op_7481_pad_type_0"), val = string("valid")]; tensor var_7481_strides_0 = const()[name = string("op_7481_strides_0"), val = tensor([1, 1])]; tensor var_7481_pad_0 = const()[name = string("op_7481_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7481_dilations_0 = const()[name = string("op_7481_dilations_0"), val = tensor([1, 1])]; int32 var_7481_groups_0 = const()[name = string("op_7481_groups_0"), val = int32(1)]; tensor var_7481 = conv(dilations = var_7481_dilations_0, groups = var_7481_groups_0, pad = var_7481_pad_0, pad_type = var_7481_pad_type_0, strides = var_7481_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_7443_cast_fp16)[name = string("op_7481")]; tensor var_7486 = const()[name = string("op_7486"), val = tensor([1, 8, 1, 128])]; tensor var_7487 = reshape(shape = var_7486, x = var_7481)[name = string("op_7487")]; string var_7503_pad_type_0 = const()[name = string("op_7503_pad_type_0"), val = string("valid")]; tensor var_7503_strides_0 = const()[name = string("op_7503_strides_0"), val = tensor([1, 1])]; tensor var_7503_pad_0 = const()[name = string("op_7503_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7503_dilations_0 = const()[name = string("op_7503_dilations_0"), val = tensor([1, 1])]; int32 var_7503_groups_0 = const()[name = string("op_7503_groups_0"), val = int32(1)]; tensor var_7503 = conv(dilations = var_7503_dilations_0, groups = var_7503_groups_0, pad = var_7503_pad_0, pad_type = var_7503_pad_type_0, strides = var_7503_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_7443_cast_fp16)[name = string("op_7503")]; tensor var_7508 = const()[name = string("op_7508"), val = tensor([1, 8, 1, 128])]; tensor var_7509 = reshape(shape = var_7508, x = var_7503)[name = string("op_7509")]; int32 var_7526 = const()[name = string("op_7526"), val = int32(-1)]; fp16 const_199_promoted = const()[name = string("const_199_promoted"), val = fp16(-0x1p+0)]; tensor var_7528 = mul(x = var_7465, y = const_199_promoted)[name = string("op_7528")]; bool input_257_interleave_0 = const()[name = string("input_257_interleave_0"), val = bool(false)]; tensor input_257 = concat(axis = var_7526, interleave = input_257_interleave_0, values = (var_7465, var_7528))[name = string("input_257")]; tensor normed_229_axes_0 = const()[name = string("normed_229_axes_0"), val = tensor([-1])]; fp16 var_7523_to_fp16 = const()[name = string("op_7523_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_229_cast_fp16 = layer_norm(axes = normed_229_axes_0, epsilon = var_7523_to_fp16, x = input_257)[name = string("normed_229_cast_fp16")]; tensor normed_231_begin_0 = const()[name = string("normed_231_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_231_end_0 = const()[name = string("normed_231_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_231_end_mask_0 = const()[name = string("normed_231_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_231 = slice_by_index(begin = normed_231_begin_0, end = normed_231_end_0, end_mask = normed_231_end_mask_0, x = normed_229_cast_fp16)[name = string("normed_231")]; tensor const_201 = const()[name = string("const_201"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315223552)))]; tensor q_29 = mul(x = normed_231, y = const_201)[name = string("q_29")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(-1)]; fp16 const_202_promoted = const()[name = string("const_202_promoted"), val = fp16(-0x1p+0)]; tensor var_7550 = mul(x = var_7487, y = const_202_promoted)[name = string("op_7550")]; bool input_259_interleave_0 = const()[name = string("input_259_interleave_0"), val = bool(false)]; tensor input_259 = concat(axis = var_7548, interleave = input_259_interleave_0, values = (var_7487, var_7550))[name = string("input_259")]; tensor normed_233_axes_0 = const()[name = string("normed_233_axes_0"), val = tensor([-1])]; fp16 var_7545_to_fp16 = const()[name = string("op_7545_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_233_cast_fp16 = layer_norm(axes = normed_233_axes_0, epsilon = var_7545_to_fp16, x = input_259)[name = string("normed_233_cast_fp16")]; tensor normed_235_begin_0 = const()[name = string("normed_235_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_235_end_0 = const()[name = string("normed_235_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_235_end_mask_0 = const()[name = string("normed_235_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_235 = slice_by_index(begin = normed_235_begin_0, end = normed_235_end_0, end_mask = normed_235_end_mask_0, x = normed_233_cast_fp16)[name = string("normed_235")]; tensor const_204 = const()[name = string("const_204"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315223872)))]; tensor k_29 = mul(x = normed_235, y = const_204)[name = string("k_29")]; tensor var_7559 = mul(x = q_29, y = cos_1_cast_fp16)[name = string("op_7559")]; tensor var_7564_begin_0 = const()[name = string("op_7564_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7564_end_0 = const()[name = string("op_7564_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_7564_end_mask_0 = const()[name = string("op_7564_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7564 = slice_by_index(begin = var_7564_begin_0, end = var_7564_end_0, end_mask = var_7564_end_mask_0, x = q_29)[name = string("op_7564")]; fp16 const_205_promoted = const()[name = string("const_205_promoted"), val = fp16(-0x1p+0)]; tensor var_7565 = mul(x = var_7564, y = const_205_promoted)[name = string("op_7565")]; tensor var_7570_begin_0 = const()[name = string("op_7570_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7570_end_0 = const()[name = string("op_7570_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_7570_end_mask_0 = const()[name = string("op_7570_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7570 = slice_by_index(begin = var_7570_begin_0, end = var_7570_end_0, end_mask = var_7570_end_mask_0, x = q_29)[name = string("op_7570")]; int32 var_7572 = const()[name = string("op_7572"), val = int32(-1)]; bool var_7573_interleave_0 = const()[name = string("op_7573_interleave_0"), val = bool(false)]; tensor var_7573 = concat(axis = var_7572, interleave = var_7573_interleave_0, values = (var_7565, var_7570))[name = string("op_7573")]; tensor var_7574 = mul(x = var_7573, y = sin_1_cast_fp16)[name = string("op_7574")]; tensor query_states_57 = add(x = var_7559, y = var_7574)[name = string("query_states_57")]; tensor var_7577 = mul(x = k_29, y = cos_1_cast_fp16)[name = string("op_7577")]; tensor var_7582_begin_0 = const()[name = string("op_7582_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7582_end_0 = const()[name = string("op_7582_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_7582_end_mask_0 = const()[name = string("op_7582_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7582 = slice_by_index(begin = var_7582_begin_0, end = var_7582_end_0, end_mask = var_7582_end_mask_0, x = k_29)[name = string("op_7582")]; fp16 const_206_promoted = const()[name = string("const_206_promoted"), val = fp16(-0x1p+0)]; tensor var_7583 = mul(x = var_7582, y = const_206_promoted)[name = string("op_7583")]; tensor var_7588_begin_0 = const()[name = string("op_7588_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7588_end_0 = const()[name = string("op_7588_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_7588_end_mask_0 = const()[name = string("op_7588_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7588 = slice_by_index(begin = var_7588_begin_0, end = var_7588_end_0, end_mask = var_7588_end_mask_0, x = k_29)[name = string("op_7588")]; int32 var_7590 = const()[name = string("op_7590"), val = int32(-1)]; bool var_7591_interleave_0 = const()[name = string("op_7591_interleave_0"), val = bool(false)]; tensor var_7591 = concat(axis = var_7590, interleave = var_7591_interleave_0, values = (var_7583, var_7588))[name = string("op_7591")]; tensor var_7592 = mul(x = var_7591, y = sin_1_cast_fp16)[name = string("op_7592")]; tensor key_states_57 = add(x = var_7577, y = var_7592)[name = string("key_states_57")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_168, expand_dims_169, current_pos, expand_dims_171))[name = string("concat_114")]; tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_172, concat_115_values1_0, var_1717, concat_115_values3_0))[name = string("concat_115")]; tensor model_model_kv_cache_0_internal_tensor_assign_29_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_29_stride_0, update = key_states_57, x = coreml_update_state_83)[name = string("model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_140_write_state")]; tensor coreml_update_state_84 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_140")]; tensor expand_dims_174 = const()[name = string("expand_dims_174"), val = tensor([42])]; tensor expand_dims_175 = const()[name = string("expand_dims_175"), val = tensor([0])]; tensor expand_dims_177 = const()[name = string("expand_dims_177"), val = tensor([0])]; tensor expand_dims_178 = const()[name = string("expand_dims_178"), val = tensor([43])]; int32 concat_118_axis_0 = const()[name = string("concat_118_axis_0"), val = int32(0)]; bool concat_118_interleave_0 = const()[name = string("concat_118_interleave_0"), val = bool(false)]; tensor concat_118 = concat(axis = concat_118_axis_0, interleave = concat_118_interleave_0, values = (expand_dims_174, expand_dims_175, current_pos, expand_dims_177))[name = string("concat_118")]; tensor concat_119_values1_0 = const()[name = string("concat_119_values1_0"), val = tensor([0])]; tensor concat_119_values3_0 = const()[name = string("concat_119_values3_0"), val = tensor([0])]; int32 concat_119_axis_0 = const()[name = string("concat_119_axis_0"), val = int32(0)]; bool concat_119_interleave_0 = const()[name = string("concat_119_interleave_0"), val = bool(false)]; tensor concat_119 = concat(axis = concat_119_axis_0, interleave = concat_119_interleave_0, values = (expand_dims_178, concat_119_values1_0, var_1717, concat_119_values3_0))[name = string("concat_119")]; tensor model_model_kv_cache_0_internal_tensor_assign_30_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16 = slice_update(begin = concat_118, begin_mask = model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0, end = concat_119, end_mask = model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_30_stride_0, update = var_7509, x = coreml_update_state_84)[name = string("model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_141_write_state")]; tensor coreml_update_state_85 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_141")]; tensor var_7647_begin_0 = const()[name = string("op_7647_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_7647_end_0 = const()[name = string("op_7647_end_0"), val = tensor([15, 8, 1536, 128])]; tensor var_7647_end_mask_0 = const()[name = string("op_7647_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7647_cast_fp16 = slice_by_index(begin = var_7647_begin_0, end = var_7647_end_0, end_mask = var_7647_end_mask_0, x = coreml_update_state_85)[name = string("op_7647_cast_fp16")]; tensor key_cache_29_axes_0 = const()[name = string("key_cache_29_axes_0"), val = tensor([0])]; tensor key_cache_29_cast_fp16 = squeeze(axes = key_cache_29_axes_0, x = var_7647_cast_fp16)[name = string("key_cache_29_cast_fp16")]; tensor var_7654_begin_0 = const()[name = string("op_7654_begin_0"), val = tensor([42, 0, 0, 0])]; tensor var_7654_end_0 = const()[name = string("op_7654_end_0"), val = tensor([43, 8, 1536, 128])]; tensor var_7654_end_mask_0 = const()[name = string("op_7654_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7654_cast_fp16 = slice_by_index(begin = var_7654_begin_0, end = var_7654_end_0, end_mask = var_7654_end_mask_0, x = coreml_update_state_85)[name = string("op_7654_cast_fp16")]; tensor value_cache_29_axes_0 = const()[name = string("value_cache_29_axes_0"), val = tensor([0])]; tensor value_cache_29_cast_fp16 = squeeze(axes = value_cache_29_axes_0, x = var_7654_cast_fp16)[name = string("value_cache_29_cast_fp16")]; tensor var_7678_axes_0 = const()[name = string("op_7678_axes_0"), val = tensor([1])]; tensor var_7678_cast_fp16 = expand_dims(axes = var_7678_axes_0, x = key_cache_29_cast_fp16)[name = string("op_7678_cast_fp16")]; tensor var_7683 = const()[name = string("op_7683"), val = tensor([1, 2, 1, 1])]; tensor value_115_cast_fp16 = tile(reps = var_7683, x = var_7678_cast_fp16)[name = string("value_115_cast_fp16")]; tensor var_7689 = const()[name = string("op_7689"), val = tensor([1, 16, 1536, 128])]; tensor key_states_59_cast_fp16 = reshape(shape = var_7689, x = value_115_cast_fp16)[name = string("key_states_59_cast_fp16")]; tensor var_7692_axes_0 = const()[name = string("op_7692_axes_0"), val = tensor([1])]; tensor var_7692_cast_fp16 = expand_dims(axes = var_7692_axes_0, x = value_cache_29_cast_fp16)[name = string("op_7692_cast_fp16")]; tensor var_7697 = const()[name = string("op_7697"), val = tensor([1, 2, 1, 1])]; tensor value_119_cast_fp16 = tile(reps = var_7697, x = var_7692_cast_fp16)[name = string("value_119_cast_fp16")]; tensor var_7703 = const()[name = string("op_7703"), val = tensor([1, 16, 1536, 128])]; tensor value_states_87_cast_fp16 = reshape(shape = var_7703, x = value_119_cast_fp16)[name = string("value_states_87_cast_fp16")]; bool var_7718_transpose_x_1 = const()[name = string("op_7718_transpose_x_1"), val = bool(false)]; bool var_7718_transpose_y_1 = const()[name = string("op_7718_transpose_y_1"), val = bool(true)]; tensor var_7718 = matmul(transpose_x = var_7718_transpose_x_1, transpose_y = var_7718_transpose_y_1, x = query_states_57, y = key_states_59_cast_fp16)[name = string("op_7718")]; fp16 var_7719_to_fp16 = const()[name = string("op_7719_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_57_cast_fp16 = mul(x = var_7718, y = var_7719_to_fp16)[name = string("attention_57_cast_fp16")]; tensor attention_59_cast_fp16 = add(x = attention_57_cast_fp16, y = causal_mask)[name = string("attention_59_cast_fp16")]; int32 var_7728 = const()[name = string("op_7728"), val = int32(-1)]; tensor probabilities_29_cast_fp16 = softmax(axis = var_7728, x = attention_59_cast_fp16)[name = string("probabilities_29_cast_fp16")]; bool output_85_transpose_x_0 = const()[name = string("output_85_transpose_x_0"), val = bool(false)]; bool output_85_transpose_y_0 = const()[name = string("output_85_transpose_y_0"), val = bool(false)]; tensor output_85_cast_fp16 = matmul(transpose_x = output_85_transpose_x_0, transpose_y = output_85_transpose_y_0, x = probabilities_29_cast_fp16, y = value_states_87_cast_fp16)[name = string("output_85_cast_fp16")]; tensor var_7739_perm_0 = const()[name = string("op_7739_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_7745 = const()[name = string("op_7745"), val = tensor([1, 1, 2048])]; tensor var_7739_cast_fp16 = transpose(perm = var_7739_perm_0, x = output_85_cast_fp16)[name = string("transpose_82")]; tensor output_87_cast_fp16 = reshape(shape = var_7745, x = var_7739_cast_fp16)[name = string("output_87_cast_fp16")]; tensor var_7750 = const()[name = string("op_7750"), val = tensor([0, 2, 1])]; string var_7766_pad_type_0 = const()[name = string("op_7766_pad_type_0"), val = string("valid")]; int32 var_7766_groups_0 = const()[name = string("op_7766_groups_0"), val = int32(1)]; tensor var_7766_strides_0 = const()[name = string("op_7766_strides_0"), val = tensor([1])]; tensor var_7766_pad_0 = const()[name = string("op_7766_pad_0"), val = tensor([0, 0])]; tensor var_7766_dilations_0 = const()[name = string("op_7766_dilations_0"), val = tensor([1])]; tensor squeeze_14_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315224192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316797120))))[name = string("squeeze_14_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7751_cast_fp16 = transpose(perm = var_7750, x = output_87_cast_fp16)[name = string("transpose_81")]; tensor var_7766_cast_fp16 = conv(dilations = var_7766_dilations_0, groups = var_7766_groups_0, pad = var_7766_pad_0, pad_type = var_7766_pad_type_0, strides = var_7766_strides_0, weight = squeeze_14_cast_fp16_to_fp32_to_fp16_palettized, x = var_7751_cast_fp16)[name = string("op_7766_cast_fp16")]; tensor var_7770 = const()[name = string("op_7770"), val = tensor([0, 2, 1])]; tensor attn_output_29_cast_fp16 = transpose(perm = var_7770, x = var_7766_cast_fp16)[name = string("transpose_80")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_141_cast_fp16, y = attn_output_29_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; int32 var_7785 = const()[name = string("op_7785"), val = int32(-1)]; fp16 const_207_promoted_to_fp16 = const()[name = string("const_207_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7787_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_207_promoted_to_fp16)[name = string("op_7787_cast_fp16")]; bool input_263_interleave_0 = const()[name = string("input_263_interleave_0"), val = bool(false)]; tensor input_263_cast_fp16 = concat(axis = var_7785, interleave = input_263_interleave_0, values = (hidden_states_149_cast_fp16, var_7787_cast_fp16))[name = string("input_263_cast_fp16")]; tensor normed_237_axes_0 = const()[name = string("normed_237_axes_0"), val = tensor([-1])]; fp16 var_7782_to_fp16 = const()[name = string("op_7782_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_237_cast_fp16 = layer_norm(axes = normed_237_axes_0, epsilon = var_7782_to_fp16, x = input_263_cast_fp16)[name = string("normed_237_cast_fp16")]; tensor normed_239_begin_0 = const()[name = string("normed_239_begin_0"), val = tensor([0, 0, 0])]; tensor normed_239_end_0 = const()[name = string("normed_239_end_0"), val = tensor([1, 1, 1024])]; tensor normed_239_end_mask_0 = const()[name = string("normed_239_end_mask_0"), val = tensor([true, true, false])]; tensor normed_239_cast_fp16 = slice_by_index(begin = normed_239_begin_0, end = normed_239_end_0, end_mask = normed_239_end_mask_0, x = normed_237_cast_fp16)[name = string("normed_239_cast_fp16")]; tensor const_209_promoted_to_fp16 = const()[name = string("const_209_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316813568)))]; tensor x_57_cast_fp16 = mul(x = normed_239_cast_fp16, y = const_209_promoted_to_fp16)[name = string("x_57_cast_fp16")]; tensor var_7807 = const()[name = string("op_7807"), val = tensor([0, 2, 1])]; tensor input_265_axes_0 = const()[name = string("input_265_axes_0"), val = tensor([2])]; tensor var_7808 = transpose(perm = var_7807, x = x_57_cast_fp16)[name = string("transpose_79")]; tensor input_265 = expand_dims(axes = input_265_axes_0, x = var_7808)[name = string("input_265")]; string input_267_pad_type_0 = const()[name = string("input_267_pad_type_0"), val = string("valid")]; tensor input_267_strides_0 = const()[name = string("input_267_strides_0"), val = tensor([1, 1])]; tensor input_267_pad_0 = const()[name = string("input_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_267_dilations_0 = const()[name = string("input_267_dilations_0"), val = tensor([1, 1])]; int32 input_267_groups_0 = const()[name = string("input_267_groups_0"), val = int32(1)]; tensor input_267 = conv(dilations = input_267_dilations_0, groups = input_267_groups_0, pad = input_267_pad_0, pad_type = input_267_pad_type_0, strides = input_267_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_265)[name = string("input_267")]; string b_29_pad_type_0 = const()[name = string("b_29_pad_type_0"), val = string("valid")]; tensor b_29_strides_0 = const()[name = string("b_29_strides_0"), val = tensor([1, 1])]; tensor b_29_pad_0 = const()[name = string("b_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_29_dilations_0 = const()[name = string("b_29_dilations_0"), val = tensor([1, 1])]; int32 b_29_groups_0 = const()[name = string("b_29_groups_0"), val = int32(1)]; tensor b_29 = conv(dilations = b_29_dilations_0, groups = b_29_groups_0, pad = b_29_pad_0, pad_type = b_29_pad_type_0, strides = b_29_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_265)[name = string("b_29")]; tensor c_29 = silu(x = input_267)[name = string("c_29")]; tensor input_269 = mul(x = c_29, y = b_29)[name = string("input_269")]; string e_29_pad_type_0 = const()[name = string("e_29_pad_type_0"), val = string("valid")]; tensor e_29_strides_0 = const()[name = string("e_29_strides_0"), val = tensor([1, 1])]; tensor e_29_pad_0 = const()[name = string("e_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_29_dilations_0 = const()[name = string("e_29_dilations_0"), val = tensor([1, 1])]; int32 e_29_groups_0 = const()[name = string("e_29_groups_0"), val = int32(1)]; tensor e_29 = conv(dilations = e_29_dilations_0, groups = e_29_groups_0, pad = e_29_pad_0, pad_type = e_29_pad_type_0, strides = e_29_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_269)[name = string("e_29")]; tensor var_7830_axes_0 = const()[name = string("op_7830_axes_0"), val = tensor([2])]; tensor var_7830 = squeeze(axes = var_7830_axes_0, x = e_29)[name = string("op_7830")]; tensor var_7831 = const()[name = string("op_7831"), val = tensor([0, 2, 1])]; tensor var_7832 = transpose(perm = var_7831, x = var_7830)[name = string("transpose_78")]; tensor hidden_states_151_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = var_7832)[name = string("hidden_states_151_cast_fp16")]; int32 var_7846 = const()[name = string("op_7846"), val = int32(-1)]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7848_cast_fp16 = mul(x = hidden_states_151_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7848_cast_fp16")]; bool input_271_interleave_0 = const()[name = string("input_271_interleave_0"), val = bool(false)]; tensor input_271_cast_fp16 = concat(axis = var_7846, interleave = input_271_interleave_0, values = (hidden_states_151_cast_fp16, var_7848_cast_fp16))[name = string("input_271_cast_fp16")]; tensor normed_241_axes_0 = const()[name = string("normed_241_axes_0"), val = tensor([-1])]; fp16 var_7843_to_fp16 = const()[name = string("op_7843_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_241_cast_fp16 = layer_norm(axes = normed_241_axes_0, epsilon = var_7843_to_fp16, x = input_271_cast_fp16)[name = string("normed_241_cast_fp16")]; tensor normed_243_begin_0 = const()[name = string("normed_243_begin_0"), val = tensor([0, 0, 0])]; tensor normed_243_end_0 = const()[name = string("normed_243_end_0"), val = tensor([1, 1, 1024])]; tensor normed_243_end_mask_0 = const()[name = string("normed_243_end_mask_0"), val = tensor([true, true, false])]; tensor normed_243_cast_fp16 = slice_by_index(begin = normed_243_begin_0, end = normed_243_end_0, end_mask = normed_243_end_mask_0, x = normed_241_cast_fp16)[name = string("normed_243_cast_fp16")]; tensor const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316815680)))]; tensor hidden_states_153_cast_fp16 = mul(x = normed_243_cast_fp16, y = const_212_promoted_to_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor var_7860 = const()[name = string("op_7860"), val = tensor([0, 2, 1])]; tensor var_7863_axes_0 = const()[name = string("op_7863_axes_0"), val = tensor([2])]; tensor var_7861_cast_fp16 = transpose(perm = var_7860, x = hidden_states_153_cast_fp16)[name = string("transpose_77")]; tensor var_7863_cast_fp16 = expand_dims(axes = var_7863_axes_0, x = var_7861_cast_fp16)[name = string("op_7863_cast_fp16")]; string var_7879_pad_type_0 = const()[name = string("op_7879_pad_type_0"), val = string("valid")]; tensor var_7879_strides_0 = const()[name = string("op_7879_strides_0"), val = tensor([1, 1])]; tensor var_7879_pad_0 = const()[name = string("op_7879_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7879_dilations_0 = const()[name = string("op_7879_dilations_0"), val = tensor([1, 1])]; int32 var_7879_groups_0 = const()[name = string("op_7879_groups_0"), val = int32(1)]; tensor var_7879 = conv(dilations = var_7879_dilations_0, groups = var_7879_groups_0, pad = var_7879_pad_0, pad_type = var_7879_pad_type_0, strides = var_7879_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_7863_cast_fp16)[name = string("op_7879")]; tensor var_7884 = const()[name = string("op_7884"), val = tensor([1, 16, 1, 128])]; tensor var_7885 = reshape(shape = var_7884, x = var_7879)[name = string("op_7885")]; string var_7901_pad_type_0 = const()[name = string("op_7901_pad_type_0"), val = string("valid")]; tensor var_7901_strides_0 = const()[name = string("op_7901_strides_0"), val = tensor([1, 1])]; tensor var_7901_pad_0 = const()[name = string("op_7901_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7901_dilations_0 = const()[name = string("op_7901_dilations_0"), val = tensor([1, 1])]; int32 var_7901_groups_0 = const()[name = string("op_7901_groups_0"), val = int32(1)]; tensor var_7901 = conv(dilations = var_7901_dilations_0, groups = var_7901_groups_0, pad = var_7901_pad_0, pad_type = var_7901_pad_type_0, strides = var_7901_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_7863_cast_fp16)[name = string("op_7901")]; tensor var_7906 = const()[name = string("op_7906"), val = tensor([1, 8, 1, 128])]; tensor var_7907 = reshape(shape = var_7906, x = var_7901)[name = string("op_7907")]; string var_7923_pad_type_0 = const()[name = string("op_7923_pad_type_0"), val = string("valid")]; tensor var_7923_strides_0 = const()[name = string("op_7923_strides_0"), val = tensor([1, 1])]; tensor var_7923_pad_0 = const()[name = string("op_7923_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7923_dilations_0 = const()[name = string("op_7923_dilations_0"), val = tensor([1, 1])]; int32 var_7923_groups_0 = const()[name = string("op_7923_groups_0"), val = int32(1)]; tensor var_7923 = conv(dilations = var_7923_dilations_0, groups = var_7923_groups_0, pad = var_7923_pad_0, pad_type = var_7923_pad_type_0, strides = var_7923_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_7863_cast_fp16)[name = string("op_7923")]; tensor var_7928 = const()[name = string("op_7928"), val = tensor([1, 8, 1, 128])]; tensor var_7929 = reshape(shape = var_7928, x = var_7923)[name = string("op_7929")]; int32 var_7946 = const()[name = string("op_7946"), val = int32(-1)]; fp16 const_213_promoted = const()[name = string("const_213_promoted"), val = fp16(-0x1p+0)]; tensor var_7948 = mul(x = var_7885, y = const_213_promoted)[name = string("op_7948")]; bool input_275_interleave_0 = const()[name = string("input_275_interleave_0"), val = bool(false)]; tensor input_275 = concat(axis = var_7946, interleave = input_275_interleave_0, values = (var_7885, var_7948))[name = string("input_275")]; tensor normed_245_axes_0 = const()[name = string("normed_245_axes_0"), val = tensor([-1])]; fp16 var_7943_to_fp16 = const()[name = string("op_7943_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_245_cast_fp16 = layer_norm(axes = normed_245_axes_0, epsilon = var_7943_to_fp16, x = input_275)[name = string("normed_245_cast_fp16")]; tensor normed_247_begin_0 = const()[name = string("normed_247_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_247_end_0 = const()[name = string("normed_247_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_247_end_mask_0 = const()[name = string("normed_247_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_247 = slice_by_index(begin = normed_247_begin_0, end = normed_247_end_0, end_mask = normed_247_end_mask_0, x = normed_245_cast_fp16)[name = string("normed_247")]; tensor const_215 = const()[name = string("const_215"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316817792)))]; tensor q_31 = mul(x = normed_247, y = const_215)[name = string("q_31")]; int32 var_7968 = const()[name = string("op_7968"), val = int32(-1)]; fp16 const_216_promoted = const()[name = string("const_216_promoted"), val = fp16(-0x1p+0)]; tensor var_7970 = mul(x = var_7907, y = const_216_promoted)[name = string("op_7970")]; bool input_277_interleave_0 = const()[name = string("input_277_interleave_0"), val = bool(false)]; tensor input_277 = concat(axis = var_7968, interleave = input_277_interleave_0, values = (var_7907, var_7970))[name = string("input_277")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; fp16 var_7965_to_fp16 = const()[name = string("op_7965_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_7965_to_fp16, x = input_277)[name = string("normed_249_cast_fp16")]; tensor normed_251_begin_0 = const()[name = string("normed_251_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_251_end_0 = const()[name = string("normed_251_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_251_end_mask_0 = const()[name = string("normed_251_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_251 = slice_by_index(begin = normed_251_begin_0, end = normed_251_end_0, end_mask = normed_251_end_mask_0, x = normed_249_cast_fp16)[name = string("normed_251")]; tensor const_218 = const()[name = string("const_218"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316818112)))]; tensor k_31 = mul(x = normed_251, y = const_218)[name = string("k_31")]; tensor var_7979 = mul(x = q_31, y = cos_1_cast_fp16)[name = string("op_7979")]; tensor var_7984_begin_0 = const()[name = string("op_7984_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7984_end_0 = const()[name = string("op_7984_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_7984_end_mask_0 = const()[name = string("op_7984_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7984 = slice_by_index(begin = var_7984_begin_0, end = var_7984_end_0, end_mask = var_7984_end_mask_0, x = q_31)[name = string("op_7984")]; fp16 const_219_promoted = const()[name = string("const_219_promoted"), val = fp16(-0x1p+0)]; tensor var_7985 = mul(x = var_7984, y = const_219_promoted)[name = string("op_7985")]; tensor var_7990_begin_0 = const()[name = string("op_7990_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7990_end_0 = const()[name = string("op_7990_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_7990_end_mask_0 = const()[name = string("op_7990_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7990 = slice_by_index(begin = var_7990_begin_0, end = var_7990_end_0, end_mask = var_7990_end_mask_0, x = q_31)[name = string("op_7990")]; int32 var_7992 = const()[name = string("op_7992"), val = int32(-1)]; bool var_7993_interleave_0 = const()[name = string("op_7993_interleave_0"), val = bool(false)]; tensor var_7993 = concat(axis = var_7992, interleave = var_7993_interleave_0, values = (var_7985, var_7990))[name = string("op_7993")]; tensor var_7994 = mul(x = var_7993, y = sin_1_cast_fp16)[name = string("op_7994")]; tensor query_states_61 = add(x = var_7979, y = var_7994)[name = string("query_states_61")]; tensor var_7997 = mul(x = k_31, y = cos_1_cast_fp16)[name = string("op_7997")]; tensor var_8002_begin_0 = const()[name = string("op_8002_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8002_end_0 = const()[name = string("op_8002_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_8002_end_mask_0 = const()[name = string("op_8002_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8002 = slice_by_index(begin = var_8002_begin_0, end = var_8002_end_0, end_mask = var_8002_end_mask_0, x = k_31)[name = string("op_8002")]; fp16 const_220_promoted = const()[name = string("const_220_promoted"), val = fp16(-0x1p+0)]; tensor var_8003 = mul(x = var_8002, y = const_220_promoted)[name = string("op_8003")]; tensor var_8008_begin_0 = const()[name = string("op_8008_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8008_end_0 = const()[name = string("op_8008_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_8008_end_mask_0 = const()[name = string("op_8008_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8008 = slice_by_index(begin = var_8008_begin_0, end = var_8008_end_0, end_mask = var_8008_end_mask_0, x = k_31)[name = string("op_8008")]; int32 var_8010 = const()[name = string("op_8010"), val = int32(-1)]; bool var_8011_interleave_0 = const()[name = string("op_8011_interleave_0"), val = bool(false)]; tensor var_8011 = concat(axis = var_8010, interleave = var_8011_interleave_0, values = (var_8003, var_8008))[name = string("op_8011")]; tensor var_8012 = mul(x = var_8011, y = sin_1_cast_fp16)[name = string("op_8012")]; tensor key_states_61 = add(x = var_7997, y = var_8012)[name = string("key_states_61")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; int32 concat_122_axis_0 = const()[name = string("concat_122_axis_0"), val = int32(0)]; bool concat_122_interleave_0 = const()[name = string("concat_122_interleave_0"), val = bool(false)]; tensor concat_122 = concat(axis = concat_122_axis_0, interleave = concat_122_interleave_0, values = (expand_dims_180, expand_dims_181, current_pos, expand_dims_183))[name = string("concat_122")]; tensor concat_123_values1_0 = const()[name = string("concat_123_values1_0"), val = tensor([0])]; tensor concat_123_values3_0 = const()[name = string("concat_123_values3_0"), val = tensor([0])]; int32 concat_123_axis_0 = const()[name = string("concat_123_axis_0"), val = int32(0)]; bool concat_123_interleave_0 = const()[name = string("concat_123_interleave_0"), val = bool(false)]; tensor concat_123 = concat(axis = concat_123_axis_0, interleave = concat_123_interleave_0, values = (expand_dims_184, concat_123_values1_0, var_1717, concat_123_values3_0))[name = string("concat_123")]; tensor model_model_kv_cache_0_internal_tensor_assign_31_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16 = slice_update(begin = concat_122, begin_mask = model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0, end = concat_123, end_mask = model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_31_stride_0, update = key_states_61, x = coreml_update_state_85)[name = string("model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_142_write_state")]; tensor coreml_update_state_86 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_142")]; tensor expand_dims_186 = const()[name = string("expand_dims_186"), val = tensor([43])]; tensor expand_dims_187 = const()[name = string("expand_dims_187"), val = tensor([0])]; tensor expand_dims_189 = const()[name = string("expand_dims_189"), val = tensor([0])]; tensor expand_dims_190 = const()[name = string("expand_dims_190"), val = tensor([44])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_186, expand_dims_187, current_pos, expand_dims_189))[name = string("concat_126")]; tensor concat_127_values1_0 = const()[name = string("concat_127_values1_0"), val = tensor([0])]; tensor concat_127_values3_0 = const()[name = string("concat_127_values3_0"), val = tensor([0])]; int32 concat_127_axis_0 = const()[name = string("concat_127_axis_0"), val = int32(0)]; bool concat_127_interleave_0 = const()[name = string("concat_127_interleave_0"), val = bool(false)]; tensor concat_127 = concat(axis = concat_127_axis_0, interleave = concat_127_interleave_0, values = (expand_dims_190, concat_127_values1_0, var_1717, concat_127_values3_0))[name = string("concat_127")]; tensor model_model_kv_cache_0_internal_tensor_assign_32_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16 = slice_update(begin = concat_126, begin_mask = model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0, end = concat_127, end_mask = model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_32_stride_0, update = var_7929, x = coreml_update_state_86)[name = string("model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_143_write_state")]; tensor coreml_update_state_87 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_143")]; tensor var_8067_begin_0 = const()[name = string("op_8067_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_8067_end_0 = const()[name = string("op_8067_end_0"), val = tensor([16, 8, 1536, 128])]; tensor var_8067_end_mask_0 = const()[name = string("op_8067_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8067_cast_fp16 = slice_by_index(begin = var_8067_begin_0, end = var_8067_end_0, end_mask = var_8067_end_mask_0, x = coreml_update_state_87)[name = string("op_8067_cast_fp16")]; tensor key_cache_31_axes_0 = const()[name = string("key_cache_31_axes_0"), val = tensor([0])]; tensor key_cache_31_cast_fp16 = squeeze(axes = key_cache_31_axes_0, x = var_8067_cast_fp16)[name = string("key_cache_31_cast_fp16")]; tensor var_8074_begin_0 = const()[name = string("op_8074_begin_0"), val = tensor([43, 0, 0, 0])]; tensor var_8074_end_0 = const()[name = string("op_8074_end_0"), val = tensor([44, 8, 1536, 128])]; tensor var_8074_end_mask_0 = const()[name = string("op_8074_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8074_cast_fp16 = slice_by_index(begin = var_8074_begin_0, end = var_8074_end_0, end_mask = var_8074_end_mask_0, x = coreml_update_state_87)[name = string("op_8074_cast_fp16")]; tensor value_cache_31_axes_0 = const()[name = string("value_cache_31_axes_0"), val = tensor([0])]; tensor value_cache_31_cast_fp16 = squeeze(axes = value_cache_31_axes_0, x = var_8074_cast_fp16)[name = string("value_cache_31_cast_fp16")]; tensor var_8098_axes_0 = const()[name = string("op_8098_axes_0"), val = tensor([1])]; tensor var_8098_cast_fp16 = expand_dims(axes = var_8098_axes_0, x = key_cache_31_cast_fp16)[name = string("op_8098_cast_fp16")]; tensor var_8103 = const()[name = string("op_8103"), val = tensor([1, 2, 1, 1])]; tensor value_123_cast_fp16 = tile(reps = var_8103, x = var_8098_cast_fp16)[name = string("value_123_cast_fp16")]; tensor var_8109 = const()[name = string("op_8109"), val = tensor([1, 16, 1536, 128])]; tensor key_states_63_cast_fp16 = reshape(shape = var_8109, x = value_123_cast_fp16)[name = string("key_states_63_cast_fp16")]; tensor var_8112_axes_0 = const()[name = string("op_8112_axes_0"), val = tensor([1])]; tensor var_8112_cast_fp16 = expand_dims(axes = var_8112_axes_0, x = value_cache_31_cast_fp16)[name = string("op_8112_cast_fp16")]; tensor var_8117 = const()[name = string("op_8117"), val = tensor([1, 2, 1, 1])]; tensor value_127_cast_fp16 = tile(reps = var_8117, x = var_8112_cast_fp16)[name = string("value_127_cast_fp16")]; tensor var_8123 = const()[name = string("op_8123"), val = tensor([1, 16, 1536, 128])]; tensor value_states_93_cast_fp16 = reshape(shape = var_8123, x = value_127_cast_fp16)[name = string("value_states_93_cast_fp16")]; bool var_8138_transpose_x_1 = const()[name = string("op_8138_transpose_x_1"), val = bool(false)]; bool var_8138_transpose_y_1 = const()[name = string("op_8138_transpose_y_1"), val = bool(true)]; tensor var_8138 = matmul(transpose_x = var_8138_transpose_x_1, transpose_y = var_8138_transpose_y_1, x = query_states_61, y = key_states_63_cast_fp16)[name = string("op_8138")]; fp16 var_8139_to_fp16 = const()[name = string("op_8139_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_61_cast_fp16 = mul(x = var_8138, y = var_8139_to_fp16)[name = string("attention_61_cast_fp16")]; tensor attention_63_cast_fp16 = add(x = attention_61_cast_fp16, y = causal_mask)[name = string("attention_63_cast_fp16")]; int32 var_8148 = const()[name = string("op_8148"), val = int32(-1)]; tensor probabilities_31_cast_fp16 = softmax(axis = var_8148, x = attention_63_cast_fp16)[name = string("probabilities_31_cast_fp16")]; bool output_91_transpose_x_0 = const()[name = string("output_91_transpose_x_0"), val = bool(false)]; bool output_91_transpose_y_0 = const()[name = string("output_91_transpose_y_0"), val = bool(false)]; tensor output_91_cast_fp16 = matmul(transpose_x = output_91_transpose_x_0, transpose_y = output_91_transpose_y_0, x = probabilities_31_cast_fp16, y = value_states_93_cast_fp16)[name = string("output_91_cast_fp16")]; tensor var_8159_perm_0 = const()[name = string("op_8159_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_8165 = const()[name = string("op_8165"), val = tensor([1, 1, 2048])]; tensor var_8159_cast_fp16 = transpose(perm = var_8159_perm_0, x = output_91_cast_fp16)[name = string("transpose_76")]; tensor output_93_cast_fp16 = reshape(shape = var_8165, x = var_8159_cast_fp16)[name = string("output_93_cast_fp16")]; tensor var_8170 = const()[name = string("op_8170"), val = tensor([0, 2, 1])]; string var_8186_pad_type_0 = const()[name = string("op_8186_pad_type_0"), val = string("valid")]; int32 var_8186_groups_0 = const()[name = string("op_8186_groups_0"), val = int32(1)]; tensor var_8186_strides_0 = const()[name = string("op_8186_strides_0"), val = tensor([1])]; tensor var_8186_pad_0 = const()[name = string("op_8186_pad_0"), val = tensor([0, 0])]; tensor var_8186_dilations_0 = const()[name = string("op_8186_dilations_0"), val = tensor([1])]; tensor squeeze_15_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316818432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318391360))))[name = string("squeeze_15_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_8171_cast_fp16 = transpose(perm = var_8170, x = output_93_cast_fp16)[name = string("transpose_75")]; tensor var_8186_cast_fp16 = conv(dilations = var_8186_dilations_0, groups = var_8186_groups_0, pad = var_8186_pad_0, pad_type = var_8186_pad_type_0, strides = var_8186_strides_0, weight = squeeze_15_cast_fp16_to_fp32_to_fp16_palettized, x = var_8171_cast_fp16)[name = string("op_8186_cast_fp16")]; tensor var_8190 = const()[name = string("op_8190"), val = tensor([0, 2, 1])]; tensor attn_output_31_cast_fp16 = transpose(perm = var_8190, x = var_8186_cast_fp16)[name = string("transpose_74")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_151_cast_fp16, y = attn_output_31_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; int32 var_8205 = const()[name = string("op_8205"), val = int32(-1)]; fp16 const_221_promoted_to_fp16 = const()[name = string("const_221_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8207_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_221_promoted_to_fp16)[name = string("op_8207_cast_fp16")]; bool input_281_interleave_0 = const()[name = string("input_281_interleave_0"), val = bool(false)]; tensor input_281_cast_fp16 = concat(axis = var_8205, interleave = input_281_interleave_0, values = (hidden_states_159_cast_fp16, var_8207_cast_fp16))[name = string("input_281_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; fp16 var_8202_to_fp16 = const()[name = string("op_8202_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_8202_to_fp16, x = input_281_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor normed_255_begin_0 = const()[name = string("normed_255_begin_0"), val = tensor([0, 0, 0])]; tensor normed_255_end_0 = const()[name = string("normed_255_end_0"), val = tensor([1, 1, 1024])]; tensor normed_255_end_mask_0 = const()[name = string("normed_255_end_mask_0"), val = tensor([true, true, false])]; tensor normed_255_cast_fp16 = slice_by_index(begin = normed_255_begin_0, end = normed_255_end_0, end_mask = normed_255_end_mask_0, x = normed_253_cast_fp16)[name = string("normed_255_cast_fp16")]; tensor const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318407808)))]; tensor x_61_cast_fp16 = mul(x = normed_255_cast_fp16, y = const_223_promoted_to_fp16)[name = string("x_61_cast_fp16")]; tensor var_8227 = const()[name = string("op_8227"), val = tensor([0, 2, 1])]; tensor input_283_axes_0 = const()[name = string("input_283_axes_0"), val = tensor([2])]; tensor var_8228 = transpose(perm = var_8227, x = x_61_cast_fp16)[name = string("transpose_73")]; tensor input_283 = expand_dims(axes = input_283_axes_0, x = var_8228)[name = string("input_283")]; string input_285_pad_type_0 = const()[name = string("input_285_pad_type_0"), val = string("valid")]; tensor input_285_strides_0 = const()[name = string("input_285_strides_0"), val = tensor([1, 1])]; tensor input_285_pad_0 = const()[name = string("input_285_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_285_dilations_0 = const()[name = string("input_285_dilations_0"), val = tensor([1, 1])]; int32 input_285_groups_0 = const()[name = string("input_285_groups_0"), val = int32(1)]; tensor input_285 = conv(dilations = input_285_dilations_0, groups = input_285_groups_0, pad = input_285_pad_0, pad_type = input_285_pad_type_0, strides = input_285_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_283)[name = string("input_285")]; string b_31_pad_type_0 = const()[name = string("b_31_pad_type_0"), val = string("valid")]; tensor b_31_strides_0 = const()[name = string("b_31_strides_0"), val = tensor([1, 1])]; tensor b_31_pad_0 = const()[name = string("b_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_31_dilations_0 = const()[name = string("b_31_dilations_0"), val = tensor([1, 1])]; int32 b_31_groups_0 = const()[name = string("b_31_groups_0"), val = int32(1)]; tensor b_31 = conv(dilations = b_31_dilations_0, groups = b_31_groups_0, pad = b_31_pad_0, pad_type = b_31_pad_type_0, strides = b_31_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_283)[name = string("b_31")]; tensor c_31 = silu(x = input_285)[name = string("c_31")]; tensor input_287 = mul(x = c_31, y = b_31)[name = string("input_287")]; string e_31_pad_type_0 = const()[name = string("e_31_pad_type_0"), val = string("valid")]; tensor e_31_strides_0 = const()[name = string("e_31_strides_0"), val = tensor([1, 1])]; tensor e_31_pad_0 = const()[name = string("e_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_31_dilations_0 = const()[name = string("e_31_dilations_0"), val = tensor([1, 1])]; int32 e_31_groups_0 = const()[name = string("e_31_groups_0"), val = int32(1)]; tensor e_31 = conv(dilations = e_31_dilations_0, groups = e_31_groups_0, pad = e_31_pad_0, pad_type = e_31_pad_type_0, strides = e_31_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_287)[name = string("e_31")]; tensor var_8250_axes_0 = const()[name = string("op_8250_axes_0"), val = tensor([2])]; tensor var_8250 = squeeze(axes = var_8250_axes_0, x = e_31)[name = string("op_8250")]; tensor var_8251 = const()[name = string("op_8251"), val = tensor([0, 2, 1])]; tensor var_8252 = transpose(perm = var_8251, x = var_8250)[name = string("transpose_72")]; tensor hidden_states_161_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = var_8252)[name = string("hidden_states_161_cast_fp16")]; int32 var_8266 = const()[name = string("op_8266"), val = int32(-1)]; fp16 const_224_promoted_to_fp16 = const()[name = string("const_224_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8268_cast_fp16 = mul(x = hidden_states_161_cast_fp16, y = const_224_promoted_to_fp16)[name = string("op_8268_cast_fp16")]; bool input_289_interleave_0 = const()[name = string("input_289_interleave_0"), val = bool(false)]; tensor input_289_cast_fp16 = concat(axis = var_8266, interleave = input_289_interleave_0, values = (hidden_states_161_cast_fp16, var_8268_cast_fp16))[name = string("input_289_cast_fp16")]; tensor normed_257_axes_0 = const()[name = string("normed_257_axes_0"), val = tensor([-1])]; fp16 var_8263_to_fp16 = const()[name = string("op_8263_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_257_cast_fp16 = layer_norm(axes = normed_257_axes_0, epsilon = var_8263_to_fp16, x = input_289_cast_fp16)[name = string("normed_257_cast_fp16")]; tensor normed_259_begin_0 = const()[name = string("normed_259_begin_0"), val = tensor([0, 0, 0])]; tensor normed_259_end_0 = const()[name = string("normed_259_end_0"), val = tensor([1, 1, 1024])]; tensor normed_259_end_mask_0 = const()[name = string("normed_259_end_mask_0"), val = tensor([true, true, false])]; tensor normed_259_cast_fp16 = slice_by_index(begin = normed_259_begin_0, end = normed_259_end_0, end_mask = normed_259_end_mask_0, x = normed_257_cast_fp16)[name = string("normed_259_cast_fp16")]; tensor const_226_promoted_to_fp16 = const()[name = string("const_226_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318409920)))]; tensor hidden_states_163_cast_fp16 = mul(x = normed_259_cast_fp16, y = const_226_promoted_to_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor var_8280 = const()[name = string("op_8280"), val = tensor([0, 2, 1])]; tensor var_8283_axes_0 = const()[name = string("op_8283_axes_0"), val = tensor([2])]; tensor var_8281_cast_fp16 = transpose(perm = var_8280, x = hidden_states_163_cast_fp16)[name = string("transpose_71")]; tensor var_8283_cast_fp16 = expand_dims(axes = var_8283_axes_0, x = var_8281_cast_fp16)[name = string("op_8283_cast_fp16")]; string var_8299_pad_type_0 = const()[name = string("op_8299_pad_type_0"), val = string("valid")]; tensor var_8299_strides_0 = const()[name = string("op_8299_strides_0"), val = tensor([1, 1])]; tensor var_8299_pad_0 = const()[name = string("op_8299_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8299_dilations_0 = const()[name = string("op_8299_dilations_0"), val = tensor([1, 1])]; int32 var_8299_groups_0 = const()[name = string("op_8299_groups_0"), val = int32(1)]; tensor var_8299 = conv(dilations = var_8299_dilations_0, groups = var_8299_groups_0, pad = var_8299_pad_0, pad_type = var_8299_pad_type_0, strides = var_8299_strides_0, weight = model_model_layers_16_self_attn_q_proj_weight_palettized, x = var_8283_cast_fp16)[name = string("op_8299")]; tensor var_8304 = const()[name = string("op_8304"), val = tensor([1, 16, 1, 128])]; tensor var_8305 = reshape(shape = var_8304, x = var_8299)[name = string("op_8305")]; string var_8321_pad_type_0 = const()[name = string("op_8321_pad_type_0"), val = string("valid")]; tensor var_8321_strides_0 = const()[name = string("op_8321_strides_0"), val = tensor([1, 1])]; tensor var_8321_pad_0 = const()[name = string("op_8321_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8321_dilations_0 = const()[name = string("op_8321_dilations_0"), val = tensor([1, 1])]; int32 var_8321_groups_0 = const()[name = string("op_8321_groups_0"), val = int32(1)]; tensor var_8321 = conv(dilations = var_8321_dilations_0, groups = var_8321_groups_0, pad = var_8321_pad_0, pad_type = var_8321_pad_type_0, strides = var_8321_strides_0, weight = model_model_layers_16_self_attn_k_proj_weight_palettized, x = var_8283_cast_fp16)[name = string("op_8321")]; tensor var_8326 = const()[name = string("op_8326"), val = tensor([1, 8, 1, 128])]; tensor var_8327 = reshape(shape = var_8326, x = var_8321)[name = string("op_8327")]; string var_8343_pad_type_0 = const()[name = string("op_8343_pad_type_0"), val = string("valid")]; tensor var_8343_strides_0 = const()[name = string("op_8343_strides_0"), val = tensor([1, 1])]; tensor var_8343_pad_0 = const()[name = string("op_8343_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8343_dilations_0 = const()[name = string("op_8343_dilations_0"), val = tensor([1, 1])]; int32 var_8343_groups_0 = const()[name = string("op_8343_groups_0"), val = int32(1)]; tensor var_8343 = conv(dilations = var_8343_dilations_0, groups = var_8343_groups_0, pad = var_8343_pad_0, pad_type = var_8343_pad_type_0, strides = var_8343_strides_0, weight = model_model_layers_16_self_attn_v_proj_weight_palettized, x = var_8283_cast_fp16)[name = string("op_8343")]; tensor var_8348 = const()[name = string("op_8348"), val = tensor([1, 8, 1, 128])]; tensor var_8349 = reshape(shape = var_8348, x = var_8343)[name = string("op_8349")]; int32 var_8366 = const()[name = string("op_8366"), val = int32(-1)]; fp16 const_227_promoted = const()[name = string("const_227_promoted"), val = fp16(-0x1p+0)]; tensor var_8368 = mul(x = var_8305, y = const_227_promoted)[name = string("op_8368")]; bool input_293_interleave_0 = const()[name = string("input_293_interleave_0"), val = bool(false)]; tensor input_293 = concat(axis = var_8366, interleave = input_293_interleave_0, values = (var_8305, var_8368))[name = string("input_293")]; tensor normed_261_axes_0 = const()[name = string("normed_261_axes_0"), val = tensor([-1])]; fp16 var_8363_to_fp16 = const()[name = string("op_8363_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_261_cast_fp16 = layer_norm(axes = normed_261_axes_0, epsilon = var_8363_to_fp16, x = input_293)[name = string("normed_261_cast_fp16")]; tensor normed_263_begin_0 = const()[name = string("normed_263_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_263_end_0 = const()[name = string("normed_263_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_263_end_mask_0 = const()[name = string("normed_263_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_263 = slice_by_index(begin = normed_263_begin_0, end = normed_263_end_0, end_mask = normed_263_end_mask_0, x = normed_261_cast_fp16)[name = string("normed_263")]; tensor const_229 = const()[name = string("const_229"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318412032)))]; tensor q_33 = mul(x = normed_263, y = const_229)[name = string("q_33")]; int32 var_8388 = const()[name = string("op_8388"), val = int32(-1)]; fp16 const_230_promoted = const()[name = string("const_230_promoted"), val = fp16(-0x1p+0)]; tensor var_8390 = mul(x = var_8327, y = const_230_promoted)[name = string("op_8390")]; bool input_295_interleave_0 = const()[name = string("input_295_interleave_0"), val = bool(false)]; tensor input_295 = concat(axis = var_8388, interleave = input_295_interleave_0, values = (var_8327, var_8390))[name = string("input_295")]; tensor normed_265_axes_0 = const()[name = string("normed_265_axes_0"), val = tensor([-1])]; fp16 var_8385_to_fp16 = const()[name = string("op_8385_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_265_cast_fp16 = layer_norm(axes = normed_265_axes_0, epsilon = var_8385_to_fp16, x = input_295)[name = string("normed_265_cast_fp16")]; tensor normed_267_begin_0 = const()[name = string("normed_267_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_267_end_0 = const()[name = string("normed_267_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_267_end_mask_0 = const()[name = string("normed_267_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_267 = slice_by_index(begin = normed_267_begin_0, end = normed_267_end_0, end_mask = normed_267_end_mask_0, x = normed_265_cast_fp16)[name = string("normed_267")]; tensor const_232 = const()[name = string("const_232"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318412352)))]; tensor k_33 = mul(x = normed_267, y = const_232)[name = string("k_33")]; tensor var_8399 = mul(x = q_33, y = cos_1_cast_fp16)[name = string("op_8399")]; tensor var_8404_begin_0 = const()[name = string("op_8404_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8404_end_0 = const()[name = string("op_8404_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_8404_end_mask_0 = const()[name = string("op_8404_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8404 = slice_by_index(begin = var_8404_begin_0, end = var_8404_end_0, end_mask = var_8404_end_mask_0, x = q_33)[name = string("op_8404")]; fp16 const_233_promoted = const()[name = string("const_233_promoted"), val = fp16(-0x1p+0)]; tensor var_8405 = mul(x = var_8404, y = const_233_promoted)[name = string("op_8405")]; tensor var_8410_begin_0 = const()[name = string("op_8410_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8410_end_0 = const()[name = string("op_8410_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_8410_end_mask_0 = const()[name = string("op_8410_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8410 = slice_by_index(begin = var_8410_begin_0, end = var_8410_end_0, end_mask = var_8410_end_mask_0, x = q_33)[name = string("op_8410")]; int32 var_8412 = const()[name = string("op_8412"), val = int32(-1)]; bool var_8413_interleave_0 = const()[name = string("op_8413_interleave_0"), val = bool(false)]; tensor var_8413 = concat(axis = var_8412, interleave = var_8413_interleave_0, values = (var_8405, var_8410))[name = string("op_8413")]; tensor var_8414 = mul(x = var_8413, y = sin_1_cast_fp16)[name = string("op_8414")]; tensor query_states_65 = add(x = var_8399, y = var_8414)[name = string("query_states_65")]; tensor var_8417 = mul(x = k_33, y = cos_1_cast_fp16)[name = string("op_8417")]; tensor var_8422_begin_0 = const()[name = string("op_8422_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8422_end_0 = const()[name = string("op_8422_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_8422_end_mask_0 = const()[name = string("op_8422_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8422 = slice_by_index(begin = var_8422_begin_0, end = var_8422_end_0, end_mask = var_8422_end_mask_0, x = k_33)[name = string("op_8422")]; fp16 const_234_promoted = const()[name = string("const_234_promoted"), val = fp16(-0x1p+0)]; tensor var_8423 = mul(x = var_8422, y = const_234_promoted)[name = string("op_8423")]; tensor var_8428_begin_0 = const()[name = string("op_8428_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8428_end_0 = const()[name = string("op_8428_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_8428_end_mask_0 = const()[name = string("op_8428_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8428 = slice_by_index(begin = var_8428_begin_0, end = var_8428_end_0, end_mask = var_8428_end_mask_0, x = k_33)[name = string("op_8428")]; int32 var_8430 = const()[name = string("op_8430"), val = int32(-1)]; bool var_8431_interleave_0 = const()[name = string("op_8431_interleave_0"), val = bool(false)]; tensor var_8431 = concat(axis = var_8430, interleave = var_8431_interleave_0, values = (var_8423, var_8428))[name = string("op_8431")]; tensor var_8432 = mul(x = var_8431, y = sin_1_cast_fp16)[name = string("op_8432")]; tensor key_states_65 = add(x = var_8417, y = var_8432)[name = string("key_states_65")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; int32 concat_130_axis_0 = const()[name = string("concat_130_axis_0"), val = int32(0)]; bool concat_130_interleave_0 = const()[name = string("concat_130_interleave_0"), val = bool(false)]; tensor concat_130 = concat(axis = concat_130_axis_0, interleave = concat_130_interleave_0, values = (expand_dims_192, expand_dims_193, current_pos, expand_dims_195))[name = string("concat_130")]; tensor concat_131_values1_0 = const()[name = string("concat_131_values1_0"), val = tensor([0])]; tensor concat_131_values3_0 = const()[name = string("concat_131_values3_0"), val = tensor([0])]; int32 concat_131_axis_0 = const()[name = string("concat_131_axis_0"), val = int32(0)]; bool concat_131_interleave_0 = const()[name = string("concat_131_interleave_0"), val = bool(false)]; tensor concat_131 = concat(axis = concat_131_axis_0, interleave = concat_131_interleave_0, values = (expand_dims_196, concat_131_values1_0, var_1717, concat_131_values3_0))[name = string("concat_131")]; tensor model_model_kv_cache_0_internal_tensor_assign_33_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_cast_fp16 = slice_update(begin = concat_130, begin_mask = model_model_kv_cache_0_internal_tensor_assign_33_begin_mask_0, end = concat_131, end_mask = model_model_kv_cache_0_internal_tensor_assign_33_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_33_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_33_stride_0, update = key_states_65, x = coreml_update_state_87)[name = string("model_model_kv_cache_0_internal_tensor_assign_33_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_33_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_144_write_state")]; tensor coreml_update_state_88 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_144")]; tensor expand_dims_198 = const()[name = string("expand_dims_198"), val = tensor([44])]; tensor expand_dims_199 = const()[name = string("expand_dims_199"), val = tensor([0])]; tensor expand_dims_201 = const()[name = string("expand_dims_201"), val = tensor([0])]; tensor expand_dims_202 = const()[name = string("expand_dims_202"), val = tensor([45])]; int32 concat_134_axis_0 = const()[name = string("concat_134_axis_0"), val = int32(0)]; bool concat_134_interleave_0 = const()[name = string("concat_134_interleave_0"), val = bool(false)]; tensor concat_134 = concat(axis = concat_134_axis_0, interleave = concat_134_interleave_0, values = (expand_dims_198, expand_dims_199, current_pos, expand_dims_201))[name = string("concat_134")]; tensor concat_135_values1_0 = const()[name = string("concat_135_values1_0"), val = tensor([0])]; tensor concat_135_values3_0 = const()[name = string("concat_135_values3_0"), val = tensor([0])]; int32 concat_135_axis_0 = const()[name = string("concat_135_axis_0"), val = int32(0)]; bool concat_135_interleave_0 = const()[name = string("concat_135_interleave_0"), val = bool(false)]; tensor concat_135 = concat(axis = concat_135_axis_0, interleave = concat_135_interleave_0, values = (expand_dims_202, concat_135_values1_0, var_1717, concat_135_values3_0))[name = string("concat_135")]; tensor model_model_kv_cache_0_internal_tensor_assign_34_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_cast_fp16 = slice_update(begin = concat_134, begin_mask = model_model_kv_cache_0_internal_tensor_assign_34_begin_mask_0, end = concat_135, end_mask = model_model_kv_cache_0_internal_tensor_assign_34_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_34_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_34_stride_0, update = var_8349, x = coreml_update_state_88)[name = string("model_model_kv_cache_0_internal_tensor_assign_34_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_34_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_145_write_state")]; tensor coreml_update_state_89 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_145")]; tensor var_8487_begin_0 = const()[name = string("op_8487_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_8487_end_0 = const()[name = string("op_8487_end_0"), val = tensor([17, 8, 1536, 128])]; tensor var_8487_end_mask_0 = const()[name = string("op_8487_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8487_cast_fp16 = slice_by_index(begin = var_8487_begin_0, end = var_8487_end_0, end_mask = var_8487_end_mask_0, x = coreml_update_state_89)[name = string("op_8487_cast_fp16")]; tensor key_cache_33_axes_0 = const()[name = string("key_cache_33_axes_0"), val = tensor([0])]; tensor key_cache_33_cast_fp16 = squeeze(axes = key_cache_33_axes_0, x = var_8487_cast_fp16)[name = string("key_cache_33_cast_fp16")]; tensor var_8494_begin_0 = const()[name = string("op_8494_begin_0"), val = tensor([44, 0, 0, 0])]; tensor var_8494_end_0 = const()[name = string("op_8494_end_0"), val = tensor([45, 8, 1536, 128])]; tensor var_8494_end_mask_0 = const()[name = string("op_8494_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8494_cast_fp16 = slice_by_index(begin = var_8494_begin_0, end = var_8494_end_0, end_mask = var_8494_end_mask_0, x = coreml_update_state_89)[name = string("op_8494_cast_fp16")]; tensor value_cache_33_axes_0 = const()[name = string("value_cache_33_axes_0"), val = tensor([0])]; tensor value_cache_33_cast_fp16 = squeeze(axes = value_cache_33_axes_0, x = var_8494_cast_fp16)[name = string("value_cache_33_cast_fp16")]; tensor var_8518_axes_0 = const()[name = string("op_8518_axes_0"), val = tensor([1])]; tensor var_8518_cast_fp16 = expand_dims(axes = var_8518_axes_0, x = key_cache_33_cast_fp16)[name = string("op_8518_cast_fp16")]; tensor var_8523 = const()[name = string("op_8523"), val = tensor([1, 2, 1, 1])]; tensor value_131_cast_fp16 = tile(reps = var_8523, x = var_8518_cast_fp16)[name = string("value_131_cast_fp16")]; tensor var_8529 = const()[name = string("op_8529"), val = tensor([1, 16, 1536, 128])]; tensor key_states_67_cast_fp16 = reshape(shape = var_8529, x = value_131_cast_fp16)[name = string("key_states_67_cast_fp16")]; tensor var_8532_axes_0 = const()[name = string("op_8532_axes_0"), val = tensor([1])]; tensor var_8532_cast_fp16 = expand_dims(axes = var_8532_axes_0, x = value_cache_33_cast_fp16)[name = string("op_8532_cast_fp16")]; tensor var_8537 = const()[name = string("op_8537"), val = tensor([1, 2, 1, 1])]; tensor value_135_cast_fp16 = tile(reps = var_8537, x = var_8532_cast_fp16)[name = string("value_135_cast_fp16")]; tensor var_8543 = const()[name = string("op_8543"), val = tensor([1, 16, 1536, 128])]; tensor value_states_99_cast_fp16 = reshape(shape = var_8543, x = value_135_cast_fp16)[name = string("value_states_99_cast_fp16")]; bool var_8558_transpose_x_1 = const()[name = string("op_8558_transpose_x_1"), val = bool(false)]; bool var_8558_transpose_y_1 = const()[name = string("op_8558_transpose_y_1"), val = bool(true)]; tensor var_8558 = matmul(transpose_x = var_8558_transpose_x_1, transpose_y = var_8558_transpose_y_1, x = query_states_65, y = key_states_67_cast_fp16)[name = string("op_8558")]; fp16 var_8559_to_fp16 = const()[name = string("op_8559_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_65_cast_fp16 = mul(x = var_8558, y = var_8559_to_fp16)[name = string("attention_65_cast_fp16")]; tensor attention_67_cast_fp16 = add(x = attention_65_cast_fp16, y = causal_mask)[name = string("attention_67_cast_fp16")]; int32 var_8568 = const()[name = string("op_8568"), val = int32(-1)]; tensor probabilities_33_cast_fp16 = softmax(axis = var_8568, x = attention_67_cast_fp16)[name = string("probabilities_33_cast_fp16")]; bool output_97_transpose_x_0 = const()[name = string("output_97_transpose_x_0"), val = bool(false)]; bool output_97_transpose_y_0 = const()[name = string("output_97_transpose_y_0"), val = bool(false)]; tensor output_97_cast_fp16 = matmul(transpose_x = output_97_transpose_x_0, transpose_y = output_97_transpose_y_0, x = probabilities_33_cast_fp16, y = value_states_99_cast_fp16)[name = string("output_97_cast_fp16")]; tensor var_8579_perm_0 = const()[name = string("op_8579_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_8585 = const()[name = string("op_8585"), val = tensor([1, 1, 2048])]; tensor var_8579_cast_fp16 = transpose(perm = var_8579_perm_0, x = output_97_cast_fp16)[name = string("transpose_70")]; tensor output_99_cast_fp16 = reshape(shape = var_8585, x = var_8579_cast_fp16)[name = string("output_99_cast_fp16")]; tensor var_8590 = const()[name = string("op_8590"), val = tensor([0, 2, 1])]; string var_8606_pad_type_0 = const()[name = string("op_8606_pad_type_0"), val = string("valid")]; int32 var_8606_groups_0 = const()[name = string("op_8606_groups_0"), val = int32(1)]; tensor var_8606_strides_0 = const()[name = string("op_8606_strides_0"), val = tensor([1])]; tensor var_8606_pad_0 = const()[name = string("op_8606_pad_0"), val = tensor([0, 0])]; tensor var_8606_dilations_0 = const()[name = string("op_8606_dilations_0"), val = tensor([1])]; tensor squeeze_16_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318412672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319985600))))[name = string("squeeze_16_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_8591_cast_fp16 = transpose(perm = var_8590, x = output_99_cast_fp16)[name = string("transpose_69")]; tensor var_8606_cast_fp16 = conv(dilations = var_8606_dilations_0, groups = var_8606_groups_0, pad = var_8606_pad_0, pad_type = var_8606_pad_type_0, strides = var_8606_strides_0, weight = squeeze_16_cast_fp16_to_fp32_to_fp16_palettized, x = var_8591_cast_fp16)[name = string("op_8606_cast_fp16")]; tensor var_8610 = const()[name = string("op_8610"), val = tensor([0, 2, 1])]; tensor attn_output_33_cast_fp16 = transpose(perm = var_8610, x = var_8606_cast_fp16)[name = string("transpose_68")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_161_cast_fp16, y = attn_output_33_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; int32 var_8625 = const()[name = string("op_8625"), val = int32(-1)]; fp16 const_235_promoted_to_fp16 = const()[name = string("const_235_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8627_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_235_promoted_to_fp16)[name = string("op_8627_cast_fp16")]; bool input_299_interleave_0 = const()[name = string("input_299_interleave_0"), val = bool(false)]; tensor input_299_cast_fp16 = concat(axis = var_8625, interleave = input_299_interleave_0, values = (hidden_states_169_cast_fp16, var_8627_cast_fp16))[name = string("input_299_cast_fp16")]; tensor normed_269_axes_0 = const()[name = string("normed_269_axes_0"), val = tensor([-1])]; fp16 var_8622_to_fp16 = const()[name = string("op_8622_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_269_cast_fp16 = layer_norm(axes = normed_269_axes_0, epsilon = var_8622_to_fp16, x = input_299_cast_fp16)[name = string("normed_269_cast_fp16")]; tensor normed_271_begin_0 = const()[name = string("normed_271_begin_0"), val = tensor([0, 0, 0])]; tensor normed_271_end_0 = const()[name = string("normed_271_end_0"), val = tensor([1, 1, 1024])]; tensor normed_271_end_mask_0 = const()[name = string("normed_271_end_mask_0"), val = tensor([true, true, false])]; tensor normed_271_cast_fp16 = slice_by_index(begin = normed_271_begin_0, end = normed_271_end_0, end_mask = normed_271_end_mask_0, x = normed_269_cast_fp16)[name = string("normed_271_cast_fp16")]; tensor const_237_promoted_to_fp16 = const()[name = string("const_237_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320002048)))]; tensor x_65_cast_fp16 = mul(x = normed_271_cast_fp16, y = const_237_promoted_to_fp16)[name = string("x_65_cast_fp16")]; tensor var_8647 = const()[name = string("op_8647"), val = tensor([0, 2, 1])]; tensor input_301_axes_0 = const()[name = string("input_301_axes_0"), val = tensor([2])]; tensor var_8648 = transpose(perm = var_8647, x = x_65_cast_fp16)[name = string("transpose_67")]; tensor input_301 = expand_dims(axes = input_301_axes_0, x = var_8648)[name = string("input_301")]; string input_303_pad_type_0 = const()[name = string("input_303_pad_type_0"), val = string("valid")]; tensor input_303_strides_0 = const()[name = string("input_303_strides_0"), val = tensor([1, 1])]; tensor input_303_pad_0 = const()[name = string("input_303_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_303_dilations_0 = const()[name = string("input_303_dilations_0"), val = tensor([1, 1])]; int32 input_303_groups_0 = const()[name = string("input_303_groups_0"), val = int32(1)]; tensor input_303 = conv(dilations = input_303_dilations_0, groups = input_303_groups_0, pad = input_303_pad_0, pad_type = input_303_pad_type_0, strides = input_303_strides_0, weight = model_model_layers_16_mlp_gate_proj_weight_palettized, x = input_301)[name = string("input_303")]; string b_33_pad_type_0 = const()[name = string("b_33_pad_type_0"), val = string("valid")]; tensor b_33_strides_0 = const()[name = string("b_33_strides_0"), val = tensor([1, 1])]; tensor b_33_pad_0 = const()[name = string("b_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_33_dilations_0 = const()[name = string("b_33_dilations_0"), val = tensor([1, 1])]; int32 b_33_groups_0 = const()[name = string("b_33_groups_0"), val = int32(1)]; tensor b_33 = conv(dilations = b_33_dilations_0, groups = b_33_groups_0, pad = b_33_pad_0, pad_type = b_33_pad_type_0, strides = b_33_strides_0, weight = model_model_layers_16_mlp_up_proj_weight_palettized, x = input_301)[name = string("b_33")]; tensor c_33 = silu(x = input_303)[name = string("c_33")]; tensor input_305 = mul(x = c_33, y = b_33)[name = string("input_305")]; string e_33_pad_type_0 = const()[name = string("e_33_pad_type_0"), val = string("valid")]; tensor e_33_strides_0 = const()[name = string("e_33_strides_0"), val = tensor([1, 1])]; tensor e_33_pad_0 = const()[name = string("e_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_33_dilations_0 = const()[name = string("e_33_dilations_0"), val = tensor([1, 1])]; int32 e_33_groups_0 = const()[name = string("e_33_groups_0"), val = int32(1)]; tensor e_33 = conv(dilations = e_33_dilations_0, groups = e_33_groups_0, pad = e_33_pad_0, pad_type = e_33_pad_type_0, strides = e_33_strides_0, weight = model_model_layers_16_mlp_down_proj_weight_palettized, x = input_305)[name = string("e_33")]; tensor var_8670_axes_0 = const()[name = string("op_8670_axes_0"), val = tensor([2])]; tensor var_8670 = squeeze(axes = var_8670_axes_0, x = e_33)[name = string("op_8670")]; tensor var_8671 = const()[name = string("op_8671"), val = tensor([0, 2, 1])]; tensor var_8672 = transpose(perm = var_8671, x = var_8670)[name = string("transpose_66")]; tensor hidden_states_171_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = var_8672)[name = string("hidden_states_171_cast_fp16")]; int32 var_8686 = const()[name = string("op_8686"), val = int32(-1)]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8688_cast_fp16 = mul(x = hidden_states_171_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8688_cast_fp16")]; bool input_307_interleave_0 = const()[name = string("input_307_interleave_0"), val = bool(false)]; tensor input_307_cast_fp16 = concat(axis = var_8686, interleave = input_307_interleave_0, values = (hidden_states_171_cast_fp16, var_8688_cast_fp16))[name = string("input_307_cast_fp16")]; tensor normed_273_axes_0 = const()[name = string("normed_273_axes_0"), val = tensor([-1])]; fp16 var_8683_to_fp16 = const()[name = string("op_8683_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_273_cast_fp16 = layer_norm(axes = normed_273_axes_0, epsilon = var_8683_to_fp16, x = input_307_cast_fp16)[name = string("normed_273_cast_fp16")]; tensor normed_275_begin_0 = const()[name = string("normed_275_begin_0"), val = tensor([0, 0, 0])]; tensor normed_275_end_0 = const()[name = string("normed_275_end_0"), val = tensor([1, 1, 1024])]; tensor normed_275_end_mask_0 = const()[name = string("normed_275_end_mask_0"), val = tensor([true, true, false])]; tensor normed_275_cast_fp16 = slice_by_index(begin = normed_275_begin_0, end = normed_275_end_0, end_mask = normed_275_end_mask_0, x = normed_273_cast_fp16)[name = string("normed_275_cast_fp16")]; tensor const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320004160)))]; tensor hidden_states_173_cast_fp16 = mul(x = normed_275_cast_fp16, y = const_240_promoted_to_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor var_8700 = const()[name = string("op_8700"), val = tensor([0, 2, 1])]; tensor var_8703_axes_0 = const()[name = string("op_8703_axes_0"), val = tensor([2])]; tensor var_8701_cast_fp16 = transpose(perm = var_8700, x = hidden_states_173_cast_fp16)[name = string("transpose_65")]; tensor var_8703_cast_fp16 = expand_dims(axes = var_8703_axes_0, x = var_8701_cast_fp16)[name = string("op_8703_cast_fp16")]; string var_8719_pad_type_0 = const()[name = string("op_8719_pad_type_0"), val = string("valid")]; tensor var_8719_strides_0 = const()[name = string("op_8719_strides_0"), val = tensor([1, 1])]; tensor var_8719_pad_0 = const()[name = string("op_8719_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8719_dilations_0 = const()[name = string("op_8719_dilations_0"), val = tensor([1, 1])]; int32 var_8719_groups_0 = const()[name = string("op_8719_groups_0"), val = int32(1)]; tensor var_8719 = conv(dilations = var_8719_dilations_0, groups = var_8719_groups_0, pad = var_8719_pad_0, pad_type = var_8719_pad_type_0, strides = var_8719_strides_0, weight = model_model_layers_17_self_attn_q_proj_weight_palettized, x = var_8703_cast_fp16)[name = string("op_8719")]; tensor var_8724 = const()[name = string("op_8724"), val = tensor([1, 16, 1, 128])]; tensor var_8725 = reshape(shape = var_8724, x = var_8719)[name = string("op_8725")]; string var_8741_pad_type_0 = const()[name = string("op_8741_pad_type_0"), val = string("valid")]; tensor var_8741_strides_0 = const()[name = string("op_8741_strides_0"), val = tensor([1, 1])]; tensor var_8741_pad_0 = const()[name = string("op_8741_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8741_dilations_0 = const()[name = string("op_8741_dilations_0"), val = tensor([1, 1])]; int32 var_8741_groups_0 = const()[name = string("op_8741_groups_0"), val = int32(1)]; tensor var_8741 = conv(dilations = var_8741_dilations_0, groups = var_8741_groups_0, pad = var_8741_pad_0, pad_type = var_8741_pad_type_0, strides = var_8741_strides_0, weight = model_model_layers_17_self_attn_k_proj_weight_palettized, x = var_8703_cast_fp16)[name = string("op_8741")]; tensor var_8746 = const()[name = string("op_8746"), val = tensor([1, 8, 1, 128])]; tensor var_8747 = reshape(shape = var_8746, x = var_8741)[name = string("op_8747")]; string var_8763_pad_type_0 = const()[name = string("op_8763_pad_type_0"), val = string("valid")]; tensor var_8763_strides_0 = const()[name = string("op_8763_strides_0"), val = tensor([1, 1])]; tensor var_8763_pad_0 = const()[name = string("op_8763_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8763_dilations_0 = const()[name = string("op_8763_dilations_0"), val = tensor([1, 1])]; int32 var_8763_groups_0 = const()[name = string("op_8763_groups_0"), val = int32(1)]; tensor var_8763 = conv(dilations = var_8763_dilations_0, groups = var_8763_groups_0, pad = var_8763_pad_0, pad_type = var_8763_pad_type_0, strides = var_8763_strides_0, weight = model_model_layers_17_self_attn_v_proj_weight_palettized, x = var_8703_cast_fp16)[name = string("op_8763")]; tensor var_8768 = const()[name = string("op_8768"), val = tensor([1, 8, 1, 128])]; tensor var_8769 = reshape(shape = var_8768, x = var_8763)[name = string("op_8769")]; int32 var_8786 = const()[name = string("op_8786"), val = int32(-1)]; fp16 const_241_promoted = const()[name = string("const_241_promoted"), val = fp16(-0x1p+0)]; tensor var_8788 = mul(x = var_8725, y = const_241_promoted)[name = string("op_8788")]; bool input_311_interleave_0 = const()[name = string("input_311_interleave_0"), val = bool(false)]; tensor input_311 = concat(axis = var_8786, interleave = input_311_interleave_0, values = (var_8725, var_8788))[name = string("input_311")]; tensor normed_277_axes_0 = const()[name = string("normed_277_axes_0"), val = tensor([-1])]; fp16 var_8783_to_fp16 = const()[name = string("op_8783_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_277_cast_fp16 = layer_norm(axes = normed_277_axes_0, epsilon = var_8783_to_fp16, x = input_311)[name = string("normed_277_cast_fp16")]; tensor normed_279_begin_0 = const()[name = string("normed_279_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_279_end_0 = const()[name = string("normed_279_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_279_end_mask_0 = const()[name = string("normed_279_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_279 = slice_by_index(begin = normed_279_begin_0, end = normed_279_end_0, end_mask = normed_279_end_mask_0, x = normed_277_cast_fp16)[name = string("normed_279")]; tensor const_243 = const()[name = string("const_243"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320006272)))]; tensor q_35 = mul(x = normed_279, y = const_243)[name = string("q_35")]; int32 var_8808 = const()[name = string("op_8808"), val = int32(-1)]; fp16 const_244_promoted = const()[name = string("const_244_promoted"), val = fp16(-0x1p+0)]; tensor var_8810 = mul(x = var_8747, y = const_244_promoted)[name = string("op_8810")]; bool input_313_interleave_0 = const()[name = string("input_313_interleave_0"), val = bool(false)]; tensor input_313 = concat(axis = var_8808, interleave = input_313_interleave_0, values = (var_8747, var_8810))[name = string("input_313")]; tensor normed_281_axes_0 = const()[name = string("normed_281_axes_0"), val = tensor([-1])]; fp16 var_8805_to_fp16 = const()[name = string("op_8805_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_281_cast_fp16 = layer_norm(axes = normed_281_axes_0, epsilon = var_8805_to_fp16, x = input_313)[name = string("normed_281_cast_fp16")]; tensor normed_283_begin_0 = const()[name = string("normed_283_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_283_end_0 = const()[name = string("normed_283_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_283_end_mask_0 = const()[name = string("normed_283_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_283 = slice_by_index(begin = normed_283_begin_0, end = normed_283_end_0, end_mask = normed_283_end_mask_0, x = normed_281_cast_fp16)[name = string("normed_283")]; tensor const_246 = const()[name = string("const_246"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320006592)))]; tensor k_35 = mul(x = normed_283, y = const_246)[name = string("k_35")]; tensor var_8819 = mul(x = q_35, y = cos_1_cast_fp16)[name = string("op_8819")]; tensor var_8824_begin_0 = const()[name = string("op_8824_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8824_end_0 = const()[name = string("op_8824_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_8824_end_mask_0 = const()[name = string("op_8824_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8824 = slice_by_index(begin = var_8824_begin_0, end = var_8824_end_0, end_mask = var_8824_end_mask_0, x = q_35)[name = string("op_8824")]; fp16 const_247_promoted = const()[name = string("const_247_promoted"), val = fp16(-0x1p+0)]; tensor var_8825 = mul(x = var_8824, y = const_247_promoted)[name = string("op_8825")]; tensor var_8830_begin_0 = const()[name = string("op_8830_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8830_end_0 = const()[name = string("op_8830_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_8830_end_mask_0 = const()[name = string("op_8830_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8830 = slice_by_index(begin = var_8830_begin_0, end = var_8830_end_0, end_mask = var_8830_end_mask_0, x = q_35)[name = string("op_8830")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(-1)]; bool var_8833_interleave_0 = const()[name = string("op_8833_interleave_0"), val = bool(false)]; tensor var_8833 = concat(axis = var_8832, interleave = var_8833_interleave_0, values = (var_8825, var_8830))[name = string("op_8833")]; tensor var_8834 = mul(x = var_8833, y = sin_1_cast_fp16)[name = string("op_8834")]; tensor query_states_69 = add(x = var_8819, y = var_8834)[name = string("query_states_69")]; tensor var_8837 = mul(x = k_35, y = cos_1_cast_fp16)[name = string("op_8837")]; tensor var_8842_begin_0 = const()[name = string("op_8842_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8842_end_0 = const()[name = string("op_8842_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_8842_end_mask_0 = const()[name = string("op_8842_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8842 = slice_by_index(begin = var_8842_begin_0, end = var_8842_end_0, end_mask = var_8842_end_mask_0, x = k_35)[name = string("op_8842")]; fp16 const_248_promoted = const()[name = string("const_248_promoted"), val = fp16(-0x1p+0)]; tensor var_8843 = mul(x = var_8842, y = const_248_promoted)[name = string("op_8843")]; tensor var_8848_begin_0 = const()[name = string("op_8848_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8848_end_0 = const()[name = string("op_8848_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_8848_end_mask_0 = const()[name = string("op_8848_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8848 = slice_by_index(begin = var_8848_begin_0, end = var_8848_end_0, end_mask = var_8848_end_mask_0, x = k_35)[name = string("op_8848")]; int32 var_8850 = const()[name = string("op_8850"), val = int32(-1)]; bool var_8851_interleave_0 = const()[name = string("op_8851_interleave_0"), val = bool(false)]; tensor var_8851 = concat(axis = var_8850, interleave = var_8851_interleave_0, values = (var_8843, var_8848))[name = string("op_8851")]; tensor var_8852 = mul(x = var_8851, y = sin_1_cast_fp16)[name = string("op_8852")]; tensor key_states_69 = add(x = var_8837, y = var_8852)[name = string("key_states_69")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_204, expand_dims_205, current_pos, expand_dims_207))[name = string("concat_138")]; tensor concat_139_values1_0 = const()[name = string("concat_139_values1_0"), val = tensor([0])]; tensor concat_139_values3_0 = const()[name = string("concat_139_values3_0"), val = tensor([0])]; int32 concat_139_axis_0 = const()[name = string("concat_139_axis_0"), val = int32(0)]; bool concat_139_interleave_0 = const()[name = string("concat_139_interleave_0"), val = bool(false)]; tensor concat_139 = concat(axis = concat_139_axis_0, interleave = concat_139_interleave_0, values = (expand_dims_208, concat_139_values1_0, var_1717, concat_139_values3_0))[name = string("concat_139")]; tensor model_model_kv_cache_0_internal_tensor_assign_35_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_cast_fp16 = slice_update(begin = concat_138, begin_mask = model_model_kv_cache_0_internal_tensor_assign_35_begin_mask_0, end = concat_139, end_mask = model_model_kv_cache_0_internal_tensor_assign_35_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_35_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_35_stride_0, update = key_states_69, x = coreml_update_state_89)[name = string("model_model_kv_cache_0_internal_tensor_assign_35_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_35_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_146_write_state")]; tensor coreml_update_state_90 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_146")]; tensor expand_dims_210 = const()[name = string("expand_dims_210"), val = tensor([45])]; tensor expand_dims_211 = const()[name = string("expand_dims_211"), val = tensor([0])]; tensor expand_dims_213 = const()[name = string("expand_dims_213"), val = tensor([0])]; tensor expand_dims_214 = const()[name = string("expand_dims_214"), val = tensor([46])]; int32 concat_142_axis_0 = const()[name = string("concat_142_axis_0"), val = int32(0)]; bool concat_142_interleave_0 = const()[name = string("concat_142_interleave_0"), val = bool(false)]; tensor concat_142 = concat(axis = concat_142_axis_0, interleave = concat_142_interleave_0, values = (expand_dims_210, expand_dims_211, current_pos, expand_dims_213))[name = string("concat_142")]; tensor concat_143_values1_0 = const()[name = string("concat_143_values1_0"), val = tensor([0])]; tensor concat_143_values3_0 = const()[name = string("concat_143_values3_0"), val = tensor([0])]; int32 concat_143_axis_0 = const()[name = string("concat_143_axis_0"), val = int32(0)]; bool concat_143_interleave_0 = const()[name = string("concat_143_interleave_0"), val = bool(false)]; tensor concat_143 = concat(axis = concat_143_axis_0, interleave = concat_143_interleave_0, values = (expand_dims_214, concat_143_values1_0, var_1717, concat_143_values3_0))[name = string("concat_143")]; tensor model_model_kv_cache_0_internal_tensor_assign_36_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_cast_fp16 = slice_update(begin = concat_142, begin_mask = model_model_kv_cache_0_internal_tensor_assign_36_begin_mask_0, end = concat_143, end_mask = model_model_kv_cache_0_internal_tensor_assign_36_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_36_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_36_stride_0, update = var_8769, x = coreml_update_state_90)[name = string("model_model_kv_cache_0_internal_tensor_assign_36_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_36_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_147_write_state")]; tensor coreml_update_state_91 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_147")]; tensor var_8907_begin_0 = const()[name = string("op_8907_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_8907_end_0 = const()[name = string("op_8907_end_0"), val = tensor([18, 8, 1536, 128])]; tensor var_8907_end_mask_0 = const()[name = string("op_8907_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8907_cast_fp16 = slice_by_index(begin = var_8907_begin_0, end = var_8907_end_0, end_mask = var_8907_end_mask_0, x = coreml_update_state_91)[name = string("op_8907_cast_fp16")]; tensor key_cache_35_axes_0 = const()[name = string("key_cache_35_axes_0"), val = tensor([0])]; tensor key_cache_35_cast_fp16 = squeeze(axes = key_cache_35_axes_0, x = var_8907_cast_fp16)[name = string("key_cache_35_cast_fp16")]; tensor var_8914_begin_0 = const()[name = string("op_8914_begin_0"), val = tensor([45, 0, 0, 0])]; tensor var_8914_end_0 = const()[name = string("op_8914_end_0"), val = tensor([46, 8, 1536, 128])]; tensor var_8914_end_mask_0 = const()[name = string("op_8914_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8914_cast_fp16 = slice_by_index(begin = var_8914_begin_0, end = var_8914_end_0, end_mask = var_8914_end_mask_0, x = coreml_update_state_91)[name = string("op_8914_cast_fp16")]; tensor value_cache_35_axes_0 = const()[name = string("value_cache_35_axes_0"), val = tensor([0])]; tensor value_cache_35_cast_fp16 = squeeze(axes = value_cache_35_axes_0, x = var_8914_cast_fp16)[name = string("value_cache_35_cast_fp16")]; tensor var_8938_axes_0 = const()[name = string("op_8938_axes_0"), val = tensor([1])]; tensor var_8938_cast_fp16 = expand_dims(axes = var_8938_axes_0, x = key_cache_35_cast_fp16)[name = string("op_8938_cast_fp16")]; tensor var_8943 = const()[name = string("op_8943"), val = tensor([1, 2, 1, 1])]; tensor value_139_cast_fp16 = tile(reps = var_8943, x = var_8938_cast_fp16)[name = string("value_139_cast_fp16")]; tensor var_8949 = const()[name = string("op_8949"), val = tensor([1, 16, 1536, 128])]; tensor key_states_71_cast_fp16 = reshape(shape = var_8949, x = value_139_cast_fp16)[name = string("key_states_71_cast_fp16")]; tensor var_8952_axes_0 = const()[name = string("op_8952_axes_0"), val = tensor([1])]; tensor var_8952_cast_fp16 = expand_dims(axes = var_8952_axes_0, x = value_cache_35_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor var_8957 = const()[name = string("op_8957"), val = tensor([1, 2, 1, 1])]; tensor value_143_cast_fp16 = tile(reps = var_8957, x = var_8952_cast_fp16)[name = string("value_143_cast_fp16")]; tensor var_8963 = const()[name = string("op_8963"), val = tensor([1, 16, 1536, 128])]; tensor value_states_105_cast_fp16 = reshape(shape = var_8963, x = value_143_cast_fp16)[name = string("value_states_105_cast_fp16")]; bool var_8978_transpose_x_1 = const()[name = string("op_8978_transpose_x_1"), val = bool(false)]; bool var_8978_transpose_y_1 = const()[name = string("op_8978_transpose_y_1"), val = bool(true)]; tensor var_8978 = matmul(transpose_x = var_8978_transpose_x_1, transpose_y = var_8978_transpose_y_1, x = query_states_69, y = key_states_71_cast_fp16)[name = string("op_8978")]; fp16 var_8979_to_fp16 = const()[name = string("op_8979_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_69_cast_fp16 = mul(x = var_8978, y = var_8979_to_fp16)[name = string("attention_69_cast_fp16")]; tensor attention_71_cast_fp16 = add(x = attention_69_cast_fp16, y = causal_mask)[name = string("attention_71_cast_fp16")]; int32 var_8988 = const()[name = string("op_8988"), val = int32(-1)]; tensor probabilities_35_cast_fp16 = softmax(axis = var_8988, x = attention_71_cast_fp16)[name = string("probabilities_35_cast_fp16")]; bool output_103_transpose_x_0 = const()[name = string("output_103_transpose_x_0"), val = bool(false)]; bool output_103_transpose_y_0 = const()[name = string("output_103_transpose_y_0"), val = bool(false)]; tensor output_103_cast_fp16 = matmul(transpose_x = output_103_transpose_x_0, transpose_y = output_103_transpose_y_0, x = probabilities_35_cast_fp16, y = value_states_105_cast_fp16)[name = string("output_103_cast_fp16")]; tensor var_8999_perm_0 = const()[name = string("op_8999_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_9005 = const()[name = string("op_9005"), val = tensor([1, 1, 2048])]; tensor var_8999_cast_fp16 = transpose(perm = var_8999_perm_0, x = output_103_cast_fp16)[name = string("transpose_64")]; tensor output_105_cast_fp16 = reshape(shape = var_9005, x = var_8999_cast_fp16)[name = string("output_105_cast_fp16")]; tensor var_9010 = const()[name = string("op_9010"), val = tensor([0, 2, 1])]; string var_9026_pad_type_0 = const()[name = string("op_9026_pad_type_0"), val = string("valid")]; int32 var_9026_groups_0 = const()[name = string("op_9026_groups_0"), val = int32(1)]; tensor var_9026_strides_0 = const()[name = string("op_9026_strides_0"), val = tensor([1])]; tensor var_9026_pad_0 = const()[name = string("op_9026_pad_0"), val = tensor([0, 0])]; tensor var_9026_dilations_0 = const()[name = string("op_9026_dilations_0"), val = tensor([1])]; tensor squeeze_17_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320006912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321579840))))[name = string("squeeze_17_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9011_cast_fp16 = transpose(perm = var_9010, x = output_105_cast_fp16)[name = string("transpose_63")]; tensor var_9026_cast_fp16 = conv(dilations = var_9026_dilations_0, groups = var_9026_groups_0, pad = var_9026_pad_0, pad_type = var_9026_pad_type_0, strides = var_9026_strides_0, weight = squeeze_17_cast_fp16_to_fp32_to_fp16_palettized, x = var_9011_cast_fp16)[name = string("op_9026_cast_fp16")]; tensor var_9030 = const()[name = string("op_9030"), val = tensor([0, 2, 1])]; tensor attn_output_35_cast_fp16 = transpose(perm = var_9030, x = var_9026_cast_fp16)[name = string("transpose_62")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_171_cast_fp16, y = attn_output_35_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; int32 var_9045 = const()[name = string("op_9045"), val = int32(-1)]; fp16 const_249_promoted_to_fp16 = const()[name = string("const_249_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9047_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_249_promoted_to_fp16)[name = string("op_9047_cast_fp16")]; bool input_317_interleave_0 = const()[name = string("input_317_interleave_0"), val = bool(false)]; tensor input_317_cast_fp16 = concat(axis = var_9045, interleave = input_317_interleave_0, values = (hidden_states_179_cast_fp16, var_9047_cast_fp16))[name = string("input_317_cast_fp16")]; tensor normed_285_axes_0 = const()[name = string("normed_285_axes_0"), val = tensor([-1])]; fp16 var_9042_to_fp16 = const()[name = string("op_9042_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_285_cast_fp16 = layer_norm(axes = normed_285_axes_0, epsilon = var_9042_to_fp16, x = input_317_cast_fp16)[name = string("normed_285_cast_fp16")]; tensor normed_287_begin_0 = const()[name = string("normed_287_begin_0"), val = tensor([0, 0, 0])]; tensor normed_287_end_0 = const()[name = string("normed_287_end_0"), val = tensor([1, 1, 1024])]; tensor normed_287_end_mask_0 = const()[name = string("normed_287_end_mask_0"), val = tensor([true, true, false])]; tensor normed_287_cast_fp16 = slice_by_index(begin = normed_287_begin_0, end = normed_287_end_0, end_mask = normed_287_end_mask_0, x = normed_285_cast_fp16)[name = string("normed_287_cast_fp16")]; tensor const_251_promoted_to_fp16 = const()[name = string("const_251_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321596288)))]; tensor x_69_cast_fp16 = mul(x = normed_287_cast_fp16, y = const_251_promoted_to_fp16)[name = string("x_69_cast_fp16")]; tensor var_9067 = const()[name = string("op_9067"), val = tensor([0, 2, 1])]; tensor input_319_axes_0 = const()[name = string("input_319_axes_0"), val = tensor([2])]; tensor var_9068 = transpose(perm = var_9067, x = x_69_cast_fp16)[name = string("transpose_61")]; tensor input_319 = expand_dims(axes = input_319_axes_0, x = var_9068)[name = string("input_319")]; string input_321_pad_type_0 = const()[name = string("input_321_pad_type_0"), val = string("valid")]; tensor input_321_strides_0 = const()[name = string("input_321_strides_0"), val = tensor([1, 1])]; tensor input_321_pad_0 = const()[name = string("input_321_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_321_dilations_0 = const()[name = string("input_321_dilations_0"), val = tensor([1, 1])]; int32 input_321_groups_0 = const()[name = string("input_321_groups_0"), val = int32(1)]; tensor input_321 = conv(dilations = input_321_dilations_0, groups = input_321_groups_0, pad = input_321_pad_0, pad_type = input_321_pad_type_0, strides = input_321_strides_0, weight = model_model_layers_17_mlp_gate_proj_weight_palettized, x = input_319)[name = string("input_321")]; string b_35_pad_type_0 = const()[name = string("b_35_pad_type_0"), val = string("valid")]; tensor b_35_strides_0 = const()[name = string("b_35_strides_0"), val = tensor([1, 1])]; tensor b_35_pad_0 = const()[name = string("b_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_35_dilations_0 = const()[name = string("b_35_dilations_0"), val = tensor([1, 1])]; int32 b_35_groups_0 = const()[name = string("b_35_groups_0"), val = int32(1)]; tensor b_35 = conv(dilations = b_35_dilations_0, groups = b_35_groups_0, pad = b_35_pad_0, pad_type = b_35_pad_type_0, strides = b_35_strides_0, weight = model_model_layers_17_mlp_up_proj_weight_palettized, x = input_319)[name = string("b_35")]; tensor c_35 = silu(x = input_321)[name = string("c_35")]; tensor input_323 = mul(x = c_35, y = b_35)[name = string("input_323")]; string e_35_pad_type_0 = const()[name = string("e_35_pad_type_0"), val = string("valid")]; tensor e_35_strides_0 = const()[name = string("e_35_strides_0"), val = tensor([1, 1])]; tensor e_35_pad_0 = const()[name = string("e_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_35_dilations_0 = const()[name = string("e_35_dilations_0"), val = tensor([1, 1])]; int32 e_35_groups_0 = const()[name = string("e_35_groups_0"), val = int32(1)]; tensor e_35 = conv(dilations = e_35_dilations_0, groups = e_35_groups_0, pad = e_35_pad_0, pad_type = e_35_pad_type_0, strides = e_35_strides_0, weight = model_model_layers_17_mlp_down_proj_weight_palettized, x = input_323)[name = string("e_35")]; tensor var_9090_axes_0 = const()[name = string("op_9090_axes_0"), val = tensor([2])]; tensor var_9090 = squeeze(axes = var_9090_axes_0, x = e_35)[name = string("op_9090")]; tensor var_9091 = const()[name = string("op_9091"), val = tensor([0, 2, 1])]; tensor var_9092 = transpose(perm = var_9091, x = var_9090)[name = string("transpose_60")]; tensor hidden_states_181_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = var_9092)[name = string("hidden_states_181_cast_fp16")]; int32 var_9106 = const()[name = string("op_9106"), val = int32(-1)]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9108_cast_fp16 = mul(x = hidden_states_181_cast_fp16, y = const_252_promoted_to_fp16)[name = string("op_9108_cast_fp16")]; bool input_325_interleave_0 = const()[name = string("input_325_interleave_0"), val = bool(false)]; tensor input_325_cast_fp16 = concat(axis = var_9106, interleave = input_325_interleave_0, values = (hidden_states_181_cast_fp16, var_9108_cast_fp16))[name = string("input_325_cast_fp16")]; tensor normed_289_axes_0 = const()[name = string("normed_289_axes_0"), val = tensor([-1])]; fp16 var_9103_to_fp16 = const()[name = string("op_9103_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_289_cast_fp16 = layer_norm(axes = normed_289_axes_0, epsilon = var_9103_to_fp16, x = input_325_cast_fp16)[name = string("normed_289_cast_fp16")]; tensor normed_291_begin_0 = const()[name = string("normed_291_begin_0"), val = tensor([0, 0, 0])]; tensor normed_291_end_0 = const()[name = string("normed_291_end_0"), val = tensor([1, 1, 1024])]; tensor normed_291_end_mask_0 = const()[name = string("normed_291_end_mask_0"), val = tensor([true, true, false])]; tensor normed_291_cast_fp16 = slice_by_index(begin = normed_291_begin_0, end = normed_291_end_0, end_mask = normed_291_end_mask_0, x = normed_289_cast_fp16)[name = string("normed_291_cast_fp16")]; tensor const_254_promoted_to_fp16 = const()[name = string("const_254_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321598400)))]; tensor hidden_states_183_cast_fp16 = mul(x = normed_291_cast_fp16, y = const_254_promoted_to_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor var_9120 = const()[name = string("op_9120"), val = tensor([0, 2, 1])]; tensor var_9123_axes_0 = const()[name = string("op_9123_axes_0"), val = tensor([2])]; tensor var_9121_cast_fp16 = transpose(perm = var_9120, x = hidden_states_183_cast_fp16)[name = string("transpose_59")]; tensor var_9123_cast_fp16 = expand_dims(axes = var_9123_axes_0, x = var_9121_cast_fp16)[name = string("op_9123_cast_fp16")]; string var_9139_pad_type_0 = const()[name = string("op_9139_pad_type_0"), val = string("valid")]; tensor var_9139_strides_0 = const()[name = string("op_9139_strides_0"), val = tensor([1, 1])]; tensor var_9139_pad_0 = const()[name = string("op_9139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9139_dilations_0 = const()[name = string("op_9139_dilations_0"), val = tensor([1, 1])]; int32 var_9139_groups_0 = const()[name = string("op_9139_groups_0"), val = int32(1)]; tensor var_9139 = conv(dilations = var_9139_dilations_0, groups = var_9139_groups_0, pad = var_9139_pad_0, pad_type = var_9139_pad_type_0, strides = var_9139_strides_0, weight = model_model_layers_18_self_attn_q_proj_weight_palettized, x = var_9123_cast_fp16)[name = string("op_9139")]; tensor var_9144 = const()[name = string("op_9144"), val = tensor([1, 16, 1, 128])]; tensor var_9145 = reshape(shape = var_9144, x = var_9139)[name = string("op_9145")]; string var_9161_pad_type_0 = const()[name = string("op_9161_pad_type_0"), val = string("valid")]; tensor var_9161_strides_0 = const()[name = string("op_9161_strides_0"), val = tensor([1, 1])]; tensor var_9161_pad_0 = const()[name = string("op_9161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9161_dilations_0 = const()[name = string("op_9161_dilations_0"), val = tensor([1, 1])]; int32 var_9161_groups_0 = const()[name = string("op_9161_groups_0"), val = int32(1)]; tensor var_9161 = conv(dilations = var_9161_dilations_0, groups = var_9161_groups_0, pad = var_9161_pad_0, pad_type = var_9161_pad_type_0, strides = var_9161_strides_0, weight = model_model_layers_18_self_attn_k_proj_weight_palettized, x = var_9123_cast_fp16)[name = string("op_9161")]; tensor var_9166 = const()[name = string("op_9166"), val = tensor([1, 8, 1, 128])]; tensor var_9167 = reshape(shape = var_9166, x = var_9161)[name = string("op_9167")]; string var_9183_pad_type_0 = const()[name = string("op_9183_pad_type_0"), val = string("valid")]; tensor var_9183_strides_0 = const()[name = string("op_9183_strides_0"), val = tensor([1, 1])]; tensor var_9183_pad_0 = const()[name = string("op_9183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9183_dilations_0 = const()[name = string("op_9183_dilations_0"), val = tensor([1, 1])]; int32 var_9183_groups_0 = const()[name = string("op_9183_groups_0"), val = int32(1)]; tensor var_9183 = conv(dilations = var_9183_dilations_0, groups = var_9183_groups_0, pad = var_9183_pad_0, pad_type = var_9183_pad_type_0, strides = var_9183_strides_0, weight = model_model_layers_18_self_attn_v_proj_weight_palettized, x = var_9123_cast_fp16)[name = string("op_9183")]; tensor var_9188 = const()[name = string("op_9188"), val = tensor([1, 8, 1, 128])]; tensor var_9189 = reshape(shape = var_9188, x = var_9183)[name = string("op_9189")]; int32 var_9206 = const()[name = string("op_9206"), val = int32(-1)]; fp16 const_255_promoted = const()[name = string("const_255_promoted"), val = fp16(-0x1p+0)]; tensor var_9208 = mul(x = var_9145, y = const_255_promoted)[name = string("op_9208")]; bool input_329_interleave_0 = const()[name = string("input_329_interleave_0"), val = bool(false)]; tensor input_329 = concat(axis = var_9206, interleave = input_329_interleave_0, values = (var_9145, var_9208))[name = string("input_329")]; tensor normed_293_axes_0 = const()[name = string("normed_293_axes_0"), val = tensor([-1])]; fp16 var_9203_to_fp16 = const()[name = string("op_9203_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_293_cast_fp16 = layer_norm(axes = normed_293_axes_0, epsilon = var_9203_to_fp16, x = input_329)[name = string("normed_293_cast_fp16")]; tensor normed_295_begin_0 = const()[name = string("normed_295_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_295_end_0 = const()[name = string("normed_295_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_295_end_mask_0 = const()[name = string("normed_295_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_295 = slice_by_index(begin = normed_295_begin_0, end = normed_295_end_0, end_mask = normed_295_end_mask_0, x = normed_293_cast_fp16)[name = string("normed_295")]; tensor const_257 = const()[name = string("const_257"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321600512)))]; tensor q_37 = mul(x = normed_295, y = const_257)[name = string("q_37")]; int32 var_9228 = const()[name = string("op_9228"), val = int32(-1)]; fp16 const_258_promoted = const()[name = string("const_258_promoted"), val = fp16(-0x1p+0)]; tensor var_9230 = mul(x = var_9167, y = const_258_promoted)[name = string("op_9230")]; bool input_331_interleave_0 = const()[name = string("input_331_interleave_0"), val = bool(false)]; tensor input_331 = concat(axis = var_9228, interleave = input_331_interleave_0, values = (var_9167, var_9230))[name = string("input_331")]; tensor normed_297_axes_0 = const()[name = string("normed_297_axes_0"), val = tensor([-1])]; fp16 var_9225_to_fp16 = const()[name = string("op_9225_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_297_cast_fp16 = layer_norm(axes = normed_297_axes_0, epsilon = var_9225_to_fp16, x = input_331)[name = string("normed_297_cast_fp16")]; tensor normed_299_begin_0 = const()[name = string("normed_299_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_299_end_0 = const()[name = string("normed_299_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_299_end_mask_0 = const()[name = string("normed_299_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_299 = slice_by_index(begin = normed_299_begin_0, end = normed_299_end_0, end_mask = normed_299_end_mask_0, x = normed_297_cast_fp16)[name = string("normed_299")]; tensor const_260 = const()[name = string("const_260"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321600832)))]; tensor k_37 = mul(x = normed_299, y = const_260)[name = string("k_37")]; tensor var_9239 = mul(x = q_37, y = cos_1_cast_fp16)[name = string("op_9239")]; tensor var_9244_begin_0 = const()[name = string("op_9244_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9244_end_0 = const()[name = string("op_9244_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_9244_end_mask_0 = const()[name = string("op_9244_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9244 = slice_by_index(begin = var_9244_begin_0, end = var_9244_end_0, end_mask = var_9244_end_mask_0, x = q_37)[name = string("op_9244")]; fp16 const_261_promoted = const()[name = string("const_261_promoted"), val = fp16(-0x1p+0)]; tensor var_9245 = mul(x = var_9244, y = const_261_promoted)[name = string("op_9245")]; tensor var_9250_begin_0 = const()[name = string("op_9250_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9250_end_0 = const()[name = string("op_9250_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_9250_end_mask_0 = const()[name = string("op_9250_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9250 = slice_by_index(begin = var_9250_begin_0, end = var_9250_end_0, end_mask = var_9250_end_mask_0, x = q_37)[name = string("op_9250")]; int32 var_9252 = const()[name = string("op_9252"), val = int32(-1)]; bool var_9253_interleave_0 = const()[name = string("op_9253_interleave_0"), val = bool(false)]; tensor var_9253 = concat(axis = var_9252, interleave = var_9253_interleave_0, values = (var_9245, var_9250))[name = string("op_9253")]; tensor var_9254 = mul(x = var_9253, y = sin_1_cast_fp16)[name = string("op_9254")]; tensor query_states_73 = add(x = var_9239, y = var_9254)[name = string("query_states_73")]; tensor var_9257 = mul(x = k_37, y = cos_1_cast_fp16)[name = string("op_9257")]; tensor var_9262_begin_0 = const()[name = string("op_9262_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9262_end_0 = const()[name = string("op_9262_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_9262_end_mask_0 = const()[name = string("op_9262_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9262 = slice_by_index(begin = var_9262_begin_0, end = var_9262_end_0, end_mask = var_9262_end_mask_0, x = k_37)[name = string("op_9262")]; fp16 const_262_promoted = const()[name = string("const_262_promoted"), val = fp16(-0x1p+0)]; tensor var_9263 = mul(x = var_9262, y = const_262_promoted)[name = string("op_9263")]; tensor var_9268_begin_0 = const()[name = string("op_9268_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9268_end_0 = const()[name = string("op_9268_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_9268_end_mask_0 = const()[name = string("op_9268_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9268 = slice_by_index(begin = var_9268_begin_0, end = var_9268_end_0, end_mask = var_9268_end_mask_0, x = k_37)[name = string("op_9268")]; int32 var_9270 = const()[name = string("op_9270"), val = int32(-1)]; bool var_9271_interleave_0 = const()[name = string("op_9271_interleave_0"), val = bool(false)]; tensor var_9271 = concat(axis = var_9270, interleave = var_9271_interleave_0, values = (var_9263, var_9268))[name = string("op_9271")]; tensor var_9272 = mul(x = var_9271, y = sin_1_cast_fp16)[name = string("op_9272")]; tensor key_states_73 = add(x = var_9257, y = var_9272)[name = string("key_states_73")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; int32 concat_146_axis_0 = const()[name = string("concat_146_axis_0"), val = int32(0)]; bool concat_146_interleave_0 = const()[name = string("concat_146_interleave_0"), val = bool(false)]; tensor concat_146 = concat(axis = concat_146_axis_0, interleave = concat_146_interleave_0, values = (expand_dims_216, expand_dims_217, current_pos, expand_dims_219))[name = string("concat_146")]; tensor concat_147_values1_0 = const()[name = string("concat_147_values1_0"), val = tensor([0])]; tensor concat_147_values3_0 = const()[name = string("concat_147_values3_0"), val = tensor([0])]; int32 concat_147_axis_0 = const()[name = string("concat_147_axis_0"), val = int32(0)]; bool concat_147_interleave_0 = const()[name = string("concat_147_interleave_0"), val = bool(false)]; tensor concat_147 = concat(axis = concat_147_axis_0, interleave = concat_147_interleave_0, values = (expand_dims_220, concat_147_values1_0, var_1717, concat_147_values3_0))[name = string("concat_147")]; tensor model_model_kv_cache_0_internal_tensor_assign_37_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_cast_fp16 = slice_update(begin = concat_146, begin_mask = model_model_kv_cache_0_internal_tensor_assign_37_begin_mask_0, end = concat_147, end_mask = model_model_kv_cache_0_internal_tensor_assign_37_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_37_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_37_stride_0, update = key_states_73, x = coreml_update_state_91)[name = string("model_model_kv_cache_0_internal_tensor_assign_37_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_37_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_148_write_state")]; tensor coreml_update_state_92 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_148")]; tensor expand_dims_222 = const()[name = string("expand_dims_222"), val = tensor([46])]; tensor expand_dims_223 = const()[name = string("expand_dims_223"), val = tensor([0])]; tensor expand_dims_225 = const()[name = string("expand_dims_225"), val = tensor([0])]; tensor expand_dims_226 = const()[name = string("expand_dims_226"), val = tensor([47])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_222, expand_dims_223, current_pos, expand_dims_225))[name = string("concat_150")]; tensor concat_151_values1_0 = const()[name = string("concat_151_values1_0"), val = tensor([0])]; tensor concat_151_values3_0 = const()[name = string("concat_151_values3_0"), val = tensor([0])]; int32 concat_151_axis_0 = const()[name = string("concat_151_axis_0"), val = int32(0)]; bool concat_151_interleave_0 = const()[name = string("concat_151_interleave_0"), val = bool(false)]; tensor concat_151 = concat(axis = concat_151_axis_0, interleave = concat_151_interleave_0, values = (expand_dims_226, concat_151_values1_0, var_1717, concat_151_values3_0))[name = string("concat_151")]; tensor model_model_kv_cache_0_internal_tensor_assign_38_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_cast_fp16 = slice_update(begin = concat_150, begin_mask = model_model_kv_cache_0_internal_tensor_assign_38_begin_mask_0, end = concat_151, end_mask = model_model_kv_cache_0_internal_tensor_assign_38_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_38_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_38_stride_0, update = var_9189, x = coreml_update_state_92)[name = string("model_model_kv_cache_0_internal_tensor_assign_38_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_38_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_149_write_state")]; tensor coreml_update_state_93 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_149")]; tensor var_9327_begin_0 = const()[name = string("op_9327_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_9327_end_0 = const()[name = string("op_9327_end_0"), val = tensor([19, 8, 1536, 128])]; tensor var_9327_end_mask_0 = const()[name = string("op_9327_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9327_cast_fp16 = slice_by_index(begin = var_9327_begin_0, end = var_9327_end_0, end_mask = var_9327_end_mask_0, x = coreml_update_state_93)[name = string("op_9327_cast_fp16")]; tensor key_cache_37_axes_0 = const()[name = string("key_cache_37_axes_0"), val = tensor([0])]; tensor key_cache_37_cast_fp16 = squeeze(axes = key_cache_37_axes_0, x = var_9327_cast_fp16)[name = string("key_cache_37_cast_fp16")]; tensor var_9334_begin_0 = const()[name = string("op_9334_begin_0"), val = tensor([46, 0, 0, 0])]; tensor var_9334_end_0 = const()[name = string("op_9334_end_0"), val = tensor([47, 8, 1536, 128])]; tensor var_9334_end_mask_0 = const()[name = string("op_9334_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9334_cast_fp16 = slice_by_index(begin = var_9334_begin_0, end = var_9334_end_0, end_mask = var_9334_end_mask_0, x = coreml_update_state_93)[name = string("op_9334_cast_fp16")]; tensor value_cache_37_axes_0 = const()[name = string("value_cache_37_axes_0"), val = tensor([0])]; tensor value_cache_37_cast_fp16 = squeeze(axes = value_cache_37_axes_0, x = var_9334_cast_fp16)[name = string("value_cache_37_cast_fp16")]; tensor var_9358_axes_0 = const()[name = string("op_9358_axes_0"), val = tensor([1])]; tensor var_9358_cast_fp16 = expand_dims(axes = var_9358_axes_0, x = key_cache_37_cast_fp16)[name = string("op_9358_cast_fp16")]; tensor var_9363 = const()[name = string("op_9363"), val = tensor([1, 2, 1, 1])]; tensor value_147_cast_fp16 = tile(reps = var_9363, x = var_9358_cast_fp16)[name = string("value_147_cast_fp16")]; tensor var_9369 = const()[name = string("op_9369"), val = tensor([1, 16, 1536, 128])]; tensor key_states_75_cast_fp16 = reshape(shape = var_9369, x = value_147_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor var_9372_axes_0 = const()[name = string("op_9372_axes_0"), val = tensor([1])]; tensor var_9372_cast_fp16 = expand_dims(axes = var_9372_axes_0, x = value_cache_37_cast_fp16)[name = string("op_9372_cast_fp16")]; tensor var_9377 = const()[name = string("op_9377"), val = tensor([1, 2, 1, 1])]; tensor value_151_cast_fp16 = tile(reps = var_9377, x = var_9372_cast_fp16)[name = string("value_151_cast_fp16")]; tensor var_9383 = const()[name = string("op_9383"), val = tensor([1, 16, 1536, 128])]; tensor value_states_111_cast_fp16 = reshape(shape = var_9383, x = value_151_cast_fp16)[name = string("value_states_111_cast_fp16")]; bool var_9398_transpose_x_1 = const()[name = string("op_9398_transpose_x_1"), val = bool(false)]; bool var_9398_transpose_y_1 = const()[name = string("op_9398_transpose_y_1"), val = bool(true)]; tensor var_9398 = matmul(transpose_x = var_9398_transpose_x_1, transpose_y = var_9398_transpose_y_1, x = query_states_73, y = key_states_75_cast_fp16)[name = string("op_9398")]; fp16 var_9399_to_fp16 = const()[name = string("op_9399_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_73_cast_fp16 = mul(x = var_9398, y = var_9399_to_fp16)[name = string("attention_73_cast_fp16")]; tensor attention_75_cast_fp16 = add(x = attention_73_cast_fp16, y = causal_mask)[name = string("attention_75_cast_fp16")]; int32 var_9408 = const()[name = string("op_9408"), val = int32(-1)]; tensor probabilities_37_cast_fp16 = softmax(axis = var_9408, x = attention_75_cast_fp16)[name = string("probabilities_37_cast_fp16")]; bool output_109_transpose_x_0 = const()[name = string("output_109_transpose_x_0"), val = bool(false)]; bool output_109_transpose_y_0 = const()[name = string("output_109_transpose_y_0"), val = bool(false)]; tensor output_109_cast_fp16 = matmul(transpose_x = output_109_transpose_x_0, transpose_y = output_109_transpose_y_0, x = probabilities_37_cast_fp16, y = value_states_111_cast_fp16)[name = string("output_109_cast_fp16")]; tensor var_9419_perm_0 = const()[name = string("op_9419_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_9425 = const()[name = string("op_9425"), val = tensor([1, 1, 2048])]; tensor var_9419_cast_fp16 = transpose(perm = var_9419_perm_0, x = output_109_cast_fp16)[name = string("transpose_58")]; tensor output_111_cast_fp16 = reshape(shape = var_9425, x = var_9419_cast_fp16)[name = string("output_111_cast_fp16")]; tensor var_9430 = const()[name = string("op_9430"), val = tensor([0, 2, 1])]; string var_9446_pad_type_0 = const()[name = string("op_9446_pad_type_0"), val = string("valid")]; int32 var_9446_groups_0 = const()[name = string("op_9446_groups_0"), val = int32(1)]; tensor var_9446_strides_0 = const()[name = string("op_9446_strides_0"), val = tensor([1])]; tensor var_9446_pad_0 = const()[name = string("op_9446_pad_0"), val = tensor([0, 0])]; tensor var_9446_dilations_0 = const()[name = string("op_9446_dilations_0"), val = tensor([1])]; tensor squeeze_18_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321601152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323174080))))[name = string("squeeze_18_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9431_cast_fp16 = transpose(perm = var_9430, x = output_111_cast_fp16)[name = string("transpose_57")]; tensor var_9446_cast_fp16 = conv(dilations = var_9446_dilations_0, groups = var_9446_groups_0, pad = var_9446_pad_0, pad_type = var_9446_pad_type_0, strides = var_9446_strides_0, weight = squeeze_18_cast_fp16_to_fp32_to_fp16_palettized, x = var_9431_cast_fp16)[name = string("op_9446_cast_fp16")]; tensor var_9450 = const()[name = string("op_9450"), val = tensor([0, 2, 1])]; tensor attn_output_37_cast_fp16 = transpose(perm = var_9450, x = var_9446_cast_fp16)[name = string("transpose_56")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_181_cast_fp16, y = attn_output_37_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; int32 var_9465 = const()[name = string("op_9465"), val = int32(-1)]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9467_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_263_promoted_to_fp16)[name = string("op_9467_cast_fp16")]; bool input_335_interleave_0 = const()[name = string("input_335_interleave_0"), val = bool(false)]; tensor input_335_cast_fp16 = concat(axis = var_9465, interleave = input_335_interleave_0, values = (hidden_states_189_cast_fp16, var_9467_cast_fp16))[name = string("input_335_cast_fp16")]; tensor normed_301_axes_0 = const()[name = string("normed_301_axes_0"), val = tensor([-1])]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_301_cast_fp16 = layer_norm(axes = normed_301_axes_0, epsilon = var_9462_to_fp16, x = input_335_cast_fp16)[name = string("normed_301_cast_fp16")]; tensor normed_303_begin_0 = const()[name = string("normed_303_begin_0"), val = tensor([0, 0, 0])]; tensor normed_303_end_0 = const()[name = string("normed_303_end_0"), val = tensor([1, 1, 1024])]; tensor normed_303_end_mask_0 = const()[name = string("normed_303_end_mask_0"), val = tensor([true, true, false])]; tensor normed_303_cast_fp16 = slice_by_index(begin = normed_303_begin_0, end = normed_303_end_0, end_mask = normed_303_end_mask_0, x = normed_301_cast_fp16)[name = string("normed_303_cast_fp16")]; tensor const_265_promoted_to_fp16 = const()[name = string("const_265_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323190528)))]; tensor x_73_cast_fp16 = mul(x = normed_303_cast_fp16, y = const_265_promoted_to_fp16)[name = string("x_73_cast_fp16")]; tensor var_9487 = const()[name = string("op_9487"), val = tensor([0, 2, 1])]; tensor input_337_axes_0 = const()[name = string("input_337_axes_0"), val = tensor([2])]; tensor var_9488 = transpose(perm = var_9487, x = x_73_cast_fp16)[name = string("transpose_55")]; tensor input_337 = expand_dims(axes = input_337_axes_0, x = var_9488)[name = string("input_337")]; string input_339_pad_type_0 = const()[name = string("input_339_pad_type_0"), val = string("valid")]; tensor input_339_strides_0 = const()[name = string("input_339_strides_0"), val = tensor([1, 1])]; tensor input_339_pad_0 = const()[name = string("input_339_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_339_dilations_0 = const()[name = string("input_339_dilations_0"), val = tensor([1, 1])]; int32 input_339_groups_0 = const()[name = string("input_339_groups_0"), val = int32(1)]; tensor input_339 = conv(dilations = input_339_dilations_0, groups = input_339_groups_0, pad = input_339_pad_0, pad_type = input_339_pad_type_0, strides = input_339_strides_0, weight = model_model_layers_18_mlp_gate_proj_weight_palettized, x = input_337)[name = string("input_339")]; string b_37_pad_type_0 = const()[name = string("b_37_pad_type_0"), val = string("valid")]; tensor b_37_strides_0 = const()[name = string("b_37_strides_0"), val = tensor([1, 1])]; tensor b_37_pad_0 = const()[name = string("b_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_37_dilations_0 = const()[name = string("b_37_dilations_0"), val = tensor([1, 1])]; int32 b_37_groups_0 = const()[name = string("b_37_groups_0"), val = int32(1)]; tensor b_37 = conv(dilations = b_37_dilations_0, groups = b_37_groups_0, pad = b_37_pad_0, pad_type = b_37_pad_type_0, strides = b_37_strides_0, weight = model_model_layers_18_mlp_up_proj_weight_palettized, x = input_337)[name = string("b_37")]; tensor c_37 = silu(x = input_339)[name = string("c_37")]; tensor input_341 = mul(x = c_37, y = b_37)[name = string("input_341")]; string e_37_pad_type_0 = const()[name = string("e_37_pad_type_0"), val = string("valid")]; tensor e_37_strides_0 = const()[name = string("e_37_strides_0"), val = tensor([1, 1])]; tensor e_37_pad_0 = const()[name = string("e_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_37_dilations_0 = const()[name = string("e_37_dilations_0"), val = tensor([1, 1])]; int32 e_37_groups_0 = const()[name = string("e_37_groups_0"), val = int32(1)]; tensor e_37 = conv(dilations = e_37_dilations_0, groups = e_37_groups_0, pad = e_37_pad_0, pad_type = e_37_pad_type_0, strides = e_37_strides_0, weight = model_model_layers_18_mlp_down_proj_weight_palettized, x = input_341)[name = string("e_37")]; tensor var_9510_axes_0 = const()[name = string("op_9510_axes_0"), val = tensor([2])]; tensor var_9510 = squeeze(axes = var_9510_axes_0, x = e_37)[name = string("op_9510")]; tensor var_9511 = const()[name = string("op_9511"), val = tensor([0, 2, 1])]; tensor var_9512 = transpose(perm = var_9511, x = var_9510)[name = string("transpose_54")]; tensor hidden_states_191_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = var_9512)[name = string("hidden_states_191_cast_fp16")]; int32 var_9526 = const()[name = string("op_9526"), val = int32(-1)]; fp16 const_266_promoted_to_fp16 = const()[name = string("const_266_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9528_cast_fp16 = mul(x = hidden_states_191_cast_fp16, y = const_266_promoted_to_fp16)[name = string("op_9528_cast_fp16")]; bool input_343_interleave_0 = const()[name = string("input_343_interleave_0"), val = bool(false)]; tensor input_343_cast_fp16 = concat(axis = var_9526, interleave = input_343_interleave_0, values = (hidden_states_191_cast_fp16, var_9528_cast_fp16))[name = string("input_343_cast_fp16")]; tensor normed_305_axes_0 = const()[name = string("normed_305_axes_0"), val = tensor([-1])]; fp16 var_9523_to_fp16 = const()[name = string("op_9523_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_305_cast_fp16 = layer_norm(axes = normed_305_axes_0, epsilon = var_9523_to_fp16, x = input_343_cast_fp16)[name = string("normed_305_cast_fp16")]; tensor normed_307_begin_0 = const()[name = string("normed_307_begin_0"), val = tensor([0, 0, 0])]; tensor normed_307_end_0 = const()[name = string("normed_307_end_0"), val = tensor([1, 1, 1024])]; tensor normed_307_end_mask_0 = const()[name = string("normed_307_end_mask_0"), val = tensor([true, true, false])]; tensor normed_307_cast_fp16 = slice_by_index(begin = normed_307_begin_0, end = normed_307_end_0, end_mask = normed_307_end_mask_0, x = normed_305_cast_fp16)[name = string("normed_307_cast_fp16")]; tensor const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323192640)))]; tensor hidden_states_193_cast_fp16 = mul(x = normed_307_cast_fp16, y = const_268_promoted_to_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor var_9540 = const()[name = string("op_9540"), val = tensor([0, 2, 1])]; tensor var_9543_axes_0 = const()[name = string("op_9543_axes_0"), val = tensor([2])]; tensor var_9541_cast_fp16 = transpose(perm = var_9540, x = hidden_states_193_cast_fp16)[name = string("transpose_53")]; tensor var_9543_cast_fp16 = expand_dims(axes = var_9543_axes_0, x = var_9541_cast_fp16)[name = string("op_9543_cast_fp16")]; string var_9559_pad_type_0 = const()[name = string("op_9559_pad_type_0"), val = string("valid")]; tensor var_9559_strides_0 = const()[name = string("op_9559_strides_0"), val = tensor([1, 1])]; tensor var_9559_pad_0 = const()[name = string("op_9559_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9559_dilations_0 = const()[name = string("op_9559_dilations_0"), val = tensor([1, 1])]; int32 var_9559_groups_0 = const()[name = string("op_9559_groups_0"), val = int32(1)]; tensor var_9559 = conv(dilations = var_9559_dilations_0, groups = var_9559_groups_0, pad = var_9559_pad_0, pad_type = var_9559_pad_type_0, strides = var_9559_strides_0, weight = model_model_layers_19_self_attn_q_proj_weight_palettized, x = var_9543_cast_fp16)[name = string("op_9559")]; tensor var_9564 = const()[name = string("op_9564"), val = tensor([1, 16, 1, 128])]; tensor var_9565 = reshape(shape = var_9564, x = var_9559)[name = string("op_9565")]; string var_9581_pad_type_0 = const()[name = string("op_9581_pad_type_0"), val = string("valid")]; tensor var_9581_strides_0 = const()[name = string("op_9581_strides_0"), val = tensor([1, 1])]; tensor var_9581_pad_0 = const()[name = string("op_9581_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9581_dilations_0 = const()[name = string("op_9581_dilations_0"), val = tensor([1, 1])]; int32 var_9581_groups_0 = const()[name = string("op_9581_groups_0"), val = int32(1)]; tensor var_9581 = conv(dilations = var_9581_dilations_0, groups = var_9581_groups_0, pad = var_9581_pad_0, pad_type = var_9581_pad_type_0, strides = var_9581_strides_0, weight = model_model_layers_19_self_attn_k_proj_weight_palettized, x = var_9543_cast_fp16)[name = string("op_9581")]; tensor var_9586 = const()[name = string("op_9586"), val = tensor([1, 8, 1, 128])]; tensor var_9587 = reshape(shape = var_9586, x = var_9581)[name = string("op_9587")]; string var_9603_pad_type_0 = const()[name = string("op_9603_pad_type_0"), val = string("valid")]; tensor var_9603_strides_0 = const()[name = string("op_9603_strides_0"), val = tensor([1, 1])]; tensor var_9603_pad_0 = const()[name = string("op_9603_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9603_dilations_0 = const()[name = string("op_9603_dilations_0"), val = tensor([1, 1])]; int32 var_9603_groups_0 = const()[name = string("op_9603_groups_0"), val = int32(1)]; tensor var_9603 = conv(dilations = var_9603_dilations_0, groups = var_9603_groups_0, pad = var_9603_pad_0, pad_type = var_9603_pad_type_0, strides = var_9603_strides_0, weight = model_model_layers_19_self_attn_v_proj_weight_palettized, x = var_9543_cast_fp16)[name = string("op_9603")]; tensor var_9608 = const()[name = string("op_9608"), val = tensor([1, 8, 1, 128])]; tensor var_9609 = reshape(shape = var_9608, x = var_9603)[name = string("op_9609")]; int32 var_9626 = const()[name = string("op_9626"), val = int32(-1)]; fp16 const_269_promoted = const()[name = string("const_269_promoted"), val = fp16(-0x1p+0)]; tensor var_9628 = mul(x = var_9565, y = const_269_promoted)[name = string("op_9628")]; bool input_347_interleave_0 = const()[name = string("input_347_interleave_0"), val = bool(false)]; tensor input_347 = concat(axis = var_9626, interleave = input_347_interleave_0, values = (var_9565, var_9628))[name = string("input_347")]; tensor normed_309_axes_0 = const()[name = string("normed_309_axes_0"), val = tensor([-1])]; fp16 var_9623_to_fp16 = const()[name = string("op_9623_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_309_cast_fp16 = layer_norm(axes = normed_309_axes_0, epsilon = var_9623_to_fp16, x = input_347)[name = string("normed_309_cast_fp16")]; tensor normed_311_begin_0 = const()[name = string("normed_311_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_311_end_0 = const()[name = string("normed_311_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_311_end_mask_0 = const()[name = string("normed_311_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_311 = slice_by_index(begin = normed_311_begin_0, end = normed_311_end_0, end_mask = normed_311_end_mask_0, x = normed_309_cast_fp16)[name = string("normed_311")]; tensor const_271 = const()[name = string("const_271"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323194752)))]; tensor q_39 = mul(x = normed_311, y = const_271)[name = string("q_39")]; int32 var_9648 = const()[name = string("op_9648"), val = int32(-1)]; fp16 const_272_promoted = const()[name = string("const_272_promoted"), val = fp16(-0x1p+0)]; tensor var_9650 = mul(x = var_9587, y = const_272_promoted)[name = string("op_9650")]; bool input_349_interleave_0 = const()[name = string("input_349_interleave_0"), val = bool(false)]; tensor input_349 = concat(axis = var_9648, interleave = input_349_interleave_0, values = (var_9587, var_9650))[name = string("input_349")]; tensor normed_313_axes_0 = const()[name = string("normed_313_axes_0"), val = tensor([-1])]; fp16 var_9645_to_fp16 = const()[name = string("op_9645_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_313_cast_fp16 = layer_norm(axes = normed_313_axes_0, epsilon = var_9645_to_fp16, x = input_349)[name = string("normed_313_cast_fp16")]; tensor normed_315_begin_0 = const()[name = string("normed_315_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_315_end_0 = const()[name = string("normed_315_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_315_end_mask_0 = const()[name = string("normed_315_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_315 = slice_by_index(begin = normed_315_begin_0, end = normed_315_end_0, end_mask = normed_315_end_mask_0, x = normed_313_cast_fp16)[name = string("normed_315")]; tensor const_274 = const()[name = string("const_274"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323195072)))]; tensor k_39 = mul(x = normed_315, y = const_274)[name = string("k_39")]; tensor var_9659 = mul(x = q_39, y = cos_1_cast_fp16)[name = string("op_9659")]; tensor var_9664_begin_0 = const()[name = string("op_9664_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9664_end_0 = const()[name = string("op_9664_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_9664_end_mask_0 = const()[name = string("op_9664_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9664 = slice_by_index(begin = var_9664_begin_0, end = var_9664_end_0, end_mask = var_9664_end_mask_0, x = q_39)[name = string("op_9664")]; fp16 const_275_promoted = const()[name = string("const_275_promoted"), val = fp16(-0x1p+0)]; tensor var_9665 = mul(x = var_9664, y = const_275_promoted)[name = string("op_9665")]; tensor var_9670_begin_0 = const()[name = string("op_9670_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9670_end_0 = const()[name = string("op_9670_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_9670_end_mask_0 = const()[name = string("op_9670_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9670 = slice_by_index(begin = var_9670_begin_0, end = var_9670_end_0, end_mask = var_9670_end_mask_0, x = q_39)[name = string("op_9670")]; int32 var_9672 = const()[name = string("op_9672"), val = int32(-1)]; bool var_9673_interleave_0 = const()[name = string("op_9673_interleave_0"), val = bool(false)]; tensor var_9673 = concat(axis = var_9672, interleave = var_9673_interleave_0, values = (var_9665, var_9670))[name = string("op_9673")]; tensor var_9674 = mul(x = var_9673, y = sin_1_cast_fp16)[name = string("op_9674")]; tensor query_states_77 = add(x = var_9659, y = var_9674)[name = string("query_states_77")]; tensor var_9677 = mul(x = k_39, y = cos_1_cast_fp16)[name = string("op_9677")]; tensor var_9682_begin_0 = const()[name = string("op_9682_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9682_end_0 = const()[name = string("op_9682_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_9682_end_mask_0 = const()[name = string("op_9682_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9682 = slice_by_index(begin = var_9682_begin_0, end = var_9682_end_0, end_mask = var_9682_end_mask_0, x = k_39)[name = string("op_9682")]; fp16 const_276_promoted = const()[name = string("const_276_promoted"), val = fp16(-0x1p+0)]; tensor var_9683 = mul(x = var_9682, y = const_276_promoted)[name = string("op_9683")]; tensor var_9688_begin_0 = const()[name = string("op_9688_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9688_end_0 = const()[name = string("op_9688_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_9688_end_mask_0 = const()[name = string("op_9688_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9688 = slice_by_index(begin = var_9688_begin_0, end = var_9688_end_0, end_mask = var_9688_end_mask_0, x = k_39)[name = string("op_9688")]; int32 var_9690 = const()[name = string("op_9690"), val = int32(-1)]; bool var_9691_interleave_0 = const()[name = string("op_9691_interleave_0"), val = bool(false)]; tensor var_9691 = concat(axis = var_9690, interleave = var_9691_interleave_0, values = (var_9683, var_9688))[name = string("op_9691")]; tensor var_9692 = mul(x = var_9691, y = sin_1_cast_fp16)[name = string("op_9692")]; tensor key_states_77 = add(x = var_9677, y = var_9692)[name = string("key_states_77")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; int32 concat_154_axis_0 = const()[name = string("concat_154_axis_0"), val = int32(0)]; bool concat_154_interleave_0 = const()[name = string("concat_154_interleave_0"), val = bool(false)]; tensor concat_154 = concat(axis = concat_154_axis_0, interleave = concat_154_interleave_0, values = (expand_dims_228, expand_dims_229, current_pos, expand_dims_231))[name = string("concat_154")]; tensor concat_155_values1_0 = const()[name = string("concat_155_values1_0"), val = tensor([0])]; tensor concat_155_values3_0 = const()[name = string("concat_155_values3_0"), val = tensor([0])]; int32 concat_155_axis_0 = const()[name = string("concat_155_axis_0"), val = int32(0)]; bool concat_155_interleave_0 = const()[name = string("concat_155_interleave_0"), val = bool(false)]; tensor concat_155 = concat(axis = concat_155_axis_0, interleave = concat_155_interleave_0, values = (expand_dims_232, concat_155_values1_0, var_1717, concat_155_values3_0))[name = string("concat_155")]; tensor model_model_kv_cache_0_internal_tensor_assign_39_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_cast_fp16 = slice_update(begin = concat_154, begin_mask = model_model_kv_cache_0_internal_tensor_assign_39_begin_mask_0, end = concat_155, end_mask = model_model_kv_cache_0_internal_tensor_assign_39_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_39_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_39_stride_0, update = key_states_77, x = coreml_update_state_93)[name = string("model_model_kv_cache_0_internal_tensor_assign_39_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_39_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_150_write_state")]; tensor coreml_update_state_94 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_150")]; tensor expand_dims_234 = const()[name = string("expand_dims_234"), val = tensor([47])]; tensor expand_dims_235 = const()[name = string("expand_dims_235"), val = tensor([0])]; tensor expand_dims_237 = const()[name = string("expand_dims_237"), val = tensor([0])]; tensor expand_dims_238 = const()[name = string("expand_dims_238"), val = tensor([48])]; int32 concat_158_axis_0 = const()[name = string("concat_158_axis_0"), val = int32(0)]; bool concat_158_interleave_0 = const()[name = string("concat_158_interleave_0"), val = bool(false)]; tensor concat_158 = concat(axis = concat_158_axis_0, interleave = concat_158_interleave_0, values = (expand_dims_234, expand_dims_235, current_pos, expand_dims_237))[name = string("concat_158")]; tensor concat_159_values1_0 = const()[name = string("concat_159_values1_0"), val = tensor([0])]; tensor concat_159_values3_0 = const()[name = string("concat_159_values3_0"), val = tensor([0])]; int32 concat_159_axis_0 = const()[name = string("concat_159_axis_0"), val = int32(0)]; bool concat_159_interleave_0 = const()[name = string("concat_159_interleave_0"), val = bool(false)]; tensor concat_159 = concat(axis = concat_159_axis_0, interleave = concat_159_interleave_0, values = (expand_dims_238, concat_159_values1_0, var_1717, concat_159_values3_0))[name = string("concat_159")]; tensor model_model_kv_cache_0_internal_tensor_assign_40_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_cast_fp16 = slice_update(begin = concat_158, begin_mask = model_model_kv_cache_0_internal_tensor_assign_40_begin_mask_0, end = concat_159, end_mask = model_model_kv_cache_0_internal_tensor_assign_40_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_40_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_40_stride_0, update = var_9609, x = coreml_update_state_94)[name = string("model_model_kv_cache_0_internal_tensor_assign_40_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_40_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_151_write_state")]; tensor coreml_update_state_95 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_151")]; tensor var_9747_begin_0 = const()[name = string("op_9747_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_9747_end_0 = const()[name = string("op_9747_end_0"), val = tensor([20, 8, 1536, 128])]; tensor var_9747_end_mask_0 = const()[name = string("op_9747_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9747_cast_fp16 = slice_by_index(begin = var_9747_begin_0, end = var_9747_end_0, end_mask = var_9747_end_mask_0, x = coreml_update_state_95)[name = string("op_9747_cast_fp16")]; tensor key_cache_39_axes_0 = const()[name = string("key_cache_39_axes_0"), val = tensor([0])]; tensor key_cache_39_cast_fp16 = squeeze(axes = key_cache_39_axes_0, x = var_9747_cast_fp16)[name = string("key_cache_39_cast_fp16")]; tensor var_9754_begin_0 = const()[name = string("op_9754_begin_0"), val = tensor([47, 0, 0, 0])]; tensor var_9754_end_0 = const()[name = string("op_9754_end_0"), val = tensor([48, 8, 1536, 128])]; tensor var_9754_end_mask_0 = const()[name = string("op_9754_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9754_cast_fp16 = slice_by_index(begin = var_9754_begin_0, end = var_9754_end_0, end_mask = var_9754_end_mask_0, x = coreml_update_state_95)[name = string("op_9754_cast_fp16")]; tensor value_cache_39_axes_0 = const()[name = string("value_cache_39_axes_0"), val = tensor([0])]; tensor value_cache_39_cast_fp16 = squeeze(axes = value_cache_39_axes_0, x = var_9754_cast_fp16)[name = string("value_cache_39_cast_fp16")]; tensor var_9778_axes_0 = const()[name = string("op_9778_axes_0"), val = tensor([1])]; tensor var_9778_cast_fp16 = expand_dims(axes = var_9778_axes_0, x = key_cache_39_cast_fp16)[name = string("op_9778_cast_fp16")]; tensor var_9783 = const()[name = string("op_9783"), val = tensor([1, 2, 1, 1])]; tensor value_155_cast_fp16 = tile(reps = var_9783, x = var_9778_cast_fp16)[name = string("value_155_cast_fp16")]; tensor var_9789 = const()[name = string("op_9789"), val = tensor([1, 16, 1536, 128])]; tensor key_states_79_cast_fp16 = reshape(shape = var_9789, x = value_155_cast_fp16)[name = string("key_states_79_cast_fp16")]; tensor var_9792_axes_0 = const()[name = string("op_9792_axes_0"), val = tensor([1])]; tensor var_9792_cast_fp16 = expand_dims(axes = var_9792_axes_0, x = value_cache_39_cast_fp16)[name = string("op_9792_cast_fp16")]; tensor var_9797 = const()[name = string("op_9797"), val = tensor([1, 2, 1, 1])]; tensor value_159_cast_fp16 = tile(reps = var_9797, x = var_9792_cast_fp16)[name = string("value_159_cast_fp16")]; tensor var_9803 = const()[name = string("op_9803"), val = tensor([1, 16, 1536, 128])]; tensor value_states_117_cast_fp16 = reshape(shape = var_9803, x = value_159_cast_fp16)[name = string("value_states_117_cast_fp16")]; bool var_9818_transpose_x_1 = const()[name = string("op_9818_transpose_x_1"), val = bool(false)]; bool var_9818_transpose_y_1 = const()[name = string("op_9818_transpose_y_1"), val = bool(true)]; tensor var_9818 = matmul(transpose_x = var_9818_transpose_x_1, transpose_y = var_9818_transpose_y_1, x = query_states_77, y = key_states_79_cast_fp16)[name = string("op_9818")]; fp16 var_9819_to_fp16 = const()[name = string("op_9819_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_77_cast_fp16 = mul(x = var_9818, y = var_9819_to_fp16)[name = string("attention_77_cast_fp16")]; tensor attention_79_cast_fp16 = add(x = attention_77_cast_fp16, y = causal_mask)[name = string("attention_79_cast_fp16")]; int32 var_9828 = const()[name = string("op_9828"), val = int32(-1)]; tensor probabilities_39_cast_fp16 = softmax(axis = var_9828, x = attention_79_cast_fp16)[name = string("probabilities_39_cast_fp16")]; bool output_115_transpose_x_0 = const()[name = string("output_115_transpose_x_0"), val = bool(false)]; bool output_115_transpose_y_0 = const()[name = string("output_115_transpose_y_0"), val = bool(false)]; tensor output_115_cast_fp16 = matmul(transpose_x = output_115_transpose_x_0, transpose_y = output_115_transpose_y_0, x = probabilities_39_cast_fp16, y = value_states_117_cast_fp16)[name = string("output_115_cast_fp16")]; tensor var_9839_perm_0 = const()[name = string("op_9839_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_9845 = const()[name = string("op_9845"), val = tensor([1, 1, 2048])]; tensor var_9839_cast_fp16 = transpose(perm = var_9839_perm_0, x = output_115_cast_fp16)[name = string("transpose_52")]; tensor output_117_cast_fp16 = reshape(shape = var_9845, x = var_9839_cast_fp16)[name = string("output_117_cast_fp16")]; tensor var_9850 = const()[name = string("op_9850"), val = tensor([0, 2, 1])]; string var_9866_pad_type_0 = const()[name = string("op_9866_pad_type_0"), val = string("valid")]; int32 var_9866_groups_0 = const()[name = string("op_9866_groups_0"), val = int32(1)]; tensor var_9866_strides_0 = const()[name = string("op_9866_strides_0"), val = tensor([1])]; tensor var_9866_pad_0 = const()[name = string("op_9866_pad_0"), val = tensor([0, 0])]; tensor var_9866_dilations_0 = const()[name = string("op_9866_dilations_0"), val = tensor([1])]; tensor squeeze_19_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323195392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324768320))))[name = string("squeeze_19_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9851_cast_fp16 = transpose(perm = var_9850, x = output_117_cast_fp16)[name = string("transpose_51")]; tensor var_9866_cast_fp16 = conv(dilations = var_9866_dilations_0, groups = var_9866_groups_0, pad = var_9866_pad_0, pad_type = var_9866_pad_type_0, strides = var_9866_strides_0, weight = squeeze_19_cast_fp16_to_fp32_to_fp16_palettized, x = var_9851_cast_fp16)[name = string("op_9866_cast_fp16")]; tensor var_9870 = const()[name = string("op_9870"), val = tensor([0, 2, 1])]; tensor attn_output_39_cast_fp16 = transpose(perm = var_9870, x = var_9866_cast_fp16)[name = string("transpose_50")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_191_cast_fp16, y = attn_output_39_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; int32 var_9885 = const()[name = string("op_9885"), val = int32(-1)]; fp16 const_277_promoted_to_fp16 = const()[name = string("const_277_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9887_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_277_promoted_to_fp16)[name = string("op_9887_cast_fp16")]; bool input_353_interleave_0 = const()[name = string("input_353_interleave_0"), val = bool(false)]; tensor input_353_cast_fp16 = concat(axis = var_9885, interleave = input_353_interleave_0, values = (hidden_states_199_cast_fp16, var_9887_cast_fp16))[name = string("input_353_cast_fp16")]; tensor normed_317_axes_0 = const()[name = string("normed_317_axes_0"), val = tensor([-1])]; fp16 var_9882_to_fp16 = const()[name = string("op_9882_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_317_cast_fp16 = layer_norm(axes = normed_317_axes_0, epsilon = var_9882_to_fp16, x = input_353_cast_fp16)[name = string("normed_317_cast_fp16")]; tensor normed_319_begin_0 = const()[name = string("normed_319_begin_0"), val = tensor([0, 0, 0])]; tensor normed_319_end_0 = const()[name = string("normed_319_end_0"), val = tensor([1, 1, 1024])]; tensor normed_319_end_mask_0 = const()[name = string("normed_319_end_mask_0"), val = tensor([true, true, false])]; tensor normed_319_cast_fp16 = slice_by_index(begin = normed_319_begin_0, end = normed_319_end_0, end_mask = normed_319_end_mask_0, x = normed_317_cast_fp16)[name = string("normed_319_cast_fp16")]; tensor const_279_promoted_to_fp16 = const()[name = string("const_279_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324784768)))]; tensor x_77_cast_fp16 = mul(x = normed_319_cast_fp16, y = const_279_promoted_to_fp16)[name = string("x_77_cast_fp16")]; tensor var_9907 = const()[name = string("op_9907"), val = tensor([0, 2, 1])]; tensor input_355_axes_0 = const()[name = string("input_355_axes_0"), val = tensor([2])]; tensor var_9908 = transpose(perm = var_9907, x = x_77_cast_fp16)[name = string("transpose_49")]; tensor input_355 = expand_dims(axes = input_355_axes_0, x = var_9908)[name = string("input_355")]; string input_357_pad_type_0 = const()[name = string("input_357_pad_type_0"), val = string("valid")]; tensor input_357_strides_0 = const()[name = string("input_357_strides_0"), val = tensor([1, 1])]; tensor input_357_pad_0 = const()[name = string("input_357_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_357_dilations_0 = const()[name = string("input_357_dilations_0"), val = tensor([1, 1])]; int32 input_357_groups_0 = const()[name = string("input_357_groups_0"), val = int32(1)]; tensor input_357 = conv(dilations = input_357_dilations_0, groups = input_357_groups_0, pad = input_357_pad_0, pad_type = input_357_pad_type_0, strides = input_357_strides_0, weight = model_model_layers_19_mlp_gate_proj_weight_palettized, x = input_355)[name = string("input_357")]; string b_39_pad_type_0 = const()[name = string("b_39_pad_type_0"), val = string("valid")]; tensor b_39_strides_0 = const()[name = string("b_39_strides_0"), val = tensor([1, 1])]; tensor b_39_pad_0 = const()[name = string("b_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_39_dilations_0 = const()[name = string("b_39_dilations_0"), val = tensor([1, 1])]; int32 b_39_groups_0 = const()[name = string("b_39_groups_0"), val = int32(1)]; tensor b_39 = conv(dilations = b_39_dilations_0, groups = b_39_groups_0, pad = b_39_pad_0, pad_type = b_39_pad_type_0, strides = b_39_strides_0, weight = model_model_layers_19_mlp_up_proj_weight_palettized, x = input_355)[name = string("b_39")]; tensor c_39 = silu(x = input_357)[name = string("c_39")]; tensor input_359 = mul(x = c_39, y = b_39)[name = string("input_359")]; string e_39_pad_type_0 = const()[name = string("e_39_pad_type_0"), val = string("valid")]; tensor e_39_strides_0 = const()[name = string("e_39_strides_0"), val = tensor([1, 1])]; tensor e_39_pad_0 = const()[name = string("e_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_39_dilations_0 = const()[name = string("e_39_dilations_0"), val = tensor([1, 1])]; int32 e_39_groups_0 = const()[name = string("e_39_groups_0"), val = int32(1)]; tensor e_39 = conv(dilations = e_39_dilations_0, groups = e_39_groups_0, pad = e_39_pad_0, pad_type = e_39_pad_type_0, strides = e_39_strides_0, weight = model_model_layers_19_mlp_down_proj_weight_palettized, x = input_359)[name = string("e_39")]; tensor var_9930_axes_0 = const()[name = string("op_9930_axes_0"), val = tensor([2])]; tensor var_9930 = squeeze(axes = var_9930_axes_0, x = e_39)[name = string("op_9930")]; tensor var_9931 = const()[name = string("op_9931"), val = tensor([0, 2, 1])]; tensor var_9932 = transpose(perm = var_9931, x = var_9930)[name = string("transpose_48")]; tensor hidden_states_201_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = var_9932)[name = string("hidden_states_201_cast_fp16")]; int32 var_9946 = const()[name = string("op_9946"), val = int32(-1)]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9948_cast_fp16 = mul(x = hidden_states_201_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9948_cast_fp16")]; bool input_361_interleave_0 = const()[name = string("input_361_interleave_0"), val = bool(false)]; tensor input_361_cast_fp16 = concat(axis = var_9946, interleave = input_361_interleave_0, values = (hidden_states_201_cast_fp16, var_9948_cast_fp16))[name = string("input_361_cast_fp16")]; tensor normed_321_axes_0 = const()[name = string("normed_321_axes_0"), val = tensor([-1])]; fp16 var_9943_to_fp16 = const()[name = string("op_9943_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_321_cast_fp16 = layer_norm(axes = normed_321_axes_0, epsilon = var_9943_to_fp16, x = input_361_cast_fp16)[name = string("normed_321_cast_fp16")]; tensor normed_323_begin_0 = const()[name = string("normed_323_begin_0"), val = tensor([0, 0, 0])]; tensor normed_323_end_0 = const()[name = string("normed_323_end_0"), val = tensor([1, 1, 1024])]; tensor normed_323_end_mask_0 = const()[name = string("normed_323_end_mask_0"), val = tensor([true, true, false])]; tensor normed_323_cast_fp16 = slice_by_index(begin = normed_323_begin_0, end = normed_323_end_0, end_mask = normed_323_end_mask_0, x = normed_321_cast_fp16)[name = string("normed_323_cast_fp16")]; tensor const_282_promoted_to_fp16 = const()[name = string("const_282_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324786880)))]; tensor hidden_states_203_cast_fp16 = mul(x = normed_323_cast_fp16, y = const_282_promoted_to_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor var_9960 = const()[name = string("op_9960"), val = tensor([0, 2, 1])]; tensor var_9963_axes_0 = const()[name = string("op_9963_axes_0"), val = tensor([2])]; tensor var_9961_cast_fp16 = transpose(perm = var_9960, x = hidden_states_203_cast_fp16)[name = string("transpose_47")]; tensor var_9963_cast_fp16 = expand_dims(axes = var_9963_axes_0, x = var_9961_cast_fp16)[name = string("op_9963_cast_fp16")]; string var_9979_pad_type_0 = const()[name = string("op_9979_pad_type_0"), val = string("valid")]; tensor var_9979_strides_0 = const()[name = string("op_9979_strides_0"), val = tensor([1, 1])]; tensor var_9979_pad_0 = const()[name = string("op_9979_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9979_dilations_0 = const()[name = string("op_9979_dilations_0"), val = tensor([1, 1])]; int32 var_9979_groups_0 = const()[name = string("op_9979_groups_0"), val = int32(1)]; tensor var_9979 = conv(dilations = var_9979_dilations_0, groups = var_9979_groups_0, pad = var_9979_pad_0, pad_type = var_9979_pad_type_0, strides = var_9979_strides_0, weight = model_model_layers_20_self_attn_q_proj_weight_palettized, x = var_9963_cast_fp16)[name = string("op_9979")]; tensor var_9984 = const()[name = string("op_9984"), val = tensor([1, 16, 1, 128])]; tensor var_9985 = reshape(shape = var_9984, x = var_9979)[name = string("op_9985")]; string var_10001_pad_type_0 = const()[name = string("op_10001_pad_type_0"), val = string("valid")]; tensor var_10001_strides_0 = const()[name = string("op_10001_strides_0"), val = tensor([1, 1])]; tensor var_10001_pad_0 = const()[name = string("op_10001_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10001_dilations_0 = const()[name = string("op_10001_dilations_0"), val = tensor([1, 1])]; int32 var_10001_groups_0 = const()[name = string("op_10001_groups_0"), val = int32(1)]; tensor var_10001 = conv(dilations = var_10001_dilations_0, groups = var_10001_groups_0, pad = var_10001_pad_0, pad_type = var_10001_pad_type_0, strides = var_10001_strides_0, weight = model_model_layers_20_self_attn_k_proj_weight_palettized, x = var_9963_cast_fp16)[name = string("op_10001")]; tensor var_10006 = const()[name = string("op_10006"), val = tensor([1, 8, 1, 128])]; tensor var_10007 = reshape(shape = var_10006, x = var_10001)[name = string("op_10007")]; string var_10023_pad_type_0 = const()[name = string("op_10023_pad_type_0"), val = string("valid")]; tensor var_10023_strides_0 = const()[name = string("op_10023_strides_0"), val = tensor([1, 1])]; tensor var_10023_pad_0 = const()[name = string("op_10023_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10023_dilations_0 = const()[name = string("op_10023_dilations_0"), val = tensor([1, 1])]; int32 var_10023_groups_0 = const()[name = string("op_10023_groups_0"), val = int32(1)]; tensor var_10023 = conv(dilations = var_10023_dilations_0, groups = var_10023_groups_0, pad = var_10023_pad_0, pad_type = var_10023_pad_type_0, strides = var_10023_strides_0, weight = model_model_layers_20_self_attn_v_proj_weight_palettized, x = var_9963_cast_fp16)[name = string("op_10023")]; tensor var_10028 = const()[name = string("op_10028"), val = tensor([1, 8, 1, 128])]; tensor var_10029 = reshape(shape = var_10028, x = var_10023)[name = string("op_10029")]; int32 var_10046 = const()[name = string("op_10046"), val = int32(-1)]; fp16 const_283_promoted = const()[name = string("const_283_promoted"), val = fp16(-0x1p+0)]; tensor var_10048 = mul(x = var_9985, y = const_283_promoted)[name = string("op_10048")]; bool input_365_interleave_0 = const()[name = string("input_365_interleave_0"), val = bool(false)]; tensor input_365 = concat(axis = var_10046, interleave = input_365_interleave_0, values = (var_9985, var_10048))[name = string("input_365")]; tensor normed_325_axes_0 = const()[name = string("normed_325_axes_0"), val = tensor([-1])]; fp16 var_10043_to_fp16 = const()[name = string("op_10043_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_325_cast_fp16 = layer_norm(axes = normed_325_axes_0, epsilon = var_10043_to_fp16, x = input_365)[name = string("normed_325_cast_fp16")]; tensor normed_327_begin_0 = const()[name = string("normed_327_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_327_end_0 = const()[name = string("normed_327_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_327_end_mask_0 = const()[name = string("normed_327_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_327 = slice_by_index(begin = normed_327_begin_0, end = normed_327_end_0, end_mask = normed_327_end_mask_0, x = normed_325_cast_fp16)[name = string("normed_327")]; tensor const_285 = const()[name = string("const_285"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324788992)))]; tensor q_41 = mul(x = normed_327, y = const_285)[name = string("q_41")]; int32 var_10068 = const()[name = string("op_10068"), val = int32(-1)]; fp16 const_286_promoted = const()[name = string("const_286_promoted"), val = fp16(-0x1p+0)]; tensor var_10070 = mul(x = var_10007, y = const_286_promoted)[name = string("op_10070")]; bool input_367_interleave_0 = const()[name = string("input_367_interleave_0"), val = bool(false)]; tensor input_367 = concat(axis = var_10068, interleave = input_367_interleave_0, values = (var_10007, var_10070))[name = string("input_367")]; tensor normed_329_axes_0 = const()[name = string("normed_329_axes_0"), val = tensor([-1])]; fp16 var_10065_to_fp16 = const()[name = string("op_10065_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_329_cast_fp16 = layer_norm(axes = normed_329_axes_0, epsilon = var_10065_to_fp16, x = input_367)[name = string("normed_329_cast_fp16")]; tensor normed_331_begin_0 = const()[name = string("normed_331_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_331_end_0 = const()[name = string("normed_331_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_331_end_mask_0 = const()[name = string("normed_331_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_331 = slice_by_index(begin = normed_331_begin_0, end = normed_331_end_0, end_mask = normed_331_end_mask_0, x = normed_329_cast_fp16)[name = string("normed_331")]; tensor const_288 = const()[name = string("const_288"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324789312)))]; tensor k_41 = mul(x = normed_331, y = const_288)[name = string("k_41")]; tensor var_10079 = mul(x = q_41, y = cos_1_cast_fp16)[name = string("op_10079")]; tensor var_10084_begin_0 = const()[name = string("op_10084_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10084_end_0 = const()[name = string("op_10084_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_10084_end_mask_0 = const()[name = string("op_10084_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10084 = slice_by_index(begin = var_10084_begin_0, end = var_10084_end_0, end_mask = var_10084_end_mask_0, x = q_41)[name = string("op_10084")]; fp16 const_289_promoted = const()[name = string("const_289_promoted"), val = fp16(-0x1p+0)]; tensor var_10085 = mul(x = var_10084, y = const_289_promoted)[name = string("op_10085")]; tensor var_10090_begin_0 = const()[name = string("op_10090_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10090_end_0 = const()[name = string("op_10090_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_10090_end_mask_0 = const()[name = string("op_10090_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10090 = slice_by_index(begin = var_10090_begin_0, end = var_10090_end_0, end_mask = var_10090_end_mask_0, x = q_41)[name = string("op_10090")]; int32 var_10092 = const()[name = string("op_10092"), val = int32(-1)]; bool var_10093_interleave_0 = const()[name = string("op_10093_interleave_0"), val = bool(false)]; tensor var_10093 = concat(axis = var_10092, interleave = var_10093_interleave_0, values = (var_10085, var_10090))[name = string("op_10093")]; tensor var_10094 = mul(x = var_10093, y = sin_1_cast_fp16)[name = string("op_10094")]; tensor query_states_81 = add(x = var_10079, y = var_10094)[name = string("query_states_81")]; tensor var_10097 = mul(x = k_41, y = cos_1_cast_fp16)[name = string("op_10097")]; tensor var_10102_begin_0 = const()[name = string("op_10102_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10102_end_0 = const()[name = string("op_10102_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_10102_end_mask_0 = const()[name = string("op_10102_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10102 = slice_by_index(begin = var_10102_begin_0, end = var_10102_end_0, end_mask = var_10102_end_mask_0, x = k_41)[name = string("op_10102")]; fp16 const_290_promoted = const()[name = string("const_290_promoted"), val = fp16(-0x1p+0)]; tensor var_10103 = mul(x = var_10102, y = const_290_promoted)[name = string("op_10103")]; tensor var_10108_begin_0 = const()[name = string("op_10108_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10108_end_0 = const()[name = string("op_10108_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_10108_end_mask_0 = const()[name = string("op_10108_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10108 = slice_by_index(begin = var_10108_begin_0, end = var_10108_end_0, end_mask = var_10108_end_mask_0, x = k_41)[name = string("op_10108")]; int32 var_10110 = const()[name = string("op_10110"), val = int32(-1)]; bool var_10111_interleave_0 = const()[name = string("op_10111_interleave_0"), val = bool(false)]; tensor var_10111 = concat(axis = var_10110, interleave = var_10111_interleave_0, values = (var_10103, var_10108))[name = string("op_10111")]; tensor var_10112 = mul(x = var_10111, y = sin_1_cast_fp16)[name = string("op_10112")]; tensor key_states_81 = add(x = var_10097, y = var_10112)[name = string("key_states_81")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_240, expand_dims_241, current_pos, expand_dims_243))[name = string("concat_162")]; tensor concat_163_values1_0 = const()[name = string("concat_163_values1_0"), val = tensor([0])]; tensor concat_163_values3_0 = const()[name = string("concat_163_values3_0"), val = tensor([0])]; int32 concat_163_axis_0 = const()[name = string("concat_163_axis_0"), val = int32(0)]; bool concat_163_interleave_0 = const()[name = string("concat_163_interleave_0"), val = bool(false)]; tensor concat_163 = concat(axis = concat_163_axis_0, interleave = concat_163_interleave_0, values = (expand_dims_244, concat_163_values1_0, var_1717, concat_163_values3_0))[name = string("concat_163")]; tensor model_model_kv_cache_0_internal_tensor_assign_41_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_cast_fp16 = slice_update(begin = concat_162, begin_mask = model_model_kv_cache_0_internal_tensor_assign_41_begin_mask_0, end = concat_163, end_mask = model_model_kv_cache_0_internal_tensor_assign_41_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_41_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_41_stride_0, update = key_states_81, x = coreml_update_state_95)[name = string("model_model_kv_cache_0_internal_tensor_assign_41_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_41_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_152_write_state")]; tensor coreml_update_state_96 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_152")]; tensor expand_dims_246 = const()[name = string("expand_dims_246"), val = tensor([48])]; tensor expand_dims_247 = const()[name = string("expand_dims_247"), val = tensor([0])]; tensor expand_dims_249 = const()[name = string("expand_dims_249"), val = tensor([0])]; tensor expand_dims_250 = const()[name = string("expand_dims_250"), val = tensor([49])]; int32 concat_166_axis_0 = const()[name = string("concat_166_axis_0"), val = int32(0)]; bool concat_166_interleave_0 = const()[name = string("concat_166_interleave_0"), val = bool(false)]; tensor concat_166 = concat(axis = concat_166_axis_0, interleave = concat_166_interleave_0, values = (expand_dims_246, expand_dims_247, current_pos, expand_dims_249))[name = string("concat_166")]; tensor concat_167_values1_0 = const()[name = string("concat_167_values1_0"), val = tensor([0])]; tensor concat_167_values3_0 = const()[name = string("concat_167_values3_0"), val = tensor([0])]; int32 concat_167_axis_0 = const()[name = string("concat_167_axis_0"), val = int32(0)]; bool concat_167_interleave_0 = const()[name = string("concat_167_interleave_0"), val = bool(false)]; tensor concat_167 = concat(axis = concat_167_axis_0, interleave = concat_167_interleave_0, values = (expand_dims_250, concat_167_values1_0, var_1717, concat_167_values3_0))[name = string("concat_167")]; tensor model_model_kv_cache_0_internal_tensor_assign_42_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_cast_fp16 = slice_update(begin = concat_166, begin_mask = model_model_kv_cache_0_internal_tensor_assign_42_begin_mask_0, end = concat_167, end_mask = model_model_kv_cache_0_internal_tensor_assign_42_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_42_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_42_stride_0, update = var_10029, x = coreml_update_state_96)[name = string("model_model_kv_cache_0_internal_tensor_assign_42_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_42_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_153_write_state")]; tensor coreml_update_state_97 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_153")]; tensor var_10167_begin_0 = const()[name = string("op_10167_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_10167_end_0 = const()[name = string("op_10167_end_0"), val = tensor([21, 8, 1536, 128])]; tensor var_10167_end_mask_0 = const()[name = string("op_10167_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10167_cast_fp16 = slice_by_index(begin = var_10167_begin_0, end = var_10167_end_0, end_mask = var_10167_end_mask_0, x = coreml_update_state_97)[name = string("op_10167_cast_fp16")]; tensor key_cache_41_axes_0 = const()[name = string("key_cache_41_axes_0"), val = tensor([0])]; tensor key_cache_41_cast_fp16 = squeeze(axes = key_cache_41_axes_0, x = var_10167_cast_fp16)[name = string("key_cache_41_cast_fp16")]; tensor var_10174_begin_0 = const()[name = string("op_10174_begin_0"), val = tensor([48, 0, 0, 0])]; tensor var_10174_end_0 = const()[name = string("op_10174_end_0"), val = tensor([49, 8, 1536, 128])]; tensor var_10174_end_mask_0 = const()[name = string("op_10174_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10174_cast_fp16 = slice_by_index(begin = var_10174_begin_0, end = var_10174_end_0, end_mask = var_10174_end_mask_0, x = coreml_update_state_97)[name = string("op_10174_cast_fp16")]; tensor value_cache_41_axes_0 = const()[name = string("value_cache_41_axes_0"), val = tensor([0])]; tensor value_cache_41_cast_fp16 = squeeze(axes = value_cache_41_axes_0, x = var_10174_cast_fp16)[name = string("value_cache_41_cast_fp16")]; tensor var_10198_axes_0 = const()[name = string("op_10198_axes_0"), val = tensor([1])]; tensor var_10198_cast_fp16 = expand_dims(axes = var_10198_axes_0, x = key_cache_41_cast_fp16)[name = string("op_10198_cast_fp16")]; tensor var_10203 = const()[name = string("op_10203"), val = tensor([1, 2, 1, 1])]; tensor value_163_cast_fp16 = tile(reps = var_10203, x = var_10198_cast_fp16)[name = string("value_163_cast_fp16")]; tensor var_10209 = const()[name = string("op_10209"), val = tensor([1, 16, 1536, 128])]; tensor key_states_83_cast_fp16 = reshape(shape = var_10209, x = value_163_cast_fp16)[name = string("key_states_83_cast_fp16")]; tensor var_10212_axes_0 = const()[name = string("op_10212_axes_0"), val = tensor([1])]; tensor var_10212_cast_fp16 = expand_dims(axes = var_10212_axes_0, x = value_cache_41_cast_fp16)[name = string("op_10212_cast_fp16")]; tensor var_10217 = const()[name = string("op_10217"), val = tensor([1, 2, 1, 1])]; tensor value_167_cast_fp16 = tile(reps = var_10217, x = var_10212_cast_fp16)[name = string("value_167_cast_fp16")]; tensor var_10223 = const()[name = string("op_10223"), val = tensor([1, 16, 1536, 128])]; tensor value_states_123_cast_fp16 = reshape(shape = var_10223, x = value_167_cast_fp16)[name = string("value_states_123_cast_fp16")]; bool var_10238_transpose_x_1 = const()[name = string("op_10238_transpose_x_1"), val = bool(false)]; bool var_10238_transpose_y_1 = const()[name = string("op_10238_transpose_y_1"), val = bool(true)]; tensor var_10238 = matmul(transpose_x = var_10238_transpose_x_1, transpose_y = var_10238_transpose_y_1, x = query_states_81, y = key_states_83_cast_fp16)[name = string("op_10238")]; fp16 var_10239_to_fp16 = const()[name = string("op_10239_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_81_cast_fp16 = mul(x = var_10238, y = var_10239_to_fp16)[name = string("attention_81_cast_fp16")]; tensor attention_83_cast_fp16 = add(x = attention_81_cast_fp16, y = causal_mask)[name = string("attention_83_cast_fp16")]; int32 var_10248 = const()[name = string("op_10248"), val = int32(-1)]; tensor probabilities_41_cast_fp16 = softmax(axis = var_10248, x = attention_83_cast_fp16)[name = string("probabilities_41_cast_fp16")]; bool output_121_transpose_x_0 = const()[name = string("output_121_transpose_x_0"), val = bool(false)]; bool output_121_transpose_y_0 = const()[name = string("output_121_transpose_y_0"), val = bool(false)]; tensor output_121_cast_fp16 = matmul(transpose_x = output_121_transpose_x_0, transpose_y = output_121_transpose_y_0, x = probabilities_41_cast_fp16, y = value_states_123_cast_fp16)[name = string("output_121_cast_fp16")]; tensor var_10259_perm_0 = const()[name = string("op_10259_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_10265 = const()[name = string("op_10265"), val = tensor([1, 1, 2048])]; tensor var_10259_cast_fp16 = transpose(perm = var_10259_perm_0, x = output_121_cast_fp16)[name = string("transpose_46")]; tensor output_123_cast_fp16 = reshape(shape = var_10265, x = var_10259_cast_fp16)[name = string("output_123_cast_fp16")]; tensor var_10270 = const()[name = string("op_10270"), val = tensor([0, 2, 1])]; string var_10286_pad_type_0 = const()[name = string("op_10286_pad_type_0"), val = string("valid")]; int32 var_10286_groups_0 = const()[name = string("op_10286_groups_0"), val = int32(1)]; tensor var_10286_strides_0 = const()[name = string("op_10286_strides_0"), val = tensor([1])]; tensor var_10286_pad_0 = const()[name = string("op_10286_pad_0"), val = tensor([0, 0])]; tensor var_10286_dilations_0 = const()[name = string("op_10286_dilations_0"), val = tensor([1])]; tensor squeeze_20_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324789632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326362560))))[name = string("squeeze_20_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10271_cast_fp16 = transpose(perm = var_10270, x = output_123_cast_fp16)[name = string("transpose_45")]; tensor var_10286_cast_fp16 = conv(dilations = var_10286_dilations_0, groups = var_10286_groups_0, pad = var_10286_pad_0, pad_type = var_10286_pad_type_0, strides = var_10286_strides_0, weight = squeeze_20_cast_fp16_to_fp32_to_fp16_palettized, x = var_10271_cast_fp16)[name = string("op_10286_cast_fp16")]; tensor var_10290 = const()[name = string("op_10290"), val = tensor([0, 2, 1])]; tensor attn_output_41_cast_fp16 = transpose(perm = var_10290, x = var_10286_cast_fp16)[name = string("transpose_44")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_201_cast_fp16, y = attn_output_41_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; int32 var_10305 = const()[name = string("op_10305"), val = int32(-1)]; fp16 const_291_promoted_to_fp16 = const()[name = string("const_291_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10307_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_291_promoted_to_fp16)[name = string("op_10307_cast_fp16")]; bool input_371_interleave_0 = const()[name = string("input_371_interleave_0"), val = bool(false)]; tensor input_371_cast_fp16 = concat(axis = var_10305, interleave = input_371_interleave_0, values = (hidden_states_209_cast_fp16, var_10307_cast_fp16))[name = string("input_371_cast_fp16")]; tensor normed_333_axes_0 = const()[name = string("normed_333_axes_0"), val = tensor([-1])]; fp16 var_10302_to_fp16 = const()[name = string("op_10302_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_333_cast_fp16 = layer_norm(axes = normed_333_axes_0, epsilon = var_10302_to_fp16, x = input_371_cast_fp16)[name = string("normed_333_cast_fp16")]; tensor normed_335_begin_0 = const()[name = string("normed_335_begin_0"), val = tensor([0, 0, 0])]; tensor normed_335_end_0 = const()[name = string("normed_335_end_0"), val = tensor([1, 1, 1024])]; tensor normed_335_end_mask_0 = const()[name = string("normed_335_end_mask_0"), val = tensor([true, true, false])]; tensor normed_335_cast_fp16 = slice_by_index(begin = normed_335_begin_0, end = normed_335_end_0, end_mask = normed_335_end_mask_0, x = normed_333_cast_fp16)[name = string("normed_335_cast_fp16")]; tensor const_293_promoted_to_fp16 = const()[name = string("const_293_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326379008)))]; tensor x_81_cast_fp16 = mul(x = normed_335_cast_fp16, y = const_293_promoted_to_fp16)[name = string("x_81_cast_fp16")]; tensor var_10327 = const()[name = string("op_10327"), val = tensor([0, 2, 1])]; tensor input_373_axes_0 = const()[name = string("input_373_axes_0"), val = tensor([2])]; tensor var_10328 = transpose(perm = var_10327, x = x_81_cast_fp16)[name = string("transpose_43")]; tensor input_373 = expand_dims(axes = input_373_axes_0, x = var_10328)[name = string("input_373")]; string input_375_pad_type_0 = const()[name = string("input_375_pad_type_0"), val = string("valid")]; tensor input_375_strides_0 = const()[name = string("input_375_strides_0"), val = tensor([1, 1])]; tensor input_375_pad_0 = const()[name = string("input_375_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_375_dilations_0 = const()[name = string("input_375_dilations_0"), val = tensor([1, 1])]; int32 input_375_groups_0 = const()[name = string("input_375_groups_0"), val = int32(1)]; tensor input_375 = conv(dilations = input_375_dilations_0, groups = input_375_groups_0, pad = input_375_pad_0, pad_type = input_375_pad_type_0, strides = input_375_strides_0, weight = model_model_layers_20_mlp_gate_proj_weight_palettized, x = input_373)[name = string("input_375")]; string b_41_pad_type_0 = const()[name = string("b_41_pad_type_0"), val = string("valid")]; tensor b_41_strides_0 = const()[name = string("b_41_strides_0"), val = tensor([1, 1])]; tensor b_41_pad_0 = const()[name = string("b_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_41_dilations_0 = const()[name = string("b_41_dilations_0"), val = tensor([1, 1])]; int32 b_41_groups_0 = const()[name = string("b_41_groups_0"), val = int32(1)]; tensor b_41 = conv(dilations = b_41_dilations_0, groups = b_41_groups_0, pad = b_41_pad_0, pad_type = b_41_pad_type_0, strides = b_41_strides_0, weight = model_model_layers_20_mlp_up_proj_weight_palettized, x = input_373)[name = string("b_41")]; tensor c_41 = silu(x = input_375)[name = string("c_41")]; tensor input_377 = mul(x = c_41, y = b_41)[name = string("input_377")]; string e_41_pad_type_0 = const()[name = string("e_41_pad_type_0"), val = string("valid")]; tensor e_41_strides_0 = const()[name = string("e_41_strides_0"), val = tensor([1, 1])]; tensor e_41_pad_0 = const()[name = string("e_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_41_dilations_0 = const()[name = string("e_41_dilations_0"), val = tensor([1, 1])]; int32 e_41_groups_0 = const()[name = string("e_41_groups_0"), val = int32(1)]; tensor e_41 = conv(dilations = e_41_dilations_0, groups = e_41_groups_0, pad = e_41_pad_0, pad_type = e_41_pad_type_0, strides = e_41_strides_0, weight = model_model_layers_20_mlp_down_proj_weight_palettized, x = input_377)[name = string("e_41")]; tensor var_10350_axes_0 = const()[name = string("op_10350_axes_0"), val = tensor([2])]; tensor var_10350 = squeeze(axes = var_10350_axes_0, x = e_41)[name = string("op_10350")]; tensor var_10351 = const()[name = string("op_10351"), val = tensor([0, 2, 1])]; tensor var_10352 = transpose(perm = var_10351, x = var_10350)[name = string("transpose_42")]; tensor hidden_states_211_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = var_10352)[name = string("hidden_states_211_cast_fp16")]; int32 var_10366 = const()[name = string("op_10366"), val = int32(-1)]; fp16 const_294_promoted_to_fp16 = const()[name = string("const_294_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10368_cast_fp16 = mul(x = hidden_states_211_cast_fp16, y = const_294_promoted_to_fp16)[name = string("op_10368_cast_fp16")]; bool input_379_interleave_0 = const()[name = string("input_379_interleave_0"), val = bool(false)]; tensor input_379_cast_fp16 = concat(axis = var_10366, interleave = input_379_interleave_0, values = (hidden_states_211_cast_fp16, var_10368_cast_fp16))[name = string("input_379_cast_fp16")]; tensor normed_337_axes_0 = const()[name = string("normed_337_axes_0"), val = tensor([-1])]; fp16 var_10363_to_fp16 = const()[name = string("op_10363_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_337_cast_fp16 = layer_norm(axes = normed_337_axes_0, epsilon = var_10363_to_fp16, x = input_379_cast_fp16)[name = string("normed_337_cast_fp16")]; tensor normed_339_begin_0 = const()[name = string("normed_339_begin_0"), val = tensor([0, 0, 0])]; tensor normed_339_end_0 = const()[name = string("normed_339_end_0"), val = tensor([1, 1, 1024])]; tensor normed_339_end_mask_0 = const()[name = string("normed_339_end_mask_0"), val = tensor([true, true, false])]; tensor normed_339_cast_fp16 = slice_by_index(begin = normed_339_begin_0, end = normed_339_end_0, end_mask = normed_339_end_mask_0, x = normed_337_cast_fp16)[name = string("normed_339_cast_fp16")]; tensor const_296_promoted_to_fp16 = const()[name = string("const_296_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326381120)))]; tensor hidden_states_213_cast_fp16 = mul(x = normed_339_cast_fp16, y = const_296_promoted_to_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor var_10380 = const()[name = string("op_10380"), val = tensor([0, 2, 1])]; tensor var_10383_axes_0 = const()[name = string("op_10383_axes_0"), val = tensor([2])]; tensor var_10381_cast_fp16 = transpose(perm = var_10380, x = hidden_states_213_cast_fp16)[name = string("transpose_41")]; tensor var_10383_cast_fp16 = expand_dims(axes = var_10383_axes_0, x = var_10381_cast_fp16)[name = string("op_10383_cast_fp16")]; string var_10399_pad_type_0 = const()[name = string("op_10399_pad_type_0"), val = string("valid")]; tensor var_10399_strides_0 = const()[name = string("op_10399_strides_0"), val = tensor([1, 1])]; tensor var_10399_pad_0 = const()[name = string("op_10399_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10399_dilations_0 = const()[name = string("op_10399_dilations_0"), val = tensor([1, 1])]; int32 var_10399_groups_0 = const()[name = string("op_10399_groups_0"), val = int32(1)]; tensor var_10399 = conv(dilations = var_10399_dilations_0, groups = var_10399_groups_0, pad = var_10399_pad_0, pad_type = var_10399_pad_type_0, strides = var_10399_strides_0, weight = model_model_layers_21_self_attn_q_proj_weight_palettized, x = var_10383_cast_fp16)[name = string("op_10399")]; tensor var_10404 = const()[name = string("op_10404"), val = tensor([1, 16, 1, 128])]; tensor var_10405 = reshape(shape = var_10404, x = var_10399)[name = string("op_10405")]; string var_10421_pad_type_0 = const()[name = string("op_10421_pad_type_0"), val = string("valid")]; tensor var_10421_strides_0 = const()[name = string("op_10421_strides_0"), val = tensor([1, 1])]; tensor var_10421_pad_0 = const()[name = string("op_10421_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10421_dilations_0 = const()[name = string("op_10421_dilations_0"), val = tensor([1, 1])]; int32 var_10421_groups_0 = const()[name = string("op_10421_groups_0"), val = int32(1)]; tensor var_10421 = conv(dilations = var_10421_dilations_0, groups = var_10421_groups_0, pad = var_10421_pad_0, pad_type = var_10421_pad_type_0, strides = var_10421_strides_0, weight = model_model_layers_21_self_attn_k_proj_weight_palettized, x = var_10383_cast_fp16)[name = string("op_10421")]; tensor var_10426 = const()[name = string("op_10426"), val = tensor([1, 8, 1, 128])]; tensor var_10427 = reshape(shape = var_10426, x = var_10421)[name = string("op_10427")]; string var_10443_pad_type_0 = const()[name = string("op_10443_pad_type_0"), val = string("valid")]; tensor var_10443_strides_0 = const()[name = string("op_10443_strides_0"), val = tensor([1, 1])]; tensor var_10443_pad_0 = const()[name = string("op_10443_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10443_dilations_0 = const()[name = string("op_10443_dilations_0"), val = tensor([1, 1])]; int32 var_10443_groups_0 = const()[name = string("op_10443_groups_0"), val = int32(1)]; tensor var_10443 = conv(dilations = var_10443_dilations_0, groups = var_10443_groups_0, pad = var_10443_pad_0, pad_type = var_10443_pad_type_0, strides = var_10443_strides_0, weight = model_model_layers_21_self_attn_v_proj_weight_palettized, x = var_10383_cast_fp16)[name = string("op_10443")]; tensor var_10448 = const()[name = string("op_10448"), val = tensor([1, 8, 1, 128])]; tensor var_10449 = reshape(shape = var_10448, x = var_10443)[name = string("op_10449")]; int32 var_10466 = const()[name = string("op_10466"), val = int32(-1)]; fp16 const_297_promoted = const()[name = string("const_297_promoted"), val = fp16(-0x1p+0)]; tensor var_10468 = mul(x = var_10405, y = const_297_promoted)[name = string("op_10468")]; bool input_383_interleave_0 = const()[name = string("input_383_interleave_0"), val = bool(false)]; tensor input_383 = concat(axis = var_10466, interleave = input_383_interleave_0, values = (var_10405, var_10468))[name = string("input_383")]; tensor normed_341_axes_0 = const()[name = string("normed_341_axes_0"), val = tensor([-1])]; fp16 var_10463_to_fp16 = const()[name = string("op_10463_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_341_cast_fp16 = layer_norm(axes = normed_341_axes_0, epsilon = var_10463_to_fp16, x = input_383)[name = string("normed_341_cast_fp16")]; tensor normed_343_begin_0 = const()[name = string("normed_343_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_343_end_0 = const()[name = string("normed_343_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_343_end_mask_0 = const()[name = string("normed_343_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_343 = slice_by_index(begin = normed_343_begin_0, end = normed_343_end_0, end_mask = normed_343_end_mask_0, x = normed_341_cast_fp16)[name = string("normed_343")]; tensor const_299 = const()[name = string("const_299"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326383232)))]; tensor q_43 = mul(x = normed_343, y = const_299)[name = string("q_43")]; int32 var_10488 = const()[name = string("op_10488"), val = int32(-1)]; fp16 const_300_promoted = const()[name = string("const_300_promoted"), val = fp16(-0x1p+0)]; tensor var_10490 = mul(x = var_10427, y = const_300_promoted)[name = string("op_10490")]; bool input_385_interleave_0 = const()[name = string("input_385_interleave_0"), val = bool(false)]; tensor input_385 = concat(axis = var_10488, interleave = input_385_interleave_0, values = (var_10427, var_10490))[name = string("input_385")]; tensor normed_345_axes_0 = const()[name = string("normed_345_axes_0"), val = tensor([-1])]; fp16 var_10485_to_fp16 = const()[name = string("op_10485_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_345_cast_fp16 = layer_norm(axes = normed_345_axes_0, epsilon = var_10485_to_fp16, x = input_385)[name = string("normed_345_cast_fp16")]; tensor normed_347_begin_0 = const()[name = string("normed_347_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_347_end_0 = const()[name = string("normed_347_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_347_end_mask_0 = const()[name = string("normed_347_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_347 = slice_by_index(begin = normed_347_begin_0, end = normed_347_end_0, end_mask = normed_347_end_mask_0, x = normed_345_cast_fp16)[name = string("normed_347")]; tensor const_302 = const()[name = string("const_302"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326383552)))]; tensor k_43 = mul(x = normed_347, y = const_302)[name = string("k_43")]; tensor var_10499 = mul(x = q_43, y = cos_1_cast_fp16)[name = string("op_10499")]; tensor var_10504_begin_0 = const()[name = string("op_10504_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10504_end_0 = const()[name = string("op_10504_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_10504_end_mask_0 = const()[name = string("op_10504_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10504 = slice_by_index(begin = var_10504_begin_0, end = var_10504_end_0, end_mask = var_10504_end_mask_0, x = q_43)[name = string("op_10504")]; fp16 const_303_promoted = const()[name = string("const_303_promoted"), val = fp16(-0x1p+0)]; tensor var_10505 = mul(x = var_10504, y = const_303_promoted)[name = string("op_10505")]; tensor var_10510_begin_0 = const()[name = string("op_10510_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10510_end_0 = const()[name = string("op_10510_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_10510_end_mask_0 = const()[name = string("op_10510_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10510 = slice_by_index(begin = var_10510_begin_0, end = var_10510_end_0, end_mask = var_10510_end_mask_0, x = q_43)[name = string("op_10510")]; int32 var_10512 = const()[name = string("op_10512"), val = int32(-1)]; bool var_10513_interleave_0 = const()[name = string("op_10513_interleave_0"), val = bool(false)]; tensor var_10513 = concat(axis = var_10512, interleave = var_10513_interleave_0, values = (var_10505, var_10510))[name = string("op_10513")]; tensor var_10514 = mul(x = var_10513, y = sin_1_cast_fp16)[name = string("op_10514")]; tensor query_states_85 = add(x = var_10499, y = var_10514)[name = string("query_states_85")]; tensor var_10517 = mul(x = k_43, y = cos_1_cast_fp16)[name = string("op_10517")]; tensor var_10522_begin_0 = const()[name = string("op_10522_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10522_end_0 = const()[name = string("op_10522_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_10522_end_mask_0 = const()[name = string("op_10522_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10522 = slice_by_index(begin = var_10522_begin_0, end = var_10522_end_0, end_mask = var_10522_end_mask_0, x = k_43)[name = string("op_10522")]; fp16 const_304_promoted = const()[name = string("const_304_promoted"), val = fp16(-0x1p+0)]; tensor var_10523 = mul(x = var_10522, y = const_304_promoted)[name = string("op_10523")]; tensor var_10528_begin_0 = const()[name = string("op_10528_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10528_end_0 = const()[name = string("op_10528_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_10528_end_mask_0 = const()[name = string("op_10528_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10528 = slice_by_index(begin = var_10528_begin_0, end = var_10528_end_0, end_mask = var_10528_end_mask_0, x = k_43)[name = string("op_10528")]; int32 var_10530 = const()[name = string("op_10530"), val = int32(-1)]; bool var_10531_interleave_0 = const()[name = string("op_10531_interleave_0"), val = bool(false)]; tensor var_10531 = concat(axis = var_10530, interleave = var_10531_interleave_0, values = (var_10523, var_10528))[name = string("op_10531")]; tensor var_10532 = mul(x = var_10531, y = sin_1_cast_fp16)[name = string("op_10532")]; tensor key_states_85 = add(x = var_10517, y = var_10532)[name = string("key_states_85")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; int32 concat_170_axis_0 = const()[name = string("concat_170_axis_0"), val = int32(0)]; bool concat_170_interleave_0 = const()[name = string("concat_170_interleave_0"), val = bool(false)]; tensor concat_170 = concat(axis = concat_170_axis_0, interleave = concat_170_interleave_0, values = (expand_dims_252, expand_dims_253, current_pos, expand_dims_255))[name = string("concat_170")]; tensor concat_171_values1_0 = const()[name = string("concat_171_values1_0"), val = tensor([0])]; tensor concat_171_values3_0 = const()[name = string("concat_171_values3_0"), val = tensor([0])]; int32 concat_171_axis_0 = const()[name = string("concat_171_axis_0"), val = int32(0)]; bool concat_171_interleave_0 = const()[name = string("concat_171_interleave_0"), val = bool(false)]; tensor concat_171 = concat(axis = concat_171_axis_0, interleave = concat_171_interleave_0, values = (expand_dims_256, concat_171_values1_0, var_1717, concat_171_values3_0))[name = string("concat_171")]; tensor model_model_kv_cache_0_internal_tensor_assign_43_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_cast_fp16 = slice_update(begin = concat_170, begin_mask = model_model_kv_cache_0_internal_tensor_assign_43_begin_mask_0, end = concat_171, end_mask = model_model_kv_cache_0_internal_tensor_assign_43_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_43_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_43_stride_0, update = key_states_85, x = coreml_update_state_97)[name = string("model_model_kv_cache_0_internal_tensor_assign_43_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_43_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_154_write_state")]; tensor coreml_update_state_98 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_154")]; tensor expand_dims_258 = const()[name = string("expand_dims_258"), val = tensor([49])]; tensor expand_dims_259 = const()[name = string("expand_dims_259"), val = tensor([0])]; tensor expand_dims_261 = const()[name = string("expand_dims_261"), val = tensor([0])]; tensor expand_dims_262 = const()[name = string("expand_dims_262"), val = tensor([50])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_258, expand_dims_259, current_pos, expand_dims_261))[name = string("concat_174")]; tensor concat_175_values1_0 = const()[name = string("concat_175_values1_0"), val = tensor([0])]; tensor concat_175_values3_0 = const()[name = string("concat_175_values3_0"), val = tensor([0])]; int32 concat_175_axis_0 = const()[name = string("concat_175_axis_0"), val = int32(0)]; bool concat_175_interleave_0 = const()[name = string("concat_175_interleave_0"), val = bool(false)]; tensor concat_175 = concat(axis = concat_175_axis_0, interleave = concat_175_interleave_0, values = (expand_dims_262, concat_175_values1_0, var_1717, concat_175_values3_0))[name = string("concat_175")]; tensor model_model_kv_cache_0_internal_tensor_assign_44_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_cast_fp16 = slice_update(begin = concat_174, begin_mask = model_model_kv_cache_0_internal_tensor_assign_44_begin_mask_0, end = concat_175, end_mask = model_model_kv_cache_0_internal_tensor_assign_44_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_44_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_44_stride_0, update = var_10449, x = coreml_update_state_98)[name = string("model_model_kv_cache_0_internal_tensor_assign_44_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_44_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_155_write_state")]; tensor coreml_update_state_99 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_155")]; tensor var_10587_begin_0 = const()[name = string("op_10587_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_10587_end_0 = const()[name = string("op_10587_end_0"), val = tensor([22, 8, 1536, 128])]; tensor var_10587_end_mask_0 = const()[name = string("op_10587_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10587_cast_fp16 = slice_by_index(begin = var_10587_begin_0, end = var_10587_end_0, end_mask = var_10587_end_mask_0, x = coreml_update_state_99)[name = string("op_10587_cast_fp16")]; tensor key_cache_43_axes_0 = const()[name = string("key_cache_43_axes_0"), val = tensor([0])]; tensor key_cache_43_cast_fp16 = squeeze(axes = key_cache_43_axes_0, x = var_10587_cast_fp16)[name = string("key_cache_43_cast_fp16")]; tensor var_10594_begin_0 = const()[name = string("op_10594_begin_0"), val = tensor([49, 0, 0, 0])]; tensor var_10594_end_0 = const()[name = string("op_10594_end_0"), val = tensor([50, 8, 1536, 128])]; tensor var_10594_end_mask_0 = const()[name = string("op_10594_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10594_cast_fp16 = slice_by_index(begin = var_10594_begin_0, end = var_10594_end_0, end_mask = var_10594_end_mask_0, x = coreml_update_state_99)[name = string("op_10594_cast_fp16")]; tensor value_cache_43_axes_0 = const()[name = string("value_cache_43_axes_0"), val = tensor([0])]; tensor value_cache_43_cast_fp16 = squeeze(axes = value_cache_43_axes_0, x = var_10594_cast_fp16)[name = string("value_cache_43_cast_fp16")]; tensor var_10618_axes_0 = const()[name = string("op_10618_axes_0"), val = tensor([1])]; tensor var_10618_cast_fp16 = expand_dims(axes = var_10618_axes_0, x = key_cache_43_cast_fp16)[name = string("op_10618_cast_fp16")]; tensor var_10623 = const()[name = string("op_10623"), val = tensor([1, 2, 1, 1])]; tensor value_171_cast_fp16 = tile(reps = var_10623, x = var_10618_cast_fp16)[name = string("value_171_cast_fp16")]; tensor var_10629 = const()[name = string("op_10629"), val = tensor([1, 16, 1536, 128])]; tensor key_states_87_cast_fp16 = reshape(shape = var_10629, x = value_171_cast_fp16)[name = string("key_states_87_cast_fp16")]; tensor var_10632_axes_0 = const()[name = string("op_10632_axes_0"), val = tensor([1])]; tensor var_10632_cast_fp16 = expand_dims(axes = var_10632_axes_0, x = value_cache_43_cast_fp16)[name = string("op_10632_cast_fp16")]; tensor var_10637 = const()[name = string("op_10637"), val = tensor([1, 2, 1, 1])]; tensor value_175_cast_fp16 = tile(reps = var_10637, x = var_10632_cast_fp16)[name = string("value_175_cast_fp16")]; tensor var_10643 = const()[name = string("op_10643"), val = tensor([1, 16, 1536, 128])]; tensor value_states_129_cast_fp16 = reshape(shape = var_10643, x = value_175_cast_fp16)[name = string("value_states_129_cast_fp16")]; bool var_10658_transpose_x_1 = const()[name = string("op_10658_transpose_x_1"), val = bool(false)]; bool var_10658_transpose_y_1 = const()[name = string("op_10658_transpose_y_1"), val = bool(true)]; tensor var_10658 = matmul(transpose_x = var_10658_transpose_x_1, transpose_y = var_10658_transpose_y_1, x = query_states_85, y = key_states_87_cast_fp16)[name = string("op_10658")]; fp16 var_10659_to_fp16 = const()[name = string("op_10659_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_85_cast_fp16 = mul(x = var_10658, y = var_10659_to_fp16)[name = string("attention_85_cast_fp16")]; tensor attention_87_cast_fp16 = add(x = attention_85_cast_fp16, y = causal_mask)[name = string("attention_87_cast_fp16")]; int32 var_10668 = const()[name = string("op_10668"), val = int32(-1)]; tensor probabilities_43_cast_fp16 = softmax(axis = var_10668, x = attention_87_cast_fp16)[name = string("probabilities_43_cast_fp16")]; bool output_127_transpose_x_0 = const()[name = string("output_127_transpose_x_0"), val = bool(false)]; bool output_127_transpose_y_0 = const()[name = string("output_127_transpose_y_0"), val = bool(false)]; tensor output_127_cast_fp16 = matmul(transpose_x = output_127_transpose_x_0, transpose_y = output_127_transpose_y_0, x = probabilities_43_cast_fp16, y = value_states_129_cast_fp16)[name = string("output_127_cast_fp16")]; tensor var_10679_perm_0 = const()[name = string("op_10679_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_10685 = const()[name = string("op_10685"), val = tensor([1, 1, 2048])]; tensor var_10679_cast_fp16 = transpose(perm = var_10679_perm_0, x = output_127_cast_fp16)[name = string("transpose_40")]; tensor output_129_cast_fp16 = reshape(shape = var_10685, x = var_10679_cast_fp16)[name = string("output_129_cast_fp16")]; tensor var_10690 = const()[name = string("op_10690"), val = tensor([0, 2, 1])]; string var_10706_pad_type_0 = const()[name = string("op_10706_pad_type_0"), val = string("valid")]; int32 var_10706_groups_0 = const()[name = string("op_10706_groups_0"), val = int32(1)]; tensor var_10706_strides_0 = const()[name = string("op_10706_strides_0"), val = tensor([1])]; tensor var_10706_pad_0 = const()[name = string("op_10706_pad_0"), val = tensor([0, 0])]; tensor var_10706_dilations_0 = const()[name = string("op_10706_dilations_0"), val = tensor([1])]; tensor squeeze_21_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326383872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327956800))))[name = string("squeeze_21_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10691_cast_fp16 = transpose(perm = var_10690, x = output_129_cast_fp16)[name = string("transpose_39")]; tensor var_10706_cast_fp16 = conv(dilations = var_10706_dilations_0, groups = var_10706_groups_0, pad = var_10706_pad_0, pad_type = var_10706_pad_type_0, strides = var_10706_strides_0, weight = squeeze_21_cast_fp16_to_fp32_to_fp16_palettized, x = var_10691_cast_fp16)[name = string("op_10706_cast_fp16")]; tensor var_10710 = const()[name = string("op_10710"), val = tensor([0, 2, 1])]; tensor attn_output_43_cast_fp16 = transpose(perm = var_10710, x = var_10706_cast_fp16)[name = string("transpose_38")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_211_cast_fp16, y = attn_output_43_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; int32 var_10725 = const()[name = string("op_10725"), val = int32(-1)]; fp16 const_305_promoted_to_fp16 = const()[name = string("const_305_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10727_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_305_promoted_to_fp16)[name = string("op_10727_cast_fp16")]; bool input_389_interleave_0 = const()[name = string("input_389_interleave_0"), val = bool(false)]; tensor input_389_cast_fp16 = concat(axis = var_10725, interleave = input_389_interleave_0, values = (hidden_states_219_cast_fp16, var_10727_cast_fp16))[name = string("input_389_cast_fp16")]; tensor normed_349_axes_0 = const()[name = string("normed_349_axes_0"), val = tensor([-1])]; fp16 var_10722_to_fp16 = const()[name = string("op_10722_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_349_cast_fp16 = layer_norm(axes = normed_349_axes_0, epsilon = var_10722_to_fp16, x = input_389_cast_fp16)[name = string("normed_349_cast_fp16")]; tensor normed_351_begin_0 = const()[name = string("normed_351_begin_0"), val = tensor([0, 0, 0])]; tensor normed_351_end_0 = const()[name = string("normed_351_end_0"), val = tensor([1, 1, 1024])]; tensor normed_351_end_mask_0 = const()[name = string("normed_351_end_mask_0"), val = tensor([true, true, false])]; tensor normed_351_cast_fp16 = slice_by_index(begin = normed_351_begin_0, end = normed_351_end_0, end_mask = normed_351_end_mask_0, x = normed_349_cast_fp16)[name = string("normed_351_cast_fp16")]; tensor const_307_promoted_to_fp16 = const()[name = string("const_307_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327973248)))]; tensor x_85_cast_fp16 = mul(x = normed_351_cast_fp16, y = const_307_promoted_to_fp16)[name = string("x_85_cast_fp16")]; tensor var_10747 = const()[name = string("op_10747"), val = tensor([0, 2, 1])]; tensor input_391_axes_0 = const()[name = string("input_391_axes_0"), val = tensor([2])]; tensor var_10748 = transpose(perm = var_10747, x = x_85_cast_fp16)[name = string("transpose_37")]; tensor input_391 = expand_dims(axes = input_391_axes_0, x = var_10748)[name = string("input_391")]; string input_393_pad_type_0 = const()[name = string("input_393_pad_type_0"), val = string("valid")]; tensor input_393_strides_0 = const()[name = string("input_393_strides_0"), val = tensor([1, 1])]; tensor input_393_pad_0 = const()[name = string("input_393_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_393_dilations_0 = const()[name = string("input_393_dilations_0"), val = tensor([1, 1])]; int32 input_393_groups_0 = const()[name = string("input_393_groups_0"), val = int32(1)]; tensor input_393 = conv(dilations = input_393_dilations_0, groups = input_393_groups_0, pad = input_393_pad_0, pad_type = input_393_pad_type_0, strides = input_393_strides_0, weight = model_model_layers_21_mlp_gate_proj_weight_palettized, x = input_391)[name = string("input_393")]; string b_43_pad_type_0 = const()[name = string("b_43_pad_type_0"), val = string("valid")]; tensor b_43_strides_0 = const()[name = string("b_43_strides_0"), val = tensor([1, 1])]; tensor b_43_pad_0 = const()[name = string("b_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_43_dilations_0 = const()[name = string("b_43_dilations_0"), val = tensor([1, 1])]; int32 b_43_groups_0 = const()[name = string("b_43_groups_0"), val = int32(1)]; tensor b_43 = conv(dilations = b_43_dilations_0, groups = b_43_groups_0, pad = b_43_pad_0, pad_type = b_43_pad_type_0, strides = b_43_strides_0, weight = model_model_layers_21_mlp_up_proj_weight_palettized, x = input_391)[name = string("b_43")]; tensor c_43 = silu(x = input_393)[name = string("c_43")]; tensor input_395 = mul(x = c_43, y = b_43)[name = string("input_395")]; string e_43_pad_type_0 = const()[name = string("e_43_pad_type_0"), val = string("valid")]; tensor e_43_strides_0 = const()[name = string("e_43_strides_0"), val = tensor([1, 1])]; tensor e_43_pad_0 = const()[name = string("e_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_43_dilations_0 = const()[name = string("e_43_dilations_0"), val = tensor([1, 1])]; int32 e_43_groups_0 = const()[name = string("e_43_groups_0"), val = int32(1)]; tensor e_43 = conv(dilations = e_43_dilations_0, groups = e_43_groups_0, pad = e_43_pad_0, pad_type = e_43_pad_type_0, strides = e_43_strides_0, weight = model_model_layers_21_mlp_down_proj_weight_palettized, x = input_395)[name = string("e_43")]; tensor var_10770_axes_0 = const()[name = string("op_10770_axes_0"), val = tensor([2])]; tensor var_10770 = squeeze(axes = var_10770_axes_0, x = e_43)[name = string("op_10770")]; tensor var_10771 = const()[name = string("op_10771"), val = tensor([0, 2, 1])]; tensor var_10772 = transpose(perm = var_10771, x = var_10770)[name = string("transpose_36")]; tensor hidden_states_221_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = var_10772)[name = string("hidden_states_221_cast_fp16")]; int32 var_10786 = const()[name = string("op_10786"), val = int32(-1)]; fp16 const_308_promoted_to_fp16 = const()[name = string("const_308_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10788_cast_fp16 = mul(x = hidden_states_221_cast_fp16, y = const_308_promoted_to_fp16)[name = string("op_10788_cast_fp16")]; bool input_397_interleave_0 = const()[name = string("input_397_interleave_0"), val = bool(false)]; tensor input_397_cast_fp16 = concat(axis = var_10786, interleave = input_397_interleave_0, values = (hidden_states_221_cast_fp16, var_10788_cast_fp16))[name = string("input_397_cast_fp16")]; tensor normed_353_axes_0 = const()[name = string("normed_353_axes_0"), val = tensor([-1])]; fp16 var_10783_to_fp16 = const()[name = string("op_10783_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_353_cast_fp16 = layer_norm(axes = normed_353_axes_0, epsilon = var_10783_to_fp16, x = input_397_cast_fp16)[name = string("normed_353_cast_fp16")]; tensor normed_355_begin_0 = const()[name = string("normed_355_begin_0"), val = tensor([0, 0, 0])]; tensor normed_355_end_0 = const()[name = string("normed_355_end_0"), val = tensor([1, 1, 1024])]; tensor normed_355_end_mask_0 = const()[name = string("normed_355_end_mask_0"), val = tensor([true, true, false])]; tensor normed_355_cast_fp16 = slice_by_index(begin = normed_355_begin_0, end = normed_355_end_0, end_mask = normed_355_end_mask_0, x = normed_353_cast_fp16)[name = string("normed_355_cast_fp16")]; tensor const_310_promoted_to_fp16 = const()[name = string("const_310_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327975360)))]; tensor hidden_states_223_cast_fp16 = mul(x = normed_355_cast_fp16, y = const_310_promoted_to_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor var_10800 = const()[name = string("op_10800"), val = tensor([0, 2, 1])]; tensor var_10803_axes_0 = const()[name = string("op_10803_axes_0"), val = tensor([2])]; tensor var_10801_cast_fp16 = transpose(perm = var_10800, x = hidden_states_223_cast_fp16)[name = string("transpose_35")]; tensor var_10803_cast_fp16 = expand_dims(axes = var_10803_axes_0, x = var_10801_cast_fp16)[name = string("op_10803_cast_fp16")]; string var_10819_pad_type_0 = const()[name = string("op_10819_pad_type_0"), val = string("valid")]; tensor var_10819_strides_0 = const()[name = string("op_10819_strides_0"), val = tensor([1, 1])]; tensor var_10819_pad_0 = const()[name = string("op_10819_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10819_dilations_0 = const()[name = string("op_10819_dilations_0"), val = tensor([1, 1])]; int32 var_10819_groups_0 = const()[name = string("op_10819_groups_0"), val = int32(1)]; tensor var_10819 = conv(dilations = var_10819_dilations_0, groups = var_10819_groups_0, pad = var_10819_pad_0, pad_type = var_10819_pad_type_0, strides = var_10819_strides_0, weight = model_model_layers_22_self_attn_q_proj_weight_palettized, x = var_10803_cast_fp16)[name = string("op_10819")]; tensor var_10824 = const()[name = string("op_10824"), val = tensor([1, 16, 1, 128])]; tensor var_10825 = reshape(shape = var_10824, x = var_10819)[name = string("op_10825")]; string var_10841_pad_type_0 = const()[name = string("op_10841_pad_type_0"), val = string("valid")]; tensor var_10841_strides_0 = const()[name = string("op_10841_strides_0"), val = tensor([1, 1])]; tensor var_10841_pad_0 = const()[name = string("op_10841_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10841_dilations_0 = const()[name = string("op_10841_dilations_0"), val = tensor([1, 1])]; int32 var_10841_groups_0 = const()[name = string("op_10841_groups_0"), val = int32(1)]; tensor var_10841 = conv(dilations = var_10841_dilations_0, groups = var_10841_groups_0, pad = var_10841_pad_0, pad_type = var_10841_pad_type_0, strides = var_10841_strides_0, weight = model_model_layers_22_self_attn_k_proj_weight_palettized, x = var_10803_cast_fp16)[name = string("op_10841")]; tensor var_10846 = const()[name = string("op_10846"), val = tensor([1, 8, 1, 128])]; tensor var_10847 = reshape(shape = var_10846, x = var_10841)[name = string("op_10847")]; string var_10863_pad_type_0 = const()[name = string("op_10863_pad_type_0"), val = string("valid")]; tensor var_10863_strides_0 = const()[name = string("op_10863_strides_0"), val = tensor([1, 1])]; tensor var_10863_pad_0 = const()[name = string("op_10863_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10863_dilations_0 = const()[name = string("op_10863_dilations_0"), val = tensor([1, 1])]; int32 var_10863_groups_0 = const()[name = string("op_10863_groups_0"), val = int32(1)]; tensor var_10863 = conv(dilations = var_10863_dilations_0, groups = var_10863_groups_0, pad = var_10863_pad_0, pad_type = var_10863_pad_type_0, strides = var_10863_strides_0, weight = model_model_layers_22_self_attn_v_proj_weight_palettized, x = var_10803_cast_fp16)[name = string("op_10863")]; tensor var_10868 = const()[name = string("op_10868"), val = tensor([1, 8, 1, 128])]; tensor var_10869 = reshape(shape = var_10868, x = var_10863)[name = string("op_10869")]; int32 var_10886 = const()[name = string("op_10886"), val = int32(-1)]; fp16 const_311_promoted = const()[name = string("const_311_promoted"), val = fp16(-0x1p+0)]; tensor var_10888 = mul(x = var_10825, y = const_311_promoted)[name = string("op_10888")]; bool input_401_interleave_0 = const()[name = string("input_401_interleave_0"), val = bool(false)]; tensor input_401 = concat(axis = var_10886, interleave = input_401_interleave_0, values = (var_10825, var_10888))[name = string("input_401")]; tensor normed_357_axes_0 = const()[name = string("normed_357_axes_0"), val = tensor([-1])]; fp16 var_10883_to_fp16 = const()[name = string("op_10883_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_357_cast_fp16 = layer_norm(axes = normed_357_axes_0, epsilon = var_10883_to_fp16, x = input_401)[name = string("normed_357_cast_fp16")]; tensor normed_359_begin_0 = const()[name = string("normed_359_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_359_end_0 = const()[name = string("normed_359_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_359_end_mask_0 = const()[name = string("normed_359_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_359 = slice_by_index(begin = normed_359_begin_0, end = normed_359_end_0, end_mask = normed_359_end_mask_0, x = normed_357_cast_fp16)[name = string("normed_359")]; tensor const_313 = const()[name = string("const_313"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327977472)))]; tensor q_45 = mul(x = normed_359, y = const_313)[name = string("q_45")]; int32 var_10908 = const()[name = string("op_10908"), val = int32(-1)]; fp16 const_314_promoted = const()[name = string("const_314_promoted"), val = fp16(-0x1p+0)]; tensor var_10910 = mul(x = var_10847, y = const_314_promoted)[name = string("op_10910")]; bool input_403_interleave_0 = const()[name = string("input_403_interleave_0"), val = bool(false)]; tensor input_403 = concat(axis = var_10908, interleave = input_403_interleave_0, values = (var_10847, var_10910))[name = string("input_403")]; tensor normed_361_axes_0 = const()[name = string("normed_361_axes_0"), val = tensor([-1])]; fp16 var_10905_to_fp16 = const()[name = string("op_10905_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_361_cast_fp16 = layer_norm(axes = normed_361_axes_0, epsilon = var_10905_to_fp16, x = input_403)[name = string("normed_361_cast_fp16")]; tensor normed_363_begin_0 = const()[name = string("normed_363_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_363_end_0 = const()[name = string("normed_363_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_363_end_mask_0 = const()[name = string("normed_363_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_363 = slice_by_index(begin = normed_363_begin_0, end = normed_363_end_0, end_mask = normed_363_end_mask_0, x = normed_361_cast_fp16)[name = string("normed_363")]; tensor const_316 = const()[name = string("const_316"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327977792)))]; tensor k_45 = mul(x = normed_363, y = const_316)[name = string("k_45")]; tensor var_10919 = mul(x = q_45, y = cos_1_cast_fp16)[name = string("op_10919")]; tensor var_10924_begin_0 = const()[name = string("op_10924_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10924_end_0 = const()[name = string("op_10924_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_10924_end_mask_0 = const()[name = string("op_10924_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10924 = slice_by_index(begin = var_10924_begin_0, end = var_10924_end_0, end_mask = var_10924_end_mask_0, x = q_45)[name = string("op_10924")]; fp16 const_317_promoted = const()[name = string("const_317_promoted"), val = fp16(-0x1p+0)]; tensor var_10925 = mul(x = var_10924, y = const_317_promoted)[name = string("op_10925")]; tensor var_10930_begin_0 = const()[name = string("op_10930_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10930_end_0 = const()[name = string("op_10930_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_10930_end_mask_0 = const()[name = string("op_10930_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10930 = slice_by_index(begin = var_10930_begin_0, end = var_10930_end_0, end_mask = var_10930_end_mask_0, x = q_45)[name = string("op_10930")]; int32 var_10932 = const()[name = string("op_10932"), val = int32(-1)]; bool var_10933_interleave_0 = const()[name = string("op_10933_interleave_0"), val = bool(false)]; tensor var_10933 = concat(axis = var_10932, interleave = var_10933_interleave_0, values = (var_10925, var_10930))[name = string("op_10933")]; tensor var_10934 = mul(x = var_10933, y = sin_1_cast_fp16)[name = string("op_10934")]; tensor query_states_89 = add(x = var_10919, y = var_10934)[name = string("query_states_89")]; tensor var_10937 = mul(x = k_45, y = cos_1_cast_fp16)[name = string("op_10937")]; tensor var_10942_begin_0 = const()[name = string("op_10942_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10942_end_0 = const()[name = string("op_10942_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_10942_end_mask_0 = const()[name = string("op_10942_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10942 = slice_by_index(begin = var_10942_begin_0, end = var_10942_end_0, end_mask = var_10942_end_mask_0, x = k_45)[name = string("op_10942")]; fp16 const_318_promoted = const()[name = string("const_318_promoted"), val = fp16(-0x1p+0)]; tensor var_10943 = mul(x = var_10942, y = const_318_promoted)[name = string("op_10943")]; tensor var_10948_begin_0 = const()[name = string("op_10948_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10948_end_0 = const()[name = string("op_10948_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_10948_end_mask_0 = const()[name = string("op_10948_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10948 = slice_by_index(begin = var_10948_begin_0, end = var_10948_end_0, end_mask = var_10948_end_mask_0, x = k_45)[name = string("op_10948")]; int32 var_10950 = const()[name = string("op_10950"), val = int32(-1)]; bool var_10951_interleave_0 = const()[name = string("op_10951_interleave_0"), val = bool(false)]; tensor var_10951 = concat(axis = var_10950, interleave = var_10951_interleave_0, values = (var_10943, var_10948))[name = string("op_10951")]; tensor var_10952 = mul(x = var_10951, y = sin_1_cast_fp16)[name = string("op_10952")]; tensor key_states_89 = add(x = var_10937, y = var_10952)[name = string("key_states_89")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; int32 concat_178_axis_0 = const()[name = string("concat_178_axis_0"), val = int32(0)]; bool concat_178_interleave_0 = const()[name = string("concat_178_interleave_0"), val = bool(false)]; tensor concat_178 = concat(axis = concat_178_axis_0, interleave = concat_178_interleave_0, values = (expand_dims_264, expand_dims_265, current_pos, expand_dims_267))[name = string("concat_178")]; tensor concat_179_values1_0 = const()[name = string("concat_179_values1_0"), val = tensor([0])]; tensor concat_179_values3_0 = const()[name = string("concat_179_values3_0"), val = tensor([0])]; int32 concat_179_axis_0 = const()[name = string("concat_179_axis_0"), val = int32(0)]; bool concat_179_interleave_0 = const()[name = string("concat_179_interleave_0"), val = bool(false)]; tensor concat_179 = concat(axis = concat_179_axis_0, interleave = concat_179_interleave_0, values = (expand_dims_268, concat_179_values1_0, var_1717, concat_179_values3_0))[name = string("concat_179")]; tensor model_model_kv_cache_0_internal_tensor_assign_45_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_cast_fp16 = slice_update(begin = concat_178, begin_mask = model_model_kv_cache_0_internal_tensor_assign_45_begin_mask_0, end = concat_179, end_mask = model_model_kv_cache_0_internal_tensor_assign_45_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_45_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_45_stride_0, update = key_states_89, x = coreml_update_state_99)[name = string("model_model_kv_cache_0_internal_tensor_assign_45_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_45_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_156_write_state")]; tensor coreml_update_state_100 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_156")]; tensor expand_dims_270 = const()[name = string("expand_dims_270"), val = tensor([50])]; tensor expand_dims_271 = const()[name = string("expand_dims_271"), val = tensor([0])]; tensor expand_dims_273 = const()[name = string("expand_dims_273"), val = tensor([0])]; tensor expand_dims_274 = const()[name = string("expand_dims_274"), val = tensor([51])]; int32 concat_182_axis_0 = const()[name = string("concat_182_axis_0"), val = int32(0)]; bool concat_182_interleave_0 = const()[name = string("concat_182_interleave_0"), val = bool(false)]; tensor concat_182 = concat(axis = concat_182_axis_0, interleave = concat_182_interleave_0, values = (expand_dims_270, expand_dims_271, current_pos, expand_dims_273))[name = string("concat_182")]; tensor concat_183_values1_0 = const()[name = string("concat_183_values1_0"), val = tensor([0])]; tensor concat_183_values3_0 = const()[name = string("concat_183_values3_0"), val = tensor([0])]; int32 concat_183_axis_0 = const()[name = string("concat_183_axis_0"), val = int32(0)]; bool concat_183_interleave_0 = const()[name = string("concat_183_interleave_0"), val = bool(false)]; tensor concat_183 = concat(axis = concat_183_axis_0, interleave = concat_183_interleave_0, values = (expand_dims_274, concat_183_values1_0, var_1717, concat_183_values3_0))[name = string("concat_183")]; tensor model_model_kv_cache_0_internal_tensor_assign_46_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_cast_fp16 = slice_update(begin = concat_182, begin_mask = model_model_kv_cache_0_internal_tensor_assign_46_begin_mask_0, end = concat_183, end_mask = model_model_kv_cache_0_internal_tensor_assign_46_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_46_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_46_stride_0, update = var_10869, x = coreml_update_state_100)[name = string("model_model_kv_cache_0_internal_tensor_assign_46_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_46_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_157_write_state")]; tensor coreml_update_state_101 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_157")]; tensor var_11007_begin_0 = const()[name = string("op_11007_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_11007_end_0 = const()[name = string("op_11007_end_0"), val = tensor([23, 8, 1536, 128])]; tensor var_11007_end_mask_0 = const()[name = string("op_11007_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11007_cast_fp16 = slice_by_index(begin = var_11007_begin_0, end = var_11007_end_0, end_mask = var_11007_end_mask_0, x = coreml_update_state_101)[name = string("op_11007_cast_fp16")]; tensor key_cache_45_axes_0 = const()[name = string("key_cache_45_axes_0"), val = tensor([0])]; tensor key_cache_45_cast_fp16 = squeeze(axes = key_cache_45_axes_0, x = var_11007_cast_fp16)[name = string("key_cache_45_cast_fp16")]; tensor var_11014_begin_0 = const()[name = string("op_11014_begin_0"), val = tensor([50, 0, 0, 0])]; tensor var_11014_end_0 = const()[name = string("op_11014_end_0"), val = tensor([51, 8, 1536, 128])]; tensor var_11014_end_mask_0 = const()[name = string("op_11014_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11014_cast_fp16 = slice_by_index(begin = var_11014_begin_0, end = var_11014_end_0, end_mask = var_11014_end_mask_0, x = coreml_update_state_101)[name = string("op_11014_cast_fp16")]; tensor value_cache_45_axes_0 = const()[name = string("value_cache_45_axes_0"), val = tensor([0])]; tensor value_cache_45_cast_fp16 = squeeze(axes = value_cache_45_axes_0, x = var_11014_cast_fp16)[name = string("value_cache_45_cast_fp16")]; tensor var_11038_axes_0 = const()[name = string("op_11038_axes_0"), val = tensor([1])]; tensor var_11038_cast_fp16 = expand_dims(axes = var_11038_axes_0, x = key_cache_45_cast_fp16)[name = string("op_11038_cast_fp16")]; tensor var_11043 = const()[name = string("op_11043"), val = tensor([1, 2, 1, 1])]; tensor value_179_cast_fp16 = tile(reps = var_11043, x = var_11038_cast_fp16)[name = string("value_179_cast_fp16")]; tensor var_11049 = const()[name = string("op_11049"), val = tensor([1, 16, 1536, 128])]; tensor key_states_91_cast_fp16 = reshape(shape = var_11049, x = value_179_cast_fp16)[name = string("key_states_91_cast_fp16")]; tensor var_11052_axes_0 = const()[name = string("op_11052_axes_0"), val = tensor([1])]; tensor var_11052_cast_fp16 = expand_dims(axes = var_11052_axes_0, x = value_cache_45_cast_fp16)[name = string("op_11052_cast_fp16")]; tensor var_11057 = const()[name = string("op_11057"), val = tensor([1, 2, 1, 1])]; tensor value_183_cast_fp16 = tile(reps = var_11057, x = var_11052_cast_fp16)[name = string("value_183_cast_fp16")]; tensor var_11063 = const()[name = string("op_11063"), val = tensor([1, 16, 1536, 128])]; tensor value_states_135_cast_fp16 = reshape(shape = var_11063, x = value_183_cast_fp16)[name = string("value_states_135_cast_fp16")]; bool var_11078_transpose_x_1 = const()[name = string("op_11078_transpose_x_1"), val = bool(false)]; bool var_11078_transpose_y_1 = const()[name = string("op_11078_transpose_y_1"), val = bool(true)]; tensor var_11078 = matmul(transpose_x = var_11078_transpose_x_1, transpose_y = var_11078_transpose_y_1, x = query_states_89, y = key_states_91_cast_fp16)[name = string("op_11078")]; fp16 var_11079_to_fp16 = const()[name = string("op_11079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_89_cast_fp16 = mul(x = var_11078, y = var_11079_to_fp16)[name = string("attention_89_cast_fp16")]; tensor attention_91_cast_fp16 = add(x = attention_89_cast_fp16, y = causal_mask)[name = string("attention_91_cast_fp16")]; int32 var_11088 = const()[name = string("op_11088"), val = int32(-1)]; tensor probabilities_45_cast_fp16 = softmax(axis = var_11088, x = attention_91_cast_fp16)[name = string("probabilities_45_cast_fp16")]; bool output_133_transpose_x_0 = const()[name = string("output_133_transpose_x_0"), val = bool(false)]; bool output_133_transpose_y_0 = const()[name = string("output_133_transpose_y_0"), val = bool(false)]; tensor output_133_cast_fp16 = matmul(transpose_x = output_133_transpose_x_0, transpose_y = output_133_transpose_y_0, x = probabilities_45_cast_fp16, y = value_states_135_cast_fp16)[name = string("output_133_cast_fp16")]; tensor var_11099_perm_0 = const()[name = string("op_11099_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_11105 = const()[name = string("op_11105"), val = tensor([1, 1, 2048])]; tensor var_11099_cast_fp16 = transpose(perm = var_11099_perm_0, x = output_133_cast_fp16)[name = string("transpose_34")]; tensor output_135_cast_fp16 = reshape(shape = var_11105, x = var_11099_cast_fp16)[name = string("output_135_cast_fp16")]; tensor var_11110 = const()[name = string("op_11110"), val = tensor([0, 2, 1])]; string var_11126_pad_type_0 = const()[name = string("op_11126_pad_type_0"), val = string("valid")]; int32 var_11126_groups_0 = const()[name = string("op_11126_groups_0"), val = int32(1)]; tensor var_11126_strides_0 = const()[name = string("op_11126_strides_0"), val = tensor([1])]; tensor var_11126_pad_0 = const()[name = string("op_11126_pad_0"), val = tensor([0, 0])]; tensor var_11126_dilations_0 = const()[name = string("op_11126_dilations_0"), val = tensor([1])]; tensor squeeze_22_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327978112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329551040))))[name = string("squeeze_22_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11111_cast_fp16 = transpose(perm = var_11110, x = output_135_cast_fp16)[name = string("transpose_33")]; tensor var_11126_cast_fp16 = conv(dilations = var_11126_dilations_0, groups = var_11126_groups_0, pad = var_11126_pad_0, pad_type = var_11126_pad_type_0, strides = var_11126_strides_0, weight = squeeze_22_cast_fp16_to_fp32_to_fp16_palettized, x = var_11111_cast_fp16)[name = string("op_11126_cast_fp16")]; tensor var_11130 = const()[name = string("op_11130"), val = tensor([0, 2, 1])]; tensor attn_output_45_cast_fp16 = transpose(perm = var_11130, x = var_11126_cast_fp16)[name = string("transpose_32")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_221_cast_fp16, y = attn_output_45_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; int32 var_11145 = const()[name = string("op_11145"), val = int32(-1)]; fp16 const_319_promoted_to_fp16 = const()[name = string("const_319_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11147_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_319_promoted_to_fp16)[name = string("op_11147_cast_fp16")]; bool input_407_interleave_0 = const()[name = string("input_407_interleave_0"), val = bool(false)]; tensor input_407_cast_fp16 = concat(axis = var_11145, interleave = input_407_interleave_0, values = (hidden_states_229_cast_fp16, var_11147_cast_fp16))[name = string("input_407_cast_fp16")]; tensor normed_365_axes_0 = const()[name = string("normed_365_axes_0"), val = tensor([-1])]; fp16 var_11142_to_fp16 = const()[name = string("op_11142_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_365_cast_fp16 = layer_norm(axes = normed_365_axes_0, epsilon = var_11142_to_fp16, x = input_407_cast_fp16)[name = string("normed_365_cast_fp16")]; tensor normed_367_begin_0 = const()[name = string("normed_367_begin_0"), val = tensor([0, 0, 0])]; tensor normed_367_end_0 = const()[name = string("normed_367_end_0"), val = tensor([1, 1, 1024])]; tensor normed_367_end_mask_0 = const()[name = string("normed_367_end_mask_0"), val = tensor([true, true, false])]; tensor normed_367_cast_fp16 = slice_by_index(begin = normed_367_begin_0, end = normed_367_end_0, end_mask = normed_367_end_mask_0, x = normed_365_cast_fp16)[name = string("normed_367_cast_fp16")]; tensor const_321_promoted_to_fp16 = const()[name = string("const_321_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329567488)))]; tensor x_89_cast_fp16 = mul(x = normed_367_cast_fp16, y = const_321_promoted_to_fp16)[name = string("x_89_cast_fp16")]; tensor var_11167 = const()[name = string("op_11167"), val = tensor([0, 2, 1])]; tensor input_409_axes_0 = const()[name = string("input_409_axes_0"), val = tensor([2])]; tensor var_11168 = transpose(perm = var_11167, x = x_89_cast_fp16)[name = string("transpose_31")]; tensor input_409 = expand_dims(axes = input_409_axes_0, x = var_11168)[name = string("input_409")]; string input_411_pad_type_0 = const()[name = string("input_411_pad_type_0"), val = string("valid")]; tensor input_411_strides_0 = const()[name = string("input_411_strides_0"), val = tensor([1, 1])]; tensor input_411_pad_0 = const()[name = string("input_411_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_411_dilations_0 = const()[name = string("input_411_dilations_0"), val = tensor([1, 1])]; int32 input_411_groups_0 = const()[name = string("input_411_groups_0"), val = int32(1)]; tensor input_411 = conv(dilations = input_411_dilations_0, groups = input_411_groups_0, pad = input_411_pad_0, pad_type = input_411_pad_type_0, strides = input_411_strides_0, weight = model_model_layers_22_mlp_gate_proj_weight_palettized, x = input_409)[name = string("input_411")]; string b_45_pad_type_0 = const()[name = string("b_45_pad_type_0"), val = string("valid")]; tensor b_45_strides_0 = const()[name = string("b_45_strides_0"), val = tensor([1, 1])]; tensor b_45_pad_0 = const()[name = string("b_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_45_dilations_0 = const()[name = string("b_45_dilations_0"), val = tensor([1, 1])]; int32 b_45_groups_0 = const()[name = string("b_45_groups_0"), val = int32(1)]; tensor b_45 = conv(dilations = b_45_dilations_0, groups = b_45_groups_0, pad = b_45_pad_0, pad_type = b_45_pad_type_0, strides = b_45_strides_0, weight = model_model_layers_22_mlp_up_proj_weight_palettized, x = input_409)[name = string("b_45")]; tensor c_45 = silu(x = input_411)[name = string("c_45")]; tensor input_413 = mul(x = c_45, y = b_45)[name = string("input_413")]; string e_45_pad_type_0 = const()[name = string("e_45_pad_type_0"), val = string("valid")]; tensor e_45_strides_0 = const()[name = string("e_45_strides_0"), val = tensor([1, 1])]; tensor e_45_pad_0 = const()[name = string("e_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_45_dilations_0 = const()[name = string("e_45_dilations_0"), val = tensor([1, 1])]; int32 e_45_groups_0 = const()[name = string("e_45_groups_0"), val = int32(1)]; tensor e_45 = conv(dilations = e_45_dilations_0, groups = e_45_groups_0, pad = e_45_pad_0, pad_type = e_45_pad_type_0, strides = e_45_strides_0, weight = model_model_layers_22_mlp_down_proj_weight_palettized, x = input_413)[name = string("e_45")]; tensor var_11190_axes_0 = const()[name = string("op_11190_axes_0"), val = tensor([2])]; tensor var_11190 = squeeze(axes = var_11190_axes_0, x = e_45)[name = string("op_11190")]; tensor var_11191 = const()[name = string("op_11191"), val = tensor([0, 2, 1])]; tensor var_11192 = transpose(perm = var_11191, x = var_11190)[name = string("transpose_30")]; tensor hidden_states_231_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = var_11192)[name = string("hidden_states_231_cast_fp16")]; int32 var_11206 = const()[name = string("op_11206"), val = int32(-1)]; fp16 const_322_promoted_to_fp16 = const()[name = string("const_322_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11208_cast_fp16 = mul(x = hidden_states_231_cast_fp16, y = const_322_promoted_to_fp16)[name = string("op_11208_cast_fp16")]; bool input_415_interleave_0 = const()[name = string("input_415_interleave_0"), val = bool(false)]; tensor input_415_cast_fp16 = concat(axis = var_11206, interleave = input_415_interleave_0, values = (hidden_states_231_cast_fp16, var_11208_cast_fp16))[name = string("input_415_cast_fp16")]; tensor normed_369_axes_0 = const()[name = string("normed_369_axes_0"), val = tensor([-1])]; fp16 var_11203_to_fp16 = const()[name = string("op_11203_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_369_cast_fp16 = layer_norm(axes = normed_369_axes_0, epsilon = var_11203_to_fp16, x = input_415_cast_fp16)[name = string("normed_369_cast_fp16")]; tensor normed_371_begin_0 = const()[name = string("normed_371_begin_0"), val = tensor([0, 0, 0])]; tensor normed_371_end_0 = const()[name = string("normed_371_end_0"), val = tensor([1, 1, 1024])]; tensor normed_371_end_mask_0 = const()[name = string("normed_371_end_mask_0"), val = tensor([true, true, false])]; tensor normed_371_cast_fp16 = slice_by_index(begin = normed_371_begin_0, end = normed_371_end_0, end_mask = normed_371_end_mask_0, x = normed_369_cast_fp16)[name = string("normed_371_cast_fp16")]; tensor const_324_promoted_to_fp16 = const()[name = string("const_324_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329569600)))]; tensor hidden_states_233_cast_fp16 = mul(x = normed_371_cast_fp16, y = const_324_promoted_to_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor var_11220 = const()[name = string("op_11220"), val = tensor([0, 2, 1])]; tensor var_11223_axes_0 = const()[name = string("op_11223_axes_0"), val = tensor([2])]; tensor var_11221_cast_fp16 = transpose(perm = var_11220, x = hidden_states_233_cast_fp16)[name = string("transpose_29")]; tensor var_11223_cast_fp16 = expand_dims(axes = var_11223_axes_0, x = var_11221_cast_fp16)[name = string("op_11223_cast_fp16")]; string var_11239_pad_type_0 = const()[name = string("op_11239_pad_type_0"), val = string("valid")]; tensor var_11239_strides_0 = const()[name = string("op_11239_strides_0"), val = tensor([1, 1])]; tensor var_11239_pad_0 = const()[name = string("op_11239_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11239_dilations_0 = const()[name = string("op_11239_dilations_0"), val = tensor([1, 1])]; int32 var_11239_groups_0 = const()[name = string("op_11239_groups_0"), val = int32(1)]; tensor var_11239 = conv(dilations = var_11239_dilations_0, groups = var_11239_groups_0, pad = var_11239_pad_0, pad_type = var_11239_pad_type_0, strides = var_11239_strides_0, weight = model_model_layers_23_self_attn_q_proj_weight_palettized, x = var_11223_cast_fp16)[name = string("op_11239")]; tensor var_11244 = const()[name = string("op_11244"), val = tensor([1, 16, 1, 128])]; tensor var_11245 = reshape(shape = var_11244, x = var_11239)[name = string("op_11245")]; string var_11261_pad_type_0 = const()[name = string("op_11261_pad_type_0"), val = string("valid")]; tensor var_11261_strides_0 = const()[name = string("op_11261_strides_0"), val = tensor([1, 1])]; tensor var_11261_pad_0 = const()[name = string("op_11261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11261_dilations_0 = const()[name = string("op_11261_dilations_0"), val = tensor([1, 1])]; int32 var_11261_groups_0 = const()[name = string("op_11261_groups_0"), val = int32(1)]; tensor var_11261 = conv(dilations = var_11261_dilations_0, groups = var_11261_groups_0, pad = var_11261_pad_0, pad_type = var_11261_pad_type_0, strides = var_11261_strides_0, weight = model_model_layers_23_self_attn_k_proj_weight_palettized, x = var_11223_cast_fp16)[name = string("op_11261")]; tensor var_11266 = const()[name = string("op_11266"), val = tensor([1, 8, 1, 128])]; tensor var_11267 = reshape(shape = var_11266, x = var_11261)[name = string("op_11267")]; string var_11283_pad_type_0 = const()[name = string("op_11283_pad_type_0"), val = string("valid")]; tensor var_11283_strides_0 = const()[name = string("op_11283_strides_0"), val = tensor([1, 1])]; tensor var_11283_pad_0 = const()[name = string("op_11283_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11283_dilations_0 = const()[name = string("op_11283_dilations_0"), val = tensor([1, 1])]; int32 var_11283_groups_0 = const()[name = string("op_11283_groups_0"), val = int32(1)]; tensor var_11283 = conv(dilations = var_11283_dilations_0, groups = var_11283_groups_0, pad = var_11283_pad_0, pad_type = var_11283_pad_type_0, strides = var_11283_strides_0, weight = model_model_layers_23_self_attn_v_proj_weight_palettized, x = var_11223_cast_fp16)[name = string("op_11283")]; tensor var_11288 = const()[name = string("op_11288"), val = tensor([1, 8, 1, 128])]; tensor var_11289 = reshape(shape = var_11288, x = var_11283)[name = string("op_11289")]; int32 var_11306 = const()[name = string("op_11306"), val = int32(-1)]; fp16 const_325_promoted = const()[name = string("const_325_promoted"), val = fp16(-0x1p+0)]; tensor var_11308 = mul(x = var_11245, y = const_325_promoted)[name = string("op_11308")]; bool input_419_interleave_0 = const()[name = string("input_419_interleave_0"), val = bool(false)]; tensor input_419 = concat(axis = var_11306, interleave = input_419_interleave_0, values = (var_11245, var_11308))[name = string("input_419")]; tensor normed_373_axes_0 = const()[name = string("normed_373_axes_0"), val = tensor([-1])]; fp16 var_11303_to_fp16 = const()[name = string("op_11303_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_373_cast_fp16 = layer_norm(axes = normed_373_axes_0, epsilon = var_11303_to_fp16, x = input_419)[name = string("normed_373_cast_fp16")]; tensor normed_375_begin_0 = const()[name = string("normed_375_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_375_end_0 = const()[name = string("normed_375_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_375_end_mask_0 = const()[name = string("normed_375_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_375 = slice_by_index(begin = normed_375_begin_0, end = normed_375_end_0, end_mask = normed_375_end_mask_0, x = normed_373_cast_fp16)[name = string("normed_375")]; tensor const_327 = const()[name = string("const_327"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329571712)))]; tensor q_47 = mul(x = normed_375, y = const_327)[name = string("q_47")]; int32 var_11328 = const()[name = string("op_11328"), val = int32(-1)]; fp16 const_328_promoted = const()[name = string("const_328_promoted"), val = fp16(-0x1p+0)]; tensor var_11330 = mul(x = var_11267, y = const_328_promoted)[name = string("op_11330")]; bool input_421_interleave_0 = const()[name = string("input_421_interleave_0"), val = bool(false)]; tensor input_421 = concat(axis = var_11328, interleave = input_421_interleave_0, values = (var_11267, var_11330))[name = string("input_421")]; tensor normed_377_axes_0 = const()[name = string("normed_377_axes_0"), val = tensor([-1])]; fp16 var_11325_to_fp16 = const()[name = string("op_11325_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_377_cast_fp16 = layer_norm(axes = normed_377_axes_0, epsilon = var_11325_to_fp16, x = input_421)[name = string("normed_377_cast_fp16")]; tensor normed_379_begin_0 = const()[name = string("normed_379_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_379_end_0 = const()[name = string("normed_379_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_379_end_mask_0 = const()[name = string("normed_379_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_379 = slice_by_index(begin = normed_379_begin_0, end = normed_379_end_0, end_mask = normed_379_end_mask_0, x = normed_377_cast_fp16)[name = string("normed_379")]; tensor const_330 = const()[name = string("const_330"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329572032)))]; tensor k_47 = mul(x = normed_379, y = const_330)[name = string("k_47")]; tensor var_11339 = mul(x = q_47, y = cos_1_cast_fp16)[name = string("op_11339")]; tensor var_11344_begin_0 = const()[name = string("op_11344_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11344_end_0 = const()[name = string("op_11344_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_11344_end_mask_0 = const()[name = string("op_11344_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11344 = slice_by_index(begin = var_11344_begin_0, end = var_11344_end_0, end_mask = var_11344_end_mask_0, x = q_47)[name = string("op_11344")]; fp16 const_331_promoted = const()[name = string("const_331_promoted"), val = fp16(-0x1p+0)]; tensor var_11345 = mul(x = var_11344, y = const_331_promoted)[name = string("op_11345")]; tensor var_11350_begin_0 = const()[name = string("op_11350_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11350_end_0 = const()[name = string("op_11350_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_11350_end_mask_0 = const()[name = string("op_11350_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11350 = slice_by_index(begin = var_11350_begin_0, end = var_11350_end_0, end_mask = var_11350_end_mask_0, x = q_47)[name = string("op_11350")]; int32 var_11352 = const()[name = string("op_11352"), val = int32(-1)]; bool var_11353_interleave_0 = const()[name = string("op_11353_interleave_0"), val = bool(false)]; tensor var_11353 = concat(axis = var_11352, interleave = var_11353_interleave_0, values = (var_11345, var_11350))[name = string("op_11353")]; tensor var_11354 = mul(x = var_11353, y = sin_1_cast_fp16)[name = string("op_11354")]; tensor query_states_93 = add(x = var_11339, y = var_11354)[name = string("query_states_93")]; tensor var_11357 = mul(x = k_47, y = cos_1_cast_fp16)[name = string("op_11357")]; tensor var_11362_begin_0 = const()[name = string("op_11362_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11362_end_0 = const()[name = string("op_11362_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_11362_end_mask_0 = const()[name = string("op_11362_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11362 = slice_by_index(begin = var_11362_begin_0, end = var_11362_end_0, end_mask = var_11362_end_mask_0, x = k_47)[name = string("op_11362")]; fp16 const_332_promoted = const()[name = string("const_332_promoted"), val = fp16(-0x1p+0)]; tensor var_11363 = mul(x = var_11362, y = const_332_promoted)[name = string("op_11363")]; tensor var_11368_begin_0 = const()[name = string("op_11368_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11368_end_0 = const()[name = string("op_11368_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_11368_end_mask_0 = const()[name = string("op_11368_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11368 = slice_by_index(begin = var_11368_begin_0, end = var_11368_end_0, end_mask = var_11368_end_mask_0, x = k_47)[name = string("op_11368")]; int32 var_11370 = const()[name = string("op_11370"), val = int32(-1)]; bool var_11371_interleave_0 = const()[name = string("op_11371_interleave_0"), val = bool(false)]; tensor var_11371 = concat(axis = var_11370, interleave = var_11371_interleave_0, values = (var_11363, var_11368))[name = string("op_11371")]; tensor var_11372 = mul(x = var_11371, y = sin_1_cast_fp16)[name = string("op_11372")]; tensor key_states_93 = add(x = var_11357, y = var_11372)[name = string("key_states_93")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_276, expand_dims_277, current_pos, expand_dims_279))[name = string("concat_186")]; tensor concat_187_values1_0 = const()[name = string("concat_187_values1_0"), val = tensor([0])]; tensor concat_187_values3_0 = const()[name = string("concat_187_values3_0"), val = tensor([0])]; int32 concat_187_axis_0 = const()[name = string("concat_187_axis_0"), val = int32(0)]; bool concat_187_interleave_0 = const()[name = string("concat_187_interleave_0"), val = bool(false)]; tensor concat_187 = concat(axis = concat_187_axis_0, interleave = concat_187_interleave_0, values = (expand_dims_280, concat_187_values1_0, var_1717, concat_187_values3_0))[name = string("concat_187")]; tensor model_model_kv_cache_0_internal_tensor_assign_47_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_cast_fp16 = slice_update(begin = concat_186, begin_mask = model_model_kv_cache_0_internal_tensor_assign_47_begin_mask_0, end = concat_187, end_mask = model_model_kv_cache_0_internal_tensor_assign_47_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_47_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_47_stride_0, update = key_states_93, x = coreml_update_state_101)[name = string("model_model_kv_cache_0_internal_tensor_assign_47_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_47_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_158_write_state")]; tensor coreml_update_state_102 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_158")]; tensor expand_dims_282 = const()[name = string("expand_dims_282"), val = tensor([51])]; tensor expand_dims_283 = const()[name = string("expand_dims_283"), val = tensor([0])]; tensor expand_dims_285 = const()[name = string("expand_dims_285"), val = tensor([0])]; tensor expand_dims_286 = const()[name = string("expand_dims_286"), val = tensor([52])]; int32 concat_190_axis_0 = const()[name = string("concat_190_axis_0"), val = int32(0)]; bool concat_190_interleave_0 = const()[name = string("concat_190_interleave_0"), val = bool(false)]; tensor concat_190 = concat(axis = concat_190_axis_0, interleave = concat_190_interleave_0, values = (expand_dims_282, expand_dims_283, current_pos, expand_dims_285))[name = string("concat_190")]; tensor concat_191_values1_0 = const()[name = string("concat_191_values1_0"), val = tensor([0])]; tensor concat_191_values3_0 = const()[name = string("concat_191_values3_0"), val = tensor([0])]; int32 concat_191_axis_0 = const()[name = string("concat_191_axis_0"), val = int32(0)]; bool concat_191_interleave_0 = const()[name = string("concat_191_interleave_0"), val = bool(false)]; tensor concat_191 = concat(axis = concat_191_axis_0, interleave = concat_191_interleave_0, values = (expand_dims_286, concat_191_values1_0, var_1717, concat_191_values3_0))[name = string("concat_191")]; tensor model_model_kv_cache_0_internal_tensor_assign_48_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_cast_fp16 = slice_update(begin = concat_190, begin_mask = model_model_kv_cache_0_internal_tensor_assign_48_begin_mask_0, end = concat_191, end_mask = model_model_kv_cache_0_internal_tensor_assign_48_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_48_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_48_stride_0, update = var_11289, x = coreml_update_state_102)[name = string("model_model_kv_cache_0_internal_tensor_assign_48_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_48_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_159_write_state")]; tensor coreml_update_state_103 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_159")]; tensor var_11427_begin_0 = const()[name = string("op_11427_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_11427_end_0 = const()[name = string("op_11427_end_0"), val = tensor([24, 8, 1536, 128])]; tensor var_11427_end_mask_0 = const()[name = string("op_11427_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11427_cast_fp16 = slice_by_index(begin = var_11427_begin_0, end = var_11427_end_0, end_mask = var_11427_end_mask_0, x = coreml_update_state_103)[name = string("op_11427_cast_fp16")]; tensor key_cache_47_axes_0 = const()[name = string("key_cache_47_axes_0"), val = tensor([0])]; tensor key_cache_47_cast_fp16 = squeeze(axes = key_cache_47_axes_0, x = var_11427_cast_fp16)[name = string("key_cache_47_cast_fp16")]; tensor var_11434_begin_0 = const()[name = string("op_11434_begin_0"), val = tensor([51, 0, 0, 0])]; tensor var_11434_end_0 = const()[name = string("op_11434_end_0"), val = tensor([52, 8, 1536, 128])]; tensor var_11434_end_mask_0 = const()[name = string("op_11434_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11434_cast_fp16 = slice_by_index(begin = var_11434_begin_0, end = var_11434_end_0, end_mask = var_11434_end_mask_0, x = coreml_update_state_103)[name = string("op_11434_cast_fp16")]; tensor value_cache_47_axes_0 = const()[name = string("value_cache_47_axes_0"), val = tensor([0])]; tensor value_cache_47_cast_fp16 = squeeze(axes = value_cache_47_axes_0, x = var_11434_cast_fp16)[name = string("value_cache_47_cast_fp16")]; tensor var_11458_axes_0 = const()[name = string("op_11458_axes_0"), val = tensor([1])]; tensor var_11458_cast_fp16 = expand_dims(axes = var_11458_axes_0, x = key_cache_47_cast_fp16)[name = string("op_11458_cast_fp16")]; tensor var_11463 = const()[name = string("op_11463"), val = tensor([1, 2, 1, 1])]; tensor value_187_cast_fp16 = tile(reps = var_11463, x = var_11458_cast_fp16)[name = string("value_187_cast_fp16")]; tensor var_11469 = const()[name = string("op_11469"), val = tensor([1, 16, 1536, 128])]; tensor key_states_95_cast_fp16 = reshape(shape = var_11469, x = value_187_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor var_11472_axes_0 = const()[name = string("op_11472_axes_0"), val = tensor([1])]; tensor var_11472_cast_fp16 = expand_dims(axes = var_11472_axes_0, x = value_cache_47_cast_fp16)[name = string("op_11472_cast_fp16")]; tensor var_11477 = const()[name = string("op_11477"), val = tensor([1, 2, 1, 1])]; tensor value_191_cast_fp16 = tile(reps = var_11477, x = var_11472_cast_fp16)[name = string("value_191_cast_fp16")]; tensor var_11483 = const()[name = string("op_11483"), val = tensor([1, 16, 1536, 128])]; tensor value_states_141_cast_fp16 = reshape(shape = var_11483, x = value_191_cast_fp16)[name = string("value_states_141_cast_fp16")]; bool var_11498_transpose_x_1 = const()[name = string("op_11498_transpose_x_1"), val = bool(false)]; bool var_11498_transpose_y_1 = const()[name = string("op_11498_transpose_y_1"), val = bool(true)]; tensor var_11498 = matmul(transpose_x = var_11498_transpose_x_1, transpose_y = var_11498_transpose_y_1, x = query_states_93, y = key_states_95_cast_fp16)[name = string("op_11498")]; fp16 var_11499_to_fp16 = const()[name = string("op_11499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_93_cast_fp16 = mul(x = var_11498, y = var_11499_to_fp16)[name = string("attention_93_cast_fp16")]; tensor attention_95_cast_fp16 = add(x = attention_93_cast_fp16, y = causal_mask)[name = string("attention_95_cast_fp16")]; int32 var_11508 = const()[name = string("op_11508"), val = int32(-1)]; tensor probabilities_47_cast_fp16 = softmax(axis = var_11508, x = attention_95_cast_fp16)[name = string("probabilities_47_cast_fp16")]; bool output_139_transpose_x_0 = const()[name = string("output_139_transpose_x_0"), val = bool(false)]; bool output_139_transpose_y_0 = const()[name = string("output_139_transpose_y_0"), val = bool(false)]; tensor output_139_cast_fp16 = matmul(transpose_x = output_139_transpose_x_0, transpose_y = output_139_transpose_y_0, x = probabilities_47_cast_fp16, y = value_states_141_cast_fp16)[name = string("output_139_cast_fp16")]; tensor var_11519_perm_0 = const()[name = string("op_11519_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_11525 = const()[name = string("op_11525"), val = tensor([1, 1, 2048])]; tensor var_11519_cast_fp16 = transpose(perm = var_11519_perm_0, x = output_139_cast_fp16)[name = string("transpose_28")]; tensor output_141_cast_fp16 = reshape(shape = var_11525, x = var_11519_cast_fp16)[name = string("output_141_cast_fp16")]; tensor var_11530 = const()[name = string("op_11530"), val = tensor([0, 2, 1])]; string var_11546_pad_type_0 = const()[name = string("op_11546_pad_type_0"), val = string("valid")]; int32 var_11546_groups_0 = const()[name = string("op_11546_groups_0"), val = int32(1)]; tensor var_11546_strides_0 = const()[name = string("op_11546_strides_0"), val = tensor([1])]; tensor var_11546_pad_0 = const()[name = string("op_11546_pad_0"), val = tensor([0, 0])]; tensor var_11546_dilations_0 = const()[name = string("op_11546_dilations_0"), val = tensor([1])]; tensor squeeze_23_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329572352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331145280))))[name = string("squeeze_23_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11531_cast_fp16 = transpose(perm = var_11530, x = output_141_cast_fp16)[name = string("transpose_27")]; tensor var_11546_cast_fp16 = conv(dilations = var_11546_dilations_0, groups = var_11546_groups_0, pad = var_11546_pad_0, pad_type = var_11546_pad_type_0, strides = var_11546_strides_0, weight = squeeze_23_cast_fp16_to_fp32_to_fp16_palettized, x = var_11531_cast_fp16)[name = string("op_11546_cast_fp16")]; tensor var_11550 = const()[name = string("op_11550"), val = tensor([0, 2, 1])]; tensor attn_output_47_cast_fp16 = transpose(perm = var_11550, x = var_11546_cast_fp16)[name = string("transpose_26")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_231_cast_fp16, y = attn_output_47_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; int32 var_11565 = const()[name = string("op_11565"), val = int32(-1)]; fp16 const_333_promoted_to_fp16 = const()[name = string("const_333_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11567_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_333_promoted_to_fp16)[name = string("op_11567_cast_fp16")]; bool input_425_interleave_0 = const()[name = string("input_425_interleave_0"), val = bool(false)]; tensor input_425_cast_fp16 = concat(axis = var_11565, interleave = input_425_interleave_0, values = (hidden_states_239_cast_fp16, var_11567_cast_fp16))[name = string("input_425_cast_fp16")]; tensor normed_381_axes_0 = const()[name = string("normed_381_axes_0"), val = tensor([-1])]; fp16 var_11562_to_fp16 = const()[name = string("op_11562_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_381_cast_fp16 = layer_norm(axes = normed_381_axes_0, epsilon = var_11562_to_fp16, x = input_425_cast_fp16)[name = string("normed_381_cast_fp16")]; tensor normed_383_begin_0 = const()[name = string("normed_383_begin_0"), val = tensor([0, 0, 0])]; tensor normed_383_end_0 = const()[name = string("normed_383_end_0"), val = tensor([1, 1, 1024])]; tensor normed_383_end_mask_0 = const()[name = string("normed_383_end_mask_0"), val = tensor([true, true, false])]; tensor normed_383_cast_fp16 = slice_by_index(begin = normed_383_begin_0, end = normed_383_end_0, end_mask = normed_383_end_mask_0, x = normed_381_cast_fp16)[name = string("normed_383_cast_fp16")]; tensor const_335_promoted_to_fp16 = const()[name = string("const_335_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331161728)))]; tensor x_93_cast_fp16 = mul(x = normed_383_cast_fp16, y = const_335_promoted_to_fp16)[name = string("x_93_cast_fp16")]; tensor var_11587 = const()[name = string("op_11587"), val = tensor([0, 2, 1])]; tensor input_427_axes_0 = const()[name = string("input_427_axes_0"), val = tensor([2])]; tensor var_11588 = transpose(perm = var_11587, x = x_93_cast_fp16)[name = string("transpose_25")]; tensor input_427 = expand_dims(axes = input_427_axes_0, x = var_11588)[name = string("input_427")]; string input_429_pad_type_0 = const()[name = string("input_429_pad_type_0"), val = string("valid")]; tensor input_429_strides_0 = const()[name = string("input_429_strides_0"), val = tensor([1, 1])]; tensor input_429_pad_0 = const()[name = string("input_429_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_429_dilations_0 = const()[name = string("input_429_dilations_0"), val = tensor([1, 1])]; int32 input_429_groups_0 = const()[name = string("input_429_groups_0"), val = int32(1)]; tensor input_429 = conv(dilations = input_429_dilations_0, groups = input_429_groups_0, pad = input_429_pad_0, pad_type = input_429_pad_type_0, strides = input_429_strides_0, weight = model_model_layers_23_mlp_gate_proj_weight_palettized, x = input_427)[name = string("input_429")]; string b_47_pad_type_0 = const()[name = string("b_47_pad_type_0"), val = string("valid")]; tensor b_47_strides_0 = const()[name = string("b_47_strides_0"), val = tensor([1, 1])]; tensor b_47_pad_0 = const()[name = string("b_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_47_dilations_0 = const()[name = string("b_47_dilations_0"), val = tensor([1, 1])]; int32 b_47_groups_0 = const()[name = string("b_47_groups_0"), val = int32(1)]; tensor b_47 = conv(dilations = b_47_dilations_0, groups = b_47_groups_0, pad = b_47_pad_0, pad_type = b_47_pad_type_0, strides = b_47_strides_0, weight = model_model_layers_23_mlp_up_proj_weight_palettized, x = input_427)[name = string("b_47")]; tensor c_47 = silu(x = input_429)[name = string("c_47")]; tensor input_431 = mul(x = c_47, y = b_47)[name = string("input_431")]; string e_47_pad_type_0 = const()[name = string("e_47_pad_type_0"), val = string("valid")]; tensor e_47_strides_0 = const()[name = string("e_47_strides_0"), val = tensor([1, 1])]; tensor e_47_pad_0 = const()[name = string("e_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_47_dilations_0 = const()[name = string("e_47_dilations_0"), val = tensor([1, 1])]; int32 e_47_groups_0 = const()[name = string("e_47_groups_0"), val = int32(1)]; tensor e_47 = conv(dilations = e_47_dilations_0, groups = e_47_groups_0, pad = e_47_pad_0, pad_type = e_47_pad_type_0, strides = e_47_strides_0, weight = model_model_layers_23_mlp_down_proj_weight_palettized, x = input_431)[name = string("e_47")]; tensor var_11610_axes_0 = const()[name = string("op_11610_axes_0"), val = tensor([2])]; tensor var_11610 = squeeze(axes = var_11610_axes_0, x = e_47)[name = string("op_11610")]; tensor var_11611 = const()[name = string("op_11611"), val = tensor([0, 2, 1])]; tensor var_11612 = transpose(perm = var_11611, x = var_11610)[name = string("transpose_24")]; tensor hidden_states_241_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = var_11612)[name = string("hidden_states_241_cast_fp16")]; int32 var_11626 = const()[name = string("op_11626"), val = int32(-1)]; fp16 const_336_promoted_to_fp16 = const()[name = string("const_336_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11628_cast_fp16 = mul(x = hidden_states_241_cast_fp16, y = const_336_promoted_to_fp16)[name = string("op_11628_cast_fp16")]; bool input_433_interleave_0 = const()[name = string("input_433_interleave_0"), val = bool(false)]; tensor input_433_cast_fp16 = concat(axis = var_11626, interleave = input_433_interleave_0, values = (hidden_states_241_cast_fp16, var_11628_cast_fp16))[name = string("input_433_cast_fp16")]; tensor normed_385_axes_0 = const()[name = string("normed_385_axes_0"), val = tensor([-1])]; fp16 var_11623_to_fp16 = const()[name = string("op_11623_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_385_cast_fp16 = layer_norm(axes = normed_385_axes_0, epsilon = var_11623_to_fp16, x = input_433_cast_fp16)[name = string("normed_385_cast_fp16")]; tensor normed_387_begin_0 = const()[name = string("normed_387_begin_0"), val = tensor([0, 0, 0])]; tensor normed_387_end_0 = const()[name = string("normed_387_end_0"), val = tensor([1, 1, 1024])]; tensor normed_387_end_mask_0 = const()[name = string("normed_387_end_mask_0"), val = tensor([true, true, false])]; tensor normed_387_cast_fp16 = slice_by_index(begin = normed_387_begin_0, end = normed_387_end_0, end_mask = normed_387_end_mask_0, x = normed_385_cast_fp16)[name = string("normed_387_cast_fp16")]; tensor const_338_promoted_to_fp16 = const()[name = string("const_338_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331163840)))]; tensor hidden_states_243_cast_fp16 = mul(x = normed_387_cast_fp16, y = const_338_promoted_to_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor var_11640 = const()[name = string("op_11640"), val = tensor([0, 2, 1])]; tensor var_11643_axes_0 = const()[name = string("op_11643_axes_0"), val = tensor([2])]; tensor var_11641_cast_fp16 = transpose(perm = var_11640, x = hidden_states_243_cast_fp16)[name = string("transpose_23")]; tensor var_11643_cast_fp16 = expand_dims(axes = var_11643_axes_0, x = var_11641_cast_fp16)[name = string("op_11643_cast_fp16")]; string var_11659_pad_type_0 = const()[name = string("op_11659_pad_type_0"), val = string("valid")]; tensor var_11659_strides_0 = const()[name = string("op_11659_strides_0"), val = tensor([1, 1])]; tensor var_11659_pad_0 = const()[name = string("op_11659_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11659_dilations_0 = const()[name = string("op_11659_dilations_0"), val = tensor([1, 1])]; int32 var_11659_groups_0 = const()[name = string("op_11659_groups_0"), val = int32(1)]; tensor var_11659 = conv(dilations = var_11659_dilations_0, groups = var_11659_groups_0, pad = var_11659_pad_0, pad_type = var_11659_pad_type_0, strides = var_11659_strides_0, weight = model_model_layers_24_self_attn_q_proj_weight_palettized, x = var_11643_cast_fp16)[name = string("op_11659")]; tensor var_11664 = const()[name = string("op_11664"), val = tensor([1, 16, 1, 128])]; tensor var_11665 = reshape(shape = var_11664, x = var_11659)[name = string("op_11665")]; string var_11681_pad_type_0 = const()[name = string("op_11681_pad_type_0"), val = string("valid")]; tensor var_11681_strides_0 = const()[name = string("op_11681_strides_0"), val = tensor([1, 1])]; tensor var_11681_pad_0 = const()[name = string("op_11681_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11681_dilations_0 = const()[name = string("op_11681_dilations_0"), val = tensor([1, 1])]; int32 var_11681_groups_0 = const()[name = string("op_11681_groups_0"), val = int32(1)]; tensor var_11681 = conv(dilations = var_11681_dilations_0, groups = var_11681_groups_0, pad = var_11681_pad_0, pad_type = var_11681_pad_type_0, strides = var_11681_strides_0, weight = model_model_layers_24_self_attn_k_proj_weight_palettized, x = var_11643_cast_fp16)[name = string("op_11681")]; tensor var_11686 = const()[name = string("op_11686"), val = tensor([1, 8, 1, 128])]; tensor var_11687 = reshape(shape = var_11686, x = var_11681)[name = string("op_11687")]; string var_11703_pad_type_0 = const()[name = string("op_11703_pad_type_0"), val = string("valid")]; tensor var_11703_strides_0 = const()[name = string("op_11703_strides_0"), val = tensor([1, 1])]; tensor var_11703_pad_0 = const()[name = string("op_11703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11703_dilations_0 = const()[name = string("op_11703_dilations_0"), val = tensor([1, 1])]; int32 var_11703_groups_0 = const()[name = string("op_11703_groups_0"), val = int32(1)]; tensor var_11703 = conv(dilations = var_11703_dilations_0, groups = var_11703_groups_0, pad = var_11703_pad_0, pad_type = var_11703_pad_type_0, strides = var_11703_strides_0, weight = model_model_layers_24_self_attn_v_proj_weight_palettized, x = var_11643_cast_fp16)[name = string("op_11703")]; tensor var_11708 = const()[name = string("op_11708"), val = tensor([1, 8, 1, 128])]; tensor var_11709 = reshape(shape = var_11708, x = var_11703)[name = string("op_11709")]; int32 var_11726 = const()[name = string("op_11726"), val = int32(-1)]; fp16 const_339_promoted = const()[name = string("const_339_promoted"), val = fp16(-0x1p+0)]; tensor var_11728 = mul(x = var_11665, y = const_339_promoted)[name = string("op_11728")]; bool input_437_interleave_0 = const()[name = string("input_437_interleave_0"), val = bool(false)]; tensor input_437 = concat(axis = var_11726, interleave = input_437_interleave_0, values = (var_11665, var_11728))[name = string("input_437")]; tensor normed_389_axes_0 = const()[name = string("normed_389_axes_0"), val = tensor([-1])]; fp16 var_11723_to_fp16 = const()[name = string("op_11723_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_389_cast_fp16 = layer_norm(axes = normed_389_axes_0, epsilon = var_11723_to_fp16, x = input_437)[name = string("normed_389_cast_fp16")]; tensor normed_391_begin_0 = const()[name = string("normed_391_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_391_end_0 = const()[name = string("normed_391_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_391_end_mask_0 = const()[name = string("normed_391_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_391 = slice_by_index(begin = normed_391_begin_0, end = normed_391_end_0, end_mask = normed_391_end_mask_0, x = normed_389_cast_fp16)[name = string("normed_391")]; tensor const_341 = const()[name = string("const_341"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331165952)))]; tensor q_49 = mul(x = normed_391, y = const_341)[name = string("q_49")]; int32 var_11748 = const()[name = string("op_11748"), val = int32(-1)]; fp16 const_342_promoted = const()[name = string("const_342_promoted"), val = fp16(-0x1p+0)]; tensor var_11750 = mul(x = var_11687, y = const_342_promoted)[name = string("op_11750")]; bool input_439_interleave_0 = const()[name = string("input_439_interleave_0"), val = bool(false)]; tensor input_439 = concat(axis = var_11748, interleave = input_439_interleave_0, values = (var_11687, var_11750))[name = string("input_439")]; tensor normed_393_axes_0 = const()[name = string("normed_393_axes_0"), val = tensor([-1])]; fp16 var_11745_to_fp16 = const()[name = string("op_11745_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_393_cast_fp16 = layer_norm(axes = normed_393_axes_0, epsilon = var_11745_to_fp16, x = input_439)[name = string("normed_393_cast_fp16")]; tensor normed_395_begin_0 = const()[name = string("normed_395_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_395_end_0 = const()[name = string("normed_395_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_395_end_mask_0 = const()[name = string("normed_395_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_395 = slice_by_index(begin = normed_395_begin_0, end = normed_395_end_0, end_mask = normed_395_end_mask_0, x = normed_393_cast_fp16)[name = string("normed_395")]; tensor const_344 = const()[name = string("const_344"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331166272)))]; tensor k_49 = mul(x = normed_395, y = const_344)[name = string("k_49")]; tensor var_11759 = mul(x = q_49, y = cos_1_cast_fp16)[name = string("op_11759")]; tensor var_11764_begin_0 = const()[name = string("op_11764_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11764_end_0 = const()[name = string("op_11764_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_11764_end_mask_0 = const()[name = string("op_11764_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11764 = slice_by_index(begin = var_11764_begin_0, end = var_11764_end_0, end_mask = var_11764_end_mask_0, x = q_49)[name = string("op_11764")]; fp16 const_345_promoted = const()[name = string("const_345_promoted"), val = fp16(-0x1p+0)]; tensor var_11765 = mul(x = var_11764, y = const_345_promoted)[name = string("op_11765")]; tensor var_11770_begin_0 = const()[name = string("op_11770_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11770_end_0 = const()[name = string("op_11770_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_11770_end_mask_0 = const()[name = string("op_11770_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11770 = slice_by_index(begin = var_11770_begin_0, end = var_11770_end_0, end_mask = var_11770_end_mask_0, x = q_49)[name = string("op_11770")]; int32 var_11772 = const()[name = string("op_11772"), val = int32(-1)]; bool var_11773_interleave_0 = const()[name = string("op_11773_interleave_0"), val = bool(false)]; tensor var_11773 = concat(axis = var_11772, interleave = var_11773_interleave_0, values = (var_11765, var_11770))[name = string("op_11773")]; tensor var_11774 = mul(x = var_11773, y = sin_1_cast_fp16)[name = string("op_11774")]; tensor query_states_97 = add(x = var_11759, y = var_11774)[name = string("query_states_97")]; tensor var_11777 = mul(x = k_49, y = cos_1_cast_fp16)[name = string("op_11777")]; tensor var_11782_begin_0 = const()[name = string("op_11782_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11782_end_0 = const()[name = string("op_11782_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_11782_end_mask_0 = const()[name = string("op_11782_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11782 = slice_by_index(begin = var_11782_begin_0, end = var_11782_end_0, end_mask = var_11782_end_mask_0, x = k_49)[name = string("op_11782")]; fp16 const_346_promoted = const()[name = string("const_346_promoted"), val = fp16(-0x1p+0)]; tensor var_11783 = mul(x = var_11782, y = const_346_promoted)[name = string("op_11783")]; tensor var_11788_begin_0 = const()[name = string("op_11788_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11788_end_0 = const()[name = string("op_11788_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_11788_end_mask_0 = const()[name = string("op_11788_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11788 = slice_by_index(begin = var_11788_begin_0, end = var_11788_end_0, end_mask = var_11788_end_mask_0, x = k_49)[name = string("op_11788")]; int32 var_11790 = const()[name = string("op_11790"), val = int32(-1)]; bool var_11791_interleave_0 = const()[name = string("op_11791_interleave_0"), val = bool(false)]; tensor var_11791 = concat(axis = var_11790, interleave = var_11791_interleave_0, values = (var_11783, var_11788))[name = string("op_11791")]; tensor var_11792 = mul(x = var_11791, y = sin_1_cast_fp16)[name = string("op_11792")]; tensor key_states_97 = add(x = var_11777, y = var_11792)[name = string("key_states_97")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; int32 concat_194_axis_0 = const()[name = string("concat_194_axis_0"), val = int32(0)]; bool concat_194_interleave_0 = const()[name = string("concat_194_interleave_0"), val = bool(false)]; tensor concat_194 = concat(axis = concat_194_axis_0, interleave = concat_194_interleave_0, values = (expand_dims_288, expand_dims_289, current_pos, expand_dims_291))[name = string("concat_194")]; tensor concat_195_values1_0 = const()[name = string("concat_195_values1_0"), val = tensor([0])]; tensor concat_195_values3_0 = const()[name = string("concat_195_values3_0"), val = tensor([0])]; int32 concat_195_axis_0 = const()[name = string("concat_195_axis_0"), val = int32(0)]; bool concat_195_interleave_0 = const()[name = string("concat_195_interleave_0"), val = bool(false)]; tensor concat_195 = concat(axis = concat_195_axis_0, interleave = concat_195_interleave_0, values = (expand_dims_292, concat_195_values1_0, var_1717, concat_195_values3_0))[name = string("concat_195")]; tensor model_model_kv_cache_0_internal_tensor_assign_49_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_cast_fp16 = slice_update(begin = concat_194, begin_mask = model_model_kv_cache_0_internal_tensor_assign_49_begin_mask_0, end = concat_195, end_mask = model_model_kv_cache_0_internal_tensor_assign_49_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_49_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_49_stride_0, update = key_states_97, x = coreml_update_state_103)[name = string("model_model_kv_cache_0_internal_tensor_assign_49_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_49_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_160_write_state")]; tensor coreml_update_state_104 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_160")]; tensor expand_dims_294 = const()[name = string("expand_dims_294"), val = tensor([52])]; tensor expand_dims_295 = const()[name = string("expand_dims_295"), val = tensor([0])]; tensor expand_dims_297 = const()[name = string("expand_dims_297"), val = tensor([0])]; tensor expand_dims_298 = const()[name = string("expand_dims_298"), val = tensor([53])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_294, expand_dims_295, current_pos, expand_dims_297))[name = string("concat_198")]; tensor concat_199_values1_0 = const()[name = string("concat_199_values1_0"), val = tensor([0])]; tensor concat_199_values3_0 = const()[name = string("concat_199_values3_0"), val = tensor([0])]; int32 concat_199_axis_0 = const()[name = string("concat_199_axis_0"), val = int32(0)]; bool concat_199_interleave_0 = const()[name = string("concat_199_interleave_0"), val = bool(false)]; tensor concat_199 = concat(axis = concat_199_axis_0, interleave = concat_199_interleave_0, values = (expand_dims_298, concat_199_values1_0, var_1717, concat_199_values3_0))[name = string("concat_199")]; tensor model_model_kv_cache_0_internal_tensor_assign_50_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_cast_fp16 = slice_update(begin = concat_198, begin_mask = model_model_kv_cache_0_internal_tensor_assign_50_begin_mask_0, end = concat_199, end_mask = model_model_kv_cache_0_internal_tensor_assign_50_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_50_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_50_stride_0, update = var_11709, x = coreml_update_state_104)[name = string("model_model_kv_cache_0_internal_tensor_assign_50_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_50_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_161_write_state")]; tensor coreml_update_state_105 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_161")]; tensor var_11847_begin_0 = const()[name = string("op_11847_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_11847_end_0 = const()[name = string("op_11847_end_0"), val = tensor([25, 8, 1536, 128])]; tensor var_11847_end_mask_0 = const()[name = string("op_11847_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11847_cast_fp16 = slice_by_index(begin = var_11847_begin_0, end = var_11847_end_0, end_mask = var_11847_end_mask_0, x = coreml_update_state_105)[name = string("op_11847_cast_fp16")]; tensor key_cache_49_axes_0 = const()[name = string("key_cache_49_axes_0"), val = tensor([0])]; tensor key_cache_49_cast_fp16 = squeeze(axes = key_cache_49_axes_0, x = var_11847_cast_fp16)[name = string("key_cache_49_cast_fp16")]; tensor var_11854_begin_0 = const()[name = string("op_11854_begin_0"), val = tensor([52, 0, 0, 0])]; tensor var_11854_end_0 = const()[name = string("op_11854_end_0"), val = tensor([53, 8, 1536, 128])]; tensor var_11854_end_mask_0 = const()[name = string("op_11854_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11854_cast_fp16 = slice_by_index(begin = var_11854_begin_0, end = var_11854_end_0, end_mask = var_11854_end_mask_0, x = coreml_update_state_105)[name = string("op_11854_cast_fp16")]; tensor value_cache_49_axes_0 = const()[name = string("value_cache_49_axes_0"), val = tensor([0])]; tensor value_cache_49_cast_fp16 = squeeze(axes = value_cache_49_axes_0, x = var_11854_cast_fp16)[name = string("value_cache_49_cast_fp16")]; tensor var_11878_axes_0 = const()[name = string("op_11878_axes_0"), val = tensor([1])]; tensor var_11878_cast_fp16 = expand_dims(axes = var_11878_axes_0, x = key_cache_49_cast_fp16)[name = string("op_11878_cast_fp16")]; tensor var_11883 = const()[name = string("op_11883"), val = tensor([1, 2, 1, 1])]; tensor value_195_cast_fp16 = tile(reps = var_11883, x = var_11878_cast_fp16)[name = string("value_195_cast_fp16")]; tensor var_11889 = const()[name = string("op_11889"), val = tensor([1, 16, 1536, 128])]; tensor key_states_99_cast_fp16 = reshape(shape = var_11889, x = value_195_cast_fp16)[name = string("key_states_99_cast_fp16")]; tensor var_11892_axes_0 = const()[name = string("op_11892_axes_0"), val = tensor([1])]; tensor var_11892_cast_fp16 = expand_dims(axes = var_11892_axes_0, x = value_cache_49_cast_fp16)[name = string("op_11892_cast_fp16")]; tensor var_11897 = const()[name = string("op_11897"), val = tensor([1, 2, 1, 1])]; tensor value_199_cast_fp16 = tile(reps = var_11897, x = var_11892_cast_fp16)[name = string("value_199_cast_fp16")]; tensor var_11903 = const()[name = string("op_11903"), val = tensor([1, 16, 1536, 128])]; tensor value_states_147_cast_fp16 = reshape(shape = var_11903, x = value_199_cast_fp16)[name = string("value_states_147_cast_fp16")]; bool var_11918_transpose_x_1 = const()[name = string("op_11918_transpose_x_1"), val = bool(false)]; bool var_11918_transpose_y_1 = const()[name = string("op_11918_transpose_y_1"), val = bool(true)]; tensor var_11918 = matmul(transpose_x = var_11918_transpose_x_1, transpose_y = var_11918_transpose_y_1, x = query_states_97, y = key_states_99_cast_fp16)[name = string("op_11918")]; fp16 var_11919_to_fp16 = const()[name = string("op_11919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_97_cast_fp16 = mul(x = var_11918, y = var_11919_to_fp16)[name = string("attention_97_cast_fp16")]; tensor attention_99_cast_fp16 = add(x = attention_97_cast_fp16, y = causal_mask)[name = string("attention_99_cast_fp16")]; int32 var_11928 = const()[name = string("op_11928"), val = int32(-1)]; tensor probabilities_49_cast_fp16 = softmax(axis = var_11928, x = attention_99_cast_fp16)[name = string("probabilities_49_cast_fp16")]; bool output_145_transpose_x_0 = const()[name = string("output_145_transpose_x_0"), val = bool(false)]; bool output_145_transpose_y_0 = const()[name = string("output_145_transpose_y_0"), val = bool(false)]; tensor output_145_cast_fp16 = matmul(transpose_x = output_145_transpose_x_0, transpose_y = output_145_transpose_y_0, x = probabilities_49_cast_fp16, y = value_states_147_cast_fp16)[name = string("output_145_cast_fp16")]; tensor var_11939_perm_0 = const()[name = string("op_11939_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_11945 = const()[name = string("op_11945"), val = tensor([1, 1, 2048])]; tensor var_11939_cast_fp16 = transpose(perm = var_11939_perm_0, x = output_145_cast_fp16)[name = string("transpose_22")]; tensor output_147_cast_fp16 = reshape(shape = var_11945, x = var_11939_cast_fp16)[name = string("output_147_cast_fp16")]; tensor var_11950 = const()[name = string("op_11950"), val = tensor([0, 2, 1])]; string var_11966_pad_type_0 = const()[name = string("op_11966_pad_type_0"), val = string("valid")]; int32 var_11966_groups_0 = const()[name = string("op_11966_groups_0"), val = int32(1)]; tensor var_11966_strides_0 = const()[name = string("op_11966_strides_0"), val = tensor([1])]; tensor var_11966_pad_0 = const()[name = string("op_11966_pad_0"), val = tensor([0, 0])]; tensor var_11966_dilations_0 = const()[name = string("op_11966_dilations_0"), val = tensor([1])]; tensor squeeze_24_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331166592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332739520))))[name = string("squeeze_24_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11951_cast_fp16 = transpose(perm = var_11950, x = output_147_cast_fp16)[name = string("transpose_21")]; tensor var_11966_cast_fp16 = conv(dilations = var_11966_dilations_0, groups = var_11966_groups_0, pad = var_11966_pad_0, pad_type = var_11966_pad_type_0, strides = var_11966_strides_0, weight = squeeze_24_cast_fp16_to_fp32_to_fp16_palettized, x = var_11951_cast_fp16)[name = string("op_11966_cast_fp16")]; tensor var_11970 = const()[name = string("op_11970"), val = tensor([0, 2, 1])]; tensor attn_output_49_cast_fp16 = transpose(perm = var_11970, x = var_11966_cast_fp16)[name = string("transpose_20")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_241_cast_fp16, y = attn_output_49_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; int32 var_11985 = const()[name = string("op_11985"), val = int32(-1)]; fp16 const_347_promoted_to_fp16 = const()[name = string("const_347_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11987_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_347_promoted_to_fp16)[name = string("op_11987_cast_fp16")]; bool input_443_interleave_0 = const()[name = string("input_443_interleave_0"), val = bool(false)]; tensor input_443_cast_fp16 = concat(axis = var_11985, interleave = input_443_interleave_0, values = (hidden_states_249_cast_fp16, var_11987_cast_fp16))[name = string("input_443_cast_fp16")]; tensor normed_397_axes_0 = const()[name = string("normed_397_axes_0"), val = tensor([-1])]; fp16 var_11982_to_fp16 = const()[name = string("op_11982_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_397_cast_fp16 = layer_norm(axes = normed_397_axes_0, epsilon = var_11982_to_fp16, x = input_443_cast_fp16)[name = string("normed_397_cast_fp16")]; tensor normed_399_begin_0 = const()[name = string("normed_399_begin_0"), val = tensor([0, 0, 0])]; tensor normed_399_end_0 = const()[name = string("normed_399_end_0"), val = tensor([1, 1, 1024])]; tensor normed_399_end_mask_0 = const()[name = string("normed_399_end_mask_0"), val = tensor([true, true, false])]; tensor normed_399_cast_fp16 = slice_by_index(begin = normed_399_begin_0, end = normed_399_end_0, end_mask = normed_399_end_mask_0, x = normed_397_cast_fp16)[name = string("normed_399_cast_fp16")]; tensor const_349_promoted_to_fp16 = const()[name = string("const_349_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332755968)))]; tensor x_97_cast_fp16 = mul(x = normed_399_cast_fp16, y = const_349_promoted_to_fp16)[name = string("x_97_cast_fp16")]; tensor var_12007 = const()[name = string("op_12007"), val = tensor([0, 2, 1])]; tensor input_445_axes_0 = const()[name = string("input_445_axes_0"), val = tensor([2])]; tensor var_12008 = transpose(perm = var_12007, x = x_97_cast_fp16)[name = string("transpose_19")]; tensor input_445 = expand_dims(axes = input_445_axes_0, x = var_12008)[name = string("input_445")]; string input_447_pad_type_0 = const()[name = string("input_447_pad_type_0"), val = string("valid")]; tensor input_447_strides_0 = const()[name = string("input_447_strides_0"), val = tensor([1, 1])]; tensor input_447_pad_0 = const()[name = string("input_447_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_447_dilations_0 = const()[name = string("input_447_dilations_0"), val = tensor([1, 1])]; int32 input_447_groups_0 = const()[name = string("input_447_groups_0"), val = int32(1)]; tensor input_447 = conv(dilations = input_447_dilations_0, groups = input_447_groups_0, pad = input_447_pad_0, pad_type = input_447_pad_type_0, strides = input_447_strides_0, weight = model_model_layers_24_mlp_gate_proj_weight_palettized, x = input_445)[name = string("input_447")]; string b_49_pad_type_0 = const()[name = string("b_49_pad_type_0"), val = string("valid")]; tensor b_49_strides_0 = const()[name = string("b_49_strides_0"), val = tensor([1, 1])]; tensor b_49_pad_0 = const()[name = string("b_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_49_dilations_0 = const()[name = string("b_49_dilations_0"), val = tensor([1, 1])]; int32 b_49_groups_0 = const()[name = string("b_49_groups_0"), val = int32(1)]; tensor b_49 = conv(dilations = b_49_dilations_0, groups = b_49_groups_0, pad = b_49_pad_0, pad_type = b_49_pad_type_0, strides = b_49_strides_0, weight = model_model_layers_24_mlp_up_proj_weight_palettized, x = input_445)[name = string("b_49")]; tensor c_49 = silu(x = input_447)[name = string("c_49")]; tensor input_449 = mul(x = c_49, y = b_49)[name = string("input_449")]; string e_49_pad_type_0 = const()[name = string("e_49_pad_type_0"), val = string("valid")]; tensor e_49_strides_0 = const()[name = string("e_49_strides_0"), val = tensor([1, 1])]; tensor e_49_pad_0 = const()[name = string("e_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_49_dilations_0 = const()[name = string("e_49_dilations_0"), val = tensor([1, 1])]; int32 e_49_groups_0 = const()[name = string("e_49_groups_0"), val = int32(1)]; tensor e_49 = conv(dilations = e_49_dilations_0, groups = e_49_groups_0, pad = e_49_pad_0, pad_type = e_49_pad_type_0, strides = e_49_strides_0, weight = model_model_layers_24_mlp_down_proj_weight_palettized, x = input_449)[name = string("e_49")]; tensor var_12030_axes_0 = const()[name = string("op_12030_axes_0"), val = tensor([2])]; tensor var_12030 = squeeze(axes = var_12030_axes_0, x = e_49)[name = string("op_12030")]; tensor var_12031 = const()[name = string("op_12031"), val = tensor([0, 2, 1])]; tensor var_12032 = transpose(perm = var_12031, x = var_12030)[name = string("transpose_18")]; tensor hidden_states_251_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = var_12032)[name = string("hidden_states_251_cast_fp16")]; int32 var_12046 = const()[name = string("op_12046"), val = int32(-1)]; fp16 const_350_promoted_to_fp16 = const()[name = string("const_350_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12048_cast_fp16 = mul(x = hidden_states_251_cast_fp16, y = const_350_promoted_to_fp16)[name = string("op_12048_cast_fp16")]; bool input_451_interleave_0 = const()[name = string("input_451_interleave_0"), val = bool(false)]; tensor input_451_cast_fp16 = concat(axis = var_12046, interleave = input_451_interleave_0, values = (hidden_states_251_cast_fp16, var_12048_cast_fp16))[name = string("input_451_cast_fp16")]; tensor normed_401_axes_0 = const()[name = string("normed_401_axes_0"), val = tensor([-1])]; fp16 var_12043_to_fp16 = const()[name = string("op_12043_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_401_cast_fp16 = layer_norm(axes = normed_401_axes_0, epsilon = var_12043_to_fp16, x = input_451_cast_fp16)[name = string("normed_401_cast_fp16")]; tensor normed_403_begin_0 = const()[name = string("normed_403_begin_0"), val = tensor([0, 0, 0])]; tensor normed_403_end_0 = const()[name = string("normed_403_end_0"), val = tensor([1, 1, 1024])]; tensor normed_403_end_mask_0 = const()[name = string("normed_403_end_mask_0"), val = tensor([true, true, false])]; tensor normed_403_cast_fp16 = slice_by_index(begin = normed_403_begin_0, end = normed_403_end_0, end_mask = normed_403_end_mask_0, x = normed_401_cast_fp16)[name = string("normed_403_cast_fp16")]; tensor const_352_promoted_to_fp16 = const()[name = string("const_352_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332758080)))]; tensor hidden_states_253_cast_fp16 = mul(x = normed_403_cast_fp16, y = const_352_promoted_to_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor var_12060 = const()[name = string("op_12060"), val = tensor([0, 2, 1])]; tensor var_12063_axes_0 = const()[name = string("op_12063_axes_0"), val = tensor([2])]; tensor var_12061_cast_fp16 = transpose(perm = var_12060, x = hidden_states_253_cast_fp16)[name = string("transpose_17")]; tensor var_12063_cast_fp16 = expand_dims(axes = var_12063_axes_0, x = var_12061_cast_fp16)[name = string("op_12063_cast_fp16")]; string var_12079_pad_type_0 = const()[name = string("op_12079_pad_type_0"), val = string("valid")]; tensor var_12079_strides_0 = const()[name = string("op_12079_strides_0"), val = tensor([1, 1])]; tensor var_12079_pad_0 = const()[name = string("op_12079_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12079_dilations_0 = const()[name = string("op_12079_dilations_0"), val = tensor([1, 1])]; int32 var_12079_groups_0 = const()[name = string("op_12079_groups_0"), val = int32(1)]; tensor var_12079 = conv(dilations = var_12079_dilations_0, groups = var_12079_groups_0, pad = var_12079_pad_0, pad_type = var_12079_pad_type_0, strides = var_12079_strides_0, weight = model_model_layers_25_self_attn_q_proj_weight_palettized, x = var_12063_cast_fp16)[name = string("op_12079")]; tensor var_12084 = const()[name = string("op_12084"), val = tensor([1, 16, 1, 128])]; tensor var_12085 = reshape(shape = var_12084, x = var_12079)[name = string("op_12085")]; string var_12101_pad_type_0 = const()[name = string("op_12101_pad_type_0"), val = string("valid")]; tensor var_12101_strides_0 = const()[name = string("op_12101_strides_0"), val = tensor([1, 1])]; tensor var_12101_pad_0 = const()[name = string("op_12101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12101_dilations_0 = const()[name = string("op_12101_dilations_0"), val = tensor([1, 1])]; int32 var_12101_groups_0 = const()[name = string("op_12101_groups_0"), val = int32(1)]; tensor var_12101 = conv(dilations = var_12101_dilations_0, groups = var_12101_groups_0, pad = var_12101_pad_0, pad_type = var_12101_pad_type_0, strides = var_12101_strides_0, weight = model_model_layers_25_self_attn_k_proj_weight_palettized, x = var_12063_cast_fp16)[name = string("op_12101")]; tensor var_12106 = const()[name = string("op_12106"), val = tensor([1, 8, 1, 128])]; tensor var_12107 = reshape(shape = var_12106, x = var_12101)[name = string("op_12107")]; string var_12123_pad_type_0 = const()[name = string("op_12123_pad_type_0"), val = string("valid")]; tensor var_12123_strides_0 = const()[name = string("op_12123_strides_0"), val = tensor([1, 1])]; tensor var_12123_pad_0 = const()[name = string("op_12123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12123_dilations_0 = const()[name = string("op_12123_dilations_0"), val = tensor([1, 1])]; int32 var_12123_groups_0 = const()[name = string("op_12123_groups_0"), val = int32(1)]; tensor var_12123 = conv(dilations = var_12123_dilations_0, groups = var_12123_groups_0, pad = var_12123_pad_0, pad_type = var_12123_pad_type_0, strides = var_12123_strides_0, weight = model_model_layers_25_self_attn_v_proj_weight_palettized, x = var_12063_cast_fp16)[name = string("op_12123")]; tensor var_12128 = const()[name = string("op_12128"), val = tensor([1, 8, 1, 128])]; tensor var_12129 = reshape(shape = var_12128, x = var_12123)[name = string("op_12129")]; int32 var_12146 = const()[name = string("op_12146"), val = int32(-1)]; fp16 const_353_promoted = const()[name = string("const_353_promoted"), val = fp16(-0x1p+0)]; tensor var_12148 = mul(x = var_12085, y = const_353_promoted)[name = string("op_12148")]; bool input_455_interleave_0 = const()[name = string("input_455_interleave_0"), val = bool(false)]; tensor input_455 = concat(axis = var_12146, interleave = input_455_interleave_0, values = (var_12085, var_12148))[name = string("input_455")]; tensor normed_405_axes_0 = const()[name = string("normed_405_axes_0"), val = tensor([-1])]; fp16 var_12143_to_fp16 = const()[name = string("op_12143_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_405_cast_fp16 = layer_norm(axes = normed_405_axes_0, epsilon = var_12143_to_fp16, x = input_455)[name = string("normed_405_cast_fp16")]; tensor normed_407_begin_0 = const()[name = string("normed_407_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_407_end_0 = const()[name = string("normed_407_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_407_end_mask_0 = const()[name = string("normed_407_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_407 = slice_by_index(begin = normed_407_begin_0, end = normed_407_end_0, end_mask = normed_407_end_mask_0, x = normed_405_cast_fp16)[name = string("normed_407")]; tensor const_355 = const()[name = string("const_355"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332760192)))]; tensor q_51 = mul(x = normed_407, y = const_355)[name = string("q_51")]; int32 var_12168 = const()[name = string("op_12168"), val = int32(-1)]; fp16 const_356_promoted = const()[name = string("const_356_promoted"), val = fp16(-0x1p+0)]; tensor var_12170 = mul(x = var_12107, y = const_356_promoted)[name = string("op_12170")]; bool input_457_interleave_0 = const()[name = string("input_457_interleave_0"), val = bool(false)]; tensor input_457 = concat(axis = var_12168, interleave = input_457_interleave_0, values = (var_12107, var_12170))[name = string("input_457")]; tensor normed_409_axes_0 = const()[name = string("normed_409_axes_0"), val = tensor([-1])]; fp16 var_12165_to_fp16 = const()[name = string("op_12165_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_409_cast_fp16 = layer_norm(axes = normed_409_axes_0, epsilon = var_12165_to_fp16, x = input_457)[name = string("normed_409_cast_fp16")]; tensor normed_411_begin_0 = const()[name = string("normed_411_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_411_end_0 = const()[name = string("normed_411_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_411_end_mask_0 = const()[name = string("normed_411_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_411 = slice_by_index(begin = normed_411_begin_0, end = normed_411_end_0, end_mask = normed_411_end_mask_0, x = normed_409_cast_fp16)[name = string("normed_411")]; tensor const_358 = const()[name = string("const_358"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332760512)))]; tensor k_51 = mul(x = normed_411, y = const_358)[name = string("k_51")]; tensor var_12179 = mul(x = q_51, y = cos_1_cast_fp16)[name = string("op_12179")]; tensor var_12184_begin_0 = const()[name = string("op_12184_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12184_end_0 = const()[name = string("op_12184_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_12184_end_mask_0 = const()[name = string("op_12184_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12184 = slice_by_index(begin = var_12184_begin_0, end = var_12184_end_0, end_mask = var_12184_end_mask_0, x = q_51)[name = string("op_12184")]; fp16 const_359_promoted = const()[name = string("const_359_promoted"), val = fp16(-0x1p+0)]; tensor var_12185 = mul(x = var_12184, y = const_359_promoted)[name = string("op_12185")]; tensor var_12190_begin_0 = const()[name = string("op_12190_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12190_end_0 = const()[name = string("op_12190_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_12190_end_mask_0 = const()[name = string("op_12190_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12190 = slice_by_index(begin = var_12190_begin_0, end = var_12190_end_0, end_mask = var_12190_end_mask_0, x = q_51)[name = string("op_12190")]; int32 var_12192 = const()[name = string("op_12192"), val = int32(-1)]; bool var_12193_interleave_0 = const()[name = string("op_12193_interleave_0"), val = bool(false)]; tensor var_12193 = concat(axis = var_12192, interleave = var_12193_interleave_0, values = (var_12185, var_12190))[name = string("op_12193")]; tensor var_12194 = mul(x = var_12193, y = sin_1_cast_fp16)[name = string("op_12194")]; tensor query_states_101 = add(x = var_12179, y = var_12194)[name = string("query_states_101")]; tensor var_12197 = mul(x = k_51, y = cos_1_cast_fp16)[name = string("op_12197")]; tensor var_12202_begin_0 = const()[name = string("op_12202_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12202_end_0 = const()[name = string("op_12202_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_12202_end_mask_0 = const()[name = string("op_12202_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12202 = slice_by_index(begin = var_12202_begin_0, end = var_12202_end_0, end_mask = var_12202_end_mask_0, x = k_51)[name = string("op_12202")]; fp16 const_360_promoted = const()[name = string("const_360_promoted"), val = fp16(-0x1p+0)]; tensor var_12203 = mul(x = var_12202, y = const_360_promoted)[name = string("op_12203")]; tensor var_12208_begin_0 = const()[name = string("op_12208_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12208_end_0 = const()[name = string("op_12208_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_12208_end_mask_0 = const()[name = string("op_12208_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12208 = slice_by_index(begin = var_12208_begin_0, end = var_12208_end_0, end_mask = var_12208_end_mask_0, x = k_51)[name = string("op_12208")]; int32 var_12210 = const()[name = string("op_12210"), val = int32(-1)]; bool var_12211_interleave_0 = const()[name = string("op_12211_interleave_0"), val = bool(false)]; tensor var_12211 = concat(axis = var_12210, interleave = var_12211_interleave_0, values = (var_12203, var_12208))[name = string("op_12211")]; tensor var_12212 = mul(x = var_12211, y = sin_1_cast_fp16)[name = string("op_12212")]; tensor key_states_101 = add(x = var_12197, y = var_12212)[name = string("key_states_101")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; int32 concat_202_axis_0 = const()[name = string("concat_202_axis_0"), val = int32(0)]; bool concat_202_interleave_0 = const()[name = string("concat_202_interleave_0"), val = bool(false)]; tensor concat_202 = concat(axis = concat_202_axis_0, interleave = concat_202_interleave_0, values = (expand_dims_300, expand_dims_301, current_pos, expand_dims_303))[name = string("concat_202")]; tensor concat_203_values1_0 = const()[name = string("concat_203_values1_0"), val = tensor([0])]; tensor concat_203_values3_0 = const()[name = string("concat_203_values3_0"), val = tensor([0])]; int32 concat_203_axis_0 = const()[name = string("concat_203_axis_0"), val = int32(0)]; bool concat_203_interleave_0 = const()[name = string("concat_203_interleave_0"), val = bool(false)]; tensor concat_203 = concat(axis = concat_203_axis_0, interleave = concat_203_interleave_0, values = (expand_dims_304, concat_203_values1_0, var_1717, concat_203_values3_0))[name = string("concat_203")]; tensor model_model_kv_cache_0_internal_tensor_assign_51_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_cast_fp16 = slice_update(begin = concat_202, begin_mask = model_model_kv_cache_0_internal_tensor_assign_51_begin_mask_0, end = concat_203, end_mask = model_model_kv_cache_0_internal_tensor_assign_51_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_51_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_51_stride_0, update = key_states_101, x = coreml_update_state_105)[name = string("model_model_kv_cache_0_internal_tensor_assign_51_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_51_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_162_write_state")]; tensor coreml_update_state_106 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_162")]; tensor expand_dims_306 = const()[name = string("expand_dims_306"), val = tensor([53])]; tensor expand_dims_307 = const()[name = string("expand_dims_307"), val = tensor([0])]; tensor expand_dims_309 = const()[name = string("expand_dims_309"), val = tensor([0])]; tensor expand_dims_310 = const()[name = string("expand_dims_310"), val = tensor([54])]; int32 concat_206_axis_0 = const()[name = string("concat_206_axis_0"), val = int32(0)]; bool concat_206_interleave_0 = const()[name = string("concat_206_interleave_0"), val = bool(false)]; tensor concat_206 = concat(axis = concat_206_axis_0, interleave = concat_206_interleave_0, values = (expand_dims_306, expand_dims_307, current_pos, expand_dims_309))[name = string("concat_206")]; tensor concat_207_values1_0 = const()[name = string("concat_207_values1_0"), val = tensor([0])]; tensor concat_207_values3_0 = const()[name = string("concat_207_values3_0"), val = tensor([0])]; int32 concat_207_axis_0 = const()[name = string("concat_207_axis_0"), val = int32(0)]; bool concat_207_interleave_0 = const()[name = string("concat_207_interleave_0"), val = bool(false)]; tensor concat_207 = concat(axis = concat_207_axis_0, interleave = concat_207_interleave_0, values = (expand_dims_310, concat_207_values1_0, var_1717, concat_207_values3_0))[name = string("concat_207")]; tensor model_model_kv_cache_0_internal_tensor_assign_52_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_cast_fp16 = slice_update(begin = concat_206, begin_mask = model_model_kv_cache_0_internal_tensor_assign_52_begin_mask_0, end = concat_207, end_mask = model_model_kv_cache_0_internal_tensor_assign_52_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_52_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_52_stride_0, update = var_12129, x = coreml_update_state_106)[name = string("model_model_kv_cache_0_internal_tensor_assign_52_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_52_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_163_write_state")]; tensor coreml_update_state_107 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_163")]; tensor var_12267_begin_0 = const()[name = string("op_12267_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_12267_end_0 = const()[name = string("op_12267_end_0"), val = tensor([26, 8, 1536, 128])]; tensor var_12267_end_mask_0 = const()[name = string("op_12267_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12267_cast_fp16 = slice_by_index(begin = var_12267_begin_0, end = var_12267_end_0, end_mask = var_12267_end_mask_0, x = coreml_update_state_107)[name = string("op_12267_cast_fp16")]; tensor key_cache_51_axes_0 = const()[name = string("key_cache_51_axes_0"), val = tensor([0])]; tensor key_cache_51_cast_fp16 = squeeze(axes = key_cache_51_axes_0, x = var_12267_cast_fp16)[name = string("key_cache_51_cast_fp16")]; tensor var_12274_begin_0 = const()[name = string("op_12274_begin_0"), val = tensor([53, 0, 0, 0])]; tensor var_12274_end_0 = const()[name = string("op_12274_end_0"), val = tensor([54, 8, 1536, 128])]; tensor var_12274_end_mask_0 = const()[name = string("op_12274_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12274_cast_fp16 = slice_by_index(begin = var_12274_begin_0, end = var_12274_end_0, end_mask = var_12274_end_mask_0, x = coreml_update_state_107)[name = string("op_12274_cast_fp16")]; tensor value_cache_51_axes_0 = const()[name = string("value_cache_51_axes_0"), val = tensor([0])]; tensor value_cache_51_cast_fp16 = squeeze(axes = value_cache_51_axes_0, x = var_12274_cast_fp16)[name = string("value_cache_51_cast_fp16")]; tensor var_12298_axes_0 = const()[name = string("op_12298_axes_0"), val = tensor([1])]; tensor var_12298_cast_fp16 = expand_dims(axes = var_12298_axes_0, x = key_cache_51_cast_fp16)[name = string("op_12298_cast_fp16")]; tensor var_12303 = const()[name = string("op_12303"), val = tensor([1, 2, 1, 1])]; tensor value_203_cast_fp16 = tile(reps = var_12303, x = var_12298_cast_fp16)[name = string("value_203_cast_fp16")]; tensor var_12309 = const()[name = string("op_12309"), val = tensor([1, 16, 1536, 128])]; tensor key_states_103_cast_fp16 = reshape(shape = var_12309, x = value_203_cast_fp16)[name = string("key_states_103_cast_fp16")]; tensor var_12312_axes_0 = const()[name = string("op_12312_axes_0"), val = tensor([1])]; tensor var_12312_cast_fp16 = expand_dims(axes = var_12312_axes_0, x = value_cache_51_cast_fp16)[name = string("op_12312_cast_fp16")]; tensor var_12317 = const()[name = string("op_12317"), val = tensor([1, 2, 1, 1])]; tensor value_207_cast_fp16 = tile(reps = var_12317, x = var_12312_cast_fp16)[name = string("value_207_cast_fp16")]; tensor var_12323 = const()[name = string("op_12323"), val = tensor([1, 16, 1536, 128])]; tensor value_states_153_cast_fp16 = reshape(shape = var_12323, x = value_207_cast_fp16)[name = string("value_states_153_cast_fp16")]; bool var_12338_transpose_x_1 = const()[name = string("op_12338_transpose_x_1"), val = bool(false)]; bool var_12338_transpose_y_1 = const()[name = string("op_12338_transpose_y_1"), val = bool(true)]; tensor var_12338 = matmul(transpose_x = var_12338_transpose_x_1, transpose_y = var_12338_transpose_y_1, x = query_states_101, y = key_states_103_cast_fp16)[name = string("op_12338")]; fp16 var_12339_to_fp16 = const()[name = string("op_12339_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_101_cast_fp16 = mul(x = var_12338, y = var_12339_to_fp16)[name = string("attention_101_cast_fp16")]; tensor attention_103_cast_fp16 = add(x = attention_101_cast_fp16, y = causal_mask)[name = string("attention_103_cast_fp16")]; int32 var_12348 = const()[name = string("op_12348"), val = int32(-1)]; tensor probabilities_51_cast_fp16 = softmax(axis = var_12348, x = attention_103_cast_fp16)[name = string("probabilities_51_cast_fp16")]; bool output_151_transpose_x_0 = const()[name = string("output_151_transpose_x_0"), val = bool(false)]; bool output_151_transpose_y_0 = const()[name = string("output_151_transpose_y_0"), val = bool(false)]; tensor output_151_cast_fp16 = matmul(transpose_x = output_151_transpose_x_0, transpose_y = output_151_transpose_y_0, x = probabilities_51_cast_fp16, y = value_states_153_cast_fp16)[name = string("output_151_cast_fp16")]; tensor var_12359_perm_0 = const()[name = string("op_12359_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_12365 = const()[name = string("op_12365"), val = tensor([1, 1, 2048])]; tensor var_12359_cast_fp16 = transpose(perm = var_12359_perm_0, x = output_151_cast_fp16)[name = string("transpose_16")]; tensor output_153_cast_fp16 = reshape(shape = var_12365, x = var_12359_cast_fp16)[name = string("output_153_cast_fp16")]; tensor var_12370 = const()[name = string("op_12370"), val = tensor([0, 2, 1])]; string var_12386_pad_type_0 = const()[name = string("op_12386_pad_type_0"), val = string("valid")]; int32 var_12386_groups_0 = const()[name = string("op_12386_groups_0"), val = int32(1)]; tensor var_12386_strides_0 = const()[name = string("op_12386_strides_0"), val = tensor([1])]; tensor var_12386_pad_0 = const()[name = string("op_12386_pad_0"), val = tensor([0, 0])]; tensor var_12386_dilations_0 = const()[name = string("op_12386_dilations_0"), val = tensor([1])]; tensor squeeze_25_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332760832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334333760))))[name = string("squeeze_25_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12371_cast_fp16 = transpose(perm = var_12370, x = output_153_cast_fp16)[name = string("transpose_15")]; tensor var_12386_cast_fp16 = conv(dilations = var_12386_dilations_0, groups = var_12386_groups_0, pad = var_12386_pad_0, pad_type = var_12386_pad_type_0, strides = var_12386_strides_0, weight = squeeze_25_cast_fp16_to_fp32_to_fp16_palettized, x = var_12371_cast_fp16)[name = string("op_12386_cast_fp16")]; tensor var_12390 = const()[name = string("op_12390"), val = tensor([0, 2, 1])]; tensor attn_output_51_cast_fp16 = transpose(perm = var_12390, x = var_12386_cast_fp16)[name = string("transpose_14")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_251_cast_fp16, y = attn_output_51_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; int32 var_12405 = const()[name = string("op_12405"), val = int32(-1)]; fp16 const_361_promoted_to_fp16 = const()[name = string("const_361_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12407_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_361_promoted_to_fp16)[name = string("op_12407_cast_fp16")]; bool input_461_interleave_0 = const()[name = string("input_461_interleave_0"), val = bool(false)]; tensor input_461_cast_fp16 = concat(axis = var_12405, interleave = input_461_interleave_0, values = (hidden_states_259_cast_fp16, var_12407_cast_fp16))[name = string("input_461_cast_fp16")]; tensor normed_413_axes_0 = const()[name = string("normed_413_axes_0"), val = tensor([-1])]; fp16 var_12402_to_fp16 = const()[name = string("op_12402_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_413_cast_fp16 = layer_norm(axes = normed_413_axes_0, epsilon = var_12402_to_fp16, x = input_461_cast_fp16)[name = string("normed_413_cast_fp16")]; tensor normed_415_begin_0 = const()[name = string("normed_415_begin_0"), val = tensor([0, 0, 0])]; tensor normed_415_end_0 = const()[name = string("normed_415_end_0"), val = tensor([1, 1, 1024])]; tensor normed_415_end_mask_0 = const()[name = string("normed_415_end_mask_0"), val = tensor([true, true, false])]; tensor normed_415_cast_fp16 = slice_by_index(begin = normed_415_begin_0, end = normed_415_end_0, end_mask = normed_415_end_mask_0, x = normed_413_cast_fp16)[name = string("normed_415_cast_fp16")]; tensor const_363_promoted_to_fp16 = const()[name = string("const_363_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334350208)))]; tensor x_101_cast_fp16 = mul(x = normed_415_cast_fp16, y = const_363_promoted_to_fp16)[name = string("x_101_cast_fp16")]; tensor var_12427 = const()[name = string("op_12427"), val = tensor([0, 2, 1])]; tensor input_463_axes_0 = const()[name = string("input_463_axes_0"), val = tensor([2])]; tensor var_12428 = transpose(perm = var_12427, x = x_101_cast_fp16)[name = string("transpose_13")]; tensor input_463 = expand_dims(axes = input_463_axes_0, x = var_12428)[name = string("input_463")]; string input_465_pad_type_0 = const()[name = string("input_465_pad_type_0"), val = string("valid")]; tensor input_465_strides_0 = const()[name = string("input_465_strides_0"), val = tensor([1, 1])]; tensor input_465_pad_0 = const()[name = string("input_465_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_465_dilations_0 = const()[name = string("input_465_dilations_0"), val = tensor([1, 1])]; int32 input_465_groups_0 = const()[name = string("input_465_groups_0"), val = int32(1)]; tensor input_465 = conv(dilations = input_465_dilations_0, groups = input_465_groups_0, pad = input_465_pad_0, pad_type = input_465_pad_type_0, strides = input_465_strides_0, weight = model_model_layers_25_mlp_gate_proj_weight_palettized, x = input_463)[name = string("input_465")]; string b_51_pad_type_0 = const()[name = string("b_51_pad_type_0"), val = string("valid")]; tensor b_51_strides_0 = const()[name = string("b_51_strides_0"), val = tensor([1, 1])]; tensor b_51_pad_0 = const()[name = string("b_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_51_dilations_0 = const()[name = string("b_51_dilations_0"), val = tensor([1, 1])]; int32 b_51_groups_0 = const()[name = string("b_51_groups_0"), val = int32(1)]; tensor b_51 = conv(dilations = b_51_dilations_0, groups = b_51_groups_0, pad = b_51_pad_0, pad_type = b_51_pad_type_0, strides = b_51_strides_0, weight = model_model_layers_25_mlp_up_proj_weight_palettized, x = input_463)[name = string("b_51")]; tensor c_51 = silu(x = input_465)[name = string("c_51")]; tensor input_467 = mul(x = c_51, y = b_51)[name = string("input_467")]; string e_51_pad_type_0 = const()[name = string("e_51_pad_type_0"), val = string("valid")]; tensor e_51_strides_0 = const()[name = string("e_51_strides_0"), val = tensor([1, 1])]; tensor e_51_pad_0 = const()[name = string("e_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_51_dilations_0 = const()[name = string("e_51_dilations_0"), val = tensor([1, 1])]; int32 e_51_groups_0 = const()[name = string("e_51_groups_0"), val = int32(1)]; tensor e_51 = conv(dilations = e_51_dilations_0, groups = e_51_groups_0, pad = e_51_pad_0, pad_type = e_51_pad_type_0, strides = e_51_strides_0, weight = model_model_layers_25_mlp_down_proj_weight_palettized, x = input_467)[name = string("e_51")]; tensor var_12450_axes_0 = const()[name = string("op_12450_axes_0"), val = tensor([2])]; tensor var_12450 = squeeze(axes = var_12450_axes_0, x = e_51)[name = string("op_12450")]; tensor var_12451 = const()[name = string("op_12451"), val = tensor([0, 2, 1])]; tensor var_12452 = transpose(perm = var_12451, x = var_12450)[name = string("transpose_12")]; tensor hidden_states_261_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = var_12452)[name = string("hidden_states_261_cast_fp16")]; int32 var_12466 = const()[name = string("op_12466"), val = int32(-1)]; fp16 const_364_promoted_to_fp16 = const()[name = string("const_364_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12468_cast_fp16 = mul(x = hidden_states_261_cast_fp16, y = const_364_promoted_to_fp16)[name = string("op_12468_cast_fp16")]; bool input_469_interleave_0 = const()[name = string("input_469_interleave_0"), val = bool(false)]; tensor input_469_cast_fp16 = concat(axis = var_12466, interleave = input_469_interleave_0, values = (hidden_states_261_cast_fp16, var_12468_cast_fp16))[name = string("input_469_cast_fp16")]; tensor normed_417_axes_0 = const()[name = string("normed_417_axes_0"), val = tensor([-1])]; fp16 var_12463_to_fp16 = const()[name = string("op_12463_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_417_cast_fp16 = layer_norm(axes = normed_417_axes_0, epsilon = var_12463_to_fp16, x = input_469_cast_fp16)[name = string("normed_417_cast_fp16")]; tensor normed_419_begin_0 = const()[name = string("normed_419_begin_0"), val = tensor([0, 0, 0])]; tensor normed_419_end_0 = const()[name = string("normed_419_end_0"), val = tensor([1, 1, 1024])]; tensor normed_419_end_mask_0 = const()[name = string("normed_419_end_mask_0"), val = tensor([true, true, false])]; tensor normed_419_cast_fp16 = slice_by_index(begin = normed_419_begin_0, end = normed_419_end_0, end_mask = normed_419_end_mask_0, x = normed_417_cast_fp16)[name = string("normed_419_cast_fp16")]; tensor const_366_promoted_to_fp16 = const()[name = string("const_366_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334352320)))]; tensor hidden_states_263_cast_fp16 = mul(x = normed_419_cast_fp16, y = const_366_promoted_to_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor var_12480 = const()[name = string("op_12480"), val = tensor([0, 2, 1])]; tensor var_12483_axes_0 = const()[name = string("op_12483_axes_0"), val = tensor([2])]; tensor var_12481_cast_fp16 = transpose(perm = var_12480, x = hidden_states_263_cast_fp16)[name = string("transpose_11")]; tensor var_12483_cast_fp16 = expand_dims(axes = var_12483_axes_0, x = var_12481_cast_fp16)[name = string("op_12483_cast_fp16")]; string var_12499_pad_type_0 = const()[name = string("op_12499_pad_type_0"), val = string("valid")]; tensor var_12499_strides_0 = const()[name = string("op_12499_strides_0"), val = tensor([1, 1])]; tensor var_12499_pad_0 = const()[name = string("op_12499_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12499_dilations_0 = const()[name = string("op_12499_dilations_0"), val = tensor([1, 1])]; int32 var_12499_groups_0 = const()[name = string("op_12499_groups_0"), val = int32(1)]; tensor var_12499 = conv(dilations = var_12499_dilations_0, groups = var_12499_groups_0, pad = var_12499_pad_0, pad_type = var_12499_pad_type_0, strides = var_12499_strides_0, weight = model_model_layers_26_self_attn_q_proj_weight_palettized, x = var_12483_cast_fp16)[name = string("op_12499")]; tensor var_12504 = const()[name = string("op_12504"), val = tensor([1, 16, 1, 128])]; tensor var_12505 = reshape(shape = var_12504, x = var_12499)[name = string("op_12505")]; string var_12521_pad_type_0 = const()[name = string("op_12521_pad_type_0"), val = string("valid")]; tensor var_12521_strides_0 = const()[name = string("op_12521_strides_0"), val = tensor([1, 1])]; tensor var_12521_pad_0 = const()[name = string("op_12521_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12521_dilations_0 = const()[name = string("op_12521_dilations_0"), val = tensor([1, 1])]; int32 var_12521_groups_0 = const()[name = string("op_12521_groups_0"), val = int32(1)]; tensor var_12521 = conv(dilations = var_12521_dilations_0, groups = var_12521_groups_0, pad = var_12521_pad_0, pad_type = var_12521_pad_type_0, strides = var_12521_strides_0, weight = model_model_layers_26_self_attn_k_proj_weight_palettized, x = var_12483_cast_fp16)[name = string("op_12521")]; tensor var_12526 = const()[name = string("op_12526"), val = tensor([1, 8, 1, 128])]; tensor var_12527 = reshape(shape = var_12526, x = var_12521)[name = string("op_12527")]; string var_12543_pad_type_0 = const()[name = string("op_12543_pad_type_0"), val = string("valid")]; tensor var_12543_strides_0 = const()[name = string("op_12543_strides_0"), val = tensor([1, 1])]; tensor var_12543_pad_0 = const()[name = string("op_12543_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12543_dilations_0 = const()[name = string("op_12543_dilations_0"), val = tensor([1, 1])]; int32 var_12543_groups_0 = const()[name = string("op_12543_groups_0"), val = int32(1)]; tensor var_12543 = conv(dilations = var_12543_dilations_0, groups = var_12543_groups_0, pad = var_12543_pad_0, pad_type = var_12543_pad_type_0, strides = var_12543_strides_0, weight = model_model_layers_26_self_attn_v_proj_weight_palettized, x = var_12483_cast_fp16)[name = string("op_12543")]; tensor var_12548 = const()[name = string("op_12548"), val = tensor([1, 8, 1, 128])]; tensor var_12549 = reshape(shape = var_12548, x = var_12543)[name = string("op_12549")]; int32 var_12566 = const()[name = string("op_12566"), val = int32(-1)]; fp16 const_367_promoted = const()[name = string("const_367_promoted"), val = fp16(-0x1p+0)]; tensor var_12568 = mul(x = var_12505, y = const_367_promoted)[name = string("op_12568")]; bool input_473_interleave_0 = const()[name = string("input_473_interleave_0"), val = bool(false)]; tensor input_473 = concat(axis = var_12566, interleave = input_473_interleave_0, values = (var_12505, var_12568))[name = string("input_473")]; tensor normed_421_axes_0 = const()[name = string("normed_421_axes_0"), val = tensor([-1])]; fp16 var_12563_to_fp16 = const()[name = string("op_12563_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_421_cast_fp16 = layer_norm(axes = normed_421_axes_0, epsilon = var_12563_to_fp16, x = input_473)[name = string("normed_421_cast_fp16")]; tensor normed_423_begin_0 = const()[name = string("normed_423_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_423_end_0 = const()[name = string("normed_423_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_423_end_mask_0 = const()[name = string("normed_423_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_423 = slice_by_index(begin = normed_423_begin_0, end = normed_423_end_0, end_mask = normed_423_end_mask_0, x = normed_421_cast_fp16)[name = string("normed_423")]; tensor const_369 = const()[name = string("const_369"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334354432)))]; tensor q_53 = mul(x = normed_423, y = const_369)[name = string("q_53")]; int32 var_12588 = const()[name = string("op_12588"), val = int32(-1)]; fp16 const_370_promoted = const()[name = string("const_370_promoted"), val = fp16(-0x1p+0)]; tensor var_12590 = mul(x = var_12527, y = const_370_promoted)[name = string("op_12590")]; bool input_475_interleave_0 = const()[name = string("input_475_interleave_0"), val = bool(false)]; tensor input_475 = concat(axis = var_12588, interleave = input_475_interleave_0, values = (var_12527, var_12590))[name = string("input_475")]; tensor normed_425_axes_0 = const()[name = string("normed_425_axes_0"), val = tensor([-1])]; fp16 var_12585_to_fp16 = const()[name = string("op_12585_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_425_cast_fp16 = layer_norm(axes = normed_425_axes_0, epsilon = var_12585_to_fp16, x = input_475)[name = string("normed_425_cast_fp16")]; tensor normed_427_begin_0 = const()[name = string("normed_427_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_427_end_0 = const()[name = string("normed_427_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_427_end_mask_0 = const()[name = string("normed_427_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_427 = slice_by_index(begin = normed_427_begin_0, end = normed_427_end_0, end_mask = normed_427_end_mask_0, x = normed_425_cast_fp16)[name = string("normed_427")]; tensor const_372 = const()[name = string("const_372"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334354752)))]; tensor k_53 = mul(x = normed_427, y = const_372)[name = string("k_53")]; tensor var_12599 = mul(x = q_53, y = cos_1_cast_fp16)[name = string("op_12599")]; tensor var_12604_begin_0 = const()[name = string("op_12604_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12604_end_0 = const()[name = string("op_12604_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_12604_end_mask_0 = const()[name = string("op_12604_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12604 = slice_by_index(begin = var_12604_begin_0, end = var_12604_end_0, end_mask = var_12604_end_mask_0, x = q_53)[name = string("op_12604")]; fp16 const_373_promoted = const()[name = string("const_373_promoted"), val = fp16(-0x1p+0)]; tensor var_12605 = mul(x = var_12604, y = const_373_promoted)[name = string("op_12605")]; tensor var_12610_begin_0 = const()[name = string("op_12610_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12610_end_0 = const()[name = string("op_12610_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_12610_end_mask_0 = const()[name = string("op_12610_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12610 = slice_by_index(begin = var_12610_begin_0, end = var_12610_end_0, end_mask = var_12610_end_mask_0, x = q_53)[name = string("op_12610")]; int32 var_12612 = const()[name = string("op_12612"), val = int32(-1)]; bool var_12613_interleave_0 = const()[name = string("op_12613_interleave_0"), val = bool(false)]; tensor var_12613 = concat(axis = var_12612, interleave = var_12613_interleave_0, values = (var_12605, var_12610))[name = string("op_12613")]; tensor var_12614 = mul(x = var_12613, y = sin_1_cast_fp16)[name = string("op_12614")]; tensor query_states_105 = add(x = var_12599, y = var_12614)[name = string("query_states_105")]; tensor var_12617 = mul(x = k_53, y = cos_1_cast_fp16)[name = string("op_12617")]; tensor var_12622_begin_0 = const()[name = string("op_12622_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12622_end_0 = const()[name = string("op_12622_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_12622_end_mask_0 = const()[name = string("op_12622_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12622 = slice_by_index(begin = var_12622_begin_0, end = var_12622_end_0, end_mask = var_12622_end_mask_0, x = k_53)[name = string("op_12622")]; fp16 const_374_promoted = const()[name = string("const_374_promoted"), val = fp16(-0x1p+0)]; tensor var_12623 = mul(x = var_12622, y = const_374_promoted)[name = string("op_12623")]; tensor var_12628_begin_0 = const()[name = string("op_12628_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12628_end_0 = const()[name = string("op_12628_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_12628_end_mask_0 = const()[name = string("op_12628_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12628 = slice_by_index(begin = var_12628_begin_0, end = var_12628_end_0, end_mask = var_12628_end_mask_0, x = k_53)[name = string("op_12628")]; int32 var_12630 = const()[name = string("op_12630"), val = int32(-1)]; bool var_12631_interleave_0 = const()[name = string("op_12631_interleave_0"), val = bool(false)]; tensor var_12631 = concat(axis = var_12630, interleave = var_12631_interleave_0, values = (var_12623, var_12628))[name = string("op_12631")]; tensor var_12632 = mul(x = var_12631, y = sin_1_cast_fp16)[name = string("op_12632")]; tensor key_states_105 = add(x = var_12617, y = var_12632)[name = string("key_states_105")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_312, expand_dims_313, current_pos, expand_dims_315))[name = string("concat_210")]; tensor concat_211_values1_0 = const()[name = string("concat_211_values1_0"), val = tensor([0])]; tensor concat_211_values3_0 = const()[name = string("concat_211_values3_0"), val = tensor([0])]; int32 concat_211_axis_0 = const()[name = string("concat_211_axis_0"), val = int32(0)]; bool concat_211_interleave_0 = const()[name = string("concat_211_interleave_0"), val = bool(false)]; tensor concat_211 = concat(axis = concat_211_axis_0, interleave = concat_211_interleave_0, values = (expand_dims_316, concat_211_values1_0, var_1717, concat_211_values3_0))[name = string("concat_211")]; tensor model_model_kv_cache_0_internal_tensor_assign_53_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_cast_fp16 = slice_update(begin = concat_210, begin_mask = model_model_kv_cache_0_internal_tensor_assign_53_begin_mask_0, end = concat_211, end_mask = model_model_kv_cache_0_internal_tensor_assign_53_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_53_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_53_stride_0, update = key_states_105, x = coreml_update_state_107)[name = string("model_model_kv_cache_0_internal_tensor_assign_53_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_53_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_164_write_state")]; tensor coreml_update_state_108 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_164")]; tensor expand_dims_318 = const()[name = string("expand_dims_318"), val = tensor([54])]; tensor expand_dims_319 = const()[name = string("expand_dims_319"), val = tensor([0])]; tensor expand_dims_321 = const()[name = string("expand_dims_321"), val = tensor([0])]; tensor expand_dims_322 = const()[name = string("expand_dims_322"), val = tensor([55])]; int32 concat_214_axis_0 = const()[name = string("concat_214_axis_0"), val = int32(0)]; bool concat_214_interleave_0 = const()[name = string("concat_214_interleave_0"), val = bool(false)]; tensor concat_214 = concat(axis = concat_214_axis_0, interleave = concat_214_interleave_0, values = (expand_dims_318, expand_dims_319, current_pos, expand_dims_321))[name = string("concat_214")]; tensor concat_215_values1_0 = const()[name = string("concat_215_values1_0"), val = tensor([0])]; tensor concat_215_values3_0 = const()[name = string("concat_215_values3_0"), val = tensor([0])]; int32 concat_215_axis_0 = const()[name = string("concat_215_axis_0"), val = int32(0)]; bool concat_215_interleave_0 = const()[name = string("concat_215_interleave_0"), val = bool(false)]; tensor concat_215 = concat(axis = concat_215_axis_0, interleave = concat_215_interleave_0, values = (expand_dims_322, concat_215_values1_0, var_1717, concat_215_values3_0))[name = string("concat_215")]; tensor model_model_kv_cache_0_internal_tensor_assign_54_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_cast_fp16 = slice_update(begin = concat_214, begin_mask = model_model_kv_cache_0_internal_tensor_assign_54_begin_mask_0, end = concat_215, end_mask = model_model_kv_cache_0_internal_tensor_assign_54_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_54_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_54_stride_0, update = var_12549, x = coreml_update_state_108)[name = string("model_model_kv_cache_0_internal_tensor_assign_54_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_54_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_165_write_state")]; tensor coreml_update_state_109 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_165")]; tensor var_12687_begin_0 = const()[name = string("op_12687_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_12687_end_0 = const()[name = string("op_12687_end_0"), val = tensor([27, 8, 1536, 128])]; tensor var_12687_end_mask_0 = const()[name = string("op_12687_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12687_cast_fp16 = slice_by_index(begin = var_12687_begin_0, end = var_12687_end_0, end_mask = var_12687_end_mask_0, x = coreml_update_state_109)[name = string("op_12687_cast_fp16")]; tensor key_cache_53_axes_0 = const()[name = string("key_cache_53_axes_0"), val = tensor([0])]; tensor key_cache_53_cast_fp16 = squeeze(axes = key_cache_53_axes_0, x = var_12687_cast_fp16)[name = string("key_cache_53_cast_fp16")]; tensor var_12694_begin_0 = const()[name = string("op_12694_begin_0"), val = tensor([54, 0, 0, 0])]; tensor var_12694_end_0 = const()[name = string("op_12694_end_0"), val = tensor([55, 8, 1536, 128])]; tensor var_12694_end_mask_0 = const()[name = string("op_12694_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12694_cast_fp16 = slice_by_index(begin = var_12694_begin_0, end = var_12694_end_0, end_mask = var_12694_end_mask_0, x = coreml_update_state_109)[name = string("op_12694_cast_fp16")]; tensor value_cache_53_axes_0 = const()[name = string("value_cache_53_axes_0"), val = tensor([0])]; tensor value_cache_53_cast_fp16 = squeeze(axes = value_cache_53_axes_0, x = var_12694_cast_fp16)[name = string("value_cache_53_cast_fp16")]; tensor var_12718_axes_0 = const()[name = string("op_12718_axes_0"), val = tensor([1])]; tensor var_12718_cast_fp16 = expand_dims(axes = var_12718_axes_0, x = key_cache_53_cast_fp16)[name = string("op_12718_cast_fp16")]; tensor var_12723 = const()[name = string("op_12723"), val = tensor([1, 2, 1, 1])]; tensor value_211_cast_fp16 = tile(reps = var_12723, x = var_12718_cast_fp16)[name = string("value_211_cast_fp16")]; tensor var_12729 = const()[name = string("op_12729"), val = tensor([1, 16, 1536, 128])]; tensor key_states_107_cast_fp16 = reshape(shape = var_12729, x = value_211_cast_fp16)[name = string("key_states_107_cast_fp16")]; tensor var_12732_axes_0 = const()[name = string("op_12732_axes_0"), val = tensor([1])]; tensor var_12732_cast_fp16 = expand_dims(axes = var_12732_axes_0, x = value_cache_53_cast_fp16)[name = string("op_12732_cast_fp16")]; tensor var_12737 = const()[name = string("op_12737"), val = tensor([1, 2, 1, 1])]; tensor value_215_cast_fp16 = tile(reps = var_12737, x = var_12732_cast_fp16)[name = string("value_215_cast_fp16")]; tensor var_12743 = const()[name = string("op_12743"), val = tensor([1, 16, 1536, 128])]; tensor value_states_159_cast_fp16 = reshape(shape = var_12743, x = value_215_cast_fp16)[name = string("value_states_159_cast_fp16")]; bool var_12758_transpose_x_1 = const()[name = string("op_12758_transpose_x_1"), val = bool(false)]; bool var_12758_transpose_y_1 = const()[name = string("op_12758_transpose_y_1"), val = bool(true)]; tensor var_12758 = matmul(transpose_x = var_12758_transpose_x_1, transpose_y = var_12758_transpose_y_1, x = query_states_105, y = key_states_107_cast_fp16)[name = string("op_12758")]; fp16 var_12759_to_fp16 = const()[name = string("op_12759_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_105_cast_fp16 = mul(x = var_12758, y = var_12759_to_fp16)[name = string("attention_105_cast_fp16")]; tensor attention_107_cast_fp16 = add(x = attention_105_cast_fp16, y = causal_mask)[name = string("attention_107_cast_fp16")]; int32 var_12768 = const()[name = string("op_12768"), val = int32(-1)]; tensor probabilities_53_cast_fp16 = softmax(axis = var_12768, x = attention_107_cast_fp16)[name = string("probabilities_53_cast_fp16")]; bool output_157_transpose_x_0 = const()[name = string("output_157_transpose_x_0"), val = bool(false)]; bool output_157_transpose_y_0 = const()[name = string("output_157_transpose_y_0"), val = bool(false)]; tensor output_157_cast_fp16 = matmul(transpose_x = output_157_transpose_x_0, transpose_y = output_157_transpose_y_0, x = probabilities_53_cast_fp16, y = value_states_159_cast_fp16)[name = string("output_157_cast_fp16")]; tensor var_12779_perm_0 = const()[name = string("op_12779_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_12785 = const()[name = string("op_12785"), val = tensor([1, 1, 2048])]; tensor var_12779_cast_fp16 = transpose(perm = var_12779_perm_0, x = output_157_cast_fp16)[name = string("transpose_10")]; tensor output_159_cast_fp16 = reshape(shape = var_12785, x = var_12779_cast_fp16)[name = string("output_159_cast_fp16")]; tensor var_12790 = const()[name = string("op_12790"), val = tensor([0, 2, 1])]; string var_12806_pad_type_0 = const()[name = string("op_12806_pad_type_0"), val = string("valid")]; int32 var_12806_groups_0 = const()[name = string("op_12806_groups_0"), val = int32(1)]; tensor var_12806_strides_0 = const()[name = string("op_12806_strides_0"), val = tensor([1])]; tensor var_12806_pad_0 = const()[name = string("op_12806_pad_0"), val = tensor([0, 0])]; tensor var_12806_dilations_0 = const()[name = string("op_12806_dilations_0"), val = tensor([1])]; tensor squeeze_26_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334355072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335928000))))[name = string("squeeze_26_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12791_cast_fp16 = transpose(perm = var_12790, x = output_159_cast_fp16)[name = string("transpose_9")]; tensor var_12806_cast_fp16 = conv(dilations = var_12806_dilations_0, groups = var_12806_groups_0, pad = var_12806_pad_0, pad_type = var_12806_pad_type_0, strides = var_12806_strides_0, weight = squeeze_26_cast_fp16_to_fp32_to_fp16_palettized, x = var_12791_cast_fp16)[name = string("op_12806_cast_fp16")]; tensor var_12810 = const()[name = string("op_12810"), val = tensor([0, 2, 1])]; tensor attn_output_53_cast_fp16 = transpose(perm = var_12810, x = var_12806_cast_fp16)[name = string("transpose_8")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_261_cast_fp16, y = attn_output_53_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; int32 var_12825 = const()[name = string("op_12825"), val = int32(-1)]; fp16 const_375_promoted_to_fp16 = const()[name = string("const_375_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12827_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_375_promoted_to_fp16)[name = string("op_12827_cast_fp16")]; bool input_479_interleave_0 = const()[name = string("input_479_interleave_0"), val = bool(false)]; tensor input_479_cast_fp16 = concat(axis = var_12825, interleave = input_479_interleave_0, values = (hidden_states_269_cast_fp16, var_12827_cast_fp16))[name = string("input_479_cast_fp16")]; tensor normed_429_axes_0 = const()[name = string("normed_429_axes_0"), val = tensor([-1])]; fp16 var_12822_to_fp16 = const()[name = string("op_12822_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_429_cast_fp16 = layer_norm(axes = normed_429_axes_0, epsilon = var_12822_to_fp16, x = input_479_cast_fp16)[name = string("normed_429_cast_fp16")]; tensor normed_431_begin_0 = const()[name = string("normed_431_begin_0"), val = tensor([0, 0, 0])]; tensor normed_431_end_0 = const()[name = string("normed_431_end_0"), val = tensor([1, 1, 1024])]; tensor normed_431_end_mask_0 = const()[name = string("normed_431_end_mask_0"), val = tensor([true, true, false])]; tensor normed_431_cast_fp16 = slice_by_index(begin = normed_431_begin_0, end = normed_431_end_0, end_mask = normed_431_end_mask_0, x = normed_429_cast_fp16)[name = string("normed_431_cast_fp16")]; tensor const_377_promoted_to_fp16 = const()[name = string("const_377_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335944448)))]; tensor x_105_cast_fp16 = mul(x = normed_431_cast_fp16, y = const_377_promoted_to_fp16)[name = string("x_105_cast_fp16")]; tensor var_12847 = const()[name = string("op_12847"), val = tensor([0, 2, 1])]; tensor input_481_axes_0 = const()[name = string("input_481_axes_0"), val = tensor([2])]; tensor var_12848 = transpose(perm = var_12847, x = x_105_cast_fp16)[name = string("transpose_7")]; tensor input_481 = expand_dims(axes = input_481_axes_0, x = var_12848)[name = string("input_481")]; string input_483_pad_type_0 = const()[name = string("input_483_pad_type_0"), val = string("valid")]; tensor input_483_strides_0 = const()[name = string("input_483_strides_0"), val = tensor([1, 1])]; tensor input_483_pad_0 = const()[name = string("input_483_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_483_dilations_0 = const()[name = string("input_483_dilations_0"), val = tensor([1, 1])]; int32 input_483_groups_0 = const()[name = string("input_483_groups_0"), val = int32(1)]; tensor input_483 = conv(dilations = input_483_dilations_0, groups = input_483_groups_0, pad = input_483_pad_0, pad_type = input_483_pad_type_0, strides = input_483_strides_0, weight = model_model_layers_26_mlp_gate_proj_weight_palettized, x = input_481)[name = string("input_483")]; string b_53_pad_type_0 = const()[name = string("b_53_pad_type_0"), val = string("valid")]; tensor b_53_strides_0 = const()[name = string("b_53_strides_0"), val = tensor([1, 1])]; tensor b_53_pad_0 = const()[name = string("b_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_53_dilations_0 = const()[name = string("b_53_dilations_0"), val = tensor([1, 1])]; int32 b_53_groups_0 = const()[name = string("b_53_groups_0"), val = int32(1)]; tensor b_53 = conv(dilations = b_53_dilations_0, groups = b_53_groups_0, pad = b_53_pad_0, pad_type = b_53_pad_type_0, strides = b_53_strides_0, weight = model_model_layers_26_mlp_up_proj_weight_palettized, x = input_481)[name = string("b_53")]; tensor c_53 = silu(x = input_483)[name = string("c_53")]; tensor input_485 = mul(x = c_53, y = b_53)[name = string("input_485")]; string e_53_pad_type_0 = const()[name = string("e_53_pad_type_0"), val = string("valid")]; tensor e_53_strides_0 = const()[name = string("e_53_strides_0"), val = tensor([1, 1])]; tensor e_53_pad_0 = const()[name = string("e_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_53_dilations_0 = const()[name = string("e_53_dilations_0"), val = tensor([1, 1])]; int32 e_53_groups_0 = const()[name = string("e_53_groups_0"), val = int32(1)]; tensor e_53 = conv(dilations = e_53_dilations_0, groups = e_53_groups_0, pad = e_53_pad_0, pad_type = e_53_pad_type_0, strides = e_53_strides_0, weight = model_model_layers_26_mlp_down_proj_weight_palettized, x = input_485)[name = string("e_53")]; tensor var_12870_axes_0 = const()[name = string("op_12870_axes_0"), val = tensor([2])]; tensor var_12870 = squeeze(axes = var_12870_axes_0, x = e_53)[name = string("op_12870")]; tensor var_12871 = const()[name = string("op_12871"), val = tensor([0, 2, 1])]; tensor var_12872 = transpose(perm = var_12871, x = var_12870)[name = string("transpose_6")]; tensor hidden_states_271_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = var_12872)[name = string("hidden_states_271_cast_fp16")]; int32 var_12886 = const()[name = string("op_12886"), val = int32(-1)]; fp16 const_378_promoted_to_fp16 = const()[name = string("const_378_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12888_cast_fp16 = mul(x = hidden_states_271_cast_fp16, y = const_378_promoted_to_fp16)[name = string("op_12888_cast_fp16")]; bool input_487_interleave_0 = const()[name = string("input_487_interleave_0"), val = bool(false)]; tensor input_487_cast_fp16 = concat(axis = var_12886, interleave = input_487_interleave_0, values = (hidden_states_271_cast_fp16, var_12888_cast_fp16))[name = string("input_487_cast_fp16")]; tensor normed_433_axes_0 = const()[name = string("normed_433_axes_0"), val = tensor([-1])]; fp16 var_12883_to_fp16 = const()[name = string("op_12883_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_433_cast_fp16 = layer_norm(axes = normed_433_axes_0, epsilon = var_12883_to_fp16, x = input_487_cast_fp16)[name = string("normed_433_cast_fp16")]; tensor normed_435_begin_0 = const()[name = string("normed_435_begin_0"), val = tensor([0, 0, 0])]; tensor normed_435_end_0 = const()[name = string("normed_435_end_0"), val = tensor([1, 1, 1024])]; tensor normed_435_end_mask_0 = const()[name = string("normed_435_end_mask_0"), val = tensor([true, true, false])]; tensor normed_435_cast_fp16 = slice_by_index(begin = normed_435_begin_0, end = normed_435_end_0, end_mask = normed_435_end_mask_0, x = normed_433_cast_fp16)[name = string("normed_435_cast_fp16")]; tensor const_380_promoted_to_fp16 = const()[name = string("const_380_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335946560)))]; tensor hidden_states_273_cast_fp16 = mul(x = normed_435_cast_fp16, y = const_380_promoted_to_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor var_12900 = const()[name = string("op_12900"), val = tensor([0, 2, 1])]; tensor var_12903_axes_0 = const()[name = string("op_12903_axes_0"), val = tensor([2])]; tensor var_12901_cast_fp16 = transpose(perm = var_12900, x = hidden_states_273_cast_fp16)[name = string("transpose_5")]; tensor var_12903_cast_fp16 = expand_dims(axes = var_12903_axes_0, x = var_12901_cast_fp16)[name = string("op_12903_cast_fp16")]; string var_12919_pad_type_0 = const()[name = string("op_12919_pad_type_0"), val = string("valid")]; tensor var_12919_strides_0 = const()[name = string("op_12919_strides_0"), val = tensor([1, 1])]; tensor var_12919_pad_0 = const()[name = string("op_12919_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12919_dilations_0 = const()[name = string("op_12919_dilations_0"), val = tensor([1, 1])]; int32 var_12919_groups_0 = const()[name = string("op_12919_groups_0"), val = int32(1)]; tensor var_12919 = conv(dilations = var_12919_dilations_0, groups = var_12919_groups_0, pad = var_12919_pad_0, pad_type = var_12919_pad_type_0, strides = var_12919_strides_0, weight = model_model_layers_27_self_attn_q_proj_weight_palettized, x = var_12903_cast_fp16)[name = string("op_12919")]; tensor var_12924 = const()[name = string("op_12924"), val = tensor([1, 16, 1, 128])]; tensor var_12925 = reshape(shape = var_12924, x = var_12919)[name = string("op_12925")]; string var_12941_pad_type_0 = const()[name = string("op_12941_pad_type_0"), val = string("valid")]; tensor var_12941_strides_0 = const()[name = string("op_12941_strides_0"), val = tensor([1, 1])]; tensor var_12941_pad_0 = const()[name = string("op_12941_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12941_dilations_0 = const()[name = string("op_12941_dilations_0"), val = tensor([1, 1])]; int32 var_12941_groups_0 = const()[name = string("op_12941_groups_0"), val = int32(1)]; tensor var_12941 = conv(dilations = var_12941_dilations_0, groups = var_12941_groups_0, pad = var_12941_pad_0, pad_type = var_12941_pad_type_0, strides = var_12941_strides_0, weight = model_model_layers_27_self_attn_k_proj_weight_palettized, x = var_12903_cast_fp16)[name = string("op_12941")]; tensor var_12946 = const()[name = string("op_12946"), val = tensor([1, 8, 1, 128])]; tensor var_12947 = reshape(shape = var_12946, x = var_12941)[name = string("op_12947")]; string var_12963_pad_type_0 = const()[name = string("op_12963_pad_type_0"), val = string("valid")]; tensor var_12963_strides_0 = const()[name = string("op_12963_strides_0"), val = tensor([1, 1])]; tensor var_12963_pad_0 = const()[name = string("op_12963_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12963_dilations_0 = const()[name = string("op_12963_dilations_0"), val = tensor([1, 1])]; int32 var_12963_groups_0 = const()[name = string("op_12963_groups_0"), val = int32(1)]; tensor var_12963 = conv(dilations = var_12963_dilations_0, groups = var_12963_groups_0, pad = var_12963_pad_0, pad_type = var_12963_pad_type_0, strides = var_12963_strides_0, weight = model_model_layers_27_self_attn_v_proj_weight_palettized, x = var_12903_cast_fp16)[name = string("op_12963")]; tensor var_12968 = const()[name = string("op_12968"), val = tensor([1, 8, 1, 128])]; tensor var_12969 = reshape(shape = var_12968, x = var_12963)[name = string("op_12969")]; int32 var_12986 = const()[name = string("op_12986"), val = int32(-1)]; fp16 const_381_promoted = const()[name = string("const_381_promoted"), val = fp16(-0x1p+0)]; tensor var_12988 = mul(x = var_12925, y = const_381_promoted)[name = string("op_12988")]; bool input_491_interleave_0 = const()[name = string("input_491_interleave_0"), val = bool(false)]; tensor input_491 = concat(axis = var_12986, interleave = input_491_interleave_0, values = (var_12925, var_12988))[name = string("input_491")]; tensor normed_437_axes_0 = const()[name = string("normed_437_axes_0"), val = tensor([-1])]; fp16 var_12983_to_fp16 = const()[name = string("op_12983_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_437_cast_fp16 = layer_norm(axes = normed_437_axes_0, epsilon = var_12983_to_fp16, x = input_491)[name = string("normed_437_cast_fp16")]; tensor normed_439_begin_0 = const()[name = string("normed_439_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_439_end_0 = const()[name = string("normed_439_end_0"), val = tensor([1, 16, 1, 128])]; tensor normed_439_end_mask_0 = const()[name = string("normed_439_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_439 = slice_by_index(begin = normed_439_begin_0, end = normed_439_end_0, end_mask = normed_439_end_mask_0, x = normed_437_cast_fp16)[name = string("normed_439")]; tensor const_383 = const()[name = string("const_383"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335948672)))]; tensor q = mul(x = normed_439, y = const_383)[name = string("q")]; int32 var_13008 = const()[name = string("op_13008"), val = int32(-1)]; fp16 const_384_promoted = const()[name = string("const_384_promoted"), val = fp16(-0x1p+0)]; tensor var_13010 = mul(x = var_12947, y = const_384_promoted)[name = string("op_13010")]; bool input_493_interleave_0 = const()[name = string("input_493_interleave_0"), val = bool(false)]; tensor input_493 = concat(axis = var_13008, interleave = input_493_interleave_0, values = (var_12947, var_13010))[name = string("input_493")]; tensor normed_441_axes_0 = const()[name = string("normed_441_axes_0"), val = tensor([-1])]; fp16 var_13005_to_fp16 = const()[name = string("op_13005_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_441_cast_fp16 = layer_norm(axes = normed_441_axes_0, epsilon = var_13005_to_fp16, x = input_493)[name = string("normed_441_cast_fp16")]; tensor normed_443_begin_0 = const()[name = string("normed_443_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_443_end_0 = const()[name = string("normed_443_end_0"), val = tensor([1, 8, 1, 128])]; tensor normed_443_end_mask_0 = const()[name = string("normed_443_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_443 = slice_by_index(begin = normed_443_begin_0, end = normed_443_end_0, end_mask = normed_443_end_mask_0, x = normed_441_cast_fp16)[name = string("normed_443")]; tensor const_386 = const()[name = string("const_386"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335948992)))]; tensor k = mul(x = normed_443, y = const_386)[name = string("k")]; tensor var_13019 = mul(x = q, y = cos_1_cast_fp16)[name = string("op_13019")]; tensor var_13024_begin_0 = const()[name = string("op_13024_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13024_end_0 = const()[name = string("op_13024_end_0"), val = tensor([1, 16, 1, 128])]; tensor var_13024_end_mask_0 = const()[name = string("op_13024_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13024 = slice_by_index(begin = var_13024_begin_0, end = var_13024_end_0, end_mask = var_13024_end_mask_0, x = q)[name = string("op_13024")]; fp16 const_387_promoted = const()[name = string("const_387_promoted"), val = fp16(-0x1p+0)]; tensor var_13025 = mul(x = var_13024, y = const_387_promoted)[name = string("op_13025")]; tensor var_13030_begin_0 = const()[name = string("op_13030_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13030_end_0 = const()[name = string("op_13030_end_0"), val = tensor([1, 16, 1, 64])]; tensor var_13030_end_mask_0 = const()[name = string("op_13030_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13030 = slice_by_index(begin = var_13030_begin_0, end = var_13030_end_0, end_mask = var_13030_end_mask_0, x = q)[name = string("op_13030")]; int32 var_13032 = const()[name = string("op_13032"), val = int32(-1)]; bool var_13033_interleave_0 = const()[name = string("op_13033_interleave_0"), val = bool(false)]; tensor var_13033 = concat(axis = var_13032, interleave = var_13033_interleave_0, values = (var_13025, var_13030))[name = string("op_13033")]; tensor var_13034 = mul(x = var_13033, y = sin_1_cast_fp16)[name = string("op_13034")]; tensor query_states_109 = add(x = var_13019, y = var_13034)[name = string("query_states_109")]; tensor var_13037 = mul(x = k, y = cos_1_cast_fp16)[name = string("op_13037")]; tensor var_13042_begin_0 = const()[name = string("op_13042_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13042_end_0 = const()[name = string("op_13042_end_0"), val = tensor([1, 8, 1, 128])]; tensor var_13042_end_mask_0 = const()[name = string("op_13042_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13042 = slice_by_index(begin = var_13042_begin_0, end = var_13042_end_0, end_mask = var_13042_end_mask_0, x = k)[name = string("op_13042")]; fp16 const_388_promoted = const()[name = string("const_388_promoted"), val = fp16(-0x1p+0)]; tensor var_13043 = mul(x = var_13042, y = const_388_promoted)[name = string("op_13043")]; tensor var_13048_begin_0 = const()[name = string("op_13048_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13048_end_0 = const()[name = string("op_13048_end_0"), val = tensor([1, 8, 1, 64])]; tensor var_13048_end_mask_0 = const()[name = string("op_13048_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13048 = slice_by_index(begin = var_13048_begin_0, end = var_13048_end_0, end_mask = var_13048_end_mask_0, x = k)[name = string("op_13048")]; int32 var_13050 = const()[name = string("op_13050"), val = int32(-1)]; bool var_13051_interleave_0 = const()[name = string("op_13051_interleave_0"), val = bool(false)]; tensor var_13051 = concat(axis = var_13050, interleave = var_13051_interleave_0, values = (var_13043, var_13048))[name = string("op_13051")]; tensor var_13052 = mul(x = var_13051, y = sin_1_cast_fp16)[name = string("op_13052")]; tensor key_states_109 = add(x = var_13037, y = var_13052)[name = string("key_states_109")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; int32 concat_218_axis_0 = const()[name = string("concat_218_axis_0"), val = int32(0)]; bool concat_218_interleave_0 = const()[name = string("concat_218_interleave_0"), val = bool(false)]; tensor concat_218 = concat(axis = concat_218_axis_0, interleave = concat_218_interleave_0, values = (expand_dims_324, expand_dims_325, current_pos, expand_dims_327))[name = string("concat_218")]; tensor concat_219_values1_0 = const()[name = string("concat_219_values1_0"), val = tensor([0])]; tensor concat_219_values3_0 = const()[name = string("concat_219_values3_0"), val = tensor([0])]; int32 concat_219_axis_0 = const()[name = string("concat_219_axis_0"), val = int32(0)]; bool concat_219_interleave_0 = const()[name = string("concat_219_interleave_0"), val = bool(false)]; tensor concat_219 = concat(axis = concat_219_axis_0, interleave = concat_219_interleave_0, values = (expand_dims_328, concat_219_values1_0, var_1717, concat_219_values3_0))[name = string("concat_219")]; tensor model_model_kv_cache_0_internal_tensor_assign_55_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_cast_fp16 = slice_update(begin = concat_218, begin_mask = model_model_kv_cache_0_internal_tensor_assign_55_begin_mask_0, end = concat_219, end_mask = model_model_kv_cache_0_internal_tensor_assign_55_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_55_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_55_stride_0, update = key_states_109, x = coreml_update_state_109)[name = string("model_model_kv_cache_0_internal_tensor_assign_55_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_55_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_166_write_state")]; tensor coreml_update_state_110 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_166")]; tensor expand_dims_330 = const()[name = string("expand_dims_330"), val = tensor([55])]; tensor expand_dims_331 = const()[name = string("expand_dims_331"), val = tensor([0])]; tensor expand_dims_333 = const()[name = string("expand_dims_333"), val = tensor([0])]; tensor expand_dims_334 = const()[name = string("expand_dims_334"), val = tensor([56])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_330, expand_dims_331, current_pos, expand_dims_333))[name = string("concat_222")]; tensor concat_223_values1_0 = const()[name = string("concat_223_values1_0"), val = tensor([0])]; tensor concat_223_values3_0 = const()[name = string("concat_223_values3_0"), val = tensor([0])]; int32 concat_223_axis_0 = const()[name = string("concat_223_axis_0"), val = int32(0)]; bool concat_223_interleave_0 = const()[name = string("concat_223_interleave_0"), val = bool(false)]; tensor concat_223 = concat(axis = concat_223_axis_0, interleave = concat_223_interleave_0, values = (expand_dims_334, concat_223_values1_0, var_1717, concat_223_values3_0))[name = string("concat_223")]; tensor model_model_kv_cache_0_internal_tensor_assign_56_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_cast_fp16 = slice_update(begin = concat_222, begin_mask = model_model_kv_cache_0_internal_tensor_assign_56_begin_mask_0, end = concat_223, end_mask = model_model_kv_cache_0_internal_tensor_assign_56_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_56_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_56_stride_0, update = var_12969, x = coreml_update_state_110)[name = string("model_model_kv_cache_0_internal_tensor_assign_56_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_56_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_167_write_state")]; tensor coreml_update_state_111 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_167")]; tensor var_13107_begin_0 = const()[name = string("op_13107_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_13107_end_0 = const()[name = string("op_13107_end_0"), val = tensor([28, 8, 1536, 128])]; tensor var_13107_end_mask_0 = const()[name = string("op_13107_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_13107_cast_fp16 = slice_by_index(begin = var_13107_begin_0, end = var_13107_end_0, end_mask = var_13107_end_mask_0, x = coreml_update_state_111)[name = string("op_13107_cast_fp16")]; tensor key_cache_axes_0 = const()[name = string("key_cache_axes_0"), val = tensor([0])]; tensor key_cache_cast_fp16 = squeeze(axes = key_cache_axes_0, x = var_13107_cast_fp16)[name = string("key_cache_cast_fp16")]; tensor var_13114_begin_0 = const()[name = string("op_13114_begin_0"), val = tensor([55, 0, 0, 0])]; tensor var_13114_end_0 = const()[name = string("op_13114_end_0"), val = tensor([1, 8, 1536, 128])]; tensor var_13114_end_mask_0 = const()[name = string("op_13114_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13114_cast_fp16 = slice_by_index(begin = var_13114_begin_0, end = var_13114_end_0, end_mask = var_13114_end_mask_0, x = coreml_update_state_111)[name = string("op_13114_cast_fp16")]; tensor value_cache_axes_0 = const()[name = string("value_cache_axes_0"), val = tensor([0])]; tensor value_cache_cast_fp16 = squeeze(axes = value_cache_axes_0, x = var_13114_cast_fp16)[name = string("value_cache_cast_fp16")]; tensor var_13138_axes_0 = const()[name = string("op_13138_axes_0"), val = tensor([1])]; tensor var_13138_cast_fp16 = expand_dims(axes = var_13138_axes_0, x = key_cache_cast_fp16)[name = string("op_13138_cast_fp16")]; tensor var_13143 = const()[name = string("op_13143"), val = tensor([1, 2, 1, 1])]; tensor value_219_cast_fp16 = tile(reps = var_13143, x = var_13138_cast_fp16)[name = string("value_219_cast_fp16")]; tensor var_13149 = const()[name = string("op_13149"), val = tensor([1, 16, 1536, 128])]; tensor key_states_cast_fp16 = reshape(shape = var_13149, x = value_219_cast_fp16)[name = string("key_states_cast_fp16")]; tensor var_13152_axes_0 = const()[name = string("op_13152_axes_0"), val = tensor([1])]; tensor var_13152_cast_fp16 = expand_dims(axes = var_13152_axes_0, x = value_cache_cast_fp16)[name = string("op_13152_cast_fp16")]; tensor var_13157 = const()[name = string("op_13157"), val = tensor([1, 2, 1, 1])]; tensor value_cast_fp16 = tile(reps = var_13157, x = var_13152_cast_fp16)[name = string("value_cast_fp16")]; tensor var_13163 = const()[name = string("op_13163"), val = tensor([1, 16, 1536, 128])]; tensor value_states_165_cast_fp16 = reshape(shape = var_13163, x = value_cast_fp16)[name = string("value_states_165_cast_fp16")]; bool var_13178_transpose_x_1 = const()[name = string("op_13178_transpose_x_1"), val = bool(false)]; bool var_13178_transpose_y_1 = const()[name = string("op_13178_transpose_y_1"), val = bool(true)]; tensor var_13178 = matmul(transpose_x = var_13178_transpose_x_1, transpose_y = var_13178_transpose_y_1, x = query_states_109, y = key_states_cast_fp16)[name = string("op_13178")]; fp16 var_13179_to_fp16 = const()[name = string("op_13179_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_109_cast_fp16 = mul(x = var_13178, y = var_13179_to_fp16)[name = string("attention_109_cast_fp16")]; tensor attention_cast_fp16 = add(x = attention_109_cast_fp16, y = causal_mask)[name = string("attention_cast_fp16")]; int32 var_13188 = const()[name = string("op_13188"), val = int32(-1)]; tensor probabilities_cast_fp16 = softmax(axis = var_13188, x = attention_cast_fp16)[name = string("probabilities_cast_fp16")]; bool output_163_transpose_x_0 = const()[name = string("output_163_transpose_x_0"), val = bool(false)]; bool output_163_transpose_y_0 = const()[name = string("output_163_transpose_y_0"), val = bool(false)]; tensor output_163_cast_fp16 = matmul(transpose_x = output_163_transpose_x_0, transpose_y = output_163_transpose_y_0, x = probabilities_cast_fp16, y = value_states_165_cast_fp16)[name = string("output_163_cast_fp16")]; tensor var_13199_perm_0 = const()[name = string("op_13199_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_13205 = const()[name = string("op_13205"), val = tensor([1, 1, 2048])]; tensor var_13199_cast_fp16 = transpose(perm = var_13199_perm_0, x = output_163_cast_fp16)[name = string("transpose_4")]; tensor output_165_cast_fp16 = reshape(shape = var_13205, x = var_13199_cast_fp16)[name = string("output_165_cast_fp16")]; tensor var_13210 = const()[name = string("op_13210"), val = tensor([0, 2, 1])]; string var_13226_pad_type_0 = const()[name = string("op_13226_pad_type_0"), val = string("valid")]; int32 var_13226_groups_0 = const()[name = string("op_13226_groups_0"), val = int32(1)]; tensor var_13226_strides_0 = const()[name = string("op_13226_strides_0"), val = tensor([1])]; tensor var_13226_pad_0 = const()[name = string("op_13226_pad_0"), val = tensor([0, 0])]; tensor var_13226_dilations_0 = const()[name = string("op_13226_dilations_0"), val = tensor([1])]; tensor squeeze_27_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335949312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337522240))))[name = string("squeeze_27_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_13211_cast_fp16 = transpose(perm = var_13210, x = output_165_cast_fp16)[name = string("transpose_3")]; tensor var_13226_cast_fp16 = conv(dilations = var_13226_dilations_0, groups = var_13226_groups_0, pad = var_13226_pad_0, pad_type = var_13226_pad_type_0, strides = var_13226_strides_0, weight = squeeze_27_cast_fp16_to_fp32_to_fp16_palettized, x = var_13211_cast_fp16)[name = string("op_13226_cast_fp16")]; tensor var_13230 = const()[name = string("op_13230"), val = tensor([0, 2, 1])]; tensor attn_output_cast_fp16 = transpose(perm = var_13230, x = var_13226_cast_fp16)[name = string("transpose_2")]; tensor hidden_states_279_cast_fp16 = add(x = hidden_states_271_cast_fp16, y = attn_output_cast_fp16)[name = string("hidden_states_279_cast_fp16")]; int32 var_13245 = const()[name = string("op_13245"), val = int32(-1)]; fp16 const_389_promoted_to_fp16 = const()[name = string("const_389_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13247_cast_fp16 = mul(x = hidden_states_279_cast_fp16, y = const_389_promoted_to_fp16)[name = string("op_13247_cast_fp16")]; bool input_497_interleave_0 = const()[name = string("input_497_interleave_0"), val = bool(false)]; tensor input_497_cast_fp16 = concat(axis = var_13245, interleave = input_497_interleave_0, values = (hidden_states_279_cast_fp16, var_13247_cast_fp16))[name = string("input_497_cast_fp16")]; tensor normed_445_axes_0 = const()[name = string("normed_445_axes_0"), val = tensor([-1])]; fp16 var_13242_to_fp16 = const()[name = string("op_13242_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_445_cast_fp16 = layer_norm(axes = normed_445_axes_0, epsilon = var_13242_to_fp16, x = input_497_cast_fp16)[name = string("normed_445_cast_fp16")]; tensor normed_447_begin_0 = const()[name = string("normed_447_begin_0"), val = tensor([0, 0, 0])]; tensor normed_447_end_0 = const()[name = string("normed_447_end_0"), val = tensor([1, 1, 1024])]; tensor normed_447_end_mask_0 = const()[name = string("normed_447_end_mask_0"), val = tensor([true, true, false])]; tensor normed_447_cast_fp16 = slice_by_index(begin = normed_447_begin_0, end = normed_447_end_0, end_mask = normed_447_end_mask_0, x = normed_445_cast_fp16)[name = string("normed_447_cast_fp16")]; tensor const_391_promoted_to_fp16 = const()[name = string("const_391_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337538688)))]; tensor x_109_cast_fp16 = mul(x = normed_447_cast_fp16, y = const_391_promoted_to_fp16)[name = string("x_109_cast_fp16")]; tensor var_13267 = const()[name = string("op_13267"), val = tensor([0, 2, 1])]; tensor input_499_axes_0 = const()[name = string("input_499_axes_0"), val = tensor([2])]; tensor var_13268 = transpose(perm = var_13267, x = x_109_cast_fp16)[name = string("transpose_1")]; tensor input_499 = expand_dims(axes = input_499_axes_0, x = var_13268)[name = string("input_499")]; string input_501_pad_type_0 = const()[name = string("input_501_pad_type_0"), val = string("valid")]; tensor input_501_strides_0 = const()[name = string("input_501_strides_0"), val = tensor([1, 1])]; tensor input_501_pad_0 = const()[name = string("input_501_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_501_dilations_0 = const()[name = string("input_501_dilations_0"), val = tensor([1, 1])]; int32 input_501_groups_0 = const()[name = string("input_501_groups_0"), val = int32(1)]; tensor input_501 = conv(dilations = input_501_dilations_0, groups = input_501_groups_0, pad = input_501_pad_0, pad_type = input_501_pad_type_0, strides = input_501_strides_0, weight = model_model_layers_27_mlp_gate_proj_weight_palettized, x = input_499)[name = string("input_501")]; string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_27_mlp_up_proj_weight_palettized, x = input_499)[name = string("b")]; tensor c = silu(x = input_501)[name = string("c")]; tensor input_503 = mul(x = c, y = b)[name = string("input_503")]; string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_27_mlp_down_proj_weight_palettized, x = input_503)[name = string("e")]; tensor var_13290_axes_0 = const()[name = string("op_13290_axes_0"), val = tensor([2])]; tensor var_13290 = squeeze(axes = var_13290_axes_0, x = e)[name = string("op_13290")]; tensor var_13291 = const()[name = string("op_13291"), val = tensor([0, 2, 1])]; tensor var_13292 = transpose(perm = var_13291, x = var_13290)[name = string("transpose_0")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_279_cast_fp16, y = var_13292)[name = string("hidden_states_cast_fp16")]; int32 var_13306 = const()[name = string("op_13306"), val = int32(-1)]; fp16 const_392_promoted_to_fp16 = const()[name = string("const_392_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13308_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_392_promoted_to_fp16)[name = string("op_13308_cast_fp16")]; bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; tensor input_cast_fp16 = concat(axis = var_13306, interleave = input_interleave_0, values = (hidden_states_cast_fp16, var_13308_cast_fp16))[name = string("input_cast_fp16")]; tensor normed_449_axes_0 = const()[name = string("normed_449_axes_0"), val = tensor([-1])]; fp16 var_13303_to_fp16 = const()[name = string("op_13303_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_449_cast_fp16 = layer_norm(axes = normed_449_axes_0, epsilon = var_13303_to_fp16, x = input_cast_fp16)[name = string("normed_449_cast_fp16")]; tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 1, 1024])]; tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_449_cast_fp16)[name = string("normed_cast_fp16")]; tensor const_394_promoted_to_fp16 = const()[name = string("const_394_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337540800)))]; tensor output_hidden_states = mul(x = normed_cast_fp16, y = const_394_promoted_to_fp16)[name = string("op_13316_cast_fp16")]; tensor position_ids_tmp = identity(x = position_ids)[name = string("position_ids_tmp")]; } -> (output_hidden_states); func prefill(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { tensor model_model_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1572992))))[name = string("model_model_layers_0_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1605824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2392320))))[name = string("model_model_layers_0_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2408768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3195264))))[name = string("model_model_layers_0_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3211712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5571072))))[name = string("model_model_layers_0_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5620288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(7979648))))[name = string("model_model_layers_0_mlp_up_proj_weight_palettized")]; tensor model_model_layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8028864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10388224))))[name = string("model_model_layers_0_mlp_down_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10404672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11977600))))[name = string("model_model_layers_1_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12010432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12796928))))[name = string("model_model_layers_1_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12813376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13599872))))[name = string("model_model_layers_1_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13616320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15975680))))[name = string("model_model_layers_1_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16024896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18384256))))[name = string("model_model_layers_1_mlp_up_proj_weight_palettized")]; tensor model_model_layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18433472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20792832))))[name = string("model_model_layers_1_mlp_down_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20809280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22382208))))[name = string("model_model_layers_2_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22415040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23201536))))[name = string("model_model_layers_2_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23217984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24004480))))[name = string("model_model_layers_2_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24020928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26380288))))[name = string("model_model_layers_2_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26429504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28788864))))[name = string("model_model_layers_2_mlp_up_proj_weight_palettized")]; tensor model_model_layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28838080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31197440))))[name = string("model_model_layers_2_mlp_down_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31213888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(32786816))))[name = string("model_model_layers_3_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(32819648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33606144))))[name = string("model_model_layers_3_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33622592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34409088))))[name = string("model_model_layers_3_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34425536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36784896))))[name = string("model_model_layers_3_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36834112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39193472))))[name = string("model_model_layers_3_mlp_up_proj_weight_palettized")]; tensor model_model_layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39242688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41602048))))[name = string("model_model_layers_3_mlp_down_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41618496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43191424))))[name = string("model_model_layers_4_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43224256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44010752))))[name = string("model_model_layers_4_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44027200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44813696))))[name = string("model_model_layers_4_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44830144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47189504))))[name = string("model_model_layers_4_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47238720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49598080))))[name = string("model_model_layers_4_mlp_up_proj_weight_palettized")]; tensor model_model_layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49647296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52006656))))[name = string("model_model_layers_4_mlp_down_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52023104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53596032))))[name = string("model_model_layers_5_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53628864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(54415360))))[name = string("model_model_layers_5_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(54431808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55218304))))[name = string("model_model_layers_5_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55234752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57594112))))[name = string("model_model_layers_5_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57643328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60002688))))[name = string("model_model_layers_5_mlp_up_proj_weight_palettized")]; tensor model_model_layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60051904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62411264))))[name = string("model_model_layers_5_mlp_down_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62427712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64000640))))[name = string("model_model_layers_6_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64033472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64819968))))[name = string("model_model_layers_6_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64836416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65622912))))[name = string("model_model_layers_6_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65639360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67998720))))[name = string("model_model_layers_6_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68047936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70407296))))[name = string("model_model_layers_6_mlp_up_proj_weight_palettized")]; tensor model_model_layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70456512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72815872))))[name = string("model_model_layers_6_mlp_down_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72832320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74405248))))[name = string("model_model_layers_7_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74438080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75224576))))[name = string("model_model_layers_7_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75241024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76027520))))[name = string("model_model_layers_7_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76043968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78403328))))[name = string("model_model_layers_7_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78452544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80811904))))[name = string("model_model_layers_7_mlp_up_proj_weight_palettized")]; tensor model_model_layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80861120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(83220480))))[name = string("model_model_layers_7_mlp_down_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(83236928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84809856))))[name = string("model_model_layers_8_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84842688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85629184))))[name = string("model_model_layers_8_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85645632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86432128))))[name = string("model_model_layers_8_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86448576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88807936))))[name = string("model_model_layers_8_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88857152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91216512))))[name = string("model_model_layers_8_mlp_up_proj_weight_palettized")]; tensor model_model_layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91265728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93625088))))[name = string("model_model_layers_8_mlp_down_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93641536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95214464))))[name = string("model_model_layers_9_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95247296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96033792))))[name = string("model_model_layers_9_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96050240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96836736))))[name = string("model_model_layers_9_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96853184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99212544))))[name = string("model_model_layers_9_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99261760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101621120))))[name = string("model_model_layers_9_mlp_up_proj_weight_palettized")]; tensor model_model_layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101670336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104029696))))[name = string("model_model_layers_9_mlp_down_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104046144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105619072))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105651904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106438400))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106454848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(107241344))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(107257792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109617152))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(109666368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112025728))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112074944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(114434304))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(114450752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116023680))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116056512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116843008))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116859456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117645952))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(117662400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120021760))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120070976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(122430336))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(122479552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124838912))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124855360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(126428288))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(126461120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(127247616))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(127264064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128050560))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128067008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130426368))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(130475584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(132834944))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(132884160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135243520))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(135259968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136832896))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136865728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137652224))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137668672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138455168))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138471616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(140830976))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(140880192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143239552))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143288768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145648128))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145664576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147237504))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(147270336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148056832))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148073280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148859776))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148876224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151235584))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151284800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153644160))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153693376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(156052736))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(156069184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(157642112))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(157674944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158461440))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158477888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159264384))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159280832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161640192))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161689408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(164048768))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(164097984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(166457344))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; tensor model_model_layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(166473792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168046720))))[name = string("model_model_layers_16_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_16_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168079552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168866048))))[name = string("model_model_layers_16_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_16_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(168882496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169668992))))[name = string("model_model_layers_16_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(169685440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172044800))))[name = string("model_model_layers_16_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172094016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174453376))))[name = string("model_model_layers_16_mlp_up_proj_weight_palettized")]; tensor model_model_layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174502592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176861952))))[name = string("model_model_layers_16_mlp_down_proj_weight_palettized")]; tensor model_model_layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176878400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178451328))))[name = string("model_model_layers_17_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_17_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(178484160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(179270656))))[name = string("model_model_layers_17_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_17_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(179287104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180073600))))[name = string("model_model_layers_17_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180090048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(182449408))))[name = string("model_model_layers_17_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(182498624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184857984))))[name = string("model_model_layers_17_mlp_up_proj_weight_palettized")]; tensor model_model_layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184907200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(187266560))))[name = string("model_model_layers_17_mlp_down_proj_weight_palettized")]; tensor model_model_layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(187283008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(188855936))))[name = string("model_model_layers_18_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_18_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(188888768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189675264))))[name = string("model_model_layers_18_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_18_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189691712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190478208))))[name = string("model_model_layers_18_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190494656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(192854016))))[name = string("model_model_layers_18_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(192903232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195262592))))[name = string("model_model_layers_18_mlp_up_proj_weight_palettized")]; tensor model_model_layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195311808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197671168))))[name = string("model_model_layers_18_mlp_down_proj_weight_palettized")]; tensor model_model_layers_19_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197687616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(199260544))))[name = string("model_model_layers_19_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_19_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(199293376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200079872))))[name = string("model_model_layers_19_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_19_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200096320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200882816))))[name = string("model_model_layers_19_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_19_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200899264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203258624))))[name = string("model_model_layers_19_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_19_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(203307840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205667200))))[name = string("model_model_layers_19_mlp_up_proj_weight_palettized")]; tensor model_model_layers_19_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205716416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(208075776))))[name = string("model_model_layers_19_mlp_down_proj_weight_palettized")]; tensor model_model_layers_20_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(208092224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(209665152))))[name = string("model_model_layers_20_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_20_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(209697984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210484480))))[name = string("model_model_layers_20_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_20_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210500928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211287424))))[name = string("model_model_layers_20_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_20_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(211303872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213663232))))[name = string("model_model_layers_20_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_20_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213712448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216071808))))[name = string("model_model_layers_20_mlp_up_proj_weight_palettized")]; tensor model_model_layers_20_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216121024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218480384))))[name = string("model_model_layers_20_mlp_down_proj_weight_palettized")]; tensor model_model_layers_21_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218496832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220069760))))[name = string("model_model_layers_21_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_21_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220102592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220889088))))[name = string("model_model_layers_21_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_21_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(220905536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(221692032))))[name = string("model_model_layers_21_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_21_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(221708480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224067840))))[name = string("model_model_layers_21_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_21_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224117056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226476416))))[name = string("model_model_layers_21_mlp_up_proj_weight_palettized")]; tensor model_model_layers_21_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(226525632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228884992))))[name = string("model_model_layers_21_mlp_down_proj_weight_palettized")]; tensor model_model_layers_22_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(228901440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230474368))))[name = string("model_model_layers_22_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_22_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230507200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(231293696))))[name = string("model_model_layers_22_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_22_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(231310144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232096640))))[name = string("model_model_layers_22_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_22_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(232113088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234472448))))[name = string("model_model_layers_22_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_22_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234521664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236881024))))[name = string("model_model_layers_22_mlp_up_proj_weight_palettized")]; tensor model_model_layers_22_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236930240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239289600))))[name = string("model_model_layers_22_mlp_down_proj_weight_palettized")]; tensor model_model_layers_23_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239306048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240878976))))[name = string("model_model_layers_23_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_23_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240911808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241698304))))[name = string("model_model_layers_23_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_23_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(241714752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242501248))))[name = string("model_model_layers_23_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_23_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(242517696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244877056))))[name = string("model_model_layers_23_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_23_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244926272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247285632))))[name = string("model_model_layers_23_mlp_up_proj_weight_palettized")]; tensor model_model_layers_23_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247334848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249694208))))[name = string("model_model_layers_23_mlp_down_proj_weight_palettized")]; tensor model_model_layers_24_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(249710656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251283584))))[name = string("model_model_layers_24_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_24_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251316416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252102912))))[name = string("model_model_layers_24_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_24_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252119360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252905856))))[name = string("model_model_layers_24_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_24_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(252922304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255281664))))[name = string("model_model_layers_24_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_24_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255330880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257690240))))[name = string("model_model_layers_24_mlp_up_proj_weight_palettized")]; tensor model_model_layers_24_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257739456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260098816))))[name = string("model_model_layers_24_mlp_down_proj_weight_palettized")]; tensor model_model_layers_25_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260115264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261688192))))[name = string("model_model_layers_25_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_25_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261721024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262507520))))[name = string("model_model_layers_25_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_25_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262523968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263310464))))[name = string("model_model_layers_25_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_25_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(263326912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265686272))))[name = string("model_model_layers_25_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_25_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265735488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268094848))))[name = string("model_model_layers_25_mlp_up_proj_weight_palettized")]; tensor model_model_layers_25_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(268144064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270503424))))[name = string("model_model_layers_25_mlp_down_proj_weight_palettized")]; tensor model_model_layers_26_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270519872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272092800))))[name = string("model_model_layers_26_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_26_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272125632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272912128))))[name = string("model_model_layers_26_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_26_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272928576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273715072))))[name = string("model_model_layers_26_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_26_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273731520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276090880))))[name = string("model_model_layers_26_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_26_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(276140096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278499456))))[name = string("model_model_layers_26_mlp_up_proj_weight_palettized")]; tensor model_model_layers_26_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278548672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280908032))))[name = string("model_model_layers_26_mlp_down_proj_weight_palettized")]; tensor model_model_layers_27_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(280924480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282497408))))[name = string("model_model_layers_27_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_27_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(282530240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283316736))))[name = string("model_model_layers_27_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_27_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283333184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284119680))))[name = string("model_model_layers_27_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_27_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284136128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286495488))))[name = string("model_model_layers_27_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_27_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286544704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288904064))))[name = string("model_model_layers_27_mlp_up_proj_weight_palettized")]; tensor model_model_layers_27_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288953280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291312640))))[name = string("model_model_layers_27_mlp_down_proj_weight_palettized")]; int32 var_1500_batch_dims_0 = const()[name = string("op_1500_batch_dims_0"), val = int32(0)]; bool var_1500_validate_indices_0 = const()[name = string("op_1500_validate_indices_0"), val = bool(false)]; tensor var_1492_to_fp16 = const()[name = string("op_1492_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292115584)))]; string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; string cast_230_dtype_0 = const()[name = string("cast_230_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; tensor cast_230 = cast(dtype = cast_230_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; tensor greater_equal_0 = greater_equal(x = cast_230, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_224 = const()[name = string("slice_by_index_224"), val = int32(3072)]; tensor add_0 = add(x = cast_230, y = slice_by_index_224)[name = string("add_0")]; tensor select_0 = select(a = cast_230, b = add_0, cond = greater_equal_0)[name = string("select_0")]; string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(3072)]; tensor add_0_1 = add(x = cast_0, y = slice_by_index_0)[name = string("add_0_1")]; tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_1500_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_1500_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; tensor op_1500_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_1500_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_1500_batch_dims_0, indices = select_0_1, validate_indices = var_1500_validate_indices_0, x = var_1492_to_fp16)[name = string("op_1500_cast_fp16_cast_uint16_cast_uint16")]; tensor var_1505 = const()[name = string("op_1505"), val = tensor([1, 64, 1, 128])]; tensor cosine_1_cast_fp16 = reshape(shape = var_1505, x = op_1500_cast_fp16_cast_uint16_cast_uint16)[name = string("cosine_1_cast_fp16")]; int32 var_1515_axis_0 = const()[name = string("op_1515_axis_0"), val = int32(1)]; int32 var_1515_batch_dims_0 = const()[name = string("op_1515_batch_dims_0"), val = int32(0)]; bool var_1515_validate_indices_0 = const()[name = string("op_1515_validate_indices_0"), val = bool(false)]; tensor var_1507_to_fp16 = const()[name = string("op_1507_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(291329088)))]; string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; tensor var_1515_cast_fp16_cast_uint16 = gather(axis = var_1515_axis_0, batch_dims = var_1515_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_1515_validate_indices_0, x = var_1507_to_fp16)[name = string("op_1515_cast_fp16_cast_uint16")]; tensor var_1520 = const()[name = string("op_1520"), val = tensor([1, 64, 1, 128])]; tensor sine_1_cast_fp16 = reshape(shape = var_1520, x = var_1515_cast_fp16_cast_uint16)[name = string("sine_1_cast_fp16")]; int32 var_1543 = const()[name = string("op_1543"), val = int32(-1)]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1545_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_1545_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_1543, interleave = input_1_interleave_0, values = (hidden_states, var_1545_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_1540_to_fp16 = const()[name = string("op_1540_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_1540_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 64, 1024])]; tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; tensor const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292902080)))]; tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_2_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor var_1557 = const()[name = string("op_1557"), val = tensor([0, 2, 1])]; tensor var_1560_axes_0 = const()[name = string("op_1560_axes_0"), val = tensor([2])]; tensor var_1558_cast_fp16 = transpose(perm = var_1557, x = hidden_states_3_cast_fp16)[name = string("transpose_253")]; tensor var_1560_cast_fp16 = expand_dims(axes = var_1560_axes_0, x = var_1558_cast_fp16)[name = string("op_1560_cast_fp16")]; string var_1576_pad_type_0 = const()[name = string("op_1576_pad_type_0"), val = string("valid")]; tensor var_1576_strides_0 = const()[name = string("op_1576_strides_0"), val = tensor([1, 1])]; tensor var_1576_pad_0 = const()[name = string("op_1576_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1576_dilations_0 = const()[name = string("op_1576_dilations_0"), val = tensor([1, 1])]; int32 var_1576_groups_0 = const()[name = string("op_1576_groups_0"), val = int32(1)]; tensor var_1576 = conv(dilations = var_1576_dilations_0, groups = var_1576_groups_0, pad = var_1576_pad_0, pad_type = var_1576_pad_type_0, strides = var_1576_strides_0, weight = model_model_layers_0_self_attn_q_proj_weight_palettized, x = var_1560_cast_fp16)[name = string("op_1576")]; tensor var_1581 = const()[name = string("op_1581"), val = tensor([1, 16, 128, 64])]; tensor var_1582 = reshape(shape = var_1581, x = var_1576)[name = string("op_1582")]; tensor var_1587 = const()[name = string("op_1587"), val = tensor([0, 1, 3, 2])]; string var_1599_pad_type_0 = const()[name = string("op_1599_pad_type_0"), val = string("valid")]; tensor var_1599_strides_0 = const()[name = string("op_1599_strides_0"), val = tensor([1, 1])]; tensor var_1599_pad_0 = const()[name = string("op_1599_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1599_dilations_0 = const()[name = string("op_1599_dilations_0"), val = tensor([1, 1])]; int32 var_1599_groups_0 = const()[name = string("op_1599_groups_0"), val = int32(1)]; tensor var_1599 = conv(dilations = var_1599_dilations_0, groups = var_1599_groups_0, pad = var_1599_pad_0, pad_type = var_1599_pad_type_0, strides = var_1599_strides_0, weight = model_model_layers_0_self_attn_k_proj_weight_palettized, x = var_1560_cast_fp16)[name = string("op_1599")]; tensor var_1604 = const()[name = string("op_1604"), val = tensor([1, 8, 128, 64])]; tensor var_1605 = reshape(shape = var_1604, x = var_1599)[name = string("op_1605")]; tensor var_1610 = const()[name = string("op_1610"), val = tensor([0, 1, 3, 2])]; string var_1622_pad_type_0 = const()[name = string("op_1622_pad_type_0"), val = string("valid")]; tensor var_1622_strides_0 = const()[name = string("op_1622_strides_0"), val = tensor([1, 1])]; tensor var_1622_pad_0 = const()[name = string("op_1622_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1622_dilations_0 = const()[name = string("op_1622_dilations_0"), val = tensor([1, 1])]; int32 var_1622_groups_0 = const()[name = string("op_1622_groups_0"), val = int32(1)]; tensor var_1622 = conv(dilations = var_1622_dilations_0, groups = var_1622_groups_0, pad = var_1622_pad_0, pad_type = var_1622_pad_type_0, strides = var_1622_strides_0, weight = model_model_layers_0_self_attn_v_proj_weight_palettized, x = var_1560_cast_fp16)[name = string("op_1622")]; tensor var_1627 = const()[name = string("op_1627"), val = tensor([1, 8, 128, 64])]; tensor var_1628 = reshape(shape = var_1627, x = var_1622)[name = string("op_1628")]; tensor var_1633 = const()[name = string("op_1633"), val = tensor([0, 1, 3, 2])]; int32 var_1646 = const()[name = string("op_1646"), val = int32(-1)]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_5 = transpose(perm = var_1587, x = var_1582)[name = string("transpose_252")]; tensor var_1648 = mul(x = hidden_states_5, y = const_3_promoted)[name = string("op_1648")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_1646, interleave = input_5_interleave_0, values = (hidden_states_5, var_1648))[name = string("input_5")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_1643_to_fp16 = const()[name = string("op_1643_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_1643_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_7 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7")]; tensor const_5 = const()[name = string("const_5"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292904192)))]; tensor q_1 = mul(x = normed_7, y = const_5)[name = string("q_1")]; int32 var_1668 = const()[name = string("op_1668"), val = int32(-1)]; fp16 const_6_promoted = const()[name = string("const_6_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_7 = transpose(perm = var_1610, x = var_1605)[name = string("transpose_251")]; tensor var_1670 = mul(x = hidden_states_7, y = const_6_promoted)[name = string("op_1670")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7 = concat(axis = var_1668, interleave = input_7_interleave_0, values = (hidden_states_7, var_1670))[name = string("input_7")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_1665_to_fp16 = const()[name = string("op_1665_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_1665_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_11 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11")]; tensor const_8 = const()[name = string("const_8"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292904512)))]; tensor k_1 = mul(x = normed_11, y = const_8)[name = string("k_1")]; tensor var_1683 = const()[name = string("op_1683"), val = tensor([0, 2, 1, 3])]; tensor var_1689 = const()[name = string("op_1689"), val = tensor([0, 2, 1, 3])]; tensor cos_1 = transpose(perm = var_1683, x = cosine_1_cast_fp16)[name = string("transpose_250")]; tensor var_1691 = mul(x = q_1, y = cos_1)[name = string("op_1691")]; tensor var_1696_begin_0 = const()[name = string("op_1696_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_1696_end_0 = const()[name = string("op_1696_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_1696_end_mask_0 = const()[name = string("op_1696_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1696 = slice_by_index(begin = var_1696_begin_0, end = var_1696_end_0, end_mask = var_1696_end_mask_0, x = q_1)[name = string("op_1696")]; fp16 const_9_promoted = const()[name = string("const_9_promoted"), val = fp16(-0x1p+0)]; tensor var_1697 = mul(x = var_1696, y = const_9_promoted)[name = string("op_1697")]; tensor var_1702_begin_0 = const()[name = string("op_1702_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1702_end_0 = const()[name = string("op_1702_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_1702_end_mask_0 = const()[name = string("op_1702_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_1702 = slice_by_index(begin = var_1702_begin_0, end = var_1702_end_0, end_mask = var_1702_end_mask_0, x = q_1)[name = string("op_1702")]; int32 var_1704 = const()[name = string("op_1704"), val = int32(-1)]; bool var_1705_interleave_0 = const()[name = string("op_1705_interleave_0"), val = bool(false)]; tensor var_1705 = concat(axis = var_1704, interleave = var_1705_interleave_0, values = (var_1697, var_1702))[name = string("op_1705")]; tensor sin_1 = transpose(perm = var_1689, x = sine_1_cast_fp16)[name = string("transpose_249")]; tensor var_1706 = mul(x = var_1705, y = sin_1)[name = string("op_1706")]; tensor query_1 = add(x = var_1691, y = var_1706)[name = string("query_1")]; tensor var_1709 = mul(x = k_1, y = cos_1)[name = string("op_1709")]; tensor var_1714_begin_0 = const()[name = string("op_1714_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_1714_end_0 = const()[name = string("op_1714_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_1714_end_mask_0 = const()[name = string("op_1714_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_1714 = slice_by_index(begin = var_1714_begin_0, end = var_1714_end_0, end_mask = var_1714_end_mask_0, x = k_1)[name = string("op_1714")]; fp16 const_10_promoted = const()[name = string("const_10_promoted"), val = fp16(-0x1p+0)]; tensor var_1715 = mul(x = var_1714, y = const_10_promoted)[name = string("op_1715")]; tensor var_1720_begin_0 = const()[name = string("op_1720_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1720_end_0 = const()[name = string("op_1720_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_1720_end_mask_0 = const()[name = string("op_1720_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_1720 = slice_by_index(begin = var_1720_begin_0, end = var_1720_end_0, end_mask = var_1720_end_mask_0, x = k_1)[name = string("op_1720")]; int32 var_1722 = const()[name = string("op_1722"), val = int32(-1)]; bool var_1723_interleave_0 = const()[name = string("op_1723_interleave_0"), val = bool(false)]; tensor var_1723 = concat(axis = var_1722, interleave = var_1723_interleave_0, values = (var_1715, var_1720))[name = string("op_1723")]; tensor var_1724 = mul(x = var_1723, y = sin_1)[name = string("op_1724")]; tensor key_1 = add(x = var_1709, y = var_1724)[name = string("key_1")]; tensor seq_length_1 = const()[name = string("seq_length_1"), val = tensor([64])]; tensor var_1746 = add(x = current_pos, y = seq_length_1)[name = string("op_1746")]; tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_1746, concat_3_values3_0))[name = string("concat_3")]; tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_1, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_168_write_state")]; tensor coreml_update_state_56 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_168")]; tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([28])]; tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([29])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_1746, concat_7_values3_0))[name = string("concat_7")]; tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_1 = transpose(perm = var_1633, x = var_1628)[name = string("transpose_248")]; tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = value_1, x = coreml_update_state_56)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_169_write_state")]; tensor coreml_update_state_57 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_169")]; tensor var_1795_begin_0 = const()[name = string("op_1795_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1795_end_0 = const()[name = string("op_1795_end_0"), val = tensor([1, 8, 1536, 128])]; tensor var_1795_end_mask_0 = const()[name = string("op_1795_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1795_cast_fp16 = slice_by_index(begin = var_1795_begin_0, end = var_1795_end_0, end_mask = var_1795_end_mask_0, x = coreml_update_state_57)[name = string("op_1795_cast_fp16")]; tensor key_cache_1_axes_0 = const()[name = string("key_cache_1_axes_0"), val = tensor([0])]; tensor key_cache_1_cast_fp16 = squeeze(axes = key_cache_1_axes_0, x = var_1795_cast_fp16)[name = string("key_cache_1_cast_fp16")]; tensor var_1802_begin_0 = const()[name = string("op_1802_begin_0"), val = tensor([28, 0, 0, 0])]; tensor var_1802_end_0 = const()[name = string("op_1802_end_0"), val = tensor([29, 8, 1536, 128])]; tensor var_1802_end_mask_0 = const()[name = string("op_1802_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1802_cast_fp16 = slice_by_index(begin = var_1802_begin_0, end = var_1802_end_0, end_mask = var_1802_end_mask_0, x = coreml_update_state_57)[name = string("op_1802_cast_fp16")]; tensor value_cache_1_axes_0 = const()[name = string("value_cache_1_axes_0"), val = tensor([0])]; tensor value_cache_1_cast_fp16 = squeeze(axes = value_cache_1_axes_0, x = var_1802_cast_fp16)[name = string("value_cache_1_cast_fp16")]; tensor var_1826_axes_0 = const()[name = string("op_1826_axes_0"), val = tensor([1])]; tensor var_1826_cast_fp16 = expand_dims(axes = var_1826_axes_0, x = key_cache_1_cast_fp16)[name = string("op_1826_cast_fp16")]; tensor var_1831 = const()[name = string("op_1831"), val = tensor([1, 2, 1, 1])]; tensor value_5_cast_fp16 = tile(reps = var_1831, x = var_1826_cast_fp16)[name = string("value_5_cast_fp16")]; tensor var_1837 = const()[name = string("op_1837"), val = tensor([1, 16, 1536, 128])]; tensor key_states_3_cast_fp16 = reshape(shape = var_1837, x = value_5_cast_fp16)[name = string("key_states_3_cast_fp16")]; tensor var_1840_axes_0 = const()[name = string("op_1840_axes_0"), val = tensor([1])]; tensor var_1840_cast_fp16 = expand_dims(axes = var_1840_axes_0, x = value_cache_1_cast_fp16)[name = string("op_1840_cast_fp16")]; tensor var_1845 = const()[name = string("op_1845"), val = tensor([1, 2, 1, 1])]; tensor value_9_cast_fp16 = tile(reps = var_1845, x = var_1840_cast_fp16)[name = string("value_9_cast_fp16")]; bool var_1866_transpose_x_0 = const()[name = string("op_1866_transpose_x_0"), val = bool(false)]; bool var_1866_transpose_y_0 = const()[name = string("op_1866_transpose_y_0"), val = bool(true)]; tensor var_1866 = matmul(transpose_x = var_1866_transpose_x_0, transpose_y = var_1866_transpose_y_0, x = query_1, y = key_states_3_cast_fp16)[name = string("op_1866")]; fp16 var_1867_to_fp16 = const()[name = string("op_1867_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_1_cast_fp16 = mul(x = var_1866, y = var_1867_to_fp16)[name = string("attention_1_cast_fp16")]; tensor attention_3_cast_fp16 = add(x = attention_1_cast_fp16, y = causal_mask)[name = string("attention_3_cast_fp16")]; int32 var_1876 = const()[name = string("op_1876"), val = int32(-1)]; tensor var_1878_cast_fp16 = softmax(axis = var_1876, x = attention_3_cast_fp16)[name = string("op_1878_cast_fp16")]; tensor concat_12 = const()[name = string("concat_12"), val = tensor([16, 64, 1536])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_12, x = var_1878_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor concat_13 = const()[name = string("concat_13"), val = tensor([16, 1536, 128])]; tensor reshape_1_cast_fp16 = reshape(shape = concat_13, x = value_9_cast_fp16)[name = string("reshape_1_cast_fp16")]; bool matmul_0_transpose_x_0 = const()[name = string("matmul_0_transpose_x_0"), val = bool(false)]; bool matmul_0_transpose_y_0 = const()[name = string("matmul_0_transpose_y_0"), val = bool(false)]; tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = reshape_0_cast_fp16, y = reshape_1_cast_fp16)[name = string("matmul_0_cast_fp16")]; tensor concat_17 = const()[name = string("concat_17"), val = tensor([1, 16, 64, 128])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_17, x = matmul_0_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor var_1890_perm_0 = const()[name = string("op_1890_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1896 = const()[name = string("op_1896"), val = tensor([1, 64, 2048])]; tensor var_1890_cast_fp16 = transpose(perm = var_1890_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_247")]; tensor output_3_cast_fp16 = reshape(shape = var_1896, x = var_1890_cast_fp16)[name = string("output_3_cast_fp16")]; tensor var_1901 = const()[name = string("op_1901"), val = tensor([0, 2, 1])]; string var_1917_pad_type_0 = const()[name = string("op_1917_pad_type_0"), val = string("valid")]; int32 var_1917_groups_0 = const()[name = string("op_1917_groups_0"), val = int32(1)]; tensor var_1917_strides_0 = const()[name = string("op_1917_strides_0"), val = tensor([1])]; tensor var_1917_pad_0 = const()[name = string("op_1917_pad_0"), val = tensor([0, 0])]; tensor var_1917_dilations_0 = const()[name = string("op_1917_dilations_0"), val = tensor([1])]; tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292904832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294477760))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_1902_cast_fp16 = transpose(perm = var_1901, x = output_3_cast_fp16)[name = string("transpose_246")]; tensor var_1917_cast_fp16 = conv(dilations = var_1917_dilations_0, groups = var_1917_groups_0, pad = var_1917_pad_0, pad_type = var_1917_pad_type_0, strides = var_1917_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1902_cast_fp16)[name = string("op_1917_cast_fp16")]; tensor var_1921 = const()[name = string("op_1921"), val = tensor([0, 2, 1])]; tensor attn_output_1_cast_fp16 = transpose(perm = var_1921, x = var_1917_cast_fp16)[name = string("transpose_245")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states, y = attn_output_1_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; int32 var_1936 = const()[name = string("op_1936"), val = int32(-1)]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1938_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_1938_cast_fp16")]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11_cast_fp16 = concat(axis = var_1936, interleave = input_11_interleave_0, values = (hidden_states_9_cast_fp16, var_1938_cast_fp16))[name = string("input_11_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_1933_to_fp16 = const()[name = string("op_1933_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1933_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 64, 1024])]; tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; tensor const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294494208)))]; tensor x_1_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_14_promoted_to_fp16)[name = string("x_1_cast_fp16")]; tensor var_1958 = const()[name = string("op_1958"), val = tensor([0, 2, 1])]; tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; tensor var_1959 = transpose(perm = var_1958, x = x_1_cast_fp16)[name = string("transpose_244")]; tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_1959)[name = string("input_13")]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = model_model_layers_0_mlp_gate_proj_weight_palettized, x = input_13)[name = string("input_15")]; string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_0_mlp_up_proj_weight_palettized, x = input_13)[name = string("b_1")]; tensor c_1 = silu(x = input_15)[name = string("c_1")]; tensor input_17 = mul(x = c_1, y = b_1)[name = string("input_17")]; string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_0_mlp_down_proj_weight_palettized, x = input_17)[name = string("e_1")]; tensor var_1981_axes_0 = const()[name = string("op_1981_axes_0"), val = tensor([2])]; tensor var_1981 = squeeze(axes = var_1981_axes_0, x = e_1)[name = string("op_1981")]; tensor var_1982 = const()[name = string("op_1982"), val = tensor([0, 2, 1])]; tensor var_1983 = transpose(perm = var_1982, x = var_1981)[name = string("transpose_243")]; tensor hidden_states_11_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = var_1983)[name = string("hidden_states_11_cast_fp16")]; int32 var_1997 = const()[name = string("op_1997"), val = int32(-1)]; fp16 const_15_promoted_to_fp16 = const()[name = string("const_15_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1999_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_15_promoted_to_fp16)[name = string("op_1999_cast_fp16")]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_1997, interleave = input_19_interleave_0, values = (hidden_states_11_cast_fp16, var_1999_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_1994_to_fp16 = const()[name = string("op_1994_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1994_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 64, 1024])]; tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; tensor const_17_promoted_to_fp16 = const()[name = string("const_17_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294496320)))]; tensor hidden_states_13_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_17_promoted_to_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor var_2011 = const()[name = string("op_2011"), val = tensor([0, 2, 1])]; tensor var_2014_axes_0 = const()[name = string("op_2014_axes_0"), val = tensor([2])]; tensor var_2012_cast_fp16 = transpose(perm = var_2011, x = hidden_states_13_cast_fp16)[name = string("transpose_242")]; tensor var_2014_cast_fp16 = expand_dims(axes = var_2014_axes_0, x = var_2012_cast_fp16)[name = string("op_2014_cast_fp16")]; string var_2030_pad_type_0 = const()[name = string("op_2030_pad_type_0"), val = string("valid")]; tensor var_2030_strides_0 = const()[name = string("op_2030_strides_0"), val = tensor([1, 1])]; tensor var_2030_pad_0 = const()[name = string("op_2030_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2030_dilations_0 = const()[name = string("op_2030_dilations_0"), val = tensor([1, 1])]; int32 var_2030_groups_0 = const()[name = string("op_2030_groups_0"), val = int32(1)]; tensor var_2030 = conv(dilations = var_2030_dilations_0, groups = var_2030_groups_0, pad = var_2030_pad_0, pad_type = var_2030_pad_type_0, strides = var_2030_strides_0, weight = model_model_layers_1_self_attn_q_proj_weight_palettized, x = var_2014_cast_fp16)[name = string("op_2030")]; tensor var_2035 = const()[name = string("op_2035"), val = tensor([1, 16, 128, 64])]; tensor var_2036 = reshape(shape = var_2035, x = var_2030)[name = string("op_2036")]; tensor var_2041 = const()[name = string("op_2041"), val = tensor([0, 1, 3, 2])]; string var_2053_pad_type_0 = const()[name = string("op_2053_pad_type_0"), val = string("valid")]; tensor var_2053_strides_0 = const()[name = string("op_2053_strides_0"), val = tensor([1, 1])]; tensor var_2053_pad_0 = const()[name = string("op_2053_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2053_dilations_0 = const()[name = string("op_2053_dilations_0"), val = tensor([1, 1])]; int32 var_2053_groups_0 = const()[name = string("op_2053_groups_0"), val = int32(1)]; tensor var_2053 = conv(dilations = var_2053_dilations_0, groups = var_2053_groups_0, pad = var_2053_pad_0, pad_type = var_2053_pad_type_0, strides = var_2053_strides_0, weight = model_model_layers_1_self_attn_k_proj_weight_palettized, x = var_2014_cast_fp16)[name = string("op_2053")]; tensor var_2058 = const()[name = string("op_2058"), val = tensor([1, 8, 128, 64])]; tensor var_2059 = reshape(shape = var_2058, x = var_2053)[name = string("op_2059")]; tensor var_2064 = const()[name = string("op_2064"), val = tensor([0, 1, 3, 2])]; string var_2076_pad_type_0 = const()[name = string("op_2076_pad_type_0"), val = string("valid")]; tensor var_2076_strides_0 = const()[name = string("op_2076_strides_0"), val = tensor([1, 1])]; tensor var_2076_pad_0 = const()[name = string("op_2076_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2076_dilations_0 = const()[name = string("op_2076_dilations_0"), val = tensor([1, 1])]; int32 var_2076_groups_0 = const()[name = string("op_2076_groups_0"), val = int32(1)]; tensor var_2076 = conv(dilations = var_2076_dilations_0, groups = var_2076_groups_0, pad = var_2076_pad_0, pad_type = var_2076_pad_type_0, strides = var_2076_strides_0, weight = model_model_layers_1_self_attn_v_proj_weight_palettized, x = var_2014_cast_fp16)[name = string("op_2076")]; tensor var_2081 = const()[name = string("op_2081"), val = tensor([1, 8, 128, 64])]; tensor var_2082 = reshape(shape = var_2081, x = var_2076)[name = string("op_2082")]; tensor var_2087 = const()[name = string("op_2087"), val = tensor([0, 1, 3, 2])]; int32 var_2100 = const()[name = string("op_2100"), val = int32(-1)]; fp16 const_18_promoted = const()[name = string("const_18_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_15 = transpose(perm = var_2041, x = var_2036)[name = string("transpose_241")]; tensor var_2102 = mul(x = hidden_states_15, y = const_18_promoted)[name = string("op_2102")]; bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; tensor input_23 = concat(axis = var_2100, interleave = input_23_interleave_0, values = (hidden_states_15, var_2102))[name = string("input_23")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_2097_to_fp16 = const()[name = string("op_2097_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_2097_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_23 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23")]; tensor const_20 = const()[name = string("const_20"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294498432)))]; tensor q_3 = mul(x = normed_23, y = const_20)[name = string("q_3")]; int32 var_2122 = const()[name = string("op_2122"), val = int32(-1)]; fp16 const_21_promoted = const()[name = string("const_21_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_17 = transpose(perm = var_2064, x = var_2059)[name = string("transpose_240")]; tensor var_2124 = mul(x = hidden_states_17, y = const_21_promoted)[name = string("op_2124")]; bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; tensor input_25 = concat(axis = var_2122, interleave = input_25_interleave_0, values = (hidden_states_17, var_2124))[name = string("input_25")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_2119_to_fp16 = const()[name = string("op_2119_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_2119_to_fp16, x = input_25)[name = string("normed_25_cast_fp16")]; tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_27 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27")]; tensor const_23 = const()[name = string("const_23"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294498752)))]; tensor k_3 = mul(x = normed_27, y = const_23)[name = string("k_3")]; tensor var_2145 = mul(x = q_3, y = cos_1)[name = string("op_2145")]; tensor var_2150_begin_0 = const()[name = string("op_2150_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2150_end_0 = const()[name = string("op_2150_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_2150_end_mask_0 = const()[name = string("op_2150_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2150 = slice_by_index(begin = var_2150_begin_0, end = var_2150_end_0, end_mask = var_2150_end_mask_0, x = q_3)[name = string("op_2150")]; fp16 const_24_promoted = const()[name = string("const_24_promoted"), val = fp16(-0x1p+0)]; tensor var_2151 = mul(x = var_2150, y = const_24_promoted)[name = string("op_2151")]; tensor var_2156_begin_0 = const()[name = string("op_2156_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2156_end_0 = const()[name = string("op_2156_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_2156_end_mask_0 = const()[name = string("op_2156_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2156 = slice_by_index(begin = var_2156_begin_0, end = var_2156_end_0, end_mask = var_2156_end_mask_0, x = q_3)[name = string("op_2156")]; int32 var_2158 = const()[name = string("op_2158"), val = int32(-1)]; bool var_2159_interleave_0 = const()[name = string("op_2159_interleave_0"), val = bool(false)]; tensor var_2159 = concat(axis = var_2158, interleave = var_2159_interleave_0, values = (var_2151, var_2156))[name = string("op_2159")]; tensor var_2160 = mul(x = var_2159, y = sin_1)[name = string("op_2160")]; tensor query_3 = add(x = var_2145, y = var_2160)[name = string("query_3")]; tensor var_2163 = mul(x = k_3, y = cos_1)[name = string("op_2163")]; tensor var_2168_begin_0 = const()[name = string("op_2168_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2168_end_0 = const()[name = string("op_2168_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_2168_end_mask_0 = const()[name = string("op_2168_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2168 = slice_by_index(begin = var_2168_begin_0, end = var_2168_end_0, end_mask = var_2168_end_mask_0, x = k_3)[name = string("op_2168")]; fp16 const_25_promoted = const()[name = string("const_25_promoted"), val = fp16(-0x1p+0)]; tensor var_2169 = mul(x = var_2168, y = const_25_promoted)[name = string("op_2169")]; tensor var_2174_begin_0 = const()[name = string("op_2174_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2174_end_0 = const()[name = string("op_2174_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_2174_end_mask_0 = const()[name = string("op_2174_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2174 = slice_by_index(begin = var_2174_begin_0, end = var_2174_end_0, end_mask = var_2174_end_mask_0, x = k_3)[name = string("op_2174")]; int32 var_2176 = const()[name = string("op_2176"), val = int32(-1)]; bool var_2177_interleave_0 = const()[name = string("op_2177_interleave_0"), val = bool(false)]; tensor var_2177 = concat(axis = var_2176, interleave = var_2177_interleave_0, values = (var_2169, var_2174))[name = string("op_2177")]; tensor var_2178 = mul(x = var_2177, y = sin_1)[name = string("op_2178")]; tensor key_3 = add(x = var_2163, y = var_2178)[name = string("key_3")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; int32 concat_20_axis_0 = const()[name = string("concat_20_axis_0"), val = int32(0)]; bool concat_20_interleave_0 = const()[name = string("concat_20_interleave_0"), val = bool(false)]; tensor concat_20 = concat(axis = concat_20_axis_0, interleave = concat_20_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_20")]; tensor concat_21_values1_0 = const()[name = string("concat_21_values1_0"), val = tensor([0])]; tensor concat_21_values3_0 = const()[name = string("concat_21_values3_0"), val = tensor([0])]; int32 concat_21_axis_0 = const()[name = string("concat_21_axis_0"), val = int32(0)]; bool concat_21_interleave_0 = const()[name = string("concat_21_interleave_0"), val = bool(false)]; tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (expand_dims_16, concat_21_values1_0, var_1746, concat_21_values3_0))[name = string("concat_21")]; tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_20, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_21, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_3, x = coreml_update_state_57)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_170_write_state")]; tensor coreml_update_state_58 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_170")]; tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([29])]; tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([30])]; int32 concat_24_axis_0 = const()[name = string("concat_24_axis_0"), val = int32(0)]; bool concat_24_interleave_0 = const()[name = string("concat_24_interleave_0"), val = bool(false)]; tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_24")]; tensor concat_25_values1_0 = const()[name = string("concat_25_values1_0"), val = tensor([0])]; tensor concat_25_values3_0 = const()[name = string("concat_25_values3_0"), val = tensor([0])]; int32 concat_25_axis_0 = const()[name = string("concat_25_axis_0"), val = int32(0)]; bool concat_25_interleave_0 = const()[name = string("concat_25_interleave_0"), val = bool(false)]; tensor concat_25 = concat(axis = concat_25_axis_0, interleave = concat_25_interleave_0, values = (expand_dims_22, concat_25_values1_0, var_1746, concat_25_values3_0))[name = string("concat_25")]; tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_11 = transpose(perm = var_2087, x = var_2082)[name = string("transpose_239")]; tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_24, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_25, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = value_11, x = coreml_update_state_58)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_171_write_state")]; tensor coreml_update_state_59 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_171")]; tensor var_2249_begin_0 = const()[name = string("op_2249_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_2249_end_0 = const()[name = string("op_2249_end_0"), val = tensor([2, 8, 1536, 128])]; tensor var_2249_end_mask_0 = const()[name = string("op_2249_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2249_cast_fp16 = slice_by_index(begin = var_2249_begin_0, end = var_2249_end_0, end_mask = var_2249_end_mask_0, x = coreml_update_state_59)[name = string("op_2249_cast_fp16")]; tensor key_cache_3_axes_0 = const()[name = string("key_cache_3_axes_0"), val = tensor([0])]; tensor key_cache_3_cast_fp16 = squeeze(axes = key_cache_3_axes_0, x = var_2249_cast_fp16)[name = string("key_cache_3_cast_fp16")]; tensor var_2256_begin_0 = const()[name = string("op_2256_begin_0"), val = tensor([29, 0, 0, 0])]; tensor var_2256_end_0 = const()[name = string("op_2256_end_0"), val = tensor([30, 8, 1536, 128])]; tensor var_2256_end_mask_0 = const()[name = string("op_2256_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2256_cast_fp16 = slice_by_index(begin = var_2256_begin_0, end = var_2256_end_0, end_mask = var_2256_end_mask_0, x = coreml_update_state_59)[name = string("op_2256_cast_fp16")]; tensor value_cache_3_axes_0 = const()[name = string("value_cache_3_axes_0"), val = tensor([0])]; tensor value_cache_3_cast_fp16 = squeeze(axes = value_cache_3_axes_0, x = var_2256_cast_fp16)[name = string("value_cache_3_cast_fp16")]; tensor var_2280_axes_0 = const()[name = string("op_2280_axes_0"), val = tensor([1])]; tensor var_2280_cast_fp16 = expand_dims(axes = var_2280_axes_0, x = key_cache_3_cast_fp16)[name = string("op_2280_cast_fp16")]; tensor var_2285 = const()[name = string("op_2285"), val = tensor([1, 2, 1, 1])]; tensor value_15_cast_fp16 = tile(reps = var_2285, x = var_2280_cast_fp16)[name = string("value_15_cast_fp16")]; tensor var_2291 = const()[name = string("op_2291"), val = tensor([1, 16, 1536, 128])]; tensor key_states_7_cast_fp16 = reshape(shape = var_2291, x = value_15_cast_fp16)[name = string("key_states_7_cast_fp16")]; tensor var_2294_axes_0 = const()[name = string("op_2294_axes_0"), val = tensor([1])]; tensor var_2294_cast_fp16 = expand_dims(axes = var_2294_axes_0, x = value_cache_3_cast_fp16)[name = string("op_2294_cast_fp16")]; tensor var_2299 = const()[name = string("op_2299"), val = tensor([1, 2, 1, 1])]; tensor value_19_cast_fp16 = tile(reps = var_2299, x = var_2294_cast_fp16)[name = string("value_19_cast_fp16")]; bool var_2320_transpose_x_0 = const()[name = string("op_2320_transpose_x_0"), val = bool(false)]; bool var_2320_transpose_y_0 = const()[name = string("op_2320_transpose_y_0"), val = bool(true)]; tensor var_2320 = matmul(transpose_x = var_2320_transpose_x_0, transpose_y = var_2320_transpose_y_0, x = query_3, y = key_states_7_cast_fp16)[name = string("op_2320")]; fp16 var_2321_to_fp16 = const()[name = string("op_2321_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_5_cast_fp16 = mul(x = var_2320, y = var_2321_to_fp16)[name = string("attention_5_cast_fp16")]; tensor attention_7_cast_fp16 = add(x = attention_5_cast_fp16, y = causal_mask)[name = string("attention_7_cast_fp16")]; int32 var_2330 = const()[name = string("op_2330"), val = int32(-1)]; tensor var_2332_cast_fp16 = softmax(axis = var_2330, x = attention_7_cast_fp16)[name = string("op_2332_cast_fp16")]; tensor concat_30 = const()[name = string("concat_30"), val = tensor([16, 64, 1536])]; tensor reshape_3_cast_fp16 = reshape(shape = concat_30, x = var_2332_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor concat_31 = const()[name = string("concat_31"), val = tensor([16, 1536, 128])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_31, x = value_19_cast_fp16)[name = string("reshape_4_cast_fp16")]; bool matmul_1_transpose_x_0 = const()[name = string("matmul_1_transpose_x_0"), val = bool(false)]; bool matmul_1_transpose_y_0 = const()[name = string("matmul_1_transpose_y_0"), val = bool(false)]; tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = reshape_3_cast_fp16, y = reshape_4_cast_fp16)[name = string("matmul_1_cast_fp16")]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([1, 16, 64, 128])]; tensor reshape_5_cast_fp16 = reshape(shape = concat_35, x = matmul_1_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor var_2344_perm_0 = const()[name = string("op_2344_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2350 = const()[name = string("op_2350"), val = tensor([1, 64, 2048])]; tensor var_2344_cast_fp16 = transpose(perm = var_2344_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_238")]; tensor output_9_cast_fp16 = reshape(shape = var_2350, x = var_2344_cast_fp16)[name = string("output_9_cast_fp16")]; tensor var_2355 = const()[name = string("op_2355"), val = tensor([0, 2, 1])]; string var_2371_pad_type_0 = const()[name = string("op_2371_pad_type_0"), val = string("valid")]; int32 var_2371_groups_0 = const()[name = string("op_2371_groups_0"), val = int32(1)]; tensor var_2371_strides_0 = const()[name = string("op_2371_strides_0"), val = tensor([1])]; tensor var_2371_pad_0 = const()[name = string("op_2371_pad_0"), val = tensor([0, 0])]; tensor var_2371_dilations_0 = const()[name = string("op_2371_dilations_0"), val = tensor([1])]; tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294499072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296072000))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2356_cast_fp16 = transpose(perm = var_2355, x = output_9_cast_fp16)[name = string("transpose_237")]; tensor var_2371_cast_fp16 = conv(dilations = var_2371_dilations_0, groups = var_2371_groups_0, pad = var_2371_pad_0, pad_type = var_2371_pad_type_0, strides = var_2371_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_2356_cast_fp16)[name = string("op_2371_cast_fp16")]; tensor var_2375 = const()[name = string("op_2375"), val = tensor([0, 2, 1])]; tensor attn_output_3_cast_fp16 = transpose(perm = var_2375, x = var_2371_cast_fp16)[name = string("transpose_236")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = attn_output_3_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; int32 var_2390 = const()[name = string("op_2390"), val = int32(-1)]; fp16 const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2392_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_27_promoted_to_fp16)[name = string("op_2392_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_2390, interleave = input_29_interleave_0, values = (hidden_states_19_cast_fp16, var_2392_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_2387_to_fp16 = const()[name = string("op_2387_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_2387_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 64, 1024])]; tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; tensor const_29_promoted_to_fp16 = const()[name = string("const_29_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296088448)))]; tensor x_5_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_29_promoted_to_fp16)[name = string("x_5_cast_fp16")]; tensor var_2412 = const()[name = string("op_2412"), val = tensor([0, 2, 1])]; tensor input_31_axes_0 = const()[name = string("input_31_axes_0"), val = tensor([2])]; tensor var_2413 = transpose(perm = var_2412, x = x_5_cast_fp16)[name = string("transpose_235")]; tensor input_31 = expand_dims(axes = input_31_axes_0, x = var_2413)[name = string("input_31")]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = model_model_layers_1_mlp_gate_proj_weight_palettized, x = input_31)[name = string("input_33")]; string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_1_mlp_up_proj_weight_palettized, x = input_31)[name = string("b_3")]; tensor c_3 = silu(x = input_33)[name = string("c_3")]; tensor input_35 = mul(x = c_3, y = b_3)[name = string("input_35")]; string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_1_mlp_down_proj_weight_palettized, x = input_35)[name = string("e_3")]; tensor var_2435_axes_0 = const()[name = string("op_2435_axes_0"), val = tensor([2])]; tensor var_2435 = squeeze(axes = var_2435_axes_0, x = e_3)[name = string("op_2435")]; tensor var_2436 = const()[name = string("op_2436"), val = tensor([0, 2, 1])]; tensor var_2437 = transpose(perm = var_2436, x = var_2435)[name = string("transpose_234")]; tensor hidden_states_21_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = var_2437)[name = string("hidden_states_21_cast_fp16")]; int32 var_2451 = const()[name = string("op_2451"), val = int32(-1)]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2453_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_2453_cast_fp16")]; bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; tensor input_37_cast_fp16 = concat(axis = var_2451, interleave = input_37_interleave_0, values = (hidden_states_21_cast_fp16, var_2453_cast_fp16))[name = string("input_37_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_2448_to_fp16 = const()[name = string("op_2448_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_2448_to_fp16, x = input_37_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 64, 1024])]; tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; tensor const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296090560)))]; tensor hidden_states_23_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_32_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor var_2465 = const()[name = string("op_2465"), val = tensor([0, 2, 1])]; tensor var_2468_axes_0 = const()[name = string("op_2468_axes_0"), val = tensor([2])]; tensor var_2466_cast_fp16 = transpose(perm = var_2465, x = hidden_states_23_cast_fp16)[name = string("transpose_233")]; tensor var_2468_cast_fp16 = expand_dims(axes = var_2468_axes_0, x = var_2466_cast_fp16)[name = string("op_2468_cast_fp16")]; string var_2484_pad_type_0 = const()[name = string("op_2484_pad_type_0"), val = string("valid")]; tensor var_2484_strides_0 = const()[name = string("op_2484_strides_0"), val = tensor([1, 1])]; tensor var_2484_pad_0 = const()[name = string("op_2484_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2484_dilations_0 = const()[name = string("op_2484_dilations_0"), val = tensor([1, 1])]; int32 var_2484_groups_0 = const()[name = string("op_2484_groups_0"), val = int32(1)]; tensor var_2484 = conv(dilations = var_2484_dilations_0, groups = var_2484_groups_0, pad = var_2484_pad_0, pad_type = var_2484_pad_type_0, strides = var_2484_strides_0, weight = model_model_layers_2_self_attn_q_proj_weight_palettized, x = var_2468_cast_fp16)[name = string("op_2484")]; tensor var_2489 = const()[name = string("op_2489"), val = tensor([1, 16, 128, 64])]; tensor var_2490 = reshape(shape = var_2489, x = var_2484)[name = string("op_2490")]; tensor var_2495 = const()[name = string("op_2495"), val = tensor([0, 1, 3, 2])]; string var_2507_pad_type_0 = const()[name = string("op_2507_pad_type_0"), val = string("valid")]; tensor var_2507_strides_0 = const()[name = string("op_2507_strides_0"), val = tensor([1, 1])]; tensor var_2507_pad_0 = const()[name = string("op_2507_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2507_dilations_0 = const()[name = string("op_2507_dilations_0"), val = tensor([1, 1])]; int32 var_2507_groups_0 = const()[name = string("op_2507_groups_0"), val = int32(1)]; tensor var_2507 = conv(dilations = var_2507_dilations_0, groups = var_2507_groups_0, pad = var_2507_pad_0, pad_type = var_2507_pad_type_0, strides = var_2507_strides_0, weight = model_model_layers_2_self_attn_k_proj_weight_palettized, x = var_2468_cast_fp16)[name = string("op_2507")]; tensor var_2512 = const()[name = string("op_2512"), val = tensor([1, 8, 128, 64])]; tensor var_2513 = reshape(shape = var_2512, x = var_2507)[name = string("op_2513")]; tensor var_2518 = const()[name = string("op_2518"), val = tensor([0, 1, 3, 2])]; string var_2530_pad_type_0 = const()[name = string("op_2530_pad_type_0"), val = string("valid")]; tensor var_2530_strides_0 = const()[name = string("op_2530_strides_0"), val = tensor([1, 1])]; tensor var_2530_pad_0 = const()[name = string("op_2530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2530_dilations_0 = const()[name = string("op_2530_dilations_0"), val = tensor([1, 1])]; int32 var_2530_groups_0 = const()[name = string("op_2530_groups_0"), val = int32(1)]; tensor var_2530 = conv(dilations = var_2530_dilations_0, groups = var_2530_groups_0, pad = var_2530_pad_0, pad_type = var_2530_pad_type_0, strides = var_2530_strides_0, weight = model_model_layers_2_self_attn_v_proj_weight_palettized, x = var_2468_cast_fp16)[name = string("op_2530")]; tensor var_2535 = const()[name = string("op_2535"), val = tensor([1, 8, 128, 64])]; tensor var_2536 = reshape(shape = var_2535, x = var_2530)[name = string("op_2536")]; tensor var_2541 = const()[name = string("op_2541"), val = tensor([0, 1, 3, 2])]; int32 var_2554 = const()[name = string("op_2554"), val = int32(-1)]; fp16 const_33_promoted = const()[name = string("const_33_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_25 = transpose(perm = var_2495, x = var_2490)[name = string("transpose_232")]; tensor var_2556 = mul(x = hidden_states_25, y = const_33_promoted)[name = string("op_2556")]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41 = concat(axis = var_2554, interleave = input_41_interleave_0, values = (hidden_states_25, var_2556))[name = string("input_41")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_2551_to_fp16 = const()[name = string("op_2551_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_2551_to_fp16, x = input_41)[name = string("normed_37_cast_fp16")]; tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_39 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39")]; tensor const_35 = const()[name = string("const_35"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296092672)))]; tensor q_5 = mul(x = normed_39, y = const_35)[name = string("q_5")]; int32 var_2576 = const()[name = string("op_2576"), val = int32(-1)]; fp16 const_36_promoted = const()[name = string("const_36_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_27 = transpose(perm = var_2518, x = var_2513)[name = string("transpose_231")]; tensor var_2578 = mul(x = hidden_states_27, y = const_36_promoted)[name = string("op_2578")]; bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; tensor input_43 = concat(axis = var_2576, interleave = input_43_interleave_0, values = (hidden_states_27, var_2578))[name = string("input_43")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_2573_to_fp16 = const()[name = string("op_2573_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2573_to_fp16, x = input_43)[name = string("normed_41_cast_fp16")]; tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_43 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43")]; tensor const_38 = const()[name = string("const_38"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296092992)))]; tensor k_5 = mul(x = normed_43, y = const_38)[name = string("k_5")]; tensor var_2599 = mul(x = q_5, y = cos_1)[name = string("op_2599")]; tensor var_2604_begin_0 = const()[name = string("op_2604_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2604_end_0 = const()[name = string("op_2604_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_2604_end_mask_0 = const()[name = string("op_2604_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2604 = slice_by_index(begin = var_2604_begin_0, end = var_2604_end_0, end_mask = var_2604_end_mask_0, x = q_5)[name = string("op_2604")]; fp16 const_39_promoted = const()[name = string("const_39_promoted"), val = fp16(-0x1p+0)]; tensor var_2605 = mul(x = var_2604, y = const_39_promoted)[name = string("op_2605")]; tensor var_2610_begin_0 = const()[name = string("op_2610_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2610_end_0 = const()[name = string("op_2610_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_2610_end_mask_0 = const()[name = string("op_2610_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2610 = slice_by_index(begin = var_2610_begin_0, end = var_2610_end_0, end_mask = var_2610_end_mask_0, x = q_5)[name = string("op_2610")]; int32 var_2612 = const()[name = string("op_2612"), val = int32(-1)]; bool var_2613_interleave_0 = const()[name = string("op_2613_interleave_0"), val = bool(false)]; tensor var_2613 = concat(axis = var_2612, interleave = var_2613_interleave_0, values = (var_2605, var_2610))[name = string("op_2613")]; tensor var_2614 = mul(x = var_2613, y = sin_1)[name = string("op_2614")]; tensor query_5 = add(x = var_2599, y = var_2614)[name = string("query_5")]; tensor var_2617 = mul(x = k_5, y = cos_1)[name = string("op_2617")]; tensor var_2622_begin_0 = const()[name = string("op_2622_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_2622_end_0 = const()[name = string("op_2622_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_2622_end_mask_0 = const()[name = string("op_2622_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2622 = slice_by_index(begin = var_2622_begin_0, end = var_2622_end_0, end_mask = var_2622_end_mask_0, x = k_5)[name = string("op_2622")]; fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; tensor var_2623 = mul(x = var_2622, y = const_40_promoted)[name = string("op_2623")]; tensor var_2628_begin_0 = const()[name = string("op_2628_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2628_end_0 = const()[name = string("op_2628_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_2628_end_mask_0 = const()[name = string("op_2628_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_2628 = slice_by_index(begin = var_2628_begin_0, end = var_2628_end_0, end_mask = var_2628_end_mask_0, x = k_5)[name = string("op_2628")]; int32 var_2630 = const()[name = string("op_2630"), val = int32(-1)]; bool var_2631_interleave_0 = const()[name = string("op_2631_interleave_0"), val = bool(false)]; tensor var_2631 = concat(axis = var_2630, interleave = var_2631_interleave_0, values = (var_2623, var_2628))[name = string("op_2631")]; tensor var_2632 = mul(x = var_2631, y = sin_1)[name = string("op_2632")]; tensor key_5 = add(x = var_2617, y = var_2632)[name = string("key_5")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_38")]; tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_28, concat_39_values1_0, var_1746, concat_39_values3_0))[name = string("concat_39")]; tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_5, x = coreml_update_state_59)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_172_write_state")]; tensor coreml_update_state_60 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_172")]; tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([30])]; tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([31])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_42")]; tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_34, concat_43_values1_0, var_1746, concat_43_values3_0))[name = string("concat_43")]; tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_21 = transpose(perm = var_2541, x = var_2536)[name = string("transpose_230")]; tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = value_21, x = coreml_update_state_60)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_173_write_state")]; tensor coreml_update_state_61 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_173")]; tensor var_2703_begin_0 = const()[name = string("op_2703_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_2703_end_0 = const()[name = string("op_2703_end_0"), val = tensor([3, 8, 1536, 128])]; tensor var_2703_end_mask_0 = const()[name = string("op_2703_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2703_cast_fp16 = slice_by_index(begin = var_2703_begin_0, end = var_2703_end_0, end_mask = var_2703_end_mask_0, x = coreml_update_state_61)[name = string("op_2703_cast_fp16")]; tensor key_cache_5_axes_0 = const()[name = string("key_cache_5_axes_0"), val = tensor([0])]; tensor key_cache_5_cast_fp16 = squeeze(axes = key_cache_5_axes_0, x = var_2703_cast_fp16)[name = string("key_cache_5_cast_fp16")]; tensor var_2710_begin_0 = const()[name = string("op_2710_begin_0"), val = tensor([30, 0, 0, 0])]; tensor var_2710_end_0 = const()[name = string("op_2710_end_0"), val = tensor([31, 8, 1536, 128])]; tensor var_2710_end_mask_0 = const()[name = string("op_2710_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2710_cast_fp16 = slice_by_index(begin = var_2710_begin_0, end = var_2710_end_0, end_mask = var_2710_end_mask_0, x = coreml_update_state_61)[name = string("op_2710_cast_fp16")]; tensor value_cache_5_axes_0 = const()[name = string("value_cache_5_axes_0"), val = tensor([0])]; tensor value_cache_5_cast_fp16 = squeeze(axes = value_cache_5_axes_0, x = var_2710_cast_fp16)[name = string("value_cache_5_cast_fp16")]; tensor var_2734_axes_0 = const()[name = string("op_2734_axes_0"), val = tensor([1])]; tensor var_2734_cast_fp16 = expand_dims(axes = var_2734_axes_0, x = key_cache_5_cast_fp16)[name = string("op_2734_cast_fp16")]; tensor var_2739 = const()[name = string("op_2739"), val = tensor([1, 2, 1, 1])]; tensor value_25_cast_fp16 = tile(reps = var_2739, x = var_2734_cast_fp16)[name = string("value_25_cast_fp16")]; tensor var_2745 = const()[name = string("op_2745"), val = tensor([1, 16, 1536, 128])]; tensor key_states_11_cast_fp16 = reshape(shape = var_2745, x = value_25_cast_fp16)[name = string("key_states_11_cast_fp16")]; tensor var_2748_axes_0 = const()[name = string("op_2748_axes_0"), val = tensor([1])]; tensor var_2748_cast_fp16 = expand_dims(axes = var_2748_axes_0, x = value_cache_5_cast_fp16)[name = string("op_2748_cast_fp16")]; tensor var_2753 = const()[name = string("op_2753"), val = tensor([1, 2, 1, 1])]; tensor value_29_cast_fp16 = tile(reps = var_2753, x = var_2748_cast_fp16)[name = string("value_29_cast_fp16")]; bool var_2774_transpose_x_0 = const()[name = string("op_2774_transpose_x_0"), val = bool(false)]; bool var_2774_transpose_y_0 = const()[name = string("op_2774_transpose_y_0"), val = bool(true)]; tensor var_2774 = matmul(transpose_x = var_2774_transpose_x_0, transpose_y = var_2774_transpose_y_0, x = query_5, y = key_states_11_cast_fp16)[name = string("op_2774")]; fp16 var_2775_to_fp16 = const()[name = string("op_2775_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_9_cast_fp16 = mul(x = var_2774, y = var_2775_to_fp16)[name = string("attention_9_cast_fp16")]; tensor attention_11_cast_fp16 = add(x = attention_9_cast_fp16, y = causal_mask)[name = string("attention_11_cast_fp16")]; int32 var_2784 = const()[name = string("op_2784"), val = int32(-1)]; tensor var_2786_cast_fp16 = softmax(axis = var_2784, x = attention_11_cast_fp16)[name = string("op_2786_cast_fp16")]; tensor concat_48 = const()[name = string("concat_48"), val = tensor([16, 64, 1536])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_48, x = var_2786_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor concat_49 = const()[name = string("concat_49"), val = tensor([16, 1536, 128])]; tensor reshape_7_cast_fp16 = reshape(shape = concat_49, x = value_29_cast_fp16)[name = string("reshape_7_cast_fp16")]; bool matmul_2_transpose_x_0 = const()[name = string("matmul_2_transpose_x_0"), val = bool(false)]; bool matmul_2_transpose_y_0 = const()[name = string("matmul_2_transpose_y_0"), val = bool(false)]; tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = reshape_6_cast_fp16, y = reshape_7_cast_fp16)[name = string("matmul_2_cast_fp16")]; tensor concat_53 = const()[name = string("concat_53"), val = tensor([1, 16, 64, 128])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_53, x = matmul_2_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor var_2798_perm_0 = const()[name = string("op_2798_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2804 = const()[name = string("op_2804"), val = tensor([1, 64, 2048])]; tensor var_2798_cast_fp16 = transpose(perm = var_2798_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_229")]; tensor output_15_cast_fp16 = reshape(shape = var_2804, x = var_2798_cast_fp16)[name = string("output_15_cast_fp16")]; tensor var_2809 = const()[name = string("op_2809"), val = tensor([0, 2, 1])]; string var_2825_pad_type_0 = const()[name = string("op_2825_pad_type_0"), val = string("valid")]; int32 var_2825_groups_0 = const()[name = string("op_2825_groups_0"), val = int32(1)]; tensor var_2825_strides_0 = const()[name = string("op_2825_strides_0"), val = tensor([1])]; tensor var_2825_pad_0 = const()[name = string("op_2825_pad_0"), val = tensor([0, 0])]; tensor var_2825_dilations_0 = const()[name = string("op_2825_dilations_0"), val = tensor([1])]; tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(296093312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297666240))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_2810_cast_fp16 = transpose(perm = var_2809, x = output_15_cast_fp16)[name = string("transpose_228")]; tensor var_2825_cast_fp16 = conv(dilations = var_2825_dilations_0, groups = var_2825_groups_0, pad = var_2825_pad_0, pad_type = var_2825_pad_type_0, strides = var_2825_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2810_cast_fp16)[name = string("op_2825_cast_fp16")]; tensor var_2829 = const()[name = string("op_2829"), val = tensor([0, 2, 1])]; tensor attn_output_5_cast_fp16 = transpose(perm = var_2829, x = var_2825_cast_fp16)[name = string("transpose_227")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = attn_output_5_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; int32 var_2844 = const()[name = string("op_2844"), val = int32(-1)]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2846_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_42_promoted_to_fp16)[name = string("op_2846_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_2844, interleave = input_47_interleave_0, values = (hidden_states_29_cast_fp16, var_2846_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_2841_to_fp16 = const()[name = string("op_2841_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2841_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 64, 1024])]; tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; tensor const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297682688)))]; tensor x_9_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_44_promoted_to_fp16)[name = string("x_9_cast_fp16")]; tensor var_2866 = const()[name = string("op_2866"), val = tensor([0, 2, 1])]; tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; tensor var_2867 = transpose(perm = var_2866, x = x_9_cast_fp16)[name = string("transpose_226")]; tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2867)[name = string("input_49")]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = model_model_layers_2_mlp_gate_proj_weight_palettized, x = input_49)[name = string("input_51")]; string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_2_mlp_up_proj_weight_palettized, x = input_49)[name = string("b_5")]; tensor c_5 = silu(x = input_51)[name = string("c_5")]; tensor input_53 = mul(x = c_5, y = b_5)[name = string("input_53")]; string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_2_mlp_down_proj_weight_palettized, x = input_53)[name = string("e_5")]; tensor var_2889_axes_0 = const()[name = string("op_2889_axes_0"), val = tensor([2])]; tensor var_2889 = squeeze(axes = var_2889_axes_0, x = e_5)[name = string("op_2889")]; tensor var_2890 = const()[name = string("op_2890"), val = tensor([0, 2, 1])]; tensor var_2891 = transpose(perm = var_2890, x = var_2889)[name = string("transpose_225")]; tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2891)[name = string("hidden_states_31_cast_fp16")]; int32 var_2905 = const()[name = string("op_2905"), val = int32(-1)]; fp16 const_45_promoted_to_fp16 = const()[name = string("const_45_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2907_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_45_promoted_to_fp16)[name = string("op_2907_cast_fp16")]; bool input_55_interleave_0 = const()[name = string("input_55_interleave_0"), val = bool(false)]; tensor input_55_cast_fp16 = concat(axis = var_2905, interleave = input_55_interleave_0, values = (hidden_states_31_cast_fp16, var_2907_cast_fp16))[name = string("input_55_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_2902_to_fp16 = const()[name = string("op_2902_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2902_to_fp16, x = input_55_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 64, 1024])]; tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; tensor const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297684800)))]; tensor hidden_states_33_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_47_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor var_2919 = const()[name = string("op_2919"), val = tensor([0, 2, 1])]; tensor var_2922_axes_0 = const()[name = string("op_2922_axes_0"), val = tensor([2])]; tensor var_2920_cast_fp16 = transpose(perm = var_2919, x = hidden_states_33_cast_fp16)[name = string("transpose_224")]; tensor var_2922_cast_fp16 = expand_dims(axes = var_2922_axes_0, x = var_2920_cast_fp16)[name = string("op_2922_cast_fp16")]; string var_2938_pad_type_0 = const()[name = string("op_2938_pad_type_0"), val = string("valid")]; tensor var_2938_strides_0 = const()[name = string("op_2938_strides_0"), val = tensor([1, 1])]; tensor var_2938_pad_0 = const()[name = string("op_2938_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2938_dilations_0 = const()[name = string("op_2938_dilations_0"), val = tensor([1, 1])]; int32 var_2938_groups_0 = const()[name = string("op_2938_groups_0"), val = int32(1)]; tensor var_2938 = conv(dilations = var_2938_dilations_0, groups = var_2938_groups_0, pad = var_2938_pad_0, pad_type = var_2938_pad_type_0, strides = var_2938_strides_0, weight = model_model_layers_3_self_attn_q_proj_weight_palettized, x = var_2922_cast_fp16)[name = string("op_2938")]; tensor var_2943 = const()[name = string("op_2943"), val = tensor([1, 16, 128, 64])]; tensor var_2944 = reshape(shape = var_2943, x = var_2938)[name = string("op_2944")]; tensor var_2949 = const()[name = string("op_2949"), val = tensor([0, 1, 3, 2])]; string var_2961_pad_type_0 = const()[name = string("op_2961_pad_type_0"), val = string("valid")]; tensor var_2961_strides_0 = const()[name = string("op_2961_strides_0"), val = tensor([1, 1])]; tensor var_2961_pad_0 = const()[name = string("op_2961_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2961_dilations_0 = const()[name = string("op_2961_dilations_0"), val = tensor([1, 1])]; int32 var_2961_groups_0 = const()[name = string("op_2961_groups_0"), val = int32(1)]; tensor var_2961 = conv(dilations = var_2961_dilations_0, groups = var_2961_groups_0, pad = var_2961_pad_0, pad_type = var_2961_pad_type_0, strides = var_2961_strides_0, weight = model_model_layers_3_self_attn_k_proj_weight_palettized, x = var_2922_cast_fp16)[name = string("op_2961")]; tensor var_2966 = const()[name = string("op_2966"), val = tensor([1, 8, 128, 64])]; tensor var_2967 = reshape(shape = var_2966, x = var_2961)[name = string("op_2967")]; tensor var_2972 = const()[name = string("op_2972"), val = tensor([0, 1, 3, 2])]; string var_2984_pad_type_0 = const()[name = string("op_2984_pad_type_0"), val = string("valid")]; tensor var_2984_strides_0 = const()[name = string("op_2984_strides_0"), val = tensor([1, 1])]; tensor var_2984_pad_0 = const()[name = string("op_2984_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2984_dilations_0 = const()[name = string("op_2984_dilations_0"), val = tensor([1, 1])]; int32 var_2984_groups_0 = const()[name = string("op_2984_groups_0"), val = int32(1)]; tensor var_2984 = conv(dilations = var_2984_dilations_0, groups = var_2984_groups_0, pad = var_2984_pad_0, pad_type = var_2984_pad_type_0, strides = var_2984_strides_0, weight = model_model_layers_3_self_attn_v_proj_weight_palettized, x = var_2922_cast_fp16)[name = string("op_2984")]; tensor var_2989 = const()[name = string("op_2989"), val = tensor([1, 8, 128, 64])]; tensor var_2990 = reshape(shape = var_2989, x = var_2984)[name = string("op_2990")]; tensor var_2995 = const()[name = string("op_2995"), val = tensor([0, 1, 3, 2])]; int32 var_3008 = const()[name = string("op_3008"), val = int32(-1)]; fp16 const_48_promoted = const()[name = string("const_48_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_35 = transpose(perm = var_2949, x = var_2944)[name = string("transpose_223")]; tensor var_3010 = mul(x = hidden_states_35, y = const_48_promoted)[name = string("op_3010")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59 = concat(axis = var_3008, interleave = input_59_interleave_0, values = (hidden_states_35, var_3010))[name = string("input_59")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_3005_to_fp16 = const()[name = string("op_3005_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_3005_to_fp16, x = input_59)[name = string("normed_53_cast_fp16")]; tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_55 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55")]; tensor const_50 = const()[name = string("const_50"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297686912)))]; tensor q_7 = mul(x = normed_55, y = const_50)[name = string("q_7")]; int32 var_3030 = const()[name = string("op_3030"), val = int32(-1)]; fp16 const_51_promoted = const()[name = string("const_51_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_37 = transpose(perm = var_2972, x = var_2967)[name = string("transpose_222")]; tensor var_3032 = mul(x = hidden_states_37, y = const_51_promoted)[name = string("op_3032")]; bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; tensor input_61 = concat(axis = var_3030, interleave = input_61_interleave_0, values = (hidden_states_37, var_3032))[name = string("input_61")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_3027_to_fp16 = const()[name = string("op_3027_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_3027_to_fp16, x = input_61)[name = string("normed_57_cast_fp16")]; tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_59 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59")]; tensor const_53 = const()[name = string("const_53"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297687232)))]; tensor k_7 = mul(x = normed_59, y = const_53)[name = string("k_7")]; tensor var_3053 = mul(x = q_7, y = cos_1)[name = string("op_3053")]; tensor var_3058_begin_0 = const()[name = string("op_3058_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3058_end_0 = const()[name = string("op_3058_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_3058_end_mask_0 = const()[name = string("op_3058_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3058 = slice_by_index(begin = var_3058_begin_0, end = var_3058_end_0, end_mask = var_3058_end_mask_0, x = q_7)[name = string("op_3058")]; fp16 const_54_promoted = const()[name = string("const_54_promoted"), val = fp16(-0x1p+0)]; tensor var_3059 = mul(x = var_3058, y = const_54_promoted)[name = string("op_3059")]; tensor var_3064_begin_0 = const()[name = string("op_3064_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3064_end_0 = const()[name = string("op_3064_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_3064_end_mask_0 = const()[name = string("op_3064_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3064 = slice_by_index(begin = var_3064_begin_0, end = var_3064_end_0, end_mask = var_3064_end_mask_0, x = q_7)[name = string("op_3064")]; int32 var_3066 = const()[name = string("op_3066"), val = int32(-1)]; bool var_3067_interleave_0 = const()[name = string("op_3067_interleave_0"), val = bool(false)]; tensor var_3067 = concat(axis = var_3066, interleave = var_3067_interleave_0, values = (var_3059, var_3064))[name = string("op_3067")]; tensor var_3068 = mul(x = var_3067, y = sin_1)[name = string("op_3068")]; tensor query_7 = add(x = var_3053, y = var_3068)[name = string("query_7")]; tensor var_3071 = mul(x = k_7, y = cos_1)[name = string("op_3071")]; tensor var_3076_begin_0 = const()[name = string("op_3076_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3076_end_0 = const()[name = string("op_3076_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_3076_end_mask_0 = const()[name = string("op_3076_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3076 = slice_by_index(begin = var_3076_begin_0, end = var_3076_end_0, end_mask = var_3076_end_mask_0, x = k_7)[name = string("op_3076")]; fp16 const_55_promoted = const()[name = string("const_55_promoted"), val = fp16(-0x1p+0)]; tensor var_3077 = mul(x = var_3076, y = const_55_promoted)[name = string("op_3077")]; tensor var_3082_begin_0 = const()[name = string("op_3082_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3082_end_0 = const()[name = string("op_3082_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_3082_end_mask_0 = const()[name = string("op_3082_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3082 = slice_by_index(begin = var_3082_begin_0, end = var_3082_end_0, end_mask = var_3082_end_mask_0, x = k_7)[name = string("op_3082")]; int32 var_3084 = const()[name = string("op_3084"), val = int32(-1)]; bool var_3085_interleave_0 = const()[name = string("op_3085_interleave_0"), val = bool(false)]; tensor var_3085 = concat(axis = var_3084, interleave = var_3085_interleave_0, values = (var_3077, var_3082))[name = string("op_3085")]; tensor var_3086 = mul(x = var_3085, y = sin_1)[name = string("op_3086")]; tensor key_7 = add(x = var_3071, y = var_3086)[name = string("key_7")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; int32 concat_56_axis_0 = const()[name = string("concat_56_axis_0"), val = int32(0)]; bool concat_56_interleave_0 = const()[name = string("concat_56_interleave_0"), val = bool(false)]; tensor concat_56 = concat(axis = concat_56_axis_0, interleave = concat_56_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_56")]; tensor concat_57_values1_0 = const()[name = string("concat_57_values1_0"), val = tensor([0])]; tensor concat_57_values3_0 = const()[name = string("concat_57_values3_0"), val = tensor([0])]; int32 concat_57_axis_0 = const()[name = string("concat_57_axis_0"), val = int32(0)]; bool concat_57_interleave_0 = const()[name = string("concat_57_interleave_0"), val = bool(false)]; tensor concat_57 = concat(axis = concat_57_axis_0, interleave = concat_57_interleave_0, values = (expand_dims_40, concat_57_values1_0, var_1746, concat_57_values3_0))[name = string("concat_57")]; tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_56, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_57, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_7, x = coreml_update_state_61)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_174_write_state")]; tensor coreml_update_state_62 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_174")]; tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([31])]; tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([32])]; int32 concat_60_axis_0 = const()[name = string("concat_60_axis_0"), val = int32(0)]; bool concat_60_interleave_0 = const()[name = string("concat_60_interleave_0"), val = bool(false)]; tensor concat_60 = concat(axis = concat_60_axis_0, interleave = concat_60_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_60")]; tensor concat_61_values1_0 = const()[name = string("concat_61_values1_0"), val = tensor([0])]; tensor concat_61_values3_0 = const()[name = string("concat_61_values3_0"), val = tensor([0])]; int32 concat_61_axis_0 = const()[name = string("concat_61_axis_0"), val = int32(0)]; bool concat_61_interleave_0 = const()[name = string("concat_61_interleave_0"), val = bool(false)]; tensor concat_61 = concat(axis = concat_61_axis_0, interleave = concat_61_interleave_0, values = (expand_dims_46, concat_61_values1_0, var_1746, concat_61_values3_0))[name = string("concat_61")]; tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_31 = transpose(perm = var_2995, x = var_2990)[name = string("transpose_221")]; tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_60, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_61, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = value_31, x = coreml_update_state_62)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_175_write_state")]; tensor coreml_update_state_63 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_175")]; tensor var_3157_begin_0 = const()[name = string("op_3157_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_3157_end_0 = const()[name = string("op_3157_end_0"), val = tensor([4, 8, 1536, 128])]; tensor var_3157_end_mask_0 = const()[name = string("op_3157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3157_cast_fp16 = slice_by_index(begin = var_3157_begin_0, end = var_3157_end_0, end_mask = var_3157_end_mask_0, x = coreml_update_state_63)[name = string("op_3157_cast_fp16")]; tensor key_cache_7_axes_0 = const()[name = string("key_cache_7_axes_0"), val = tensor([0])]; tensor key_cache_7_cast_fp16 = squeeze(axes = key_cache_7_axes_0, x = var_3157_cast_fp16)[name = string("key_cache_7_cast_fp16")]; tensor var_3164_begin_0 = const()[name = string("op_3164_begin_0"), val = tensor([31, 0, 0, 0])]; tensor var_3164_end_0 = const()[name = string("op_3164_end_0"), val = tensor([32, 8, 1536, 128])]; tensor var_3164_end_mask_0 = const()[name = string("op_3164_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3164_cast_fp16 = slice_by_index(begin = var_3164_begin_0, end = var_3164_end_0, end_mask = var_3164_end_mask_0, x = coreml_update_state_63)[name = string("op_3164_cast_fp16")]; tensor value_cache_7_axes_0 = const()[name = string("value_cache_7_axes_0"), val = tensor([0])]; tensor value_cache_7_cast_fp16 = squeeze(axes = value_cache_7_axes_0, x = var_3164_cast_fp16)[name = string("value_cache_7_cast_fp16")]; tensor var_3188_axes_0 = const()[name = string("op_3188_axes_0"), val = tensor([1])]; tensor var_3188_cast_fp16 = expand_dims(axes = var_3188_axes_0, x = key_cache_7_cast_fp16)[name = string("op_3188_cast_fp16")]; tensor var_3193 = const()[name = string("op_3193"), val = tensor([1, 2, 1, 1])]; tensor value_35_cast_fp16 = tile(reps = var_3193, x = var_3188_cast_fp16)[name = string("value_35_cast_fp16")]; tensor var_3199 = const()[name = string("op_3199"), val = tensor([1, 16, 1536, 128])]; tensor key_states_15_cast_fp16 = reshape(shape = var_3199, x = value_35_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor var_3202_axes_0 = const()[name = string("op_3202_axes_0"), val = tensor([1])]; tensor var_3202_cast_fp16 = expand_dims(axes = var_3202_axes_0, x = value_cache_7_cast_fp16)[name = string("op_3202_cast_fp16")]; tensor var_3207 = const()[name = string("op_3207"), val = tensor([1, 2, 1, 1])]; tensor value_39_cast_fp16 = tile(reps = var_3207, x = var_3202_cast_fp16)[name = string("value_39_cast_fp16")]; bool var_3228_transpose_x_0 = const()[name = string("op_3228_transpose_x_0"), val = bool(false)]; bool var_3228_transpose_y_0 = const()[name = string("op_3228_transpose_y_0"), val = bool(true)]; tensor var_3228 = matmul(transpose_x = var_3228_transpose_x_0, transpose_y = var_3228_transpose_y_0, x = query_7, y = key_states_15_cast_fp16)[name = string("op_3228")]; fp16 var_3229_to_fp16 = const()[name = string("op_3229_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_13_cast_fp16 = mul(x = var_3228, y = var_3229_to_fp16)[name = string("attention_13_cast_fp16")]; tensor attention_15_cast_fp16 = add(x = attention_13_cast_fp16, y = causal_mask)[name = string("attention_15_cast_fp16")]; int32 var_3238 = const()[name = string("op_3238"), val = int32(-1)]; tensor var_3240_cast_fp16 = softmax(axis = var_3238, x = attention_15_cast_fp16)[name = string("op_3240_cast_fp16")]; tensor concat_66 = const()[name = string("concat_66"), val = tensor([16, 64, 1536])]; tensor reshape_9_cast_fp16 = reshape(shape = concat_66, x = var_3240_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor concat_67 = const()[name = string("concat_67"), val = tensor([16, 1536, 128])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_67, x = value_39_cast_fp16)[name = string("reshape_10_cast_fp16")]; bool matmul_3_transpose_x_0 = const()[name = string("matmul_3_transpose_x_0"), val = bool(false)]; bool matmul_3_transpose_y_0 = const()[name = string("matmul_3_transpose_y_0"), val = bool(false)]; tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = reshape_9_cast_fp16, y = reshape_10_cast_fp16)[name = string("matmul_3_cast_fp16")]; tensor concat_71 = const()[name = string("concat_71"), val = tensor([1, 16, 64, 128])]; tensor reshape_11_cast_fp16 = reshape(shape = concat_71, x = matmul_3_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor var_3252_perm_0 = const()[name = string("op_3252_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3258 = const()[name = string("op_3258"), val = tensor([1, 64, 2048])]; tensor var_3252_cast_fp16 = transpose(perm = var_3252_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_220")]; tensor output_21_cast_fp16 = reshape(shape = var_3258, x = var_3252_cast_fp16)[name = string("output_21_cast_fp16")]; tensor var_3263 = const()[name = string("op_3263"), val = tensor([0, 2, 1])]; string var_3279_pad_type_0 = const()[name = string("op_3279_pad_type_0"), val = string("valid")]; int32 var_3279_groups_0 = const()[name = string("op_3279_groups_0"), val = int32(1)]; tensor var_3279_strides_0 = const()[name = string("op_3279_strides_0"), val = tensor([1])]; tensor var_3279_pad_0 = const()[name = string("op_3279_pad_0"), val = tensor([0, 0])]; tensor var_3279_dilations_0 = const()[name = string("op_3279_dilations_0"), val = tensor([1])]; tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(297687552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299260480))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3264_cast_fp16 = transpose(perm = var_3263, x = output_21_cast_fp16)[name = string("transpose_219")]; tensor var_3279_cast_fp16 = conv(dilations = var_3279_dilations_0, groups = var_3279_groups_0, pad = var_3279_pad_0, pad_type = var_3279_pad_type_0, strides = var_3279_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_3264_cast_fp16)[name = string("op_3279_cast_fp16")]; tensor var_3283 = const()[name = string("op_3283"), val = tensor([0, 2, 1])]; tensor attn_output_7_cast_fp16 = transpose(perm = var_3283, x = var_3279_cast_fp16)[name = string("transpose_218")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_7_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; int32 var_3298 = const()[name = string("op_3298"), val = int32(-1)]; fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3300_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_57_promoted_to_fp16)[name = string("op_3300_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_3298, interleave = input_65_interleave_0, values = (hidden_states_39_cast_fp16, var_3300_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_3295_to_fp16 = const()[name = string("op_3295_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_3295_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 64, 1024])]; tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; tensor const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299276928)))]; tensor x_13_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_59_promoted_to_fp16)[name = string("x_13_cast_fp16")]; tensor var_3320 = const()[name = string("op_3320"), val = tensor([0, 2, 1])]; tensor input_67_axes_0 = const()[name = string("input_67_axes_0"), val = tensor([2])]; tensor var_3321 = transpose(perm = var_3320, x = x_13_cast_fp16)[name = string("transpose_217")]; tensor input_67 = expand_dims(axes = input_67_axes_0, x = var_3321)[name = string("input_67")]; string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; tensor input_69 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = model_model_layers_3_mlp_gate_proj_weight_palettized, x = input_67)[name = string("input_69")]; string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_3_mlp_up_proj_weight_palettized, x = input_67)[name = string("b_7")]; tensor c_7 = silu(x = input_69)[name = string("c_7")]; tensor input_71 = mul(x = c_7, y = b_7)[name = string("input_71")]; string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_3_mlp_down_proj_weight_palettized, x = input_71)[name = string("e_7")]; tensor var_3343_axes_0 = const()[name = string("op_3343_axes_0"), val = tensor([2])]; tensor var_3343 = squeeze(axes = var_3343_axes_0, x = e_7)[name = string("op_3343")]; tensor var_3344 = const()[name = string("op_3344"), val = tensor([0, 2, 1])]; tensor var_3345 = transpose(perm = var_3344, x = var_3343)[name = string("transpose_216")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = var_3345)[name = string("hidden_states_41_cast_fp16")]; int32 var_3359 = const()[name = string("op_3359"), val = int32(-1)]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3361_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_3361_cast_fp16")]; bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; tensor input_73_cast_fp16 = concat(axis = var_3359, interleave = input_73_interleave_0, values = (hidden_states_41_cast_fp16, var_3361_cast_fp16))[name = string("input_73_cast_fp16")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_3356_to_fp16 = const()[name = string("op_3356_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_3356_to_fp16, x = input_73_cast_fp16)[name = string("normed_65_cast_fp16")]; tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 64, 1024])]; tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; tensor const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299279040)))]; tensor hidden_states_43_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_62_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor var_3373 = const()[name = string("op_3373"), val = tensor([0, 2, 1])]; tensor var_3376_axes_0 = const()[name = string("op_3376_axes_0"), val = tensor([2])]; tensor var_3374_cast_fp16 = transpose(perm = var_3373, x = hidden_states_43_cast_fp16)[name = string("transpose_215")]; tensor var_3376_cast_fp16 = expand_dims(axes = var_3376_axes_0, x = var_3374_cast_fp16)[name = string("op_3376_cast_fp16")]; string var_3392_pad_type_0 = const()[name = string("op_3392_pad_type_0"), val = string("valid")]; tensor var_3392_strides_0 = const()[name = string("op_3392_strides_0"), val = tensor([1, 1])]; tensor var_3392_pad_0 = const()[name = string("op_3392_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3392_dilations_0 = const()[name = string("op_3392_dilations_0"), val = tensor([1, 1])]; int32 var_3392_groups_0 = const()[name = string("op_3392_groups_0"), val = int32(1)]; tensor var_3392 = conv(dilations = var_3392_dilations_0, groups = var_3392_groups_0, pad = var_3392_pad_0, pad_type = var_3392_pad_type_0, strides = var_3392_strides_0, weight = model_model_layers_4_self_attn_q_proj_weight_palettized, x = var_3376_cast_fp16)[name = string("op_3392")]; tensor var_3397 = const()[name = string("op_3397"), val = tensor([1, 16, 128, 64])]; tensor var_3398 = reshape(shape = var_3397, x = var_3392)[name = string("op_3398")]; tensor var_3403 = const()[name = string("op_3403"), val = tensor([0, 1, 3, 2])]; string var_3415_pad_type_0 = const()[name = string("op_3415_pad_type_0"), val = string("valid")]; tensor var_3415_strides_0 = const()[name = string("op_3415_strides_0"), val = tensor([1, 1])]; tensor var_3415_pad_0 = const()[name = string("op_3415_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3415_dilations_0 = const()[name = string("op_3415_dilations_0"), val = tensor([1, 1])]; int32 var_3415_groups_0 = const()[name = string("op_3415_groups_0"), val = int32(1)]; tensor var_3415 = conv(dilations = var_3415_dilations_0, groups = var_3415_groups_0, pad = var_3415_pad_0, pad_type = var_3415_pad_type_0, strides = var_3415_strides_0, weight = model_model_layers_4_self_attn_k_proj_weight_palettized, x = var_3376_cast_fp16)[name = string("op_3415")]; tensor var_3420 = const()[name = string("op_3420"), val = tensor([1, 8, 128, 64])]; tensor var_3421 = reshape(shape = var_3420, x = var_3415)[name = string("op_3421")]; tensor var_3426 = const()[name = string("op_3426"), val = tensor([0, 1, 3, 2])]; string var_3438_pad_type_0 = const()[name = string("op_3438_pad_type_0"), val = string("valid")]; tensor var_3438_strides_0 = const()[name = string("op_3438_strides_0"), val = tensor([1, 1])]; tensor var_3438_pad_0 = const()[name = string("op_3438_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3438_dilations_0 = const()[name = string("op_3438_dilations_0"), val = tensor([1, 1])]; int32 var_3438_groups_0 = const()[name = string("op_3438_groups_0"), val = int32(1)]; tensor var_3438 = conv(dilations = var_3438_dilations_0, groups = var_3438_groups_0, pad = var_3438_pad_0, pad_type = var_3438_pad_type_0, strides = var_3438_strides_0, weight = model_model_layers_4_self_attn_v_proj_weight_palettized, x = var_3376_cast_fp16)[name = string("op_3438")]; tensor var_3443 = const()[name = string("op_3443"), val = tensor([1, 8, 128, 64])]; tensor var_3444 = reshape(shape = var_3443, x = var_3438)[name = string("op_3444")]; tensor var_3449 = const()[name = string("op_3449"), val = tensor([0, 1, 3, 2])]; int32 var_3462 = const()[name = string("op_3462"), val = int32(-1)]; fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_45 = transpose(perm = var_3403, x = var_3398)[name = string("transpose_214")]; tensor var_3464 = mul(x = hidden_states_45, y = const_63_promoted)[name = string("op_3464")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77 = concat(axis = var_3462, interleave = input_77_interleave_0, values = (hidden_states_45, var_3464))[name = string("input_77")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_3459_to_fp16 = const()[name = string("op_3459_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_3459_to_fp16, x = input_77)[name = string("normed_69_cast_fp16")]; tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_71 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71")]; tensor const_65 = const()[name = string("const_65"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299281152)))]; tensor q_9 = mul(x = normed_71, y = const_65)[name = string("q_9")]; int32 var_3484 = const()[name = string("op_3484"), val = int32(-1)]; fp16 const_66_promoted = const()[name = string("const_66_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_47 = transpose(perm = var_3426, x = var_3421)[name = string("transpose_213")]; tensor var_3486 = mul(x = hidden_states_47, y = const_66_promoted)[name = string("op_3486")]; bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; tensor input_79 = concat(axis = var_3484, interleave = input_79_interleave_0, values = (hidden_states_47, var_3486))[name = string("input_79")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_3481_to_fp16 = const()[name = string("op_3481_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_3481_to_fp16, x = input_79)[name = string("normed_73_cast_fp16")]; tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_75 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75")]; tensor const_68 = const()[name = string("const_68"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299281472)))]; tensor k_9 = mul(x = normed_75, y = const_68)[name = string("k_9")]; tensor var_3507 = mul(x = q_9, y = cos_1)[name = string("op_3507")]; tensor var_3512_begin_0 = const()[name = string("op_3512_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3512_end_0 = const()[name = string("op_3512_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_3512_end_mask_0 = const()[name = string("op_3512_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3512 = slice_by_index(begin = var_3512_begin_0, end = var_3512_end_0, end_mask = var_3512_end_mask_0, x = q_9)[name = string("op_3512")]; fp16 const_69_promoted = const()[name = string("const_69_promoted"), val = fp16(-0x1p+0)]; tensor var_3513 = mul(x = var_3512, y = const_69_promoted)[name = string("op_3513")]; tensor var_3518_begin_0 = const()[name = string("op_3518_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3518_end_0 = const()[name = string("op_3518_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_3518_end_mask_0 = const()[name = string("op_3518_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3518 = slice_by_index(begin = var_3518_begin_0, end = var_3518_end_0, end_mask = var_3518_end_mask_0, x = q_9)[name = string("op_3518")]; int32 var_3520 = const()[name = string("op_3520"), val = int32(-1)]; bool var_3521_interleave_0 = const()[name = string("op_3521_interleave_0"), val = bool(false)]; tensor var_3521 = concat(axis = var_3520, interleave = var_3521_interleave_0, values = (var_3513, var_3518))[name = string("op_3521")]; tensor var_3522 = mul(x = var_3521, y = sin_1)[name = string("op_3522")]; tensor query_9 = add(x = var_3507, y = var_3522)[name = string("query_9")]; tensor var_3525 = mul(x = k_9, y = cos_1)[name = string("op_3525")]; tensor var_3530_begin_0 = const()[name = string("op_3530_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3530_end_0 = const()[name = string("op_3530_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_3530_end_mask_0 = const()[name = string("op_3530_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3530 = slice_by_index(begin = var_3530_begin_0, end = var_3530_end_0, end_mask = var_3530_end_mask_0, x = k_9)[name = string("op_3530")]; fp16 const_70_promoted = const()[name = string("const_70_promoted"), val = fp16(-0x1p+0)]; tensor var_3531 = mul(x = var_3530, y = const_70_promoted)[name = string("op_3531")]; tensor var_3536_begin_0 = const()[name = string("op_3536_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3536_end_0 = const()[name = string("op_3536_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_3536_end_mask_0 = const()[name = string("op_3536_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3536 = slice_by_index(begin = var_3536_begin_0, end = var_3536_end_0, end_mask = var_3536_end_mask_0, x = k_9)[name = string("op_3536")]; int32 var_3538 = const()[name = string("op_3538"), val = int32(-1)]; bool var_3539_interleave_0 = const()[name = string("op_3539_interleave_0"), val = bool(false)]; tensor var_3539 = concat(axis = var_3538, interleave = var_3539_interleave_0, values = (var_3531, var_3536))[name = string("op_3539")]; tensor var_3540 = mul(x = var_3539, y = sin_1)[name = string("op_3540")]; tensor key_9 = add(x = var_3525, y = var_3540)[name = string("key_9")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_74")]; tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_52, concat_75_values1_0, var_1746, concat_75_values3_0))[name = string("concat_75")]; tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_9, x = coreml_update_state_63)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_176_write_state")]; tensor coreml_update_state_64 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_176")]; tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([32])]; tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([33])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_78")]; tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_58, concat_79_values1_0, var_1746, concat_79_values3_0))[name = string("concat_79")]; tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_41 = transpose(perm = var_3449, x = var_3444)[name = string("transpose_212")]; tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = value_41, x = coreml_update_state_64)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_177_write_state")]; tensor coreml_update_state_65 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_177")]; tensor var_3611_begin_0 = const()[name = string("op_3611_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_3611_end_0 = const()[name = string("op_3611_end_0"), val = tensor([5, 8, 1536, 128])]; tensor var_3611_end_mask_0 = const()[name = string("op_3611_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3611_cast_fp16 = slice_by_index(begin = var_3611_begin_0, end = var_3611_end_0, end_mask = var_3611_end_mask_0, x = coreml_update_state_65)[name = string("op_3611_cast_fp16")]; tensor key_cache_9_axes_0 = const()[name = string("key_cache_9_axes_0"), val = tensor([0])]; tensor key_cache_9_cast_fp16 = squeeze(axes = key_cache_9_axes_0, x = var_3611_cast_fp16)[name = string("key_cache_9_cast_fp16")]; tensor var_3618_begin_0 = const()[name = string("op_3618_begin_0"), val = tensor([32, 0, 0, 0])]; tensor var_3618_end_0 = const()[name = string("op_3618_end_0"), val = tensor([33, 8, 1536, 128])]; tensor var_3618_end_mask_0 = const()[name = string("op_3618_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3618_cast_fp16 = slice_by_index(begin = var_3618_begin_0, end = var_3618_end_0, end_mask = var_3618_end_mask_0, x = coreml_update_state_65)[name = string("op_3618_cast_fp16")]; tensor value_cache_9_axes_0 = const()[name = string("value_cache_9_axes_0"), val = tensor([0])]; tensor value_cache_9_cast_fp16 = squeeze(axes = value_cache_9_axes_0, x = var_3618_cast_fp16)[name = string("value_cache_9_cast_fp16")]; tensor var_3642_axes_0 = const()[name = string("op_3642_axes_0"), val = tensor([1])]; tensor var_3642_cast_fp16 = expand_dims(axes = var_3642_axes_0, x = key_cache_9_cast_fp16)[name = string("op_3642_cast_fp16")]; tensor var_3647 = const()[name = string("op_3647"), val = tensor([1, 2, 1, 1])]; tensor value_45_cast_fp16 = tile(reps = var_3647, x = var_3642_cast_fp16)[name = string("value_45_cast_fp16")]; tensor var_3653 = const()[name = string("op_3653"), val = tensor([1, 16, 1536, 128])]; tensor key_states_19_cast_fp16 = reshape(shape = var_3653, x = value_45_cast_fp16)[name = string("key_states_19_cast_fp16")]; tensor var_3656_axes_0 = const()[name = string("op_3656_axes_0"), val = tensor([1])]; tensor var_3656_cast_fp16 = expand_dims(axes = var_3656_axes_0, x = value_cache_9_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3661 = const()[name = string("op_3661"), val = tensor([1, 2, 1, 1])]; tensor value_49_cast_fp16 = tile(reps = var_3661, x = var_3656_cast_fp16)[name = string("value_49_cast_fp16")]; bool var_3682_transpose_x_0 = const()[name = string("op_3682_transpose_x_0"), val = bool(false)]; bool var_3682_transpose_y_0 = const()[name = string("op_3682_transpose_y_0"), val = bool(true)]; tensor var_3682 = matmul(transpose_x = var_3682_transpose_x_0, transpose_y = var_3682_transpose_y_0, x = query_9, y = key_states_19_cast_fp16)[name = string("op_3682")]; fp16 var_3683_to_fp16 = const()[name = string("op_3683_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_17_cast_fp16 = mul(x = var_3682, y = var_3683_to_fp16)[name = string("attention_17_cast_fp16")]; tensor attention_19_cast_fp16 = add(x = attention_17_cast_fp16, y = causal_mask)[name = string("attention_19_cast_fp16")]; int32 var_3692 = const()[name = string("op_3692"), val = int32(-1)]; tensor var_3694_cast_fp16 = softmax(axis = var_3692, x = attention_19_cast_fp16)[name = string("op_3694_cast_fp16")]; tensor concat_84 = const()[name = string("concat_84"), val = tensor([16, 64, 1536])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_84, x = var_3694_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor concat_85 = const()[name = string("concat_85"), val = tensor([16, 1536, 128])]; tensor reshape_13_cast_fp16 = reshape(shape = concat_85, x = value_49_cast_fp16)[name = string("reshape_13_cast_fp16")]; bool matmul_4_transpose_x_0 = const()[name = string("matmul_4_transpose_x_0"), val = bool(false)]; bool matmul_4_transpose_y_0 = const()[name = string("matmul_4_transpose_y_0"), val = bool(false)]; tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = reshape_12_cast_fp16, y = reshape_13_cast_fp16)[name = string("matmul_4_cast_fp16")]; tensor concat_89 = const()[name = string("concat_89"), val = tensor([1, 16, 64, 128])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_89, x = matmul_4_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor var_3706_perm_0 = const()[name = string("op_3706_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3712 = const()[name = string("op_3712"), val = tensor([1, 64, 2048])]; tensor var_3706_cast_fp16 = transpose(perm = var_3706_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_211")]; tensor output_27_cast_fp16 = reshape(shape = var_3712, x = var_3706_cast_fp16)[name = string("output_27_cast_fp16")]; tensor var_3717 = const()[name = string("op_3717"), val = tensor([0, 2, 1])]; string var_3733_pad_type_0 = const()[name = string("op_3733_pad_type_0"), val = string("valid")]; int32 var_3733_groups_0 = const()[name = string("op_3733_groups_0"), val = int32(1)]; tensor var_3733_strides_0 = const()[name = string("op_3733_strides_0"), val = tensor([1])]; tensor var_3733_pad_0 = const()[name = string("op_3733_pad_0"), val = tensor([0, 0])]; tensor var_3733_dilations_0 = const()[name = string("op_3733_dilations_0"), val = tensor([1])]; tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(299281792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300854720))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_3718_cast_fp16 = transpose(perm = var_3717, x = output_27_cast_fp16)[name = string("transpose_210")]; tensor var_3733_cast_fp16 = conv(dilations = var_3733_dilations_0, groups = var_3733_groups_0, pad = var_3733_pad_0, pad_type = var_3733_pad_type_0, strides = var_3733_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3718_cast_fp16)[name = string("op_3733_cast_fp16")]; tensor var_3737 = const()[name = string("op_3737"), val = tensor([0, 2, 1])]; tensor attn_output_9_cast_fp16 = transpose(perm = var_3737, x = var_3733_cast_fp16)[name = string("transpose_209")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = attn_output_9_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; int32 var_3752 = const()[name = string("op_3752"), val = int32(-1)]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3754_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_72_promoted_to_fp16)[name = string("op_3754_cast_fp16")]; bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; tensor input_83_cast_fp16 = concat(axis = var_3752, interleave = input_83_interleave_0, values = (hidden_states_49_cast_fp16, var_3754_cast_fp16))[name = string("input_83_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_3749_to_fp16 = const()[name = string("op_3749_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3749_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 64, 1024])]; tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; tensor const_74_promoted_to_fp16 = const()[name = string("const_74_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300871168)))]; tensor x_17_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_74_promoted_to_fp16)[name = string("x_17_cast_fp16")]; tensor var_3774 = const()[name = string("op_3774"), val = tensor([0, 2, 1])]; tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; tensor var_3775 = transpose(perm = var_3774, x = x_17_cast_fp16)[name = string("transpose_208")]; tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_3775)[name = string("input_85")]; string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; tensor input_87 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = model_model_layers_4_mlp_gate_proj_weight_palettized, x = input_85)[name = string("input_87")]; string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_4_mlp_up_proj_weight_palettized, x = input_85)[name = string("b_9")]; tensor c_9 = silu(x = input_87)[name = string("c_9")]; tensor input_89 = mul(x = c_9, y = b_9)[name = string("input_89")]; string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_4_mlp_down_proj_weight_palettized, x = input_89)[name = string("e_9")]; tensor var_3797_axes_0 = const()[name = string("op_3797_axes_0"), val = tensor([2])]; tensor var_3797 = squeeze(axes = var_3797_axes_0, x = e_9)[name = string("op_3797")]; tensor var_3798 = const()[name = string("op_3798"), val = tensor([0, 2, 1])]; tensor var_3799 = transpose(perm = var_3798, x = var_3797)[name = string("transpose_207")]; tensor hidden_states_51_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = var_3799)[name = string("hidden_states_51_cast_fp16")]; int32 var_3813 = const()[name = string("op_3813"), val = int32(-1)]; fp16 const_75_promoted_to_fp16 = const()[name = string("const_75_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3815_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_75_promoted_to_fp16)[name = string("op_3815_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_3813, interleave = input_91_interleave_0, values = (hidden_states_51_cast_fp16, var_3815_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_3810_to_fp16 = const()[name = string("op_3810_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3810_to_fp16, x = input_91_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 64, 1024])]; tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; tensor const_77_promoted_to_fp16 = const()[name = string("const_77_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300873280)))]; tensor hidden_states_53_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_77_promoted_to_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor var_3827 = const()[name = string("op_3827"), val = tensor([0, 2, 1])]; tensor var_3830_axes_0 = const()[name = string("op_3830_axes_0"), val = tensor([2])]; tensor var_3828_cast_fp16 = transpose(perm = var_3827, x = hidden_states_53_cast_fp16)[name = string("transpose_206")]; tensor var_3830_cast_fp16 = expand_dims(axes = var_3830_axes_0, x = var_3828_cast_fp16)[name = string("op_3830_cast_fp16")]; string var_3846_pad_type_0 = const()[name = string("op_3846_pad_type_0"), val = string("valid")]; tensor var_3846_strides_0 = const()[name = string("op_3846_strides_0"), val = tensor([1, 1])]; tensor var_3846_pad_0 = const()[name = string("op_3846_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3846_dilations_0 = const()[name = string("op_3846_dilations_0"), val = tensor([1, 1])]; int32 var_3846_groups_0 = const()[name = string("op_3846_groups_0"), val = int32(1)]; tensor var_3846 = conv(dilations = var_3846_dilations_0, groups = var_3846_groups_0, pad = var_3846_pad_0, pad_type = var_3846_pad_type_0, strides = var_3846_strides_0, weight = model_model_layers_5_self_attn_q_proj_weight_palettized, x = var_3830_cast_fp16)[name = string("op_3846")]; tensor var_3851 = const()[name = string("op_3851"), val = tensor([1, 16, 128, 64])]; tensor var_3852 = reshape(shape = var_3851, x = var_3846)[name = string("op_3852")]; tensor var_3857 = const()[name = string("op_3857"), val = tensor([0, 1, 3, 2])]; string var_3869_pad_type_0 = const()[name = string("op_3869_pad_type_0"), val = string("valid")]; tensor var_3869_strides_0 = const()[name = string("op_3869_strides_0"), val = tensor([1, 1])]; tensor var_3869_pad_0 = const()[name = string("op_3869_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3869_dilations_0 = const()[name = string("op_3869_dilations_0"), val = tensor([1, 1])]; int32 var_3869_groups_0 = const()[name = string("op_3869_groups_0"), val = int32(1)]; tensor var_3869 = conv(dilations = var_3869_dilations_0, groups = var_3869_groups_0, pad = var_3869_pad_0, pad_type = var_3869_pad_type_0, strides = var_3869_strides_0, weight = model_model_layers_5_self_attn_k_proj_weight_palettized, x = var_3830_cast_fp16)[name = string("op_3869")]; tensor var_3874 = const()[name = string("op_3874"), val = tensor([1, 8, 128, 64])]; tensor var_3875 = reshape(shape = var_3874, x = var_3869)[name = string("op_3875")]; tensor var_3880 = const()[name = string("op_3880"), val = tensor([0, 1, 3, 2])]; string var_3892_pad_type_0 = const()[name = string("op_3892_pad_type_0"), val = string("valid")]; tensor var_3892_strides_0 = const()[name = string("op_3892_strides_0"), val = tensor([1, 1])]; tensor var_3892_pad_0 = const()[name = string("op_3892_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3892_dilations_0 = const()[name = string("op_3892_dilations_0"), val = tensor([1, 1])]; int32 var_3892_groups_0 = const()[name = string("op_3892_groups_0"), val = int32(1)]; tensor var_3892 = conv(dilations = var_3892_dilations_0, groups = var_3892_groups_0, pad = var_3892_pad_0, pad_type = var_3892_pad_type_0, strides = var_3892_strides_0, weight = model_model_layers_5_self_attn_v_proj_weight_palettized, x = var_3830_cast_fp16)[name = string("op_3892")]; tensor var_3897 = const()[name = string("op_3897"), val = tensor([1, 8, 128, 64])]; tensor var_3898 = reshape(shape = var_3897, x = var_3892)[name = string("op_3898")]; tensor var_3903 = const()[name = string("op_3903"), val = tensor([0, 1, 3, 2])]; int32 var_3916 = const()[name = string("op_3916"), val = int32(-1)]; fp16 const_78_promoted = const()[name = string("const_78_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_55 = transpose(perm = var_3857, x = var_3852)[name = string("transpose_205")]; tensor var_3918 = mul(x = hidden_states_55, y = const_78_promoted)[name = string("op_3918")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95 = concat(axis = var_3916, interleave = input_95_interleave_0, values = (hidden_states_55, var_3918))[name = string("input_95")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_3913_to_fp16 = const()[name = string("op_3913_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3913_to_fp16, x = input_95)[name = string("normed_85_cast_fp16")]; tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_87 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87")]; tensor const_80 = const()[name = string("const_80"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300875392)))]; tensor q_11 = mul(x = normed_87, y = const_80)[name = string("q_11")]; int32 var_3938 = const()[name = string("op_3938"), val = int32(-1)]; fp16 const_81_promoted = const()[name = string("const_81_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_57 = transpose(perm = var_3880, x = var_3875)[name = string("transpose_204")]; tensor var_3940 = mul(x = hidden_states_57, y = const_81_promoted)[name = string("op_3940")]; bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; tensor input_97 = concat(axis = var_3938, interleave = input_97_interleave_0, values = (hidden_states_57, var_3940))[name = string("input_97")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_3935_to_fp16 = const()[name = string("op_3935_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3935_to_fp16, x = input_97)[name = string("normed_89_cast_fp16")]; tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_91 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91")]; tensor const_83 = const()[name = string("const_83"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300875712)))]; tensor k_11 = mul(x = normed_91, y = const_83)[name = string("k_11")]; tensor var_3961 = mul(x = q_11, y = cos_1)[name = string("op_3961")]; tensor var_3966_begin_0 = const()[name = string("op_3966_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3966_end_0 = const()[name = string("op_3966_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_3966_end_mask_0 = const()[name = string("op_3966_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3966 = slice_by_index(begin = var_3966_begin_0, end = var_3966_end_0, end_mask = var_3966_end_mask_0, x = q_11)[name = string("op_3966")]; fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; tensor var_3967 = mul(x = var_3966, y = const_84_promoted)[name = string("op_3967")]; tensor var_3972_begin_0 = const()[name = string("op_3972_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3972_end_0 = const()[name = string("op_3972_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_3972_end_mask_0 = const()[name = string("op_3972_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3972 = slice_by_index(begin = var_3972_begin_0, end = var_3972_end_0, end_mask = var_3972_end_mask_0, x = q_11)[name = string("op_3972")]; int32 var_3974 = const()[name = string("op_3974"), val = int32(-1)]; bool var_3975_interleave_0 = const()[name = string("op_3975_interleave_0"), val = bool(false)]; tensor var_3975 = concat(axis = var_3974, interleave = var_3975_interleave_0, values = (var_3967, var_3972))[name = string("op_3975")]; tensor var_3976 = mul(x = var_3975, y = sin_1)[name = string("op_3976")]; tensor query_11 = add(x = var_3961, y = var_3976)[name = string("query_11")]; tensor var_3979 = mul(x = k_11, y = cos_1)[name = string("op_3979")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3984 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = k_11)[name = string("op_3984")]; fp16 const_85_promoted = const()[name = string("const_85_promoted"), val = fp16(-0x1p+0)]; tensor var_3985 = mul(x = var_3984, y = const_85_promoted)[name = string("op_3985")]; tensor var_3990_begin_0 = const()[name = string("op_3990_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_3990_end_0 = const()[name = string("op_3990_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_3990_end_mask_0 = const()[name = string("op_3990_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_3990 = slice_by_index(begin = var_3990_begin_0, end = var_3990_end_0, end_mask = var_3990_end_mask_0, x = k_11)[name = string("op_3990")]; int32 var_3992 = const()[name = string("op_3992"), val = int32(-1)]; bool var_3993_interleave_0 = const()[name = string("op_3993_interleave_0"), val = bool(false)]; tensor var_3993 = concat(axis = var_3992, interleave = var_3993_interleave_0, values = (var_3985, var_3990))[name = string("op_3993")]; tensor var_3994 = mul(x = var_3993, y = sin_1)[name = string("op_3994")]; tensor key_11 = add(x = var_3979, y = var_3994)[name = string("key_11")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; int32 concat_92_axis_0 = const()[name = string("concat_92_axis_0"), val = int32(0)]; bool concat_92_interleave_0 = const()[name = string("concat_92_interleave_0"), val = bool(false)]; tensor concat_92 = concat(axis = concat_92_axis_0, interleave = concat_92_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_92")]; tensor concat_93_values1_0 = const()[name = string("concat_93_values1_0"), val = tensor([0])]; tensor concat_93_values3_0 = const()[name = string("concat_93_values3_0"), val = tensor([0])]; int32 concat_93_axis_0 = const()[name = string("concat_93_axis_0"), val = int32(0)]; bool concat_93_interleave_0 = const()[name = string("concat_93_interleave_0"), val = bool(false)]; tensor concat_93 = concat(axis = concat_93_axis_0, interleave = concat_93_interleave_0, values = (expand_dims_64, concat_93_values1_0, var_1746, concat_93_values3_0))[name = string("concat_93")]; tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_92, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_93, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_11, x = coreml_update_state_65)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_178_write_state")]; tensor coreml_update_state_66 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_178")]; tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([33])]; tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([34])]; int32 concat_96_axis_0 = const()[name = string("concat_96_axis_0"), val = int32(0)]; bool concat_96_interleave_0 = const()[name = string("concat_96_interleave_0"), val = bool(false)]; tensor concat_96 = concat(axis = concat_96_axis_0, interleave = concat_96_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_96")]; tensor concat_97_values1_0 = const()[name = string("concat_97_values1_0"), val = tensor([0])]; tensor concat_97_values3_0 = const()[name = string("concat_97_values3_0"), val = tensor([0])]; int32 concat_97_axis_0 = const()[name = string("concat_97_axis_0"), val = int32(0)]; bool concat_97_interleave_0 = const()[name = string("concat_97_interleave_0"), val = bool(false)]; tensor concat_97 = concat(axis = concat_97_axis_0, interleave = concat_97_interleave_0, values = (expand_dims_70, concat_97_values1_0, var_1746, concat_97_values3_0))[name = string("concat_97")]; tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_51 = transpose(perm = var_3903, x = var_3898)[name = string("transpose_203")]; tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_96, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_97, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = value_51, x = coreml_update_state_66)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_179_write_state")]; tensor coreml_update_state_67 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_179")]; tensor var_4065_begin_0 = const()[name = string("op_4065_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_4065_end_0 = const()[name = string("op_4065_end_0"), val = tensor([6, 8, 1536, 128])]; tensor var_4065_end_mask_0 = const()[name = string("op_4065_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4065_cast_fp16 = slice_by_index(begin = var_4065_begin_0, end = var_4065_end_0, end_mask = var_4065_end_mask_0, x = coreml_update_state_67)[name = string("op_4065_cast_fp16")]; tensor key_cache_11_axes_0 = const()[name = string("key_cache_11_axes_0"), val = tensor([0])]; tensor key_cache_11_cast_fp16 = squeeze(axes = key_cache_11_axes_0, x = var_4065_cast_fp16)[name = string("key_cache_11_cast_fp16")]; tensor var_4072_begin_0 = const()[name = string("op_4072_begin_0"), val = tensor([33, 0, 0, 0])]; tensor var_4072_end_0 = const()[name = string("op_4072_end_0"), val = tensor([34, 8, 1536, 128])]; tensor var_4072_end_mask_0 = const()[name = string("op_4072_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4072_cast_fp16 = slice_by_index(begin = var_4072_begin_0, end = var_4072_end_0, end_mask = var_4072_end_mask_0, x = coreml_update_state_67)[name = string("op_4072_cast_fp16")]; tensor value_cache_11_axes_0 = const()[name = string("value_cache_11_axes_0"), val = tensor([0])]; tensor value_cache_11_cast_fp16 = squeeze(axes = value_cache_11_axes_0, x = var_4072_cast_fp16)[name = string("value_cache_11_cast_fp16")]; tensor var_4096_axes_0 = const()[name = string("op_4096_axes_0"), val = tensor([1])]; tensor var_4096_cast_fp16 = expand_dims(axes = var_4096_axes_0, x = key_cache_11_cast_fp16)[name = string("op_4096_cast_fp16")]; tensor var_4101 = const()[name = string("op_4101"), val = tensor([1, 2, 1, 1])]; tensor value_55_cast_fp16 = tile(reps = var_4101, x = var_4096_cast_fp16)[name = string("value_55_cast_fp16")]; tensor var_4107 = const()[name = string("op_4107"), val = tensor([1, 16, 1536, 128])]; tensor key_states_23_cast_fp16 = reshape(shape = var_4107, x = value_55_cast_fp16)[name = string("key_states_23_cast_fp16")]; tensor var_4110_axes_0 = const()[name = string("op_4110_axes_0"), val = tensor([1])]; tensor var_4110_cast_fp16 = expand_dims(axes = var_4110_axes_0, x = value_cache_11_cast_fp16)[name = string("op_4110_cast_fp16")]; tensor var_4115 = const()[name = string("op_4115"), val = tensor([1, 2, 1, 1])]; tensor value_59_cast_fp16 = tile(reps = var_4115, x = var_4110_cast_fp16)[name = string("value_59_cast_fp16")]; bool var_4136_transpose_x_0 = const()[name = string("op_4136_transpose_x_0"), val = bool(false)]; bool var_4136_transpose_y_0 = const()[name = string("op_4136_transpose_y_0"), val = bool(true)]; tensor var_4136 = matmul(transpose_x = var_4136_transpose_x_0, transpose_y = var_4136_transpose_y_0, x = query_11, y = key_states_23_cast_fp16)[name = string("op_4136")]; fp16 var_4137_to_fp16 = const()[name = string("op_4137_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_21_cast_fp16 = mul(x = var_4136, y = var_4137_to_fp16)[name = string("attention_21_cast_fp16")]; tensor attention_23_cast_fp16 = add(x = attention_21_cast_fp16, y = causal_mask)[name = string("attention_23_cast_fp16")]; int32 var_4146 = const()[name = string("op_4146"), val = int32(-1)]; tensor var_4148_cast_fp16 = softmax(axis = var_4146, x = attention_23_cast_fp16)[name = string("op_4148_cast_fp16")]; tensor concat_102 = const()[name = string("concat_102"), val = tensor([16, 64, 1536])]; tensor reshape_15_cast_fp16 = reshape(shape = concat_102, x = var_4148_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor concat_103 = const()[name = string("concat_103"), val = tensor([16, 1536, 128])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_103, x = value_59_cast_fp16)[name = string("reshape_16_cast_fp16")]; bool matmul_5_transpose_x_0 = const()[name = string("matmul_5_transpose_x_0"), val = bool(false)]; bool matmul_5_transpose_y_0 = const()[name = string("matmul_5_transpose_y_0"), val = bool(false)]; tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = reshape_15_cast_fp16, y = reshape_16_cast_fp16)[name = string("matmul_5_cast_fp16")]; tensor concat_107 = const()[name = string("concat_107"), val = tensor([1, 16, 64, 128])]; tensor reshape_17_cast_fp16 = reshape(shape = concat_107, x = matmul_5_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor var_4160_perm_0 = const()[name = string("op_4160_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4166 = const()[name = string("op_4166"), val = tensor([1, 64, 2048])]; tensor var_4160_cast_fp16 = transpose(perm = var_4160_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_202")]; tensor output_33_cast_fp16 = reshape(shape = var_4166, x = var_4160_cast_fp16)[name = string("output_33_cast_fp16")]; tensor var_4171 = const()[name = string("op_4171"), val = tensor([0, 2, 1])]; string var_4187_pad_type_0 = const()[name = string("op_4187_pad_type_0"), val = string("valid")]; int32 var_4187_groups_0 = const()[name = string("op_4187_groups_0"), val = int32(1)]; tensor var_4187_strides_0 = const()[name = string("op_4187_strides_0"), val = tensor([1])]; tensor var_4187_pad_0 = const()[name = string("op_4187_pad_0"), val = tensor([0, 0])]; tensor var_4187_dilations_0 = const()[name = string("op_4187_dilations_0"), val = tensor([1])]; tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(300876032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302448960))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4172_cast_fp16 = transpose(perm = var_4171, x = output_33_cast_fp16)[name = string("transpose_201")]; tensor var_4187_cast_fp16 = conv(dilations = var_4187_dilations_0, groups = var_4187_groups_0, pad = var_4187_pad_0, pad_type = var_4187_pad_type_0, strides = var_4187_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_4172_cast_fp16)[name = string("op_4187_cast_fp16")]; tensor var_4191 = const()[name = string("op_4191"), val = tensor([0, 2, 1])]; tensor attn_output_11_cast_fp16 = transpose(perm = var_4191, x = var_4187_cast_fp16)[name = string("transpose_200")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_51_cast_fp16, y = attn_output_11_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; int32 var_4206 = const()[name = string("op_4206"), val = int32(-1)]; fp16 const_87_promoted_to_fp16 = const()[name = string("const_87_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_87_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; tensor input_101_cast_fp16 = concat(axis = var_4206, interleave = input_101_interleave_0, values = (hidden_states_59_cast_fp16, var_4208_cast_fp16))[name = string("input_101_cast_fp16")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_4203_to_fp16 = const()[name = string("op_4203_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_4203_to_fp16, x = input_101_cast_fp16)[name = string("normed_93_cast_fp16")]; tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 64, 1024])]; tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; tensor const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302465408)))]; tensor x_21_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_89_promoted_to_fp16)[name = string("x_21_cast_fp16")]; tensor var_4228 = const()[name = string("op_4228"), val = tensor([0, 2, 1])]; tensor input_103_axes_0 = const()[name = string("input_103_axes_0"), val = tensor([2])]; tensor var_4229 = transpose(perm = var_4228, x = x_21_cast_fp16)[name = string("transpose_199")]; tensor input_103 = expand_dims(axes = input_103_axes_0, x = var_4229)[name = string("input_103")]; string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; tensor input_105 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = model_model_layers_5_mlp_gate_proj_weight_palettized, x = input_103)[name = string("input_105")]; string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_5_mlp_up_proj_weight_palettized, x = input_103)[name = string("b_11")]; tensor c_11 = silu(x = input_105)[name = string("c_11")]; tensor input_107 = mul(x = c_11, y = b_11)[name = string("input_107")]; string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_5_mlp_down_proj_weight_palettized, x = input_107)[name = string("e_11")]; tensor var_4251_axes_0 = const()[name = string("op_4251_axes_0"), val = tensor([2])]; tensor var_4251 = squeeze(axes = var_4251_axes_0, x = e_11)[name = string("op_4251")]; tensor var_4252 = const()[name = string("op_4252"), val = tensor([0, 2, 1])]; tensor var_4253 = transpose(perm = var_4252, x = var_4251)[name = string("transpose_198")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = var_4253)[name = string("hidden_states_61_cast_fp16")]; int32 var_4267 = const()[name = string("op_4267"), val = int32(-1)]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4269_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_4269_cast_fp16")]; bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; tensor input_109_cast_fp16 = concat(axis = var_4267, interleave = input_109_interleave_0, values = (hidden_states_61_cast_fp16, var_4269_cast_fp16))[name = string("input_109_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_4264_to_fp16 = const()[name = string("op_4264_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_4264_to_fp16, x = input_109_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 64, 1024])]; tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; tensor const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302467520)))]; tensor hidden_states_63_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_92_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor var_4281 = const()[name = string("op_4281"), val = tensor([0, 2, 1])]; tensor var_4284_axes_0 = const()[name = string("op_4284_axes_0"), val = tensor([2])]; tensor var_4282_cast_fp16 = transpose(perm = var_4281, x = hidden_states_63_cast_fp16)[name = string("transpose_197")]; tensor var_4284_cast_fp16 = expand_dims(axes = var_4284_axes_0, x = var_4282_cast_fp16)[name = string("op_4284_cast_fp16")]; string var_4300_pad_type_0 = const()[name = string("op_4300_pad_type_0"), val = string("valid")]; tensor var_4300_strides_0 = const()[name = string("op_4300_strides_0"), val = tensor([1, 1])]; tensor var_4300_pad_0 = const()[name = string("op_4300_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4300_dilations_0 = const()[name = string("op_4300_dilations_0"), val = tensor([1, 1])]; int32 var_4300_groups_0 = const()[name = string("op_4300_groups_0"), val = int32(1)]; tensor var_4300 = conv(dilations = var_4300_dilations_0, groups = var_4300_groups_0, pad = var_4300_pad_0, pad_type = var_4300_pad_type_0, strides = var_4300_strides_0, weight = model_model_layers_6_self_attn_q_proj_weight_palettized, x = var_4284_cast_fp16)[name = string("op_4300")]; tensor var_4305 = const()[name = string("op_4305"), val = tensor([1, 16, 128, 64])]; tensor var_4306 = reshape(shape = var_4305, x = var_4300)[name = string("op_4306")]; tensor var_4311 = const()[name = string("op_4311"), val = tensor([0, 1, 3, 2])]; string var_4323_pad_type_0 = const()[name = string("op_4323_pad_type_0"), val = string("valid")]; tensor var_4323_strides_0 = const()[name = string("op_4323_strides_0"), val = tensor([1, 1])]; tensor var_4323_pad_0 = const()[name = string("op_4323_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4323_dilations_0 = const()[name = string("op_4323_dilations_0"), val = tensor([1, 1])]; int32 var_4323_groups_0 = const()[name = string("op_4323_groups_0"), val = int32(1)]; tensor var_4323 = conv(dilations = var_4323_dilations_0, groups = var_4323_groups_0, pad = var_4323_pad_0, pad_type = var_4323_pad_type_0, strides = var_4323_strides_0, weight = model_model_layers_6_self_attn_k_proj_weight_palettized, x = var_4284_cast_fp16)[name = string("op_4323")]; tensor var_4328 = const()[name = string("op_4328"), val = tensor([1, 8, 128, 64])]; tensor var_4329 = reshape(shape = var_4328, x = var_4323)[name = string("op_4329")]; tensor var_4334 = const()[name = string("op_4334"), val = tensor([0, 1, 3, 2])]; string var_4346_pad_type_0 = const()[name = string("op_4346_pad_type_0"), val = string("valid")]; tensor var_4346_strides_0 = const()[name = string("op_4346_strides_0"), val = tensor([1, 1])]; tensor var_4346_pad_0 = const()[name = string("op_4346_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4346_dilations_0 = const()[name = string("op_4346_dilations_0"), val = tensor([1, 1])]; int32 var_4346_groups_0 = const()[name = string("op_4346_groups_0"), val = int32(1)]; tensor var_4346 = conv(dilations = var_4346_dilations_0, groups = var_4346_groups_0, pad = var_4346_pad_0, pad_type = var_4346_pad_type_0, strides = var_4346_strides_0, weight = model_model_layers_6_self_attn_v_proj_weight_palettized, x = var_4284_cast_fp16)[name = string("op_4346")]; tensor var_4351 = const()[name = string("op_4351"), val = tensor([1, 8, 128, 64])]; tensor var_4352 = reshape(shape = var_4351, x = var_4346)[name = string("op_4352")]; tensor var_4357 = const()[name = string("op_4357"), val = tensor([0, 1, 3, 2])]; int32 var_4370 = const()[name = string("op_4370"), val = int32(-1)]; fp16 const_93_promoted = const()[name = string("const_93_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_65 = transpose(perm = var_4311, x = var_4306)[name = string("transpose_196")]; tensor var_4372 = mul(x = hidden_states_65, y = const_93_promoted)[name = string("op_4372")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113 = concat(axis = var_4370, interleave = input_113_interleave_0, values = (hidden_states_65, var_4372))[name = string("input_113")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_4367_to_fp16 = const()[name = string("op_4367_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_4367_to_fp16, x = input_113)[name = string("normed_101_cast_fp16")]; tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_103 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103")]; tensor const_95 = const()[name = string("const_95"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302469632)))]; tensor q_13 = mul(x = normed_103, y = const_95)[name = string("q_13")]; int32 var_4392 = const()[name = string("op_4392"), val = int32(-1)]; fp16 const_96_promoted = const()[name = string("const_96_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_67 = transpose(perm = var_4334, x = var_4329)[name = string("transpose_195")]; tensor var_4394 = mul(x = hidden_states_67, y = const_96_promoted)[name = string("op_4394")]; bool input_115_interleave_0 = const()[name = string("input_115_interleave_0"), val = bool(false)]; tensor input_115 = concat(axis = var_4392, interleave = input_115_interleave_0, values = (hidden_states_67, var_4394))[name = string("input_115")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_4389_to_fp16 = const()[name = string("op_4389_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_4389_to_fp16, x = input_115)[name = string("normed_105_cast_fp16")]; tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_107 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107")]; tensor const_98 = const()[name = string("const_98"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302469952)))]; tensor k_13 = mul(x = normed_107, y = const_98)[name = string("k_13")]; tensor var_4415 = mul(x = q_13, y = cos_1)[name = string("op_4415")]; tensor var_4420_begin_0 = const()[name = string("op_4420_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4420_end_0 = const()[name = string("op_4420_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_4420_end_mask_0 = const()[name = string("op_4420_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4420 = slice_by_index(begin = var_4420_begin_0, end = var_4420_end_0, end_mask = var_4420_end_mask_0, x = q_13)[name = string("op_4420")]; fp16 const_99_promoted = const()[name = string("const_99_promoted"), val = fp16(-0x1p+0)]; tensor var_4421 = mul(x = var_4420, y = const_99_promoted)[name = string("op_4421")]; tensor var_4426_begin_0 = const()[name = string("op_4426_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4426_end_0 = const()[name = string("op_4426_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_4426_end_mask_0 = const()[name = string("op_4426_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4426 = slice_by_index(begin = var_4426_begin_0, end = var_4426_end_0, end_mask = var_4426_end_mask_0, x = q_13)[name = string("op_4426")]; int32 var_4428 = const()[name = string("op_4428"), val = int32(-1)]; bool var_4429_interleave_0 = const()[name = string("op_4429_interleave_0"), val = bool(false)]; tensor var_4429 = concat(axis = var_4428, interleave = var_4429_interleave_0, values = (var_4421, var_4426))[name = string("op_4429")]; tensor var_4430 = mul(x = var_4429, y = sin_1)[name = string("op_4430")]; tensor query_13 = add(x = var_4415, y = var_4430)[name = string("query_13")]; tensor var_4433 = mul(x = k_13, y = cos_1)[name = string("op_4433")]; tensor var_4438_begin_0 = const()[name = string("op_4438_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4438_end_0 = const()[name = string("op_4438_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_4438_end_mask_0 = const()[name = string("op_4438_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4438 = slice_by_index(begin = var_4438_begin_0, end = var_4438_end_0, end_mask = var_4438_end_mask_0, x = k_13)[name = string("op_4438")]; fp16 const_100_promoted = const()[name = string("const_100_promoted"), val = fp16(-0x1p+0)]; tensor var_4439 = mul(x = var_4438, y = const_100_promoted)[name = string("op_4439")]; tensor var_4444_begin_0 = const()[name = string("op_4444_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4444_end_0 = const()[name = string("op_4444_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_4444_end_mask_0 = const()[name = string("op_4444_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4444 = slice_by_index(begin = var_4444_begin_0, end = var_4444_end_0, end_mask = var_4444_end_mask_0, x = k_13)[name = string("op_4444")]; int32 var_4446 = const()[name = string("op_4446"), val = int32(-1)]; bool var_4447_interleave_0 = const()[name = string("op_4447_interleave_0"), val = bool(false)]; tensor var_4447 = concat(axis = var_4446, interleave = var_4447_interleave_0, values = (var_4439, var_4444))[name = string("op_4447")]; tensor var_4448 = mul(x = var_4447, y = sin_1)[name = string("op_4448")]; tensor key_13 = add(x = var_4433, y = var_4448)[name = string("key_13")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_110")]; tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_76, concat_111_values1_0, var_1746, concat_111_values3_0))[name = string("concat_111")]; tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_13, x = coreml_update_state_67)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_180_write_state")]; tensor coreml_update_state_68 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_180")]; tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([34])]; tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([35])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_114")]; tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_82, concat_115_values1_0, var_1746, concat_115_values3_0))[name = string("concat_115")]; tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_61 = transpose(perm = var_4357, x = var_4352)[name = string("transpose_194")]; tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = value_61, x = coreml_update_state_68)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_181_write_state")]; tensor coreml_update_state_69 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_181")]; tensor var_4519_begin_0 = const()[name = string("op_4519_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_4519_end_0 = const()[name = string("op_4519_end_0"), val = tensor([7, 8, 1536, 128])]; tensor var_4519_end_mask_0 = const()[name = string("op_4519_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4519_cast_fp16 = slice_by_index(begin = var_4519_begin_0, end = var_4519_end_0, end_mask = var_4519_end_mask_0, x = coreml_update_state_69)[name = string("op_4519_cast_fp16")]; tensor key_cache_13_axes_0 = const()[name = string("key_cache_13_axes_0"), val = tensor([0])]; tensor key_cache_13_cast_fp16 = squeeze(axes = key_cache_13_axes_0, x = var_4519_cast_fp16)[name = string("key_cache_13_cast_fp16")]; tensor var_4526_begin_0 = const()[name = string("op_4526_begin_0"), val = tensor([34, 0, 0, 0])]; tensor var_4526_end_0 = const()[name = string("op_4526_end_0"), val = tensor([35, 8, 1536, 128])]; tensor var_4526_end_mask_0 = const()[name = string("op_4526_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4526_cast_fp16 = slice_by_index(begin = var_4526_begin_0, end = var_4526_end_0, end_mask = var_4526_end_mask_0, x = coreml_update_state_69)[name = string("op_4526_cast_fp16")]; tensor value_cache_13_axes_0 = const()[name = string("value_cache_13_axes_0"), val = tensor([0])]; tensor value_cache_13_cast_fp16 = squeeze(axes = value_cache_13_axes_0, x = var_4526_cast_fp16)[name = string("value_cache_13_cast_fp16")]; tensor var_4550_axes_0 = const()[name = string("op_4550_axes_0"), val = tensor([1])]; tensor var_4550_cast_fp16 = expand_dims(axes = var_4550_axes_0, x = key_cache_13_cast_fp16)[name = string("op_4550_cast_fp16")]; tensor var_4555 = const()[name = string("op_4555"), val = tensor([1, 2, 1, 1])]; tensor value_65_cast_fp16 = tile(reps = var_4555, x = var_4550_cast_fp16)[name = string("value_65_cast_fp16")]; tensor var_4561 = const()[name = string("op_4561"), val = tensor([1, 16, 1536, 128])]; tensor key_states_27_cast_fp16 = reshape(shape = var_4561, x = value_65_cast_fp16)[name = string("key_states_27_cast_fp16")]; tensor var_4564_axes_0 = const()[name = string("op_4564_axes_0"), val = tensor([1])]; tensor var_4564_cast_fp16 = expand_dims(axes = var_4564_axes_0, x = value_cache_13_cast_fp16)[name = string("op_4564_cast_fp16")]; tensor var_4569 = const()[name = string("op_4569"), val = tensor([1, 2, 1, 1])]; tensor value_69_cast_fp16 = tile(reps = var_4569, x = var_4564_cast_fp16)[name = string("value_69_cast_fp16")]; bool var_4590_transpose_x_0 = const()[name = string("op_4590_transpose_x_0"), val = bool(false)]; bool var_4590_transpose_y_0 = const()[name = string("op_4590_transpose_y_0"), val = bool(true)]; tensor var_4590 = matmul(transpose_x = var_4590_transpose_x_0, transpose_y = var_4590_transpose_y_0, x = query_13, y = key_states_27_cast_fp16)[name = string("op_4590")]; fp16 var_4591_to_fp16 = const()[name = string("op_4591_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_25_cast_fp16 = mul(x = var_4590, y = var_4591_to_fp16)[name = string("attention_25_cast_fp16")]; tensor attention_27_cast_fp16 = add(x = attention_25_cast_fp16, y = causal_mask)[name = string("attention_27_cast_fp16")]; int32 var_4600 = const()[name = string("op_4600"), val = int32(-1)]; tensor var_4602_cast_fp16 = softmax(axis = var_4600, x = attention_27_cast_fp16)[name = string("op_4602_cast_fp16")]; tensor concat_120 = const()[name = string("concat_120"), val = tensor([16, 64, 1536])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_120, x = var_4602_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor concat_121 = const()[name = string("concat_121"), val = tensor([16, 1536, 128])]; tensor reshape_19_cast_fp16 = reshape(shape = concat_121, x = value_69_cast_fp16)[name = string("reshape_19_cast_fp16")]; bool matmul_6_transpose_x_0 = const()[name = string("matmul_6_transpose_x_0"), val = bool(false)]; bool matmul_6_transpose_y_0 = const()[name = string("matmul_6_transpose_y_0"), val = bool(false)]; tensor matmul_6_cast_fp16 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = reshape_18_cast_fp16, y = reshape_19_cast_fp16)[name = string("matmul_6_cast_fp16")]; tensor concat_125 = const()[name = string("concat_125"), val = tensor([1, 16, 64, 128])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_125, x = matmul_6_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor var_4614_perm_0 = const()[name = string("op_4614_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4620 = const()[name = string("op_4620"), val = tensor([1, 64, 2048])]; tensor var_4614_cast_fp16 = transpose(perm = var_4614_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_193")]; tensor output_39_cast_fp16 = reshape(shape = var_4620, x = var_4614_cast_fp16)[name = string("output_39_cast_fp16")]; tensor var_4625 = const()[name = string("op_4625"), val = tensor([0, 2, 1])]; string var_4641_pad_type_0 = const()[name = string("op_4641_pad_type_0"), val = string("valid")]; int32 var_4641_groups_0 = const()[name = string("op_4641_groups_0"), val = int32(1)]; tensor var_4641_strides_0 = const()[name = string("op_4641_strides_0"), val = tensor([1])]; tensor var_4641_pad_0 = const()[name = string("op_4641_pad_0"), val = tensor([0, 0])]; tensor var_4641_dilations_0 = const()[name = string("op_4641_dilations_0"), val = tensor([1])]; tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(302470272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304043200))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_4626_cast_fp16 = transpose(perm = var_4625, x = output_39_cast_fp16)[name = string("transpose_192")]; tensor var_4641_cast_fp16 = conv(dilations = var_4641_dilations_0, groups = var_4641_groups_0, pad = var_4641_pad_0, pad_type = var_4641_pad_type_0, strides = var_4641_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4626_cast_fp16)[name = string("op_4641_cast_fp16")]; tensor var_4645 = const()[name = string("op_4645"), val = tensor([0, 2, 1])]; tensor attn_output_13_cast_fp16 = transpose(perm = var_4645, x = var_4641_cast_fp16)[name = string("transpose_191")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = attn_output_13_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; int32 var_4660 = const()[name = string("op_4660"), val = int32(-1)]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4662_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_102_promoted_to_fp16)[name = string("op_4662_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_4660, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_4662_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_4657_to_fp16 = const()[name = string("op_4657_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_4657_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 64, 1024])]; tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; tensor const_104_promoted_to_fp16 = const()[name = string("const_104_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304059648)))]; tensor x_25_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_104_promoted_to_fp16)[name = string("x_25_cast_fp16")]; tensor var_4682 = const()[name = string("op_4682"), val = tensor([0, 2, 1])]; tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; tensor var_4683 = transpose(perm = var_4682, x = x_25_cast_fp16)[name = string("transpose_190")]; tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4683)[name = string("input_121")]; string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_6_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_6_mlp_up_proj_weight_palettized, x = input_121)[name = string("b_13")]; tensor c_13 = silu(x = input_123)[name = string("c_13")]; tensor input_125 = mul(x = c_13, y = b_13)[name = string("input_125")]; string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_6_mlp_down_proj_weight_palettized, x = input_125)[name = string("e_13")]; tensor var_4705_axes_0 = const()[name = string("op_4705_axes_0"), val = tensor([2])]; tensor var_4705 = squeeze(axes = var_4705_axes_0, x = e_13)[name = string("op_4705")]; tensor var_4706 = const()[name = string("op_4706"), val = tensor([0, 2, 1])]; tensor var_4707 = transpose(perm = var_4706, x = var_4705)[name = string("transpose_189")]; tensor hidden_states_71_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_4707)[name = string("hidden_states_71_cast_fp16")]; int32 var_4721 = const()[name = string("op_4721"), val = int32(-1)]; fp16 const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4723_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_105_promoted_to_fp16)[name = string("op_4723_cast_fp16")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127_cast_fp16 = concat(axis = var_4721, interleave = input_127_interleave_0, values = (hidden_states_71_cast_fp16, var_4723_cast_fp16))[name = string("input_127_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_4718_to_fp16 = const()[name = string("op_4718_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4718_to_fp16, x = input_127_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 64, 1024])]; tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; tensor const_107_promoted_to_fp16 = const()[name = string("const_107_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304061760)))]; tensor hidden_states_73_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_107_promoted_to_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor var_4735 = const()[name = string("op_4735"), val = tensor([0, 2, 1])]; tensor var_4738_axes_0 = const()[name = string("op_4738_axes_0"), val = tensor([2])]; tensor var_4736_cast_fp16 = transpose(perm = var_4735, x = hidden_states_73_cast_fp16)[name = string("transpose_188")]; tensor var_4738_cast_fp16 = expand_dims(axes = var_4738_axes_0, x = var_4736_cast_fp16)[name = string("op_4738_cast_fp16")]; string var_4754_pad_type_0 = const()[name = string("op_4754_pad_type_0"), val = string("valid")]; tensor var_4754_strides_0 = const()[name = string("op_4754_strides_0"), val = tensor([1, 1])]; tensor var_4754_pad_0 = const()[name = string("op_4754_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4754_dilations_0 = const()[name = string("op_4754_dilations_0"), val = tensor([1, 1])]; int32 var_4754_groups_0 = const()[name = string("op_4754_groups_0"), val = int32(1)]; tensor var_4754 = conv(dilations = var_4754_dilations_0, groups = var_4754_groups_0, pad = var_4754_pad_0, pad_type = var_4754_pad_type_0, strides = var_4754_strides_0, weight = model_model_layers_7_self_attn_q_proj_weight_palettized, x = var_4738_cast_fp16)[name = string("op_4754")]; tensor var_4759 = const()[name = string("op_4759"), val = tensor([1, 16, 128, 64])]; tensor var_4760 = reshape(shape = var_4759, x = var_4754)[name = string("op_4760")]; tensor var_4765 = const()[name = string("op_4765"), val = tensor([0, 1, 3, 2])]; string var_4777_pad_type_0 = const()[name = string("op_4777_pad_type_0"), val = string("valid")]; tensor var_4777_strides_0 = const()[name = string("op_4777_strides_0"), val = tensor([1, 1])]; tensor var_4777_pad_0 = const()[name = string("op_4777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4777_dilations_0 = const()[name = string("op_4777_dilations_0"), val = tensor([1, 1])]; int32 var_4777_groups_0 = const()[name = string("op_4777_groups_0"), val = int32(1)]; tensor var_4777 = conv(dilations = var_4777_dilations_0, groups = var_4777_groups_0, pad = var_4777_pad_0, pad_type = var_4777_pad_type_0, strides = var_4777_strides_0, weight = model_model_layers_7_self_attn_k_proj_weight_palettized, x = var_4738_cast_fp16)[name = string("op_4777")]; tensor var_4782 = const()[name = string("op_4782"), val = tensor([1, 8, 128, 64])]; tensor var_4783 = reshape(shape = var_4782, x = var_4777)[name = string("op_4783")]; tensor var_4788 = const()[name = string("op_4788"), val = tensor([0, 1, 3, 2])]; string var_4800_pad_type_0 = const()[name = string("op_4800_pad_type_0"), val = string("valid")]; tensor var_4800_strides_0 = const()[name = string("op_4800_strides_0"), val = tensor([1, 1])]; tensor var_4800_pad_0 = const()[name = string("op_4800_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4800_dilations_0 = const()[name = string("op_4800_dilations_0"), val = tensor([1, 1])]; int32 var_4800_groups_0 = const()[name = string("op_4800_groups_0"), val = int32(1)]; tensor var_4800 = conv(dilations = var_4800_dilations_0, groups = var_4800_groups_0, pad = var_4800_pad_0, pad_type = var_4800_pad_type_0, strides = var_4800_strides_0, weight = model_model_layers_7_self_attn_v_proj_weight_palettized, x = var_4738_cast_fp16)[name = string("op_4800")]; tensor var_4805 = const()[name = string("op_4805"), val = tensor([1, 8, 128, 64])]; tensor var_4806 = reshape(shape = var_4805, x = var_4800)[name = string("op_4806")]; tensor var_4811 = const()[name = string("op_4811"), val = tensor([0, 1, 3, 2])]; int32 var_4824 = const()[name = string("op_4824"), val = int32(-1)]; fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_75 = transpose(perm = var_4765, x = var_4760)[name = string("transpose_187")]; tensor var_4826 = mul(x = hidden_states_75, y = const_108_promoted)[name = string("op_4826")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131 = concat(axis = var_4824, interleave = input_131_interleave_0, values = (hidden_states_75, var_4826))[name = string("input_131")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_4821_to_fp16 = const()[name = string("op_4821_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4821_to_fp16, x = input_131)[name = string("normed_117_cast_fp16")]; tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_119 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119")]; tensor const_110 = const()[name = string("const_110"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304063872)))]; tensor q_15 = mul(x = normed_119, y = const_110)[name = string("q_15")]; int32 var_4846 = const()[name = string("op_4846"), val = int32(-1)]; fp16 const_111_promoted = const()[name = string("const_111_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_77 = transpose(perm = var_4788, x = var_4783)[name = string("transpose_186")]; tensor var_4848 = mul(x = hidden_states_77, y = const_111_promoted)[name = string("op_4848")]; bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; tensor input_133 = concat(axis = var_4846, interleave = input_133_interleave_0, values = (hidden_states_77, var_4848))[name = string("input_133")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_4843_to_fp16 = const()[name = string("op_4843_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4843_to_fp16, x = input_133)[name = string("normed_121_cast_fp16")]; tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_123 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123")]; tensor const_113 = const()[name = string("const_113"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304064192)))]; tensor k_15 = mul(x = normed_123, y = const_113)[name = string("k_15")]; tensor var_4869 = mul(x = q_15, y = cos_1)[name = string("op_4869")]; tensor var_4874_begin_0 = const()[name = string("op_4874_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4874_end_0 = const()[name = string("op_4874_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_4874_end_mask_0 = const()[name = string("op_4874_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4874 = slice_by_index(begin = var_4874_begin_0, end = var_4874_end_0, end_mask = var_4874_end_mask_0, x = q_15)[name = string("op_4874")]; fp16 const_114_promoted = const()[name = string("const_114_promoted"), val = fp16(-0x1p+0)]; tensor var_4875 = mul(x = var_4874, y = const_114_promoted)[name = string("op_4875")]; tensor var_4880_begin_0 = const()[name = string("op_4880_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4880_end_0 = const()[name = string("op_4880_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_4880_end_mask_0 = const()[name = string("op_4880_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4880 = slice_by_index(begin = var_4880_begin_0, end = var_4880_end_0, end_mask = var_4880_end_mask_0, x = q_15)[name = string("op_4880")]; int32 var_4882 = const()[name = string("op_4882"), val = int32(-1)]; bool var_4883_interleave_0 = const()[name = string("op_4883_interleave_0"), val = bool(false)]; tensor var_4883 = concat(axis = var_4882, interleave = var_4883_interleave_0, values = (var_4875, var_4880))[name = string("op_4883")]; tensor var_4884 = mul(x = var_4883, y = sin_1)[name = string("op_4884")]; tensor query_15 = add(x = var_4869, y = var_4884)[name = string("query_15")]; tensor var_4887 = mul(x = k_15, y = cos_1)[name = string("op_4887")]; tensor var_4892_begin_0 = const()[name = string("op_4892_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_4892_end_0 = const()[name = string("op_4892_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_4892_end_mask_0 = const()[name = string("op_4892_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_4892 = slice_by_index(begin = var_4892_begin_0, end = var_4892_end_0, end_mask = var_4892_end_mask_0, x = k_15)[name = string("op_4892")]; fp16 const_115_promoted = const()[name = string("const_115_promoted"), val = fp16(-0x1p+0)]; tensor var_4893 = mul(x = var_4892, y = const_115_promoted)[name = string("op_4893")]; tensor var_4898_begin_0 = const()[name = string("op_4898_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_4898_end_0 = const()[name = string("op_4898_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_4898_end_mask_0 = const()[name = string("op_4898_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_4898 = slice_by_index(begin = var_4898_begin_0, end = var_4898_end_0, end_mask = var_4898_end_mask_0, x = k_15)[name = string("op_4898")]; int32 var_4900 = const()[name = string("op_4900"), val = int32(-1)]; bool var_4901_interleave_0 = const()[name = string("op_4901_interleave_0"), val = bool(false)]; tensor var_4901 = concat(axis = var_4900, interleave = var_4901_interleave_0, values = (var_4893, var_4898))[name = string("op_4901")]; tensor var_4902 = mul(x = var_4901, y = sin_1)[name = string("op_4902")]; tensor key_15 = add(x = var_4887, y = var_4902)[name = string("key_15")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; int32 concat_128_axis_0 = const()[name = string("concat_128_axis_0"), val = int32(0)]; bool concat_128_interleave_0 = const()[name = string("concat_128_interleave_0"), val = bool(false)]; tensor concat_128 = concat(axis = concat_128_axis_0, interleave = concat_128_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_128")]; tensor concat_129_values1_0 = const()[name = string("concat_129_values1_0"), val = tensor([0])]; tensor concat_129_values3_0 = const()[name = string("concat_129_values3_0"), val = tensor([0])]; int32 concat_129_axis_0 = const()[name = string("concat_129_axis_0"), val = int32(0)]; bool concat_129_interleave_0 = const()[name = string("concat_129_interleave_0"), val = bool(false)]; tensor concat_129 = concat(axis = concat_129_axis_0, interleave = concat_129_interleave_0, values = (expand_dims_88, concat_129_values1_0, var_1746, concat_129_values3_0))[name = string("concat_129")]; tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_128, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_129, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_15, x = coreml_update_state_69)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_182_write_state")]; tensor coreml_update_state_70 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_182")]; tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([35])]; tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([36])]; int32 concat_132_axis_0 = const()[name = string("concat_132_axis_0"), val = int32(0)]; bool concat_132_interleave_0 = const()[name = string("concat_132_interleave_0"), val = bool(false)]; tensor concat_132 = concat(axis = concat_132_axis_0, interleave = concat_132_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_132")]; tensor concat_133_values1_0 = const()[name = string("concat_133_values1_0"), val = tensor([0])]; tensor concat_133_values3_0 = const()[name = string("concat_133_values3_0"), val = tensor([0])]; int32 concat_133_axis_0 = const()[name = string("concat_133_axis_0"), val = int32(0)]; bool concat_133_interleave_0 = const()[name = string("concat_133_interleave_0"), val = bool(false)]; tensor concat_133 = concat(axis = concat_133_axis_0, interleave = concat_133_interleave_0, values = (expand_dims_94, concat_133_values1_0, var_1746, concat_133_values3_0))[name = string("concat_133")]; tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_71 = transpose(perm = var_4811, x = var_4806)[name = string("transpose_185")]; tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_132, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_133, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = value_71, x = coreml_update_state_70)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_183_write_state")]; tensor coreml_update_state_71 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_183")]; tensor var_4973_begin_0 = const()[name = string("op_4973_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_4973_end_0 = const()[name = string("op_4973_end_0"), val = tensor([8, 8, 1536, 128])]; tensor var_4973_end_mask_0 = const()[name = string("op_4973_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4973_cast_fp16 = slice_by_index(begin = var_4973_begin_0, end = var_4973_end_0, end_mask = var_4973_end_mask_0, x = coreml_update_state_71)[name = string("op_4973_cast_fp16")]; tensor key_cache_15_axes_0 = const()[name = string("key_cache_15_axes_0"), val = tensor([0])]; tensor key_cache_15_cast_fp16 = squeeze(axes = key_cache_15_axes_0, x = var_4973_cast_fp16)[name = string("key_cache_15_cast_fp16")]; tensor var_4980_begin_0 = const()[name = string("op_4980_begin_0"), val = tensor([35, 0, 0, 0])]; tensor var_4980_end_0 = const()[name = string("op_4980_end_0"), val = tensor([36, 8, 1536, 128])]; tensor var_4980_end_mask_0 = const()[name = string("op_4980_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4980_cast_fp16 = slice_by_index(begin = var_4980_begin_0, end = var_4980_end_0, end_mask = var_4980_end_mask_0, x = coreml_update_state_71)[name = string("op_4980_cast_fp16")]; tensor value_cache_15_axes_0 = const()[name = string("value_cache_15_axes_0"), val = tensor([0])]; tensor value_cache_15_cast_fp16 = squeeze(axes = value_cache_15_axes_0, x = var_4980_cast_fp16)[name = string("value_cache_15_cast_fp16")]; tensor var_5004_axes_0 = const()[name = string("op_5004_axes_0"), val = tensor([1])]; tensor var_5004_cast_fp16 = expand_dims(axes = var_5004_axes_0, x = key_cache_15_cast_fp16)[name = string("op_5004_cast_fp16")]; tensor var_5009 = const()[name = string("op_5009"), val = tensor([1, 2, 1, 1])]; tensor value_75_cast_fp16 = tile(reps = var_5009, x = var_5004_cast_fp16)[name = string("value_75_cast_fp16")]; tensor var_5015 = const()[name = string("op_5015"), val = tensor([1, 16, 1536, 128])]; tensor key_states_31_cast_fp16 = reshape(shape = var_5015, x = value_75_cast_fp16)[name = string("key_states_31_cast_fp16")]; tensor var_5018_axes_0 = const()[name = string("op_5018_axes_0"), val = tensor([1])]; tensor var_5018_cast_fp16 = expand_dims(axes = var_5018_axes_0, x = value_cache_15_cast_fp16)[name = string("op_5018_cast_fp16")]; tensor var_5023 = const()[name = string("op_5023"), val = tensor([1, 2, 1, 1])]; tensor value_79_cast_fp16 = tile(reps = var_5023, x = var_5018_cast_fp16)[name = string("value_79_cast_fp16")]; bool var_5044_transpose_x_0 = const()[name = string("op_5044_transpose_x_0"), val = bool(false)]; bool var_5044_transpose_y_0 = const()[name = string("op_5044_transpose_y_0"), val = bool(true)]; tensor var_5044 = matmul(transpose_x = var_5044_transpose_x_0, transpose_y = var_5044_transpose_y_0, x = query_15, y = key_states_31_cast_fp16)[name = string("op_5044")]; fp16 var_5045_to_fp16 = const()[name = string("op_5045_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_29_cast_fp16 = mul(x = var_5044, y = var_5045_to_fp16)[name = string("attention_29_cast_fp16")]; tensor attention_31_cast_fp16 = add(x = attention_29_cast_fp16, y = causal_mask)[name = string("attention_31_cast_fp16")]; int32 var_5054 = const()[name = string("op_5054"), val = int32(-1)]; tensor var_5056_cast_fp16 = softmax(axis = var_5054, x = attention_31_cast_fp16)[name = string("op_5056_cast_fp16")]; tensor concat_138 = const()[name = string("concat_138"), val = tensor([16, 64, 1536])]; tensor reshape_21_cast_fp16 = reshape(shape = concat_138, x = var_5056_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor concat_139 = const()[name = string("concat_139"), val = tensor([16, 1536, 128])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_139, x = value_79_cast_fp16)[name = string("reshape_22_cast_fp16")]; bool matmul_7_transpose_x_0 = const()[name = string("matmul_7_transpose_x_0"), val = bool(false)]; bool matmul_7_transpose_y_0 = const()[name = string("matmul_7_transpose_y_0"), val = bool(false)]; tensor matmul_7_cast_fp16 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = reshape_21_cast_fp16, y = reshape_22_cast_fp16)[name = string("matmul_7_cast_fp16")]; tensor concat_143 = const()[name = string("concat_143"), val = tensor([1, 16, 64, 128])]; tensor reshape_23_cast_fp16 = reshape(shape = concat_143, x = matmul_7_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor var_5068_perm_0 = const()[name = string("op_5068_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_5074 = const()[name = string("op_5074"), val = tensor([1, 64, 2048])]; tensor var_5068_cast_fp16 = transpose(perm = var_5068_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_184")]; tensor output_45_cast_fp16 = reshape(shape = var_5074, x = var_5068_cast_fp16)[name = string("output_45_cast_fp16")]; tensor var_5079 = const()[name = string("op_5079"), val = tensor([0, 2, 1])]; string var_5095_pad_type_0 = const()[name = string("op_5095_pad_type_0"), val = string("valid")]; int32 var_5095_groups_0 = const()[name = string("op_5095_groups_0"), val = int32(1)]; tensor var_5095_strides_0 = const()[name = string("op_5095_strides_0"), val = tensor([1])]; tensor var_5095_pad_0 = const()[name = string("op_5095_pad_0"), val = tensor([0, 0])]; tensor var_5095_dilations_0 = const()[name = string("op_5095_dilations_0"), val = tensor([1])]; tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304064512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305637440))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5080_cast_fp16 = transpose(perm = var_5079, x = output_45_cast_fp16)[name = string("transpose_183")]; tensor var_5095_cast_fp16 = conv(dilations = var_5095_dilations_0, groups = var_5095_groups_0, pad = var_5095_pad_0, pad_type = var_5095_pad_type_0, strides = var_5095_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_5080_cast_fp16)[name = string("op_5095_cast_fp16")]; tensor var_5099 = const()[name = string("op_5099"), val = tensor([0, 2, 1])]; tensor attn_output_15_cast_fp16 = transpose(perm = var_5099, x = var_5095_cast_fp16)[name = string("transpose_182")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_71_cast_fp16, y = attn_output_15_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; int32 var_5114 = const()[name = string("op_5114"), val = int32(-1)]; fp16 const_117_promoted_to_fp16 = const()[name = string("const_117_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5116_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_117_promoted_to_fp16)[name = string("op_5116_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_5114, interleave = input_137_interleave_0, values = (hidden_states_79_cast_fp16, var_5116_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_5111_to_fp16 = const()[name = string("op_5111_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_5111_to_fp16, x = input_137_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 64, 1024])]; tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; tensor const_119_promoted_to_fp16 = const()[name = string("const_119_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305653888)))]; tensor x_29_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_119_promoted_to_fp16)[name = string("x_29_cast_fp16")]; tensor var_5136 = const()[name = string("op_5136"), val = tensor([0, 2, 1])]; tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; tensor var_5137 = transpose(perm = var_5136, x = x_29_cast_fp16)[name = string("transpose_181")]; tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_5137)[name = string("input_139")]; string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; tensor input_141 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = model_model_layers_7_mlp_gate_proj_weight_palettized, x = input_139)[name = string("input_141")]; string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_7_mlp_up_proj_weight_palettized, x = input_139)[name = string("b_15")]; tensor c_15 = silu(x = input_141)[name = string("c_15")]; tensor input_143 = mul(x = c_15, y = b_15)[name = string("input_143")]; string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_7_mlp_down_proj_weight_palettized, x = input_143)[name = string("e_15")]; tensor var_5159_axes_0 = const()[name = string("op_5159_axes_0"), val = tensor([2])]; tensor var_5159 = squeeze(axes = var_5159_axes_0, x = e_15)[name = string("op_5159")]; tensor var_5160 = const()[name = string("op_5160"), val = tensor([0, 2, 1])]; tensor var_5161 = transpose(perm = var_5160, x = var_5159)[name = string("transpose_180")]; tensor hidden_states_81_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = var_5161)[name = string("hidden_states_81_cast_fp16")]; int32 var_5175 = const()[name = string("op_5175"), val = int32(-1)]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5177_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_5177_cast_fp16")]; bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; tensor input_145_cast_fp16 = concat(axis = var_5175, interleave = input_145_interleave_0, values = (hidden_states_81_cast_fp16, var_5177_cast_fp16))[name = string("input_145_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_5172_to_fp16 = const()[name = string("op_5172_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_5172_to_fp16, x = input_145_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor normed_131_begin_0 = const()[name = string("normed_131_begin_0"), val = tensor([0, 0, 0])]; tensor normed_131_end_0 = const()[name = string("normed_131_end_0"), val = tensor([1, 64, 1024])]; tensor normed_131_end_mask_0 = const()[name = string("normed_131_end_mask_0"), val = tensor([true, true, false])]; tensor normed_131_cast_fp16 = slice_by_index(begin = normed_131_begin_0, end = normed_131_end_0, end_mask = normed_131_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_131_cast_fp16")]; tensor const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305656000)))]; tensor hidden_states_83_cast_fp16 = mul(x = normed_131_cast_fp16, y = const_122_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor var_5189 = const()[name = string("op_5189"), val = tensor([0, 2, 1])]; tensor var_5192_axes_0 = const()[name = string("op_5192_axes_0"), val = tensor([2])]; tensor var_5190_cast_fp16 = transpose(perm = var_5189, x = hidden_states_83_cast_fp16)[name = string("transpose_179")]; tensor var_5192_cast_fp16 = expand_dims(axes = var_5192_axes_0, x = var_5190_cast_fp16)[name = string("op_5192_cast_fp16")]; string var_5208_pad_type_0 = const()[name = string("op_5208_pad_type_0"), val = string("valid")]; tensor var_5208_strides_0 = const()[name = string("op_5208_strides_0"), val = tensor([1, 1])]; tensor var_5208_pad_0 = const()[name = string("op_5208_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5208_dilations_0 = const()[name = string("op_5208_dilations_0"), val = tensor([1, 1])]; int32 var_5208_groups_0 = const()[name = string("op_5208_groups_0"), val = int32(1)]; tensor var_5208 = conv(dilations = var_5208_dilations_0, groups = var_5208_groups_0, pad = var_5208_pad_0, pad_type = var_5208_pad_type_0, strides = var_5208_strides_0, weight = model_model_layers_8_self_attn_q_proj_weight_palettized, x = var_5192_cast_fp16)[name = string("op_5208")]; tensor var_5213 = const()[name = string("op_5213"), val = tensor([1, 16, 128, 64])]; tensor var_5214 = reshape(shape = var_5213, x = var_5208)[name = string("op_5214")]; tensor var_5219 = const()[name = string("op_5219"), val = tensor([0, 1, 3, 2])]; string var_5231_pad_type_0 = const()[name = string("op_5231_pad_type_0"), val = string("valid")]; tensor var_5231_strides_0 = const()[name = string("op_5231_strides_0"), val = tensor([1, 1])]; tensor var_5231_pad_0 = const()[name = string("op_5231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5231_dilations_0 = const()[name = string("op_5231_dilations_0"), val = tensor([1, 1])]; int32 var_5231_groups_0 = const()[name = string("op_5231_groups_0"), val = int32(1)]; tensor var_5231 = conv(dilations = var_5231_dilations_0, groups = var_5231_groups_0, pad = var_5231_pad_0, pad_type = var_5231_pad_type_0, strides = var_5231_strides_0, weight = model_model_layers_8_self_attn_k_proj_weight_palettized, x = var_5192_cast_fp16)[name = string("op_5231")]; tensor var_5236 = const()[name = string("op_5236"), val = tensor([1, 8, 128, 64])]; tensor var_5237 = reshape(shape = var_5236, x = var_5231)[name = string("op_5237")]; tensor var_5242 = const()[name = string("op_5242"), val = tensor([0, 1, 3, 2])]; string var_5254_pad_type_0 = const()[name = string("op_5254_pad_type_0"), val = string("valid")]; tensor var_5254_strides_0 = const()[name = string("op_5254_strides_0"), val = tensor([1, 1])]; tensor var_5254_pad_0 = const()[name = string("op_5254_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5254_dilations_0 = const()[name = string("op_5254_dilations_0"), val = tensor([1, 1])]; int32 var_5254_groups_0 = const()[name = string("op_5254_groups_0"), val = int32(1)]; tensor var_5254 = conv(dilations = var_5254_dilations_0, groups = var_5254_groups_0, pad = var_5254_pad_0, pad_type = var_5254_pad_type_0, strides = var_5254_strides_0, weight = model_model_layers_8_self_attn_v_proj_weight_palettized, x = var_5192_cast_fp16)[name = string("op_5254")]; tensor var_5259 = const()[name = string("op_5259"), val = tensor([1, 8, 128, 64])]; tensor var_5260 = reshape(shape = var_5259, x = var_5254)[name = string("op_5260")]; tensor var_5265 = const()[name = string("op_5265"), val = tensor([0, 1, 3, 2])]; int32 var_5278 = const()[name = string("op_5278"), val = int32(-1)]; fp16 const_123_promoted = const()[name = string("const_123_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_85 = transpose(perm = var_5219, x = var_5214)[name = string("transpose_178")]; tensor var_5280 = mul(x = hidden_states_85, y = const_123_promoted)[name = string("op_5280")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149 = concat(axis = var_5278, interleave = input_149_interleave_0, values = (hidden_states_85, var_5280))[name = string("input_149")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_5275_to_fp16 = const()[name = string("op_5275_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_5275_to_fp16, x = input_149)[name = string("normed_133_cast_fp16")]; tensor normed_135_begin_0 = const()[name = string("normed_135_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_135_end_0 = const()[name = string("normed_135_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_135_end_mask_0 = const()[name = string("normed_135_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_135 = slice_by_index(begin = normed_135_begin_0, end = normed_135_end_0, end_mask = normed_135_end_mask_0, x = normed_133_cast_fp16)[name = string("normed_135")]; tensor const_125 = const()[name = string("const_125"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305658112)))]; tensor q_17 = mul(x = normed_135, y = const_125)[name = string("q_17")]; int32 var_5300 = const()[name = string("op_5300"), val = int32(-1)]; fp16 const_126_promoted = const()[name = string("const_126_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_87 = transpose(perm = var_5242, x = var_5237)[name = string("transpose_177")]; tensor var_5302 = mul(x = hidden_states_87, y = const_126_promoted)[name = string("op_5302")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151 = concat(axis = var_5300, interleave = input_151_interleave_0, values = (hidden_states_87, var_5302))[name = string("input_151")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_5297_to_fp16 = const()[name = string("op_5297_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_5297_to_fp16, x = input_151)[name = string("normed_137_cast_fp16")]; tensor normed_139_begin_0 = const()[name = string("normed_139_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_139_end_0 = const()[name = string("normed_139_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_139_end_mask_0 = const()[name = string("normed_139_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_139 = slice_by_index(begin = normed_139_begin_0, end = normed_139_end_0, end_mask = normed_139_end_mask_0, x = normed_137_cast_fp16)[name = string("normed_139")]; tensor const_128 = const()[name = string("const_128"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305658432)))]; tensor k_17 = mul(x = normed_139, y = const_128)[name = string("k_17")]; tensor var_5323 = mul(x = q_17, y = cos_1)[name = string("op_5323")]; tensor var_5328_begin_0 = const()[name = string("op_5328_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5328_end_0 = const()[name = string("op_5328_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_5328_end_mask_0 = const()[name = string("op_5328_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5328 = slice_by_index(begin = var_5328_begin_0, end = var_5328_end_0, end_mask = var_5328_end_mask_0, x = q_17)[name = string("op_5328")]; fp16 const_129_promoted = const()[name = string("const_129_promoted"), val = fp16(-0x1p+0)]; tensor var_5329 = mul(x = var_5328, y = const_129_promoted)[name = string("op_5329")]; tensor var_5334_begin_0 = const()[name = string("op_5334_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5334_end_0 = const()[name = string("op_5334_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_5334_end_mask_0 = const()[name = string("op_5334_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5334 = slice_by_index(begin = var_5334_begin_0, end = var_5334_end_0, end_mask = var_5334_end_mask_0, x = q_17)[name = string("op_5334")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(-1)]; bool var_5337_interleave_0 = const()[name = string("op_5337_interleave_0"), val = bool(false)]; tensor var_5337 = concat(axis = var_5336, interleave = var_5337_interleave_0, values = (var_5329, var_5334))[name = string("op_5337")]; tensor var_5338 = mul(x = var_5337, y = sin_1)[name = string("op_5338")]; tensor query_17 = add(x = var_5323, y = var_5338)[name = string("query_17")]; tensor var_5341 = mul(x = k_17, y = cos_1)[name = string("op_5341")]; tensor var_5346_begin_0 = const()[name = string("op_5346_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5346_end_0 = const()[name = string("op_5346_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_5346_end_mask_0 = const()[name = string("op_5346_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5346 = slice_by_index(begin = var_5346_begin_0, end = var_5346_end_0, end_mask = var_5346_end_mask_0, x = k_17)[name = string("op_5346")]; fp16 const_130_promoted = const()[name = string("const_130_promoted"), val = fp16(-0x1p+0)]; tensor var_5347 = mul(x = var_5346, y = const_130_promoted)[name = string("op_5347")]; tensor var_5352_begin_0 = const()[name = string("op_5352_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5352_end_0 = const()[name = string("op_5352_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_5352_end_mask_0 = const()[name = string("op_5352_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5352 = slice_by_index(begin = var_5352_begin_0, end = var_5352_end_0, end_mask = var_5352_end_mask_0, x = k_17)[name = string("op_5352")]; int32 var_5354 = const()[name = string("op_5354"), val = int32(-1)]; bool var_5355_interleave_0 = const()[name = string("op_5355_interleave_0"), val = bool(false)]; tensor var_5355 = concat(axis = var_5354, interleave = var_5355_interleave_0, values = (var_5347, var_5352))[name = string("op_5355")]; tensor var_5356 = mul(x = var_5355, y = sin_1)[name = string("op_5356")]; tensor key_17 = add(x = var_5341, y = var_5356)[name = string("key_17")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; int32 concat_146_axis_0 = const()[name = string("concat_146_axis_0"), val = int32(0)]; bool concat_146_interleave_0 = const()[name = string("concat_146_interleave_0"), val = bool(false)]; tensor concat_146 = concat(axis = concat_146_axis_0, interleave = concat_146_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_146")]; tensor concat_147_values1_0 = const()[name = string("concat_147_values1_0"), val = tensor([0])]; tensor concat_147_values3_0 = const()[name = string("concat_147_values3_0"), val = tensor([0])]; int32 concat_147_axis_0 = const()[name = string("concat_147_axis_0"), val = int32(0)]; bool concat_147_interleave_0 = const()[name = string("concat_147_interleave_0"), val = bool(false)]; tensor concat_147 = concat(axis = concat_147_axis_0, interleave = concat_147_interleave_0, values = (expand_dims_100, concat_147_values1_0, var_1746, concat_147_values3_0))[name = string("concat_147")]; tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_146, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_147, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_17, x = coreml_update_state_71)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_184_write_state")]; tensor coreml_update_state_72 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_184")]; tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([36])]; tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([37])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_150")]; tensor concat_151_values1_0 = const()[name = string("concat_151_values1_0"), val = tensor([0])]; tensor concat_151_values3_0 = const()[name = string("concat_151_values3_0"), val = tensor([0])]; int32 concat_151_axis_0 = const()[name = string("concat_151_axis_0"), val = int32(0)]; bool concat_151_interleave_0 = const()[name = string("concat_151_interleave_0"), val = bool(false)]; tensor concat_151 = concat(axis = concat_151_axis_0, interleave = concat_151_interleave_0, values = (expand_dims_106, concat_151_values1_0, var_1746, concat_151_values3_0))[name = string("concat_151")]; tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_81 = transpose(perm = var_5265, x = var_5260)[name = string("transpose_176")]; tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_150, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_151, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = value_81, x = coreml_update_state_72)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_185_write_state")]; tensor coreml_update_state_73 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_185")]; tensor var_5427_begin_0 = const()[name = string("op_5427_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_5427_end_0 = const()[name = string("op_5427_end_0"), val = tensor([9, 8, 1536, 128])]; tensor var_5427_end_mask_0 = const()[name = string("op_5427_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5427_cast_fp16 = slice_by_index(begin = var_5427_begin_0, end = var_5427_end_0, end_mask = var_5427_end_mask_0, x = coreml_update_state_73)[name = string("op_5427_cast_fp16")]; tensor key_cache_17_axes_0 = const()[name = string("key_cache_17_axes_0"), val = tensor([0])]; tensor key_cache_17_cast_fp16 = squeeze(axes = key_cache_17_axes_0, x = var_5427_cast_fp16)[name = string("key_cache_17_cast_fp16")]; tensor var_5434_begin_0 = const()[name = string("op_5434_begin_0"), val = tensor([36, 0, 0, 0])]; tensor var_5434_end_0 = const()[name = string("op_5434_end_0"), val = tensor([37, 8, 1536, 128])]; tensor var_5434_end_mask_0 = const()[name = string("op_5434_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5434_cast_fp16 = slice_by_index(begin = var_5434_begin_0, end = var_5434_end_0, end_mask = var_5434_end_mask_0, x = coreml_update_state_73)[name = string("op_5434_cast_fp16")]; tensor value_cache_17_axes_0 = const()[name = string("value_cache_17_axes_0"), val = tensor([0])]; tensor value_cache_17_cast_fp16 = squeeze(axes = value_cache_17_axes_0, x = var_5434_cast_fp16)[name = string("value_cache_17_cast_fp16")]; tensor var_5458_axes_0 = const()[name = string("op_5458_axes_0"), val = tensor([1])]; tensor var_5458_cast_fp16 = expand_dims(axes = var_5458_axes_0, x = key_cache_17_cast_fp16)[name = string("op_5458_cast_fp16")]; tensor var_5463 = const()[name = string("op_5463"), val = tensor([1, 2, 1, 1])]; tensor value_85_cast_fp16 = tile(reps = var_5463, x = var_5458_cast_fp16)[name = string("value_85_cast_fp16")]; tensor var_5469 = const()[name = string("op_5469"), val = tensor([1, 16, 1536, 128])]; tensor key_states_35_cast_fp16 = reshape(shape = var_5469, x = value_85_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor var_5472_axes_0 = const()[name = string("op_5472_axes_0"), val = tensor([1])]; tensor var_5472_cast_fp16 = expand_dims(axes = var_5472_axes_0, x = value_cache_17_cast_fp16)[name = string("op_5472_cast_fp16")]; tensor var_5477 = const()[name = string("op_5477"), val = tensor([1, 2, 1, 1])]; tensor value_89_cast_fp16 = tile(reps = var_5477, x = var_5472_cast_fp16)[name = string("value_89_cast_fp16")]; bool var_5498_transpose_x_0 = const()[name = string("op_5498_transpose_x_0"), val = bool(false)]; bool var_5498_transpose_y_0 = const()[name = string("op_5498_transpose_y_0"), val = bool(true)]; tensor var_5498 = matmul(transpose_x = var_5498_transpose_x_0, transpose_y = var_5498_transpose_y_0, x = query_17, y = key_states_35_cast_fp16)[name = string("op_5498")]; fp16 var_5499_to_fp16 = const()[name = string("op_5499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_33_cast_fp16 = mul(x = var_5498, y = var_5499_to_fp16)[name = string("attention_33_cast_fp16")]; tensor attention_35_cast_fp16 = add(x = attention_33_cast_fp16, y = causal_mask)[name = string("attention_35_cast_fp16")]; int32 var_5508 = const()[name = string("op_5508"), val = int32(-1)]; tensor var_5510_cast_fp16 = softmax(axis = var_5508, x = attention_35_cast_fp16)[name = string("op_5510_cast_fp16")]; tensor concat_156 = const()[name = string("concat_156"), val = tensor([16, 64, 1536])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_156, x = var_5510_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor concat_157 = const()[name = string("concat_157"), val = tensor([16, 1536, 128])]; tensor reshape_25_cast_fp16 = reshape(shape = concat_157, x = value_89_cast_fp16)[name = string("reshape_25_cast_fp16")]; bool matmul_8_transpose_x_0 = const()[name = string("matmul_8_transpose_x_0"), val = bool(false)]; bool matmul_8_transpose_y_0 = const()[name = string("matmul_8_transpose_y_0"), val = bool(false)]; tensor matmul_8_cast_fp16 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = reshape_24_cast_fp16, y = reshape_25_cast_fp16)[name = string("matmul_8_cast_fp16")]; tensor concat_161 = const()[name = string("concat_161"), val = tensor([1, 16, 64, 128])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_161, x = matmul_8_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor var_5522_perm_0 = const()[name = string("op_5522_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_5528 = const()[name = string("op_5528"), val = tensor([1, 64, 2048])]; tensor var_5522_cast_fp16 = transpose(perm = var_5522_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_175")]; tensor output_51_cast_fp16 = reshape(shape = var_5528, x = var_5522_cast_fp16)[name = string("output_51_cast_fp16")]; tensor var_5533 = const()[name = string("op_5533"), val = tensor([0, 2, 1])]; string var_5549_pad_type_0 = const()[name = string("op_5549_pad_type_0"), val = string("valid")]; int32 var_5549_groups_0 = const()[name = string("op_5549_groups_0"), val = int32(1)]; tensor var_5549_strides_0 = const()[name = string("op_5549_strides_0"), val = tensor([1])]; tensor var_5549_pad_0 = const()[name = string("op_5549_pad_0"), val = tensor([0, 0])]; tensor var_5549_dilations_0 = const()[name = string("op_5549_dilations_0"), val = tensor([1])]; tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305658752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307231680))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5534_cast_fp16 = transpose(perm = var_5533, x = output_51_cast_fp16)[name = string("transpose_174")]; tensor var_5549_cast_fp16 = conv(dilations = var_5549_dilations_0, groups = var_5549_groups_0, pad = var_5549_pad_0, pad_type = var_5549_pad_type_0, strides = var_5549_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5534_cast_fp16)[name = string("op_5549_cast_fp16")]; tensor var_5553 = const()[name = string("op_5553"), val = tensor([0, 2, 1])]; tensor attn_output_17_cast_fp16 = transpose(perm = var_5553, x = var_5549_cast_fp16)[name = string("transpose_173")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = attn_output_17_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; int32 var_5568 = const()[name = string("op_5568"), val = int32(-1)]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5570_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_132_promoted_to_fp16)[name = string("op_5570_cast_fp16")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155_cast_fp16 = concat(axis = var_5568, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_5570_cast_fp16))[name = string("input_155_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_5565_to_fp16 = const()[name = string("op_5565_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_5565_to_fp16, x = input_155_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor normed_143_begin_0 = const()[name = string("normed_143_begin_0"), val = tensor([0, 0, 0])]; tensor normed_143_end_0 = const()[name = string("normed_143_end_0"), val = tensor([1, 64, 1024])]; tensor normed_143_end_mask_0 = const()[name = string("normed_143_end_mask_0"), val = tensor([true, true, false])]; tensor normed_143_cast_fp16 = slice_by_index(begin = normed_143_begin_0, end = normed_143_end_0, end_mask = normed_143_end_mask_0, x = normed_141_cast_fp16)[name = string("normed_143_cast_fp16")]; tensor const_134_promoted_to_fp16 = const()[name = string("const_134_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307248128)))]; tensor x_33_cast_fp16 = mul(x = normed_143_cast_fp16, y = const_134_promoted_to_fp16)[name = string("x_33_cast_fp16")]; tensor var_5590 = const()[name = string("op_5590"), val = tensor([0, 2, 1])]; tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; tensor var_5591 = transpose(perm = var_5590, x = x_33_cast_fp16)[name = string("transpose_172")]; tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_5591)[name = string("input_157")]; string input_159_pad_type_0 = const()[name = string("input_159_pad_type_0"), val = string("valid")]; tensor input_159_strides_0 = const()[name = string("input_159_strides_0"), val = tensor([1, 1])]; tensor input_159_pad_0 = const()[name = string("input_159_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_159_dilations_0 = const()[name = string("input_159_dilations_0"), val = tensor([1, 1])]; int32 input_159_groups_0 = const()[name = string("input_159_groups_0"), val = int32(1)]; tensor input_159 = conv(dilations = input_159_dilations_0, groups = input_159_groups_0, pad = input_159_pad_0, pad_type = input_159_pad_type_0, strides = input_159_strides_0, weight = model_model_layers_8_mlp_gate_proj_weight_palettized, x = input_157)[name = string("input_159")]; string b_17_pad_type_0 = const()[name = string("b_17_pad_type_0"), val = string("valid")]; tensor b_17_strides_0 = const()[name = string("b_17_strides_0"), val = tensor([1, 1])]; tensor b_17_pad_0 = const()[name = string("b_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_17_dilations_0 = const()[name = string("b_17_dilations_0"), val = tensor([1, 1])]; int32 b_17_groups_0 = const()[name = string("b_17_groups_0"), val = int32(1)]; tensor b_17 = conv(dilations = b_17_dilations_0, groups = b_17_groups_0, pad = b_17_pad_0, pad_type = b_17_pad_type_0, strides = b_17_strides_0, weight = model_model_layers_8_mlp_up_proj_weight_palettized, x = input_157)[name = string("b_17")]; tensor c_17 = silu(x = input_159)[name = string("c_17")]; tensor input_161 = mul(x = c_17, y = b_17)[name = string("input_161")]; string e_17_pad_type_0 = const()[name = string("e_17_pad_type_0"), val = string("valid")]; tensor e_17_strides_0 = const()[name = string("e_17_strides_0"), val = tensor([1, 1])]; tensor e_17_pad_0 = const()[name = string("e_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_17_dilations_0 = const()[name = string("e_17_dilations_0"), val = tensor([1, 1])]; int32 e_17_groups_0 = const()[name = string("e_17_groups_0"), val = int32(1)]; tensor e_17 = conv(dilations = e_17_dilations_0, groups = e_17_groups_0, pad = e_17_pad_0, pad_type = e_17_pad_type_0, strides = e_17_strides_0, weight = model_model_layers_8_mlp_down_proj_weight_palettized, x = input_161)[name = string("e_17")]; tensor var_5613_axes_0 = const()[name = string("op_5613_axes_0"), val = tensor([2])]; tensor var_5613 = squeeze(axes = var_5613_axes_0, x = e_17)[name = string("op_5613")]; tensor var_5614 = const()[name = string("op_5614"), val = tensor([0, 2, 1])]; tensor var_5615 = transpose(perm = var_5614, x = var_5613)[name = string("transpose_171")]; tensor hidden_states_91_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = var_5615)[name = string("hidden_states_91_cast_fp16")]; int32 var_5629 = const()[name = string("op_5629"), val = int32(-1)]; fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5631_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_135_promoted_to_fp16)[name = string("op_5631_cast_fp16")]; bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; tensor input_163_cast_fp16 = concat(axis = var_5629, interleave = input_163_interleave_0, values = (hidden_states_91_cast_fp16, var_5631_cast_fp16))[name = string("input_163_cast_fp16")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_5626_to_fp16 = const()[name = string("op_5626_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_5626_to_fp16, x = input_163_cast_fp16)[name = string("normed_145_cast_fp16")]; tensor normed_147_begin_0 = const()[name = string("normed_147_begin_0"), val = tensor([0, 0, 0])]; tensor normed_147_end_0 = const()[name = string("normed_147_end_0"), val = tensor([1, 64, 1024])]; tensor normed_147_end_mask_0 = const()[name = string("normed_147_end_mask_0"), val = tensor([true, true, false])]; tensor normed_147_cast_fp16 = slice_by_index(begin = normed_147_begin_0, end = normed_147_end_0, end_mask = normed_147_end_mask_0, x = normed_145_cast_fp16)[name = string("normed_147_cast_fp16")]; tensor const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307250240)))]; tensor hidden_states_93_cast_fp16 = mul(x = normed_147_cast_fp16, y = const_137_promoted_to_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor var_5643 = const()[name = string("op_5643"), val = tensor([0, 2, 1])]; tensor var_5646_axes_0 = const()[name = string("op_5646_axes_0"), val = tensor([2])]; tensor var_5644_cast_fp16 = transpose(perm = var_5643, x = hidden_states_93_cast_fp16)[name = string("transpose_170")]; tensor var_5646_cast_fp16 = expand_dims(axes = var_5646_axes_0, x = var_5644_cast_fp16)[name = string("op_5646_cast_fp16")]; string var_5662_pad_type_0 = const()[name = string("op_5662_pad_type_0"), val = string("valid")]; tensor var_5662_strides_0 = const()[name = string("op_5662_strides_0"), val = tensor([1, 1])]; tensor var_5662_pad_0 = const()[name = string("op_5662_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5662_dilations_0 = const()[name = string("op_5662_dilations_0"), val = tensor([1, 1])]; int32 var_5662_groups_0 = const()[name = string("op_5662_groups_0"), val = int32(1)]; tensor var_5662 = conv(dilations = var_5662_dilations_0, groups = var_5662_groups_0, pad = var_5662_pad_0, pad_type = var_5662_pad_type_0, strides = var_5662_strides_0, weight = model_model_layers_9_self_attn_q_proj_weight_palettized, x = var_5646_cast_fp16)[name = string("op_5662")]; tensor var_5667 = const()[name = string("op_5667"), val = tensor([1, 16, 128, 64])]; tensor var_5668 = reshape(shape = var_5667, x = var_5662)[name = string("op_5668")]; tensor var_5673 = const()[name = string("op_5673"), val = tensor([0, 1, 3, 2])]; string var_5685_pad_type_0 = const()[name = string("op_5685_pad_type_0"), val = string("valid")]; tensor var_5685_strides_0 = const()[name = string("op_5685_strides_0"), val = tensor([1, 1])]; tensor var_5685_pad_0 = const()[name = string("op_5685_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5685_dilations_0 = const()[name = string("op_5685_dilations_0"), val = tensor([1, 1])]; int32 var_5685_groups_0 = const()[name = string("op_5685_groups_0"), val = int32(1)]; tensor var_5685 = conv(dilations = var_5685_dilations_0, groups = var_5685_groups_0, pad = var_5685_pad_0, pad_type = var_5685_pad_type_0, strides = var_5685_strides_0, weight = model_model_layers_9_self_attn_k_proj_weight_palettized, x = var_5646_cast_fp16)[name = string("op_5685")]; tensor var_5690 = const()[name = string("op_5690"), val = tensor([1, 8, 128, 64])]; tensor var_5691 = reshape(shape = var_5690, x = var_5685)[name = string("op_5691")]; tensor var_5696 = const()[name = string("op_5696"), val = tensor([0, 1, 3, 2])]; string var_5708_pad_type_0 = const()[name = string("op_5708_pad_type_0"), val = string("valid")]; tensor var_5708_strides_0 = const()[name = string("op_5708_strides_0"), val = tensor([1, 1])]; tensor var_5708_pad_0 = const()[name = string("op_5708_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5708_dilations_0 = const()[name = string("op_5708_dilations_0"), val = tensor([1, 1])]; int32 var_5708_groups_0 = const()[name = string("op_5708_groups_0"), val = int32(1)]; tensor var_5708 = conv(dilations = var_5708_dilations_0, groups = var_5708_groups_0, pad = var_5708_pad_0, pad_type = var_5708_pad_type_0, strides = var_5708_strides_0, weight = model_model_layers_9_self_attn_v_proj_weight_palettized, x = var_5646_cast_fp16)[name = string("op_5708")]; tensor var_5713 = const()[name = string("op_5713"), val = tensor([1, 8, 128, 64])]; tensor var_5714 = reshape(shape = var_5713, x = var_5708)[name = string("op_5714")]; tensor var_5719 = const()[name = string("op_5719"), val = tensor([0, 1, 3, 2])]; int32 var_5732 = const()[name = string("op_5732"), val = int32(-1)]; fp16 const_138_promoted = const()[name = string("const_138_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_95 = transpose(perm = var_5673, x = var_5668)[name = string("transpose_169")]; tensor var_5734 = mul(x = hidden_states_95, y = const_138_promoted)[name = string("op_5734")]; bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; tensor input_167 = concat(axis = var_5732, interleave = input_167_interleave_0, values = (hidden_states_95, var_5734))[name = string("input_167")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_5729_to_fp16 = const()[name = string("op_5729_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_5729_to_fp16, x = input_167)[name = string("normed_149_cast_fp16")]; tensor normed_151_begin_0 = const()[name = string("normed_151_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_151_end_0 = const()[name = string("normed_151_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_151_end_mask_0 = const()[name = string("normed_151_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_151 = slice_by_index(begin = normed_151_begin_0, end = normed_151_end_0, end_mask = normed_151_end_mask_0, x = normed_149_cast_fp16)[name = string("normed_151")]; tensor const_140 = const()[name = string("const_140"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307252352)))]; tensor q_19 = mul(x = normed_151, y = const_140)[name = string("q_19")]; int32 var_5754 = const()[name = string("op_5754"), val = int32(-1)]; fp16 const_141_promoted = const()[name = string("const_141_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_97 = transpose(perm = var_5696, x = var_5691)[name = string("transpose_168")]; tensor var_5756 = mul(x = hidden_states_97, y = const_141_promoted)[name = string("op_5756")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169 = concat(axis = var_5754, interleave = input_169_interleave_0, values = (hidden_states_97, var_5756))[name = string("input_169")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_5751_to_fp16 = const()[name = string("op_5751_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_5751_to_fp16, x = input_169)[name = string("normed_153_cast_fp16")]; tensor normed_155_begin_0 = const()[name = string("normed_155_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_155_end_0 = const()[name = string("normed_155_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_155_end_mask_0 = const()[name = string("normed_155_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_155 = slice_by_index(begin = normed_155_begin_0, end = normed_155_end_0, end_mask = normed_155_end_mask_0, x = normed_153_cast_fp16)[name = string("normed_155")]; tensor const_143 = const()[name = string("const_143"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307252672)))]; tensor k_19 = mul(x = normed_155, y = const_143)[name = string("k_19")]; tensor var_5777 = mul(x = q_19, y = cos_1)[name = string("op_5777")]; tensor var_5782_begin_0 = const()[name = string("op_5782_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5782_end_0 = const()[name = string("op_5782_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_5782_end_mask_0 = const()[name = string("op_5782_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5782 = slice_by_index(begin = var_5782_begin_0, end = var_5782_end_0, end_mask = var_5782_end_mask_0, x = q_19)[name = string("op_5782")]; fp16 const_144_promoted = const()[name = string("const_144_promoted"), val = fp16(-0x1p+0)]; tensor var_5783 = mul(x = var_5782, y = const_144_promoted)[name = string("op_5783")]; tensor var_5788_begin_0 = const()[name = string("op_5788_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5788_end_0 = const()[name = string("op_5788_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_5788_end_mask_0 = const()[name = string("op_5788_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5788 = slice_by_index(begin = var_5788_begin_0, end = var_5788_end_0, end_mask = var_5788_end_mask_0, x = q_19)[name = string("op_5788")]; int32 var_5790 = const()[name = string("op_5790"), val = int32(-1)]; bool var_5791_interleave_0 = const()[name = string("op_5791_interleave_0"), val = bool(false)]; tensor var_5791 = concat(axis = var_5790, interleave = var_5791_interleave_0, values = (var_5783, var_5788))[name = string("op_5791")]; tensor var_5792 = mul(x = var_5791, y = sin_1)[name = string("op_5792")]; tensor query_19 = add(x = var_5777, y = var_5792)[name = string("query_19")]; tensor var_5795 = mul(x = k_19, y = cos_1)[name = string("op_5795")]; tensor var_5800_begin_0 = const()[name = string("op_5800_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_5800_end_0 = const()[name = string("op_5800_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_5800_end_mask_0 = const()[name = string("op_5800_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_5800 = slice_by_index(begin = var_5800_begin_0, end = var_5800_end_0, end_mask = var_5800_end_mask_0, x = k_19)[name = string("op_5800")]; fp16 const_145_promoted = const()[name = string("const_145_promoted"), val = fp16(-0x1p+0)]; tensor var_5801 = mul(x = var_5800, y = const_145_promoted)[name = string("op_5801")]; tensor var_5806_begin_0 = const()[name = string("op_5806_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_5806_end_0 = const()[name = string("op_5806_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_5806_end_mask_0 = const()[name = string("op_5806_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_5806 = slice_by_index(begin = var_5806_begin_0, end = var_5806_end_0, end_mask = var_5806_end_mask_0, x = k_19)[name = string("op_5806")]; int32 var_5808 = const()[name = string("op_5808"), val = int32(-1)]; bool var_5809_interleave_0 = const()[name = string("op_5809_interleave_0"), val = bool(false)]; tensor var_5809 = concat(axis = var_5808, interleave = var_5809_interleave_0, values = (var_5801, var_5806))[name = string("op_5809")]; tensor var_5810 = mul(x = var_5809, y = sin_1)[name = string("op_5810")]; tensor key_19 = add(x = var_5795, y = var_5810)[name = string("key_19")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; int32 concat_164_axis_0 = const()[name = string("concat_164_axis_0"), val = int32(0)]; bool concat_164_interleave_0 = const()[name = string("concat_164_interleave_0"), val = bool(false)]; tensor concat_164 = concat(axis = concat_164_axis_0, interleave = concat_164_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_164")]; tensor concat_165_values1_0 = const()[name = string("concat_165_values1_0"), val = tensor([0])]; tensor concat_165_values3_0 = const()[name = string("concat_165_values3_0"), val = tensor([0])]; int32 concat_165_axis_0 = const()[name = string("concat_165_axis_0"), val = int32(0)]; bool concat_165_interleave_0 = const()[name = string("concat_165_interleave_0"), val = bool(false)]; tensor concat_165 = concat(axis = concat_165_axis_0, interleave = concat_165_interleave_0, values = (expand_dims_112, concat_165_values1_0, var_1746, concat_165_values3_0))[name = string("concat_165")]; tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_164, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_165, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = key_19, x = coreml_update_state_73)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_186_write_state")]; tensor coreml_update_state_74 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_186")]; tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([37])]; tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([38])]; int32 concat_168_axis_0 = const()[name = string("concat_168_axis_0"), val = int32(0)]; bool concat_168_interleave_0 = const()[name = string("concat_168_interleave_0"), val = bool(false)]; tensor concat_168 = concat(axis = concat_168_axis_0, interleave = concat_168_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_168")]; tensor concat_169_values1_0 = const()[name = string("concat_169_values1_0"), val = tensor([0])]; tensor concat_169_values3_0 = const()[name = string("concat_169_values3_0"), val = tensor([0])]; int32 concat_169_axis_0 = const()[name = string("concat_169_axis_0"), val = int32(0)]; bool concat_169_interleave_0 = const()[name = string("concat_169_interleave_0"), val = bool(false)]; tensor concat_169 = concat(axis = concat_169_axis_0, interleave = concat_169_interleave_0, values = (expand_dims_118, concat_169_values1_0, var_1746, concat_169_values3_0))[name = string("concat_169")]; tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_91 = transpose(perm = var_5719, x = var_5714)[name = string("transpose_167")]; tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_168, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_169, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = value_91, x = coreml_update_state_74)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_187_write_state")]; tensor coreml_update_state_75 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_187")]; tensor var_5881_begin_0 = const()[name = string("op_5881_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_5881_end_0 = const()[name = string("op_5881_end_0"), val = tensor([10, 8, 1536, 128])]; tensor var_5881_end_mask_0 = const()[name = string("op_5881_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5881_cast_fp16 = slice_by_index(begin = var_5881_begin_0, end = var_5881_end_0, end_mask = var_5881_end_mask_0, x = coreml_update_state_75)[name = string("op_5881_cast_fp16")]; tensor key_cache_19_axes_0 = const()[name = string("key_cache_19_axes_0"), val = tensor([0])]; tensor key_cache_19_cast_fp16 = squeeze(axes = key_cache_19_axes_0, x = var_5881_cast_fp16)[name = string("key_cache_19_cast_fp16")]; tensor var_5888_begin_0 = const()[name = string("op_5888_begin_0"), val = tensor([37, 0, 0, 0])]; tensor var_5888_end_0 = const()[name = string("op_5888_end_0"), val = tensor([38, 8, 1536, 128])]; tensor var_5888_end_mask_0 = const()[name = string("op_5888_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5888_cast_fp16 = slice_by_index(begin = var_5888_begin_0, end = var_5888_end_0, end_mask = var_5888_end_mask_0, x = coreml_update_state_75)[name = string("op_5888_cast_fp16")]; tensor value_cache_19_axes_0 = const()[name = string("value_cache_19_axes_0"), val = tensor([0])]; tensor value_cache_19_cast_fp16 = squeeze(axes = value_cache_19_axes_0, x = var_5888_cast_fp16)[name = string("value_cache_19_cast_fp16")]; tensor var_5912_axes_0 = const()[name = string("op_5912_axes_0"), val = tensor([1])]; tensor var_5912_cast_fp16 = expand_dims(axes = var_5912_axes_0, x = key_cache_19_cast_fp16)[name = string("op_5912_cast_fp16")]; tensor var_5917 = const()[name = string("op_5917"), val = tensor([1, 2, 1, 1])]; tensor value_95_cast_fp16 = tile(reps = var_5917, x = var_5912_cast_fp16)[name = string("value_95_cast_fp16")]; tensor var_5923 = const()[name = string("op_5923"), val = tensor([1, 16, 1536, 128])]; tensor key_states_39_cast_fp16 = reshape(shape = var_5923, x = value_95_cast_fp16)[name = string("key_states_39_cast_fp16")]; tensor var_5926_axes_0 = const()[name = string("op_5926_axes_0"), val = tensor([1])]; tensor var_5926_cast_fp16 = expand_dims(axes = var_5926_axes_0, x = value_cache_19_cast_fp16)[name = string("op_5926_cast_fp16")]; tensor var_5931 = const()[name = string("op_5931"), val = tensor([1, 2, 1, 1])]; tensor value_99_cast_fp16 = tile(reps = var_5931, x = var_5926_cast_fp16)[name = string("value_99_cast_fp16")]; bool var_5952_transpose_x_0 = const()[name = string("op_5952_transpose_x_0"), val = bool(false)]; bool var_5952_transpose_y_0 = const()[name = string("op_5952_transpose_y_0"), val = bool(true)]; tensor var_5952 = matmul(transpose_x = var_5952_transpose_x_0, transpose_y = var_5952_transpose_y_0, x = query_19, y = key_states_39_cast_fp16)[name = string("op_5952")]; fp16 var_5953_to_fp16 = const()[name = string("op_5953_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_37_cast_fp16 = mul(x = var_5952, y = var_5953_to_fp16)[name = string("attention_37_cast_fp16")]; tensor attention_39_cast_fp16 = add(x = attention_37_cast_fp16, y = causal_mask)[name = string("attention_39_cast_fp16")]; int32 var_5962 = const()[name = string("op_5962"), val = int32(-1)]; tensor var_5964_cast_fp16 = softmax(axis = var_5962, x = attention_39_cast_fp16)[name = string("op_5964_cast_fp16")]; tensor concat_174 = const()[name = string("concat_174"), val = tensor([16, 64, 1536])]; tensor reshape_27_cast_fp16 = reshape(shape = concat_174, x = var_5964_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor concat_175 = const()[name = string("concat_175"), val = tensor([16, 1536, 128])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_175, x = value_99_cast_fp16)[name = string("reshape_28_cast_fp16")]; bool matmul_9_transpose_x_0 = const()[name = string("matmul_9_transpose_x_0"), val = bool(false)]; bool matmul_9_transpose_y_0 = const()[name = string("matmul_9_transpose_y_0"), val = bool(false)]; tensor matmul_9_cast_fp16 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = reshape_27_cast_fp16, y = reshape_28_cast_fp16)[name = string("matmul_9_cast_fp16")]; tensor concat_179 = const()[name = string("concat_179"), val = tensor([1, 16, 64, 128])]; tensor reshape_29_cast_fp16 = reshape(shape = concat_179, x = matmul_9_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor var_5976_perm_0 = const()[name = string("op_5976_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_5982 = const()[name = string("op_5982"), val = tensor([1, 64, 2048])]; tensor var_5976_cast_fp16 = transpose(perm = var_5976_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_166")]; tensor output_57_cast_fp16 = reshape(shape = var_5982, x = var_5976_cast_fp16)[name = string("output_57_cast_fp16")]; tensor var_5987 = const()[name = string("op_5987"), val = tensor([0, 2, 1])]; string var_6003_pad_type_0 = const()[name = string("op_6003_pad_type_0"), val = string("valid")]; int32 var_6003_groups_0 = const()[name = string("op_6003_groups_0"), val = int32(1)]; tensor var_6003_strides_0 = const()[name = string("op_6003_strides_0"), val = tensor([1])]; tensor var_6003_pad_0 = const()[name = string("op_6003_pad_0"), val = tensor([0, 0])]; tensor var_6003_dilations_0 = const()[name = string("op_6003_dilations_0"), val = tensor([1])]; tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307252992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308825920))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_5988_cast_fp16 = transpose(perm = var_5987, x = output_57_cast_fp16)[name = string("transpose_165")]; tensor var_6003_cast_fp16 = conv(dilations = var_6003_dilations_0, groups = var_6003_groups_0, pad = var_6003_pad_0, pad_type = var_6003_pad_type_0, strides = var_6003_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_5988_cast_fp16)[name = string("op_6003_cast_fp16")]; tensor var_6007 = const()[name = string("op_6007"), val = tensor([0, 2, 1])]; tensor attn_output_19_cast_fp16 = transpose(perm = var_6007, x = var_6003_cast_fp16)[name = string("transpose_164")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_91_cast_fp16, y = attn_output_19_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; int32 var_6022 = const()[name = string("op_6022"), val = int32(-1)]; fp16 const_147_promoted_to_fp16 = const()[name = string("const_147_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6024_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_147_promoted_to_fp16)[name = string("op_6024_cast_fp16")]; bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; tensor input_173_cast_fp16 = concat(axis = var_6022, interleave = input_173_interleave_0, values = (hidden_states_99_cast_fp16, var_6024_cast_fp16))[name = string("input_173_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_6019_to_fp16 = const()[name = string("op_6019_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_6019_to_fp16, x = input_173_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor normed_159_begin_0 = const()[name = string("normed_159_begin_0"), val = tensor([0, 0, 0])]; tensor normed_159_end_0 = const()[name = string("normed_159_end_0"), val = tensor([1, 64, 1024])]; tensor normed_159_end_mask_0 = const()[name = string("normed_159_end_mask_0"), val = tensor([true, true, false])]; tensor normed_159_cast_fp16 = slice_by_index(begin = normed_159_begin_0, end = normed_159_end_0, end_mask = normed_159_end_mask_0, x = normed_157_cast_fp16)[name = string("normed_159_cast_fp16")]; tensor const_149_promoted_to_fp16 = const()[name = string("const_149_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308842368)))]; tensor x_37_cast_fp16 = mul(x = normed_159_cast_fp16, y = const_149_promoted_to_fp16)[name = string("x_37_cast_fp16")]; tensor var_6044 = const()[name = string("op_6044"), val = tensor([0, 2, 1])]; tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; tensor var_6045 = transpose(perm = var_6044, x = x_37_cast_fp16)[name = string("transpose_163")]; tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_6045)[name = string("input_175")]; string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; tensor input_177 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = model_model_layers_9_mlp_gate_proj_weight_palettized, x = input_175)[name = string("input_177")]; string b_19_pad_type_0 = const()[name = string("b_19_pad_type_0"), val = string("valid")]; tensor b_19_strides_0 = const()[name = string("b_19_strides_0"), val = tensor([1, 1])]; tensor b_19_pad_0 = const()[name = string("b_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_19_dilations_0 = const()[name = string("b_19_dilations_0"), val = tensor([1, 1])]; int32 b_19_groups_0 = const()[name = string("b_19_groups_0"), val = int32(1)]; tensor b_19 = conv(dilations = b_19_dilations_0, groups = b_19_groups_0, pad = b_19_pad_0, pad_type = b_19_pad_type_0, strides = b_19_strides_0, weight = model_model_layers_9_mlp_up_proj_weight_palettized, x = input_175)[name = string("b_19")]; tensor c_19 = silu(x = input_177)[name = string("c_19")]; tensor input_179 = mul(x = c_19, y = b_19)[name = string("input_179")]; string e_19_pad_type_0 = const()[name = string("e_19_pad_type_0"), val = string("valid")]; tensor e_19_strides_0 = const()[name = string("e_19_strides_0"), val = tensor([1, 1])]; tensor e_19_pad_0 = const()[name = string("e_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_19_dilations_0 = const()[name = string("e_19_dilations_0"), val = tensor([1, 1])]; int32 e_19_groups_0 = const()[name = string("e_19_groups_0"), val = int32(1)]; tensor e_19 = conv(dilations = e_19_dilations_0, groups = e_19_groups_0, pad = e_19_pad_0, pad_type = e_19_pad_type_0, strides = e_19_strides_0, weight = model_model_layers_9_mlp_down_proj_weight_palettized, x = input_179)[name = string("e_19")]; tensor var_6067_axes_0 = const()[name = string("op_6067_axes_0"), val = tensor([2])]; tensor var_6067 = squeeze(axes = var_6067_axes_0, x = e_19)[name = string("op_6067")]; tensor var_6068 = const()[name = string("op_6068"), val = tensor([0, 2, 1])]; tensor var_6069 = transpose(perm = var_6068, x = var_6067)[name = string("transpose_162")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = var_6069)[name = string("hidden_states_101_cast_fp16")]; int32 var_6083 = const()[name = string("op_6083"), val = int32(-1)]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6085_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_6085_cast_fp16")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181_cast_fp16 = concat(axis = var_6083, interleave = input_181_interleave_0, values = (hidden_states_101_cast_fp16, var_6085_cast_fp16))[name = string("input_181_cast_fp16")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_6080_to_fp16 = const()[name = string("op_6080_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_6080_to_fp16, x = input_181_cast_fp16)[name = string("normed_161_cast_fp16")]; tensor normed_163_begin_0 = const()[name = string("normed_163_begin_0"), val = tensor([0, 0, 0])]; tensor normed_163_end_0 = const()[name = string("normed_163_end_0"), val = tensor([1, 64, 1024])]; tensor normed_163_end_mask_0 = const()[name = string("normed_163_end_mask_0"), val = tensor([true, true, false])]; tensor normed_163_cast_fp16 = slice_by_index(begin = normed_163_begin_0, end = normed_163_end_0, end_mask = normed_163_end_mask_0, x = normed_161_cast_fp16)[name = string("normed_163_cast_fp16")]; tensor const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308844480)))]; tensor hidden_states_103_cast_fp16 = mul(x = normed_163_cast_fp16, y = const_152_promoted_to_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor var_6097 = const()[name = string("op_6097"), val = tensor([0, 2, 1])]; tensor var_6100_axes_0 = const()[name = string("op_6100_axes_0"), val = tensor([2])]; tensor var_6098_cast_fp16 = transpose(perm = var_6097, x = hidden_states_103_cast_fp16)[name = string("transpose_161")]; tensor var_6100_cast_fp16 = expand_dims(axes = var_6100_axes_0, x = var_6098_cast_fp16)[name = string("op_6100_cast_fp16")]; string var_6116_pad_type_0 = const()[name = string("op_6116_pad_type_0"), val = string("valid")]; tensor var_6116_strides_0 = const()[name = string("op_6116_strides_0"), val = tensor([1, 1])]; tensor var_6116_pad_0 = const()[name = string("op_6116_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6116_dilations_0 = const()[name = string("op_6116_dilations_0"), val = tensor([1, 1])]; int32 var_6116_groups_0 = const()[name = string("op_6116_groups_0"), val = int32(1)]; tensor var_6116 = conv(dilations = var_6116_dilations_0, groups = var_6116_groups_0, pad = var_6116_pad_0, pad_type = var_6116_pad_type_0, strides = var_6116_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_6100_cast_fp16)[name = string("op_6116")]; tensor var_6121 = const()[name = string("op_6121"), val = tensor([1, 16, 128, 64])]; tensor var_6122 = reshape(shape = var_6121, x = var_6116)[name = string("op_6122")]; tensor var_6127 = const()[name = string("op_6127"), val = tensor([0, 1, 3, 2])]; string var_6139_pad_type_0 = const()[name = string("op_6139_pad_type_0"), val = string("valid")]; tensor var_6139_strides_0 = const()[name = string("op_6139_strides_0"), val = tensor([1, 1])]; tensor var_6139_pad_0 = const()[name = string("op_6139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6139_dilations_0 = const()[name = string("op_6139_dilations_0"), val = tensor([1, 1])]; int32 var_6139_groups_0 = const()[name = string("op_6139_groups_0"), val = int32(1)]; tensor var_6139 = conv(dilations = var_6139_dilations_0, groups = var_6139_groups_0, pad = var_6139_pad_0, pad_type = var_6139_pad_type_0, strides = var_6139_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_6100_cast_fp16)[name = string("op_6139")]; tensor var_6144 = const()[name = string("op_6144"), val = tensor([1, 8, 128, 64])]; tensor var_6145 = reshape(shape = var_6144, x = var_6139)[name = string("op_6145")]; tensor var_6150 = const()[name = string("op_6150"), val = tensor([0, 1, 3, 2])]; string var_6162_pad_type_0 = const()[name = string("op_6162_pad_type_0"), val = string("valid")]; tensor var_6162_strides_0 = const()[name = string("op_6162_strides_0"), val = tensor([1, 1])]; tensor var_6162_pad_0 = const()[name = string("op_6162_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6162_dilations_0 = const()[name = string("op_6162_dilations_0"), val = tensor([1, 1])]; int32 var_6162_groups_0 = const()[name = string("op_6162_groups_0"), val = int32(1)]; tensor var_6162 = conv(dilations = var_6162_dilations_0, groups = var_6162_groups_0, pad = var_6162_pad_0, pad_type = var_6162_pad_type_0, strides = var_6162_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_6100_cast_fp16)[name = string("op_6162")]; tensor var_6167 = const()[name = string("op_6167"), val = tensor([1, 8, 128, 64])]; tensor var_6168 = reshape(shape = var_6167, x = var_6162)[name = string("op_6168")]; tensor var_6173 = const()[name = string("op_6173"), val = tensor([0, 1, 3, 2])]; int32 var_6186 = const()[name = string("op_6186"), val = int32(-1)]; fp16 const_153_promoted = const()[name = string("const_153_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_105 = transpose(perm = var_6127, x = var_6122)[name = string("transpose_160")]; tensor var_6188 = mul(x = hidden_states_105, y = const_153_promoted)[name = string("op_6188")]; bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; tensor input_185 = concat(axis = var_6186, interleave = input_185_interleave_0, values = (hidden_states_105, var_6188))[name = string("input_185")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_6183_to_fp16 = const()[name = string("op_6183_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_6183_to_fp16, x = input_185)[name = string("normed_165_cast_fp16")]; tensor normed_167_begin_0 = const()[name = string("normed_167_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_167_end_0 = const()[name = string("normed_167_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_167_end_mask_0 = const()[name = string("normed_167_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_167 = slice_by_index(begin = normed_167_begin_0, end = normed_167_end_0, end_mask = normed_167_end_mask_0, x = normed_165_cast_fp16)[name = string("normed_167")]; tensor const_155 = const()[name = string("const_155"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308846592)))]; tensor q_21 = mul(x = normed_167, y = const_155)[name = string("q_21")]; int32 var_6208 = const()[name = string("op_6208"), val = int32(-1)]; fp16 const_156_promoted = const()[name = string("const_156_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_107 = transpose(perm = var_6150, x = var_6145)[name = string("transpose_159")]; tensor var_6210 = mul(x = hidden_states_107, y = const_156_promoted)[name = string("op_6210")]; bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; tensor input_187 = concat(axis = var_6208, interleave = input_187_interleave_0, values = (hidden_states_107, var_6210))[name = string("input_187")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_6205_to_fp16 = const()[name = string("op_6205_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_6205_to_fp16, x = input_187)[name = string("normed_169_cast_fp16")]; tensor normed_171_begin_0 = const()[name = string("normed_171_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_171_end_0 = const()[name = string("normed_171_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_171_end_mask_0 = const()[name = string("normed_171_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_171 = slice_by_index(begin = normed_171_begin_0, end = normed_171_end_0, end_mask = normed_171_end_mask_0, x = normed_169_cast_fp16)[name = string("normed_171")]; tensor const_158 = const()[name = string("const_158"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308846912)))]; tensor k_21 = mul(x = normed_171, y = const_158)[name = string("k_21")]; tensor var_6231 = mul(x = q_21, y = cos_1)[name = string("op_6231")]; tensor var_6236_begin_0 = const()[name = string("op_6236_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6236_end_0 = const()[name = string("op_6236_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_6236_end_mask_0 = const()[name = string("op_6236_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6236 = slice_by_index(begin = var_6236_begin_0, end = var_6236_end_0, end_mask = var_6236_end_mask_0, x = q_21)[name = string("op_6236")]; fp16 const_159_promoted = const()[name = string("const_159_promoted"), val = fp16(-0x1p+0)]; tensor var_6237 = mul(x = var_6236, y = const_159_promoted)[name = string("op_6237")]; tensor var_6242_begin_0 = const()[name = string("op_6242_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6242_end_0 = const()[name = string("op_6242_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_6242_end_mask_0 = const()[name = string("op_6242_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6242 = slice_by_index(begin = var_6242_begin_0, end = var_6242_end_0, end_mask = var_6242_end_mask_0, x = q_21)[name = string("op_6242")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-1)]; bool var_6245_interleave_0 = const()[name = string("op_6245_interleave_0"), val = bool(false)]; tensor var_6245 = concat(axis = var_6244, interleave = var_6245_interleave_0, values = (var_6237, var_6242))[name = string("op_6245")]; tensor var_6246 = mul(x = var_6245, y = sin_1)[name = string("op_6246")]; tensor query_21 = add(x = var_6231, y = var_6246)[name = string("query_21")]; tensor var_6249 = mul(x = k_21, y = cos_1)[name = string("op_6249")]; tensor var_6254_begin_0 = const()[name = string("op_6254_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6254_end_0 = const()[name = string("op_6254_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_6254_end_mask_0 = const()[name = string("op_6254_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6254 = slice_by_index(begin = var_6254_begin_0, end = var_6254_end_0, end_mask = var_6254_end_mask_0, x = k_21)[name = string("op_6254")]; fp16 const_160_promoted = const()[name = string("const_160_promoted"), val = fp16(-0x1p+0)]; tensor var_6255 = mul(x = var_6254, y = const_160_promoted)[name = string("op_6255")]; tensor var_6260_begin_0 = const()[name = string("op_6260_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6260_end_0 = const()[name = string("op_6260_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_6260_end_mask_0 = const()[name = string("op_6260_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6260 = slice_by_index(begin = var_6260_begin_0, end = var_6260_end_0, end_mask = var_6260_end_mask_0, x = k_21)[name = string("op_6260")]; int32 var_6262 = const()[name = string("op_6262"), val = int32(-1)]; bool var_6263_interleave_0 = const()[name = string("op_6263_interleave_0"), val = bool(false)]; tensor var_6263 = concat(axis = var_6262, interleave = var_6263_interleave_0, values = (var_6255, var_6260))[name = string("op_6263")]; tensor var_6264 = mul(x = var_6263, y = sin_1)[name = string("op_6264")]; tensor key_21 = add(x = var_6249, y = var_6264)[name = string("key_21")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; int32 concat_182_axis_0 = const()[name = string("concat_182_axis_0"), val = int32(0)]; bool concat_182_interleave_0 = const()[name = string("concat_182_interleave_0"), val = bool(false)]; tensor concat_182 = concat(axis = concat_182_axis_0, interleave = concat_182_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_182")]; tensor concat_183_values1_0 = const()[name = string("concat_183_values1_0"), val = tensor([0])]; tensor concat_183_values3_0 = const()[name = string("concat_183_values3_0"), val = tensor([0])]; int32 concat_183_axis_0 = const()[name = string("concat_183_axis_0"), val = int32(0)]; bool concat_183_interleave_0 = const()[name = string("concat_183_interleave_0"), val = bool(false)]; tensor concat_183 = concat(axis = concat_183_axis_0, interleave = concat_183_interleave_0, values = (expand_dims_124, concat_183_values1_0, var_1746, concat_183_values3_0))[name = string("concat_183")]; tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_182, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_183, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = key_21, x = coreml_update_state_75)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_188_write_state")]; tensor coreml_update_state_76 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_188")]; tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([38])]; tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([39])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_186")]; tensor concat_187_values1_0 = const()[name = string("concat_187_values1_0"), val = tensor([0])]; tensor concat_187_values3_0 = const()[name = string("concat_187_values3_0"), val = tensor([0])]; int32 concat_187_axis_0 = const()[name = string("concat_187_axis_0"), val = int32(0)]; bool concat_187_interleave_0 = const()[name = string("concat_187_interleave_0"), val = bool(false)]; tensor concat_187 = concat(axis = concat_187_axis_0, interleave = concat_187_interleave_0, values = (expand_dims_130, concat_187_values1_0, var_1746, concat_187_values3_0))[name = string("concat_187")]; tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_101 = transpose(perm = var_6173, x = var_6168)[name = string("transpose_158")]; tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_186, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_187, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = value_101, x = coreml_update_state_76)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_189_write_state")]; tensor coreml_update_state_77 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_189")]; tensor var_6335_begin_0 = const()[name = string("op_6335_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_6335_end_0 = const()[name = string("op_6335_end_0"), val = tensor([11, 8, 1536, 128])]; tensor var_6335_end_mask_0 = const()[name = string("op_6335_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6335_cast_fp16 = slice_by_index(begin = var_6335_begin_0, end = var_6335_end_0, end_mask = var_6335_end_mask_0, x = coreml_update_state_77)[name = string("op_6335_cast_fp16")]; tensor key_cache_21_axes_0 = const()[name = string("key_cache_21_axes_0"), val = tensor([0])]; tensor key_cache_21_cast_fp16 = squeeze(axes = key_cache_21_axes_0, x = var_6335_cast_fp16)[name = string("key_cache_21_cast_fp16")]; tensor var_6342_begin_0 = const()[name = string("op_6342_begin_0"), val = tensor([38, 0, 0, 0])]; tensor var_6342_end_0 = const()[name = string("op_6342_end_0"), val = tensor([39, 8, 1536, 128])]; tensor var_6342_end_mask_0 = const()[name = string("op_6342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6342_cast_fp16 = slice_by_index(begin = var_6342_begin_0, end = var_6342_end_0, end_mask = var_6342_end_mask_0, x = coreml_update_state_77)[name = string("op_6342_cast_fp16")]; tensor value_cache_21_axes_0 = const()[name = string("value_cache_21_axes_0"), val = tensor([0])]; tensor value_cache_21_cast_fp16 = squeeze(axes = value_cache_21_axes_0, x = var_6342_cast_fp16)[name = string("value_cache_21_cast_fp16")]; tensor var_6366_axes_0 = const()[name = string("op_6366_axes_0"), val = tensor([1])]; tensor var_6366_cast_fp16 = expand_dims(axes = var_6366_axes_0, x = key_cache_21_cast_fp16)[name = string("op_6366_cast_fp16")]; tensor var_6371 = const()[name = string("op_6371"), val = tensor([1, 2, 1, 1])]; tensor value_105_cast_fp16 = tile(reps = var_6371, x = var_6366_cast_fp16)[name = string("value_105_cast_fp16")]; tensor var_6377 = const()[name = string("op_6377"), val = tensor([1, 16, 1536, 128])]; tensor key_states_43_cast_fp16 = reshape(shape = var_6377, x = value_105_cast_fp16)[name = string("key_states_43_cast_fp16")]; tensor var_6380_axes_0 = const()[name = string("op_6380_axes_0"), val = tensor([1])]; tensor var_6380_cast_fp16 = expand_dims(axes = var_6380_axes_0, x = value_cache_21_cast_fp16)[name = string("op_6380_cast_fp16")]; tensor var_6385 = const()[name = string("op_6385"), val = tensor([1, 2, 1, 1])]; tensor value_109_cast_fp16 = tile(reps = var_6385, x = var_6380_cast_fp16)[name = string("value_109_cast_fp16")]; bool var_6406_transpose_x_0 = const()[name = string("op_6406_transpose_x_0"), val = bool(false)]; bool var_6406_transpose_y_0 = const()[name = string("op_6406_transpose_y_0"), val = bool(true)]; tensor var_6406 = matmul(transpose_x = var_6406_transpose_x_0, transpose_y = var_6406_transpose_y_0, x = query_21, y = key_states_43_cast_fp16)[name = string("op_6406")]; fp16 var_6407_to_fp16 = const()[name = string("op_6407_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_41_cast_fp16 = mul(x = var_6406, y = var_6407_to_fp16)[name = string("attention_41_cast_fp16")]; tensor attention_43_cast_fp16 = add(x = attention_41_cast_fp16, y = causal_mask)[name = string("attention_43_cast_fp16")]; int32 var_6416 = const()[name = string("op_6416"), val = int32(-1)]; tensor var_6418_cast_fp16 = softmax(axis = var_6416, x = attention_43_cast_fp16)[name = string("op_6418_cast_fp16")]; tensor concat_192 = const()[name = string("concat_192"), val = tensor([16, 64, 1536])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_192, x = var_6418_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor concat_193 = const()[name = string("concat_193"), val = tensor([16, 1536, 128])]; tensor reshape_31_cast_fp16 = reshape(shape = concat_193, x = value_109_cast_fp16)[name = string("reshape_31_cast_fp16")]; bool matmul_10_transpose_x_0 = const()[name = string("matmul_10_transpose_x_0"), val = bool(false)]; bool matmul_10_transpose_y_0 = const()[name = string("matmul_10_transpose_y_0"), val = bool(false)]; tensor matmul_10_cast_fp16 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = reshape_30_cast_fp16, y = reshape_31_cast_fp16)[name = string("matmul_10_cast_fp16")]; tensor concat_197 = const()[name = string("concat_197"), val = tensor([1, 16, 64, 128])]; tensor reshape_32_cast_fp16 = reshape(shape = concat_197, x = matmul_10_cast_fp16)[name = string("reshape_32_cast_fp16")]; tensor var_6430_perm_0 = const()[name = string("op_6430_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_6436 = const()[name = string("op_6436"), val = tensor([1, 64, 2048])]; tensor var_6430_cast_fp16 = transpose(perm = var_6430_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_157")]; tensor output_63_cast_fp16 = reshape(shape = var_6436, x = var_6430_cast_fp16)[name = string("output_63_cast_fp16")]; tensor var_6441 = const()[name = string("op_6441"), val = tensor([0, 2, 1])]; string var_6457_pad_type_0 = const()[name = string("op_6457_pad_type_0"), val = string("valid")]; int32 var_6457_groups_0 = const()[name = string("op_6457_groups_0"), val = int32(1)]; tensor var_6457_strides_0 = const()[name = string("op_6457_strides_0"), val = tensor([1])]; tensor var_6457_pad_0 = const()[name = string("op_6457_pad_0"), val = tensor([0, 0])]; tensor var_6457_dilations_0 = const()[name = string("op_6457_dilations_0"), val = tensor([1])]; tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308847232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310420160))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6442_cast_fp16 = transpose(perm = var_6441, x = output_63_cast_fp16)[name = string("transpose_156")]; tensor var_6457_cast_fp16 = conv(dilations = var_6457_dilations_0, groups = var_6457_groups_0, pad = var_6457_pad_0, pad_type = var_6457_pad_type_0, strides = var_6457_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6442_cast_fp16)[name = string("op_6457_cast_fp16")]; tensor var_6461 = const()[name = string("op_6461"), val = tensor([0, 2, 1])]; tensor attn_output_21_cast_fp16 = transpose(perm = var_6461, x = var_6457_cast_fp16)[name = string("transpose_155")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = attn_output_21_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; int32 var_6476 = const()[name = string("op_6476"), val = int32(-1)]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6478_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_162_promoted_to_fp16)[name = string("op_6478_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_6476, interleave = input_191_interleave_0, values = (hidden_states_109_cast_fp16, var_6478_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_6473_to_fp16 = const()[name = string("op_6473_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_6473_to_fp16, x = input_191_cast_fp16)[name = string("normed_173_cast_fp16")]; tensor normed_175_begin_0 = const()[name = string("normed_175_begin_0"), val = tensor([0, 0, 0])]; tensor normed_175_end_0 = const()[name = string("normed_175_end_0"), val = tensor([1, 64, 1024])]; tensor normed_175_end_mask_0 = const()[name = string("normed_175_end_mask_0"), val = tensor([true, true, false])]; tensor normed_175_cast_fp16 = slice_by_index(begin = normed_175_begin_0, end = normed_175_end_0, end_mask = normed_175_end_mask_0, x = normed_173_cast_fp16)[name = string("normed_175_cast_fp16")]; tensor const_164_promoted_to_fp16 = const()[name = string("const_164_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310436608)))]; tensor x_41_cast_fp16 = mul(x = normed_175_cast_fp16, y = const_164_promoted_to_fp16)[name = string("x_41_cast_fp16")]; tensor var_6498 = const()[name = string("op_6498"), val = tensor([0, 2, 1])]; tensor input_193_axes_0 = const()[name = string("input_193_axes_0"), val = tensor([2])]; tensor var_6499 = transpose(perm = var_6498, x = x_41_cast_fp16)[name = string("transpose_154")]; tensor input_193 = expand_dims(axes = input_193_axes_0, x = var_6499)[name = string("input_193")]; string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; tensor input_195 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_193)[name = string("input_195")]; string b_21_pad_type_0 = const()[name = string("b_21_pad_type_0"), val = string("valid")]; tensor b_21_strides_0 = const()[name = string("b_21_strides_0"), val = tensor([1, 1])]; tensor b_21_pad_0 = const()[name = string("b_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_21_dilations_0 = const()[name = string("b_21_dilations_0"), val = tensor([1, 1])]; int32 b_21_groups_0 = const()[name = string("b_21_groups_0"), val = int32(1)]; tensor b_21 = conv(dilations = b_21_dilations_0, groups = b_21_groups_0, pad = b_21_pad_0, pad_type = b_21_pad_type_0, strides = b_21_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_193)[name = string("b_21")]; tensor c_21 = silu(x = input_195)[name = string("c_21")]; tensor input_197 = mul(x = c_21, y = b_21)[name = string("input_197")]; string e_21_pad_type_0 = const()[name = string("e_21_pad_type_0"), val = string("valid")]; tensor e_21_strides_0 = const()[name = string("e_21_strides_0"), val = tensor([1, 1])]; tensor e_21_pad_0 = const()[name = string("e_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_21_dilations_0 = const()[name = string("e_21_dilations_0"), val = tensor([1, 1])]; int32 e_21_groups_0 = const()[name = string("e_21_groups_0"), val = int32(1)]; tensor e_21 = conv(dilations = e_21_dilations_0, groups = e_21_groups_0, pad = e_21_pad_0, pad_type = e_21_pad_type_0, strides = e_21_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_197)[name = string("e_21")]; tensor var_6521_axes_0 = const()[name = string("op_6521_axes_0"), val = tensor([2])]; tensor var_6521 = squeeze(axes = var_6521_axes_0, x = e_21)[name = string("op_6521")]; tensor var_6522 = const()[name = string("op_6522"), val = tensor([0, 2, 1])]; tensor var_6523 = transpose(perm = var_6522, x = var_6521)[name = string("transpose_153")]; tensor hidden_states_111_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_6523)[name = string("hidden_states_111_cast_fp16")]; int32 var_6537 = const()[name = string("op_6537"), val = int32(-1)]; fp16 const_165_promoted_to_fp16 = const()[name = string("const_165_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6539_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_165_promoted_to_fp16)[name = string("op_6539_cast_fp16")]; bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; tensor input_199_cast_fp16 = concat(axis = var_6537, interleave = input_199_interleave_0, values = (hidden_states_111_cast_fp16, var_6539_cast_fp16))[name = string("input_199_cast_fp16")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_6534_to_fp16 = const()[name = string("op_6534_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_6534_to_fp16, x = input_199_cast_fp16)[name = string("normed_177_cast_fp16")]; tensor normed_179_begin_0 = const()[name = string("normed_179_begin_0"), val = tensor([0, 0, 0])]; tensor normed_179_end_0 = const()[name = string("normed_179_end_0"), val = tensor([1, 64, 1024])]; tensor normed_179_end_mask_0 = const()[name = string("normed_179_end_mask_0"), val = tensor([true, true, false])]; tensor normed_179_cast_fp16 = slice_by_index(begin = normed_179_begin_0, end = normed_179_end_0, end_mask = normed_179_end_mask_0, x = normed_177_cast_fp16)[name = string("normed_179_cast_fp16")]; tensor const_167_promoted_to_fp16 = const()[name = string("const_167_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310438720)))]; tensor hidden_states_113_cast_fp16 = mul(x = normed_179_cast_fp16, y = const_167_promoted_to_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor var_6551 = const()[name = string("op_6551"), val = tensor([0, 2, 1])]; tensor var_6554_axes_0 = const()[name = string("op_6554_axes_0"), val = tensor([2])]; tensor var_6552_cast_fp16 = transpose(perm = var_6551, x = hidden_states_113_cast_fp16)[name = string("transpose_152")]; tensor var_6554_cast_fp16 = expand_dims(axes = var_6554_axes_0, x = var_6552_cast_fp16)[name = string("op_6554_cast_fp16")]; string var_6570_pad_type_0 = const()[name = string("op_6570_pad_type_0"), val = string("valid")]; tensor var_6570_strides_0 = const()[name = string("op_6570_strides_0"), val = tensor([1, 1])]; tensor var_6570_pad_0 = const()[name = string("op_6570_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6570_dilations_0 = const()[name = string("op_6570_dilations_0"), val = tensor([1, 1])]; int32 var_6570_groups_0 = const()[name = string("op_6570_groups_0"), val = int32(1)]; tensor var_6570 = conv(dilations = var_6570_dilations_0, groups = var_6570_groups_0, pad = var_6570_pad_0, pad_type = var_6570_pad_type_0, strides = var_6570_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_6554_cast_fp16)[name = string("op_6570")]; tensor var_6575 = const()[name = string("op_6575"), val = tensor([1, 16, 128, 64])]; tensor var_6576 = reshape(shape = var_6575, x = var_6570)[name = string("op_6576")]; tensor var_6581 = const()[name = string("op_6581"), val = tensor([0, 1, 3, 2])]; string var_6593_pad_type_0 = const()[name = string("op_6593_pad_type_0"), val = string("valid")]; tensor var_6593_strides_0 = const()[name = string("op_6593_strides_0"), val = tensor([1, 1])]; tensor var_6593_pad_0 = const()[name = string("op_6593_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6593_dilations_0 = const()[name = string("op_6593_dilations_0"), val = tensor([1, 1])]; int32 var_6593_groups_0 = const()[name = string("op_6593_groups_0"), val = int32(1)]; tensor var_6593 = conv(dilations = var_6593_dilations_0, groups = var_6593_groups_0, pad = var_6593_pad_0, pad_type = var_6593_pad_type_0, strides = var_6593_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_6554_cast_fp16)[name = string("op_6593")]; tensor var_6598 = const()[name = string("op_6598"), val = tensor([1, 8, 128, 64])]; tensor var_6599 = reshape(shape = var_6598, x = var_6593)[name = string("op_6599")]; tensor var_6604 = const()[name = string("op_6604"), val = tensor([0, 1, 3, 2])]; string var_6616_pad_type_0 = const()[name = string("op_6616_pad_type_0"), val = string("valid")]; tensor var_6616_strides_0 = const()[name = string("op_6616_strides_0"), val = tensor([1, 1])]; tensor var_6616_pad_0 = const()[name = string("op_6616_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6616_dilations_0 = const()[name = string("op_6616_dilations_0"), val = tensor([1, 1])]; int32 var_6616_groups_0 = const()[name = string("op_6616_groups_0"), val = int32(1)]; tensor var_6616 = conv(dilations = var_6616_dilations_0, groups = var_6616_groups_0, pad = var_6616_pad_0, pad_type = var_6616_pad_type_0, strides = var_6616_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_6554_cast_fp16)[name = string("op_6616")]; tensor var_6621 = const()[name = string("op_6621"), val = tensor([1, 8, 128, 64])]; tensor var_6622 = reshape(shape = var_6621, x = var_6616)[name = string("op_6622")]; tensor var_6627 = const()[name = string("op_6627"), val = tensor([0, 1, 3, 2])]; int32 var_6640 = const()[name = string("op_6640"), val = int32(-1)]; fp16 const_168_promoted = const()[name = string("const_168_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_115 = transpose(perm = var_6581, x = var_6576)[name = string("transpose_151")]; tensor var_6642 = mul(x = hidden_states_115, y = const_168_promoted)[name = string("op_6642")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203 = concat(axis = var_6640, interleave = input_203_interleave_0, values = (hidden_states_115, var_6642))[name = string("input_203")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_6637_to_fp16 = const()[name = string("op_6637_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_6637_to_fp16, x = input_203)[name = string("normed_181_cast_fp16")]; tensor normed_183_begin_0 = const()[name = string("normed_183_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_183_end_0 = const()[name = string("normed_183_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_183_end_mask_0 = const()[name = string("normed_183_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_183 = slice_by_index(begin = normed_183_begin_0, end = normed_183_end_0, end_mask = normed_183_end_mask_0, x = normed_181_cast_fp16)[name = string("normed_183")]; tensor const_170 = const()[name = string("const_170"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310440832)))]; tensor q_23 = mul(x = normed_183, y = const_170)[name = string("q_23")]; int32 var_6662 = const()[name = string("op_6662"), val = int32(-1)]; fp16 const_171_promoted = const()[name = string("const_171_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_117 = transpose(perm = var_6604, x = var_6599)[name = string("transpose_150")]; tensor var_6664 = mul(x = hidden_states_117, y = const_171_promoted)[name = string("op_6664")]; bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; tensor input_205 = concat(axis = var_6662, interleave = input_205_interleave_0, values = (hidden_states_117, var_6664))[name = string("input_205")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_6659_to_fp16 = const()[name = string("op_6659_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_6659_to_fp16, x = input_205)[name = string("normed_185_cast_fp16")]; tensor normed_187_begin_0 = const()[name = string("normed_187_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_187_end_0 = const()[name = string("normed_187_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_187_end_mask_0 = const()[name = string("normed_187_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_187 = slice_by_index(begin = normed_187_begin_0, end = normed_187_end_0, end_mask = normed_187_end_mask_0, x = normed_185_cast_fp16)[name = string("normed_187")]; tensor const_173 = const()[name = string("const_173"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310441152)))]; tensor k_23 = mul(x = normed_187, y = const_173)[name = string("k_23")]; tensor var_6685 = mul(x = q_23, y = cos_1)[name = string("op_6685")]; tensor var_6690_begin_0 = const()[name = string("op_6690_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6690_end_0 = const()[name = string("op_6690_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_6690_end_mask_0 = const()[name = string("op_6690_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6690 = slice_by_index(begin = var_6690_begin_0, end = var_6690_end_0, end_mask = var_6690_end_mask_0, x = q_23)[name = string("op_6690")]; fp16 const_174_promoted = const()[name = string("const_174_promoted"), val = fp16(-0x1p+0)]; tensor var_6691 = mul(x = var_6690, y = const_174_promoted)[name = string("op_6691")]; tensor var_6696_begin_0 = const()[name = string("op_6696_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6696_end_0 = const()[name = string("op_6696_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_6696_end_mask_0 = const()[name = string("op_6696_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6696 = slice_by_index(begin = var_6696_begin_0, end = var_6696_end_0, end_mask = var_6696_end_mask_0, x = q_23)[name = string("op_6696")]; int32 var_6698 = const()[name = string("op_6698"), val = int32(-1)]; bool var_6699_interleave_0 = const()[name = string("op_6699_interleave_0"), val = bool(false)]; tensor var_6699 = concat(axis = var_6698, interleave = var_6699_interleave_0, values = (var_6691, var_6696))[name = string("op_6699")]; tensor var_6700 = mul(x = var_6699, y = sin_1)[name = string("op_6700")]; tensor query_23 = add(x = var_6685, y = var_6700)[name = string("query_23")]; tensor var_6703 = mul(x = k_23, y = cos_1)[name = string("op_6703")]; tensor var_6708_begin_0 = const()[name = string("op_6708_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_6708_end_0 = const()[name = string("op_6708_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_6708_end_mask_0 = const()[name = string("op_6708_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_6708 = slice_by_index(begin = var_6708_begin_0, end = var_6708_end_0, end_mask = var_6708_end_mask_0, x = k_23)[name = string("op_6708")]; fp16 const_175_promoted = const()[name = string("const_175_promoted"), val = fp16(-0x1p+0)]; tensor var_6709 = mul(x = var_6708, y = const_175_promoted)[name = string("op_6709")]; tensor var_6714_begin_0 = const()[name = string("op_6714_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_6714_end_0 = const()[name = string("op_6714_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_6714_end_mask_0 = const()[name = string("op_6714_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_6714 = slice_by_index(begin = var_6714_begin_0, end = var_6714_end_0, end_mask = var_6714_end_mask_0, x = k_23)[name = string("op_6714")]; int32 var_6716 = const()[name = string("op_6716"), val = int32(-1)]; bool var_6717_interleave_0 = const()[name = string("op_6717_interleave_0"), val = bool(false)]; tensor var_6717 = concat(axis = var_6716, interleave = var_6717_interleave_0, values = (var_6709, var_6714))[name = string("op_6717")]; tensor var_6718 = mul(x = var_6717, y = sin_1)[name = string("op_6718")]; tensor key_23 = add(x = var_6703, y = var_6718)[name = string("key_23")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; int32 concat_200_axis_0 = const()[name = string("concat_200_axis_0"), val = int32(0)]; bool concat_200_interleave_0 = const()[name = string("concat_200_interleave_0"), val = bool(false)]; tensor concat_200 = concat(axis = concat_200_axis_0, interleave = concat_200_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_200")]; tensor concat_201_values1_0 = const()[name = string("concat_201_values1_0"), val = tensor([0])]; tensor concat_201_values3_0 = const()[name = string("concat_201_values3_0"), val = tensor([0])]; int32 concat_201_axis_0 = const()[name = string("concat_201_axis_0"), val = int32(0)]; bool concat_201_interleave_0 = const()[name = string("concat_201_interleave_0"), val = bool(false)]; tensor concat_201 = concat(axis = concat_201_axis_0, interleave = concat_201_interleave_0, values = (expand_dims_136, concat_201_values1_0, var_1746, concat_201_values3_0))[name = string("concat_201")]; tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_200, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_201, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = key_23, x = coreml_update_state_77)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_190_write_state")]; tensor coreml_update_state_78 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_190")]; tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([39])]; tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([40])]; int32 concat_204_axis_0 = const()[name = string("concat_204_axis_0"), val = int32(0)]; bool concat_204_interleave_0 = const()[name = string("concat_204_interleave_0"), val = bool(false)]; tensor concat_204 = concat(axis = concat_204_axis_0, interleave = concat_204_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_204")]; tensor concat_205_values1_0 = const()[name = string("concat_205_values1_0"), val = tensor([0])]; tensor concat_205_values3_0 = const()[name = string("concat_205_values3_0"), val = tensor([0])]; int32 concat_205_axis_0 = const()[name = string("concat_205_axis_0"), val = int32(0)]; bool concat_205_interleave_0 = const()[name = string("concat_205_interleave_0"), val = bool(false)]; tensor concat_205 = concat(axis = concat_205_axis_0, interleave = concat_205_interleave_0, values = (expand_dims_142, concat_205_values1_0, var_1746, concat_205_values3_0))[name = string("concat_205")]; tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_111 = transpose(perm = var_6627, x = var_6622)[name = string("transpose_149")]; tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_204, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_205, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = value_111, x = coreml_update_state_78)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_191_write_state")]; tensor coreml_update_state_79 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_191")]; tensor var_6789_begin_0 = const()[name = string("op_6789_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_6789_end_0 = const()[name = string("op_6789_end_0"), val = tensor([12, 8, 1536, 128])]; tensor var_6789_end_mask_0 = const()[name = string("op_6789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6789_cast_fp16 = slice_by_index(begin = var_6789_begin_0, end = var_6789_end_0, end_mask = var_6789_end_mask_0, x = coreml_update_state_79)[name = string("op_6789_cast_fp16")]; tensor key_cache_23_axes_0 = const()[name = string("key_cache_23_axes_0"), val = tensor([0])]; tensor key_cache_23_cast_fp16 = squeeze(axes = key_cache_23_axes_0, x = var_6789_cast_fp16)[name = string("key_cache_23_cast_fp16")]; tensor var_6796_begin_0 = const()[name = string("op_6796_begin_0"), val = tensor([39, 0, 0, 0])]; tensor var_6796_end_0 = const()[name = string("op_6796_end_0"), val = tensor([40, 8, 1536, 128])]; tensor var_6796_end_mask_0 = const()[name = string("op_6796_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6796_cast_fp16 = slice_by_index(begin = var_6796_begin_0, end = var_6796_end_0, end_mask = var_6796_end_mask_0, x = coreml_update_state_79)[name = string("op_6796_cast_fp16")]; tensor value_cache_23_axes_0 = const()[name = string("value_cache_23_axes_0"), val = tensor([0])]; tensor value_cache_23_cast_fp16 = squeeze(axes = value_cache_23_axes_0, x = var_6796_cast_fp16)[name = string("value_cache_23_cast_fp16")]; tensor var_6820_axes_0 = const()[name = string("op_6820_axes_0"), val = tensor([1])]; tensor var_6820_cast_fp16 = expand_dims(axes = var_6820_axes_0, x = key_cache_23_cast_fp16)[name = string("op_6820_cast_fp16")]; tensor var_6825 = const()[name = string("op_6825"), val = tensor([1, 2, 1, 1])]; tensor value_115_cast_fp16 = tile(reps = var_6825, x = var_6820_cast_fp16)[name = string("value_115_cast_fp16")]; tensor var_6831 = const()[name = string("op_6831"), val = tensor([1, 16, 1536, 128])]; tensor key_states_47_cast_fp16 = reshape(shape = var_6831, x = value_115_cast_fp16)[name = string("key_states_47_cast_fp16")]; tensor var_6834_axes_0 = const()[name = string("op_6834_axes_0"), val = tensor([1])]; tensor var_6834_cast_fp16 = expand_dims(axes = var_6834_axes_0, x = value_cache_23_cast_fp16)[name = string("op_6834_cast_fp16")]; tensor var_6839 = const()[name = string("op_6839"), val = tensor([1, 2, 1, 1])]; tensor value_119_cast_fp16 = tile(reps = var_6839, x = var_6834_cast_fp16)[name = string("value_119_cast_fp16")]; bool var_6860_transpose_x_0 = const()[name = string("op_6860_transpose_x_0"), val = bool(false)]; bool var_6860_transpose_y_0 = const()[name = string("op_6860_transpose_y_0"), val = bool(true)]; tensor var_6860 = matmul(transpose_x = var_6860_transpose_x_0, transpose_y = var_6860_transpose_y_0, x = query_23, y = key_states_47_cast_fp16)[name = string("op_6860")]; fp16 var_6861_to_fp16 = const()[name = string("op_6861_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_45_cast_fp16 = mul(x = var_6860, y = var_6861_to_fp16)[name = string("attention_45_cast_fp16")]; tensor attention_47_cast_fp16 = add(x = attention_45_cast_fp16, y = causal_mask)[name = string("attention_47_cast_fp16")]; int32 var_6870 = const()[name = string("op_6870"), val = int32(-1)]; tensor var_6872_cast_fp16 = softmax(axis = var_6870, x = attention_47_cast_fp16)[name = string("op_6872_cast_fp16")]; tensor concat_210 = const()[name = string("concat_210"), val = tensor([16, 64, 1536])]; tensor reshape_33_cast_fp16 = reshape(shape = concat_210, x = var_6872_cast_fp16)[name = string("reshape_33_cast_fp16")]; tensor concat_211 = const()[name = string("concat_211"), val = tensor([16, 1536, 128])]; tensor reshape_34_cast_fp16 = reshape(shape = concat_211, x = value_119_cast_fp16)[name = string("reshape_34_cast_fp16")]; bool matmul_11_transpose_x_0 = const()[name = string("matmul_11_transpose_x_0"), val = bool(false)]; bool matmul_11_transpose_y_0 = const()[name = string("matmul_11_transpose_y_0"), val = bool(false)]; tensor matmul_11_cast_fp16 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = reshape_33_cast_fp16, y = reshape_34_cast_fp16)[name = string("matmul_11_cast_fp16")]; tensor concat_215 = const()[name = string("concat_215"), val = tensor([1, 16, 64, 128])]; tensor reshape_35_cast_fp16 = reshape(shape = concat_215, x = matmul_11_cast_fp16)[name = string("reshape_35_cast_fp16")]; tensor var_6884_perm_0 = const()[name = string("op_6884_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_6890 = const()[name = string("op_6890"), val = tensor([1, 64, 2048])]; tensor var_6884_cast_fp16 = transpose(perm = var_6884_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_148")]; tensor output_69_cast_fp16 = reshape(shape = var_6890, x = var_6884_cast_fp16)[name = string("output_69_cast_fp16")]; tensor var_6895 = const()[name = string("op_6895"), val = tensor([0, 2, 1])]; string var_6911_pad_type_0 = const()[name = string("op_6911_pad_type_0"), val = string("valid")]; int32 var_6911_groups_0 = const()[name = string("op_6911_groups_0"), val = int32(1)]; tensor var_6911_strides_0 = const()[name = string("op_6911_strides_0"), val = tensor([1])]; tensor var_6911_pad_0 = const()[name = string("op_6911_pad_0"), val = tensor([0, 0])]; tensor var_6911_dilations_0 = const()[name = string("op_6911_dilations_0"), val = tensor([1])]; tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310441472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312014400))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_6896_cast_fp16 = transpose(perm = var_6895, x = output_69_cast_fp16)[name = string("transpose_147")]; tensor var_6911_cast_fp16 = conv(dilations = var_6911_dilations_0, groups = var_6911_groups_0, pad = var_6911_pad_0, pad_type = var_6911_pad_type_0, strides = var_6911_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_6896_cast_fp16)[name = string("op_6911_cast_fp16")]; tensor var_6915 = const()[name = string("op_6915"), val = tensor([0, 2, 1])]; tensor attn_output_23_cast_fp16 = transpose(perm = var_6915, x = var_6911_cast_fp16)[name = string("transpose_146")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = attn_output_23_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; int32 var_6930 = const()[name = string("op_6930"), val = int32(-1)]; fp16 const_177_promoted_to_fp16 = const()[name = string("const_177_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6932_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_177_promoted_to_fp16)[name = string("op_6932_cast_fp16")]; bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; tensor input_209_cast_fp16 = concat(axis = var_6930, interleave = input_209_interleave_0, values = (hidden_states_119_cast_fp16, var_6932_cast_fp16))[name = string("input_209_cast_fp16")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_6927_to_fp16 = const()[name = string("op_6927_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_6927_to_fp16, x = input_209_cast_fp16)[name = string("normed_189_cast_fp16")]; tensor normed_191_begin_0 = const()[name = string("normed_191_begin_0"), val = tensor([0, 0, 0])]; tensor normed_191_end_0 = const()[name = string("normed_191_end_0"), val = tensor([1, 64, 1024])]; tensor normed_191_end_mask_0 = const()[name = string("normed_191_end_mask_0"), val = tensor([true, true, false])]; tensor normed_191_cast_fp16 = slice_by_index(begin = normed_191_begin_0, end = normed_191_end_0, end_mask = normed_191_end_mask_0, x = normed_189_cast_fp16)[name = string("normed_191_cast_fp16")]; tensor const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312030848)))]; tensor x_45_cast_fp16 = mul(x = normed_191_cast_fp16, y = const_179_promoted_to_fp16)[name = string("x_45_cast_fp16")]; tensor var_6952 = const()[name = string("op_6952"), val = tensor([0, 2, 1])]; tensor input_211_axes_0 = const()[name = string("input_211_axes_0"), val = tensor([2])]; tensor var_6953 = transpose(perm = var_6952, x = x_45_cast_fp16)[name = string("transpose_145")]; tensor input_211 = expand_dims(axes = input_211_axes_0, x = var_6953)[name = string("input_211")]; string input_213_pad_type_0 = const()[name = string("input_213_pad_type_0"), val = string("valid")]; tensor input_213_strides_0 = const()[name = string("input_213_strides_0"), val = tensor([1, 1])]; tensor input_213_pad_0 = const()[name = string("input_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_213_dilations_0 = const()[name = string("input_213_dilations_0"), val = tensor([1, 1])]; int32 input_213_groups_0 = const()[name = string("input_213_groups_0"), val = int32(1)]; tensor input_213 = conv(dilations = input_213_dilations_0, groups = input_213_groups_0, pad = input_213_pad_0, pad_type = input_213_pad_type_0, strides = input_213_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_211)[name = string("input_213")]; string b_23_pad_type_0 = const()[name = string("b_23_pad_type_0"), val = string("valid")]; tensor b_23_strides_0 = const()[name = string("b_23_strides_0"), val = tensor([1, 1])]; tensor b_23_pad_0 = const()[name = string("b_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_23_dilations_0 = const()[name = string("b_23_dilations_0"), val = tensor([1, 1])]; int32 b_23_groups_0 = const()[name = string("b_23_groups_0"), val = int32(1)]; tensor b_23 = conv(dilations = b_23_dilations_0, groups = b_23_groups_0, pad = b_23_pad_0, pad_type = b_23_pad_type_0, strides = b_23_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_211)[name = string("b_23")]; tensor c_23 = silu(x = input_213)[name = string("c_23")]; tensor input_215 = mul(x = c_23, y = b_23)[name = string("input_215")]; string e_23_pad_type_0 = const()[name = string("e_23_pad_type_0"), val = string("valid")]; tensor e_23_strides_0 = const()[name = string("e_23_strides_0"), val = tensor([1, 1])]; tensor e_23_pad_0 = const()[name = string("e_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_23_dilations_0 = const()[name = string("e_23_dilations_0"), val = tensor([1, 1])]; int32 e_23_groups_0 = const()[name = string("e_23_groups_0"), val = int32(1)]; tensor e_23 = conv(dilations = e_23_dilations_0, groups = e_23_groups_0, pad = e_23_pad_0, pad_type = e_23_pad_type_0, strides = e_23_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_215)[name = string("e_23")]; tensor var_6975_axes_0 = const()[name = string("op_6975_axes_0"), val = tensor([2])]; tensor var_6975 = squeeze(axes = var_6975_axes_0, x = e_23)[name = string("op_6975")]; tensor var_6976 = const()[name = string("op_6976"), val = tensor([0, 2, 1])]; tensor var_6977 = transpose(perm = var_6976, x = var_6975)[name = string("transpose_144")]; tensor hidden_states_121_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = var_6977)[name = string("hidden_states_121_cast_fp16")]; int32 var_6991 = const()[name = string("op_6991"), val = int32(-1)]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6993_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6993_cast_fp16")]; bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; tensor input_217_cast_fp16 = concat(axis = var_6991, interleave = input_217_interleave_0, values = (hidden_states_121_cast_fp16, var_6993_cast_fp16))[name = string("input_217_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_6988_to_fp16 = const()[name = string("op_6988_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_6988_to_fp16, x = input_217_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor normed_195_begin_0 = const()[name = string("normed_195_begin_0"), val = tensor([0, 0, 0])]; tensor normed_195_end_0 = const()[name = string("normed_195_end_0"), val = tensor([1, 64, 1024])]; tensor normed_195_end_mask_0 = const()[name = string("normed_195_end_mask_0"), val = tensor([true, true, false])]; tensor normed_195_cast_fp16 = slice_by_index(begin = normed_195_begin_0, end = normed_195_end_0, end_mask = normed_195_end_mask_0, x = normed_193_cast_fp16)[name = string("normed_195_cast_fp16")]; tensor const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312032960)))]; tensor hidden_states_123_cast_fp16 = mul(x = normed_195_cast_fp16, y = const_182_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor var_7005 = const()[name = string("op_7005"), val = tensor([0, 2, 1])]; tensor var_7008_axes_0 = const()[name = string("op_7008_axes_0"), val = tensor([2])]; tensor var_7006_cast_fp16 = transpose(perm = var_7005, x = hidden_states_123_cast_fp16)[name = string("transpose_143")]; tensor var_7008_cast_fp16 = expand_dims(axes = var_7008_axes_0, x = var_7006_cast_fp16)[name = string("op_7008_cast_fp16")]; string var_7024_pad_type_0 = const()[name = string("op_7024_pad_type_0"), val = string("valid")]; tensor var_7024_strides_0 = const()[name = string("op_7024_strides_0"), val = tensor([1, 1])]; tensor var_7024_pad_0 = const()[name = string("op_7024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7024_dilations_0 = const()[name = string("op_7024_dilations_0"), val = tensor([1, 1])]; int32 var_7024_groups_0 = const()[name = string("op_7024_groups_0"), val = int32(1)]; tensor var_7024 = conv(dilations = var_7024_dilations_0, groups = var_7024_groups_0, pad = var_7024_pad_0, pad_type = var_7024_pad_type_0, strides = var_7024_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_7008_cast_fp16)[name = string("op_7024")]; tensor var_7029 = const()[name = string("op_7029"), val = tensor([1, 16, 128, 64])]; tensor var_7030 = reshape(shape = var_7029, x = var_7024)[name = string("op_7030")]; tensor var_7035 = const()[name = string("op_7035"), val = tensor([0, 1, 3, 2])]; string var_7047_pad_type_0 = const()[name = string("op_7047_pad_type_0"), val = string("valid")]; tensor var_7047_strides_0 = const()[name = string("op_7047_strides_0"), val = tensor([1, 1])]; tensor var_7047_pad_0 = const()[name = string("op_7047_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7047_dilations_0 = const()[name = string("op_7047_dilations_0"), val = tensor([1, 1])]; int32 var_7047_groups_0 = const()[name = string("op_7047_groups_0"), val = int32(1)]; tensor var_7047 = conv(dilations = var_7047_dilations_0, groups = var_7047_groups_0, pad = var_7047_pad_0, pad_type = var_7047_pad_type_0, strides = var_7047_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_7008_cast_fp16)[name = string("op_7047")]; tensor var_7052 = const()[name = string("op_7052"), val = tensor([1, 8, 128, 64])]; tensor var_7053 = reshape(shape = var_7052, x = var_7047)[name = string("op_7053")]; tensor var_7058 = const()[name = string("op_7058"), val = tensor([0, 1, 3, 2])]; string var_7070_pad_type_0 = const()[name = string("op_7070_pad_type_0"), val = string("valid")]; tensor var_7070_strides_0 = const()[name = string("op_7070_strides_0"), val = tensor([1, 1])]; tensor var_7070_pad_0 = const()[name = string("op_7070_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7070_dilations_0 = const()[name = string("op_7070_dilations_0"), val = tensor([1, 1])]; int32 var_7070_groups_0 = const()[name = string("op_7070_groups_0"), val = int32(1)]; tensor var_7070 = conv(dilations = var_7070_dilations_0, groups = var_7070_groups_0, pad = var_7070_pad_0, pad_type = var_7070_pad_type_0, strides = var_7070_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_7008_cast_fp16)[name = string("op_7070")]; tensor var_7075 = const()[name = string("op_7075"), val = tensor([1, 8, 128, 64])]; tensor var_7076 = reshape(shape = var_7075, x = var_7070)[name = string("op_7076")]; tensor var_7081 = const()[name = string("op_7081"), val = tensor([0, 1, 3, 2])]; int32 var_7094 = const()[name = string("op_7094"), val = int32(-1)]; fp16 const_183_promoted = const()[name = string("const_183_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_125 = transpose(perm = var_7035, x = var_7030)[name = string("transpose_142")]; tensor var_7096 = mul(x = hidden_states_125, y = const_183_promoted)[name = string("op_7096")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221 = concat(axis = var_7094, interleave = input_221_interleave_0, values = (hidden_states_125, var_7096))[name = string("input_221")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_7091_to_fp16 = const()[name = string("op_7091_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_7091_to_fp16, x = input_221)[name = string("normed_197_cast_fp16")]; tensor normed_199_begin_0 = const()[name = string("normed_199_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_199_end_0 = const()[name = string("normed_199_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_199_end_mask_0 = const()[name = string("normed_199_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_199 = slice_by_index(begin = normed_199_begin_0, end = normed_199_end_0, end_mask = normed_199_end_mask_0, x = normed_197_cast_fp16)[name = string("normed_199")]; tensor const_185 = const()[name = string("const_185"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312035072)))]; tensor q_25 = mul(x = normed_199, y = const_185)[name = string("q_25")]; int32 var_7116 = const()[name = string("op_7116"), val = int32(-1)]; fp16 const_186_promoted = const()[name = string("const_186_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_127 = transpose(perm = var_7058, x = var_7053)[name = string("transpose_141")]; tensor var_7118 = mul(x = hidden_states_127, y = const_186_promoted)[name = string("op_7118")]; bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; tensor input_223 = concat(axis = var_7116, interleave = input_223_interleave_0, values = (hidden_states_127, var_7118))[name = string("input_223")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_7113_to_fp16 = const()[name = string("op_7113_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_7113_to_fp16, x = input_223)[name = string("normed_201_cast_fp16")]; tensor normed_203_begin_0 = const()[name = string("normed_203_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_203_end_0 = const()[name = string("normed_203_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_203_end_mask_0 = const()[name = string("normed_203_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_203 = slice_by_index(begin = normed_203_begin_0, end = normed_203_end_0, end_mask = normed_203_end_mask_0, x = normed_201_cast_fp16)[name = string("normed_203")]; tensor const_188 = const()[name = string("const_188"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312035392)))]; tensor k_25 = mul(x = normed_203, y = const_188)[name = string("k_25")]; tensor var_7139 = mul(x = q_25, y = cos_1)[name = string("op_7139")]; tensor var_7144_begin_0 = const()[name = string("op_7144_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7144_end_0 = const()[name = string("op_7144_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_7144_end_mask_0 = const()[name = string("op_7144_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7144 = slice_by_index(begin = var_7144_begin_0, end = var_7144_end_0, end_mask = var_7144_end_mask_0, x = q_25)[name = string("op_7144")]; fp16 const_189_promoted = const()[name = string("const_189_promoted"), val = fp16(-0x1p+0)]; tensor var_7145 = mul(x = var_7144, y = const_189_promoted)[name = string("op_7145")]; tensor var_7150_begin_0 = const()[name = string("op_7150_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7150_end_0 = const()[name = string("op_7150_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_7150_end_mask_0 = const()[name = string("op_7150_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7150 = slice_by_index(begin = var_7150_begin_0, end = var_7150_end_0, end_mask = var_7150_end_mask_0, x = q_25)[name = string("op_7150")]; int32 var_7152 = const()[name = string("op_7152"), val = int32(-1)]; bool var_7153_interleave_0 = const()[name = string("op_7153_interleave_0"), val = bool(false)]; tensor var_7153 = concat(axis = var_7152, interleave = var_7153_interleave_0, values = (var_7145, var_7150))[name = string("op_7153")]; tensor var_7154 = mul(x = var_7153, y = sin_1)[name = string("op_7154")]; tensor query_25 = add(x = var_7139, y = var_7154)[name = string("query_25")]; tensor var_7157 = mul(x = k_25, y = cos_1)[name = string("op_7157")]; tensor var_7162_begin_0 = const()[name = string("op_7162_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7162_end_0 = const()[name = string("op_7162_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_7162_end_mask_0 = const()[name = string("op_7162_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7162 = slice_by_index(begin = var_7162_begin_0, end = var_7162_end_0, end_mask = var_7162_end_mask_0, x = k_25)[name = string("op_7162")]; fp16 const_190_promoted = const()[name = string("const_190_promoted"), val = fp16(-0x1p+0)]; tensor var_7163 = mul(x = var_7162, y = const_190_promoted)[name = string("op_7163")]; tensor var_7168_begin_0 = const()[name = string("op_7168_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7168_end_0 = const()[name = string("op_7168_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_7168_end_mask_0 = const()[name = string("op_7168_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7168 = slice_by_index(begin = var_7168_begin_0, end = var_7168_end_0, end_mask = var_7168_end_mask_0, x = k_25)[name = string("op_7168")]; int32 var_7170 = const()[name = string("op_7170"), val = int32(-1)]; bool var_7171_interleave_0 = const()[name = string("op_7171_interleave_0"), val = bool(false)]; tensor var_7171 = concat(axis = var_7170, interleave = var_7171_interleave_0, values = (var_7163, var_7168))[name = string("op_7171")]; tensor var_7172 = mul(x = var_7171, y = sin_1)[name = string("op_7172")]; tensor key_25 = add(x = var_7157, y = var_7172)[name = string("key_25")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; int32 concat_218_axis_0 = const()[name = string("concat_218_axis_0"), val = int32(0)]; bool concat_218_interleave_0 = const()[name = string("concat_218_interleave_0"), val = bool(false)]; tensor concat_218 = concat(axis = concat_218_axis_0, interleave = concat_218_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_218")]; tensor concat_219_values1_0 = const()[name = string("concat_219_values1_0"), val = tensor([0])]; tensor concat_219_values3_0 = const()[name = string("concat_219_values3_0"), val = tensor([0])]; int32 concat_219_axis_0 = const()[name = string("concat_219_axis_0"), val = int32(0)]; bool concat_219_interleave_0 = const()[name = string("concat_219_interleave_0"), val = bool(false)]; tensor concat_219 = concat(axis = concat_219_axis_0, interleave = concat_219_interleave_0, values = (expand_dims_148, concat_219_values1_0, var_1746, concat_219_values3_0))[name = string("concat_219")]; tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_218, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_219, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = key_25, x = coreml_update_state_79)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_192_write_state")]; tensor coreml_update_state_80 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_192")]; tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([40])]; tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([41])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_222")]; tensor concat_223_values1_0 = const()[name = string("concat_223_values1_0"), val = tensor([0])]; tensor concat_223_values3_0 = const()[name = string("concat_223_values3_0"), val = tensor([0])]; int32 concat_223_axis_0 = const()[name = string("concat_223_axis_0"), val = int32(0)]; bool concat_223_interleave_0 = const()[name = string("concat_223_interleave_0"), val = bool(false)]; tensor concat_223 = concat(axis = concat_223_axis_0, interleave = concat_223_interleave_0, values = (expand_dims_154, concat_223_values1_0, var_1746, concat_223_values3_0))[name = string("concat_223")]; tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_121 = transpose(perm = var_7081, x = var_7076)[name = string("transpose_140")]; tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_222, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_223, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = value_121, x = coreml_update_state_80)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_193_write_state")]; tensor coreml_update_state_81 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_193")]; tensor var_7243_begin_0 = const()[name = string("op_7243_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_7243_end_0 = const()[name = string("op_7243_end_0"), val = tensor([13, 8, 1536, 128])]; tensor var_7243_end_mask_0 = const()[name = string("op_7243_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7243_cast_fp16 = slice_by_index(begin = var_7243_begin_0, end = var_7243_end_0, end_mask = var_7243_end_mask_0, x = coreml_update_state_81)[name = string("op_7243_cast_fp16")]; tensor key_cache_25_axes_0 = const()[name = string("key_cache_25_axes_0"), val = tensor([0])]; tensor key_cache_25_cast_fp16 = squeeze(axes = key_cache_25_axes_0, x = var_7243_cast_fp16)[name = string("key_cache_25_cast_fp16")]; tensor var_7250_begin_0 = const()[name = string("op_7250_begin_0"), val = tensor([40, 0, 0, 0])]; tensor var_7250_end_0 = const()[name = string("op_7250_end_0"), val = tensor([41, 8, 1536, 128])]; tensor var_7250_end_mask_0 = const()[name = string("op_7250_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7250_cast_fp16 = slice_by_index(begin = var_7250_begin_0, end = var_7250_end_0, end_mask = var_7250_end_mask_0, x = coreml_update_state_81)[name = string("op_7250_cast_fp16")]; tensor value_cache_25_axes_0 = const()[name = string("value_cache_25_axes_0"), val = tensor([0])]; tensor value_cache_25_cast_fp16 = squeeze(axes = value_cache_25_axes_0, x = var_7250_cast_fp16)[name = string("value_cache_25_cast_fp16")]; tensor var_7274_axes_0 = const()[name = string("op_7274_axes_0"), val = tensor([1])]; tensor var_7274_cast_fp16 = expand_dims(axes = var_7274_axes_0, x = key_cache_25_cast_fp16)[name = string("op_7274_cast_fp16")]; tensor var_7279 = const()[name = string("op_7279"), val = tensor([1, 2, 1, 1])]; tensor value_125_cast_fp16 = tile(reps = var_7279, x = var_7274_cast_fp16)[name = string("value_125_cast_fp16")]; tensor var_7285 = const()[name = string("op_7285"), val = tensor([1, 16, 1536, 128])]; tensor key_states_51_cast_fp16 = reshape(shape = var_7285, x = value_125_cast_fp16)[name = string("key_states_51_cast_fp16")]; tensor var_7288_axes_0 = const()[name = string("op_7288_axes_0"), val = tensor([1])]; tensor var_7288_cast_fp16 = expand_dims(axes = var_7288_axes_0, x = value_cache_25_cast_fp16)[name = string("op_7288_cast_fp16")]; tensor var_7293 = const()[name = string("op_7293"), val = tensor([1, 2, 1, 1])]; tensor value_129_cast_fp16 = tile(reps = var_7293, x = var_7288_cast_fp16)[name = string("value_129_cast_fp16")]; bool var_7314_transpose_x_0 = const()[name = string("op_7314_transpose_x_0"), val = bool(false)]; bool var_7314_transpose_y_0 = const()[name = string("op_7314_transpose_y_0"), val = bool(true)]; tensor var_7314 = matmul(transpose_x = var_7314_transpose_x_0, transpose_y = var_7314_transpose_y_0, x = query_25, y = key_states_51_cast_fp16)[name = string("op_7314")]; fp16 var_7315_to_fp16 = const()[name = string("op_7315_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_49_cast_fp16 = mul(x = var_7314, y = var_7315_to_fp16)[name = string("attention_49_cast_fp16")]; tensor attention_51_cast_fp16 = add(x = attention_49_cast_fp16, y = causal_mask)[name = string("attention_51_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(-1)]; tensor var_7326_cast_fp16 = softmax(axis = var_7324, x = attention_51_cast_fp16)[name = string("op_7326_cast_fp16")]; tensor concat_228 = const()[name = string("concat_228"), val = tensor([16, 64, 1536])]; tensor reshape_36_cast_fp16 = reshape(shape = concat_228, x = var_7326_cast_fp16)[name = string("reshape_36_cast_fp16")]; tensor concat_229 = const()[name = string("concat_229"), val = tensor([16, 1536, 128])]; tensor reshape_37_cast_fp16 = reshape(shape = concat_229, x = value_129_cast_fp16)[name = string("reshape_37_cast_fp16")]; bool matmul_12_transpose_x_0 = const()[name = string("matmul_12_transpose_x_0"), val = bool(false)]; bool matmul_12_transpose_y_0 = const()[name = string("matmul_12_transpose_y_0"), val = bool(false)]; tensor matmul_12_cast_fp16 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = reshape_36_cast_fp16, y = reshape_37_cast_fp16)[name = string("matmul_12_cast_fp16")]; tensor concat_233 = const()[name = string("concat_233"), val = tensor([1, 16, 64, 128])]; tensor reshape_38_cast_fp16 = reshape(shape = concat_233, x = matmul_12_cast_fp16)[name = string("reshape_38_cast_fp16")]; tensor var_7338_perm_0 = const()[name = string("op_7338_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_7344 = const()[name = string("op_7344"), val = tensor([1, 64, 2048])]; tensor var_7338_cast_fp16 = transpose(perm = var_7338_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_139")]; tensor output_75_cast_fp16 = reshape(shape = var_7344, x = var_7338_cast_fp16)[name = string("output_75_cast_fp16")]; tensor var_7349 = const()[name = string("op_7349"), val = tensor([0, 2, 1])]; string var_7365_pad_type_0 = const()[name = string("op_7365_pad_type_0"), val = string("valid")]; int32 var_7365_groups_0 = const()[name = string("op_7365_groups_0"), val = int32(1)]; tensor var_7365_strides_0 = const()[name = string("op_7365_strides_0"), val = tensor([1])]; tensor var_7365_pad_0 = const()[name = string("op_7365_pad_0"), val = tensor([0, 0])]; tensor var_7365_dilations_0 = const()[name = string("op_7365_dilations_0"), val = tensor([1])]; tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312035712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313608640))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7350_cast_fp16 = transpose(perm = var_7349, x = output_75_cast_fp16)[name = string("transpose_138")]; tensor var_7365_cast_fp16 = conv(dilations = var_7365_dilations_0, groups = var_7365_groups_0, pad = var_7365_pad_0, pad_type = var_7365_pad_type_0, strides = var_7365_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_7350_cast_fp16)[name = string("op_7365_cast_fp16")]; tensor var_7369 = const()[name = string("op_7369"), val = tensor([0, 2, 1])]; tensor attn_output_25_cast_fp16 = transpose(perm = var_7369, x = var_7365_cast_fp16)[name = string("transpose_137")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = attn_output_25_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; int32 var_7384 = const()[name = string("op_7384"), val = int32(-1)]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7386_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_192_promoted_to_fp16)[name = string("op_7386_cast_fp16")]; bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; tensor input_227_cast_fp16 = concat(axis = var_7384, interleave = input_227_interleave_0, values = (hidden_states_129_cast_fp16, var_7386_cast_fp16))[name = string("input_227_cast_fp16")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_7381_to_fp16 = const()[name = string("op_7381_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_7381_to_fp16, x = input_227_cast_fp16)[name = string("normed_205_cast_fp16")]; tensor normed_207_begin_0 = const()[name = string("normed_207_begin_0"), val = tensor([0, 0, 0])]; tensor normed_207_end_0 = const()[name = string("normed_207_end_0"), val = tensor([1, 64, 1024])]; tensor normed_207_end_mask_0 = const()[name = string("normed_207_end_mask_0"), val = tensor([true, true, false])]; tensor normed_207_cast_fp16 = slice_by_index(begin = normed_207_begin_0, end = normed_207_end_0, end_mask = normed_207_end_mask_0, x = normed_205_cast_fp16)[name = string("normed_207_cast_fp16")]; tensor const_194_promoted_to_fp16 = const()[name = string("const_194_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313625088)))]; tensor x_49_cast_fp16 = mul(x = normed_207_cast_fp16, y = const_194_promoted_to_fp16)[name = string("x_49_cast_fp16")]; tensor var_7406 = const()[name = string("op_7406"), val = tensor([0, 2, 1])]; tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; tensor var_7407 = transpose(perm = var_7406, x = x_49_cast_fp16)[name = string("transpose_136")]; tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_7407)[name = string("input_229")]; string input_231_pad_type_0 = const()[name = string("input_231_pad_type_0"), val = string("valid")]; tensor input_231_strides_0 = const()[name = string("input_231_strides_0"), val = tensor([1, 1])]; tensor input_231_pad_0 = const()[name = string("input_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_231_dilations_0 = const()[name = string("input_231_dilations_0"), val = tensor([1, 1])]; int32 input_231_groups_0 = const()[name = string("input_231_groups_0"), val = int32(1)]; tensor input_231 = conv(dilations = input_231_dilations_0, groups = input_231_groups_0, pad = input_231_pad_0, pad_type = input_231_pad_type_0, strides = input_231_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_229)[name = string("input_231")]; string b_25_pad_type_0 = const()[name = string("b_25_pad_type_0"), val = string("valid")]; tensor b_25_strides_0 = const()[name = string("b_25_strides_0"), val = tensor([1, 1])]; tensor b_25_pad_0 = const()[name = string("b_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_25_dilations_0 = const()[name = string("b_25_dilations_0"), val = tensor([1, 1])]; int32 b_25_groups_0 = const()[name = string("b_25_groups_0"), val = int32(1)]; tensor b_25 = conv(dilations = b_25_dilations_0, groups = b_25_groups_0, pad = b_25_pad_0, pad_type = b_25_pad_type_0, strides = b_25_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_229)[name = string("b_25")]; tensor c_25 = silu(x = input_231)[name = string("c_25")]; tensor input_233 = mul(x = c_25, y = b_25)[name = string("input_233")]; string e_25_pad_type_0 = const()[name = string("e_25_pad_type_0"), val = string("valid")]; tensor e_25_strides_0 = const()[name = string("e_25_strides_0"), val = tensor([1, 1])]; tensor e_25_pad_0 = const()[name = string("e_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_25_dilations_0 = const()[name = string("e_25_dilations_0"), val = tensor([1, 1])]; int32 e_25_groups_0 = const()[name = string("e_25_groups_0"), val = int32(1)]; tensor e_25 = conv(dilations = e_25_dilations_0, groups = e_25_groups_0, pad = e_25_pad_0, pad_type = e_25_pad_type_0, strides = e_25_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_233)[name = string("e_25")]; tensor var_7429_axes_0 = const()[name = string("op_7429_axes_0"), val = tensor([2])]; tensor var_7429 = squeeze(axes = var_7429_axes_0, x = e_25)[name = string("op_7429")]; tensor var_7430 = const()[name = string("op_7430"), val = tensor([0, 2, 1])]; tensor var_7431 = transpose(perm = var_7430, x = var_7429)[name = string("transpose_135")]; tensor hidden_states_131_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_7431)[name = string("hidden_states_131_cast_fp16")]; int32 var_7445 = const()[name = string("op_7445"), val = int32(-1)]; fp16 const_195_promoted_to_fp16 = const()[name = string("const_195_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7447_cast_fp16 = mul(x = hidden_states_131_cast_fp16, y = const_195_promoted_to_fp16)[name = string("op_7447_cast_fp16")]; bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; tensor input_235_cast_fp16 = concat(axis = var_7445, interleave = input_235_interleave_0, values = (hidden_states_131_cast_fp16, var_7447_cast_fp16))[name = string("input_235_cast_fp16")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_7442_to_fp16 = const()[name = string("op_7442_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_7442_to_fp16, x = input_235_cast_fp16)[name = string("normed_209_cast_fp16")]; tensor normed_211_begin_0 = const()[name = string("normed_211_begin_0"), val = tensor([0, 0, 0])]; tensor normed_211_end_0 = const()[name = string("normed_211_end_0"), val = tensor([1, 64, 1024])]; tensor normed_211_end_mask_0 = const()[name = string("normed_211_end_mask_0"), val = tensor([true, true, false])]; tensor normed_211_cast_fp16 = slice_by_index(begin = normed_211_begin_0, end = normed_211_end_0, end_mask = normed_211_end_mask_0, x = normed_209_cast_fp16)[name = string("normed_211_cast_fp16")]; tensor const_197_promoted_to_fp16 = const()[name = string("const_197_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313627200)))]; tensor hidden_states_133_cast_fp16 = mul(x = normed_211_cast_fp16, y = const_197_promoted_to_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor var_7459 = const()[name = string("op_7459"), val = tensor([0, 2, 1])]; tensor var_7462_axes_0 = const()[name = string("op_7462_axes_0"), val = tensor([2])]; tensor var_7460_cast_fp16 = transpose(perm = var_7459, x = hidden_states_133_cast_fp16)[name = string("transpose_134")]; tensor var_7462_cast_fp16 = expand_dims(axes = var_7462_axes_0, x = var_7460_cast_fp16)[name = string("op_7462_cast_fp16")]; string var_7478_pad_type_0 = const()[name = string("op_7478_pad_type_0"), val = string("valid")]; tensor var_7478_strides_0 = const()[name = string("op_7478_strides_0"), val = tensor([1, 1])]; tensor var_7478_pad_0 = const()[name = string("op_7478_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7478_dilations_0 = const()[name = string("op_7478_dilations_0"), val = tensor([1, 1])]; int32 var_7478_groups_0 = const()[name = string("op_7478_groups_0"), val = int32(1)]; tensor var_7478 = conv(dilations = var_7478_dilations_0, groups = var_7478_groups_0, pad = var_7478_pad_0, pad_type = var_7478_pad_type_0, strides = var_7478_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_7462_cast_fp16)[name = string("op_7478")]; tensor var_7483 = const()[name = string("op_7483"), val = tensor([1, 16, 128, 64])]; tensor var_7484 = reshape(shape = var_7483, x = var_7478)[name = string("op_7484")]; tensor var_7489 = const()[name = string("op_7489"), val = tensor([0, 1, 3, 2])]; string var_7501_pad_type_0 = const()[name = string("op_7501_pad_type_0"), val = string("valid")]; tensor var_7501_strides_0 = const()[name = string("op_7501_strides_0"), val = tensor([1, 1])]; tensor var_7501_pad_0 = const()[name = string("op_7501_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7501_dilations_0 = const()[name = string("op_7501_dilations_0"), val = tensor([1, 1])]; int32 var_7501_groups_0 = const()[name = string("op_7501_groups_0"), val = int32(1)]; tensor var_7501 = conv(dilations = var_7501_dilations_0, groups = var_7501_groups_0, pad = var_7501_pad_0, pad_type = var_7501_pad_type_0, strides = var_7501_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_7462_cast_fp16)[name = string("op_7501")]; tensor var_7506 = const()[name = string("op_7506"), val = tensor([1, 8, 128, 64])]; tensor var_7507 = reshape(shape = var_7506, x = var_7501)[name = string("op_7507")]; tensor var_7512 = const()[name = string("op_7512"), val = tensor([0, 1, 3, 2])]; string var_7524_pad_type_0 = const()[name = string("op_7524_pad_type_0"), val = string("valid")]; tensor var_7524_strides_0 = const()[name = string("op_7524_strides_0"), val = tensor([1, 1])]; tensor var_7524_pad_0 = const()[name = string("op_7524_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7524_dilations_0 = const()[name = string("op_7524_dilations_0"), val = tensor([1, 1])]; int32 var_7524_groups_0 = const()[name = string("op_7524_groups_0"), val = int32(1)]; tensor var_7524 = conv(dilations = var_7524_dilations_0, groups = var_7524_groups_0, pad = var_7524_pad_0, pad_type = var_7524_pad_type_0, strides = var_7524_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_7462_cast_fp16)[name = string("op_7524")]; tensor var_7529 = const()[name = string("op_7529"), val = tensor([1, 8, 128, 64])]; tensor var_7530 = reshape(shape = var_7529, x = var_7524)[name = string("op_7530")]; tensor var_7535 = const()[name = string("op_7535"), val = tensor([0, 1, 3, 2])]; int32 var_7548 = const()[name = string("op_7548"), val = int32(-1)]; fp16 const_198_promoted = const()[name = string("const_198_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_135 = transpose(perm = var_7489, x = var_7484)[name = string("transpose_133")]; tensor var_7550 = mul(x = hidden_states_135, y = const_198_promoted)[name = string("op_7550")]; bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; tensor input_239 = concat(axis = var_7548, interleave = input_239_interleave_0, values = (hidden_states_135, var_7550))[name = string("input_239")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_7545_to_fp16 = const()[name = string("op_7545_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_7545_to_fp16, x = input_239)[name = string("normed_213_cast_fp16")]; tensor normed_215_begin_0 = const()[name = string("normed_215_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_215_end_0 = const()[name = string("normed_215_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_215_end_mask_0 = const()[name = string("normed_215_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_215 = slice_by_index(begin = normed_215_begin_0, end = normed_215_end_0, end_mask = normed_215_end_mask_0, x = normed_213_cast_fp16)[name = string("normed_215")]; tensor const_200 = const()[name = string("const_200"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313629312)))]; tensor q_27 = mul(x = normed_215, y = const_200)[name = string("q_27")]; int32 var_7570 = const()[name = string("op_7570"), val = int32(-1)]; fp16 const_201_promoted = const()[name = string("const_201_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_137 = transpose(perm = var_7512, x = var_7507)[name = string("transpose_132")]; tensor var_7572 = mul(x = hidden_states_137, y = const_201_promoted)[name = string("op_7572")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241 = concat(axis = var_7570, interleave = input_241_interleave_0, values = (hidden_states_137, var_7572))[name = string("input_241")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_7567_to_fp16 = const()[name = string("op_7567_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_7567_to_fp16, x = input_241)[name = string("normed_217_cast_fp16")]; tensor normed_219_begin_0 = const()[name = string("normed_219_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_219_end_0 = const()[name = string("normed_219_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_219_end_mask_0 = const()[name = string("normed_219_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_219 = slice_by_index(begin = normed_219_begin_0, end = normed_219_end_0, end_mask = normed_219_end_mask_0, x = normed_217_cast_fp16)[name = string("normed_219")]; tensor const_203 = const()[name = string("const_203"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313629632)))]; tensor k_27 = mul(x = normed_219, y = const_203)[name = string("k_27")]; tensor var_7593 = mul(x = q_27, y = cos_1)[name = string("op_7593")]; tensor var_7598_begin_0 = const()[name = string("op_7598_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7598_end_0 = const()[name = string("op_7598_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_7598_end_mask_0 = const()[name = string("op_7598_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7598 = slice_by_index(begin = var_7598_begin_0, end = var_7598_end_0, end_mask = var_7598_end_mask_0, x = q_27)[name = string("op_7598")]; fp16 const_204_promoted = const()[name = string("const_204_promoted"), val = fp16(-0x1p+0)]; tensor var_7599 = mul(x = var_7598, y = const_204_promoted)[name = string("op_7599")]; tensor var_7604_begin_0 = const()[name = string("op_7604_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7604_end_0 = const()[name = string("op_7604_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_7604_end_mask_0 = const()[name = string("op_7604_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7604 = slice_by_index(begin = var_7604_begin_0, end = var_7604_end_0, end_mask = var_7604_end_mask_0, x = q_27)[name = string("op_7604")]; int32 var_7606 = const()[name = string("op_7606"), val = int32(-1)]; bool var_7607_interleave_0 = const()[name = string("op_7607_interleave_0"), val = bool(false)]; tensor var_7607 = concat(axis = var_7606, interleave = var_7607_interleave_0, values = (var_7599, var_7604))[name = string("op_7607")]; tensor var_7608 = mul(x = var_7607, y = sin_1)[name = string("op_7608")]; tensor query_27 = add(x = var_7593, y = var_7608)[name = string("query_27")]; tensor var_7611 = mul(x = k_27, y = cos_1)[name = string("op_7611")]; tensor var_7616_begin_0 = const()[name = string("op_7616_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_7616_end_0 = const()[name = string("op_7616_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_7616_end_mask_0 = const()[name = string("op_7616_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_7616 = slice_by_index(begin = var_7616_begin_0, end = var_7616_end_0, end_mask = var_7616_end_mask_0, x = k_27)[name = string("op_7616")]; fp16 const_205_promoted = const()[name = string("const_205_promoted"), val = fp16(-0x1p+0)]; tensor var_7617 = mul(x = var_7616, y = const_205_promoted)[name = string("op_7617")]; tensor var_7622_begin_0 = const()[name = string("op_7622_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_7622_end_0 = const()[name = string("op_7622_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_7622_end_mask_0 = const()[name = string("op_7622_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_7622 = slice_by_index(begin = var_7622_begin_0, end = var_7622_end_0, end_mask = var_7622_end_mask_0, x = k_27)[name = string("op_7622")]; int32 var_7624 = const()[name = string("op_7624"), val = int32(-1)]; bool var_7625_interleave_0 = const()[name = string("op_7625_interleave_0"), val = bool(false)]; tensor var_7625 = concat(axis = var_7624, interleave = var_7625_interleave_0, values = (var_7617, var_7622))[name = string("op_7625")]; tensor var_7626 = mul(x = var_7625, y = sin_1)[name = string("op_7626")]; tensor key_27 = add(x = var_7611, y = var_7626)[name = string("key_27")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; int32 concat_236_axis_0 = const()[name = string("concat_236_axis_0"), val = int32(0)]; bool concat_236_interleave_0 = const()[name = string("concat_236_interleave_0"), val = bool(false)]; tensor concat_236 = concat(axis = concat_236_axis_0, interleave = concat_236_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_236")]; tensor concat_237_values1_0 = const()[name = string("concat_237_values1_0"), val = tensor([0])]; tensor concat_237_values3_0 = const()[name = string("concat_237_values3_0"), val = tensor([0])]; int32 concat_237_axis_0 = const()[name = string("concat_237_axis_0"), val = int32(0)]; bool concat_237_interleave_0 = const()[name = string("concat_237_interleave_0"), val = bool(false)]; tensor concat_237 = concat(axis = concat_237_axis_0, interleave = concat_237_interleave_0, values = (expand_dims_160, concat_237_values1_0, var_1746, concat_237_values3_0))[name = string("concat_237")]; tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_236, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_237, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = key_27, x = coreml_update_state_81)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_194_write_state")]; tensor coreml_update_state_82 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_194")]; tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([41])]; tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([42])]; int32 concat_240_axis_0 = const()[name = string("concat_240_axis_0"), val = int32(0)]; bool concat_240_interleave_0 = const()[name = string("concat_240_interleave_0"), val = bool(false)]; tensor concat_240 = concat(axis = concat_240_axis_0, interleave = concat_240_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_240")]; tensor concat_241_values1_0 = const()[name = string("concat_241_values1_0"), val = tensor([0])]; tensor concat_241_values3_0 = const()[name = string("concat_241_values3_0"), val = tensor([0])]; int32 concat_241_axis_0 = const()[name = string("concat_241_axis_0"), val = int32(0)]; bool concat_241_interleave_0 = const()[name = string("concat_241_interleave_0"), val = bool(false)]; tensor concat_241 = concat(axis = concat_241_axis_0, interleave = concat_241_interleave_0, values = (expand_dims_166, concat_241_values1_0, var_1746, concat_241_values3_0))[name = string("concat_241")]; tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_131 = transpose(perm = var_7535, x = var_7530)[name = string("transpose_131")]; tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_240, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_241, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = value_131, x = coreml_update_state_82)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_195_write_state")]; tensor coreml_update_state_83 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_195")]; tensor var_7697_begin_0 = const()[name = string("op_7697_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_7697_end_0 = const()[name = string("op_7697_end_0"), val = tensor([14, 8, 1536, 128])]; tensor var_7697_end_mask_0 = const()[name = string("op_7697_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7697_cast_fp16 = slice_by_index(begin = var_7697_begin_0, end = var_7697_end_0, end_mask = var_7697_end_mask_0, x = coreml_update_state_83)[name = string("op_7697_cast_fp16")]; tensor key_cache_27_axes_0 = const()[name = string("key_cache_27_axes_0"), val = tensor([0])]; tensor key_cache_27_cast_fp16 = squeeze(axes = key_cache_27_axes_0, x = var_7697_cast_fp16)[name = string("key_cache_27_cast_fp16")]; tensor var_7704_begin_0 = const()[name = string("op_7704_begin_0"), val = tensor([41, 0, 0, 0])]; tensor var_7704_end_0 = const()[name = string("op_7704_end_0"), val = tensor([42, 8, 1536, 128])]; tensor var_7704_end_mask_0 = const()[name = string("op_7704_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7704_cast_fp16 = slice_by_index(begin = var_7704_begin_0, end = var_7704_end_0, end_mask = var_7704_end_mask_0, x = coreml_update_state_83)[name = string("op_7704_cast_fp16")]; tensor value_cache_27_axes_0 = const()[name = string("value_cache_27_axes_0"), val = tensor([0])]; tensor value_cache_27_cast_fp16 = squeeze(axes = value_cache_27_axes_0, x = var_7704_cast_fp16)[name = string("value_cache_27_cast_fp16")]; tensor var_7728_axes_0 = const()[name = string("op_7728_axes_0"), val = tensor([1])]; tensor var_7728_cast_fp16 = expand_dims(axes = var_7728_axes_0, x = key_cache_27_cast_fp16)[name = string("op_7728_cast_fp16")]; tensor var_7733 = const()[name = string("op_7733"), val = tensor([1, 2, 1, 1])]; tensor value_135_cast_fp16 = tile(reps = var_7733, x = var_7728_cast_fp16)[name = string("value_135_cast_fp16")]; tensor var_7739 = const()[name = string("op_7739"), val = tensor([1, 16, 1536, 128])]; tensor key_states_55_cast_fp16 = reshape(shape = var_7739, x = value_135_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor var_7742_axes_0 = const()[name = string("op_7742_axes_0"), val = tensor([1])]; tensor var_7742_cast_fp16 = expand_dims(axes = var_7742_axes_0, x = value_cache_27_cast_fp16)[name = string("op_7742_cast_fp16")]; tensor var_7747 = const()[name = string("op_7747"), val = tensor([1, 2, 1, 1])]; tensor value_139_cast_fp16 = tile(reps = var_7747, x = var_7742_cast_fp16)[name = string("value_139_cast_fp16")]; bool var_7768_transpose_x_0 = const()[name = string("op_7768_transpose_x_0"), val = bool(false)]; bool var_7768_transpose_y_0 = const()[name = string("op_7768_transpose_y_0"), val = bool(true)]; tensor var_7768 = matmul(transpose_x = var_7768_transpose_x_0, transpose_y = var_7768_transpose_y_0, x = query_27, y = key_states_55_cast_fp16)[name = string("op_7768")]; fp16 var_7769_to_fp16 = const()[name = string("op_7769_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_53_cast_fp16 = mul(x = var_7768, y = var_7769_to_fp16)[name = string("attention_53_cast_fp16")]; tensor attention_55_cast_fp16 = add(x = attention_53_cast_fp16, y = causal_mask)[name = string("attention_55_cast_fp16")]; int32 var_7778 = const()[name = string("op_7778"), val = int32(-1)]; tensor var_7780_cast_fp16 = softmax(axis = var_7778, x = attention_55_cast_fp16)[name = string("op_7780_cast_fp16")]; tensor concat_246 = const()[name = string("concat_246"), val = tensor([16, 64, 1536])]; tensor reshape_39_cast_fp16 = reshape(shape = concat_246, x = var_7780_cast_fp16)[name = string("reshape_39_cast_fp16")]; tensor concat_247 = const()[name = string("concat_247"), val = tensor([16, 1536, 128])]; tensor reshape_40_cast_fp16 = reshape(shape = concat_247, x = value_139_cast_fp16)[name = string("reshape_40_cast_fp16")]; bool matmul_13_transpose_x_0 = const()[name = string("matmul_13_transpose_x_0"), val = bool(false)]; bool matmul_13_transpose_y_0 = const()[name = string("matmul_13_transpose_y_0"), val = bool(false)]; tensor matmul_13_cast_fp16 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = reshape_39_cast_fp16, y = reshape_40_cast_fp16)[name = string("matmul_13_cast_fp16")]; tensor concat_251 = const()[name = string("concat_251"), val = tensor([1, 16, 64, 128])]; tensor reshape_41_cast_fp16 = reshape(shape = concat_251, x = matmul_13_cast_fp16)[name = string("reshape_41_cast_fp16")]; tensor var_7792_perm_0 = const()[name = string("op_7792_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_7798 = const()[name = string("op_7798"), val = tensor([1, 64, 2048])]; tensor var_7792_cast_fp16 = transpose(perm = var_7792_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_130")]; tensor output_81_cast_fp16 = reshape(shape = var_7798, x = var_7792_cast_fp16)[name = string("output_81_cast_fp16")]; tensor var_7803 = const()[name = string("op_7803"), val = tensor([0, 2, 1])]; string var_7819_pad_type_0 = const()[name = string("op_7819_pad_type_0"), val = string("valid")]; int32 var_7819_groups_0 = const()[name = string("op_7819_groups_0"), val = int32(1)]; tensor var_7819_strides_0 = const()[name = string("op_7819_strides_0"), val = tensor([1])]; tensor var_7819_pad_0 = const()[name = string("op_7819_pad_0"), val = tensor([0, 0])]; tensor var_7819_dilations_0 = const()[name = string("op_7819_dilations_0"), val = tensor([1])]; tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313629952))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315202880))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_7804_cast_fp16 = transpose(perm = var_7803, x = output_81_cast_fp16)[name = string("transpose_129")]; tensor var_7819_cast_fp16 = conv(dilations = var_7819_dilations_0, groups = var_7819_groups_0, pad = var_7819_pad_0, pad_type = var_7819_pad_type_0, strides = var_7819_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_7804_cast_fp16)[name = string("op_7819_cast_fp16")]; tensor var_7823 = const()[name = string("op_7823"), val = tensor([0, 2, 1])]; tensor attn_output_27_cast_fp16 = transpose(perm = var_7823, x = var_7819_cast_fp16)[name = string("transpose_128")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_131_cast_fp16, y = attn_output_27_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; int32 var_7838 = const()[name = string("op_7838"), val = int32(-1)]; fp16 const_207_promoted_to_fp16 = const()[name = string("const_207_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7840_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_207_promoted_to_fp16)[name = string("op_7840_cast_fp16")]; bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; tensor input_245_cast_fp16 = concat(axis = var_7838, interleave = input_245_interleave_0, values = (hidden_states_139_cast_fp16, var_7840_cast_fp16))[name = string("input_245_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_7835_to_fp16 = const()[name = string("op_7835_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_7835_to_fp16, x = input_245_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor normed_223_begin_0 = const()[name = string("normed_223_begin_0"), val = tensor([0, 0, 0])]; tensor normed_223_end_0 = const()[name = string("normed_223_end_0"), val = tensor([1, 64, 1024])]; tensor normed_223_end_mask_0 = const()[name = string("normed_223_end_mask_0"), val = tensor([true, true, false])]; tensor normed_223_cast_fp16 = slice_by_index(begin = normed_223_begin_0, end = normed_223_end_0, end_mask = normed_223_end_mask_0, x = normed_221_cast_fp16)[name = string("normed_223_cast_fp16")]; tensor const_209_promoted_to_fp16 = const()[name = string("const_209_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315219328)))]; tensor x_53_cast_fp16 = mul(x = normed_223_cast_fp16, y = const_209_promoted_to_fp16)[name = string("x_53_cast_fp16")]; tensor var_7860 = const()[name = string("op_7860"), val = tensor([0, 2, 1])]; tensor input_247_axes_0 = const()[name = string("input_247_axes_0"), val = tensor([2])]; tensor var_7861 = transpose(perm = var_7860, x = x_53_cast_fp16)[name = string("transpose_127")]; tensor input_247 = expand_dims(axes = input_247_axes_0, x = var_7861)[name = string("input_247")]; string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; tensor input_249 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_247)[name = string("input_249")]; string b_27_pad_type_0 = const()[name = string("b_27_pad_type_0"), val = string("valid")]; tensor b_27_strides_0 = const()[name = string("b_27_strides_0"), val = tensor([1, 1])]; tensor b_27_pad_0 = const()[name = string("b_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_27_dilations_0 = const()[name = string("b_27_dilations_0"), val = tensor([1, 1])]; int32 b_27_groups_0 = const()[name = string("b_27_groups_0"), val = int32(1)]; tensor b_27 = conv(dilations = b_27_dilations_0, groups = b_27_groups_0, pad = b_27_pad_0, pad_type = b_27_pad_type_0, strides = b_27_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_247)[name = string("b_27")]; tensor c_27 = silu(x = input_249)[name = string("c_27")]; tensor input_251 = mul(x = c_27, y = b_27)[name = string("input_251")]; string e_27_pad_type_0 = const()[name = string("e_27_pad_type_0"), val = string("valid")]; tensor e_27_strides_0 = const()[name = string("e_27_strides_0"), val = tensor([1, 1])]; tensor e_27_pad_0 = const()[name = string("e_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_27_dilations_0 = const()[name = string("e_27_dilations_0"), val = tensor([1, 1])]; int32 e_27_groups_0 = const()[name = string("e_27_groups_0"), val = int32(1)]; tensor e_27 = conv(dilations = e_27_dilations_0, groups = e_27_groups_0, pad = e_27_pad_0, pad_type = e_27_pad_type_0, strides = e_27_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input_251)[name = string("e_27")]; tensor var_7883_axes_0 = const()[name = string("op_7883_axes_0"), val = tensor([2])]; tensor var_7883 = squeeze(axes = var_7883_axes_0, x = e_27)[name = string("op_7883")]; tensor var_7884 = const()[name = string("op_7884"), val = tensor([0, 2, 1])]; tensor var_7885 = transpose(perm = var_7884, x = var_7883)[name = string("transpose_126")]; tensor hidden_states_141_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = var_7885)[name = string("hidden_states_141_cast_fp16")]; int32 var_7899 = const()[name = string("op_7899"), val = int32(-1)]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7901_cast_fp16 = mul(x = hidden_states_141_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7901_cast_fp16")]; bool input_253_interleave_0 = const()[name = string("input_253_interleave_0"), val = bool(false)]; tensor input_253_cast_fp16 = concat(axis = var_7899, interleave = input_253_interleave_0, values = (hidden_states_141_cast_fp16, var_7901_cast_fp16))[name = string("input_253_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_7896_to_fp16 = const()[name = string("op_7896_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_7896_to_fp16, x = input_253_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor normed_227_begin_0 = const()[name = string("normed_227_begin_0"), val = tensor([0, 0, 0])]; tensor normed_227_end_0 = const()[name = string("normed_227_end_0"), val = tensor([1, 64, 1024])]; tensor normed_227_end_mask_0 = const()[name = string("normed_227_end_mask_0"), val = tensor([true, true, false])]; tensor normed_227_cast_fp16 = slice_by_index(begin = normed_227_begin_0, end = normed_227_end_0, end_mask = normed_227_end_mask_0, x = normed_225_cast_fp16)[name = string("normed_227_cast_fp16")]; tensor const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315221440)))]; tensor hidden_states_143_cast_fp16 = mul(x = normed_227_cast_fp16, y = const_212_promoted_to_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor var_7913 = const()[name = string("op_7913"), val = tensor([0, 2, 1])]; tensor var_7916_axes_0 = const()[name = string("op_7916_axes_0"), val = tensor([2])]; tensor var_7914_cast_fp16 = transpose(perm = var_7913, x = hidden_states_143_cast_fp16)[name = string("transpose_125")]; tensor var_7916_cast_fp16 = expand_dims(axes = var_7916_axes_0, x = var_7914_cast_fp16)[name = string("op_7916_cast_fp16")]; string var_7932_pad_type_0 = const()[name = string("op_7932_pad_type_0"), val = string("valid")]; tensor var_7932_strides_0 = const()[name = string("op_7932_strides_0"), val = tensor([1, 1])]; tensor var_7932_pad_0 = const()[name = string("op_7932_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7932_dilations_0 = const()[name = string("op_7932_dilations_0"), val = tensor([1, 1])]; int32 var_7932_groups_0 = const()[name = string("op_7932_groups_0"), val = int32(1)]; tensor var_7932 = conv(dilations = var_7932_dilations_0, groups = var_7932_groups_0, pad = var_7932_pad_0, pad_type = var_7932_pad_type_0, strides = var_7932_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_7916_cast_fp16)[name = string("op_7932")]; tensor var_7937 = const()[name = string("op_7937"), val = tensor([1, 16, 128, 64])]; tensor var_7938 = reshape(shape = var_7937, x = var_7932)[name = string("op_7938")]; tensor var_7943 = const()[name = string("op_7943"), val = tensor([0, 1, 3, 2])]; string var_7955_pad_type_0 = const()[name = string("op_7955_pad_type_0"), val = string("valid")]; tensor var_7955_strides_0 = const()[name = string("op_7955_strides_0"), val = tensor([1, 1])]; tensor var_7955_pad_0 = const()[name = string("op_7955_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7955_dilations_0 = const()[name = string("op_7955_dilations_0"), val = tensor([1, 1])]; int32 var_7955_groups_0 = const()[name = string("op_7955_groups_0"), val = int32(1)]; tensor var_7955 = conv(dilations = var_7955_dilations_0, groups = var_7955_groups_0, pad = var_7955_pad_0, pad_type = var_7955_pad_type_0, strides = var_7955_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_7916_cast_fp16)[name = string("op_7955")]; tensor var_7960 = const()[name = string("op_7960"), val = tensor([1, 8, 128, 64])]; tensor var_7961 = reshape(shape = var_7960, x = var_7955)[name = string("op_7961")]; tensor var_7966 = const()[name = string("op_7966"), val = tensor([0, 1, 3, 2])]; string var_7978_pad_type_0 = const()[name = string("op_7978_pad_type_0"), val = string("valid")]; tensor var_7978_strides_0 = const()[name = string("op_7978_strides_0"), val = tensor([1, 1])]; tensor var_7978_pad_0 = const()[name = string("op_7978_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7978_dilations_0 = const()[name = string("op_7978_dilations_0"), val = tensor([1, 1])]; int32 var_7978_groups_0 = const()[name = string("op_7978_groups_0"), val = int32(1)]; tensor var_7978 = conv(dilations = var_7978_dilations_0, groups = var_7978_groups_0, pad = var_7978_pad_0, pad_type = var_7978_pad_type_0, strides = var_7978_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_7916_cast_fp16)[name = string("op_7978")]; tensor var_7983 = const()[name = string("op_7983"), val = tensor([1, 8, 128, 64])]; tensor var_7984 = reshape(shape = var_7983, x = var_7978)[name = string("op_7984")]; tensor var_7989 = const()[name = string("op_7989"), val = tensor([0, 1, 3, 2])]; int32 var_8002 = const()[name = string("op_8002"), val = int32(-1)]; fp16 const_213_promoted = const()[name = string("const_213_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_145 = transpose(perm = var_7943, x = var_7938)[name = string("transpose_124")]; tensor var_8004 = mul(x = hidden_states_145, y = const_213_promoted)[name = string("op_8004")]; bool input_257_interleave_0 = const()[name = string("input_257_interleave_0"), val = bool(false)]; tensor input_257 = concat(axis = var_8002, interleave = input_257_interleave_0, values = (hidden_states_145, var_8004))[name = string("input_257")]; tensor normed_229_axes_0 = const()[name = string("normed_229_axes_0"), val = tensor([-1])]; fp16 var_7999_to_fp16 = const()[name = string("op_7999_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_229_cast_fp16 = layer_norm(axes = normed_229_axes_0, epsilon = var_7999_to_fp16, x = input_257)[name = string("normed_229_cast_fp16")]; tensor normed_231_begin_0 = const()[name = string("normed_231_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_231_end_0 = const()[name = string("normed_231_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_231_end_mask_0 = const()[name = string("normed_231_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_231 = slice_by_index(begin = normed_231_begin_0, end = normed_231_end_0, end_mask = normed_231_end_mask_0, x = normed_229_cast_fp16)[name = string("normed_231")]; tensor const_215 = const()[name = string("const_215"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315223552)))]; tensor q_29 = mul(x = normed_231, y = const_215)[name = string("q_29")]; int32 var_8024 = const()[name = string("op_8024"), val = int32(-1)]; fp16 const_216_promoted = const()[name = string("const_216_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_147 = transpose(perm = var_7966, x = var_7961)[name = string("transpose_123")]; tensor var_8026 = mul(x = hidden_states_147, y = const_216_promoted)[name = string("op_8026")]; bool input_259_interleave_0 = const()[name = string("input_259_interleave_0"), val = bool(false)]; tensor input_259 = concat(axis = var_8024, interleave = input_259_interleave_0, values = (hidden_states_147, var_8026))[name = string("input_259")]; tensor normed_233_axes_0 = const()[name = string("normed_233_axes_0"), val = tensor([-1])]; fp16 var_8021_to_fp16 = const()[name = string("op_8021_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_233_cast_fp16 = layer_norm(axes = normed_233_axes_0, epsilon = var_8021_to_fp16, x = input_259)[name = string("normed_233_cast_fp16")]; tensor normed_235_begin_0 = const()[name = string("normed_235_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_235_end_0 = const()[name = string("normed_235_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_235_end_mask_0 = const()[name = string("normed_235_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_235 = slice_by_index(begin = normed_235_begin_0, end = normed_235_end_0, end_mask = normed_235_end_mask_0, x = normed_233_cast_fp16)[name = string("normed_235")]; tensor const_218 = const()[name = string("const_218"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315223872)))]; tensor k_29 = mul(x = normed_235, y = const_218)[name = string("k_29")]; tensor var_8047 = mul(x = q_29, y = cos_1)[name = string("op_8047")]; tensor var_8052_begin_0 = const()[name = string("op_8052_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8052_end_0 = const()[name = string("op_8052_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_8052_end_mask_0 = const()[name = string("op_8052_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8052 = slice_by_index(begin = var_8052_begin_0, end = var_8052_end_0, end_mask = var_8052_end_mask_0, x = q_29)[name = string("op_8052")]; fp16 const_219_promoted = const()[name = string("const_219_promoted"), val = fp16(-0x1p+0)]; tensor var_8053 = mul(x = var_8052, y = const_219_promoted)[name = string("op_8053")]; tensor var_8058_begin_0 = const()[name = string("op_8058_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8058_end_0 = const()[name = string("op_8058_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_8058_end_mask_0 = const()[name = string("op_8058_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8058 = slice_by_index(begin = var_8058_begin_0, end = var_8058_end_0, end_mask = var_8058_end_mask_0, x = q_29)[name = string("op_8058")]; int32 var_8060 = const()[name = string("op_8060"), val = int32(-1)]; bool var_8061_interleave_0 = const()[name = string("op_8061_interleave_0"), val = bool(false)]; tensor var_8061 = concat(axis = var_8060, interleave = var_8061_interleave_0, values = (var_8053, var_8058))[name = string("op_8061")]; tensor var_8062 = mul(x = var_8061, y = sin_1)[name = string("op_8062")]; tensor query_29 = add(x = var_8047, y = var_8062)[name = string("query_29")]; tensor var_8065 = mul(x = k_29, y = cos_1)[name = string("op_8065")]; tensor var_8070_begin_0 = const()[name = string("op_8070_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8070_end_0 = const()[name = string("op_8070_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_8070_end_mask_0 = const()[name = string("op_8070_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8070 = slice_by_index(begin = var_8070_begin_0, end = var_8070_end_0, end_mask = var_8070_end_mask_0, x = k_29)[name = string("op_8070")]; fp16 const_220_promoted = const()[name = string("const_220_promoted"), val = fp16(-0x1p+0)]; tensor var_8071 = mul(x = var_8070, y = const_220_promoted)[name = string("op_8071")]; tensor var_8076_begin_0 = const()[name = string("op_8076_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8076_end_0 = const()[name = string("op_8076_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_8076_end_mask_0 = const()[name = string("op_8076_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8076 = slice_by_index(begin = var_8076_begin_0, end = var_8076_end_0, end_mask = var_8076_end_mask_0, x = k_29)[name = string("op_8076")]; int32 var_8078 = const()[name = string("op_8078"), val = int32(-1)]; bool var_8079_interleave_0 = const()[name = string("op_8079_interleave_0"), val = bool(false)]; tensor var_8079 = concat(axis = var_8078, interleave = var_8079_interleave_0, values = (var_8071, var_8076))[name = string("op_8079")]; tensor var_8080 = mul(x = var_8079, y = sin_1)[name = string("op_8080")]; tensor key_29 = add(x = var_8065, y = var_8080)[name = string("key_29")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; int32 concat_254_axis_0 = const()[name = string("concat_254_axis_0"), val = int32(0)]; bool concat_254_interleave_0 = const()[name = string("concat_254_interleave_0"), val = bool(false)]; tensor concat_254 = concat(axis = concat_254_axis_0, interleave = concat_254_interleave_0, values = (expand_dims_168, expand_dims_169, current_pos, expand_dims_171))[name = string("concat_254")]; tensor concat_255_values1_0 = const()[name = string("concat_255_values1_0"), val = tensor([0])]; tensor concat_255_values3_0 = const()[name = string("concat_255_values3_0"), val = tensor([0])]; int32 concat_255_axis_0 = const()[name = string("concat_255_axis_0"), val = int32(0)]; bool concat_255_interleave_0 = const()[name = string("concat_255_interleave_0"), val = bool(false)]; tensor concat_255 = concat(axis = concat_255_axis_0, interleave = concat_255_interleave_0, values = (expand_dims_172, concat_255_values1_0, var_1746, concat_255_values3_0))[name = string("concat_255")]; tensor model_model_kv_cache_0_internal_tensor_assign_29_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16 = slice_update(begin = concat_254, begin_mask = model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0, end = concat_255, end_mask = model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_29_stride_0, update = key_29, x = coreml_update_state_83)[name = string("model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_196_write_state")]; tensor coreml_update_state_84 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_196")]; tensor expand_dims_174 = const()[name = string("expand_dims_174"), val = tensor([42])]; tensor expand_dims_175 = const()[name = string("expand_dims_175"), val = tensor([0])]; tensor expand_dims_177 = const()[name = string("expand_dims_177"), val = tensor([0])]; tensor expand_dims_178 = const()[name = string("expand_dims_178"), val = tensor([43])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_174, expand_dims_175, current_pos, expand_dims_177))[name = string("concat_258")]; tensor concat_259_values1_0 = const()[name = string("concat_259_values1_0"), val = tensor([0])]; tensor concat_259_values3_0 = const()[name = string("concat_259_values3_0"), val = tensor([0])]; int32 concat_259_axis_0 = const()[name = string("concat_259_axis_0"), val = int32(0)]; bool concat_259_interleave_0 = const()[name = string("concat_259_interleave_0"), val = bool(false)]; tensor concat_259 = concat(axis = concat_259_axis_0, interleave = concat_259_interleave_0, values = (expand_dims_178, concat_259_values1_0, var_1746, concat_259_values3_0))[name = string("concat_259")]; tensor model_model_kv_cache_0_internal_tensor_assign_30_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_141 = transpose(perm = var_7989, x = var_7984)[name = string("transpose_122")]; tensor model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16 = slice_update(begin = concat_258, begin_mask = model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0, end = concat_259, end_mask = model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_30_stride_0, update = value_141, x = coreml_update_state_84)[name = string("model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_197_write_state")]; tensor coreml_update_state_85 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_197")]; tensor var_8151_begin_0 = const()[name = string("op_8151_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_8151_end_0 = const()[name = string("op_8151_end_0"), val = tensor([15, 8, 1536, 128])]; tensor var_8151_end_mask_0 = const()[name = string("op_8151_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8151_cast_fp16 = slice_by_index(begin = var_8151_begin_0, end = var_8151_end_0, end_mask = var_8151_end_mask_0, x = coreml_update_state_85)[name = string("op_8151_cast_fp16")]; tensor key_cache_29_axes_0 = const()[name = string("key_cache_29_axes_0"), val = tensor([0])]; tensor key_cache_29_cast_fp16 = squeeze(axes = key_cache_29_axes_0, x = var_8151_cast_fp16)[name = string("key_cache_29_cast_fp16")]; tensor var_8158_begin_0 = const()[name = string("op_8158_begin_0"), val = tensor([42, 0, 0, 0])]; tensor var_8158_end_0 = const()[name = string("op_8158_end_0"), val = tensor([43, 8, 1536, 128])]; tensor var_8158_end_mask_0 = const()[name = string("op_8158_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8158_cast_fp16 = slice_by_index(begin = var_8158_begin_0, end = var_8158_end_0, end_mask = var_8158_end_mask_0, x = coreml_update_state_85)[name = string("op_8158_cast_fp16")]; tensor value_cache_29_axes_0 = const()[name = string("value_cache_29_axes_0"), val = tensor([0])]; tensor value_cache_29_cast_fp16 = squeeze(axes = value_cache_29_axes_0, x = var_8158_cast_fp16)[name = string("value_cache_29_cast_fp16")]; tensor var_8182_axes_0 = const()[name = string("op_8182_axes_0"), val = tensor([1])]; tensor var_8182_cast_fp16 = expand_dims(axes = var_8182_axes_0, x = key_cache_29_cast_fp16)[name = string("op_8182_cast_fp16")]; tensor var_8187 = const()[name = string("op_8187"), val = tensor([1, 2, 1, 1])]; tensor value_145_cast_fp16 = tile(reps = var_8187, x = var_8182_cast_fp16)[name = string("value_145_cast_fp16")]; tensor var_8193 = const()[name = string("op_8193"), val = tensor([1, 16, 1536, 128])]; tensor key_states_59_cast_fp16 = reshape(shape = var_8193, x = value_145_cast_fp16)[name = string("key_states_59_cast_fp16")]; tensor var_8196_axes_0 = const()[name = string("op_8196_axes_0"), val = tensor([1])]; tensor var_8196_cast_fp16 = expand_dims(axes = var_8196_axes_0, x = value_cache_29_cast_fp16)[name = string("op_8196_cast_fp16")]; tensor var_8201 = const()[name = string("op_8201"), val = tensor([1, 2, 1, 1])]; tensor value_149_cast_fp16 = tile(reps = var_8201, x = var_8196_cast_fp16)[name = string("value_149_cast_fp16")]; bool var_8222_transpose_x_0 = const()[name = string("op_8222_transpose_x_0"), val = bool(false)]; bool var_8222_transpose_y_0 = const()[name = string("op_8222_transpose_y_0"), val = bool(true)]; tensor var_8222 = matmul(transpose_x = var_8222_transpose_x_0, transpose_y = var_8222_transpose_y_0, x = query_29, y = key_states_59_cast_fp16)[name = string("op_8222")]; fp16 var_8223_to_fp16 = const()[name = string("op_8223_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_57_cast_fp16 = mul(x = var_8222, y = var_8223_to_fp16)[name = string("attention_57_cast_fp16")]; tensor attention_59_cast_fp16 = add(x = attention_57_cast_fp16, y = causal_mask)[name = string("attention_59_cast_fp16")]; int32 var_8232 = const()[name = string("op_8232"), val = int32(-1)]; tensor var_8234_cast_fp16 = softmax(axis = var_8232, x = attention_59_cast_fp16)[name = string("op_8234_cast_fp16")]; tensor concat_264 = const()[name = string("concat_264"), val = tensor([16, 64, 1536])]; tensor reshape_42_cast_fp16 = reshape(shape = concat_264, x = var_8234_cast_fp16)[name = string("reshape_42_cast_fp16")]; tensor concat_265 = const()[name = string("concat_265"), val = tensor([16, 1536, 128])]; tensor reshape_43_cast_fp16 = reshape(shape = concat_265, x = value_149_cast_fp16)[name = string("reshape_43_cast_fp16")]; bool matmul_14_transpose_x_0 = const()[name = string("matmul_14_transpose_x_0"), val = bool(false)]; bool matmul_14_transpose_y_0 = const()[name = string("matmul_14_transpose_y_0"), val = bool(false)]; tensor matmul_14_cast_fp16 = matmul(transpose_x = matmul_14_transpose_x_0, transpose_y = matmul_14_transpose_y_0, x = reshape_42_cast_fp16, y = reshape_43_cast_fp16)[name = string("matmul_14_cast_fp16")]; tensor concat_269 = const()[name = string("concat_269"), val = tensor([1, 16, 64, 128])]; tensor reshape_44_cast_fp16 = reshape(shape = concat_269, x = matmul_14_cast_fp16)[name = string("reshape_44_cast_fp16")]; tensor var_8246_perm_0 = const()[name = string("op_8246_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_8252 = const()[name = string("op_8252"), val = tensor([1, 64, 2048])]; tensor var_8246_cast_fp16 = transpose(perm = var_8246_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_121")]; tensor output_87_cast_fp16 = reshape(shape = var_8252, x = var_8246_cast_fp16)[name = string("output_87_cast_fp16")]; tensor var_8257 = const()[name = string("op_8257"), val = tensor([0, 2, 1])]; string var_8273_pad_type_0 = const()[name = string("op_8273_pad_type_0"), val = string("valid")]; int32 var_8273_groups_0 = const()[name = string("op_8273_groups_0"), val = int32(1)]; tensor var_8273_strides_0 = const()[name = string("op_8273_strides_0"), val = tensor([1])]; tensor var_8273_pad_0 = const()[name = string("op_8273_pad_0"), val = tensor([0, 0])]; tensor var_8273_dilations_0 = const()[name = string("op_8273_dilations_0"), val = tensor([1])]; tensor squeeze_14_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315224192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316797120))))[name = string("squeeze_14_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_8258_cast_fp16 = transpose(perm = var_8257, x = output_87_cast_fp16)[name = string("transpose_120")]; tensor var_8273_cast_fp16 = conv(dilations = var_8273_dilations_0, groups = var_8273_groups_0, pad = var_8273_pad_0, pad_type = var_8273_pad_type_0, strides = var_8273_strides_0, weight = squeeze_14_cast_fp16_to_fp32_to_fp16_palettized, x = var_8258_cast_fp16)[name = string("op_8273_cast_fp16")]; tensor var_8277 = const()[name = string("op_8277"), val = tensor([0, 2, 1])]; tensor attn_output_29_cast_fp16 = transpose(perm = var_8277, x = var_8273_cast_fp16)[name = string("transpose_119")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_141_cast_fp16, y = attn_output_29_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; int32 var_8292 = const()[name = string("op_8292"), val = int32(-1)]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8294_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_222_promoted_to_fp16)[name = string("op_8294_cast_fp16")]; bool input_263_interleave_0 = const()[name = string("input_263_interleave_0"), val = bool(false)]; tensor input_263_cast_fp16 = concat(axis = var_8292, interleave = input_263_interleave_0, values = (hidden_states_149_cast_fp16, var_8294_cast_fp16))[name = string("input_263_cast_fp16")]; tensor normed_237_axes_0 = const()[name = string("normed_237_axes_0"), val = tensor([-1])]; fp16 var_8289_to_fp16 = const()[name = string("op_8289_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_237_cast_fp16 = layer_norm(axes = normed_237_axes_0, epsilon = var_8289_to_fp16, x = input_263_cast_fp16)[name = string("normed_237_cast_fp16")]; tensor normed_239_begin_0 = const()[name = string("normed_239_begin_0"), val = tensor([0, 0, 0])]; tensor normed_239_end_0 = const()[name = string("normed_239_end_0"), val = tensor([1, 64, 1024])]; tensor normed_239_end_mask_0 = const()[name = string("normed_239_end_mask_0"), val = tensor([true, true, false])]; tensor normed_239_cast_fp16 = slice_by_index(begin = normed_239_begin_0, end = normed_239_end_0, end_mask = normed_239_end_mask_0, x = normed_237_cast_fp16)[name = string("normed_239_cast_fp16")]; tensor const_224_promoted_to_fp16 = const()[name = string("const_224_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316813568)))]; tensor x_57_cast_fp16 = mul(x = normed_239_cast_fp16, y = const_224_promoted_to_fp16)[name = string("x_57_cast_fp16")]; tensor var_8314 = const()[name = string("op_8314"), val = tensor([0, 2, 1])]; tensor input_265_axes_0 = const()[name = string("input_265_axes_0"), val = tensor([2])]; tensor var_8315 = transpose(perm = var_8314, x = x_57_cast_fp16)[name = string("transpose_118")]; tensor input_265 = expand_dims(axes = input_265_axes_0, x = var_8315)[name = string("input_265")]; string input_267_pad_type_0 = const()[name = string("input_267_pad_type_0"), val = string("valid")]; tensor input_267_strides_0 = const()[name = string("input_267_strides_0"), val = tensor([1, 1])]; tensor input_267_pad_0 = const()[name = string("input_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_267_dilations_0 = const()[name = string("input_267_dilations_0"), val = tensor([1, 1])]; int32 input_267_groups_0 = const()[name = string("input_267_groups_0"), val = int32(1)]; tensor input_267 = conv(dilations = input_267_dilations_0, groups = input_267_groups_0, pad = input_267_pad_0, pad_type = input_267_pad_type_0, strides = input_267_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_265)[name = string("input_267")]; string b_29_pad_type_0 = const()[name = string("b_29_pad_type_0"), val = string("valid")]; tensor b_29_strides_0 = const()[name = string("b_29_strides_0"), val = tensor([1, 1])]; tensor b_29_pad_0 = const()[name = string("b_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_29_dilations_0 = const()[name = string("b_29_dilations_0"), val = tensor([1, 1])]; int32 b_29_groups_0 = const()[name = string("b_29_groups_0"), val = int32(1)]; tensor b_29 = conv(dilations = b_29_dilations_0, groups = b_29_groups_0, pad = b_29_pad_0, pad_type = b_29_pad_type_0, strides = b_29_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_265)[name = string("b_29")]; tensor c_29 = silu(x = input_267)[name = string("c_29")]; tensor input_269 = mul(x = c_29, y = b_29)[name = string("input_269")]; string e_29_pad_type_0 = const()[name = string("e_29_pad_type_0"), val = string("valid")]; tensor e_29_strides_0 = const()[name = string("e_29_strides_0"), val = tensor([1, 1])]; tensor e_29_pad_0 = const()[name = string("e_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_29_dilations_0 = const()[name = string("e_29_dilations_0"), val = tensor([1, 1])]; int32 e_29_groups_0 = const()[name = string("e_29_groups_0"), val = int32(1)]; tensor e_29 = conv(dilations = e_29_dilations_0, groups = e_29_groups_0, pad = e_29_pad_0, pad_type = e_29_pad_type_0, strides = e_29_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_269)[name = string("e_29")]; tensor var_8337_axes_0 = const()[name = string("op_8337_axes_0"), val = tensor([2])]; tensor var_8337 = squeeze(axes = var_8337_axes_0, x = e_29)[name = string("op_8337")]; tensor var_8338 = const()[name = string("op_8338"), val = tensor([0, 2, 1])]; tensor var_8339 = transpose(perm = var_8338, x = var_8337)[name = string("transpose_117")]; tensor hidden_states_151_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = var_8339)[name = string("hidden_states_151_cast_fp16")]; int32 var_8353 = const()[name = string("op_8353"), val = int32(-1)]; fp16 const_225_promoted_to_fp16 = const()[name = string("const_225_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8355_cast_fp16 = mul(x = hidden_states_151_cast_fp16, y = const_225_promoted_to_fp16)[name = string("op_8355_cast_fp16")]; bool input_271_interleave_0 = const()[name = string("input_271_interleave_0"), val = bool(false)]; tensor input_271_cast_fp16 = concat(axis = var_8353, interleave = input_271_interleave_0, values = (hidden_states_151_cast_fp16, var_8355_cast_fp16))[name = string("input_271_cast_fp16")]; tensor normed_241_axes_0 = const()[name = string("normed_241_axes_0"), val = tensor([-1])]; fp16 var_8350_to_fp16 = const()[name = string("op_8350_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_241_cast_fp16 = layer_norm(axes = normed_241_axes_0, epsilon = var_8350_to_fp16, x = input_271_cast_fp16)[name = string("normed_241_cast_fp16")]; tensor normed_243_begin_0 = const()[name = string("normed_243_begin_0"), val = tensor([0, 0, 0])]; tensor normed_243_end_0 = const()[name = string("normed_243_end_0"), val = tensor([1, 64, 1024])]; tensor normed_243_end_mask_0 = const()[name = string("normed_243_end_mask_0"), val = tensor([true, true, false])]; tensor normed_243_cast_fp16 = slice_by_index(begin = normed_243_begin_0, end = normed_243_end_0, end_mask = normed_243_end_mask_0, x = normed_241_cast_fp16)[name = string("normed_243_cast_fp16")]; tensor const_227_promoted_to_fp16 = const()[name = string("const_227_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316815680)))]; tensor hidden_states_153_cast_fp16 = mul(x = normed_243_cast_fp16, y = const_227_promoted_to_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor var_8367 = const()[name = string("op_8367"), val = tensor([0, 2, 1])]; tensor var_8370_axes_0 = const()[name = string("op_8370_axes_0"), val = tensor([2])]; tensor var_8368_cast_fp16 = transpose(perm = var_8367, x = hidden_states_153_cast_fp16)[name = string("transpose_116")]; tensor var_8370_cast_fp16 = expand_dims(axes = var_8370_axes_0, x = var_8368_cast_fp16)[name = string("op_8370_cast_fp16")]; string var_8386_pad_type_0 = const()[name = string("op_8386_pad_type_0"), val = string("valid")]; tensor var_8386_strides_0 = const()[name = string("op_8386_strides_0"), val = tensor([1, 1])]; tensor var_8386_pad_0 = const()[name = string("op_8386_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8386_dilations_0 = const()[name = string("op_8386_dilations_0"), val = tensor([1, 1])]; int32 var_8386_groups_0 = const()[name = string("op_8386_groups_0"), val = int32(1)]; tensor var_8386 = conv(dilations = var_8386_dilations_0, groups = var_8386_groups_0, pad = var_8386_pad_0, pad_type = var_8386_pad_type_0, strides = var_8386_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_8370_cast_fp16)[name = string("op_8386")]; tensor var_8391 = const()[name = string("op_8391"), val = tensor([1, 16, 128, 64])]; tensor var_8392 = reshape(shape = var_8391, x = var_8386)[name = string("op_8392")]; tensor var_8397 = const()[name = string("op_8397"), val = tensor([0, 1, 3, 2])]; string var_8409_pad_type_0 = const()[name = string("op_8409_pad_type_0"), val = string("valid")]; tensor var_8409_strides_0 = const()[name = string("op_8409_strides_0"), val = tensor([1, 1])]; tensor var_8409_pad_0 = const()[name = string("op_8409_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8409_dilations_0 = const()[name = string("op_8409_dilations_0"), val = tensor([1, 1])]; int32 var_8409_groups_0 = const()[name = string("op_8409_groups_0"), val = int32(1)]; tensor var_8409 = conv(dilations = var_8409_dilations_0, groups = var_8409_groups_0, pad = var_8409_pad_0, pad_type = var_8409_pad_type_0, strides = var_8409_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_8370_cast_fp16)[name = string("op_8409")]; tensor var_8414 = const()[name = string("op_8414"), val = tensor([1, 8, 128, 64])]; tensor var_8415 = reshape(shape = var_8414, x = var_8409)[name = string("op_8415")]; tensor var_8420 = const()[name = string("op_8420"), val = tensor([0, 1, 3, 2])]; string var_8432_pad_type_0 = const()[name = string("op_8432_pad_type_0"), val = string("valid")]; tensor var_8432_strides_0 = const()[name = string("op_8432_strides_0"), val = tensor([1, 1])]; tensor var_8432_pad_0 = const()[name = string("op_8432_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8432_dilations_0 = const()[name = string("op_8432_dilations_0"), val = tensor([1, 1])]; int32 var_8432_groups_0 = const()[name = string("op_8432_groups_0"), val = int32(1)]; tensor var_8432 = conv(dilations = var_8432_dilations_0, groups = var_8432_groups_0, pad = var_8432_pad_0, pad_type = var_8432_pad_type_0, strides = var_8432_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_8370_cast_fp16)[name = string("op_8432")]; tensor var_8437 = const()[name = string("op_8437"), val = tensor([1, 8, 128, 64])]; tensor var_8438 = reshape(shape = var_8437, x = var_8432)[name = string("op_8438")]; tensor var_8443 = const()[name = string("op_8443"), val = tensor([0, 1, 3, 2])]; int32 var_8456 = const()[name = string("op_8456"), val = int32(-1)]; fp16 const_228_promoted = const()[name = string("const_228_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_155 = transpose(perm = var_8397, x = var_8392)[name = string("transpose_115")]; tensor var_8458 = mul(x = hidden_states_155, y = const_228_promoted)[name = string("op_8458")]; bool input_275_interleave_0 = const()[name = string("input_275_interleave_0"), val = bool(false)]; tensor input_275 = concat(axis = var_8456, interleave = input_275_interleave_0, values = (hidden_states_155, var_8458))[name = string("input_275")]; tensor normed_245_axes_0 = const()[name = string("normed_245_axes_0"), val = tensor([-1])]; fp16 var_8453_to_fp16 = const()[name = string("op_8453_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_245_cast_fp16 = layer_norm(axes = normed_245_axes_0, epsilon = var_8453_to_fp16, x = input_275)[name = string("normed_245_cast_fp16")]; tensor normed_247_begin_0 = const()[name = string("normed_247_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_247_end_0 = const()[name = string("normed_247_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_247_end_mask_0 = const()[name = string("normed_247_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_247 = slice_by_index(begin = normed_247_begin_0, end = normed_247_end_0, end_mask = normed_247_end_mask_0, x = normed_245_cast_fp16)[name = string("normed_247")]; tensor const_230 = const()[name = string("const_230"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316817792)))]; tensor q_31 = mul(x = normed_247, y = const_230)[name = string("q_31")]; int32 var_8478 = const()[name = string("op_8478"), val = int32(-1)]; fp16 const_231_promoted = const()[name = string("const_231_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_157 = transpose(perm = var_8420, x = var_8415)[name = string("transpose_114")]; tensor var_8480 = mul(x = hidden_states_157, y = const_231_promoted)[name = string("op_8480")]; bool input_277_interleave_0 = const()[name = string("input_277_interleave_0"), val = bool(false)]; tensor input_277 = concat(axis = var_8478, interleave = input_277_interleave_0, values = (hidden_states_157, var_8480))[name = string("input_277")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; fp16 var_8475_to_fp16 = const()[name = string("op_8475_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_8475_to_fp16, x = input_277)[name = string("normed_249_cast_fp16")]; tensor normed_251_begin_0 = const()[name = string("normed_251_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_251_end_0 = const()[name = string("normed_251_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_251_end_mask_0 = const()[name = string("normed_251_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_251 = slice_by_index(begin = normed_251_begin_0, end = normed_251_end_0, end_mask = normed_251_end_mask_0, x = normed_249_cast_fp16)[name = string("normed_251")]; tensor const_233 = const()[name = string("const_233"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316818112)))]; tensor k_31 = mul(x = normed_251, y = const_233)[name = string("k_31")]; tensor var_8501 = mul(x = q_31, y = cos_1)[name = string("op_8501")]; tensor var_8506_begin_0 = const()[name = string("op_8506_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8506_end_0 = const()[name = string("op_8506_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_8506_end_mask_0 = const()[name = string("op_8506_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8506 = slice_by_index(begin = var_8506_begin_0, end = var_8506_end_0, end_mask = var_8506_end_mask_0, x = q_31)[name = string("op_8506")]; fp16 const_234_promoted = const()[name = string("const_234_promoted"), val = fp16(-0x1p+0)]; tensor var_8507 = mul(x = var_8506, y = const_234_promoted)[name = string("op_8507")]; tensor var_8512_begin_0 = const()[name = string("op_8512_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8512_end_0 = const()[name = string("op_8512_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_8512_end_mask_0 = const()[name = string("op_8512_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8512 = slice_by_index(begin = var_8512_begin_0, end = var_8512_end_0, end_mask = var_8512_end_mask_0, x = q_31)[name = string("op_8512")]; int32 var_8514 = const()[name = string("op_8514"), val = int32(-1)]; bool var_8515_interleave_0 = const()[name = string("op_8515_interleave_0"), val = bool(false)]; tensor var_8515 = concat(axis = var_8514, interleave = var_8515_interleave_0, values = (var_8507, var_8512))[name = string("op_8515")]; tensor var_8516 = mul(x = var_8515, y = sin_1)[name = string("op_8516")]; tensor query_31 = add(x = var_8501, y = var_8516)[name = string("query_31")]; tensor var_8519 = mul(x = k_31, y = cos_1)[name = string("op_8519")]; tensor var_8524_begin_0 = const()[name = string("op_8524_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8524_end_0 = const()[name = string("op_8524_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_8524_end_mask_0 = const()[name = string("op_8524_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8524 = slice_by_index(begin = var_8524_begin_0, end = var_8524_end_0, end_mask = var_8524_end_mask_0, x = k_31)[name = string("op_8524")]; fp16 const_235_promoted = const()[name = string("const_235_promoted"), val = fp16(-0x1p+0)]; tensor var_8525 = mul(x = var_8524, y = const_235_promoted)[name = string("op_8525")]; tensor var_8530_begin_0 = const()[name = string("op_8530_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8530_end_0 = const()[name = string("op_8530_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_8530_end_mask_0 = const()[name = string("op_8530_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8530 = slice_by_index(begin = var_8530_begin_0, end = var_8530_end_0, end_mask = var_8530_end_mask_0, x = k_31)[name = string("op_8530")]; int32 var_8532 = const()[name = string("op_8532"), val = int32(-1)]; bool var_8533_interleave_0 = const()[name = string("op_8533_interleave_0"), val = bool(false)]; tensor var_8533 = concat(axis = var_8532, interleave = var_8533_interleave_0, values = (var_8525, var_8530))[name = string("op_8533")]; tensor var_8534 = mul(x = var_8533, y = sin_1)[name = string("op_8534")]; tensor key_31 = add(x = var_8519, y = var_8534)[name = string("key_31")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; int32 concat_272_axis_0 = const()[name = string("concat_272_axis_0"), val = int32(0)]; bool concat_272_interleave_0 = const()[name = string("concat_272_interleave_0"), val = bool(false)]; tensor concat_272 = concat(axis = concat_272_axis_0, interleave = concat_272_interleave_0, values = (expand_dims_180, expand_dims_181, current_pos, expand_dims_183))[name = string("concat_272")]; tensor concat_273_values1_0 = const()[name = string("concat_273_values1_0"), val = tensor([0])]; tensor concat_273_values3_0 = const()[name = string("concat_273_values3_0"), val = tensor([0])]; int32 concat_273_axis_0 = const()[name = string("concat_273_axis_0"), val = int32(0)]; bool concat_273_interleave_0 = const()[name = string("concat_273_interleave_0"), val = bool(false)]; tensor concat_273 = concat(axis = concat_273_axis_0, interleave = concat_273_interleave_0, values = (expand_dims_184, concat_273_values1_0, var_1746, concat_273_values3_0))[name = string("concat_273")]; tensor model_model_kv_cache_0_internal_tensor_assign_31_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16 = slice_update(begin = concat_272, begin_mask = model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0, end = concat_273, end_mask = model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_31_stride_0, update = key_31, x = coreml_update_state_85)[name = string("model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_198_write_state")]; tensor coreml_update_state_86 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_198")]; tensor expand_dims_186 = const()[name = string("expand_dims_186"), val = tensor([43])]; tensor expand_dims_187 = const()[name = string("expand_dims_187"), val = tensor([0])]; tensor expand_dims_189 = const()[name = string("expand_dims_189"), val = tensor([0])]; tensor expand_dims_190 = const()[name = string("expand_dims_190"), val = tensor([44])]; int32 concat_276_axis_0 = const()[name = string("concat_276_axis_0"), val = int32(0)]; bool concat_276_interleave_0 = const()[name = string("concat_276_interleave_0"), val = bool(false)]; tensor concat_276 = concat(axis = concat_276_axis_0, interleave = concat_276_interleave_0, values = (expand_dims_186, expand_dims_187, current_pos, expand_dims_189))[name = string("concat_276")]; tensor concat_277_values1_0 = const()[name = string("concat_277_values1_0"), val = tensor([0])]; tensor concat_277_values3_0 = const()[name = string("concat_277_values3_0"), val = tensor([0])]; int32 concat_277_axis_0 = const()[name = string("concat_277_axis_0"), val = int32(0)]; bool concat_277_interleave_0 = const()[name = string("concat_277_interleave_0"), val = bool(false)]; tensor concat_277 = concat(axis = concat_277_axis_0, interleave = concat_277_interleave_0, values = (expand_dims_190, concat_277_values1_0, var_1746, concat_277_values3_0))[name = string("concat_277")]; tensor model_model_kv_cache_0_internal_tensor_assign_32_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_151 = transpose(perm = var_8443, x = var_8438)[name = string("transpose_113")]; tensor model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16 = slice_update(begin = concat_276, begin_mask = model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0, end = concat_277, end_mask = model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_32_stride_0, update = value_151, x = coreml_update_state_86)[name = string("model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_199_write_state")]; tensor coreml_update_state_87 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_199")]; tensor var_8605_begin_0 = const()[name = string("op_8605_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_8605_end_0 = const()[name = string("op_8605_end_0"), val = tensor([16, 8, 1536, 128])]; tensor var_8605_end_mask_0 = const()[name = string("op_8605_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8605_cast_fp16 = slice_by_index(begin = var_8605_begin_0, end = var_8605_end_0, end_mask = var_8605_end_mask_0, x = coreml_update_state_87)[name = string("op_8605_cast_fp16")]; tensor key_cache_31_axes_0 = const()[name = string("key_cache_31_axes_0"), val = tensor([0])]; tensor key_cache_31_cast_fp16 = squeeze(axes = key_cache_31_axes_0, x = var_8605_cast_fp16)[name = string("key_cache_31_cast_fp16")]; tensor var_8612_begin_0 = const()[name = string("op_8612_begin_0"), val = tensor([43, 0, 0, 0])]; tensor var_8612_end_0 = const()[name = string("op_8612_end_0"), val = tensor([44, 8, 1536, 128])]; tensor var_8612_end_mask_0 = const()[name = string("op_8612_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8612_cast_fp16 = slice_by_index(begin = var_8612_begin_0, end = var_8612_end_0, end_mask = var_8612_end_mask_0, x = coreml_update_state_87)[name = string("op_8612_cast_fp16")]; tensor value_cache_31_axes_0 = const()[name = string("value_cache_31_axes_0"), val = tensor([0])]; tensor value_cache_31_cast_fp16 = squeeze(axes = value_cache_31_axes_0, x = var_8612_cast_fp16)[name = string("value_cache_31_cast_fp16")]; tensor var_8636_axes_0 = const()[name = string("op_8636_axes_0"), val = tensor([1])]; tensor var_8636_cast_fp16 = expand_dims(axes = var_8636_axes_0, x = key_cache_31_cast_fp16)[name = string("op_8636_cast_fp16")]; tensor var_8641 = const()[name = string("op_8641"), val = tensor([1, 2, 1, 1])]; tensor value_155_cast_fp16 = tile(reps = var_8641, x = var_8636_cast_fp16)[name = string("value_155_cast_fp16")]; tensor var_8647 = const()[name = string("op_8647"), val = tensor([1, 16, 1536, 128])]; tensor key_states_63_cast_fp16 = reshape(shape = var_8647, x = value_155_cast_fp16)[name = string("key_states_63_cast_fp16")]; tensor var_8650_axes_0 = const()[name = string("op_8650_axes_0"), val = tensor([1])]; tensor var_8650_cast_fp16 = expand_dims(axes = var_8650_axes_0, x = value_cache_31_cast_fp16)[name = string("op_8650_cast_fp16")]; tensor var_8655 = const()[name = string("op_8655"), val = tensor([1, 2, 1, 1])]; tensor value_159_cast_fp16 = tile(reps = var_8655, x = var_8650_cast_fp16)[name = string("value_159_cast_fp16")]; bool var_8676_transpose_x_0 = const()[name = string("op_8676_transpose_x_0"), val = bool(false)]; bool var_8676_transpose_y_0 = const()[name = string("op_8676_transpose_y_0"), val = bool(true)]; tensor var_8676 = matmul(transpose_x = var_8676_transpose_x_0, transpose_y = var_8676_transpose_y_0, x = query_31, y = key_states_63_cast_fp16)[name = string("op_8676")]; fp16 var_8677_to_fp16 = const()[name = string("op_8677_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_61_cast_fp16 = mul(x = var_8676, y = var_8677_to_fp16)[name = string("attention_61_cast_fp16")]; tensor attention_63_cast_fp16 = add(x = attention_61_cast_fp16, y = causal_mask)[name = string("attention_63_cast_fp16")]; int32 var_8686 = const()[name = string("op_8686"), val = int32(-1)]; tensor var_8688_cast_fp16 = softmax(axis = var_8686, x = attention_63_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_282 = const()[name = string("concat_282"), val = tensor([16, 64, 1536])]; tensor reshape_45_cast_fp16 = reshape(shape = concat_282, x = var_8688_cast_fp16)[name = string("reshape_45_cast_fp16")]; tensor concat_283 = const()[name = string("concat_283"), val = tensor([16, 1536, 128])]; tensor reshape_46_cast_fp16 = reshape(shape = concat_283, x = value_159_cast_fp16)[name = string("reshape_46_cast_fp16")]; bool matmul_15_transpose_x_0 = const()[name = string("matmul_15_transpose_x_0"), val = bool(false)]; bool matmul_15_transpose_y_0 = const()[name = string("matmul_15_transpose_y_0"), val = bool(false)]; tensor matmul_15_cast_fp16 = matmul(transpose_x = matmul_15_transpose_x_0, transpose_y = matmul_15_transpose_y_0, x = reshape_45_cast_fp16, y = reshape_46_cast_fp16)[name = string("matmul_15_cast_fp16")]; tensor concat_287 = const()[name = string("concat_287"), val = tensor([1, 16, 64, 128])]; tensor reshape_47_cast_fp16 = reshape(shape = concat_287, x = matmul_15_cast_fp16)[name = string("reshape_47_cast_fp16")]; tensor var_8700_perm_0 = const()[name = string("op_8700_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_8706 = const()[name = string("op_8706"), val = tensor([1, 64, 2048])]; tensor var_8700_cast_fp16 = transpose(perm = var_8700_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_112")]; tensor output_93_cast_fp16 = reshape(shape = var_8706, x = var_8700_cast_fp16)[name = string("output_93_cast_fp16")]; tensor var_8711 = const()[name = string("op_8711"), val = tensor([0, 2, 1])]; string var_8727_pad_type_0 = const()[name = string("op_8727_pad_type_0"), val = string("valid")]; int32 var_8727_groups_0 = const()[name = string("op_8727_groups_0"), val = int32(1)]; tensor var_8727_strides_0 = const()[name = string("op_8727_strides_0"), val = tensor([1])]; tensor var_8727_pad_0 = const()[name = string("op_8727_pad_0"), val = tensor([0, 0])]; tensor var_8727_dilations_0 = const()[name = string("op_8727_dilations_0"), val = tensor([1])]; tensor squeeze_15_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316818432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318391360))))[name = string("squeeze_15_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_8712_cast_fp16 = transpose(perm = var_8711, x = output_93_cast_fp16)[name = string("transpose_111")]; tensor var_8727_cast_fp16 = conv(dilations = var_8727_dilations_0, groups = var_8727_groups_0, pad = var_8727_pad_0, pad_type = var_8727_pad_type_0, strides = var_8727_strides_0, weight = squeeze_15_cast_fp16_to_fp32_to_fp16_palettized, x = var_8712_cast_fp16)[name = string("op_8727_cast_fp16")]; tensor var_8731 = const()[name = string("op_8731"), val = tensor([0, 2, 1])]; tensor attn_output_31_cast_fp16 = transpose(perm = var_8731, x = var_8727_cast_fp16)[name = string("transpose_110")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_151_cast_fp16, y = attn_output_31_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; int32 var_8746 = const()[name = string("op_8746"), val = int32(-1)]; fp16 const_237_promoted_to_fp16 = const()[name = string("const_237_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8748_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_237_promoted_to_fp16)[name = string("op_8748_cast_fp16")]; bool input_281_interleave_0 = const()[name = string("input_281_interleave_0"), val = bool(false)]; tensor input_281_cast_fp16 = concat(axis = var_8746, interleave = input_281_interleave_0, values = (hidden_states_159_cast_fp16, var_8748_cast_fp16))[name = string("input_281_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; fp16 var_8743_to_fp16 = const()[name = string("op_8743_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_8743_to_fp16, x = input_281_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor normed_255_begin_0 = const()[name = string("normed_255_begin_0"), val = tensor([0, 0, 0])]; tensor normed_255_end_0 = const()[name = string("normed_255_end_0"), val = tensor([1, 64, 1024])]; tensor normed_255_end_mask_0 = const()[name = string("normed_255_end_mask_0"), val = tensor([true, true, false])]; tensor normed_255_cast_fp16 = slice_by_index(begin = normed_255_begin_0, end = normed_255_end_0, end_mask = normed_255_end_mask_0, x = normed_253_cast_fp16)[name = string("normed_255_cast_fp16")]; tensor const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318407808)))]; tensor x_61_cast_fp16 = mul(x = normed_255_cast_fp16, y = const_239_promoted_to_fp16)[name = string("x_61_cast_fp16")]; tensor var_8768 = const()[name = string("op_8768"), val = tensor([0, 2, 1])]; tensor input_283_axes_0 = const()[name = string("input_283_axes_0"), val = tensor([2])]; tensor var_8769 = transpose(perm = var_8768, x = x_61_cast_fp16)[name = string("transpose_109")]; tensor input_283 = expand_dims(axes = input_283_axes_0, x = var_8769)[name = string("input_283")]; string input_285_pad_type_0 = const()[name = string("input_285_pad_type_0"), val = string("valid")]; tensor input_285_strides_0 = const()[name = string("input_285_strides_0"), val = tensor([1, 1])]; tensor input_285_pad_0 = const()[name = string("input_285_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_285_dilations_0 = const()[name = string("input_285_dilations_0"), val = tensor([1, 1])]; int32 input_285_groups_0 = const()[name = string("input_285_groups_0"), val = int32(1)]; tensor input_285 = conv(dilations = input_285_dilations_0, groups = input_285_groups_0, pad = input_285_pad_0, pad_type = input_285_pad_type_0, strides = input_285_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_283)[name = string("input_285")]; string b_31_pad_type_0 = const()[name = string("b_31_pad_type_0"), val = string("valid")]; tensor b_31_strides_0 = const()[name = string("b_31_strides_0"), val = tensor([1, 1])]; tensor b_31_pad_0 = const()[name = string("b_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_31_dilations_0 = const()[name = string("b_31_dilations_0"), val = tensor([1, 1])]; int32 b_31_groups_0 = const()[name = string("b_31_groups_0"), val = int32(1)]; tensor b_31 = conv(dilations = b_31_dilations_0, groups = b_31_groups_0, pad = b_31_pad_0, pad_type = b_31_pad_type_0, strides = b_31_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_283)[name = string("b_31")]; tensor c_31 = silu(x = input_285)[name = string("c_31")]; tensor input_287 = mul(x = c_31, y = b_31)[name = string("input_287")]; string e_31_pad_type_0 = const()[name = string("e_31_pad_type_0"), val = string("valid")]; tensor e_31_strides_0 = const()[name = string("e_31_strides_0"), val = tensor([1, 1])]; tensor e_31_pad_0 = const()[name = string("e_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_31_dilations_0 = const()[name = string("e_31_dilations_0"), val = tensor([1, 1])]; int32 e_31_groups_0 = const()[name = string("e_31_groups_0"), val = int32(1)]; tensor e_31 = conv(dilations = e_31_dilations_0, groups = e_31_groups_0, pad = e_31_pad_0, pad_type = e_31_pad_type_0, strides = e_31_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_287)[name = string("e_31")]; tensor var_8791_axes_0 = const()[name = string("op_8791_axes_0"), val = tensor([2])]; tensor var_8791 = squeeze(axes = var_8791_axes_0, x = e_31)[name = string("op_8791")]; tensor var_8792 = const()[name = string("op_8792"), val = tensor([0, 2, 1])]; tensor var_8793 = transpose(perm = var_8792, x = var_8791)[name = string("transpose_108")]; tensor hidden_states_161_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = var_8793)[name = string("hidden_states_161_cast_fp16")]; int32 var_8807 = const()[name = string("op_8807"), val = int32(-1)]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8809_cast_fp16 = mul(x = hidden_states_161_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8809_cast_fp16")]; bool input_289_interleave_0 = const()[name = string("input_289_interleave_0"), val = bool(false)]; tensor input_289_cast_fp16 = concat(axis = var_8807, interleave = input_289_interleave_0, values = (hidden_states_161_cast_fp16, var_8809_cast_fp16))[name = string("input_289_cast_fp16")]; tensor normed_257_axes_0 = const()[name = string("normed_257_axes_0"), val = tensor([-1])]; fp16 var_8804_to_fp16 = const()[name = string("op_8804_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_257_cast_fp16 = layer_norm(axes = normed_257_axes_0, epsilon = var_8804_to_fp16, x = input_289_cast_fp16)[name = string("normed_257_cast_fp16")]; tensor normed_259_begin_0 = const()[name = string("normed_259_begin_0"), val = tensor([0, 0, 0])]; tensor normed_259_end_0 = const()[name = string("normed_259_end_0"), val = tensor([1, 64, 1024])]; tensor normed_259_end_mask_0 = const()[name = string("normed_259_end_mask_0"), val = tensor([true, true, false])]; tensor normed_259_cast_fp16 = slice_by_index(begin = normed_259_begin_0, end = normed_259_end_0, end_mask = normed_259_end_mask_0, x = normed_257_cast_fp16)[name = string("normed_259_cast_fp16")]; tensor const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318409920)))]; tensor hidden_states_163_cast_fp16 = mul(x = normed_259_cast_fp16, y = const_242_promoted_to_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor var_8821 = const()[name = string("op_8821"), val = tensor([0, 2, 1])]; tensor var_8824_axes_0 = const()[name = string("op_8824_axes_0"), val = tensor([2])]; tensor var_8822_cast_fp16 = transpose(perm = var_8821, x = hidden_states_163_cast_fp16)[name = string("transpose_107")]; tensor var_8824_cast_fp16 = expand_dims(axes = var_8824_axes_0, x = var_8822_cast_fp16)[name = string("op_8824_cast_fp16")]; string var_8840_pad_type_0 = const()[name = string("op_8840_pad_type_0"), val = string("valid")]; tensor var_8840_strides_0 = const()[name = string("op_8840_strides_0"), val = tensor([1, 1])]; tensor var_8840_pad_0 = const()[name = string("op_8840_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8840_dilations_0 = const()[name = string("op_8840_dilations_0"), val = tensor([1, 1])]; int32 var_8840_groups_0 = const()[name = string("op_8840_groups_0"), val = int32(1)]; tensor var_8840 = conv(dilations = var_8840_dilations_0, groups = var_8840_groups_0, pad = var_8840_pad_0, pad_type = var_8840_pad_type_0, strides = var_8840_strides_0, weight = model_model_layers_16_self_attn_q_proj_weight_palettized, x = var_8824_cast_fp16)[name = string("op_8840")]; tensor var_8845 = const()[name = string("op_8845"), val = tensor([1, 16, 128, 64])]; tensor var_8846 = reshape(shape = var_8845, x = var_8840)[name = string("op_8846")]; tensor var_8851 = const()[name = string("op_8851"), val = tensor([0, 1, 3, 2])]; string var_8863_pad_type_0 = const()[name = string("op_8863_pad_type_0"), val = string("valid")]; tensor var_8863_strides_0 = const()[name = string("op_8863_strides_0"), val = tensor([1, 1])]; tensor var_8863_pad_0 = const()[name = string("op_8863_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8863_dilations_0 = const()[name = string("op_8863_dilations_0"), val = tensor([1, 1])]; int32 var_8863_groups_0 = const()[name = string("op_8863_groups_0"), val = int32(1)]; tensor var_8863 = conv(dilations = var_8863_dilations_0, groups = var_8863_groups_0, pad = var_8863_pad_0, pad_type = var_8863_pad_type_0, strides = var_8863_strides_0, weight = model_model_layers_16_self_attn_k_proj_weight_palettized, x = var_8824_cast_fp16)[name = string("op_8863")]; tensor var_8868 = const()[name = string("op_8868"), val = tensor([1, 8, 128, 64])]; tensor var_8869 = reshape(shape = var_8868, x = var_8863)[name = string("op_8869")]; tensor var_8874 = const()[name = string("op_8874"), val = tensor([0, 1, 3, 2])]; string var_8886_pad_type_0 = const()[name = string("op_8886_pad_type_0"), val = string("valid")]; tensor var_8886_strides_0 = const()[name = string("op_8886_strides_0"), val = tensor([1, 1])]; tensor var_8886_pad_0 = const()[name = string("op_8886_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8886_dilations_0 = const()[name = string("op_8886_dilations_0"), val = tensor([1, 1])]; int32 var_8886_groups_0 = const()[name = string("op_8886_groups_0"), val = int32(1)]; tensor var_8886 = conv(dilations = var_8886_dilations_0, groups = var_8886_groups_0, pad = var_8886_pad_0, pad_type = var_8886_pad_type_0, strides = var_8886_strides_0, weight = model_model_layers_16_self_attn_v_proj_weight_palettized, x = var_8824_cast_fp16)[name = string("op_8886")]; tensor var_8891 = const()[name = string("op_8891"), val = tensor([1, 8, 128, 64])]; tensor var_8892 = reshape(shape = var_8891, x = var_8886)[name = string("op_8892")]; tensor var_8897 = const()[name = string("op_8897"), val = tensor([0, 1, 3, 2])]; int32 var_8910 = const()[name = string("op_8910"), val = int32(-1)]; fp16 const_243_promoted = const()[name = string("const_243_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_165 = transpose(perm = var_8851, x = var_8846)[name = string("transpose_106")]; tensor var_8912 = mul(x = hidden_states_165, y = const_243_promoted)[name = string("op_8912")]; bool input_293_interleave_0 = const()[name = string("input_293_interleave_0"), val = bool(false)]; tensor input_293 = concat(axis = var_8910, interleave = input_293_interleave_0, values = (hidden_states_165, var_8912))[name = string("input_293")]; tensor normed_261_axes_0 = const()[name = string("normed_261_axes_0"), val = tensor([-1])]; fp16 var_8907_to_fp16 = const()[name = string("op_8907_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_261_cast_fp16 = layer_norm(axes = normed_261_axes_0, epsilon = var_8907_to_fp16, x = input_293)[name = string("normed_261_cast_fp16")]; tensor normed_263_begin_0 = const()[name = string("normed_263_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_263_end_0 = const()[name = string("normed_263_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_263_end_mask_0 = const()[name = string("normed_263_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_263 = slice_by_index(begin = normed_263_begin_0, end = normed_263_end_0, end_mask = normed_263_end_mask_0, x = normed_261_cast_fp16)[name = string("normed_263")]; tensor const_245 = const()[name = string("const_245"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318412032)))]; tensor q_33 = mul(x = normed_263, y = const_245)[name = string("q_33")]; int32 var_8932 = const()[name = string("op_8932"), val = int32(-1)]; fp16 const_246_promoted = const()[name = string("const_246_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_167 = transpose(perm = var_8874, x = var_8869)[name = string("transpose_105")]; tensor var_8934 = mul(x = hidden_states_167, y = const_246_promoted)[name = string("op_8934")]; bool input_295_interleave_0 = const()[name = string("input_295_interleave_0"), val = bool(false)]; tensor input_295 = concat(axis = var_8932, interleave = input_295_interleave_0, values = (hidden_states_167, var_8934))[name = string("input_295")]; tensor normed_265_axes_0 = const()[name = string("normed_265_axes_0"), val = tensor([-1])]; fp16 var_8929_to_fp16 = const()[name = string("op_8929_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_265_cast_fp16 = layer_norm(axes = normed_265_axes_0, epsilon = var_8929_to_fp16, x = input_295)[name = string("normed_265_cast_fp16")]; tensor normed_267_begin_0 = const()[name = string("normed_267_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_267_end_0 = const()[name = string("normed_267_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_267_end_mask_0 = const()[name = string("normed_267_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_267 = slice_by_index(begin = normed_267_begin_0, end = normed_267_end_0, end_mask = normed_267_end_mask_0, x = normed_265_cast_fp16)[name = string("normed_267")]; tensor const_248 = const()[name = string("const_248"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318412352)))]; tensor k_33 = mul(x = normed_267, y = const_248)[name = string("k_33")]; tensor var_8955 = mul(x = q_33, y = cos_1)[name = string("op_8955")]; tensor var_8960_begin_0 = const()[name = string("op_8960_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8960_end_0 = const()[name = string("op_8960_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_8960_end_mask_0 = const()[name = string("op_8960_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8960 = slice_by_index(begin = var_8960_begin_0, end = var_8960_end_0, end_mask = var_8960_end_mask_0, x = q_33)[name = string("op_8960")]; fp16 const_249_promoted = const()[name = string("const_249_promoted"), val = fp16(-0x1p+0)]; tensor var_8961 = mul(x = var_8960, y = const_249_promoted)[name = string("op_8961")]; tensor var_8966_begin_0 = const()[name = string("op_8966_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8966_end_0 = const()[name = string("op_8966_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_8966_end_mask_0 = const()[name = string("op_8966_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8966 = slice_by_index(begin = var_8966_begin_0, end = var_8966_end_0, end_mask = var_8966_end_mask_0, x = q_33)[name = string("op_8966")]; int32 var_8968 = const()[name = string("op_8968"), val = int32(-1)]; bool var_8969_interleave_0 = const()[name = string("op_8969_interleave_0"), val = bool(false)]; tensor var_8969 = concat(axis = var_8968, interleave = var_8969_interleave_0, values = (var_8961, var_8966))[name = string("op_8969")]; tensor var_8970 = mul(x = var_8969, y = sin_1)[name = string("op_8970")]; tensor query_33 = add(x = var_8955, y = var_8970)[name = string("query_33")]; tensor var_8973 = mul(x = k_33, y = cos_1)[name = string("op_8973")]; tensor var_8978_begin_0 = const()[name = string("op_8978_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_8978_end_0 = const()[name = string("op_8978_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_8978_end_mask_0 = const()[name = string("op_8978_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_8978 = slice_by_index(begin = var_8978_begin_0, end = var_8978_end_0, end_mask = var_8978_end_mask_0, x = k_33)[name = string("op_8978")]; fp16 const_250_promoted = const()[name = string("const_250_promoted"), val = fp16(-0x1p+0)]; tensor var_8979 = mul(x = var_8978, y = const_250_promoted)[name = string("op_8979")]; tensor var_8984_begin_0 = const()[name = string("op_8984_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_8984_end_0 = const()[name = string("op_8984_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_8984_end_mask_0 = const()[name = string("op_8984_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_8984 = slice_by_index(begin = var_8984_begin_0, end = var_8984_end_0, end_mask = var_8984_end_mask_0, x = k_33)[name = string("op_8984")]; int32 var_8986 = const()[name = string("op_8986"), val = int32(-1)]; bool var_8987_interleave_0 = const()[name = string("op_8987_interleave_0"), val = bool(false)]; tensor var_8987 = concat(axis = var_8986, interleave = var_8987_interleave_0, values = (var_8979, var_8984))[name = string("op_8987")]; tensor var_8988 = mul(x = var_8987, y = sin_1)[name = string("op_8988")]; tensor key_33 = add(x = var_8973, y = var_8988)[name = string("key_33")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; int32 concat_290_axis_0 = const()[name = string("concat_290_axis_0"), val = int32(0)]; bool concat_290_interleave_0 = const()[name = string("concat_290_interleave_0"), val = bool(false)]; tensor concat_290 = concat(axis = concat_290_axis_0, interleave = concat_290_interleave_0, values = (expand_dims_192, expand_dims_193, current_pos, expand_dims_195))[name = string("concat_290")]; tensor concat_291_values1_0 = const()[name = string("concat_291_values1_0"), val = tensor([0])]; tensor concat_291_values3_0 = const()[name = string("concat_291_values3_0"), val = tensor([0])]; int32 concat_291_axis_0 = const()[name = string("concat_291_axis_0"), val = int32(0)]; bool concat_291_interleave_0 = const()[name = string("concat_291_interleave_0"), val = bool(false)]; tensor concat_291 = concat(axis = concat_291_axis_0, interleave = concat_291_interleave_0, values = (expand_dims_196, concat_291_values1_0, var_1746, concat_291_values3_0))[name = string("concat_291")]; tensor model_model_kv_cache_0_internal_tensor_assign_33_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_33_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_33_cast_fp16 = slice_update(begin = concat_290, begin_mask = model_model_kv_cache_0_internal_tensor_assign_33_begin_mask_0, end = concat_291, end_mask = model_model_kv_cache_0_internal_tensor_assign_33_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_33_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_33_stride_0, update = key_33, x = coreml_update_state_87)[name = string("model_model_kv_cache_0_internal_tensor_assign_33_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_33_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_200_write_state")]; tensor coreml_update_state_88 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_200")]; tensor expand_dims_198 = const()[name = string("expand_dims_198"), val = tensor([44])]; tensor expand_dims_199 = const()[name = string("expand_dims_199"), val = tensor([0])]; tensor expand_dims_201 = const()[name = string("expand_dims_201"), val = tensor([0])]; tensor expand_dims_202 = const()[name = string("expand_dims_202"), val = tensor([45])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_198, expand_dims_199, current_pos, expand_dims_201))[name = string("concat_294")]; tensor concat_295_values1_0 = const()[name = string("concat_295_values1_0"), val = tensor([0])]; tensor concat_295_values3_0 = const()[name = string("concat_295_values3_0"), val = tensor([0])]; int32 concat_295_axis_0 = const()[name = string("concat_295_axis_0"), val = int32(0)]; bool concat_295_interleave_0 = const()[name = string("concat_295_interleave_0"), val = bool(false)]; tensor concat_295 = concat(axis = concat_295_axis_0, interleave = concat_295_interleave_0, values = (expand_dims_202, concat_295_values1_0, var_1746, concat_295_values3_0))[name = string("concat_295")]; tensor model_model_kv_cache_0_internal_tensor_assign_34_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_34_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_34_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_161 = transpose(perm = var_8897, x = var_8892)[name = string("transpose_104")]; tensor model_model_kv_cache_0_internal_tensor_assign_34_cast_fp16 = slice_update(begin = concat_294, begin_mask = model_model_kv_cache_0_internal_tensor_assign_34_begin_mask_0, end = concat_295, end_mask = model_model_kv_cache_0_internal_tensor_assign_34_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_34_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_34_stride_0, update = value_161, x = coreml_update_state_88)[name = string("model_model_kv_cache_0_internal_tensor_assign_34_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_34_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_201_write_state")]; tensor coreml_update_state_89 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_201")]; tensor var_9059_begin_0 = const()[name = string("op_9059_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_9059_end_0 = const()[name = string("op_9059_end_0"), val = tensor([17, 8, 1536, 128])]; tensor var_9059_end_mask_0 = const()[name = string("op_9059_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9059_cast_fp16 = slice_by_index(begin = var_9059_begin_0, end = var_9059_end_0, end_mask = var_9059_end_mask_0, x = coreml_update_state_89)[name = string("op_9059_cast_fp16")]; tensor key_cache_33_axes_0 = const()[name = string("key_cache_33_axes_0"), val = tensor([0])]; tensor key_cache_33_cast_fp16 = squeeze(axes = key_cache_33_axes_0, x = var_9059_cast_fp16)[name = string("key_cache_33_cast_fp16")]; tensor var_9066_begin_0 = const()[name = string("op_9066_begin_0"), val = tensor([44, 0, 0, 0])]; tensor var_9066_end_0 = const()[name = string("op_9066_end_0"), val = tensor([45, 8, 1536, 128])]; tensor var_9066_end_mask_0 = const()[name = string("op_9066_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9066_cast_fp16 = slice_by_index(begin = var_9066_begin_0, end = var_9066_end_0, end_mask = var_9066_end_mask_0, x = coreml_update_state_89)[name = string("op_9066_cast_fp16")]; tensor value_cache_33_axes_0 = const()[name = string("value_cache_33_axes_0"), val = tensor([0])]; tensor value_cache_33_cast_fp16 = squeeze(axes = value_cache_33_axes_0, x = var_9066_cast_fp16)[name = string("value_cache_33_cast_fp16")]; tensor var_9090_axes_0 = const()[name = string("op_9090_axes_0"), val = tensor([1])]; tensor var_9090_cast_fp16 = expand_dims(axes = var_9090_axes_0, x = key_cache_33_cast_fp16)[name = string("op_9090_cast_fp16")]; tensor var_9095 = const()[name = string("op_9095"), val = tensor([1, 2, 1, 1])]; tensor value_165_cast_fp16 = tile(reps = var_9095, x = var_9090_cast_fp16)[name = string("value_165_cast_fp16")]; tensor var_9101 = const()[name = string("op_9101"), val = tensor([1, 16, 1536, 128])]; tensor key_states_67_cast_fp16 = reshape(shape = var_9101, x = value_165_cast_fp16)[name = string("key_states_67_cast_fp16")]; tensor var_9104_axes_0 = const()[name = string("op_9104_axes_0"), val = tensor([1])]; tensor var_9104_cast_fp16 = expand_dims(axes = var_9104_axes_0, x = value_cache_33_cast_fp16)[name = string("op_9104_cast_fp16")]; tensor var_9109 = const()[name = string("op_9109"), val = tensor([1, 2, 1, 1])]; tensor value_169_cast_fp16 = tile(reps = var_9109, x = var_9104_cast_fp16)[name = string("value_169_cast_fp16")]; bool var_9130_transpose_x_0 = const()[name = string("op_9130_transpose_x_0"), val = bool(false)]; bool var_9130_transpose_y_0 = const()[name = string("op_9130_transpose_y_0"), val = bool(true)]; tensor var_9130 = matmul(transpose_x = var_9130_transpose_x_0, transpose_y = var_9130_transpose_y_0, x = query_33, y = key_states_67_cast_fp16)[name = string("op_9130")]; fp16 var_9131_to_fp16 = const()[name = string("op_9131_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_65_cast_fp16 = mul(x = var_9130, y = var_9131_to_fp16)[name = string("attention_65_cast_fp16")]; tensor attention_67_cast_fp16 = add(x = attention_65_cast_fp16, y = causal_mask)[name = string("attention_67_cast_fp16")]; int32 var_9140 = const()[name = string("op_9140"), val = int32(-1)]; tensor var_9142_cast_fp16 = softmax(axis = var_9140, x = attention_67_cast_fp16)[name = string("op_9142_cast_fp16")]; tensor concat_300 = const()[name = string("concat_300"), val = tensor([16, 64, 1536])]; tensor reshape_48_cast_fp16 = reshape(shape = concat_300, x = var_9142_cast_fp16)[name = string("reshape_48_cast_fp16")]; tensor concat_301 = const()[name = string("concat_301"), val = tensor([16, 1536, 128])]; tensor reshape_49_cast_fp16 = reshape(shape = concat_301, x = value_169_cast_fp16)[name = string("reshape_49_cast_fp16")]; bool matmul_16_transpose_x_0 = const()[name = string("matmul_16_transpose_x_0"), val = bool(false)]; bool matmul_16_transpose_y_0 = const()[name = string("matmul_16_transpose_y_0"), val = bool(false)]; tensor matmul_16_cast_fp16 = matmul(transpose_x = matmul_16_transpose_x_0, transpose_y = matmul_16_transpose_y_0, x = reshape_48_cast_fp16, y = reshape_49_cast_fp16)[name = string("matmul_16_cast_fp16")]; tensor concat_305 = const()[name = string("concat_305"), val = tensor([1, 16, 64, 128])]; tensor reshape_50_cast_fp16 = reshape(shape = concat_305, x = matmul_16_cast_fp16)[name = string("reshape_50_cast_fp16")]; tensor var_9154_perm_0 = const()[name = string("op_9154_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_9160 = const()[name = string("op_9160"), val = tensor([1, 64, 2048])]; tensor var_9154_cast_fp16 = transpose(perm = var_9154_perm_0, x = reshape_50_cast_fp16)[name = string("transpose_103")]; tensor output_99_cast_fp16 = reshape(shape = var_9160, x = var_9154_cast_fp16)[name = string("output_99_cast_fp16")]; tensor var_9165 = const()[name = string("op_9165"), val = tensor([0, 2, 1])]; string var_9181_pad_type_0 = const()[name = string("op_9181_pad_type_0"), val = string("valid")]; int32 var_9181_groups_0 = const()[name = string("op_9181_groups_0"), val = int32(1)]; tensor var_9181_strides_0 = const()[name = string("op_9181_strides_0"), val = tensor([1])]; tensor var_9181_pad_0 = const()[name = string("op_9181_pad_0"), val = tensor([0, 0])]; tensor var_9181_dilations_0 = const()[name = string("op_9181_dilations_0"), val = tensor([1])]; tensor squeeze_16_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318412672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319985600))))[name = string("squeeze_16_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9166_cast_fp16 = transpose(perm = var_9165, x = output_99_cast_fp16)[name = string("transpose_102")]; tensor var_9181_cast_fp16 = conv(dilations = var_9181_dilations_0, groups = var_9181_groups_0, pad = var_9181_pad_0, pad_type = var_9181_pad_type_0, strides = var_9181_strides_0, weight = squeeze_16_cast_fp16_to_fp32_to_fp16_palettized, x = var_9166_cast_fp16)[name = string("op_9181_cast_fp16")]; tensor var_9185 = const()[name = string("op_9185"), val = tensor([0, 2, 1])]; tensor attn_output_33_cast_fp16 = transpose(perm = var_9185, x = var_9181_cast_fp16)[name = string("transpose_101")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_161_cast_fp16, y = attn_output_33_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; int32 var_9200 = const()[name = string("op_9200"), val = int32(-1)]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9202_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_252_promoted_to_fp16)[name = string("op_9202_cast_fp16")]; bool input_299_interleave_0 = const()[name = string("input_299_interleave_0"), val = bool(false)]; tensor input_299_cast_fp16 = concat(axis = var_9200, interleave = input_299_interleave_0, values = (hidden_states_169_cast_fp16, var_9202_cast_fp16))[name = string("input_299_cast_fp16")]; tensor normed_269_axes_0 = const()[name = string("normed_269_axes_0"), val = tensor([-1])]; fp16 var_9197_to_fp16 = const()[name = string("op_9197_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_269_cast_fp16 = layer_norm(axes = normed_269_axes_0, epsilon = var_9197_to_fp16, x = input_299_cast_fp16)[name = string("normed_269_cast_fp16")]; tensor normed_271_begin_0 = const()[name = string("normed_271_begin_0"), val = tensor([0, 0, 0])]; tensor normed_271_end_0 = const()[name = string("normed_271_end_0"), val = tensor([1, 64, 1024])]; tensor normed_271_end_mask_0 = const()[name = string("normed_271_end_mask_0"), val = tensor([true, true, false])]; tensor normed_271_cast_fp16 = slice_by_index(begin = normed_271_begin_0, end = normed_271_end_0, end_mask = normed_271_end_mask_0, x = normed_269_cast_fp16)[name = string("normed_271_cast_fp16")]; tensor const_254_promoted_to_fp16 = const()[name = string("const_254_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320002048)))]; tensor x_65_cast_fp16 = mul(x = normed_271_cast_fp16, y = const_254_promoted_to_fp16)[name = string("x_65_cast_fp16")]; tensor var_9222 = const()[name = string("op_9222"), val = tensor([0, 2, 1])]; tensor input_301_axes_0 = const()[name = string("input_301_axes_0"), val = tensor([2])]; tensor var_9223 = transpose(perm = var_9222, x = x_65_cast_fp16)[name = string("transpose_100")]; tensor input_301 = expand_dims(axes = input_301_axes_0, x = var_9223)[name = string("input_301")]; string input_303_pad_type_0 = const()[name = string("input_303_pad_type_0"), val = string("valid")]; tensor input_303_strides_0 = const()[name = string("input_303_strides_0"), val = tensor([1, 1])]; tensor input_303_pad_0 = const()[name = string("input_303_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_303_dilations_0 = const()[name = string("input_303_dilations_0"), val = tensor([1, 1])]; int32 input_303_groups_0 = const()[name = string("input_303_groups_0"), val = int32(1)]; tensor input_303 = conv(dilations = input_303_dilations_0, groups = input_303_groups_0, pad = input_303_pad_0, pad_type = input_303_pad_type_0, strides = input_303_strides_0, weight = model_model_layers_16_mlp_gate_proj_weight_palettized, x = input_301)[name = string("input_303")]; string b_33_pad_type_0 = const()[name = string("b_33_pad_type_0"), val = string("valid")]; tensor b_33_strides_0 = const()[name = string("b_33_strides_0"), val = tensor([1, 1])]; tensor b_33_pad_0 = const()[name = string("b_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_33_dilations_0 = const()[name = string("b_33_dilations_0"), val = tensor([1, 1])]; int32 b_33_groups_0 = const()[name = string("b_33_groups_0"), val = int32(1)]; tensor b_33 = conv(dilations = b_33_dilations_0, groups = b_33_groups_0, pad = b_33_pad_0, pad_type = b_33_pad_type_0, strides = b_33_strides_0, weight = model_model_layers_16_mlp_up_proj_weight_palettized, x = input_301)[name = string("b_33")]; tensor c_33 = silu(x = input_303)[name = string("c_33")]; tensor input_305 = mul(x = c_33, y = b_33)[name = string("input_305")]; string e_33_pad_type_0 = const()[name = string("e_33_pad_type_0"), val = string("valid")]; tensor e_33_strides_0 = const()[name = string("e_33_strides_0"), val = tensor([1, 1])]; tensor e_33_pad_0 = const()[name = string("e_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_33_dilations_0 = const()[name = string("e_33_dilations_0"), val = tensor([1, 1])]; int32 e_33_groups_0 = const()[name = string("e_33_groups_0"), val = int32(1)]; tensor e_33 = conv(dilations = e_33_dilations_0, groups = e_33_groups_0, pad = e_33_pad_0, pad_type = e_33_pad_type_0, strides = e_33_strides_0, weight = model_model_layers_16_mlp_down_proj_weight_palettized, x = input_305)[name = string("e_33")]; tensor var_9245_axes_0 = const()[name = string("op_9245_axes_0"), val = tensor([2])]; tensor var_9245 = squeeze(axes = var_9245_axes_0, x = e_33)[name = string("op_9245")]; tensor var_9246 = const()[name = string("op_9246"), val = tensor([0, 2, 1])]; tensor var_9247 = transpose(perm = var_9246, x = var_9245)[name = string("transpose_99")]; tensor hidden_states_171_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = var_9247)[name = string("hidden_states_171_cast_fp16")]; int32 var_9261 = const()[name = string("op_9261"), val = int32(-1)]; fp16 const_255_promoted_to_fp16 = const()[name = string("const_255_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9263_cast_fp16 = mul(x = hidden_states_171_cast_fp16, y = const_255_promoted_to_fp16)[name = string("op_9263_cast_fp16")]; bool input_307_interleave_0 = const()[name = string("input_307_interleave_0"), val = bool(false)]; tensor input_307_cast_fp16 = concat(axis = var_9261, interleave = input_307_interleave_0, values = (hidden_states_171_cast_fp16, var_9263_cast_fp16))[name = string("input_307_cast_fp16")]; tensor normed_273_axes_0 = const()[name = string("normed_273_axes_0"), val = tensor([-1])]; fp16 var_9258_to_fp16 = const()[name = string("op_9258_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_273_cast_fp16 = layer_norm(axes = normed_273_axes_0, epsilon = var_9258_to_fp16, x = input_307_cast_fp16)[name = string("normed_273_cast_fp16")]; tensor normed_275_begin_0 = const()[name = string("normed_275_begin_0"), val = tensor([0, 0, 0])]; tensor normed_275_end_0 = const()[name = string("normed_275_end_0"), val = tensor([1, 64, 1024])]; tensor normed_275_end_mask_0 = const()[name = string("normed_275_end_mask_0"), val = tensor([true, true, false])]; tensor normed_275_cast_fp16 = slice_by_index(begin = normed_275_begin_0, end = normed_275_end_0, end_mask = normed_275_end_mask_0, x = normed_273_cast_fp16)[name = string("normed_275_cast_fp16")]; tensor const_257_promoted_to_fp16 = const()[name = string("const_257_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320004160)))]; tensor hidden_states_173_cast_fp16 = mul(x = normed_275_cast_fp16, y = const_257_promoted_to_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor var_9275 = const()[name = string("op_9275"), val = tensor([0, 2, 1])]; tensor var_9278_axes_0 = const()[name = string("op_9278_axes_0"), val = tensor([2])]; tensor var_9276_cast_fp16 = transpose(perm = var_9275, x = hidden_states_173_cast_fp16)[name = string("transpose_98")]; tensor var_9278_cast_fp16 = expand_dims(axes = var_9278_axes_0, x = var_9276_cast_fp16)[name = string("op_9278_cast_fp16")]; string var_9294_pad_type_0 = const()[name = string("op_9294_pad_type_0"), val = string("valid")]; tensor var_9294_strides_0 = const()[name = string("op_9294_strides_0"), val = tensor([1, 1])]; tensor var_9294_pad_0 = const()[name = string("op_9294_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9294_dilations_0 = const()[name = string("op_9294_dilations_0"), val = tensor([1, 1])]; int32 var_9294_groups_0 = const()[name = string("op_9294_groups_0"), val = int32(1)]; tensor var_9294 = conv(dilations = var_9294_dilations_0, groups = var_9294_groups_0, pad = var_9294_pad_0, pad_type = var_9294_pad_type_0, strides = var_9294_strides_0, weight = model_model_layers_17_self_attn_q_proj_weight_palettized, x = var_9278_cast_fp16)[name = string("op_9294")]; tensor var_9299 = const()[name = string("op_9299"), val = tensor([1, 16, 128, 64])]; tensor var_9300 = reshape(shape = var_9299, x = var_9294)[name = string("op_9300")]; tensor var_9305 = const()[name = string("op_9305"), val = tensor([0, 1, 3, 2])]; string var_9317_pad_type_0 = const()[name = string("op_9317_pad_type_0"), val = string("valid")]; tensor var_9317_strides_0 = const()[name = string("op_9317_strides_0"), val = tensor([1, 1])]; tensor var_9317_pad_0 = const()[name = string("op_9317_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9317_dilations_0 = const()[name = string("op_9317_dilations_0"), val = tensor([1, 1])]; int32 var_9317_groups_0 = const()[name = string("op_9317_groups_0"), val = int32(1)]; tensor var_9317 = conv(dilations = var_9317_dilations_0, groups = var_9317_groups_0, pad = var_9317_pad_0, pad_type = var_9317_pad_type_0, strides = var_9317_strides_0, weight = model_model_layers_17_self_attn_k_proj_weight_palettized, x = var_9278_cast_fp16)[name = string("op_9317")]; tensor var_9322 = const()[name = string("op_9322"), val = tensor([1, 8, 128, 64])]; tensor var_9323 = reshape(shape = var_9322, x = var_9317)[name = string("op_9323")]; tensor var_9328 = const()[name = string("op_9328"), val = tensor([0, 1, 3, 2])]; string var_9340_pad_type_0 = const()[name = string("op_9340_pad_type_0"), val = string("valid")]; tensor var_9340_strides_0 = const()[name = string("op_9340_strides_0"), val = tensor([1, 1])]; tensor var_9340_pad_0 = const()[name = string("op_9340_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9340_dilations_0 = const()[name = string("op_9340_dilations_0"), val = tensor([1, 1])]; int32 var_9340_groups_0 = const()[name = string("op_9340_groups_0"), val = int32(1)]; tensor var_9340 = conv(dilations = var_9340_dilations_0, groups = var_9340_groups_0, pad = var_9340_pad_0, pad_type = var_9340_pad_type_0, strides = var_9340_strides_0, weight = model_model_layers_17_self_attn_v_proj_weight_palettized, x = var_9278_cast_fp16)[name = string("op_9340")]; tensor var_9345 = const()[name = string("op_9345"), val = tensor([1, 8, 128, 64])]; tensor var_9346 = reshape(shape = var_9345, x = var_9340)[name = string("op_9346")]; tensor var_9351 = const()[name = string("op_9351"), val = tensor([0, 1, 3, 2])]; int32 var_9364 = const()[name = string("op_9364"), val = int32(-1)]; fp16 const_258_promoted = const()[name = string("const_258_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_175 = transpose(perm = var_9305, x = var_9300)[name = string("transpose_97")]; tensor var_9366 = mul(x = hidden_states_175, y = const_258_promoted)[name = string("op_9366")]; bool input_311_interleave_0 = const()[name = string("input_311_interleave_0"), val = bool(false)]; tensor input_311 = concat(axis = var_9364, interleave = input_311_interleave_0, values = (hidden_states_175, var_9366))[name = string("input_311")]; tensor normed_277_axes_0 = const()[name = string("normed_277_axes_0"), val = tensor([-1])]; fp16 var_9361_to_fp16 = const()[name = string("op_9361_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_277_cast_fp16 = layer_norm(axes = normed_277_axes_0, epsilon = var_9361_to_fp16, x = input_311)[name = string("normed_277_cast_fp16")]; tensor normed_279_begin_0 = const()[name = string("normed_279_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_279_end_0 = const()[name = string("normed_279_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_279_end_mask_0 = const()[name = string("normed_279_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_279 = slice_by_index(begin = normed_279_begin_0, end = normed_279_end_0, end_mask = normed_279_end_mask_0, x = normed_277_cast_fp16)[name = string("normed_279")]; tensor const_260 = const()[name = string("const_260"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320006272)))]; tensor q_35 = mul(x = normed_279, y = const_260)[name = string("q_35")]; int32 var_9386 = const()[name = string("op_9386"), val = int32(-1)]; fp16 const_261_promoted = const()[name = string("const_261_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_177 = transpose(perm = var_9328, x = var_9323)[name = string("transpose_96")]; tensor var_9388 = mul(x = hidden_states_177, y = const_261_promoted)[name = string("op_9388")]; bool input_313_interleave_0 = const()[name = string("input_313_interleave_0"), val = bool(false)]; tensor input_313 = concat(axis = var_9386, interleave = input_313_interleave_0, values = (hidden_states_177, var_9388))[name = string("input_313")]; tensor normed_281_axes_0 = const()[name = string("normed_281_axes_0"), val = tensor([-1])]; fp16 var_9383_to_fp16 = const()[name = string("op_9383_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_281_cast_fp16 = layer_norm(axes = normed_281_axes_0, epsilon = var_9383_to_fp16, x = input_313)[name = string("normed_281_cast_fp16")]; tensor normed_283_begin_0 = const()[name = string("normed_283_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_283_end_0 = const()[name = string("normed_283_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_283_end_mask_0 = const()[name = string("normed_283_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_283 = slice_by_index(begin = normed_283_begin_0, end = normed_283_end_0, end_mask = normed_283_end_mask_0, x = normed_281_cast_fp16)[name = string("normed_283")]; tensor const_263 = const()[name = string("const_263"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320006592)))]; tensor k_35 = mul(x = normed_283, y = const_263)[name = string("k_35")]; tensor var_9409 = mul(x = q_35, y = cos_1)[name = string("op_9409")]; tensor var_9414_begin_0 = const()[name = string("op_9414_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9414_end_0 = const()[name = string("op_9414_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_9414_end_mask_0 = const()[name = string("op_9414_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9414 = slice_by_index(begin = var_9414_begin_0, end = var_9414_end_0, end_mask = var_9414_end_mask_0, x = q_35)[name = string("op_9414")]; fp16 const_264_promoted = const()[name = string("const_264_promoted"), val = fp16(-0x1p+0)]; tensor var_9415 = mul(x = var_9414, y = const_264_promoted)[name = string("op_9415")]; tensor var_9420_begin_0 = const()[name = string("op_9420_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9420_end_0 = const()[name = string("op_9420_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_9420_end_mask_0 = const()[name = string("op_9420_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9420 = slice_by_index(begin = var_9420_begin_0, end = var_9420_end_0, end_mask = var_9420_end_mask_0, x = q_35)[name = string("op_9420")]; int32 var_9422 = const()[name = string("op_9422"), val = int32(-1)]; bool var_9423_interleave_0 = const()[name = string("op_9423_interleave_0"), val = bool(false)]; tensor var_9423 = concat(axis = var_9422, interleave = var_9423_interleave_0, values = (var_9415, var_9420))[name = string("op_9423")]; tensor var_9424 = mul(x = var_9423, y = sin_1)[name = string("op_9424")]; tensor query_35 = add(x = var_9409, y = var_9424)[name = string("query_35")]; tensor var_9427 = mul(x = k_35, y = cos_1)[name = string("op_9427")]; tensor var_9432_begin_0 = const()[name = string("op_9432_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9432_end_0 = const()[name = string("op_9432_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_9432_end_mask_0 = const()[name = string("op_9432_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9432 = slice_by_index(begin = var_9432_begin_0, end = var_9432_end_0, end_mask = var_9432_end_mask_0, x = k_35)[name = string("op_9432")]; fp16 const_265_promoted = const()[name = string("const_265_promoted"), val = fp16(-0x1p+0)]; tensor var_9433 = mul(x = var_9432, y = const_265_promoted)[name = string("op_9433")]; tensor var_9438_begin_0 = const()[name = string("op_9438_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9438_end_0 = const()[name = string("op_9438_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_9438_end_mask_0 = const()[name = string("op_9438_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9438 = slice_by_index(begin = var_9438_begin_0, end = var_9438_end_0, end_mask = var_9438_end_mask_0, x = k_35)[name = string("op_9438")]; int32 var_9440 = const()[name = string("op_9440"), val = int32(-1)]; bool var_9441_interleave_0 = const()[name = string("op_9441_interleave_0"), val = bool(false)]; tensor var_9441 = concat(axis = var_9440, interleave = var_9441_interleave_0, values = (var_9433, var_9438))[name = string("op_9441")]; tensor var_9442 = mul(x = var_9441, y = sin_1)[name = string("op_9442")]; tensor key_35 = add(x = var_9427, y = var_9442)[name = string("key_35")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; int32 concat_308_axis_0 = const()[name = string("concat_308_axis_0"), val = int32(0)]; bool concat_308_interleave_0 = const()[name = string("concat_308_interleave_0"), val = bool(false)]; tensor concat_308 = concat(axis = concat_308_axis_0, interleave = concat_308_interleave_0, values = (expand_dims_204, expand_dims_205, current_pos, expand_dims_207))[name = string("concat_308")]; tensor concat_309_values1_0 = const()[name = string("concat_309_values1_0"), val = tensor([0])]; tensor concat_309_values3_0 = const()[name = string("concat_309_values3_0"), val = tensor([0])]; int32 concat_309_axis_0 = const()[name = string("concat_309_axis_0"), val = int32(0)]; bool concat_309_interleave_0 = const()[name = string("concat_309_interleave_0"), val = bool(false)]; tensor concat_309 = concat(axis = concat_309_axis_0, interleave = concat_309_interleave_0, values = (expand_dims_208, concat_309_values1_0, var_1746, concat_309_values3_0))[name = string("concat_309")]; tensor model_model_kv_cache_0_internal_tensor_assign_35_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_35_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_35_cast_fp16 = slice_update(begin = concat_308, begin_mask = model_model_kv_cache_0_internal_tensor_assign_35_begin_mask_0, end = concat_309, end_mask = model_model_kv_cache_0_internal_tensor_assign_35_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_35_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_35_stride_0, update = key_35, x = coreml_update_state_89)[name = string("model_model_kv_cache_0_internal_tensor_assign_35_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_35_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_202_write_state")]; tensor coreml_update_state_90 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_202")]; tensor expand_dims_210 = const()[name = string("expand_dims_210"), val = tensor([45])]; tensor expand_dims_211 = const()[name = string("expand_dims_211"), val = tensor([0])]; tensor expand_dims_213 = const()[name = string("expand_dims_213"), val = tensor([0])]; tensor expand_dims_214 = const()[name = string("expand_dims_214"), val = tensor([46])]; int32 concat_312_axis_0 = const()[name = string("concat_312_axis_0"), val = int32(0)]; bool concat_312_interleave_0 = const()[name = string("concat_312_interleave_0"), val = bool(false)]; tensor concat_312 = concat(axis = concat_312_axis_0, interleave = concat_312_interleave_0, values = (expand_dims_210, expand_dims_211, current_pos, expand_dims_213))[name = string("concat_312")]; tensor concat_313_values1_0 = const()[name = string("concat_313_values1_0"), val = tensor([0])]; tensor concat_313_values3_0 = const()[name = string("concat_313_values3_0"), val = tensor([0])]; int32 concat_313_axis_0 = const()[name = string("concat_313_axis_0"), val = int32(0)]; bool concat_313_interleave_0 = const()[name = string("concat_313_interleave_0"), val = bool(false)]; tensor concat_313 = concat(axis = concat_313_axis_0, interleave = concat_313_interleave_0, values = (expand_dims_214, concat_313_values1_0, var_1746, concat_313_values3_0))[name = string("concat_313")]; tensor model_model_kv_cache_0_internal_tensor_assign_36_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_36_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_36_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_171 = transpose(perm = var_9351, x = var_9346)[name = string("transpose_95")]; tensor model_model_kv_cache_0_internal_tensor_assign_36_cast_fp16 = slice_update(begin = concat_312, begin_mask = model_model_kv_cache_0_internal_tensor_assign_36_begin_mask_0, end = concat_313, end_mask = model_model_kv_cache_0_internal_tensor_assign_36_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_36_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_36_stride_0, update = value_171, x = coreml_update_state_90)[name = string("model_model_kv_cache_0_internal_tensor_assign_36_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_36_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_203_write_state")]; tensor coreml_update_state_91 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_203")]; tensor var_9513_begin_0 = const()[name = string("op_9513_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_9513_end_0 = const()[name = string("op_9513_end_0"), val = tensor([18, 8, 1536, 128])]; tensor var_9513_end_mask_0 = const()[name = string("op_9513_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9513_cast_fp16 = slice_by_index(begin = var_9513_begin_0, end = var_9513_end_0, end_mask = var_9513_end_mask_0, x = coreml_update_state_91)[name = string("op_9513_cast_fp16")]; tensor key_cache_35_axes_0 = const()[name = string("key_cache_35_axes_0"), val = tensor([0])]; tensor key_cache_35_cast_fp16 = squeeze(axes = key_cache_35_axes_0, x = var_9513_cast_fp16)[name = string("key_cache_35_cast_fp16")]; tensor var_9520_begin_0 = const()[name = string("op_9520_begin_0"), val = tensor([45, 0, 0, 0])]; tensor var_9520_end_0 = const()[name = string("op_9520_end_0"), val = tensor([46, 8, 1536, 128])]; tensor var_9520_end_mask_0 = const()[name = string("op_9520_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9520_cast_fp16 = slice_by_index(begin = var_9520_begin_0, end = var_9520_end_0, end_mask = var_9520_end_mask_0, x = coreml_update_state_91)[name = string("op_9520_cast_fp16")]; tensor value_cache_35_axes_0 = const()[name = string("value_cache_35_axes_0"), val = tensor([0])]; tensor value_cache_35_cast_fp16 = squeeze(axes = value_cache_35_axes_0, x = var_9520_cast_fp16)[name = string("value_cache_35_cast_fp16")]; tensor var_9544_axes_0 = const()[name = string("op_9544_axes_0"), val = tensor([1])]; tensor var_9544_cast_fp16 = expand_dims(axes = var_9544_axes_0, x = key_cache_35_cast_fp16)[name = string("op_9544_cast_fp16")]; tensor var_9549 = const()[name = string("op_9549"), val = tensor([1, 2, 1, 1])]; tensor value_175_cast_fp16 = tile(reps = var_9549, x = var_9544_cast_fp16)[name = string("value_175_cast_fp16")]; tensor var_9555 = const()[name = string("op_9555"), val = tensor([1, 16, 1536, 128])]; tensor key_states_71_cast_fp16 = reshape(shape = var_9555, x = value_175_cast_fp16)[name = string("key_states_71_cast_fp16")]; tensor var_9558_axes_0 = const()[name = string("op_9558_axes_0"), val = tensor([1])]; tensor var_9558_cast_fp16 = expand_dims(axes = var_9558_axes_0, x = value_cache_35_cast_fp16)[name = string("op_9558_cast_fp16")]; tensor var_9563 = const()[name = string("op_9563"), val = tensor([1, 2, 1, 1])]; tensor value_179_cast_fp16 = tile(reps = var_9563, x = var_9558_cast_fp16)[name = string("value_179_cast_fp16")]; bool var_9584_transpose_x_0 = const()[name = string("op_9584_transpose_x_0"), val = bool(false)]; bool var_9584_transpose_y_0 = const()[name = string("op_9584_transpose_y_0"), val = bool(true)]; tensor var_9584 = matmul(transpose_x = var_9584_transpose_x_0, transpose_y = var_9584_transpose_y_0, x = query_35, y = key_states_71_cast_fp16)[name = string("op_9584")]; fp16 var_9585_to_fp16 = const()[name = string("op_9585_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_69_cast_fp16 = mul(x = var_9584, y = var_9585_to_fp16)[name = string("attention_69_cast_fp16")]; tensor attention_71_cast_fp16 = add(x = attention_69_cast_fp16, y = causal_mask)[name = string("attention_71_cast_fp16")]; int32 var_9594 = const()[name = string("op_9594"), val = int32(-1)]; tensor var_9596_cast_fp16 = softmax(axis = var_9594, x = attention_71_cast_fp16)[name = string("op_9596_cast_fp16")]; tensor concat_318 = const()[name = string("concat_318"), val = tensor([16, 64, 1536])]; tensor reshape_51_cast_fp16 = reshape(shape = concat_318, x = var_9596_cast_fp16)[name = string("reshape_51_cast_fp16")]; tensor concat_319 = const()[name = string("concat_319"), val = tensor([16, 1536, 128])]; tensor reshape_52_cast_fp16 = reshape(shape = concat_319, x = value_179_cast_fp16)[name = string("reshape_52_cast_fp16")]; bool matmul_17_transpose_x_0 = const()[name = string("matmul_17_transpose_x_0"), val = bool(false)]; bool matmul_17_transpose_y_0 = const()[name = string("matmul_17_transpose_y_0"), val = bool(false)]; tensor matmul_17_cast_fp16 = matmul(transpose_x = matmul_17_transpose_x_0, transpose_y = matmul_17_transpose_y_0, x = reshape_51_cast_fp16, y = reshape_52_cast_fp16)[name = string("matmul_17_cast_fp16")]; tensor concat_323 = const()[name = string("concat_323"), val = tensor([1, 16, 64, 128])]; tensor reshape_53_cast_fp16 = reshape(shape = concat_323, x = matmul_17_cast_fp16)[name = string("reshape_53_cast_fp16")]; tensor var_9608_perm_0 = const()[name = string("op_9608_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_9614 = const()[name = string("op_9614"), val = tensor([1, 64, 2048])]; tensor var_9608_cast_fp16 = transpose(perm = var_9608_perm_0, x = reshape_53_cast_fp16)[name = string("transpose_94")]; tensor output_105_cast_fp16 = reshape(shape = var_9614, x = var_9608_cast_fp16)[name = string("output_105_cast_fp16")]; tensor var_9619 = const()[name = string("op_9619"), val = tensor([0, 2, 1])]; string var_9635_pad_type_0 = const()[name = string("op_9635_pad_type_0"), val = string("valid")]; int32 var_9635_groups_0 = const()[name = string("op_9635_groups_0"), val = int32(1)]; tensor var_9635_strides_0 = const()[name = string("op_9635_strides_0"), val = tensor([1])]; tensor var_9635_pad_0 = const()[name = string("op_9635_pad_0"), val = tensor([0, 0])]; tensor var_9635_dilations_0 = const()[name = string("op_9635_dilations_0"), val = tensor([1])]; tensor squeeze_17_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320006912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321579840))))[name = string("squeeze_17_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_9620_cast_fp16 = transpose(perm = var_9619, x = output_105_cast_fp16)[name = string("transpose_93")]; tensor var_9635_cast_fp16 = conv(dilations = var_9635_dilations_0, groups = var_9635_groups_0, pad = var_9635_pad_0, pad_type = var_9635_pad_type_0, strides = var_9635_strides_0, weight = squeeze_17_cast_fp16_to_fp32_to_fp16_palettized, x = var_9620_cast_fp16)[name = string("op_9635_cast_fp16")]; tensor var_9639 = const()[name = string("op_9639"), val = tensor([0, 2, 1])]; tensor attn_output_35_cast_fp16 = transpose(perm = var_9639, x = var_9635_cast_fp16)[name = string("transpose_92")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_171_cast_fp16, y = attn_output_35_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; int32 var_9654 = const()[name = string("op_9654"), val = int32(-1)]; fp16 const_267_promoted_to_fp16 = const()[name = string("const_267_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9656_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_267_promoted_to_fp16)[name = string("op_9656_cast_fp16")]; bool input_317_interleave_0 = const()[name = string("input_317_interleave_0"), val = bool(false)]; tensor input_317_cast_fp16 = concat(axis = var_9654, interleave = input_317_interleave_0, values = (hidden_states_179_cast_fp16, var_9656_cast_fp16))[name = string("input_317_cast_fp16")]; tensor normed_285_axes_0 = const()[name = string("normed_285_axes_0"), val = tensor([-1])]; fp16 var_9651_to_fp16 = const()[name = string("op_9651_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_285_cast_fp16 = layer_norm(axes = normed_285_axes_0, epsilon = var_9651_to_fp16, x = input_317_cast_fp16)[name = string("normed_285_cast_fp16")]; tensor normed_287_begin_0 = const()[name = string("normed_287_begin_0"), val = tensor([0, 0, 0])]; tensor normed_287_end_0 = const()[name = string("normed_287_end_0"), val = tensor([1, 64, 1024])]; tensor normed_287_end_mask_0 = const()[name = string("normed_287_end_mask_0"), val = tensor([true, true, false])]; tensor normed_287_cast_fp16 = slice_by_index(begin = normed_287_begin_0, end = normed_287_end_0, end_mask = normed_287_end_mask_0, x = normed_285_cast_fp16)[name = string("normed_287_cast_fp16")]; tensor const_269_promoted_to_fp16 = const()[name = string("const_269_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321596288)))]; tensor x_69_cast_fp16 = mul(x = normed_287_cast_fp16, y = const_269_promoted_to_fp16)[name = string("x_69_cast_fp16")]; tensor var_9676 = const()[name = string("op_9676"), val = tensor([0, 2, 1])]; tensor input_319_axes_0 = const()[name = string("input_319_axes_0"), val = tensor([2])]; tensor var_9677 = transpose(perm = var_9676, x = x_69_cast_fp16)[name = string("transpose_91")]; tensor input_319 = expand_dims(axes = input_319_axes_0, x = var_9677)[name = string("input_319")]; string input_321_pad_type_0 = const()[name = string("input_321_pad_type_0"), val = string("valid")]; tensor input_321_strides_0 = const()[name = string("input_321_strides_0"), val = tensor([1, 1])]; tensor input_321_pad_0 = const()[name = string("input_321_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_321_dilations_0 = const()[name = string("input_321_dilations_0"), val = tensor([1, 1])]; int32 input_321_groups_0 = const()[name = string("input_321_groups_0"), val = int32(1)]; tensor input_321 = conv(dilations = input_321_dilations_0, groups = input_321_groups_0, pad = input_321_pad_0, pad_type = input_321_pad_type_0, strides = input_321_strides_0, weight = model_model_layers_17_mlp_gate_proj_weight_palettized, x = input_319)[name = string("input_321")]; string b_35_pad_type_0 = const()[name = string("b_35_pad_type_0"), val = string("valid")]; tensor b_35_strides_0 = const()[name = string("b_35_strides_0"), val = tensor([1, 1])]; tensor b_35_pad_0 = const()[name = string("b_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_35_dilations_0 = const()[name = string("b_35_dilations_0"), val = tensor([1, 1])]; int32 b_35_groups_0 = const()[name = string("b_35_groups_0"), val = int32(1)]; tensor b_35 = conv(dilations = b_35_dilations_0, groups = b_35_groups_0, pad = b_35_pad_0, pad_type = b_35_pad_type_0, strides = b_35_strides_0, weight = model_model_layers_17_mlp_up_proj_weight_palettized, x = input_319)[name = string("b_35")]; tensor c_35 = silu(x = input_321)[name = string("c_35")]; tensor input_323 = mul(x = c_35, y = b_35)[name = string("input_323")]; string e_35_pad_type_0 = const()[name = string("e_35_pad_type_0"), val = string("valid")]; tensor e_35_strides_0 = const()[name = string("e_35_strides_0"), val = tensor([1, 1])]; tensor e_35_pad_0 = const()[name = string("e_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_35_dilations_0 = const()[name = string("e_35_dilations_0"), val = tensor([1, 1])]; int32 e_35_groups_0 = const()[name = string("e_35_groups_0"), val = int32(1)]; tensor e_35 = conv(dilations = e_35_dilations_0, groups = e_35_groups_0, pad = e_35_pad_0, pad_type = e_35_pad_type_0, strides = e_35_strides_0, weight = model_model_layers_17_mlp_down_proj_weight_palettized, x = input_323)[name = string("e_35")]; tensor var_9699_axes_0 = const()[name = string("op_9699_axes_0"), val = tensor([2])]; tensor var_9699 = squeeze(axes = var_9699_axes_0, x = e_35)[name = string("op_9699")]; tensor var_9700 = const()[name = string("op_9700"), val = tensor([0, 2, 1])]; tensor var_9701 = transpose(perm = var_9700, x = var_9699)[name = string("transpose_90")]; tensor hidden_states_181_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = var_9701)[name = string("hidden_states_181_cast_fp16")]; int32 var_9715 = const()[name = string("op_9715"), val = int32(-1)]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9717_cast_fp16 = mul(x = hidden_states_181_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9717_cast_fp16")]; bool input_325_interleave_0 = const()[name = string("input_325_interleave_0"), val = bool(false)]; tensor input_325_cast_fp16 = concat(axis = var_9715, interleave = input_325_interleave_0, values = (hidden_states_181_cast_fp16, var_9717_cast_fp16))[name = string("input_325_cast_fp16")]; tensor normed_289_axes_0 = const()[name = string("normed_289_axes_0"), val = tensor([-1])]; fp16 var_9712_to_fp16 = const()[name = string("op_9712_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_289_cast_fp16 = layer_norm(axes = normed_289_axes_0, epsilon = var_9712_to_fp16, x = input_325_cast_fp16)[name = string("normed_289_cast_fp16")]; tensor normed_291_begin_0 = const()[name = string("normed_291_begin_0"), val = tensor([0, 0, 0])]; tensor normed_291_end_0 = const()[name = string("normed_291_end_0"), val = tensor([1, 64, 1024])]; tensor normed_291_end_mask_0 = const()[name = string("normed_291_end_mask_0"), val = tensor([true, true, false])]; tensor normed_291_cast_fp16 = slice_by_index(begin = normed_291_begin_0, end = normed_291_end_0, end_mask = normed_291_end_mask_0, x = normed_289_cast_fp16)[name = string("normed_291_cast_fp16")]; tensor const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321598400)))]; tensor hidden_states_183_cast_fp16 = mul(x = normed_291_cast_fp16, y = const_272_promoted_to_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor var_9729 = const()[name = string("op_9729"), val = tensor([0, 2, 1])]; tensor var_9732_axes_0 = const()[name = string("op_9732_axes_0"), val = tensor([2])]; tensor var_9730_cast_fp16 = transpose(perm = var_9729, x = hidden_states_183_cast_fp16)[name = string("transpose_89")]; tensor var_9732_cast_fp16 = expand_dims(axes = var_9732_axes_0, x = var_9730_cast_fp16)[name = string("op_9732_cast_fp16")]; string var_9748_pad_type_0 = const()[name = string("op_9748_pad_type_0"), val = string("valid")]; tensor var_9748_strides_0 = const()[name = string("op_9748_strides_0"), val = tensor([1, 1])]; tensor var_9748_pad_0 = const()[name = string("op_9748_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9748_dilations_0 = const()[name = string("op_9748_dilations_0"), val = tensor([1, 1])]; int32 var_9748_groups_0 = const()[name = string("op_9748_groups_0"), val = int32(1)]; tensor var_9748 = conv(dilations = var_9748_dilations_0, groups = var_9748_groups_0, pad = var_9748_pad_0, pad_type = var_9748_pad_type_0, strides = var_9748_strides_0, weight = model_model_layers_18_self_attn_q_proj_weight_palettized, x = var_9732_cast_fp16)[name = string("op_9748")]; tensor var_9753 = const()[name = string("op_9753"), val = tensor([1, 16, 128, 64])]; tensor var_9754 = reshape(shape = var_9753, x = var_9748)[name = string("op_9754")]; tensor var_9759 = const()[name = string("op_9759"), val = tensor([0, 1, 3, 2])]; string var_9771_pad_type_0 = const()[name = string("op_9771_pad_type_0"), val = string("valid")]; tensor var_9771_strides_0 = const()[name = string("op_9771_strides_0"), val = tensor([1, 1])]; tensor var_9771_pad_0 = const()[name = string("op_9771_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9771_dilations_0 = const()[name = string("op_9771_dilations_0"), val = tensor([1, 1])]; int32 var_9771_groups_0 = const()[name = string("op_9771_groups_0"), val = int32(1)]; tensor var_9771 = conv(dilations = var_9771_dilations_0, groups = var_9771_groups_0, pad = var_9771_pad_0, pad_type = var_9771_pad_type_0, strides = var_9771_strides_0, weight = model_model_layers_18_self_attn_k_proj_weight_palettized, x = var_9732_cast_fp16)[name = string("op_9771")]; tensor var_9776 = const()[name = string("op_9776"), val = tensor([1, 8, 128, 64])]; tensor var_9777 = reshape(shape = var_9776, x = var_9771)[name = string("op_9777")]; tensor var_9782 = const()[name = string("op_9782"), val = tensor([0, 1, 3, 2])]; string var_9794_pad_type_0 = const()[name = string("op_9794_pad_type_0"), val = string("valid")]; tensor var_9794_strides_0 = const()[name = string("op_9794_strides_0"), val = tensor([1, 1])]; tensor var_9794_pad_0 = const()[name = string("op_9794_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9794_dilations_0 = const()[name = string("op_9794_dilations_0"), val = tensor([1, 1])]; int32 var_9794_groups_0 = const()[name = string("op_9794_groups_0"), val = int32(1)]; tensor var_9794 = conv(dilations = var_9794_dilations_0, groups = var_9794_groups_0, pad = var_9794_pad_0, pad_type = var_9794_pad_type_0, strides = var_9794_strides_0, weight = model_model_layers_18_self_attn_v_proj_weight_palettized, x = var_9732_cast_fp16)[name = string("op_9794")]; tensor var_9799 = const()[name = string("op_9799"), val = tensor([1, 8, 128, 64])]; tensor var_9800 = reshape(shape = var_9799, x = var_9794)[name = string("op_9800")]; tensor var_9805 = const()[name = string("op_9805"), val = tensor([0, 1, 3, 2])]; int32 var_9818 = const()[name = string("op_9818"), val = int32(-1)]; fp16 const_273_promoted = const()[name = string("const_273_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_185 = transpose(perm = var_9759, x = var_9754)[name = string("transpose_88")]; tensor var_9820 = mul(x = hidden_states_185, y = const_273_promoted)[name = string("op_9820")]; bool input_329_interleave_0 = const()[name = string("input_329_interleave_0"), val = bool(false)]; tensor input_329 = concat(axis = var_9818, interleave = input_329_interleave_0, values = (hidden_states_185, var_9820))[name = string("input_329")]; tensor normed_293_axes_0 = const()[name = string("normed_293_axes_0"), val = tensor([-1])]; fp16 var_9815_to_fp16 = const()[name = string("op_9815_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_293_cast_fp16 = layer_norm(axes = normed_293_axes_0, epsilon = var_9815_to_fp16, x = input_329)[name = string("normed_293_cast_fp16")]; tensor normed_295_begin_0 = const()[name = string("normed_295_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_295_end_0 = const()[name = string("normed_295_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_295_end_mask_0 = const()[name = string("normed_295_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_295 = slice_by_index(begin = normed_295_begin_0, end = normed_295_end_0, end_mask = normed_295_end_mask_0, x = normed_293_cast_fp16)[name = string("normed_295")]; tensor const_275 = const()[name = string("const_275"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321600512)))]; tensor q_37 = mul(x = normed_295, y = const_275)[name = string("q_37")]; int32 var_9840 = const()[name = string("op_9840"), val = int32(-1)]; fp16 const_276_promoted = const()[name = string("const_276_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_187 = transpose(perm = var_9782, x = var_9777)[name = string("transpose_87")]; tensor var_9842 = mul(x = hidden_states_187, y = const_276_promoted)[name = string("op_9842")]; bool input_331_interleave_0 = const()[name = string("input_331_interleave_0"), val = bool(false)]; tensor input_331 = concat(axis = var_9840, interleave = input_331_interleave_0, values = (hidden_states_187, var_9842))[name = string("input_331")]; tensor normed_297_axes_0 = const()[name = string("normed_297_axes_0"), val = tensor([-1])]; fp16 var_9837_to_fp16 = const()[name = string("op_9837_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_297_cast_fp16 = layer_norm(axes = normed_297_axes_0, epsilon = var_9837_to_fp16, x = input_331)[name = string("normed_297_cast_fp16")]; tensor normed_299_begin_0 = const()[name = string("normed_299_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_299_end_0 = const()[name = string("normed_299_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_299_end_mask_0 = const()[name = string("normed_299_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_299 = slice_by_index(begin = normed_299_begin_0, end = normed_299_end_0, end_mask = normed_299_end_mask_0, x = normed_297_cast_fp16)[name = string("normed_299")]; tensor const_278 = const()[name = string("const_278"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321600832)))]; tensor k_37 = mul(x = normed_299, y = const_278)[name = string("k_37")]; tensor var_9863 = mul(x = q_37, y = cos_1)[name = string("op_9863")]; tensor var_9868_begin_0 = const()[name = string("op_9868_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9868_end_0 = const()[name = string("op_9868_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_9868_end_mask_0 = const()[name = string("op_9868_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9868 = slice_by_index(begin = var_9868_begin_0, end = var_9868_end_0, end_mask = var_9868_end_mask_0, x = q_37)[name = string("op_9868")]; fp16 const_279_promoted = const()[name = string("const_279_promoted"), val = fp16(-0x1p+0)]; tensor var_9869 = mul(x = var_9868, y = const_279_promoted)[name = string("op_9869")]; tensor var_9874_begin_0 = const()[name = string("op_9874_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9874_end_0 = const()[name = string("op_9874_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_9874_end_mask_0 = const()[name = string("op_9874_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9874 = slice_by_index(begin = var_9874_begin_0, end = var_9874_end_0, end_mask = var_9874_end_mask_0, x = q_37)[name = string("op_9874")]; int32 var_9876 = const()[name = string("op_9876"), val = int32(-1)]; bool var_9877_interleave_0 = const()[name = string("op_9877_interleave_0"), val = bool(false)]; tensor var_9877 = concat(axis = var_9876, interleave = var_9877_interleave_0, values = (var_9869, var_9874))[name = string("op_9877")]; tensor var_9878 = mul(x = var_9877, y = sin_1)[name = string("op_9878")]; tensor query_37 = add(x = var_9863, y = var_9878)[name = string("query_37")]; tensor var_9881 = mul(x = k_37, y = cos_1)[name = string("op_9881")]; tensor var_9886_begin_0 = const()[name = string("op_9886_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_9886_end_0 = const()[name = string("op_9886_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_9886_end_mask_0 = const()[name = string("op_9886_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9886 = slice_by_index(begin = var_9886_begin_0, end = var_9886_end_0, end_mask = var_9886_end_mask_0, x = k_37)[name = string("op_9886")]; fp16 const_280_promoted = const()[name = string("const_280_promoted"), val = fp16(-0x1p+0)]; tensor var_9887 = mul(x = var_9886, y = const_280_promoted)[name = string("op_9887")]; tensor var_9892_begin_0 = const()[name = string("op_9892_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_9892_end_0 = const()[name = string("op_9892_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_9892_end_mask_0 = const()[name = string("op_9892_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_9892 = slice_by_index(begin = var_9892_begin_0, end = var_9892_end_0, end_mask = var_9892_end_mask_0, x = k_37)[name = string("op_9892")]; int32 var_9894 = const()[name = string("op_9894"), val = int32(-1)]; bool var_9895_interleave_0 = const()[name = string("op_9895_interleave_0"), val = bool(false)]; tensor var_9895 = concat(axis = var_9894, interleave = var_9895_interleave_0, values = (var_9887, var_9892))[name = string("op_9895")]; tensor var_9896 = mul(x = var_9895, y = sin_1)[name = string("op_9896")]; tensor key_37 = add(x = var_9881, y = var_9896)[name = string("key_37")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; int32 concat_326_axis_0 = const()[name = string("concat_326_axis_0"), val = int32(0)]; bool concat_326_interleave_0 = const()[name = string("concat_326_interleave_0"), val = bool(false)]; tensor concat_326 = concat(axis = concat_326_axis_0, interleave = concat_326_interleave_0, values = (expand_dims_216, expand_dims_217, current_pos, expand_dims_219))[name = string("concat_326")]; tensor concat_327_values1_0 = const()[name = string("concat_327_values1_0"), val = tensor([0])]; tensor concat_327_values3_0 = const()[name = string("concat_327_values3_0"), val = tensor([0])]; int32 concat_327_axis_0 = const()[name = string("concat_327_axis_0"), val = int32(0)]; bool concat_327_interleave_0 = const()[name = string("concat_327_interleave_0"), val = bool(false)]; tensor concat_327 = concat(axis = concat_327_axis_0, interleave = concat_327_interleave_0, values = (expand_dims_220, concat_327_values1_0, var_1746, concat_327_values3_0))[name = string("concat_327")]; tensor model_model_kv_cache_0_internal_tensor_assign_37_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_37_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_37_cast_fp16 = slice_update(begin = concat_326, begin_mask = model_model_kv_cache_0_internal_tensor_assign_37_begin_mask_0, end = concat_327, end_mask = model_model_kv_cache_0_internal_tensor_assign_37_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_37_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_37_stride_0, update = key_37, x = coreml_update_state_91)[name = string("model_model_kv_cache_0_internal_tensor_assign_37_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_37_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_204_write_state")]; tensor coreml_update_state_92 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_204")]; tensor expand_dims_222 = const()[name = string("expand_dims_222"), val = tensor([46])]; tensor expand_dims_223 = const()[name = string("expand_dims_223"), val = tensor([0])]; tensor expand_dims_225 = const()[name = string("expand_dims_225"), val = tensor([0])]; tensor expand_dims_226 = const()[name = string("expand_dims_226"), val = tensor([47])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_222, expand_dims_223, current_pos, expand_dims_225))[name = string("concat_330")]; tensor concat_331_values1_0 = const()[name = string("concat_331_values1_0"), val = tensor([0])]; tensor concat_331_values3_0 = const()[name = string("concat_331_values3_0"), val = tensor([0])]; int32 concat_331_axis_0 = const()[name = string("concat_331_axis_0"), val = int32(0)]; bool concat_331_interleave_0 = const()[name = string("concat_331_interleave_0"), val = bool(false)]; tensor concat_331 = concat(axis = concat_331_axis_0, interleave = concat_331_interleave_0, values = (expand_dims_226, concat_331_values1_0, var_1746, concat_331_values3_0))[name = string("concat_331")]; tensor model_model_kv_cache_0_internal_tensor_assign_38_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_38_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_38_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_181 = transpose(perm = var_9805, x = var_9800)[name = string("transpose_86")]; tensor model_model_kv_cache_0_internal_tensor_assign_38_cast_fp16 = slice_update(begin = concat_330, begin_mask = model_model_kv_cache_0_internal_tensor_assign_38_begin_mask_0, end = concat_331, end_mask = model_model_kv_cache_0_internal_tensor_assign_38_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_38_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_38_stride_0, update = value_181, x = coreml_update_state_92)[name = string("model_model_kv_cache_0_internal_tensor_assign_38_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_38_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_205_write_state")]; tensor coreml_update_state_93 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_205")]; tensor var_9967_begin_0 = const()[name = string("op_9967_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_9967_end_0 = const()[name = string("op_9967_end_0"), val = tensor([19, 8, 1536, 128])]; tensor var_9967_end_mask_0 = const()[name = string("op_9967_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9967_cast_fp16 = slice_by_index(begin = var_9967_begin_0, end = var_9967_end_0, end_mask = var_9967_end_mask_0, x = coreml_update_state_93)[name = string("op_9967_cast_fp16")]; tensor key_cache_37_axes_0 = const()[name = string("key_cache_37_axes_0"), val = tensor([0])]; tensor key_cache_37_cast_fp16 = squeeze(axes = key_cache_37_axes_0, x = var_9967_cast_fp16)[name = string("key_cache_37_cast_fp16")]; tensor var_9974_begin_0 = const()[name = string("op_9974_begin_0"), val = tensor([46, 0, 0, 0])]; tensor var_9974_end_0 = const()[name = string("op_9974_end_0"), val = tensor([47, 8, 1536, 128])]; tensor var_9974_end_mask_0 = const()[name = string("op_9974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9974_cast_fp16 = slice_by_index(begin = var_9974_begin_0, end = var_9974_end_0, end_mask = var_9974_end_mask_0, x = coreml_update_state_93)[name = string("op_9974_cast_fp16")]; tensor value_cache_37_axes_0 = const()[name = string("value_cache_37_axes_0"), val = tensor([0])]; tensor value_cache_37_cast_fp16 = squeeze(axes = value_cache_37_axes_0, x = var_9974_cast_fp16)[name = string("value_cache_37_cast_fp16")]; tensor var_9998_axes_0 = const()[name = string("op_9998_axes_0"), val = tensor([1])]; tensor var_9998_cast_fp16 = expand_dims(axes = var_9998_axes_0, x = key_cache_37_cast_fp16)[name = string("op_9998_cast_fp16")]; tensor var_10003 = const()[name = string("op_10003"), val = tensor([1, 2, 1, 1])]; tensor value_185_cast_fp16 = tile(reps = var_10003, x = var_9998_cast_fp16)[name = string("value_185_cast_fp16")]; tensor var_10009 = const()[name = string("op_10009"), val = tensor([1, 16, 1536, 128])]; tensor key_states_75_cast_fp16 = reshape(shape = var_10009, x = value_185_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor var_10012_axes_0 = const()[name = string("op_10012_axes_0"), val = tensor([1])]; tensor var_10012_cast_fp16 = expand_dims(axes = var_10012_axes_0, x = value_cache_37_cast_fp16)[name = string("op_10012_cast_fp16")]; tensor var_10017 = const()[name = string("op_10017"), val = tensor([1, 2, 1, 1])]; tensor value_189_cast_fp16 = tile(reps = var_10017, x = var_10012_cast_fp16)[name = string("value_189_cast_fp16")]; bool var_10038_transpose_x_0 = const()[name = string("op_10038_transpose_x_0"), val = bool(false)]; bool var_10038_transpose_y_0 = const()[name = string("op_10038_transpose_y_0"), val = bool(true)]; tensor var_10038 = matmul(transpose_x = var_10038_transpose_x_0, transpose_y = var_10038_transpose_y_0, x = query_37, y = key_states_75_cast_fp16)[name = string("op_10038")]; fp16 var_10039_to_fp16 = const()[name = string("op_10039_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_73_cast_fp16 = mul(x = var_10038, y = var_10039_to_fp16)[name = string("attention_73_cast_fp16")]; tensor attention_75_cast_fp16 = add(x = attention_73_cast_fp16, y = causal_mask)[name = string("attention_75_cast_fp16")]; int32 var_10048 = const()[name = string("op_10048"), val = int32(-1)]; tensor var_10050_cast_fp16 = softmax(axis = var_10048, x = attention_75_cast_fp16)[name = string("op_10050_cast_fp16")]; tensor concat_336 = const()[name = string("concat_336"), val = tensor([16, 64, 1536])]; tensor reshape_54_cast_fp16 = reshape(shape = concat_336, x = var_10050_cast_fp16)[name = string("reshape_54_cast_fp16")]; tensor concat_337 = const()[name = string("concat_337"), val = tensor([16, 1536, 128])]; tensor reshape_55_cast_fp16 = reshape(shape = concat_337, x = value_189_cast_fp16)[name = string("reshape_55_cast_fp16")]; bool matmul_18_transpose_x_0 = const()[name = string("matmul_18_transpose_x_0"), val = bool(false)]; bool matmul_18_transpose_y_0 = const()[name = string("matmul_18_transpose_y_0"), val = bool(false)]; tensor matmul_18_cast_fp16 = matmul(transpose_x = matmul_18_transpose_x_0, transpose_y = matmul_18_transpose_y_0, x = reshape_54_cast_fp16, y = reshape_55_cast_fp16)[name = string("matmul_18_cast_fp16")]; tensor concat_341 = const()[name = string("concat_341"), val = tensor([1, 16, 64, 128])]; tensor reshape_56_cast_fp16 = reshape(shape = concat_341, x = matmul_18_cast_fp16)[name = string("reshape_56_cast_fp16")]; tensor var_10062_perm_0 = const()[name = string("op_10062_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_10068 = const()[name = string("op_10068"), val = tensor([1, 64, 2048])]; tensor var_10062_cast_fp16 = transpose(perm = var_10062_perm_0, x = reshape_56_cast_fp16)[name = string("transpose_85")]; tensor output_111_cast_fp16 = reshape(shape = var_10068, x = var_10062_cast_fp16)[name = string("output_111_cast_fp16")]; tensor var_10073 = const()[name = string("op_10073"), val = tensor([0, 2, 1])]; string var_10089_pad_type_0 = const()[name = string("op_10089_pad_type_0"), val = string("valid")]; int32 var_10089_groups_0 = const()[name = string("op_10089_groups_0"), val = int32(1)]; tensor var_10089_strides_0 = const()[name = string("op_10089_strides_0"), val = tensor([1])]; tensor var_10089_pad_0 = const()[name = string("op_10089_pad_0"), val = tensor([0, 0])]; tensor var_10089_dilations_0 = const()[name = string("op_10089_dilations_0"), val = tensor([1])]; tensor squeeze_18_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321601152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323174080))))[name = string("squeeze_18_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10074_cast_fp16 = transpose(perm = var_10073, x = output_111_cast_fp16)[name = string("transpose_84")]; tensor var_10089_cast_fp16 = conv(dilations = var_10089_dilations_0, groups = var_10089_groups_0, pad = var_10089_pad_0, pad_type = var_10089_pad_type_0, strides = var_10089_strides_0, weight = squeeze_18_cast_fp16_to_fp32_to_fp16_palettized, x = var_10074_cast_fp16)[name = string("op_10089_cast_fp16")]; tensor var_10093 = const()[name = string("op_10093"), val = tensor([0, 2, 1])]; tensor attn_output_37_cast_fp16 = transpose(perm = var_10093, x = var_10089_cast_fp16)[name = string("transpose_83")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_181_cast_fp16, y = attn_output_37_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; int32 var_10108 = const()[name = string("op_10108"), val = int32(-1)]; fp16 const_282_promoted_to_fp16 = const()[name = string("const_282_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10110_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_282_promoted_to_fp16)[name = string("op_10110_cast_fp16")]; bool input_335_interleave_0 = const()[name = string("input_335_interleave_0"), val = bool(false)]; tensor input_335_cast_fp16 = concat(axis = var_10108, interleave = input_335_interleave_0, values = (hidden_states_189_cast_fp16, var_10110_cast_fp16))[name = string("input_335_cast_fp16")]; tensor normed_301_axes_0 = const()[name = string("normed_301_axes_0"), val = tensor([-1])]; fp16 var_10105_to_fp16 = const()[name = string("op_10105_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_301_cast_fp16 = layer_norm(axes = normed_301_axes_0, epsilon = var_10105_to_fp16, x = input_335_cast_fp16)[name = string("normed_301_cast_fp16")]; tensor normed_303_begin_0 = const()[name = string("normed_303_begin_0"), val = tensor([0, 0, 0])]; tensor normed_303_end_0 = const()[name = string("normed_303_end_0"), val = tensor([1, 64, 1024])]; tensor normed_303_end_mask_0 = const()[name = string("normed_303_end_mask_0"), val = tensor([true, true, false])]; tensor normed_303_cast_fp16 = slice_by_index(begin = normed_303_begin_0, end = normed_303_end_0, end_mask = normed_303_end_mask_0, x = normed_301_cast_fp16)[name = string("normed_303_cast_fp16")]; tensor const_284_promoted_to_fp16 = const()[name = string("const_284_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323190528)))]; tensor x_73_cast_fp16 = mul(x = normed_303_cast_fp16, y = const_284_promoted_to_fp16)[name = string("x_73_cast_fp16")]; tensor var_10130 = const()[name = string("op_10130"), val = tensor([0, 2, 1])]; tensor input_337_axes_0 = const()[name = string("input_337_axes_0"), val = tensor([2])]; tensor var_10131 = transpose(perm = var_10130, x = x_73_cast_fp16)[name = string("transpose_82")]; tensor input_337 = expand_dims(axes = input_337_axes_0, x = var_10131)[name = string("input_337")]; string input_339_pad_type_0 = const()[name = string("input_339_pad_type_0"), val = string("valid")]; tensor input_339_strides_0 = const()[name = string("input_339_strides_0"), val = tensor([1, 1])]; tensor input_339_pad_0 = const()[name = string("input_339_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_339_dilations_0 = const()[name = string("input_339_dilations_0"), val = tensor([1, 1])]; int32 input_339_groups_0 = const()[name = string("input_339_groups_0"), val = int32(1)]; tensor input_339 = conv(dilations = input_339_dilations_0, groups = input_339_groups_0, pad = input_339_pad_0, pad_type = input_339_pad_type_0, strides = input_339_strides_0, weight = model_model_layers_18_mlp_gate_proj_weight_palettized, x = input_337)[name = string("input_339")]; string b_37_pad_type_0 = const()[name = string("b_37_pad_type_0"), val = string("valid")]; tensor b_37_strides_0 = const()[name = string("b_37_strides_0"), val = tensor([1, 1])]; tensor b_37_pad_0 = const()[name = string("b_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_37_dilations_0 = const()[name = string("b_37_dilations_0"), val = tensor([1, 1])]; int32 b_37_groups_0 = const()[name = string("b_37_groups_0"), val = int32(1)]; tensor b_37 = conv(dilations = b_37_dilations_0, groups = b_37_groups_0, pad = b_37_pad_0, pad_type = b_37_pad_type_0, strides = b_37_strides_0, weight = model_model_layers_18_mlp_up_proj_weight_palettized, x = input_337)[name = string("b_37")]; tensor c_37 = silu(x = input_339)[name = string("c_37")]; tensor input_341 = mul(x = c_37, y = b_37)[name = string("input_341")]; string e_37_pad_type_0 = const()[name = string("e_37_pad_type_0"), val = string("valid")]; tensor e_37_strides_0 = const()[name = string("e_37_strides_0"), val = tensor([1, 1])]; tensor e_37_pad_0 = const()[name = string("e_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_37_dilations_0 = const()[name = string("e_37_dilations_0"), val = tensor([1, 1])]; int32 e_37_groups_0 = const()[name = string("e_37_groups_0"), val = int32(1)]; tensor e_37 = conv(dilations = e_37_dilations_0, groups = e_37_groups_0, pad = e_37_pad_0, pad_type = e_37_pad_type_0, strides = e_37_strides_0, weight = model_model_layers_18_mlp_down_proj_weight_palettized, x = input_341)[name = string("e_37")]; tensor var_10153_axes_0 = const()[name = string("op_10153_axes_0"), val = tensor([2])]; tensor var_10153 = squeeze(axes = var_10153_axes_0, x = e_37)[name = string("op_10153")]; tensor var_10154 = const()[name = string("op_10154"), val = tensor([0, 2, 1])]; tensor var_10155 = transpose(perm = var_10154, x = var_10153)[name = string("transpose_81")]; tensor hidden_states_191_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = var_10155)[name = string("hidden_states_191_cast_fp16")]; int32 var_10169 = const()[name = string("op_10169"), val = int32(-1)]; fp16 const_285_promoted_to_fp16 = const()[name = string("const_285_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10171_cast_fp16 = mul(x = hidden_states_191_cast_fp16, y = const_285_promoted_to_fp16)[name = string("op_10171_cast_fp16")]; bool input_343_interleave_0 = const()[name = string("input_343_interleave_0"), val = bool(false)]; tensor input_343_cast_fp16 = concat(axis = var_10169, interleave = input_343_interleave_0, values = (hidden_states_191_cast_fp16, var_10171_cast_fp16))[name = string("input_343_cast_fp16")]; tensor normed_305_axes_0 = const()[name = string("normed_305_axes_0"), val = tensor([-1])]; fp16 var_10166_to_fp16 = const()[name = string("op_10166_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_305_cast_fp16 = layer_norm(axes = normed_305_axes_0, epsilon = var_10166_to_fp16, x = input_343_cast_fp16)[name = string("normed_305_cast_fp16")]; tensor normed_307_begin_0 = const()[name = string("normed_307_begin_0"), val = tensor([0, 0, 0])]; tensor normed_307_end_0 = const()[name = string("normed_307_end_0"), val = tensor([1, 64, 1024])]; tensor normed_307_end_mask_0 = const()[name = string("normed_307_end_mask_0"), val = tensor([true, true, false])]; tensor normed_307_cast_fp16 = slice_by_index(begin = normed_307_begin_0, end = normed_307_end_0, end_mask = normed_307_end_mask_0, x = normed_305_cast_fp16)[name = string("normed_307_cast_fp16")]; tensor const_287_promoted_to_fp16 = const()[name = string("const_287_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323192640)))]; tensor hidden_states_193_cast_fp16 = mul(x = normed_307_cast_fp16, y = const_287_promoted_to_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor var_10183 = const()[name = string("op_10183"), val = tensor([0, 2, 1])]; tensor var_10186_axes_0 = const()[name = string("op_10186_axes_0"), val = tensor([2])]; tensor var_10184_cast_fp16 = transpose(perm = var_10183, x = hidden_states_193_cast_fp16)[name = string("transpose_80")]; tensor var_10186_cast_fp16 = expand_dims(axes = var_10186_axes_0, x = var_10184_cast_fp16)[name = string("op_10186_cast_fp16")]; string var_10202_pad_type_0 = const()[name = string("op_10202_pad_type_0"), val = string("valid")]; tensor var_10202_strides_0 = const()[name = string("op_10202_strides_0"), val = tensor([1, 1])]; tensor var_10202_pad_0 = const()[name = string("op_10202_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10202_dilations_0 = const()[name = string("op_10202_dilations_0"), val = tensor([1, 1])]; int32 var_10202_groups_0 = const()[name = string("op_10202_groups_0"), val = int32(1)]; tensor var_10202 = conv(dilations = var_10202_dilations_0, groups = var_10202_groups_0, pad = var_10202_pad_0, pad_type = var_10202_pad_type_0, strides = var_10202_strides_0, weight = model_model_layers_19_self_attn_q_proj_weight_palettized, x = var_10186_cast_fp16)[name = string("op_10202")]; tensor var_10207 = const()[name = string("op_10207"), val = tensor([1, 16, 128, 64])]; tensor var_10208 = reshape(shape = var_10207, x = var_10202)[name = string("op_10208")]; tensor var_10213 = const()[name = string("op_10213"), val = tensor([0, 1, 3, 2])]; string var_10225_pad_type_0 = const()[name = string("op_10225_pad_type_0"), val = string("valid")]; tensor var_10225_strides_0 = const()[name = string("op_10225_strides_0"), val = tensor([1, 1])]; tensor var_10225_pad_0 = const()[name = string("op_10225_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10225_dilations_0 = const()[name = string("op_10225_dilations_0"), val = tensor([1, 1])]; int32 var_10225_groups_0 = const()[name = string("op_10225_groups_0"), val = int32(1)]; tensor var_10225 = conv(dilations = var_10225_dilations_0, groups = var_10225_groups_0, pad = var_10225_pad_0, pad_type = var_10225_pad_type_0, strides = var_10225_strides_0, weight = model_model_layers_19_self_attn_k_proj_weight_palettized, x = var_10186_cast_fp16)[name = string("op_10225")]; tensor var_10230 = const()[name = string("op_10230"), val = tensor([1, 8, 128, 64])]; tensor var_10231 = reshape(shape = var_10230, x = var_10225)[name = string("op_10231")]; tensor var_10236 = const()[name = string("op_10236"), val = tensor([0, 1, 3, 2])]; string var_10248_pad_type_0 = const()[name = string("op_10248_pad_type_0"), val = string("valid")]; tensor var_10248_strides_0 = const()[name = string("op_10248_strides_0"), val = tensor([1, 1])]; tensor var_10248_pad_0 = const()[name = string("op_10248_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10248_dilations_0 = const()[name = string("op_10248_dilations_0"), val = tensor([1, 1])]; int32 var_10248_groups_0 = const()[name = string("op_10248_groups_0"), val = int32(1)]; tensor var_10248 = conv(dilations = var_10248_dilations_0, groups = var_10248_groups_0, pad = var_10248_pad_0, pad_type = var_10248_pad_type_0, strides = var_10248_strides_0, weight = model_model_layers_19_self_attn_v_proj_weight_palettized, x = var_10186_cast_fp16)[name = string("op_10248")]; tensor var_10253 = const()[name = string("op_10253"), val = tensor([1, 8, 128, 64])]; tensor var_10254 = reshape(shape = var_10253, x = var_10248)[name = string("op_10254")]; tensor var_10259 = const()[name = string("op_10259"), val = tensor([0, 1, 3, 2])]; int32 var_10272 = const()[name = string("op_10272"), val = int32(-1)]; fp16 const_288_promoted = const()[name = string("const_288_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_195 = transpose(perm = var_10213, x = var_10208)[name = string("transpose_79")]; tensor var_10274 = mul(x = hidden_states_195, y = const_288_promoted)[name = string("op_10274")]; bool input_347_interleave_0 = const()[name = string("input_347_interleave_0"), val = bool(false)]; tensor input_347 = concat(axis = var_10272, interleave = input_347_interleave_0, values = (hidden_states_195, var_10274))[name = string("input_347")]; tensor normed_309_axes_0 = const()[name = string("normed_309_axes_0"), val = tensor([-1])]; fp16 var_10269_to_fp16 = const()[name = string("op_10269_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_309_cast_fp16 = layer_norm(axes = normed_309_axes_0, epsilon = var_10269_to_fp16, x = input_347)[name = string("normed_309_cast_fp16")]; tensor normed_311_begin_0 = const()[name = string("normed_311_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_311_end_0 = const()[name = string("normed_311_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_311_end_mask_0 = const()[name = string("normed_311_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_311 = slice_by_index(begin = normed_311_begin_0, end = normed_311_end_0, end_mask = normed_311_end_mask_0, x = normed_309_cast_fp16)[name = string("normed_311")]; tensor const_290 = const()[name = string("const_290"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323194752)))]; tensor q_39 = mul(x = normed_311, y = const_290)[name = string("q_39")]; int32 var_10294 = const()[name = string("op_10294"), val = int32(-1)]; fp16 const_291_promoted = const()[name = string("const_291_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_197 = transpose(perm = var_10236, x = var_10231)[name = string("transpose_78")]; tensor var_10296 = mul(x = hidden_states_197, y = const_291_promoted)[name = string("op_10296")]; bool input_349_interleave_0 = const()[name = string("input_349_interleave_0"), val = bool(false)]; tensor input_349 = concat(axis = var_10294, interleave = input_349_interleave_0, values = (hidden_states_197, var_10296))[name = string("input_349")]; tensor normed_313_axes_0 = const()[name = string("normed_313_axes_0"), val = tensor([-1])]; fp16 var_10291_to_fp16 = const()[name = string("op_10291_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_313_cast_fp16 = layer_norm(axes = normed_313_axes_0, epsilon = var_10291_to_fp16, x = input_349)[name = string("normed_313_cast_fp16")]; tensor normed_315_begin_0 = const()[name = string("normed_315_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_315_end_0 = const()[name = string("normed_315_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_315_end_mask_0 = const()[name = string("normed_315_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_315 = slice_by_index(begin = normed_315_begin_0, end = normed_315_end_0, end_mask = normed_315_end_mask_0, x = normed_313_cast_fp16)[name = string("normed_315")]; tensor const_293 = const()[name = string("const_293"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323195072)))]; tensor k_39 = mul(x = normed_315, y = const_293)[name = string("k_39")]; tensor var_10317 = mul(x = q_39, y = cos_1)[name = string("op_10317")]; tensor var_10322_begin_0 = const()[name = string("op_10322_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10322_end_0 = const()[name = string("op_10322_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_10322_end_mask_0 = const()[name = string("op_10322_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10322 = slice_by_index(begin = var_10322_begin_0, end = var_10322_end_0, end_mask = var_10322_end_mask_0, x = q_39)[name = string("op_10322")]; fp16 const_294_promoted = const()[name = string("const_294_promoted"), val = fp16(-0x1p+0)]; tensor var_10323 = mul(x = var_10322, y = const_294_promoted)[name = string("op_10323")]; tensor var_10328_begin_0 = const()[name = string("op_10328_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10328_end_0 = const()[name = string("op_10328_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_10328_end_mask_0 = const()[name = string("op_10328_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10328 = slice_by_index(begin = var_10328_begin_0, end = var_10328_end_0, end_mask = var_10328_end_mask_0, x = q_39)[name = string("op_10328")]; int32 var_10330 = const()[name = string("op_10330"), val = int32(-1)]; bool var_10331_interleave_0 = const()[name = string("op_10331_interleave_0"), val = bool(false)]; tensor var_10331 = concat(axis = var_10330, interleave = var_10331_interleave_0, values = (var_10323, var_10328))[name = string("op_10331")]; tensor var_10332 = mul(x = var_10331, y = sin_1)[name = string("op_10332")]; tensor query_39 = add(x = var_10317, y = var_10332)[name = string("query_39")]; tensor var_10335 = mul(x = k_39, y = cos_1)[name = string("op_10335")]; tensor var_10340_begin_0 = const()[name = string("op_10340_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10340_end_0 = const()[name = string("op_10340_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_10340_end_mask_0 = const()[name = string("op_10340_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10340 = slice_by_index(begin = var_10340_begin_0, end = var_10340_end_0, end_mask = var_10340_end_mask_0, x = k_39)[name = string("op_10340")]; fp16 const_295_promoted = const()[name = string("const_295_promoted"), val = fp16(-0x1p+0)]; tensor var_10341 = mul(x = var_10340, y = const_295_promoted)[name = string("op_10341")]; tensor var_10346_begin_0 = const()[name = string("op_10346_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10346_end_0 = const()[name = string("op_10346_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_10346_end_mask_0 = const()[name = string("op_10346_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10346 = slice_by_index(begin = var_10346_begin_0, end = var_10346_end_0, end_mask = var_10346_end_mask_0, x = k_39)[name = string("op_10346")]; int32 var_10348 = const()[name = string("op_10348"), val = int32(-1)]; bool var_10349_interleave_0 = const()[name = string("op_10349_interleave_0"), val = bool(false)]; tensor var_10349 = concat(axis = var_10348, interleave = var_10349_interleave_0, values = (var_10341, var_10346))[name = string("op_10349")]; tensor var_10350 = mul(x = var_10349, y = sin_1)[name = string("op_10350")]; tensor key_39 = add(x = var_10335, y = var_10350)[name = string("key_39")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; int32 concat_344_axis_0 = const()[name = string("concat_344_axis_0"), val = int32(0)]; bool concat_344_interleave_0 = const()[name = string("concat_344_interleave_0"), val = bool(false)]; tensor concat_344 = concat(axis = concat_344_axis_0, interleave = concat_344_interleave_0, values = (expand_dims_228, expand_dims_229, current_pos, expand_dims_231))[name = string("concat_344")]; tensor concat_345_values1_0 = const()[name = string("concat_345_values1_0"), val = tensor([0])]; tensor concat_345_values3_0 = const()[name = string("concat_345_values3_0"), val = tensor([0])]; int32 concat_345_axis_0 = const()[name = string("concat_345_axis_0"), val = int32(0)]; bool concat_345_interleave_0 = const()[name = string("concat_345_interleave_0"), val = bool(false)]; tensor concat_345 = concat(axis = concat_345_axis_0, interleave = concat_345_interleave_0, values = (expand_dims_232, concat_345_values1_0, var_1746, concat_345_values3_0))[name = string("concat_345")]; tensor model_model_kv_cache_0_internal_tensor_assign_39_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_39_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_39_cast_fp16 = slice_update(begin = concat_344, begin_mask = model_model_kv_cache_0_internal_tensor_assign_39_begin_mask_0, end = concat_345, end_mask = model_model_kv_cache_0_internal_tensor_assign_39_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_39_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_39_stride_0, update = key_39, x = coreml_update_state_93)[name = string("model_model_kv_cache_0_internal_tensor_assign_39_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_39_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_206_write_state")]; tensor coreml_update_state_94 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_206")]; tensor expand_dims_234 = const()[name = string("expand_dims_234"), val = tensor([47])]; tensor expand_dims_235 = const()[name = string("expand_dims_235"), val = tensor([0])]; tensor expand_dims_237 = const()[name = string("expand_dims_237"), val = tensor([0])]; tensor expand_dims_238 = const()[name = string("expand_dims_238"), val = tensor([48])]; int32 concat_348_axis_0 = const()[name = string("concat_348_axis_0"), val = int32(0)]; bool concat_348_interleave_0 = const()[name = string("concat_348_interleave_0"), val = bool(false)]; tensor concat_348 = concat(axis = concat_348_axis_0, interleave = concat_348_interleave_0, values = (expand_dims_234, expand_dims_235, current_pos, expand_dims_237))[name = string("concat_348")]; tensor concat_349_values1_0 = const()[name = string("concat_349_values1_0"), val = tensor([0])]; tensor concat_349_values3_0 = const()[name = string("concat_349_values3_0"), val = tensor([0])]; int32 concat_349_axis_0 = const()[name = string("concat_349_axis_0"), val = int32(0)]; bool concat_349_interleave_0 = const()[name = string("concat_349_interleave_0"), val = bool(false)]; tensor concat_349 = concat(axis = concat_349_axis_0, interleave = concat_349_interleave_0, values = (expand_dims_238, concat_349_values1_0, var_1746, concat_349_values3_0))[name = string("concat_349")]; tensor model_model_kv_cache_0_internal_tensor_assign_40_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_40_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_40_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_191 = transpose(perm = var_10259, x = var_10254)[name = string("transpose_77")]; tensor model_model_kv_cache_0_internal_tensor_assign_40_cast_fp16 = slice_update(begin = concat_348, begin_mask = model_model_kv_cache_0_internal_tensor_assign_40_begin_mask_0, end = concat_349, end_mask = model_model_kv_cache_0_internal_tensor_assign_40_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_40_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_40_stride_0, update = value_191, x = coreml_update_state_94)[name = string("model_model_kv_cache_0_internal_tensor_assign_40_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_40_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_207_write_state")]; tensor coreml_update_state_95 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_207")]; tensor var_10421_begin_0 = const()[name = string("op_10421_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_10421_end_0 = const()[name = string("op_10421_end_0"), val = tensor([20, 8, 1536, 128])]; tensor var_10421_end_mask_0 = const()[name = string("op_10421_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10421_cast_fp16 = slice_by_index(begin = var_10421_begin_0, end = var_10421_end_0, end_mask = var_10421_end_mask_0, x = coreml_update_state_95)[name = string("op_10421_cast_fp16")]; tensor key_cache_39_axes_0 = const()[name = string("key_cache_39_axes_0"), val = tensor([0])]; tensor key_cache_39_cast_fp16 = squeeze(axes = key_cache_39_axes_0, x = var_10421_cast_fp16)[name = string("key_cache_39_cast_fp16")]; tensor var_10428_begin_0 = const()[name = string("op_10428_begin_0"), val = tensor([47, 0, 0, 0])]; tensor var_10428_end_0 = const()[name = string("op_10428_end_0"), val = tensor([48, 8, 1536, 128])]; tensor var_10428_end_mask_0 = const()[name = string("op_10428_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10428_cast_fp16 = slice_by_index(begin = var_10428_begin_0, end = var_10428_end_0, end_mask = var_10428_end_mask_0, x = coreml_update_state_95)[name = string("op_10428_cast_fp16")]; tensor value_cache_39_axes_0 = const()[name = string("value_cache_39_axes_0"), val = tensor([0])]; tensor value_cache_39_cast_fp16 = squeeze(axes = value_cache_39_axes_0, x = var_10428_cast_fp16)[name = string("value_cache_39_cast_fp16")]; tensor var_10452_axes_0 = const()[name = string("op_10452_axes_0"), val = tensor([1])]; tensor var_10452_cast_fp16 = expand_dims(axes = var_10452_axes_0, x = key_cache_39_cast_fp16)[name = string("op_10452_cast_fp16")]; tensor var_10457 = const()[name = string("op_10457"), val = tensor([1, 2, 1, 1])]; tensor value_195_cast_fp16 = tile(reps = var_10457, x = var_10452_cast_fp16)[name = string("value_195_cast_fp16")]; tensor var_10463 = const()[name = string("op_10463"), val = tensor([1, 16, 1536, 128])]; tensor key_states_79_cast_fp16 = reshape(shape = var_10463, x = value_195_cast_fp16)[name = string("key_states_79_cast_fp16")]; tensor var_10466_axes_0 = const()[name = string("op_10466_axes_0"), val = tensor([1])]; tensor var_10466_cast_fp16 = expand_dims(axes = var_10466_axes_0, x = value_cache_39_cast_fp16)[name = string("op_10466_cast_fp16")]; tensor var_10471 = const()[name = string("op_10471"), val = tensor([1, 2, 1, 1])]; tensor value_199_cast_fp16 = tile(reps = var_10471, x = var_10466_cast_fp16)[name = string("value_199_cast_fp16")]; bool var_10492_transpose_x_0 = const()[name = string("op_10492_transpose_x_0"), val = bool(false)]; bool var_10492_transpose_y_0 = const()[name = string("op_10492_transpose_y_0"), val = bool(true)]; tensor var_10492 = matmul(transpose_x = var_10492_transpose_x_0, transpose_y = var_10492_transpose_y_0, x = query_39, y = key_states_79_cast_fp16)[name = string("op_10492")]; fp16 var_10493_to_fp16 = const()[name = string("op_10493_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_77_cast_fp16 = mul(x = var_10492, y = var_10493_to_fp16)[name = string("attention_77_cast_fp16")]; tensor attention_79_cast_fp16 = add(x = attention_77_cast_fp16, y = causal_mask)[name = string("attention_79_cast_fp16")]; int32 var_10502 = const()[name = string("op_10502"), val = int32(-1)]; tensor var_10504_cast_fp16 = softmax(axis = var_10502, x = attention_79_cast_fp16)[name = string("op_10504_cast_fp16")]; tensor concat_354 = const()[name = string("concat_354"), val = tensor([16, 64, 1536])]; tensor reshape_57_cast_fp16 = reshape(shape = concat_354, x = var_10504_cast_fp16)[name = string("reshape_57_cast_fp16")]; tensor concat_355 = const()[name = string("concat_355"), val = tensor([16, 1536, 128])]; tensor reshape_58_cast_fp16 = reshape(shape = concat_355, x = value_199_cast_fp16)[name = string("reshape_58_cast_fp16")]; bool matmul_19_transpose_x_0 = const()[name = string("matmul_19_transpose_x_0"), val = bool(false)]; bool matmul_19_transpose_y_0 = const()[name = string("matmul_19_transpose_y_0"), val = bool(false)]; tensor matmul_19_cast_fp16 = matmul(transpose_x = matmul_19_transpose_x_0, transpose_y = matmul_19_transpose_y_0, x = reshape_57_cast_fp16, y = reshape_58_cast_fp16)[name = string("matmul_19_cast_fp16")]; tensor concat_359 = const()[name = string("concat_359"), val = tensor([1, 16, 64, 128])]; tensor reshape_59_cast_fp16 = reshape(shape = concat_359, x = matmul_19_cast_fp16)[name = string("reshape_59_cast_fp16")]; tensor var_10516_perm_0 = const()[name = string("op_10516_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_10522 = const()[name = string("op_10522"), val = tensor([1, 64, 2048])]; tensor var_10516_cast_fp16 = transpose(perm = var_10516_perm_0, x = reshape_59_cast_fp16)[name = string("transpose_76")]; tensor output_117_cast_fp16 = reshape(shape = var_10522, x = var_10516_cast_fp16)[name = string("output_117_cast_fp16")]; tensor var_10527 = const()[name = string("op_10527"), val = tensor([0, 2, 1])]; string var_10543_pad_type_0 = const()[name = string("op_10543_pad_type_0"), val = string("valid")]; int32 var_10543_groups_0 = const()[name = string("op_10543_groups_0"), val = int32(1)]; tensor var_10543_strides_0 = const()[name = string("op_10543_strides_0"), val = tensor([1])]; tensor var_10543_pad_0 = const()[name = string("op_10543_pad_0"), val = tensor([0, 0])]; tensor var_10543_dilations_0 = const()[name = string("op_10543_dilations_0"), val = tensor([1])]; tensor squeeze_19_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323195392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324768320))))[name = string("squeeze_19_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10528_cast_fp16 = transpose(perm = var_10527, x = output_117_cast_fp16)[name = string("transpose_75")]; tensor var_10543_cast_fp16 = conv(dilations = var_10543_dilations_0, groups = var_10543_groups_0, pad = var_10543_pad_0, pad_type = var_10543_pad_type_0, strides = var_10543_strides_0, weight = squeeze_19_cast_fp16_to_fp32_to_fp16_palettized, x = var_10528_cast_fp16)[name = string("op_10543_cast_fp16")]; tensor var_10547 = const()[name = string("op_10547"), val = tensor([0, 2, 1])]; tensor attn_output_39_cast_fp16 = transpose(perm = var_10547, x = var_10543_cast_fp16)[name = string("transpose_74")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_191_cast_fp16, y = attn_output_39_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; int32 var_10562 = const()[name = string("op_10562"), val = int32(-1)]; fp16 const_297_promoted_to_fp16 = const()[name = string("const_297_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10564_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_297_promoted_to_fp16)[name = string("op_10564_cast_fp16")]; bool input_353_interleave_0 = const()[name = string("input_353_interleave_0"), val = bool(false)]; tensor input_353_cast_fp16 = concat(axis = var_10562, interleave = input_353_interleave_0, values = (hidden_states_199_cast_fp16, var_10564_cast_fp16))[name = string("input_353_cast_fp16")]; tensor normed_317_axes_0 = const()[name = string("normed_317_axes_0"), val = tensor([-1])]; fp16 var_10559_to_fp16 = const()[name = string("op_10559_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_317_cast_fp16 = layer_norm(axes = normed_317_axes_0, epsilon = var_10559_to_fp16, x = input_353_cast_fp16)[name = string("normed_317_cast_fp16")]; tensor normed_319_begin_0 = const()[name = string("normed_319_begin_0"), val = tensor([0, 0, 0])]; tensor normed_319_end_0 = const()[name = string("normed_319_end_0"), val = tensor([1, 64, 1024])]; tensor normed_319_end_mask_0 = const()[name = string("normed_319_end_mask_0"), val = tensor([true, true, false])]; tensor normed_319_cast_fp16 = slice_by_index(begin = normed_319_begin_0, end = normed_319_end_0, end_mask = normed_319_end_mask_0, x = normed_317_cast_fp16)[name = string("normed_319_cast_fp16")]; tensor const_299_promoted_to_fp16 = const()[name = string("const_299_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324784768)))]; tensor x_77_cast_fp16 = mul(x = normed_319_cast_fp16, y = const_299_promoted_to_fp16)[name = string("x_77_cast_fp16")]; tensor var_10584 = const()[name = string("op_10584"), val = tensor([0, 2, 1])]; tensor input_355_axes_0 = const()[name = string("input_355_axes_0"), val = tensor([2])]; tensor var_10585 = transpose(perm = var_10584, x = x_77_cast_fp16)[name = string("transpose_73")]; tensor input_355 = expand_dims(axes = input_355_axes_0, x = var_10585)[name = string("input_355")]; string input_357_pad_type_0 = const()[name = string("input_357_pad_type_0"), val = string("valid")]; tensor input_357_strides_0 = const()[name = string("input_357_strides_0"), val = tensor([1, 1])]; tensor input_357_pad_0 = const()[name = string("input_357_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_357_dilations_0 = const()[name = string("input_357_dilations_0"), val = tensor([1, 1])]; int32 input_357_groups_0 = const()[name = string("input_357_groups_0"), val = int32(1)]; tensor input_357 = conv(dilations = input_357_dilations_0, groups = input_357_groups_0, pad = input_357_pad_0, pad_type = input_357_pad_type_0, strides = input_357_strides_0, weight = model_model_layers_19_mlp_gate_proj_weight_palettized, x = input_355)[name = string("input_357")]; string b_39_pad_type_0 = const()[name = string("b_39_pad_type_0"), val = string("valid")]; tensor b_39_strides_0 = const()[name = string("b_39_strides_0"), val = tensor([1, 1])]; tensor b_39_pad_0 = const()[name = string("b_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_39_dilations_0 = const()[name = string("b_39_dilations_0"), val = tensor([1, 1])]; int32 b_39_groups_0 = const()[name = string("b_39_groups_0"), val = int32(1)]; tensor b_39 = conv(dilations = b_39_dilations_0, groups = b_39_groups_0, pad = b_39_pad_0, pad_type = b_39_pad_type_0, strides = b_39_strides_0, weight = model_model_layers_19_mlp_up_proj_weight_palettized, x = input_355)[name = string("b_39")]; tensor c_39 = silu(x = input_357)[name = string("c_39")]; tensor input_359 = mul(x = c_39, y = b_39)[name = string("input_359")]; string e_39_pad_type_0 = const()[name = string("e_39_pad_type_0"), val = string("valid")]; tensor e_39_strides_0 = const()[name = string("e_39_strides_0"), val = tensor([1, 1])]; tensor e_39_pad_0 = const()[name = string("e_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_39_dilations_0 = const()[name = string("e_39_dilations_0"), val = tensor([1, 1])]; int32 e_39_groups_0 = const()[name = string("e_39_groups_0"), val = int32(1)]; tensor e_39 = conv(dilations = e_39_dilations_0, groups = e_39_groups_0, pad = e_39_pad_0, pad_type = e_39_pad_type_0, strides = e_39_strides_0, weight = model_model_layers_19_mlp_down_proj_weight_palettized, x = input_359)[name = string("e_39")]; tensor var_10607_axes_0 = const()[name = string("op_10607_axes_0"), val = tensor([2])]; tensor var_10607 = squeeze(axes = var_10607_axes_0, x = e_39)[name = string("op_10607")]; tensor var_10608 = const()[name = string("op_10608"), val = tensor([0, 2, 1])]; tensor var_10609 = transpose(perm = var_10608, x = var_10607)[name = string("transpose_72")]; tensor hidden_states_201_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = var_10609)[name = string("hidden_states_201_cast_fp16")]; int32 var_10623 = const()[name = string("op_10623"), val = int32(-1)]; fp16 const_300_promoted_to_fp16 = const()[name = string("const_300_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_10625_cast_fp16 = mul(x = hidden_states_201_cast_fp16, y = const_300_promoted_to_fp16)[name = string("op_10625_cast_fp16")]; bool input_361_interleave_0 = const()[name = string("input_361_interleave_0"), val = bool(false)]; tensor input_361_cast_fp16 = concat(axis = var_10623, interleave = input_361_interleave_0, values = (hidden_states_201_cast_fp16, var_10625_cast_fp16))[name = string("input_361_cast_fp16")]; tensor normed_321_axes_0 = const()[name = string("normed_321_axes_0"), val = tensor([-1])]; fp16 var_10620_to_fp16 = const()[name = string("op_10620_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_321_cast_fp16 = layer_norm(axes = normed_321_axes_0, epsilon = var_10620_to_fp16, x = input_361_cast_fp16)[name = string("normed_321_cast_fp16")]; tensor normed_323_begin_0 = const()[name = string("normed_323_begin_0"), val = tensor([0, 0, 0])]; tensor normed_323_end_0 = const()[name = string("normed_323_end_0"), val = tensor([1, 64, 1024])]; tensor normed_323_end_mask_0 = const()[name = string("normed_323_end_mask_0"), val = tensor([true, true, false])]; tensor normed_323_cast_fp16 = slice_by_index(begin = normed_323_begin_0, end = normed_323_end_0, end_mask = normed_323_end_mask_0, x = normed_321_cast_fp16)[name = string("normed_323_cast_fp16")]; tensor const_302_promoted_to_fp16 = const()[name = string("const_302_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324786880)))]; tensor hidden_states_203_cast_fp16 = mul(x = normed_323_cast_fp16, y = const_302_promoted_to_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor var_10637 = const()[name = string("op_10637"), val = tensor([0, 2, 1])]; tensor var_10640_axes_0 = const()[name = string("op_10640_axes_0"), val = tensor([2])]; tensor var_10638_cast_fp16 = transpose(perm = var_10637, x = hidden_states_203_cast_fp16)[name = string("transpose_71")]; tensor var_10640_cast_fp16 = expand_dims(axes = var_10640_axes_0, x = var_10638_cast_fp16)[name = string("op_10640_cast_fp16")]; string var_10656_pad_type_0 = const()[name = string("op_10656_pad_type_0"), val = string("valid")]; tensor var_10656_strides_0 = const()[name = string("op_10656_strides_0"), val = tensor([1, 1])]; tensor var_10656_pad_0 = const()[name = string("op_10656_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10656_dilations_0 = const()[name = string("op_10656_dilations_0"), val = tensor([1, 1])]; int32 var_10656_groups_0 = const()[name = string("op_10656_groups_0"), val = int32(1)]; tensor var_10656 = conv(dilations = var_10656_dilations_0, groups = var_10656_groups_0, pad = var_10656_pad_0, pad_type = var_10656_pad_type_0, strides = var_10656_strides_0, weight = model_model_layers_20_self_attn_q_proj_weight_palettized, x = var_10640_cast_fp16)[name = string("op_10656")]; tensor var_10661 = const()[name = string("op_10661"), val = tensor([1, 16, 128, 64])]; tensor var_10662 = reshape(shape = var_10661, x = var_10656)[name = string("op_10662")]; tensor var_10667 = const()[name = string("op_10667"), val = tensor([0, 1, 3, 2])]; string var_10679_pad_type_0 = const()[name = string("op_10679_pad_type_0"), val = string("valid")]; tensor var_10679_strides_0 = const()[name = string("op_10679_strides_0"), val = tensor([1, 1])]; tensor var_10679_pad_0 = const()[name = string("op_10679_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10679_dilations_0 = const()[name = string("op_10679_dilations_0"), val = tensor([1, 1])]; int32 var_10679_groups_0 = const()[name = string("op_10679_groups_0"), val = int32(1)]; tensor var_10679 = conv(dilations = var_10679_dilations_0, groups = var_10679_groups_0, pad = var_10679_pad_0, pad_type = var_10679_pad_type_0, strides = var_10679_strides_0, weight = model_model_layers_20_self_attn_k_proj_weight_palettized, x = var_10640_cast_fp16)[name = string("op_10679")]; tensor var_10684 = const()[name = string("op_10684"), val = tensor([1, 8, 128, 64])]; tensor var_10685 = reshape(shape = var_10684, x = var_10679)[name = string("op_10685")]; tensor var_10690 = const()[name = string("op_10690"), val = tensor([0, 1, 3, 2])]; string var_10702_pad_type_0 = const()[name = string("op_10702_pad_type_0"), val = string("valid")]; tensor var_10702_strides_0 = const()[name = string("op_10702_strides_0"), val = tensor([1, 1])]; tensor var_10702_pad_0 = const()[name = string("op_10702_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_10702_dilations_0 = const()[name = string("op_10702_dilations_0"), val = tensor([1, 1])]; int32 var_10702_groups_0 = const()[name = string("op_10702_groups_0"), val = int32(1)]; tensor var_10702 = conv(dilations = var_10702_dilations_0, groups = var_10702_groups_0, pad = var_10702_pad_0, pad_type = var_10702_pad_type_0, strides = var_10702_strides_0, weight = model_model_layers_20_self_attn_v_proj_weight_palettized, x = var_10640_cast_fp16)[name = string("op_10702")]; tensor var_10707 = const()[name = string("op_10707"), val = tensor([1, 8, 128, 64])]; tensor var_10708 = reshape(shape = var_10707, x = var_10702)[name = string("op_10708")]; tensor var_10713 = const()[name = string("op_10713"), val = tensor([0, 1, 3, 2])]; int32 var_10726 = const()[name = string("op_10726"), val = int32(-1)]; fp16 const_303_promoted = const()[name = string("const_303_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_205 = transpose(perm = var_10667, x = var_10662)[name = string("transpose_70")]; tensor var_10728 = mul(x = hidden_states_205, y = const_303_promoted)[name = string("op_10728")]; bool input_365_interleave_0 = const()[name = string("input_365_interleave_0"), val = bool(false)]; tensor input_365 = concat(axis = var_10726, interleave = input_365_interleave_0, values = (hidden_states_205, var_10728))[name = string("input_365")]; tensor normed_325_axes_0 = const()[name = string("normed_325_axes_0"), val = tensor([-1])]; fp16 var_10723_to_fp16 = const()[name = string("op_10723_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_325_cast_fp16 = layer_norm(axes = normed_325_axes_0, epsilon = var_10723_to_fp16, x = input_365)[name = string("normed_325_cast_fp16")]; tensor normed_327_begin_0 = const()[name = string("normed_327_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_327_end_0 = const()[name = string("normed_327_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_327_end_mask_0 = const()[name = string("normed_327_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_327 = slice_by_index(begin = normed_327_begin_0, end = normed_327_end_0, end_mask = normed_327_end_mask_0, x = normed_325_cast_fp16)[name = string("normed_327")]; tensor const_305 = const()[name = string("const_305"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324788992)))]; tensor q_41 = mul(x = normed_327, y = const_305)[name = string("q_41")]; int32 var_10748 = const()[name = string("op_10748"), val = int32(-1)]; fp16 const_306_promoted = const()[name = string("const_306_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_207 = transpose(perm = var_10690, x = var_10685)[name = string("transpose_69")]; tensor var_10750 = mul(x = hidden_states_207, y = const_306_promoted)[name = string("op_10750")]; bool input_367_interleave_0 = const()[name = string("input_367_interleave_0"), val = bool(false)]; tensor input_367 = concat(axis = var_10748, interleave = input_367_interleave_0, values = (hidden_states_207, var_10750))[name = string("input_367")]; tensor normed_329_axes_0 = const()[name = string("normed_329_axes_0"), val = tensor([-1])]; fp16 var_10745_to_fp16 = const()[name = string("op_10745_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_329_cast_fp16 = layer_norm(axes = normed_329_axes_0, epsilon = var_10745_to_fp16, x = input_367)[name = string("normed_329_cast_fp16")]; tensor normed_331_begin_0 = const()[name = string("normed_331_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_331_end_0 = const()[name = string("normed_331_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_331_end_mask_0 = const()[name = string("normed_331_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_331 = slice_by_index(begin = normed_331_begin_0, end = normed_331_end_0, end_mask = normed_331_end_mask_0, x = normed_329_cast_fp16)[name = string("normed_331")]; tensor const_308 = const()[name = string("const_308"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324789312)))]; tensor k_41 = mul(x = normed_331, y = const_308)[name = string("k_41")]; tensor var_10771 = mul(x = q_41, y = cos_1)[name = string("op_10771")]; tensor var_10776_begin_0 = const()[name = string("op_10776_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10776_end_0 = const()[name = string("op_10776_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_10776_end_mask_0 = const()[name = string("op_10776_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10776 = slice_by_index(begin = var_10776_begin_0, end = var_10776_end_0, end_mask = var_10776_end_mask_0, x = q_41)[name = string("op_10776")]; fp16 const_309_promoted = const()[name = string("const_309_promoted"), val = fp16(-0x1p+0)]; tensor var_10777 = mul(x = var_10776, y = const_309_promoted)[name = string("op_10777")]; tensor var_10782_begin_0 = const()[name = string("op_10782_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10782_end_0 = const()[name = string("op_10782_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_10782_end_mask_0 = const()[name = string("op_10782_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10782 = slice_by_index(begin = var_10782_begin_0, end = var_10782_end_0, end_mask = var_10782_end_mask_0, x = q_41)[name = string("op_10782")]; int32 var_10784 = const()[name = string("op_10784"), val = int32(-1)]; bool var_10785_interleave_0 = const()[name = string("op_10785_interleave_0"), val = bool(false)]; tensor var_10785 = concat(axis = var_10784, interleave = var_10785_interleave_0, values = (var_10777, var_10782))[name = string("op_10785")]; tensor var_10786 = mul(x = var_10785, y = sin_1)[name = string("op_10786")]; tensor query_41 = add(x = var_10771, y = var_10786)[name = string("query_41")]; tensor var_10789 = mul(x = k_41, y = cos_1)[name = string("op_10789")]; tensor var_10794_begin_0 = const()[name = string("op_10794_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_10794_end_0 = const()[name = string("op_10794_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_10794_end_mask_0 = const()[name = string("op_10794_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_10794 = slice_by_index(begin = var_10794_begin_0, end = var_10794_end_0, end_mask = var_10794_end_mask_0, x = k_41)[name = string("op_10794")]; fp16 const_310_promoted = const()[name = string("const_310_promoted"), val = fp16(-0x1p+0)]; tensor var_10795 = mul(x = var_10794, y = const_310_promoted)[name = string("op_10795")]; tensor var_10800_begin_0 = const()[name = string("op_10800_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_10800_end_0 = const()[name = string("op_10800_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_10800_end_mask_0 = const()[name = string("op_10800_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_10800 = slice_by_index(begin = var_10800_begin_0, end = var_10800_end_0, end_mask = var_10800_end_mask_0, x = k_41)[name = string("op_10800")]; int32 var_10802 = const()[name = string("op_10802"), val = int32(-1)]; bool var_10803_interleave_0 = const()[name = string("op_10803_interleave_0"), val = bool(false)]; tensor var_10803 = concat(axis = var_10802, interleave = var_10803_interleave_0, values = (var_10795, var_10800))[name = string("op_10803")]; tensor var_10804 = mul(x = var_10803, y = sin_1)[name = string("op_10804")]; tensor key_41 = add(x = var_10789, y = var_10804)[name = string("key_41")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; int32 concat_362_axis_0 = const()[name = string("concat_362_axis_0"), val = int32(0)]; bool concat_362_interleave_0 = const()[name = string("concat_362_interleave_0"), val = bool(false)]; tensor concat_362 = concat(axis = concat_362_axis_0, interleave = concat_362_interleave_0, values = (expand_dims_240, expand_dims_241, current_pos, expand_dims_243))[name = string("concat_362")]; tensor concat_363_values1_0 = const()[name = string("concat_363_values1_0"), val = tensor([0])]; tensor concat_363_values3_0 = const()[name = string("concat_363_values3_0"), val = tensor([0])]; int32 concat_363_axis_0 = const()[name = string("concat_363_axis_0"), val = int32(0)]; bool concat_363_interleave_0 = const()[name = string("concat_363_interleave_0"), val = bool(false)]; tensor concat_363 = concat(axis = concat_363_axis_0, interleave = concat_363_interleave_0, values = (expand_dims_244, concat_363_values1_0, var_1746, concat_363_values3_0))[name = string("concat_363")]; tensor model_model_kv_cache_0_internal_tensor_assign_41_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_41_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_41_cast_fp16 = slice_update(begin = concat_362, begin_mask = model_model_kv_cache_0_internal_tensor_assign_41_begin_mask_0, end = concat_363, end_mask = model_model_kv_cache_0_internal_tensor_assign_41_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_41_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_41_stride_0, update = key_41, x = coreml_update_state_95)[name = string("model_model_kv_cache_0_internal_tensor_assign_41_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_41_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_208_write_state")]; tensor coreml_update_state_96 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_208")]; tensor expand_dims_246 = const()[name = string("expand_dims_246"), val = tensor([48])]; tensor expand_dims_247 = const()[name = string("expand_dims_247"), val = tensor([0])]; tensor expand_dims_249 = const()[name = string("expand_dims_249"), val = tensor([0])]; tensor expand_dims_250 = const()[name = string("expand_dims_250"), val = tensor([49])]; int32 concat_366_axis_0 = const()[name = string("concat_366_axis_0"), val = int32(0)]; bool concat_366_interleave_0 = const()[name = string("concat_366_interleave_0"), val = bool(false)]; tensor concat_366 = concat(axis = concat_366_axis_0, interleave = concat_366_interleave_0, values = (expand_dims_246, expand_dims_247, current_pos, expand_dims_249))[name = string("concat_366")]; tensor concat_367_values1_0 = const()[name = string("concat_367_values1_0"), val = tensor([0])]; tensor concat_367_values3_0 = const()[name = string("concat_367_values3_0"), val = tensor([0])]; int32 concat_367_axis_0 = const()[name = string("concat_367_axis_0"), val = int32(0)]; bool concat_367_interleave_0 = const()[name = string("concat_367_interleave_0"), val = bool(false)]; tensor concat_367 = concat(axis = concat_367_axis_0, interleave = concat_367_interleave_0, values = (expand_dims_250, concat_367_values1_0, var_1746, concat_367_values3_0))[name = string("concat_367")]; tensor model_model_kv_cache_0_internal_tensor_assign_42_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_42_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_42_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_201 = transpose(perm = var_10713, x = var_10708)[name = string("transpose_68")]; tensor model_model_kv_cache_0_internal_tensor_assign_42_cast_fp16 = slice_update(begin = concat_366, begin_mask = model_model_kv_cache_0_internal_tensor_assign_42_begin_mask_0, end = concat_367, end_mask = model_model_kv_cache_0_internal_tensor_assign_42_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_42_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_42_stride_0, update = value_201, x = coreml_update_state_96)[name = string("model_model_kv_cache_0_internal_tensor_assign_42_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_42_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_209_write_state")]; tensor coreml_update_state_97 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_209")]; tensor var_10875_begin_0 = const()[name = string("op_10875_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_10875_end_0 = const()[name = string("op_10875_end_0"), val = tensor([21, 8, 1536, 128])]; tensor var_10875_end_mask_0 = const()[name = string("op_10875_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10875_cast_fp16 = slice_by_index(begin = var_10875_begin_0, end = var_10875_end_0, end_mask = var_10875_end_mask_0, x = coreml_update_state_97)[name = string("op_10875_cast_fp16")]; tensor key_cache_41_axes_0 = const()[name = string("key_cache_41_axes_0"), val = tensor([0])]; tensor key_cache_41_cast_fp16 = squeeze(axes = key_cache_41_axes_0, x = var_10875_cast_fp16)[name = string("key_cache_41_cast_fp16")]; tensor var_10882_begin_0 = const()[name = string("op_10882_begin_0"), val = tensor([48, 0, 0, 0])]; tensor var_10882_end_0 = const()[name = string("op_10882_end_0"), val = tensor([49, 8, 1536, 128])]; tensor var_10882_end_mask_0 = const()[name = string("op_10882_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_10882_cast_fp16 = slice_by_index(begin = var_10882_begin_0, end = var_10882_end_0, end_mask = var_10882_end_mask_0, x = coreml_update_state_97)[name = string("op_10882_cast_fp16")]; tensor value_cache_41_axes_0 = const()[name = string("value_cache_41_axes_0"), val = tensor([0])]; tensor value_cache_41_cast_fp16 = squeeze(axes = value_cache_41_axes_0, x = var_10882_cast_fp16)[name = string("value_cache_41_cast_fp16")]; tensor var_10906_axes_0 = const()[name = string("op_10906_axes_0"), val = tensor([1])]; tensor var_10906_cast_fp16 = expand_dims(axes = var_10906_axes_0, x = key_cache_41_cast_fp16)[name = string("op_10906_cast_fp16")]; tensor var_10911 = const()[name = string("op_10911"), val = tensor([1, 2, 1, 1])]; tensor value_205_cast_fp16 = tile(reps = var_10911, x = var_10906_cast_fp16)[name = string("value_205_cast_fp16")]; tensor var_10917 = const()[name = string("op_10917"), val = tensor([1, 16, 1536, 128])]; tensor key_states_83_cast_fp16 = reshape(shape = var_10917, x = value_205_cast_fp16)[name = string("key_states_83_cast_fp16")]; tensor var_10920_axes_0 = const()[name = string("op_10920_axes_0"), val = tensor([1])]; tensor var_10920_cast_fp16 = expand_dims(axes = var_10920_axes_0, x = value_cache_41_cast_fp16)[name = string("op_10920_cast_fp16")]; tensor var_10925 = const()[name = string("op_10925"), val = tensor([1, 2, 1, 1])]; tensor value_209_cast_fp16 = tile(reps = var_10925, x = var_10920_cast_fp16)[name = string("value_209_cast_fp16")]; bool var_10946_transpose_x_0 = const()[name = string("op_10946_transpose_x_0"), val = bool(false)]; bool var_10946_transpose_y_0 = const()[name = string("op_10946_transpose_y_0"), val = bool(true)]; tensor var_10946 = matmul(transpose_x = var_10946_transpose_x_0, transpose_y = var_10946_transpose_y_0, x = query_41, y = key_states_83_cast_fp16)[name = string("op_10946")]; fp16 var_10947_to_fp16 = const()[name = string("op_10947_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_81_cast_fp16 = mul(x = var_10946, y = var_10947_to_fp16)[name = string("attention_81_cast_fp16")]; tensor attention_83_cast_fp16 = add(x = attention_81_cast_fp16, y = causal_mask)[name = string("attention_83_cast_fp16")]; int32 var_10956 = const()[name = string("op_10956"), val = int32(-1)]; tensor var_10958_cast_fp16 = softmax(axis = var_10956, x = attention_83_cast_fp16)[name = string("op_10958_cast_fp16")]; tensor concat_372 = const()[name = string("concat_372"), val = tensor([16, 64, 1536])]; tensor reshape_60_cast_fp16 = reshape(shape = concat_372, x = var_10958_cast_fp16)[name = string("reshape_60_cast_fp16")]; tensor concat_373 = const()[name = string("concat_373"), val = tensor([16, 1536, 128])]; tensor reshape_61_cast_fp16 = reshape(shape = concat_373, x = value_209_cast_fp16)[name = string("reshape_61_cast_fp16")]; bool matmul_20_transpose_x_0 = const()[name = string("matmul_20_transpose_x_0"), val = bool(false)]; bool matmul_20_transpose_y_0 = const()[name = string("matmul_20_transpose_y_0"), val = bool(false)]; tensor matmul_20_cast_fp16 = matmul(transpose_x = matmul_20_transpose_x_0, transpose_y = matmul_20_transpose_y_0, x = reshape_60_cast_fp16, y = reshape_61_cast_fp16)[name = string("matmul_20_cast_fp16")]; tensor concat_377 = const()[name = string("concat_377"), val = tensor([1, 16, 64, 128])]; tensor reshape_62_cast_fp16 = reshape(shape = concat_377, x = matmul_20_cast_fp16)[name = string("reshape_62_cast_fp16")]; tensor var_10970_perm_0 = const()[name = string("op_10970_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_10976 = const()[name = string("op_10976"), val = tensor([1, 64, 2048])]; tensor var_10970_cast_fp16 = transpose(perm = var_10970_perm_0, x = reshape_62_cast_fp16)[name = string("transpose_67")]; tensor output_123_cast_fp16 = reshape(shape = var_10976, x = var_10970_cast_fp16)[name = string("output_123_cast_fp16")]; tensor var_10981 = const()[name = string("op_10981"), val = tensor([0, 2, 1])]; string var_10997_pad_type_0 = const()[name = string("op_10997_pad_type_0"), val = string("valid")]; int32 var_10997_groups_0 = const()[name = string("op_10997_groups_0"), val = int32(1)]; tensor var_10997_strides_0 = const()[name = string("op_10997_strides_0"), val = tensor([1])]; tensor var_10997_pad_0 = const()[name = string("op_10997_pad_0"), val = tensor([0, 0])]; tensor var_10997_dilations_0 = const()[name = string("op_10997_dilations_0"), val = tensor([1])]; tensor squeeze_20_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324789632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326362560))))[name = string("squeeze_20_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_10982_cast_fp16 = transpose(perm = var_10981, x = output_123_cast_fp16)[name = string("transpose_66")]; tensor var_10997_cast_fp16 = conv(dilations = var_10997_dilations_0, groups = var_10997_groups_0, pad = var_10997_pad_0, pad_type = var_10997_pad_type_0, strides = var_10997_strides_0, weight = squeeze_20_cast_fp16_to_fp32_to_fp16_palettized, x = var_10982_cast_fp16)[name = string("op_10997_cast_fp16")]; tensor var_11001 = const()[name = string("op_11001"), val = tensor([0, 2, 1])]; tensor attn_output_41_cast_fp16 = transpose(perm = var_11001, x = var_10997_cast_fp16)[name = string("transpose_65")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_201_cast_fp16, y = attn_output_41_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; int32 var_11016 = const()[name = string("op_11016"), val = int32(-1)]; fp16 const_312_promoted_to_fp16 = const()[name = string("const_312_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11018_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_312_promoted_to_fp16)[name = string("op_11018_cast_fp16")]; bool input_371_interleave_0 = const()[name = string("input_371_interleave_0"), val = bool(false)]; tensor input_371_cast_fp16 = concat(axis = var_11016, interleave = input_371_interleave_0, values = (hidden_states_209_cast_fp16, var_11018_cast_fp16))[name = string("input_371_cast_fp16")]; tensor normed_333_axes_0 = const()[name = string("normed_333_axes_0"), val = tensor([-1])]; fp16 var_11013_to_fp16 = const()[name = string("op_11013_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_333_cast_fp16 = layer_norm(axes = normed_333_axes_0, epsilon = var_11013_to_fp16, x = input_371_cast_fp16)[name = string("normed_333_cast_fp16")]; tensor normed_335_begin_0 = const()[name = string("normed_335_begin_0"), val = tensor([0, 0, 0])]; tensor normed_335_end_0 = const()[name = string("normed_335_end_0"), val = tensor([1, 64, 1024])]; tensor normed_335_end_mask_0 = const()[name = string("normed_335_end_mask_0"), val = tensor([true, true, false])]; tensor normed_335_cast_fp16 = slice_by_index(begin = normed_335_begin_0, end = normed_335_end_0, end_mask = normed_335_end_mask_0, x = normed_333_cast_fp16)[name = string("normed_335_cast_fp16")]; tensor const_314_promoted_to_fp16 = const()[name = string("const_314_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326379008)))]; tensor x_81_cast_fp16 = mul(x = normed_335_cast_fp16, y = const_314_promoted_to_fp16)[name = string("x_81_cast_fp16")]; tensor var_11038 = const()[name = string("op_11038"), val = tensor([0, 2, 1])]; tensor input_373_axes_0 = const()[name = string("input_373_axes_0"), val = tensor([2])]; tensor var_11039 = transpose(perm = var_11038, x = x_81_cast_fp16)[name = string("transpose_64")]; tensor input_373 = expand_dims(axes = input_373_axes_0, x = var_11039)[name = string("input_373")]; string input_375_pad_type_0 = const()[name = string("input_375_pad_type_0"), val = string("valid")]; tensor input_375_strides_0 = const()[name = string("input_375_strides_0"), val = tensor([1, 1])]; tensor input_375_pad_0 = const()[name = string("input_375_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_375_dilations_0 = const()[name = string("input_375_dilations_0"), val = tensor([1, 1])]; int32 input_375_groups_0 = const()[name = string("input_375_groups_0"), val = int32(1)]; tensor input_375 = conv(dilations = input_375_dilations_0, groups = input_375_groups_0, pad = input_375_pad_0, pad_type = input_375_pad_type_0, strides = input_375_strides_0, weight = model_model_layers_20_mlp_gate_proj_weight_palettized, x = input_373)[name = string("input_375")]; string b_41_pad_type_0 = const()[name = string("b_41_pad_type_0"), val = string("valid")]; tensor b_41_strides_0 = const()[name = string("b_41_strides_0"), val = tensor([1, 1])]; tensor b_41_pad_0 = const()[name = string("b_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_41_dilations_0 = const()[name = string("b_41_dilations_0"), val = tensor([1, 1])]; int32 b_41_groups_0 = const()[name = string("b_41_groups_0"), val = int32(1)]; tensor b_41 = conv(dilations = b_41_dilations_0, groups = b_41_groups_0, pad = b_41_pad_0, pad_type = b_41_pad_type_0, strides = b_41_strides_0, weight = model_model_layers_20_mlp_up_proj_weight_palettized, x = input_373)[name = string("b_41")]; tensor c_41 = silu(x = input_375)[name = string("c_41")]; tensor input_377 = mul(x = c_41, y = b_41)[name = string("input_377")]; string e_41_pad_type_0 = const()[name = string("e_41_pad_type_0"), val = string("valid")]; tensor e_41_strides_0 = const()[name = string("e_41_strides_0"), val = tensor([1, 1])]; tensor e_41_pad_0 = const()[name = string("e_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_41_dilations_0 = const()[name = string("e_41_dilations_0"), val = tensor([1, 1])]; int32 e_41_groups_0 = const()[name = string("e_41_groups_0"), val = int32(1)]; tensor e_41 = conv(dilations = e_41_dilations_0, groups = e_41_groups_0, pad = e_41_pad_0, pad_type = e_41_pad_type_0, strides = e_41_strides_0, weight = model_model_layers_20_mlp_down_proj_weight_palettized, x = input_377)[name = string("e_41")]; tensor var_11061_axes_0 = const()[name = string("op_11061_axes_0"), val = tensor([2])]; tensor var_11061 = squeeze(axes = var_11061_axes_0, x = e_41)[name = string("op_11061")]; tensor var_11062 = const()[name = string("op_11062"), val = tensor([0, 2, 1])]; tensor var_11063 = transpose(perm = var_11062, x = var_11061)[name = string("transpose_63")]; tensor hidden_states_211_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = var_11063)[name = string("hidden_states_211_cast_fp16")]; int32 var_11077 = const()[name = string("op_11077"), val = int32(-1)]; fp16 const_315_promoted_to_fp16 = const()[name = string("const_315_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11079_cast_fp16 = mul(x = hidden_states_211_cast_fp16, y = const_315_promoted_to_fp16)[name = string("op_11079_cast_fp16")]; bool input_379_interleave_0 = const()[name = string("input_379_interleave_0"), val = bool(false)]; tensor input_379_cast_fp16 = concat(axis = var_11077, interleave = input_379_interleave_0, values = (hidden_states_211_cast_fp16, var_11079_cast_fp16))[name = string("input_379_cast_fp16")]; tensor normed_337_axes_0 = const()[name = string("normed_337_axes_0"), val = tensor([-1])]; fp16 var_11074_to_fp16 = const()[name = string("op_11074_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_337_cast_fp16 = layer_norm(axes = normed_337_axes_0, epsilon = var_11074_to_fp16, x = input_379_cast_fp16)[name = string("normed_337_cast_fp16")]; tensor normed_339_begin_0 = const()[name = string("normed_339_begin_0"), val = tensor([0, 0, 0])]; tensor normed_339_end_0 = const()[name = string("normed_339_end_0"), val = tensor([1, 64, 1024])]; tensor normed_339_end_mask_0 = const()[name = string("normed_339_end_mask_0"), val = tensor([true, true, false])]; tensor normed_339_cast_fp16 = slice_by_index(begin = normed_339_begin_0, end = normed_339_end_0, end_mask = normed_339_end_mask_0, x = normed_337_cast_fp16)[name = string("normed_339_cast_fp16")]; tensor const_317_promoted_to_fp16 = const()[name = string("const_317_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326381120)))]; tensor hidden_states_213_cast_fp16 = mul(x = normed_339_cast_fp16, y = const_317_promoted_to_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor var_11091 = const()[name = string("op_11091"), val = tensor([0, 2, 1])]; tensor var_11094_axes_0 = const()[name = string("op_11094_axes_0"), val = tensor([2])]; tensor var_11092_cast_fp16 = transpose(perm = var_11091, x = hidden_states_213_cast_fp16)[name = string("transpose_62")]; tensor var_11094_cast_fp16 = expand_dims(axes = var_11094_axes_0, x = var_11092_cast_fp16)[name = string("op_11094_cast_fp16")]; string var_11110_pad_type_0 = const()[name = string("op_11110_pad_type_0"), val = string("valid")]; tensor var_11110_strides_0 = const()[name = string("op_11110_strides_0"), val = tensor([1, 1])]; tensor var_11110_pad_0 = const()[name = string("op_11110_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11110_dilations_0 = const()[name = string("op_11110_dilations_0"), val = tensor([1, 1])]; int32 var_11110_groups_0 = const()[name = string("op_11110_groups_0"), val = int32(1)]; tensor var_11110 = conv(dilations = var_11110_dilations_0, groups = var_11110_groups_0, pad = var_11110_pad_0, pad_type = var_11110_pad_type_0, strides = var_11110_strides_0, weight = model_model_layers_21_self_attn_q_proj_weight_palettized, x = var_11094_cast_fp16)[name = string("op_11110")]; tensor var_11115 = const()[name = string("op_11115"), val = tensor([1, 16, 128, 64])]; tensor var_11116 = reshape(shape = var_11115, x = var_11110)[name = string("op_11116")]; tensor var_11121 = const()[name = string("op_11121"), val = tensor([0, 1, 3, 2])]; string var_11133_pad_type_0 = const()[name = string("op_11133_pad_type_0"), val = string("valid")]; tensor var_11133_strides_0 = const()[name = string("op_11133_strides_0"), val = tensor([1, 1])]; tensor var_11133_pad_0 = const()[name = string("op_11133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11133_dilations_0 = const()[name = string("op_11133_dilations_0"), val = tensor([1, 1])]; int32 var_11133_groups_0 = const()[name = string("op_11133_groups_0"), val = int32(1)]; tensor var_11133 = conv(dilations = var_11133_dilations_0, groups = var_11133_groups_0, pad = var_11133_pad_0, pad_type = var_11133_pad_type_0, strides = var_11133_strides_0, weight = model_model_layers_21_self_attn_k_proj_weight_palettized, x = var_11094_cast_fp16)[name = string("op_11133")]; tensor var_11138 = const()[name = string("op_11138"), val = tensor([1, 8, 128, 64])]; tensor var_11139 = reshape(shape = var_11138, x = var_11133)[name = string("op_11139")]; tensor var_11144 = const()[name = string("op_11144"), val = tensor([0, 1, 3, 2])]; string var_11156_pad_type_0 = const()[name = string("op_11156_pad_type_0"), val = string("valid")]; tensor var_11156_strides_0 = const()[name = string("op_11156_strides_0"), val = tensor([1, 1])]; tensor var_11156_pad_0 = const()[name = string("op_11156_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11156_dilations_0 = const()[name = string("op_11156_dilations_0"), val = tensor([1, 1])]; int32 var_11156_groups_0 = const()[name = string("op_11156_groups_0"), val = int32(1)]; tensor var_11156 = conv(dilations = var_11156_dilations_0, groups = var_11156_groups_0, pad = var_11156_pad_0, pad_type = var_11156_pad_type_0, strides = var_11156_strides_0, weight = model_model_layers_21_self_attn_v_proj_weight_palettized, x = var_11094_cast_fp16)[name = string("op_11156")]; tensor var_11161 = const()[name = string("op_11161"), val = tensor([1, 8, 128, 64])]; tensor var_11162 = reshape(shape = var_11161, x = var_11156)[name = string("op_11162")]; tensor var_11167 = const()[name = string("op_11167"), val = tensor([0, 1, 3, 2])]; int32 var_11180 = const()[name = string("op_11180"), val = int32(-1)]; fp16 const_318_promoted = const()[name = string("const_318_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_215 = transpose(perm = var_11121, x = var_11116)[name = string("transpose_61")]; tensor var_11182 = mul(x = hidden_states_215, y = const_318_promoted)[name = string("op_11182")]; bool input_383_interleave_0 = const()[name = string("input_383_interleave_0"), val = bool(false)]; tensor input_383 = concat(axis = var_11180, interleave = input_383_interleave_0, values = (hidden_states_215, var_11182))[name = string("input_383")]; tensor normed_341_axes_0 = const()[name = string("normed_341_axes_0"), val = tensor([-1])]; fp16 var_11177_to_fp16 = const()[name = string("op_11177_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_341_cast_fp16 = layer_norm(axes = normed_341_axes_0, epsilon = var_11177_to_fp16, x = input_383)[name = string("normed_341_cast_fp16")]; tensor normed_343_begin_0 = const()[name = string("normed_343_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_343_end_0 = const()[name = string("normed_343_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_343_end_mask_0 = const()[name = string("normed_343_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_343 = slice_by_index(begin = normed_343_begin_0, end = normed_343_end_0, end_mask = normed_343_end_mask_0, x = normed_341_cast_fp16)[name = string("normed_343")]; tensor const_320 = const()[name = string("const_320"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326383232)))]; tensor q_43 = mul(x = normed_343, y = const_320)[name = string("q_43")]; int32 var_11202 = const()[name = string("op_11202"), val = int32(-1)]; fp16 const_321_promoted = const()[name = string("const_321_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_217 = transpose(perm = var_11144, x = var_11139)[name = string("transpose_60")]; tensor var_11204 = mul(x = hidden_states_217, y = const_321_promoted)[name = string("op_11204")]; bool input_385_interleave_0 = const()[name = string("input_385_interleave_0"), val = bool(false)]; tensor input_385 = concat(axis = var_11202, interleave = input_385_interleave_0, values = (hidden_states_217, var_11204))[name = string("input_385")]; tensor normed_345_axes_0 = const()[name = string("normed_345_axes_0"), val = tensor([-1])]; fp16 var_11199_to_fp16 = const()[name = string("op_11199_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_345_cast_fp16 = layer_norm(axes = normed_345_axes_0, epsilon = var_11199_to_fp16, x = input_385)[name = string("normed_345_cast_fp16")]; tensor normed_347_begin_0 = const()[name = string("normed_347_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_347_end_0 = const()[name = string("normed_347_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_347_end_mask_0 = const()[name = string("normed_347_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_347 = slice_by_index(begin = normed_347_begin_0, end = normed_347_end_0, end_mask = normed_347_end_mask_0, x = normed_345_cast_fp16)[name = string("normed_347")]; tensor const_323 = const()[name = string("const_323"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326383552)))]; tensor k_43 = mul(x = normed_347, y = const_323)[name = string("k_43")]; tensor var_11225 = mul(x = q_43, y = cos_1)[name = string("op_11225")]; tensor var_11230_begin_0 = const()[name = string("op_11230_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11230_end_0 = const()[name = string("op_11230_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_11230_end_mask_0 = const()[name = string("op_11230_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11230 = slice_by_index(begin = var_11230_begin_0, end = var_11230_end_0, end_mask = var_11230_end_mask_0, x = q_43)[name = string("op_11230")]; fp16 const_324_promoted = const()[name = string("const_324_promoted"), val = fp16(-0x1p+0)]; tensor var_11231 = mul(x = var_11230, y = const_324_promoted)[name = string("op_11231")]; tensor var_11236_begin_0 = const()[name = string("op_11236_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11236_end_0 = const()[name = string("op_11236_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_11236_end_mask_0 = const()[name = string("op_11236_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11236 = slice_by_index(begin = var_11236_begin_0, end = var_11236_end_0, end_mask = var_11236_end_mask_0, x = q_43)[name = string("op_11236")]; int32 var_11238 = const()[name = string("op_11238"), val = int32(-1)]; bool var_11239_interleave_0 = const()[name = string("op_11239_interleave_0"), val = bool(false)]; tensor var_11239 = concat(axis = var_11238, interleave = var_11239_interleave_0, values = (var_11231, var_11236))[name = string("op_11239")]; tensor var_11240 = mul(x = var_11239, y = sin_1)[name = string("op_11240")]; tensor query_43 = add(x = var_11225, y = var_11240)[name = string("query_43")]; tensor var_11243 = mul(x = k_43, y = cos_1)[name = string("op_11243")]; tensor var_11248_begin_0 = const()[name = string("op_11248_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11248_end_0 = const()[name = string("op_11248_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_11248_end_mask_0 = const()[name = string("op_11248_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11248 = slice_by_index(begin = var_11248_begin_0, end = var_11248_end_0, end_mask = var_11248_end_mask_0, x = k_43)[name = string("op_11248")]; fp16 const_325_promoted = const()[name = string("const_325_promoted"), val = fp16(-0x1p+0)]; tensor var_11249 = mul(x = var_11248, y = const_325_promoted)[name = string("op_11249")]; tensor var_11254_begin_0 = const()[name = string("op_11254_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11254_end_0 = const()[name = string("op_11254_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_11254_end_mask_0 = const()[name = string("op_11254_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11254 = slice_by_index(begin = var_11254_begin_0, end = var_11254_end_0, end_mask = var_11254_end_mask_0, x = k_43)[name = string("op_11254")]; int32 var_11256 = const()[name = string("op_11256"), val = int32(-1)]; bool var_11257_interleave_0 = const()[name = string("op_11257_interleave_0"), val = bool(false)]; tensor var_11257 = concat(axis = var_11256, interleave = var_11257_interleave_0, values = (var_11249, var_11254))[name = string("op_11257")]; tensor var_11258 = mul(x = var_11257, y = sin_1)[name = string("op_11258")]; tensor key_43 = add(x = var_11243, y = var_11258)[name = string("key_43")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; int32 concat_380_axis_0 = const()[name = string("concat_380_axis_0"), val = int32(0)]; bool concat_380_interleave_0 = const()[name = string("concat_380_interleave_0"), val = bool(false)]; tensor concat_380 = concat(axis = concat_380_axis_0, interleave = concat_380_interleave_0, values = (expand_dims_252, expand_dims_253, current_pos, expand_dims_255))[name = string("concat_380")]; tensor concat_381_values1_0 = const()[name = string("concat_381_values1_0"), val = tensor([0])]; tensor concat_381_values3_0 = const()[name = string("concat_381_values3_0"), val = tensor([0])]; int32 concat_381_axis_0 = const()[name = string("concat_381_axis_0"), val = int32(0)]; bool concat_381_interleave_0 = const()[name = string("concat_381_interleave_0"), val = bool(false)]; tensor concat_381 = concat(axis = concat_381_axis_0, interleave = concat_381_interleave_0, values = (expand_dims_256, concat_381_values1_0, var_1746, concat_381_values3_0))[name = string("concat_381")]; tensor model_model_kv_cache_0_internal_tensor_assign_43_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_43_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_43_cast_fp16 = slice_update(begin = concat_380, begin_mask = model_model_kv_cache_0_internal_tensor_assign_43_begin_mask_0, end = concat_381, end_mask = model_model_kv_cache_0_internal_tensor_assign_43_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_43_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_43_stride_0, update = key_43, x = coreml_update_state_97)[name = string("model_model_kv_cache_0_internal_tensor_assign_43_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_43_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_210_write_state")]; tensor coreml_update_state_98 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_210")]; tensor expand_dims_258 = const()[name = string("expand_dims_258"), val = tensor([49])]; tensor expand_dims_259 = const()[name = string("expand_dims_259"), val = tensor([0])]; tensor expand_dims_261 = const()[name = string("expand_dims_261"), val = tensor([0])]; tensor expand_dims_262 = const()[name = string("expand_dims_262"), val = tensor([50])]; int32 concat_384_axis_0 = const()[name = string("concat_384_axis_0"), val = int32(0)]; bool concat_384_interleave_0 = const()[name = string("concat_384_interleave_0"), val = bool(false)]; tensor concat_384 = concat(axis = concat_384_axis_0, interleave = concat_384_interleave_0, values = (expand_dims_258, expand_dims_259, current_pos, expand_dims_261))[name = string("concat_384")]; tensor concat_385_values1_0 = const()[name = string("concat_385_values1_0"), val = tensor([0])]; tensor concat_385_values3_0 = const()[name = string("concat_385_values3_0"), val = tensor([0])]; int32 concat_385_axis_0 = const()[name = string("concat_385_axis_0"), val = int32(0)]; bool concat_385_interleave_0 = const()[name = string("concat_385_interleave_0"), val = bool(false)]; tensor concat_385 = concat(axis = concat_385_axis_0, interleave = concat_385_interleave_0, values = (expand_dims_262, concat_385_values1_0, var_1746, concat_385_values3_0))[name = string("concat_385")]; tensor model_model_kv_cache_0_internal_tensor_assign_44_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_44_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_44_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_211 = transpose(perm = var_11167, x = var_11162)[name = string("transpose_59")]; tensor model_model_kv_cache_0_internal_tensor_assign_44_cast_fp16 = slice_update(begin = concat_384, begin_mask = model_model_kv_cache_0_internal_tensor_assign_44_begin_mask_0, end = concat_385, end_mask = model_model_kv_cache_0_internal_tensor_assign_44_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_44_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_44_stride_0, update = value_211, x = coreml_update_state_98)[name = string("model_model_kv_cache_0_internal_tensor_assign_44_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_44_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_211_write_state")]; tensor coreml_update_state_99 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_211")]; tensor var_11329_begin_0 = const()[name = string("op_11329_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_11329_end_0 = const()[name = string("op_11329_end_0"), val = tensor([22, 8, 1536, 128])]; tensor var_11329_end_mask_0 = const()[name = string("op_11329_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11329_cast_fp16 = slice_by_index(begin = var_11329_begin_0, end = var_11329_end_0, end_mask = var_11329_end_mask_0, x = coreml_update_state_99)[name = string("op_11329_cast_fp16")]; tensor key_cache_43_axes_0 = const()[name = string("key_cache_43_axes_0"), val = tensor([0])]; tensor key_cache_43_cast_fp16 = squeeze(axes = key_cache_43_axes_0, x = var_11329_cast_fp16)[name = string("key_cache_43_cast_fp16")]; tensor var_11336_begin_0 = const()[name = string("op_11336_begin_0"), val = tensor([49, 0, 0, 0])]; tensor var_11336_end_0 = const()[name = string("op_11336_end_0"), val = tensor([50, 8, 1536, 128])]; tensor var_11336_end_mask_0 = const()[name = string("op_11336_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11336_cast_fp16 = slice_by_index(begin = var_11336_begin_0, end = var_11336_end_0, end_mask = var_11336_end_mask_0, x = coreml_update_state_99)[name = string("op_11336_cast_fp16")]; tensor value_cache_43_axes_0 = const()[name = string("value_cache_43_axes_0"), val = tensor([0])]; tensor value_cache_43_cast_fp16 = squeeze(axes = value_cache_43_axes_0, x = var_11336_cast_fp16)[name = string("value_cache_43_cast_fp16")]; tensor var_11360_axes_0 = const()[name = string("op_11360_axes_0"), val = tensor([1])]; tensor var_11360_cast_fp16 = expand_dims(axes = var_11360_axes_0, x = key_cache_43_cast_fp16)[name = string("op_11360_cast_fp16")]; tensor var_11365 = const()[name = string("op_11365"), val = tensor([1, 2, 1, 1])]; tensor value_215_cast_fp16 = tile(reps = var_11365, x = var_11360_cast_fp16)[name = string("value_215_cast_fp16")]; tensor var_11371 = const()[name = string("op_11371"), val = tensor([1, 16, 1536, 128])]; tensor key_states_87_cast_fp16 = reshape(shape = var_11371, x = value_215_cast_fp16)[name = string("key_states_87_cast_fp16")]; tensor var_11374_axes_0 = const()[name = string("op_11374_axes_0"), val = tensor([1])]; tensor var_11374_cast_fp16 = expand_dims(axes = var_11374_axes_0, x = value_cache_43_cast_fp16)[name = string("op_11374_cast_fp16")]; tensor var_11379 = const()[name = string("op_11379"), val = tensor([1, 2, 1, 1])]; tensor value_219_cast_fp16 = tile(reps = var_11379, x = var_11374_cast_fp16)[name = string("value_219_cast_fp16")]; bool var_11400_transpose_x_0 = const()[name = string("op_11400_transpose_x_0"), val = bool(false)]; bool var_11400_transpose_y_0 = const()[name = string("op_11400_transpose_y_0"), val = bool(true)]; tensor var_11400 = matmul(transpose_x = var_11400_transpose_x_0, transpose_y = var_11400_transpose_y_0, x = query_43, y = key_states_87_cast_fp16)[name = string("op_11400")]; fp16 var_11401_to_fp16 = const()[name = string("op_11401_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_85_cast_fp16 = mul(x = var_11400, y = var_11401_to_fp16)[name = string("attention_85_cast_fp16")]; tensor attention_87_cast_fp16 = add(x = attention_85_cast_fp16, y = causal_mask)[name = string("attention_87_cast_fp16")]; int32 var_11410 = const()[name = string("op_11410"), val = int32(-1)]; tensor var_11412_cast_fp16 = softmax(axis = var_11410, x = attention_87_cast_fp16)[name = string("op_11412_cast_fp16")]; tensor concat_390 = const()[name = string("concat_390"), val = tensor([16, 64, 1536])]; tensor reshape_63_cast_fp16 = reshape(shape = concat_390, x = var_11412_cast_fp16)[name = string("reshape_63_cast_fp16")]; tensor concat_391 = const()[name = string("concat_391"), val = tensor([16, 1536, 128])]; tensor reshape_64_cast_fp16 = reshape(shape = concat_391, x = value_219_cast_fp16)[name = string("reshape_64_cast_fp16")]; bool matmul_21_transpose_x_0 = const()[name = string("matmul_21_transpose_x_0"), val = bool(false)]; bool matmul_21_transpose_y_0 = const()[name = string("matmul_21_transpose_y_0"), val = bool(false)]; tensor matmul_21_cast_fp16 = matmul(transpose_x = matmul_21_transpose_x_0, transpose_y = matmul_21_transpose_y_0, x = reshape_63_cast_fp16, y = reshape_64_cast_fp16)[name = string("matmul_21_cast_fp16")]; tensor concat_395 = const()[name = string("concat_395"), val = tensor([1, 16, 64, 128])]; tensor reshape_65_cast_fp16 = reshape(shape = concat_395, x = matmul_21_cast_fp16)[name = string("reshape_65_cast_fp16")]; tensor var_11424_perm_0 = const()[name = string("op_11424_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_11430 = const()[name = string("op_11430"), val = tensor([1, 64, 2048])]; tensor var_11424_cast_fp16 = transpose(perm = var_11424_perm_0, x = reshape_65_cast_fp16)[name = string("transpose_58")]; tensor output_129_cast_fp16 = reshape(shape = var_11430, x = var_11424_cast_fp16)[name = string("output_129_cast_fp16")]; tensor var_11435 = const()[name = string("op_11435"), val = tensor([0, 2, 1])]; string var_11451_pad_type_0 = const()[name = string("op_11451_pad_type_0"), val = string("valid")]; int32 var_11451_groups_0 = const()[name = string("op_11451_groups_0"), val = int32(1)]; tensor var_11451_strides_0 = const()[name = string("op_11451_strides_0"), val = tensor([1])]; tensor var_11451_pad_0 = const()[name = string("op_11451_pad_0"), val = tensor([0, 0])]; tensor var_11451_dilations_0 = const()[name = string("op_11451_dilations_0"), val = tensor([1])]; tensor squeeze_21_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326383872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327956800))))[name = string("squeeze_21_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11436_cast_fp16 = transpose(perm = var_11435, x = output_129_cast_fp16)[name = string("transpose_57")]; tensor var_11451_cast_fp16 = conv(dilations = var_11451_dilations_0, groups = var_11451_groups_0, pad = var_11451_pad_0, pad_type = var_11451_pad_type_0, strides = var_11451_strides_0, weight = squeeze_21_cast_fp16_to_fp32_to_fp16_palettized, x = var_11436_cast_fp16)[name = string("op_11451_cast_fp16")]; tensor var_11455 = const()[name = string("op_11455"), val = tensor([0, 2, 1])]; tensor attn_output_43_cast_fp16 = transpose(perm = var_11455, x = var_11451_cast_fp16)[name = string("transpose_56")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_211_cast_fp16, y = attn_output_43_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; int32 var_11470 = const()[name = string("op_11470"), val = int32(-1)]; fp16 const_327_promoted_to_fp16 = const()[name = string("const_327_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11472_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_327_promoted_to_fp16)[name = string("op_11472_cast_fp16")]; bool input_389_interleave_0 = const()[name = string("input_389_interleave_0"), val = bool(false)]; tensor input_389_cast_fp16 = concat(axis = var_11470, interleave = input_389_interleave_0, values = (hidden_states_219_cast_fp16, var_11472_cast_fp16))[name = string("input_389_cast_fp16")]; tensor normed_349_axes_0 = const()[name = string("normed_349_axes_0"), val = tensor([-1])]; fp16 var_11467_to_fp16 = const()[name = string("op_11467_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_349_cast_fp16 = layer_norm(axes = normed_349_axes_0, epsilon = var_11467_to_fp16, x = input_389_cast_fp16)[name = string("normed_349_cast_fp16")]; tensor normed_351_begin_0 = const()[name = string("normed_351_begin_0"), val = tensor([0, 0, 0])]; tensor normed_351_end_0 = const()[name = string("normed_351_end_0"), val = tensor([1, 64, 1024])]; tensor normed_351_end_mask_0 = const()[name = string("normed_351_end_mask_0"), val = tensor([true, true, false])]; tensor normed_351_cast_fp16 = slice_by_index(begin = normed_351_begin_0, end = normed_351_end_0, end_mask = normed_351_end_mask_0, x = normed_349_cast_fp16)[name = string("normed_351_cast_fp16")]; tensor const_329_promoted_to_fp16 = const()[name = string("const_329_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327973248)))]; tensor x_85_cast_fp16 = mul(x = normed_351_cast_fp16, y = const_329_promoted_to_fp16)[name = string("x_85_cast_fp16")]; tensor var_11492 = const()[name = string("op_11492"), val = tensor([0, 2, 1])]; tensor input_391_axes_0 = const()[name = string("input_391_axes_0"), val = tensor([2])]; tensor var_11493 = transpose(perm = var_11492, x = x_85_cast_fp16)[name = string("transpose_55")]; tensor input_391 = expand_dims(axes = input_391_axes_0, x = var_11493)[name = string("input_391")]; string input_393_pad_type_0 = const()[name = string("input_393_pad_type_0"), val = string("valid")]; tensor input_393_strides_0 = const()[name = string("input_393_strides_0"), val = tensor([1, 1])]; tensor input_393_pad_0 = const()[name = string("input_393_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_393_dilations_0 = const()[name = string("input_393_dilations_0"), val = tensor([1, 1])]; int32 input_393_groups_0 = const()[name = string("input_393_groups_0"), val = int32(1)]; tensor input_393 = conv(dilations = input_393_dilations_0, groups = input_393_groups_0, pad = input_393_pad_0, pad_type = input_393_pad_type_0, strides = input_393_strides_0, weight = model_model_layers_21_mlp_gate_proj_weight_palettized, x = input_391)[name = string("input_393")]; string b_43_pad_type_0 = const()[name = string("b_43_pad_type_0"), val = string("valid")]; tensor b_43_strides_0 = const()[name = string("b_43_strides_0"), val = tensor([1, 1])]; tensor b_43_pad_0 = const()[name = string("b_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_43_dilations_0 = const()[name = string("b_43_dilations_0"), val = tensor([1, 1])]; int32 b_43_groups_0 = const()[name = string("b_43_groups_0"), val = int32(1)]; tensor b_43 = conv(dilations = b_43_dilations_0, groups = b_43_groups_0, pad = b_43_pad_0, pad_type = b_43_pad_type_0, strides = b_43_strides_0, weight = model_model_layers_21_mlp_up_proj_weight_palettized, x = input_391)[name = string("b_43")]; tensor c_43 = silu(x = input_393)[name = string("c_43")]; tensor input_395 = mul(x = c_43, y = b_43)[name = string("input_395")]; string e_43_pad_type_0 = const()[name = string("e_43_pad_type_0"), val = string("valid")]; tensor e_43_strides_0 = const()[name = string("e_43_strides_0"), val = tensor([1, 1])]; tensor e_43_pad_0 = const()[name = string("e_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_43_dilations_0 = const()[name = string("e_43_dilations_0"), val = tensor([1, 1])]; int32 e_43_groups_0 = const()[name = string("e_43_groups_0"), val = int32(1)]; tensor e_43 = conv(dilations = e_43_dilations_0, groups = e_43_groups_0, pad = e_43_pad_0, pad_type = e_43_pad_type_0, strides = e_43_strides_0, weight = model_model_layers_21_mlp_down_proj_weight_palettized, x = input_395)[name = string("e_43")]; tensor var_11515_axes_0 = const()[name = string("op_11515_axes_0"), val = tensor([2])]; tensor var_11515 = squeeze(axes = var_11515_axes_0, x = e_43)[name = string("op_11515")]; tensor var_11516 = const()[name = string("op_11516"), val = tensor([0, 2, 1])]; tensor var_11517 = transpose(perm = var_11516, x = var_11515)[name = string("transpose_54")]; tensor hidden_states_221_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = var_11517)[name = string("hidden_states_221_cast_fp16")]; int32 var_11531 = const()[name = string("op_11531"), val = int32(-1)]; fp16 const_330_promoted_to_fp16 = const()[name = string("const_330_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11533_cast_fp16 = mul(x = hidden_states_221_cast_fp16, y = const_330_promoted_to_fp16)[name = string("op_11533_cast_fp16")]; bool input_397_interleave_0 = const()[name = string("input_397_interleave_0"), val = bool(false)]; tensor input_397_cast_fp16 = concat(axis = var_11531, interleave = input_397_interleave_0, values = (hidden_states_221_cast_fp16, var_11533_cast_fp16))[name = string("input_397_cast_fp16")]; tensor normed_353_axes_0 = const()[name = string("normed_353_axes_0"), val = tensor([-1])]; fp16 var_11528_to_fp16 = const()[name = string("op_11528_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_353_cast_fp16 = layer_norm(axes = normed_353_axes_0, epsilon = var_11528_to_fp16, x = input_397_cast_fp16)[name = string("normed_353_cast_fp16")]; tensor normed_355_begin_0 = const()[name = string("normed_355_begin_0"), val = tensor([0, 0, 0])]; tensor normed_355_end_0 = const()[name = string("normed_355_end_0"), val = tensor([1, 64, 1024])]; tensor normed_355_end_mask_0 = const()[name = string("normed_355_end_mask_0"), val = tensor([true, true, false])]; tensor normed_355_cast_fp16 = slice_by_index(begin = normed_355_begin_0, end = normed_355_end_0, end_mask = normed_355_end_mask_0, x = normed_353_cast_fp16)[name = string("normed_355_cast_fp16")]; tensor const_332_promoted_to_fp16 = const()[name = string("const_332_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327975360)))]; tensor hidden_states_223_cast_fp16 = mul(x = normed_355_cast_fp16, y = const_332_promoted_to_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor var_11545 = const()[name = string("op_11545"), val = tensor([0, 2, 1])]; tensor var_11548_axes_0 = const()[name = string("op_11548_axes_0"), val = tensor([2])]; tensor var_11546_cast_fp16 = transpose(perm = var_11545, x = hidden_states_223_cast_fp16)[name = string("transpose_53")]; tensor var_11548_cast_fp16 = expand_dims(axes = var_11548_axes_0, x = var_11546_cast_fp16)[name = string("op_11548_cast_fp16")]; string var_11564_pad_type_0 = const()[name = string("op_11564_pad_type_0"), val = string("valid")]; tensor var_11564_strides_0 = const()[name = string("op_11564_strides_0"), val = tensor([1, 1])]; tensor var_11564_pad_0 = const()[name = string("op_11564_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11564_dilations_0 = const()[name = string("op_11564_dilations_0"), val = tensor([1, 1])]; int32 var_11564_groups_0 = const()[name = string("op_11564_groups_0"), val = int32(1)]; tensor var_11564 = conv(dilations = var_11564_dilations_0, groups = var_11564_groups_0, pad = var_11564_pad_0, pad_type = var_11564_pad_type_0, strides = var_11564_strides_0, weight = model_model_layers_22_self_attn_q_proj_weight_palettized, x = var_11548_cast_fp16)[name = string("op_11564")]; tensor var_11569 = const()[name = string("op_11569"), val = tensor([1, 16, 128, 64])]; tensor var_11570 = reshape(shape = var_11569, x = var_11564)[name = string("op_11570")]; tensor var_11575 = const()[name = string("op_11575"), val = tensor([0, 1, 3, 2])]; string var_11587_pad_type_0 = const()[name = string("op_11587_pad_type_0"), val = string("valid")]; tensor var_11587_strides_0 = const()[name = string("op_11587_strides_0"), val = tensor([1, 1])]; tensor var_11587_pad_0 = const()[name = string("op_11587_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11587_dilations_0 = const()[name = string("op_11587_dilations_0"), val = tensor([1, 1])]; int32 var_11587_groups_0 = const()[name = string("op_11587_groups_0"), val = int32(1)]; tensor var_11587 = conv(dilations = var_11587_dilations_0, groups = var_11587_groups_0, pad = var_11587_pad_0, pad_type = var_11587_pad_type_0, strides = var_11587_strides_0, weight = model_model_layers_22_self_attn_k_proj_weight_palettized, x = var_11548_cast_fp16)[name = string("op_11587")]; tensor var_11592 = const()[name = string("op_11592"), val = tensor([1, 8, 128, 64])]; tensor var_11593 = reshape(shape = var_11592, x = var_11587)[name = string("op_11593")]; tensor var_11598 = const()[name = string("op_11598"), val = tensor([0, 1, 3, 2])]; string var_11610_pad_type_0 = const()[name = string("op_11610_pad_type_0"), val = string("valid")]; tensor var_11610_strides_0 = const()[name = string("op_11610_strides_0"), val = tensor([1, 1])]; tensor var_11610_pad_0 = const()[name = string("op_11610_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_11610_dilations_0 = const()[name = string("op_11610_dilations_0"), val = tensor([1, 1])]; int32 var_11610_groups_0 = const()[name = string("op_11610_groups_0"), val = int32(1)]; tensor var_11610 = conv(dilations = var_11610_dilations_0, groups = var_11610_groups_0, pad = var_11610_pad_0, pad_type = var_11610_pad_type_0, strides = var_11610_strides_0, weight = model_model_layers_22_self_attn_v_proj_weight_palettized, x = var_11548_cast_fp16)[name = string("op_11610")]; tensor var_11615 = const()[name = string("op_11615"), val = tensor([1, 8, 128, 64])]; tensor var_11616 = reshape(shape = var_11615, x = var_11610)[name = string("op_11616")]; tensor var_11621 = const()[name = string("op_11621"), val = tensor([0, 1, 3, 2])]; int32 var_11634 = const()[name = string("op_11634"), val = int32(-1)]; fp16 const_333_promoted = const()[name = string("const_333_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_225 = transpose(perm = var_11575, x = var_11570)[name = string("transpose_52")]; tensor var_11636 = mul(x = hidden_states_225, y = const_333_promoted)[name = string("op_11636")]; bool input_401_interleave_0 = const()[name = string("input_401_interleave_0"), val = bool(false)]; tensor input_401 = concat(axis = var_11634, interleave = input_401_interleave_0, values = (hidden_states_225, var_11636))[name = string("input_401")]; tensor normed_357_axes_0 = const()[name = string("normed_357_axes_0"), val = tensor([-1])]; fp16 var_11631_to_fp16 = const()[name = string("op_11631_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_357_cast_fp16 = layer_norm(axes = normed_357_axes_0, epsilon = var_11631_to_fp16, x = input_401)[name = string("normed_357_cast_fp16")]; tensor normed_359_begin_0 = const()[name = string("normed_359_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_359_end_0 = const()[name = string("normed_359_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_359_end_mask_0 = const()[name = string("normed_359_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_359 = slice_by_index(begin = normed_359_begin_0, end = normed_359_end_0, end_mask = normed_359_end_mask_0, x = normed_357_cast_fp16)[name = string("normed_359")]; tensor const_335 = const()[name = string("const_335"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327977472)))]; tensor q_45 = mul(x = normed_359, y = const_335)[name = string("q_45")]; int32 var_11656 = const()[name = string("op_11656"), val = int32(-1)]; fp16 const_336_promoted = const()[name = string("const_336_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_227 = transpose(perm = var_11598, x = var_11593)[name = string("transpose_51")]; tensor var_11658 = mul(x = hidden_states_227, y = const_336_promoted)[name = string("op_11658")]; bool input_403_interleave_0 = const()[name = string("input_403_interleave_0"), val = bool(false)]; tensor input_403 = concat(axis = var_11656, interleave = input_403_interleave_0, values = (hidden_states_227, var_11658))[name = string("input_403")]; tensor normed_361_axes_0 = const()[name = string("normed_361_axes_0"), val = tensor([-1])]; fp16 var_11653_to_fp16 = const()[name = string("op_11653_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_361_cast_fp16 = layer_norm(axes = normed_361_axes_0, epsilon = var_11653_to_fp16, x = input_403)[name = string("normed_361_cast_fp16")]; tensor normed_363_begin_0 = const()[name = string("normed_363_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_363_end_0 = const()[name = string("normed_363_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_363_end_mask_0 = const()[name = string("normed_363_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_363 = slice_by_index(begin = normed_363_begin_0, end = normed_363_end_0, end_mask = normed_363_end_mask_0, x = normed_361_cast_fp16)[name = string("normed_363")]; tensor const_338 = const()[name = string("const_338"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327977792)))]; tensor k_45 = mul(x = normed_363, y = const_338)[name = string("k_45")]; tensor var_11679 = mul(x = q_45, y = cos_1)[name = string("op_11679")]; tensor var_11684_begin_0 = const()[name = string("op_11684_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11684_end_0 = const()[name = string("op_11684_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_11684_end_mask_0 = const()[name = string("op_11684_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11684 = slice_by_index(begin = var_11684_begin_0, end = var_11684_end_0, end_mask = var_11684_end_mask_0, x = q_45)[name = string("op_11684")]; fp16 const_339_promoted = const()[name = string("const_339_promoted"), val = fp16(-0x1p+0)]; tensor var_11685 = mul(x = var_11684, y = const_339_promoted)[name = string("op_11685")]; tensor var_11690_begin_0 = const()[name = string("op_11690_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11690_end_0 = const()[name = string("op_11690_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_11690_end_mask_0 = const()[name = string("op_11690_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11690 = slice_by_index(begin = var_11690_begin_0, end = var_11690_end_0, end_mask = var_11690_end_mask_0, x = q_45)[name = string("op_11690")]; int32 var_11692 = const()[name = string("op_11692"), val = int32(-1)]; bool var_11693_interleave_0 = const()[name = string("op_11693_interleave_0"), val = bool(false)]; tensor var_11693 = concat(axis = var_11692, interleave = var_11693_interleave_0, values = (var_11685, var_11690))[name = string("op_11693")]; tensor var_11694 = mul(x = var_11693, y = sin_1)[name = string("op_11694")]; tensor query_45 = add(x = var_11679, y = var_11694)[name = string("query_45")]; tensor var_11697 = mul(x = k_45, y = cos_1)[name = string("op_11697")]; tensor var_11702_begin_0 = const()[name = string("op_11702_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_11702_end_0 = const()[name = string("op_11702_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_11702_end_mask_0 = const()[name = string("op_11702_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_11702 = slice_by_index(begin = var_11702_begin_0, end = var_11702_end_0, end_mask = var_11702_end_mask_0, x = k_45)[name = string("op_11702")]; fp16 const_340_promoted = const()[name = string("const_340_promoted"), val = fp16(-0x1p+0)]; tensor var_11703 = mul(x = var_11702, y = const_340_promoted)[name = string("op_11703")]; tensor var_11708_begin_0 = const()[name = string("op_11708_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_11708_end_0 = const()[name = string("op_11708_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_11708_end_mask_0 = const()[name = string("op_11708_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_11708 = slice_by_index(begin = var_11708_begin_0, end = var_11708_end_0, end_mask = var_11708_end_mask_0, x = k_45)[name = string("op_11708")]; int32 var_11710 = const()[name = string("op_11710"), val = int32(-1)]; bool var_11711_interleave_0 = const()[name = string("op_11711_interleave_0"), val = bool(false)]; tensor var_11711 = concat(axis = var_11710, interleave = var_11711_interleave_0, values = (var_11703, var_11708))[name = string("op_11711")]; tensor var_11712 = mul(x = var_11711, y = sin_1)[name = string("op_11712")]; tensor key_45 = add(x = var_11697, y = var_11712)[name = string("key_45")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; int32 concat_398_axis_0 = const()[name = string("concat_398_axis_0"), val = int32(0)]; bool concat_398_interleave_0 = const()[name = string("concat_398_interleave_0"), val = bool(false)]; tensor concat_398 = concat(axis = concat_398_axis_0, interleave = concat_398_interleave_0, values = (expand_dims_264, expand_dims_265, current_pos, expand_dims_267))[name = string("concat_398")]; tensor concat_399_values1_0 = const()[name = string("concat_399_values1_0"), val = tensor([0])]; tensor concat_399_values3_0 = const()[name = string("concat_399_values3_0"), val = tensor([0])]; int32 concat_399_axis_0 = const()[name = string("concat_399_axis_0"), val = int32(0)]; bool concat_399_interleave_0 = const()[name = string("concat_399_interleave_0"), val = bool(false)]; tensor concat_399 = concat(axis = concat_399_axis_0, interleave = concat_399_interleave_0, values = (expand_dims_268, concat_399_values1_0, var_1746, concat_399_values3_0))[name = string("concat_399")]; tensor model_model_kv_cache_0_internal_tensor_assign_45_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_45_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_45_cast_fp16 = slice_update(begin = concat_398, begin_mask = model_model_kv_cache_0_internal_tensor_assign_45_begin_mask_0, end = concat_399, end_mask = model_model_kv_cache_0_internal_tensor_assign_45_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_45_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_45_stride_0, update = key_45, x = coreml_update_state_99)[name = string("model_model_kv_cache_0_internal_tensor_assign_45_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_45_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_212_write_state")]; tensor coreml_update_state_100 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_212")]; tensor expand_dims_270 = const()[name = string("expand_dims_270"), val = tensor([50])]; tensor expand_dims_271 = const()[name = string("expand_dims_271"), val = tensor([0])]; tensor expand_dims_273 = const()[name = string("expand_dims_273"), val = tensor([0])]; tensor expand_dims_274 = const()[name = string("expand_dims_274"), val = tensor([51])]; int32 concat_402_axis_0 = const()[name = string("concat_402_axis_0"), val = int32(0)]; bool concat_402_interleave_0 = const()[name = string("concat_402_interleave_0"), val = bool(false)]; tensor concat_402 = concat(axis = concat_402_axis_0, interleave = concat_402_interleave_0, values = (expand_dims_270, expand_dims_271, current_pos, expand_dims_273))[name = string("concat_402")]; tensor concat_403_values1_0 = const()[name = string("concat_403_values1_0"), val = tensor([0])]; tensor concat_403_values3_0 = const()[name = string("concat_403_values3_0"), val = tensor([0])]; int32 concat_403_axis_0 = const()[name = string("concat_403_axis_0"), val = int32(0)]; bool concat_403_interleave_0 = const()[name = string("concat_403_interleave_0"), val = bool(false)]; tensor concat_403 = concat(axis = concat_403_axis_0, interleave = concat_403_interleave_0, values = (expand_dims_274, concat_403_values1_0, var_1746, concat_403_values3_0))[name = string("concat_403")]; tensor model_model_kv_cache_0_internal_tensor_assign_46_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_46_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_46_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_221 = transpose(perm = var_11621, x = var_11616)[name = string("transpose_50")]; tensor model_model_kv_cache_0_internal_tensor_assign_46_cast_fp16 = slice_update(begin = concat_402, begin_mask = model_model_kv_cache_0_internal_tensor_assign_46_begin_mask_0, end = concat_403, end_mask = model_model_kv_cache_0_internal_tensor_assign_46_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_46_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_46_stride_0, update = value_221, x = coreml_update_state_100)[name = string("model_model_kv_cache_0_internal_tensor_assign_46_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_46_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_213_write_state")]; tensor coreml_update_state_101 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_213")]; tensor var_11783_begin_0 = const()[name = string("op_11783_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_11783_end_0 = const()[name = string("op_11783_end_0"), val = tensor([23, 8, 1536, 128])]; tensor var_11783_end_mask_0 = const()[name = string("op_11783_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11783_cast_fp16 = slice_by_index(begin = var_11783_begin_0, end = var_11783_end_0, end_mask = var_11783_end_mask_0, x = coreml_update_state_101)[name = string("op_11783_cast_fp16")]; tensor key_cache_45_axes_0 = const()[name = string("key_cache_45_axes_0"), val = tensor([0])]; tensor key_cache_45_cast_fp16 = squeeze(axes = key_cache_45_axes_0, x = var_11783_cast_fp16)[name = string("key_cache_45_cast_fp16")]; tensor var_11790_begin_0 = const()[name = string("op_11790_begin_0"), val = tensor([50, 0, 0, 0])]; tensor var_11790_end_0 = const()[name = string("op_11790_end_0"), val = tensor([51, 8, 1536, 128])]; tensor var_11790_end_mask_0 = const()[name = string("op_11790_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_11790_cast_fp16 = slice_by_index(begin = var_11790_begin_0, end = var_11790_end_0, end_mask = var_11790_end_mask_0, x = coreml_update_state_101)[name = string("op_11790_cast_fp16")]; tensor value_cache_45_axes_0 = const()[name = string("value_cache_45_axes_0"), val = tensor([0])]; tensor value_cache_45_cast_fp16 = squeeze(axes = value_cache_45_axes_0, x = var_11790_cast_fp16)[name = string("value_cache_45_cast_fp16")]; tensor var_11814_axes_0 = const()[name = string("op_11814_axes_0"), val = tensor([1])]; tensor var_11814_cast_fp16 = expand_dims(axes = var_11814_axes_0, x = key_cache_45_cast_fp16)[name = string("op_11814_cast_fp16")]; tensor var_11819 = const()[name = string("op_11819"), val = tensor([1, 2, 1, 1])]; tensor value_225_cast_fp16 = tile(reps = var_11819, x = var_11814_cast_fp16)[name = string("value_225_cast_fp16")]; tensor var_11825 = const()[name = string("op_11825"), val = tensor([1, 16, 1536, 128])]; tensor key_states_91_cast_fp16 = reshape(shape = var_11825, x = value_225_cast_fp16)[name = string("key_states_91_cast_fp16")]; tensor var_11828_axes_0 = const()[name = string("op_11828_axes_0"), val = tensor([1])]; tensor var_11828_cast_fp16 = expand_dims(axes = var_11828_axes_0, x = value_cache_45_cast_fp16)[name = string("op_11828_cast_fp16")]; tensor var_11833 = const()[name = string("op_11833"), val = tensor([1, 2, 1, 1])]; tensor value_229_cast_fp16 = tile(reps = var_11833, x = var_11828_cast_fp16)[name = string("value_229_cast_fp16")]; bool var_11854_transpose_x_0 = const()[name = string("op_11854_transpose_x_0"), val = bool(false)]; bool var_11854_transpose_y_0 = const()[name = string("op_11854_transpose_y_0"), val = bool(true)]; tensor var_11854 = matmul(transpose_x = var_11854_transpose_x_0, transpose_y = var_11854_transpose_y_0, x = query_45, y = key_states_91_cast_fp16)[name = string("op_11854")]; fp16 var_11855_to_fp16 = const()[name = string("op_11855_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_89_cast_fp16 = mul(x = var_11854, y = var_11855_to_fp16)[name = string("attention_89_cast_fp16")]; tensor attention_91_cast_fp16 = add(x = attention_89_cast_fp16, y = causal_mask)[name = string("attention_91_cast_fp16")]; int32 var_11864 = const()[name = string("op_11864"), val = int32(-1)]; tensor var_11866_cast_fp16 = softmax(axis = var_11864, x = attention_91_cast_fp16)[name = string("op_11866_cast_fp16")]; tensor concat_408 = const()[name = string("concat_408"), val = tensor([16, 64, 1536])]; tensor reshape_66_cast_fp16 = reshape(shape = concat_408, x = var_11866_cast_fp16)[name = string("reshape_66_cast_fp16")]; tensor concat_409 = const()[name = string("concat_409"), val = tensor([16, 1536, 128])]; tensor reshape_67_cast_fp16 = reshape(shape = concat_409, x = value_229_cast_fp16)[name = string("reshape_67_cast_fp16")]; bool matmul_22_transpose_x_0 = const()[name = string("matmul_22_transpose_x_0"), val = bool(false)]; bool matmul_22_transpose_y_0 = const()[name = string("matmul_22_transpose_y_0"), val = bool(false)]; tensor matmul_22_cast_fp16 = matmul(transpose_x = matmul_22_transpose_x_0, transpose_y = matmul_22_transpose_y_0, x = reshape_66_cast_fp16, y = reshape_67_cast_fp16)[name = string("matmul_22_cast_fp16")]; tensor concat_413 = const()[name = string("concat_413"), val = tensor([1, 16, 64, 128])]; tensor reshape_68_cast_fp16 = reshape(shape = concat_413, x = matmul_22_cast_fp16)[name = string("reshape_68_cast_fp16")]; tensor var_11878_perm_0 = const()[name = string("op_11878_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_11884 = const()[name = string("op_11884"), val = tensor([1, 64, 2048])]; tensor var_11878_cast_fp16 = transpose(perm = var_11878_perm_0, x = reshape_68_cast_fp16)[name = string("transpose_49")]; tensor output_135_cast_fp16 = reshape(shape = var_11884, x = var_11878_cast_fp16)[name = string("output_135_cast_fp16")]; tensor var_11889 = const()[name = string("op_11889"), val = tensor([0, 2, 1])]; string var_11905_pad_type_0 = const()[name = string("op_11905_pad_type_0"), val = string("valid")]; int32 var_11905_groups_0 = const()[name = string("op_11905_groups_0"), val = int32(1)]; tensor var_11905_strides_0 = const()[name = string("op_11905_strides_0"), val = tensor([1])]; tensor var_11905_pad_0 = const()[name = string("op_11905_pad_0"), val = tensor([0, 0])]; tensor var_11905_dilations_0 = const()[name = string("op_11905_dilations_0"), val = tensor([1])]; tensor squeeze_22_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(327978112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329551040))))[name = string("squeeze_22_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_11890_cast_fp16 = transpose(perm = var_11889, x = output_135_cast_fp16)[name = string("transpose_48")]; tensor var_11905_cast_fp16 = conv(dilations = var_11905_dilations_0, groups = var_11905_groups_0, pad = var_11905_pad_0, pad_type = var_11905_pad_type_0, strides = var_11905_strides_0, weight = squeeze_22_cast_fp16_to_fp32_to_fp16_palettized, x = var_11890_cast_fp16)[name = string("op_11905_cast_fp16")]; tensor var_11909 = const()[name = string("op_11909"), val = tensor([0, 2, 1])]; tensor attn_output_45_cast_fp16 = transpose(perm = var_11909, x = var_11905_cast_fp16)[name = string("transpose_47")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_221_cast_fp16, y = attn_output_45_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; int32 var_11924 = const()[name = string("op_11924"), val = int32(-1)]; fp16 const_342_promoted_to_fp16 = const()[name = string("const_342_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11926_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_342_promoted_to_fp16)[name = string("op_11926_cast_fp16")]; bool input_407_interleave_0 = const()[name = string("input_407_interleave_0"), val = bool(false)]; tensor input_407_cast_fp16 = concat(axis = var_11924, interleave = input_407_interleave_0, values = (hidden_states_229_cast_fp16, var_11926_cast_fp16))[name = string("input_407_cast_fp16")]; tensor normed_365_axes_0 = const()[name = string("normed_365_axes_0"), val = tensor([-1])]; fp16 var_11921_to_fp16 = const()[name = string("op_11921_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_365_cast_fp16 = layer_norm(axes = normed_365_axes_0, epsilon = var_11921_to_fp16, x = input_407_cast_fp16)[name = string("normed_365_cast_fp16")]; tensor normed_367_begin_0 = const()[name = string("normed_367_begin_0"), val = tensor([0, 0, 0])]; tensor normed_367_end_0 = const()[name = string("normed_367_end_0"), val = tensor([1, 64, 1024])]; tensor normed_367_end_mask_0 = const()[name = string("normed_367_end_mask_0"), val = tensor([true, true, false])]; tensor normed_367_cast_fp16 = slice_by_index(begin = normed_367_begin_0, end = normed_367_end_0, end_mask = normed_367_end_mask_0, x = normed_365_cast_fp16)[name = string("normed_367_cast_fp16")]; tensor const_344_promoted_to_fp16 = const()[name = string("const_344_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329567488)))]; tensor x_89_cast_fp16 = mul(x = normed_367_cast_fp16, y = const_344_promoted_to_fp16)[name = string("x_89_cast_fp16")]; tensor var_11946 = const()[name = string("op_11946"), val = tensor([0, 2, 1])]; tensor input_409_axes_0 = const()[name = string("input_409_axes_0"), val = tensor([2])]; tensor var_11947 = transpose(perm = var_11946, x = x_89_cast_fp16)[name = string("transpose_46")]; tensor input_409 = expand_dims(axes = input_409_axes_0, x = var_11947)[name = string("input_409")]; string input_411_pad_type_0 = const()[name = string("input_411_pad_type_0"), val = string("valid")]; tensor input_411_strides_0 = const()[name = string("input_411_strides_0"), val = tensor([1, 1])]; tensor input_411_pad_0 = const()[name = string("input_411_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_411_dilations_0 = const()[name = string("input_411_dilations_0"), val = tensor([1, 1])]; int32 input_411_groups_0 = const()[name = string("input_411_groups_0"), val = int32(1)]; tensor input_411 = conv(dilations = input_411_dilations_0, groups = input_411_groups_0, pad = input_411_pad_0, pad_type = input_411_pad_type_0, strides = input_411_strides_0, weight = model_model_layers_22_mlp_gate_proj_weight_palettized, x = input_409)[name = string("input_411")]; string b_45_pad_type_0 = const()[name = string("b_45_pad_type_0"), val = string("valid")]; tensor b_45_strides_0 = const()[name = string("b_45_strides_0"), val = tensor([1, 1])]; tensor b_45_pad_0 = const()[name = string("b_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_45_dilations_0 = const()[name = string("b_45_dilations_0"), val = tensor([1, 1])]; int32 b_45_groups_0 = const()[name = string("b_45_groups_0"), val = int32(1)]; tensor b_45 = conv(dilations = b_45_dilations_0, groups = b_45_groups_0, pad = b_45_pad_0, pad_type = b_45_pad_type_0, strides = b_45_strides_0, weight = model_model_layers_22_mlp_up_proj_weight_palettized, x = input_409)[name = string("b_45")]; tensor c_45 = silu(x = input_411)[name = string("c_45")]; tensor input_413 = mul(x = c_45, y = b_45)[name = string("input_413")]; string e_45_pad_type_0 = const()[name = string("e_45_pad_type_0"), val = string("valid")]; tensor e_45_strides_0 = const()[name = string("e_45_strides_0"), val = tensor([1, 1])]; tensor e_45_pad_0 = const()[name = string("e_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_45_dilations_0 = const()[name = string("e_45_dilations_0"), val = tensor([1, 1])]; int32 e_45_groups_0 = const()[name = string("e_45_groups_0"), val = int32(1)]; tensor e_45 = conv(dilations = e_45_dilations_0, groups = e_45_groups_0, pad = e_45_pad_0, pad_type = e_45_pad_type_0, strides = e_45_strides_0, weight = model_model_layers_22_mlp_down_proj_weight_palettized, x = input_413)[name = string("e_45")]; tensor var_11969_axes_0 = const()[name = string("op_11969_axes_0"), val = tensor([2])]; tensor var_11969 = squeeze(axes = var_11969_axes_0, x = e_45)[name = string("op_11969")]; tensor var_11970 = const()[name = string("op_11970"), val = tensor([0, 2, 1])]; tensor var_11971 = transpose(perm = var_11970, x = var_11969)[name = string("transpose_45")]; tensor hidden_states_231_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = var_11971)[name = string("hidden_states_231_cast_fp16")]; int32 var_11985 = const()[name = string("op_11985"), val = int32(-1)]; fp16 const_345_promoted_to_fp16 = const()[name = string("const_345_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_11987_cast_fp16 = mul(x = hidden_states_231_cast_fp16, y = const_345_promoted_to_fp16)[name = string("op_11987_cast_fp16")]; bool input_415_interleave_0 = const()[name = string("input_415_interleave_0"), val = bool(false)]; tensor input_415_cast_fp16 = concat(axis = var_11985, interleave = input_415_interleave_0, values = (hidden_states_231_cast_fp16, var_11987_cast_fp16))[name = string("input_415_cast_fp16")]; tensor normed_369_axes_0 = const()[name = string("normed_369_axes_0"), val = tensor([-1])]; fp16 var_11982_to_fp16 = const()[name = string("op_11982_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_369_cast_fp16 = layer_norm(axes = normed_369_axes_0, epsilon = var_11982_to_fp16, x = input_415_cast_fp16)[name = string("normed_369_cast_fp16")]; tensor normed_371_begin_0 = const()[name = string("normed_371_begin_0"), val = tensor([0, 0, 0])]; tensor normed_371_end_0 = const()[name = string("normed_371_end_0"), val = tensor([1, 64, 1024])]; tensor normed_371_end_mask_0 = const()[name = string("normed_371_end_mask_0"), val = tensor([true, true, false])]; tensor normed_371_cast_fp16 = slice_by_index(begin = normed_371_begin_0, end = normed_371_end_0, end_mask = normed_371_end_mask_0, x = normed_369_cast_fp16)[name = string("normed_371_cast_fp16")]; tensor const_347_promoted_to_fp16 = const()[name = string("const_347_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329569600)))]; tensor hidden_states_233_cast_fp16 = mul(x = normed_371_cast_fp16, y = const_347_promoted_to_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor var_11999 = const()[name = string("op_11999"), val = tensor([0, 2, 1])]; tensor var_12002_axes_0 = const()[name = string("op_12002_axes_0"), val = tensor([2])]; tensor var_12000_cast_fp16 = transpose(perm = var_11999, x = hidden_states_233_cast_fp16)[name = string("transpose_44")]; tensor var_12002_cast_fp16 = expand_dims(axes = var_12002_axes_0, x = var_12000_cast_fp16)[name = string("op_12002_cast_fp16")]; string var_12018_pad_type_0 = const()[name = string("op_12018_pad_type_0"), val = string("valid")]; tensor var_12018_strides_0 = const()[name = string("op_12018_strides_0"), val = tensor([1, 1])]; tensor var_12018_pad_0 = const()[name = string("op_12018_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12018_dilations_0 = const()[name = string("op_12018_dilations_0"), val = tensor([1, 1])]; int32 var_12018_groups_0 = const()[name = string("op_12018_groups_0"), val = int32(1)]; tensor var_12018 = conv(dilations = var_12018_dilations_0, groups = var_12018_groups_0, pad = var_12018_pad_0, pad_type = var_12018_pad_type_0, strides = var_12018_strides_0, weight = model_model_layers_23_self_attn_q_proj_weight_palettized, x = var_12002_cast_fp16)[name = string("op_12018")]; tensor var_12023 = const()[name = string("op_12023"), val = tensor([1, 16, 128, 64])]; tensor var_12024 = reshape(shape = var_12023, x = var_12018)[name = string("op_12024")]; tensor var_12029 = const()[name = string("op_12029"), val = tensor([0, 1, 3, 2])]; string var_12041_pad_type_0 = const()[name = string("op_12041_pad_type_0"), val = string("valid")]; tensor var_12041_strides_0 = const()[name = string("op_12041_strides_0"), val = tensor([1, 1])]; tensor var_12041_pad_0 = const()[name = string("op_12041_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12041_dilations_0 = const()[name = string("op_12041_dilations_0"), val = tensor([1, 1])]; int32 var_12041_groups_0 = const()[name = string("op_12041_groups_0"), val = int32(1)]; tensor var_12041 = conv(dilations = var_12041_dilations_0, groups = var_12041_groups_0, pad = var_12041_pad_0, pad_type = var_12041_pad_type_0, strides = var_12041_strides_0, weight = model_model_layers_23_self_attn_k_proj_weight_palettized, x = var_12002_cast_fp16)[name = string("op_12041")]; tensor var_12046 = const()[name = string("op_12046"), val = tensor([1, 8, 128, 64])]; tensor var_12047 = reshape(shape = var_12046, x = var_12041)[name = string("op_12047")]; tensor var_12052 = const()[name = string("op_12052"), val = tensor([0, 1, 3, 2])]; string var_12064_pad_type_0 = const()[name = string("op_12064_pad_type_0"), val = string("valid")]; tensor var_12064_strides_0 = const()[name = string("op_12064_strides_0"), val = tensor([1, 1])]; tensor var_12064_pad_0 = const()[name = string("op_12064_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12064_dilations_0 = const()[name = string("op_12064_dilations_0"), val = tensor([1, 1])]; int32 var_12064_groups_0 = const()[name = string("op_12064_groups_0"), val = int32(1)]; tensor var_12064 = conv(dilations = var_12064_dilations_0, groups = var_12064_groups_0, pad = var_12064_pad_0, pad_type = var_12064_pad_type_0, strides = var_12064_strides_0, weight = model_model_layers_23_self_attn_v_proj_weight_palettized, x = var_12002_cast_fp16)[name = string("op_12064")]; tensor var_12069 = const()[name = string("op_12069"), val = tensor([1, 8, 128, 64])]; tensor var_12070 = reshape(shape = var_12069, x = var_12064)[name = string("op_12070")]; tensor var_12075 = const()[name = string("op_12075"), val = tensor([0, 1, 3, 2])]; int32 var_12088 = const()[name = string("op_12088"), val = int32(-1)]; fp16 const_348_promoted = const()[name = string("const_348_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_235 = transpose(perm = var_12029, x = var_12024)[name = string("transpose_43")]; tensor var_12090 = mul(x = hidden_states_235, y = const_348_promoted)[name = string("op_12090")]; bool input_419_interleave_0 = const()[name = string("input_419_interleave_0"), val = bool(false)]; tensor input_419 = concat(axis = var_12088, interleave = input_419_interleave_0, values = (hidden_states_235, var_12090))[name = string("input_419")]; tensor normed_373_axes_0 = const()[name = string("normed_373_axes_0"), val = tensor([-1])]; fp16 var_12085_to_fp16 = const()[name = string("op_12085_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_373_cast_fp16 = layer_norm(axes = normed_373_axes_0, epsilon = var_12085_to_fp16, x = input_419)[name = string("normed_373_cast_fp16")]; tensor normed_375_begin_0 = const()[name = string("normed_375_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_375_end_0 = const()[name = string("normed_375_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_375_end_mask_0 = const()[name = string("normed_375_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_375 = slice_by_index(begin = normed_375_begin_0, end = normed_375_end_0, end_mask = normed_375_end_mask_0, x = normed_373_cast_fp16)[name = string("normed_375")]; tensor const_350 = const()[name = string("const_350"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329571712)))]; tensor q_47 = mul(x = normed_375, y = const_350)[name = string("q_47")]; int32 var_12110 = const()[name = string("op_12110"), val = int32(-1)]; fp16 const_351_promoted = const()[name = string("const_351_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_237 = transpose(perm = var_12052, x = var_12047)[name = string("transpose_42")]; tensor var_12112 = mul(x = hidden_states_237, y = const_351_promoted)[name = string("op_12112")]; bool input_421_interleave_0 = const()[name = string("input_421_interleave_0"), val = bool(false)]; tensor input_421 = concat(axis = var_12110, interleave = input_421_interleave_0, values = (hidden_states_237, var_12112))[name = string("input_421")]; tensor normed_377_axes_0 = const()[name = string("normed_377_axes_0"), val = tensor([-1])]; fp16 var_12107_to_fp16 = const()[name = string("op_12107_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_377_cast_fp16 = layer_norm(axes = normed_377_axes_0, epsilon = var_12107_to_fp16, x = input_421)[name = string("normed_377_cast_fp16")]; tensor normed_379_begin_0 = const()[name = string("normed_379_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_379_end_0 = const()[name = string("normed_379_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_379_end_mask_0 = const()[name = string("normed_379_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_379 = slice_by_index(begin = normed_379_begin_0, end = normed_379_end_0, end_mask = normed_379_end_mask_0, x = normed_377_cast_fp16)[name = string("normed_379")]; tensor const_353 = const()[name = string("const_353"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329572032)))]; tensor k_47 = mul(x = normed_379, y = const_353)[name = string("k_47")]; tensor var_12133 = mul(x = q_47, y = cos_1)[name = string("op_12133")]; tensor var_12138_begin_0 = const()[name = string("op_12138_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12138_end_0 = const()[name = string("op_12138_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_12138_end_mask_0 = const()[name = string("op_12138_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12138 = slice_by_index(begin = var_12138_begin_0, end = var_12138_end_0, end_mask = var_12138_end_mask_0, x = q_47)[name = string("op_12138")]; fp16 const_354_promoted = const()[name = string("const_354_promoted"), val = fp16(-0x1p+0)]; tensor var_12139 = mul(x = var_12138, y = const_354_promoted)[name = string("op_12139")]; tensor var_12144_begin_0 = const()[name = string("op_12144_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12144_end_0 = const()[name = string("op_12144_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_12144_end_mask_0 = const()[name = string("op_12144_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12144 = slice_by_index(begin = var_12144_begin_0, end = var_12144_end_0, end_mask = var_12144_end_mask_0, x = q_47)[name = string("op_12144")]; int32 var_12146 = const()[name = string("op_12146"), val = int32(-1)]; bool var_12147_interleave_0 = const()[name = string("op_12147_interleave_0"), val = bool(false)]; tensor var_12147 = concat(axis = var_12146, interleave = var_12147_interleave_0, values = (var_12139, var_12144))[name = string("op_12147")]; tensor var_12148 = mul(x = var_12147, y = sin_1)[name = string("op_12148")]; tensor query_47 = add(x = var_12133, y = var_12148)[name = string("query_47")]; tensor var_12151 = mul(x = k_47, y = cos_1)[name = string("op_12151")]; tensor var_12156_begin_0 = const()[name = string("op_12156_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12156_end_0 = const()[name = string("op_12156_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_12156_end_mask_0 = const()[name = string("op_12156_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12156 = slice_by_index(begin = var_12156_begin_0, end = var_12156_end_0, end_mask = var_12156_end_mask_0, x = k_47)[name = string("op_12156")]; fp16 const_355_promoted = const()[name = string("const_355_promoted"), val = fp16(-0x1p+0)]; tensor var_12157 = mul(x = var_12156, y = const_355_promoted)[name = string("op_12157")]; tensor var_12162_begin_0 = const()[name = string("op_12162_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12162_end_0 = const()[name = string("op_12162_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_12162_end_mask_0 = const()[name = string("op_12162_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12162 = slice_by_index(begin = var_12162_begin_0, end = var_12162_end_0, end_mask = var_12162_end_mask_0, x = k_47)[name = string("op_12162")]; int32 var_12164 = const()[name = string("op_12164"), val = int32(-1)]; bool var_12165_interleave_0 = const()[name = string("op_12165_interleave_0"), val = bool(false)]; tensor var_12165 = concat(axis = var_12164, interleave = var_12165_interleave_0, values = (var_12157, var_12162))[name = string("op_12165")]; tensor var_12166 = mul(x = var_12165, y = sin_1)[name = string("op_12166")]; tensor key_47 = add(x = var_12151, y = var_12166)[name = string("key_47")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; int32 concat_416_axis_0 = const()[name = string("concat_416_axis_0"), val = int32(0)]; bool concat_416_interleave_0 = const()[name = string("concat_416_interleave_0"), val = bool(false)]; tensor concat_416 = concat(axis = concat_416_axis_0, interleave = concat_416_interleave_0, values = (expand_dims_276, expand_dims_277, current_pos, expand_dims_279))[name = string("concat_416")]; tensor concat_417_values1_0 = const()[name = string("concat_417_values1_0"), val = tensor([0])]; tensor concat_417_values3_0 = const()[name = string("concat_417_values3_0"), val = tensor([0])]; int32 concat_417_axis_0 = const()[name = string("concat_417_axis_0"), val = int32(0)]; bool concat_417_interleave_0 = const()[name = string("concat_417_interleave_0"), val = bool(false)]; tensor concat_417 = concat(axis = concat_417_axis_0, interleave = concat_417_interleave_0, values = (expand_dims_280, concat_417_values1_0, var_1746, concat_417_values3_0))[name = string("concat_417")]; tensor model_model_kv_cache_0_internal_tensor_assign_47_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_47_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_47_cast_fp16 = slice_update(begin = concat_416, begin_mask = model_model_kv_cache_0_internal_tensor_assign_47_begin_mask_0, end = concat_417, end_mask = model_model_kv_cache_0_internal_tensor_assign_47_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_47_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_47_stride_0, update = key_47, x = coreml_update_state_101)[name = string("model_model_kv_cache_0_internal_tensor_assign_47_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_47_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_214_write_state")]; tensor coreml_update_state_102 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_214")]; tensor expand_dims_282 = const()[name = string("expand_dims_282"), val = tensor([51])]; tensor expand_dims_283 = const()[name = string("expand_dims_283"), val = tensor([0])]; tensor expand_dims_285 = const()[name = string("expand_dims_285"), val = tensor([0])]; tensor expand_dims_286 = const()[name = string("expand_dims_286"), val = tensor([52])]; int32 concat_420_axis_0 = const()[name = string("concat_420_axis_0"), val = int32(0)]; bool concat_420_interleave_0 = const()[name = string("concat_420_interleave_0"), val = bool(false)]; tensor concat_420 = concat(axis = concat_420_axis_0, interleave = concat_420_interleave_0, values = (expand_dims_282, expand_dims_283, current_pos, expand_dims_285))[name = string("concat_420")]; tensor concat_421_values1_0 = const()[name = string("concat_421_values1_0"), val = tensor([0])]; tensor concat_421_values3_0 = const()[name = string("concat_421_values3_0"), val = tensor([0])]; int32 concat_421_axis_0 = const()[name = string("concat_421_axis_0"), val = int32(0)]; bool concat_421_interleave_0 = const()[name = string("concat_421_interleave_0"), val = bool(false)]; tensor concat_421 = concat(axis = concat_421_axis_0, interleave = concat_421_interleave_0, values = (expand_dims_286, concat_421_values1_0, var_1746, concat_421_values3_0))[name = string("concat_421")]; tensor model_model_kv_cache_0_internal_tensor_assign_48_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_48_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_48_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_231 = transpose(perm = var_12075, x = var_12070)[name = string("transpose_41")]; tensor model_model_kv_cache_0_internal_tensor_assign_48_cast_fp16 = slice_update(begin = concat_420, begin_mask = model_model_kv_cache_0_internal_tensor_assign_48_begin_mask_0, end = concat_421, end_mask = model_model_kv_cache_0_internal_tensor_assign_48_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_48_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_48_stride_0, update = value_231, x = coreml_update_state_102)[name = string("model_model_kv_cache_0_internal_tensor_assign_48_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_48_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_215_write_state")]; tensor coreml_update_state_103 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_215")]; tensor var_12237_begin_0 = const()[name = string("op_12237_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_12237_end_0 = const()[name = string("op_12237_end_0"), val = tensor([24, 8, 1536, 128])]; tensor var_12237_end_mask_0 = const()[name = string("op_12237_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12237_cast_fp16 = slice_by_index(begin = var_12237_begin_0, end = var_12237_end_0, end_mask = var_12237_end_mask_0, x = coreml_update_state_103)[name = string("op_12237_cast_fp16")]; tensor key_cache_47_axes_0 = const()[name = string("key_cache_47_axes_0"), val = tensor([0])]; tensor key_cache_47_cast_fp16 = squeeze(axes = key_cache_47_axes_0, x = var_12237_cast_fp16)[name = string("key_cache_47_cast_fp16")]; tensor var_12244_begin_0 = const()[name = string("op_12244_begin_0"), val = tensor([51, 0, 0, 0])]; tensor var_12244_end_0 = const()[name = string("op_12244_end_0"), val = tensor([52, 8, 1536, 128])]; tensor var_12244_end_mask_0 = const()[name = string("op_12244_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12244_cast_fp16 = slice_by_index(begin = var_12244_begin_0, end = var_12244_end_0, end_mask = var_12244_end_mask_0, x = coreml_update_state_103)[name = string("op_12244_cast_fp16")]; tensor value_cache_47_axes_0 = const()[name = string("value_cache_47_axes_0"), val = tensor([0])]; tensor value_cache_47_cast_fp16 = squeeze(axes = value_cache_47_axes_0, x = var_12244_cast_fp16)[name = string("value_cache_47_cast_fp16")]; tensor var_12268_axes_0 = const()[name = string("op_12268_axes_0"), val = tensor([1])]; tensor var_12268_cast_fp16 = expand_dims(axes = var_12268_axes_0, x = key_cache_47_cast_fp16)[name = string("op_12268_cast_fp16")]; tensor var_12273 = const()[name = string("op_12273"), val = tensor([1, 2, 1, 1])]; tensor value_235_cast_fp16 = tile(reps = var_12273, x = var_12268_cast_fp16)[name = string("value_235_cast_fp16")]; tensor var_12279 = const()[name = string("op_12279"), val = tensor([1, 16, 1536, 128])]; tensor key_states_95_cast_fp16 = reshape(shape = var_12279, x = value_235_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor var_12282_axes_0 = const()[name = string("op_12282_axes_0"), val = tensor([1])]; tensor var_12282_cast_fp16 = expand_dims(axes = var_12282_axes_0, x = value_cache_47_cast_fp16)[name = string("op_12282_cast_fp16")]; tensor var_12287 = const()[name = string("op_12287"), val = tensor([1, 2, 1, 1])]; tensor value_239_cast_fp16 = tile(reps = var_12287, x = var_12282_cast_fp16)[name = string("value_239_cast_fp16")]; bool var_12308_transpose_x_0 = const()[name = string("op_12308_transpose_x_0"), val = bool(false)]; bool var_12308_transpose_y_0 = const()[name = string("op_12308_transpose_y_0"), val = bool(true)]; tensor var_12308 = matmul(transpose_x = var_12308_transpose_x_0, transpose_y = var_12308_transpose_y_0, x = query_47, y = key_states_95_cast_fp16)[name = string("op_12308")]; fp16 var_12309_to_fp16 = const()[name = string("op_12309_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_93_cast_fp16 = mul(x = var_12308, y = var_12309_to_fp16)[name = string("attention_93_cast_fp16")]; tensor attention_95_cast_fp16 = add(x = attention_93_cast_fp16, y = causal_mask)[name = string("attention_95_cast_fp16")]; int32 var_12318 = const()[name = string("op_12318"), val = int32(-1)]; tensor var_12320_cast_fp16 = softmax(axis = var_12318, x = attention_95_cast_fp16)[name = string("op_12320_cast_fp16")]; tensor concat_426 = const()[name = string("concat_426"), val = tensor([16, 64, 1536])]; tensor reshape_69_cast_fp16 = reshape(shape = concat_426, x = var_12320_cast_fp16)[name = string("reshape_69_cast_fp16")]; tensor concat_427 = const()[name = string("concat_427"), val = tensor([16, 1536, 128])]; tensor reshape_70_cast_fp16 = reshape(shape = concat_427, x = value_239_cast_fp16)[name = string("reshape_70_cast_fp16")]; bool matmul_23_transpose_x_0 = const()[name = string("matmul_23_transpose_x_0"), val = bool(false)]; bool matmul_23_transpose_y_0 = const()[name = string("matmul_23_transpose_y_0"), val = bool(false)]; tensor matmul_23_cast_fp16 = matmul(transpose_x = matmul_23_transpose_x_0, transpose_y = matmul_23_transpose_y_0, x = reshape_69_cast_fp16, y = reshape_70_cast_fp16)[name = string("matmul_23_cast_fp16")]; tensor concat_431 = const()[name = string("concat_431"), val = tensor([1, 16, 64, 128])]; tensor reshape_71_cast_fp16 = reshape(shape = concat_431, x = matmul_23_cast_fp16)[name = string("reshape_71_cast_fp16")]; tensor var_12332_perm_0 = const()[name = string("op_12332_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_12338 = const()[name = string("op_12338"), val = tensor([1, 64, 2048])]; tensor var_12332_cast_fp16 = transpose(perm = var_12332_perm_0, x = reshape_71_cast_fp16)[name = string("transpose_40")]; tensor output_141_cast_fp16 = reshape(shape = var_12338, x = var_12332_cast_fp16)[name = string("output_141_cast_fp16")]; tensor var_12343 = const()[name = string("op_12343"), val = tensor([0, 2, 1])]; string var_12359_pad_type_0 = const()[name = string("op_12359_pad_type_0"), val = string("valid")]; int32 var_12359_groups_0 = const()[name = string("op_12359_groups_0"), val = int32(1)]; tensor var_12359_strides_0 = const()[name = string("op_12359_strides_0"), val = tensor([1])]; tensor var_12359_pad_0 = const()[name = string("op_12359_pad_0"), val = tensor([0, 0])]; tensor var_12359_dilations_0 = const()[name = string("op_12359_dilations_0"), val = tensor([1])]; tensor squeeze_23_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(329572352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331145280))))[name = string("squeeze_23_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12344_cast_fp16 = transpose(perm = var_12343, x = output_141_cast_fp16)[name = string("transpose_39")]; tensor var_12359_cast_fp16 = conv(dilations = var_12359_dilations_0, groups = var_12359_groups_0, pad = var_12359_pad_0, pad_type = var_12359_pad_type_0, strides = var_12359_strides_0, weight = squeeze_23_cast_fp16_to_fp32_to_fp16_palettized, x = var_12344_cast_fp16)[name = string("op_12359_cast_fp16")]; tensor var_12363 = const()[name = string("op_12363"), val = tensor([0, 2, 1])]; tensor attn_output_47_cast_fp16 = transpose(perm = var_12363, x = var_12359_cast_fp16)[name = string("transpose_38")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_231_cast_fp16, y = attn_output_47_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; int32 var_12378 = const()[name = string("op_12378"), val = int32(-1)]; fp16 const_357_promoted_to_fp16 = const()[name = string("const_357_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12380_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_357_promoted_to_fp16)[name = string("op_12380_cast_fp16")]; bool input_425_interleave_0 = const()[name = string("input_425_interleave_0"), val = bool(false)]; tensor input_425_cast_fp16 = concat(axis = var_12378, interleave = input_425_interleave_0, values = (hidden_states_239_cast_fp16, var_12380_cast_fp16))[name = string("input_425_cast_fp16")]; tensor normed_381_axes_0 = const()[name = string("normed_381_axes_0"), val = tensor([-1])]; fp16 var_12375_to_fp16 = const()[name = string("op_12375_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_381_cast_fp16 = layer_norm(axes = normed_381_axes_0, epsilon = var_12375_to_fp16, x = input_425_cast_fp16)[name = string("normed_381_cast_fp16")]; tensor normed_383_begin_0 = const()[name = string("normed_383_begin_0"), val = tensor([0, 0, 0])]; tensor normed_383_end_0 = const()[name = string("normed_383_end_0"), val = tensor([1, 64, 1024])]; tensor normed_383_end_mask_0 = const()[name = string("normed_383_end_mask_0"), val = tensor([true, true, false])]; tensor normed_383_cast_fp16 = slice_by_index(begin = normed_383_begin_0, end = normed_383_end_0, end_mask = normed_383_end_mask_0, x = normed_381_cast_fp16)[name = string("normed_383_cast_fp16")]; tensor const_359_promoted_to_fp16 = const()[name = string("const_359_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331161728)))]; tensor x_93_cast_fp16 = mul(x = normed_383_cast_fp16, y = const_359_promoted_to_fp16)[name = string("x_93_cast_fp16")]; tensor var_12400 = const()[name = string("op_12400"), val = tensor([0, 2, 1])]; tensor input_427_axes_0 = const()[name = string("input_427_axes_0"), val = tensor([2])]; tensor var_12401 = transpose(perm = var_12400, x = x_93_cast_fp16)[name = string("transpose_37")]; tensor input_427 = expand_dims(axes = input_427_axes_0, x = var_12401)[name = string("input_427")]; string input_429_pad_type_0 = const()[name = string("input_429_pad_type_0"), val = string("valid")]; tensor input_429_strides_0 = const()[name = string("input_429_strides_0"), val = tensor([1, 1])]; tensor input_429_pad_0 = const()[name = string("input_429_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_429_dilations_0 = const()[name = string("input_429_dilations_0"), val = tensor([1, 1])]; int32 input_429_groups_0 = const()[name = string("input_429_groups_0"), val = int32(1)]; tensor input_429 = conv(dilations = input_429_dilations_0, groups = input_429_groups_0, pad = input_429_pad_0, pad_type = input_429_pad_type_0, strides = input_429_strides_0, weight = model_model_layers_23_mlp_gate_proj_weight_palettized, x = input_427)[name = string("input_429")]; string b_47_pad_type_0 = const()[name = string("b_47_pad_type_0"), val = string("valid")]; tensor b_47_strides_0 = const()[name = string("b_47_strides_0"), val = tensor([1, 1])]; tensor b_47_pad_0 = const()[name = string("b_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_47_dilations_0 = const()[name = string("b_47_dilations_0"), val = tensor([1, 1])]; int32 b_47_groups_0 = const()[name = string("b_47_groups_0"), val = int32(1)]; tensor b_47 = conv(dilations = b_47_dilations_0, groups = b_47_groups_0, pad = b_47_pad_0, pad_type = b_47_pad_type_0, strides = b_47_strides_0, weight = model_model_layers_23_mlp_up_proj_weight_palettized, x = input_427)[name = string("b_47")]; tensor c_47 = silu(x = input_429)[name = string("c_47")]; tensor input_431 = mul(x = c_47, y = b_47)[name = string("input_431")]; string e_47_pad_type_0 = const()[name = string("e_47_pad_type_0"), val = string("valid")]; tensor e_47_strides_0 = const()[name = string("e_47_strides_0"), val = tensor([1, 1])]; tensor e_47_pad_0 = const()[name = string("e_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_47_dilations_0 = const()[name = string("e_47_dilations_0"), val = tensor([1, 1])]; int32 e_47_groups_0 = const()[name = string("e_47_groups_0"), val = int32(1)]; tensor e_47 = conv(dilations = e_47_dilations_0, groups = e_47_groups_0, pad = e_47_pad_0, pad_type = e_47_pad_type_0, strides = e_47_strides_0, weight = model_model_layers_23_mlp_down_proj_weight_palettized, x = input_431)[name = string("e_47")]; tensor var_12423_axes_0 = const()[name = string("op_12423_axes_0"), val = tensor([2])]; tensor var_12423 = squeeze(axes = var_12423_axes_0, x = e_47)[name = string("op_12423")]; tensor var_12424 = const()[name = string("op_12424"), val = tensor([0, 2, 1])]; tensor var_12425 = transpose(perm = var_12424, x = var_12423)[name = string("transpose_36")]; tensor hidden_states_241_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = var_12425)[name = string("hidden_states_241_cast_fp16")]; int32 var_12439 = const()[name = string("op_12439"), val = int32(-1)]; fp16 const_360_promoted_to_fp16 = const()[name = string("const_360_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12441_cast_fp16 = mul(x = hidden_states_241_cast_fp16, y = const_360_promoted_to_fp16)[name = string("op_12441_cast_fp16")]; bool input_433_interleave_0 = const()[name = string("input_433_interleave_0"), val = bool(false)]; tensor input_433_cast_fp16 = concat(axis = var_12439, interleave = input_433_interleave_0, values = (hidden_states_241_cast_fp16, var_12441_cast_fp16))[name = string("input_433_cast_fp16")]; tensor normed_385_axes_0 = const()[name = string("normed_385_axes_0"), val = tensor([-1])]; fp16 var_12436_to_fp16 = const()[name = string("op_12436_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_385_cast_fp16 = layer_norm(axes = normed_385_axes_0, epsilon = var_12436_to_fp16, x = input_433_cast_fp16)[name = string("normed_385_cast_fp16")]; tensor normed_387_begin_0 = const()[name = string("normed_387_begin_0"), val = tensor([0, 0, 0])]; tensor normed_387_end_0 = const()[name = string("normed_387_end_0"), val = tensor([1, 64, 1024])]; tensor normed_387_end_mask_0 = const()[name = string("normed_387_end_mask_0"), val = tensor([true, true, false])]; tensor normed_387_cast_fp16 = slice_by_index(begin = normed_387_begin_0, end = normed_387_end_0, end_mask = normed_387_end_mask_0, x = normed_385_cast_fp16)[name = string("normed_387_cast_fp16")]; tensor const_362_promoted_to_fp16 = const()[name = string("const_362_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331163840)))]; tensor hidden_states_243_cast_fp16 = mul(x = normed_387_cast_fp16, y = const_362_promoted_to_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor var_12453 = const()[name = string("op_12453"), val = tensor([0, 2, 1])]; tensor var_12456_axes_0 = const()[name = string("op_12456_axes_0"), val = tensor([2])]; tensor var_12454_cast_fp16 = transpose(perm = var_12453, x = hidden_states_243_cast_fp16)[name = string("transpose_35")]; tensor var_12456_cast_fp16 = expand_dims(axes = var_12456_axes_0, x = var_12454_cast_fp16)[name = string("op_12456_cast_fp16")]; string var_12472_pad_type_0 = const()[name = string("op_12472_pad_type_0"), val = string("valid")]; tensor var_12472_strides_0 = const()[name = string("op_12472_strides_0"), val = tensor([1, 1])]; tensor var_12472_pad_0 = const()[name = string("op_12472_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12472_dilations_0 = const()[name = string("op_12472_dilations_0"), val = tensor([1, 1])]; int32 var_12472_groups_0 = const()[name = string("op_12472_groups_0"), val = int32(1)]; tensor var_12472 = conv(dilations = var_12472_dilations_0, groups = var_12472_groups_0, pad = var_12472_pad_0, pad_type = var_12472_pad_type_0, strides = var_12472_strides_0, weight = model_model_layers_24_self_attn_q_proj_weight_palettized, x = var_12456_cast_fp16)[name = string("op_12472")]; tensor var_12477 = const()[name = string("op_12477"), val = tensor([1, 16, 128, 64])]; tensor var_12478 = reshape(shape = var_12477, x = var_12472)[name = string("op_12478")]; tensor var_12483 = const()[name = string("op_12483"), val = tensor([0, 1, 3, 2])]; string var_12495_pad_type_0 = const()[name = string("op_12495_pad_type_0"), val = string("valid")]; tensor var_12495_strides_0 = const()[name = string("op_12495_strides_0"), val = tensor([1, 1])]; tensor var_12495_pad_0 = const()[name = string("op_12495_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12495_dilations_0 = const()[name = string("op_12495_dilations_0"), val = tensor([1, 1])]; int32 var_12495_groups_0 = const()[name = string("op_12495_groups_0"), val = int32(1)]; tensor var_12495 = conv(dilations = var_12495_dilations_0, groups = var_12495_groups_0, pad = var_12495_pad_0, pad_type = var_12495_pad_type_0, strides = var_12495_strides_0, weight = model_model_layers_24_self_attn_k_proj_weight_palettized, x = var_12456_cast_fp16)[name = string("op_12495")]; tensor var_12500 = const()[name = string("op_12500"), val = tensor([1, 8, 128, 64])]; tensor var_12501 = reshape(shape = var_12500, x = var_12495)[name = string("op_12501")]; tensor var_12506 = const()[name = string("op_12506"), val = tensor([0, 1, 3, 2])]; string var_12518_pad_type_0 = const()[name = string("op_12518_pad_type_0"), val = string("valid")]; tensor var_12518_strides_0 = const()[name = string("op_12518_strides_0"), val = tensor([1, 1])]; tensor var_12518_pad_0 = const()[name = string("op_12518_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12518_dilations_0 = const()[name = string("op_12518_dilations_0"), val = tensor([1, 1])]; int32 var_12518_groups_0 = const()[name = string("op_12518_groups_0"), val = int32(1)]; tensor var_12518 = conv(dilations = var_12518_dilations_0, groups = var_12518_groups_0, pad = var_12518_pad_0, pad_type = var_12518_pad_type_0, strides = var_12518_strides_0, weight = model_model_layers_24_self_attn_v_proj_weight_palettized, x = var_12456_cast_fp16)[name = string("op_12518")]; tensor var_12523 = const()[name = string("op_12523"), val = tensor([1, 8, 128, 64])]; tensor var_12524 = reshape(shape = var_12523, x = var_12518)[name = string("op_12524")]; tensor var_12529 = const()[name = string("op_12529"), val = tensor([0, 1, 3, 2])]; int32 var_12542 = const()[name = string("op_12542"), val = int32(-1)]; fp16 const_363_promoted = const()[name = string("const_363_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_245 = transpose(perm = var_12483, x = var_12478)[name = string("transpose_34")]; tensor var_12544 = mul(x = hidden_states_245, y = const_363_promoted)[name = string("op_12544")]; bool input_437_interleave_0 = const()[name = string("input_437_interleave_0"), val = bool(false)]; tensor input_437 = concat(axis = var_12542, interleave = input_437_interleave_0, values = (hidden_states_245, var_12544))[name = string("input_437")]; tensor normed_389_axes_0 = const()[name = string("normed_389_axes_0"), val = tensor([-1])]; fp16 var_12539_to_fp16 = const()[name = string("op_12539_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_389_cast_fp16 = layer_norm(axes = normed_389_axes_0, epsilon = var_12539_to_fp16, x = input_437)[name = string("normed_389_cast_fp16")]; tensor normed_391_begin_0 = const()[name = string("normed_391_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_391_end_0 = const()[name = string("normed_391_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_391_end_mask_0 = const()[name = string("normed_391_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_391 = slice_by_index(begin = normed_391_begin_0, end = normed_391_end_0, end_mask = normed_391_end_mask_0, x = normed_389_cast_fp16)[name = string("normed_391")]; tensor const_365 = const()[name = string("const_365"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331165952)))]; tensor q_49 = mul(x = normed_391, y = const_365)[name = string("q_49")]; int32 var_12564 = const()[name = string("op_12564"), val = int32(-1)]; fp16 const_366_promoted = const()[name = string("const_366_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_247 = transpose(perm = var_12506, x = var_12501)[name = string("transpose_33")]; tensor var_12566 = mul(x = hidden_states_247, y = const_366_promoted)[name = string("op_12566")]; bool input_439_interleave_0 = const()[name = string("input_439_interleave_0"), val = bool(false)]; tensor input_439 = concat(axis = var_12564, interleave = input_439_interleave_0, values = (hidden_states_247, var_12566))[name = string("input_439")]; tensor normed_393_axes_0 = const()[name = string("normed_393_axes_0"), val = tensor([-1])]; fp16 var_12561_to_fp16 = const()[name = string("op_12561_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_393_cast_fp16 = layer_norm(axes = normed_393_axes_0, epsilon = var_12561_to_fp16, x = input_439)[name = string("normed_393_cast_fp16")]; tensor normed_395_begin_0 = const()[name = string("normed_395_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_395_end_0 = const()[name = string("normed_395_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_395_end_mask_0 = const()[name = string("normed_395_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_395 = slice_by_index(begin = normed_395_begin_0, end = normed_395_end_0, end_mask = normed_395_end_mask_0, x = normed_393_cast_fp16)[name = string("normed_395")]; tensor const_368 = const()[name = string("const_368"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331166272)))]; tensor k_49 = mul(x = normed_395, y = const_368)[name = string("k_49")]; tensor var_12587 = mul(x = q_49, y = cos_1)[name = string("op_12587")]; tensor var_12592_begin_0 = const()[name = string("op_12592_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12592_end_0 = const()[name = string("op_12592_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_12592_end_mask_0 = const()[name = string("op_12592_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12592 = slice_by_index(begin = var_12592_begin_0, end = var_12592_end_0, end_mask = var_12592_end_mask_0, x = q_49)[name = string("op_12592")]; fp16 const_369_promoted = const()[name = string("const_369_promoted"), val = fp16(-0x1p+0)]; tensor var_12593 = mul(x = var_12592, y = const_369_promoted)[name = string("op_12593")]; tensor var_12598_begin_0 = const()[name = string("op_12598_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12598_end_0 = const()[name = string("op_12598_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_12598_end_mask_0 = const()[name = string("op_12598_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12598 = slice_by_index(begin = var_12598_begin_0, end = var_12598_end_0, end_mask = var_12598_end_mask_0, x = q_49)[name = string("op_12598")]; int32 var_12600 = const()[name = string("op_12600"), val = int32(-1)]; bool var_12601_interleave_0 = const()[name = string("op_12601_interleave_0"), val = bool(false)]; tensor var_12601 = concat(axis = var_12600, interleave = var_12601_interleave_0, values = (var_12593, var_12598))[name = string("op_12601")]; tensor var_12602 = mul(x = var_12601, y = sin_1)[name = string("op_12602")]; tensor query_49 = add(x = var_12587, y = var_12602)[name = string("query_49")]; tensor var_12605 = mul(x = k_49, y = cos_1)[name = string("op_12605")]; tensor var_12610_begin_0 = const()[name = string("op_12610_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_12610_end_0 = const()[name = string("op_12610_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_12610_end_mask_0 = const()[name = string("op_12610_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_12610 = slice_by_index(begin = var_12610_begin_0, end = var_12610_end_0, end_mask = var_12610_end_mask_0, x = k_49)[name = string("op_12610")]; fp16 const_370_promoted = const()[name = string("const_370_promoted"), val = fp16(-0x1p+0)]; tensor var_12611 = mul(x = var_12610, y = const_370_promoted)[name = string("op_12611")]; tensor var_12616_begin_0 = const()[name = string("op_12616_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_12616_end_0 = const()[name = string("op_12616_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_12616_end_mask_0 = const()[name = string("op_12616_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_12616 = slice_by_index(begin = var_12616_begin_0, end = var_12616_end_0, end_mask = var_12616_end_mask_0, x = k_49)[name = string("op_12616")]; int32 var_12618 = const()[name = string("op_12618"), val = int32(-1)]; bool var_12619_interleave_0 = const()[name = string("op_12619_interleave_0"), val = bool(false)]; tensor var_12619 = concat(axis = var_12618, interleave = var_12619_interleave_0, values = (var_12611, var_12616))[name = string("op_12619")]; tensor var_12620 = mul(x = var_12619, y = sin_1)[name = string("op_12620")]; tensor key_49 = add(x = var_12605, y = var_12620)[name = string("key_49")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; int32 concat_434_axis_0 = const()[name = string("concat_434_axis_0"), val = int32(0)]; bool concat_434_interleave_0 = const()[name = string("concat_434_interleave_0"), val = bool(false)]; tensor concat_434 = concat(axis = concat_434_axis_0, interleave = concat_434_interleave_0, values = (expand_dims_288, expand_dims_289, current_pos, expand_dims_291))[name = string("concat_434")]; tensor concat_435_values1_0 = const()[name = string("concat_435_values1_0"), val = tensor([0])]; tensor concat_435_values3_0 = const()[name = string("concat_435_values3_0"), val = tensor([0])]; int32 concat_435_axis_0 = const()[name = string("concat_435_axis_0"), val = int32(0)]; bool concat_435_interleave_0 = const()[name = string("concat_435_interleave_0"), val = bool(false)]; tensor concat_435 = concat(axis = concat_435_axis_0, interleave = concat_435_interleave_0, values = (expand_dims_292, concat_435_values1_0, var_1746, concat_435_values3_0))[name = string("concat_435")]; tensor model_model_kv_cache_0_internal_tensor_assign_49_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_49_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_49_cast_fp16 = slice_update(begin = concat_434, begin_mask = model_model_kv_cache_0_internal_tensor_assign_49_begin_mask_0, end = concat_435, end_mask = model_model_kv_cache_0_internal_tensor_assign_49_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_49_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_49_stride_0, update = key_49, x = coreml_update_state_103)[name = string("model_model_kv_cache_0_internal_tensor_assign_49_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_49_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_216_write_state")]; tensor coreml_update_state_104 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_216")]; tensor expand_dims_294 = const()[name = string("expand_dims_294"), val = tensor([52])]; tensor expand_dims_295 = const()[name = string("expand_dims_295"), val = tensor([0])]; tensor expand_dims_297 = const()[name = string("expand_dims_297"), val = tensor([0])]; tensor expand_dims_298 = const()[name = string("expand_dims_298"), val = tensor([53])]; int32 concat_438_axis_0 = const()[name = string("concat_438_axis_0"), val = int32(0)]; bool concat_438_interleave_0 = const()[name = string("concat_438_interleave_0"), val = bool(false)]; tensor concat_438 = concat(axis = concat_438_axis_0, interleave = concat_438_interleave_0, values = (expand_dims_294, expand_dims_295, current_pos, expand_dims_297))[name = string("concat_438")]; tensor concat_439_values1_0 = const()[name = string("concat_439_values1_0"), val = tensor([0])]; tensor concat_439_values3_0 = const()[name = string("concat_439_values3_0"), val = tensor([0])]; int32 concat_439_axis_0 = const()[name = string("concat_439_axis_0"), val = int32(0)]; bool concat_439_interleave_0 = const()[name = string("concat_439_interleave_0"), val = bool(false)]; tensor concat_439 = concat(axis = concat_439_axis_0, interleave = concat_439_interleave_0, values = (expand_dims_298, concat_439_values1_0, var_1746, concat_439_values3_0))[name = string("concat_439")]; tensor model_model_kv_cache_0_internal_tensor_assign_50_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_50_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_50_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_241 = transpose(perm = var_12529, x = var_12524)[name = string("transpose_32")]; tensor model_model_kv_cache_0_internal_tensor_assign_50_cast_fp16 = slice_update(begin = concat_438, begin_mask = model_model_kv_cache_0_internal_tensor_assign_50_begin_mask_0, end = concat_439, end_mask = model_model_kv_cache_0_internal_tensor_assign_50_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_50_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_50_stride_0, update = value_241, x = coreml_update_state_104)[name = string("model_model_kv_cache_0_internal_tensor_assign_50_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_50_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_217_write_state")]; tensor coreml_update_state_105 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_217")]; tensor var_12691_begin_0 = const()[name = string("op_12691_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_12691_end_0 = const()[name = string("op_12691_end_0"), val = tensor([25, 8, 1536, 128])]; tensor var_12691_end_mask_0 = const()[name = string("op_12691_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12691_cast_fp16 = slice_by_index(begin = var_12691_begin_0, end = var_12691_end_0, end_mask = var_12691_end_mask_0, x = coreml_update_state_105)[name = string("op_12691_cast_fp16")]; tensor key_cache_49_axes_0 = const()[name = string("key_cache_49_axes_0"), val = tensor([0])]; tensor key_cache_49_cast_fp16 = squeeze(axes = key_cache_49_axes_0, x = var_12691_cast_fp16)[name = string("key_cache_49_cast_fp16")]; tensor var_12698_begin_0 = const()[name = string("op_12698_begin_0"), val = tensor([52, 0, 0, 0])]; tensor var_12698_end_0 = const()[name = string("op_12698_end_0"), val = tensor([53, 8, 1536, 128])]; tensor var_12698_end_mask_0 = const()[name = string("op_12698_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_12698_cast_fp16 = slice_by_index(begin = var_12698_begin_0, end = var_12698_end_0, end_mask = var_12698_end_mask_0, x = coreml_update_state_105)[name = string("op_12698_cast_fp16")]; tensor value_cache_49_axes_0 = const()[name = string("value_cache_49_axes_0"), val = tensor([0])]; tensor value_cache_49_cast_fp16 = squeeze(axes = value_cache_49_axes_0, x = var_12698_cast_fp16)[name = string("value_cache_49_cast_fp16")]; tensor var_12722_axes_0 = const()[name = string("op_12722_axes_0"), val = tensor([1])]; tensor var_12722_cast_fp16 = expand_dims(axes = var_12722_axes_0, x = key_cache_49_cast_fp16)[name = string("op_12722_cast_fp16")]; tensor var_12727 = const()[name = string("op_12727"), val = tensor([1, 2, 1, 1])]; tensor value_245_cast_fp16 = tile(reps = var_12727, x = var_12722_cast_fp16)[name = string("value_245_cast_fp16")]; tensor var_12733 = const()[name = string("op_12733"), val = tensor([1, 16, 1536, 128])]; tensor key_states_99_cast_fp16 = reshape(shape = var_12733, x = value_245_cast_fp16)[name = string("key_states_99_cast_fp16")]; tensor var_12736_axes_0 = const()[name = string("op_12736_axes_0"), val = tensor([1])]; tensor var_12736_cast_fp16 = expand_dims(axes = var_12736_axes_0, x = value_cache_49_cast_fp16)[name = string("op_12736_cast_fp16")]; tensor var_12741 = const()[name = string("op_12741"), val = tensor([1, 2, 1, 1])]; tensor value_249_cast_fp16 = tile(reps = var_12741, x = var_12736_cast_fp16)[name = string("value_249_cast_fp16")]; bool var_12762_transpose_x_0 = const()[name = string("op_12762_transpose_x_0"), val = bool(false)]; bool var_12762_transpose_y_0 = const()[name = string("op_12762_transpose_y_0"), val = bool(true)]; tensor var_12762 = matmul(transpose_x = var_12762_transpose_x_0, transpose_y = var_12762_transpose_y_0, x = query_49, y = key_states_99_cast_fp16)[name = string("op_12762")]; fp16 var_12763_to_fp16 = const()[name = string("op_12763_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_97_cast_fp16 = mul(x = var_12762, y = var_12763_to_fp16)[name = string("attention_97_cast_fp16")]; tensor attention_99_cast_fp16 = add(x = attention_97_cast_fp16, y = causal_mask)[name = string("attention_99_cast_fp16")]; int32 var_12772 = const()[name = string("op_12772"), val = int32(-1)]; tensor var_12774_cast_fp16 = softmax(axis = var_12772, x = attention_99_cast_fp16)[name = string("op_12774_cast_fp16")]; tensor concat_444 = const()[name = string("concat_444"), val = tensor([16, 64, 1536])]; tensor reshape_72_cast_fp16 = reshape(shape = concat_444, x = var_12774_cast_fp16)[name = string("reshape_72_cast_fp16")]; tensor concat_445 = const()[name = string("concat_445"), val = tensor([16, 1536, 128])]; tensor reshape_73_cast_fp16 = reshape(shape = concat_445, x = value_249_cast_fp16)[name = string("reshape_73_cast_fp16")]; bool matmul_24_transpose_x_0 = const()[name = string("matmul_24_transpose_x_0"), val = bool(false)]; bool matmul_24_transpose_y_0 = const()[name = string("matmul_24_transpose_y_0"), val = bool(false)]; tensor matmul_24_cast_fp16 = matmul(transpose_x = matmul_24_transpose_x_0, transpose_y = matmul_24_transpose_y_0, x = reshape_72_cast_fp16, y = reshape_73_cast_fp16)[name = string("matmul_24_cast_fp16")]; tensor concat_449 = const()[name = string("concat_449"), val = tensor([1, 16, 64, 128])]; tensor reshape_74_cast_fp16 = reshape(shape = concat_449, x = matmul_24_cast_fp16)[name = string("reshape_74_cast_fp16")]; tensor var_12786_perm_0 = const()[name = string("op_12786_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_12792 = const()[name = string("op_12792"), val = tensor([1, 64, 2048])]; tensor var_12786_cast_fp16 = transpose(perm = var_12786_perm_0, x = reshape_74_cast_fp16)[name = string("transpose_31")]; tensor output_147_cast_fp16 = reshape(shape = var_12792, x = var_12786_cast_fp16)[name = string("output_147_cast_fp16")]; tensor var_12797 = const()[name = string("op_12797"), val = tensor([0, 2, 1])]; string var_12813_pad_type_0 = const()[name = string("op_12813_pad_type_0"), val = string("valid")]; int32 var_12813_groups_0 = const()[name = string("op_12813_groups_0"), val = int32(1)]; tensor var_12813_strides_0 = const()[name = string("op_12813_strides_0"), val = tensor([1])]; tensor var_12813_pad_0 = const()[name = string("op_12813_pad_0"), val = tensor([0, 0])]; tensor var_12813_dilations_0 = const()[name = string("op_12813_dilations_0"), val = tensor([1])]; tensor squeeze_24_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331166592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332739520))))[name = string("squeeze_24_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_12798_cast_fp16 = transpose(perm = var_12797, x = output_147_cast_fp16)[name = string("transpose_30")]; tensor var_12813_cast_fp16 = conv(dilations = var_12813_dilations_0, groups = var_12813_groups_0, pad = var_12813_pad_0, pad_type = var_12813_pad_type_0, strides = var_12813_strides_0, weight = squeeze_24_cast_fp16_to_fp32_to_fp16_palettized, x = var_12798_cast_fp16)[name = string("op_12813_cast_fp16")]; tensor var_12817 = const()[name = string("op_12817"), val = tensor([0, 2, 1])]; tensor attn_output_49_cast_fp16 = transpose(perm = var_12817, x = var_12813_cast_fp16)[name = string("transpose_29")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_241_cast_fp16, y = attn_output_49_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; int32 var_12832 = const()[name = string("op_12832"), val = int32(-1)]; fp16 const_372_promoted_to_fp16 = const()[name = string("const_372_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12834_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_372_promoted_to_fp16)[name = string("op_12834_cast_fp16")]; bool input_443_interleave_0 = const()[name = string("input_443_interleave_0"), val = bool(false)]; tensor input_443_cast_fp16 = concat(axis = var_12832, interleave = input_443_interleave_0, values = (hidden_states_249_cast_fp16, var_12834_cast_fp16))[name = string("input_443_cast_fp16")]; tensor normed_397_axes_0 = const()[name = string("normed_397_axes_0"), val = tensor([-1])]; fp16 var_12829_to_fp16 = const()[name = string("op_12829_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_397_cast_fp16 = layer_norm(axes = normed_397_axes_0, epsilon = var_12829_to_fp16, x = input_443_cast_fp16)[name = string("normed_397_cast_fp16")]; tensor normed_399_begin_0 = const()[name = string("normed_399_begin_0"), val = tensor([0, 0, 0])]; tensor normed_399_end_0 = const()[name = string("normed_399_end_0"), val = tensor([1, 64, 1024])]; tensor normed_399_end_mask_0 = const()[name = string("normed_399_end_mask_0"), val = tensor([true, true, false])]; tensor normed_399_cast_fp16 = slice_by_index(begin = normed_399_begin_0, end = normed_399_end_0, end_mask = normed_399_end_mask_0, x = normed_397_cast_fp16)[name = string("normed_399_cast_fp16")]; tensor const_374_promoted_to_fp16 = const()[name = string("const_374_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332755968)))]; tensor x_97_cast_fp16 = mul(x = normed_399_cast_fp16, y = const_374_promoted_to_fp16)[name = string("x_97_cast_fp16")]; tensor var_12854 = const()[name = string("op_12854"), val = tensor([0, 2, 1])]; tensor input_445_axes_0 = const()[name = string("input_445_axes_0"), val = tensor([2])]; tensor var_12855 = transpose(perm = var_12854, x = x_97_cast_fp16)[name = string("transpose_28")]; tensor input_445 = expand_dims(axes = input_445_axes_0, x = var_12855)[name = string("input_445")]; string input_447_pad_type_0 = const()[name = string("input_447_pad_type_0"), val = string("valid")]; tensor input_447_strides_0 = const()[name = string("input_447_strides_0"), val = tensor([1, 1])]; tensor input_447_pad_0 = const()[name = string("input_447_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_447_dilations_0 = const()[name = string("input_447_dilations_0"), val = tensor([1, 1])]; int32 input_447_groups_0 = const()[name = string("input_447_groups_0"), val = int32(1)]; tensor input_447 = conv(dilations = input_447_dilations_0, groups = input_447_groups_0, pad = input_447_pad_0, pad_type = input_447_pad_type_0, strides = input_447_strides_0, weight = model_model_layers_24_mlp_gate_proj_weight_palettized, x = input_445)[name = string("input_447")]; string b_49_pad_type_0 = const()[name = string("b_49_pad_type_0"), val = string("valid")]; tensor b_49_strides_0 = const()[name = string("b_49_strides_0"), val = tensor([1, 1])]; tensor b_49_pad_0 = const()[name = string("b_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_49_dilations_0 = const()[name = string("b_49_dilations_0"), val = tensor([1, 1])]; int32 b_49_groups_0 = const()[name = string("b_49_groups_0"), val = int32(1)]; tensor b_49 = conv(dilations = b_49_dilations_0, groups = b_49_groups_0, pad = b_49_pad_0, pad_type = b_49_pad_type_0, strides = b_49_strides_0, weight = model_model_layers_24_mlp_up_proj_weight_palettized, x = input_445)[name = string("b_49")]; tensor c_49 = silu(x = input_447)[name = string("c_49")]; tensor input_449 = mul(x = c_49, y = b_49)[name = string("input_449")]; string e_49_pad_type_0 = const()[name = string("e_49_pad_type_0"), val = string("valid")]; tensor e_49_strides_0 = const()[name = string("e_49_strides_0"), val = tensor([1, 1])]; tensor e_49_pad_0 = const()[name = string("e_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_49_dilations_0 = const()[name = string("e_49_dilations_0"), val = tensor([1, 1])]; int32 e_49_groups_0 = const()[name = string("e_49_groups_0"), val = int32(1)]; tensor e_49 = conv(dilations = e_49_dilations_0, groups = e_49_groups_0, pad = e_49_pad_0, pad_type = e_49_pad_type_0, strides = e_49_strides_0, weight = model_model_layers_24_mlp_down_proj_weight_palettized, x = input_449)[name = string("e_49")]; tensor var_12877_axes_0 = const()[name = string("op_12877_axes_0"), val = tensor([2])]; tensor var_12877 = squeeze(axes = var_12877_axes_0, x = e_49)[name = string("op_12877")]; tensor var_12878 = const()[name = string("op_12878"), val = tensor([0, 2, 1])]; tensor var_12879 = transpose(perm = var_12878, x = var_12877)[name = string("transpose_27")]; tensor hidden_states_251_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = var_12879)[name = string("hidden_states_251_cast_fp16")]; int32 var_12893 = const()[name = string("op_12893"), val = int32(-1)]; fp16 const_375_promoted_to_fp16 = const()[name = string("const_375_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_12895_cast_fp16 = mul(x = hidden_states_251_cast_fp16, y = const_375_promoted_to_fp16)[name = string("op_12895_cast_fp16")]; bool input_451_interleave_0 = const()[name = string("input_451_interleave_0"), val = bool(false)]; tensor input_451_cast_fp16 = concat(axis = var_12893, interleave = input_451_interleave_0, values = (hidden_states_251_cast_fp16, var_12895_cast_fp16))[name = string("input_451_cast_fp16")]; tensor normed_401_axes_0 = const()[name = string("normed_401_axes_0"), val = tensor([-1])]; fp16 var_12890_to_fp16 = const()[name = string("op_12890_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_401_cast_fp16 = layer_norm(axes = normed_401_axes_0, epsilon = var_12890_to_fp16, x = input_451_cast_fp16)[name = string("normed_401_cast_fp16")]; tensor normed_403_begin_0 = const()[name = string("normed_403_begin_0"), val = tensor([0, 0, 0])]; tensor normed_403_end_0 = const()[name = string("normed_403_end_0"), val = tensor([1, 64, 1024])]; tensor normed_403_end_mask_0 = const()[name = string("normed_403_end_mask_0"), val = tensor([true, true, false])]; tensor normed_403_cast_fp16 = slice_by_index(begin = normed_403_begin_0, end = normed_403_end_0, end_mask = normed_403_end_mask_0, x = normed_401_cast_fp16)[name = string("normed_403_cast_fp16")]; tensor const_377_promoted_to_fp16 = const()[name = string("const_377_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332758080)))]; tensor hidden_states_253_cast_fp16 = mul(x = normed_403_cast_fp16, y = const_377_promoted_to_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor var_12907 = const()[name = string("op_12907"), val = tensor([0, 2, 1])]; tensor var_12910_axes_0 = const()[name = string("op_12910_axes_0"), val = tensor([2])]; tensor var_12908_cast_fp16 = transpose(perm = var_12907, x = hidden_states_253_cast_fp16)[name = string("transpose_26")]; tensor var_12910_cast_fp16 = expand_dims(axes = var_12910_axes_0, x = var_12908_cast_fp16)[name = string("op_12910_cast_fp16")]; string var_12926_pad_type_0 = const()[name = string("op_12926_pad_type_0"), val = string("valid")]; tensor var_12926_strides_0 = const()[name = string("op_12926_strides_0"), val = tensor([1, 1])]; tensor var_12926_pad_0 = const()[name = string("op_12926_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12926_dilations_0 = const()[name = string("op_12926_dilations_0"), val = tensor([1, 1])]; int32 var_12926_groups_0 = const()[name = string("op_12926_groups_0"), val = int32(1)]; tensor var_12926 = conv(dilations = var_12926_dilations_0, groups = var_12926_groups_0, pad = var_12926_pad_0, pad_type = var_12926_pad_type_0, strides = var_12926_strides_0, weight = model_model_layers_25_self_attn_q_proj_weight_palettized, x = var_12910_cast_fp16)[name = string("op_12926")]; tensor var_12931 = const()[name = string("op_12931"), val = tensor([1, 16, 128, 64])]; tensor var_12932 = reshape(shape = var_12931, x = var_12926)[name = string("op_12932")]; tensor var_12937 = const()[name = string("op_12937"), val = tensor([0, 1, 3, 2])]; string var_12949_pad_type_0 = const()[name = string("op_12949_pad_type_0"), val = string("valid")]; tensor var_12949_strides_0 = const()[name = string("op_12949_strides_0"), val = tensor([1, 1])]; tensor var_12949_pad_0 = const()[name = string("op_12949_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12949_dilations_0 = const()[name = string("op_12949_dilations_0"), val = tensor([1, 1])]; int32 var_12949_groups_0 = const()[name = string("op_12949_groups_0"), val = int32(1)]; tensor var_12949 = conv(dilations = var_12949_dilations_0, groups = var_12949_groups_0, pad = var_12949_pad_0, pad_type = var_12949_pad_type_0, strides = var_12949_strides_0, weight = model_model_layers_25_self_attn_k_proj_weight_palettized, x = var_12910_cast_fp16)[name = string("op_12949")]; tensor var_12954 = const()[name = string("op_12954"), val = tensor([1, 8, 128, 64])]; tensor var_12955 = reshape(shape = var_12954, x = var_12949)[name = string("op_12955")]; tensor var_12960 = const()[name = string("op_12960"), val = tensor([0, 1, 3, 2])]; string var_12972_pad_type_0 = const()[name = string("op_12972_pad_type_0"), val = string("valid")]; tensor var_12972_strides_0 = const()[name = string("op_12972_strides_0"), val = tensor([1, 1])]; tensor var_12972_pad_0 = const()[name = string("op_12972_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_12972_dilations_0 = const()[name = string("op_12972_dilations_0"), val = tensor([1, 1])]; int32 var_12972_groups_0 = const()[name = string("op_12972_groups_0"), val = int32(1)]; tensor var_12972 = conv(dilations = var_12972_dilations_0, groups = var_12972_groups_0, pad = var_12972_pad_0, pad_type = var_12972_pad_type_0, strides = var_12972_strides_0, weight = model_model_layers_25_self_attn_v_proj_weight_palettized, x = var_12910_cast_fp16)[name = string("op_12972")]; tensor var_12977 = const()[name = string("op_12977"), val = tensor([1, 8, 128, 64])]; tensor var_12978 = reshape(shape = var_12977, x = var_12972)[name = string("op_12978")]; tensor var_12983 = const()[name = string("op_12983"), val = tensor([0, 1, 3, 2])]; int32 var_12996 = const()[name = string("op_12996"), val = int32(-1)]; fp16 const_378_promoted = const()[name = string("const_378_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_255 = transpose(perm = var_12937, x = var_12932)[name = string("transpose_25")]; tensor var_12998 = mul(x = hidden_states_255, y = const_378_promoted)[name = string("op_12998")]; bool input_455_interleave_0 = const()[name = string("input_455_interleave_0"), val = bool(false)]; tensor input_455 = concat(axis = var_12996, interleave = input_455_interleave_0, values = (hidden_states_255, var_12998))[name = string("input_455")]; tensor normed_405_axes_0 = const()[name = string("normed_405_axes_0"), val = tensor([-1])]; fp16 var_12993_to_fp16 = const()[name = string("op_12993_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_405_cast_fp16 = layer_norm(axes = normed_405_axes_0, epsilon = var_12993_to_fp16, x = input_455)[name = string("normed_405_cast_fp16")]; tensor normed_407_begin_0 = const()[name = string("normed_407_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_407_end_0 = const()[name = string("normed_407_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_407_end_mask_0 = const()[name = string("normed_407_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_407 = slice_by_index(begin = normed_407_begin_0, end = normed_407_end_0, end_mask = normed_407_end_mask_0, x = normed_405_cast_fp16)[name = string("normed_407")]; tensor const_380 = const()[name = string("const_380"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332760192)))]; tensor q_51 = mul(x = normed_407, y = const_380)[name = string("q_51")]; int32 var_13018 = const()[name = string("op_13018"), val = int32(-1)]; fp16 const_381_promoted = const()[name = string("const_381_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_257 = transpose(perm = var_12960, x = var_12955)[name = string("transpose_24")]; tensor var_13020 = mul(x = hidden_states_257, y = const_381_promoted)[name = string("op_13020")]; bool input_457_interleave_0 = const()[name = string("input_457_interleave_0"), val = bool(false)]; tensor input_457 = concat(axis = var_13018, interleave = input_457_interleave_0, values = (hidden_states_257, var_13020))[name = string("input_457")]; tensor normed_409_axes_0 = const()[name = string("normed_409_axes_0"), val = tensor([-1])]; fp16 var_13015_to_fp16 = const()[name = string("op_13015_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_409_cast_fp16 = layer_norm(axes = normed_409_axes_0, epsilon = var_13015_to_fp16, x = input_457)[name = string("normed_409_cast_fp16")]; tensor normed_411_begin_0 = const()[name = string("normed_411_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_411_end_0 = const()[name = string("normed_411_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_411_end_mask_0 = const()[name = string("normed_411_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_411 = slice_by_index(begin = normed_411_begin_0, end = normed_411_end_0, end_mask = normed_411_end_mask_0, x = normed_409_cast_fp16)[name = string("normed_411")]; tensor const_383 = const()[name = string("const_383"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332760512)))]; tensor k_51 = mul(x = normed_411, y = const_383)[name = string("k_51")]; tensor var_13041 = mul(x = q_51, y = cos_1)[name = string("op_13041")]; tensor var_13046_begin_0 = const()[name = string("op_13046_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13046_end_0 = const()[name = string("op_13046_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_13046_end_mask_0 = const()[name = string("op_13046_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13046 = slice_by_index(begin = var_13046_begin_0, end = var_13046_end_0, end_mask = var_13046_end_mask_0, x = q_51)[name = string("op_13046")]; fp16 const_384_promoted = const()[name = string("const_384_promoted"), val = fp16(-0x1p+0)]; tensor var_13047 = mul(x = var_13046, y = const_384_promoted)[name = string("op_13047")]; tensor var_13052_begin_0 = const()[name = string("op_13052_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13052_end_0 = const()[name = string("op_13052_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_13052_end_mask_0 = const()[name = string("op_13052_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13052 = slice_by_index(begin = var_13052_begin_0, end = var_13052_end_0, end_mask = var_13052_end_mask_0, x = q_51)[name = string("op_13052")]; int32 var_13054 = const()[name = string("op_13054"), val = int32(-1)]; bool var_13055_interleave_0 = const()[name = string("op_13055_interleave_0"), val = bool(false)]; tensor var_13055 = concat(axis = var_13054, interleave = var_13055_interleave_0, values = (var_13047, var_13052))[name = string("op_13055")]; tensor var_13056 = mul(x = var_13055, y = sin_1)[name = string("op_13056")]; tensor query_51 = add(x = var_13041, y = var_13056)[name = string("query_51")]; tensor var_13059 = mul(x = k_51, y = cos_1)[name = string("op_13059")]; tensor var_13064_begin_0 = const()[name = string("op_13064_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13064_end_0 = const()[name = string("op_13064_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_13064_end_mask_0 = const()[name = string("op_13064_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13064 = slice_by_index(begin = var_13064_begin_0, end = var_13064_end_0, end_mask = var_13064_end_mask_0, x = k_51)[name = string("op_13064")]; fp16 const_385_promoted = const()[name = string("const_385_promoted"), val = fp16(-0x1p+0)]; tensor var_13065 = mul(x = var_13064, y = const_385_promoted)[name = string("op_13065")]; tensor var_13070_begin_0 = const()[name = string("op_13070_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13070_end_0 = const()[name = string("op_13070_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_13070_end_mask_0 = const()[name = string("op_13070_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13070 = slice_by_index(begin = var_13070_begin_0, end = var_13070_end_0, end_mask = var_13070_end_mask_0, x = k_51)[name = string("op_13070")]; int32 var_13072 = const()[name = string("op_13072"), val = int32(-1)]; bool var_13073_interleave_0 = const()[name = string("op_13073_interleave_0"), val = bool(false)]; tensor var_13073 = concat(axis = var_13072, interleave = var_13073_interleave_0, values = (var_13065, var_13070))[name = string("op_13073")]; tensor var_13074 = mul(x = var_13073, y = sin_1)[name = string("op_13074")]; tensor key_51 = add(x = var_13059, y = var_13074)[name = string("key_51")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; int32 concat_452_axis_0 = const()[name = string("concat_452_axis_0"), val = int32(0)]; bool concat_452_interleave_0 = const()[name = string("concat_452_interleave_0"), val = bool(false)]; tensor concat_452 = concat(axis = concat_452_axis_0, interleave = concat_452_interleave_0, values = (expand_dims_300, expand_dims_301, current_pos, expand_dims_303))[name = string("concat_452")]; tensor concat_453_values1_0 = const()[name = string("concat_453_values1_0"), val = tensor([0])]; tensor concat_453_values3_0 = const()[name = string("concat_453_values3_0"), val = tensor([0])]; int32 concat_453_axis_0 = const()[name = string("concat_453_axis_0"), val = int32(0)]; bool concat_453_interleave_0 = const()[name = string("concat_453_interleave_0"), val = bool(false)]; tensor concat_453 = concat(axis = concat_453_axis_0, interleave = concat_453_interleave_0, values = (expand_dims_304, concat_453_values1_0, var_1746, concat_453_values3_0))[name = string("concat_453")]; tensor model_model_kv_cache_0_internal_tensor_assign_51_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_51_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_51_cast_fp16 = slice_update(begin = concat_452, begin_mask = model_model_kv_cache_0_internal_tensor_assign_51_begin_mask_0, end = concat_453, end_mask = model_model_kv_cache_0_internal_tensor_assign_51_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_51_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_51_stride_0, update = key_51, x = coreml_update_state_105)[name = string("model_model_kv_cache_0_internal_tensor_assign_51_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_51_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_218_write_state")]; tensor coreml_update_state_106 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_218")]; tensor expand_dims_306 = const()[name = string("expand_dims_306"), val = tensor([53])]; tensor expand_dims_307 = const()[name = string("expand_dims_307"), val = tensor([0])]; tensor expand_dims_309 = const()[name = string("expand_dims_309"), val = tensor([0])]; tensor expand_dims_310 = const()[name = string("expand_dims_310"), val = tensor([54])]; int32 concat_456_axis_0 = const()[name = string("concat_456_axis_0"), val = int32(0)]; bool concat_456_interleave_0 = const()[name = string("concat_456_interleave_0"), val = bool(false)]; tensor concat_456 = concat(axis = concat_456_axis_0, interleave = concat_456_interleave_0, values = (expand_dims_306, expand_dims_307, current_pos, expand_dims_309))[name = string("concat_456")]; tensor concat_457_values1_0 = const()[name = string("concat_457_values1_0"), val = tensor([0])]; tensor concat_457_values3_0 = const()[name = string("concat_457_values3_0"), val = tensor([0])]; int32 concat_457_axis_0 = const()[name = string("concat_457_axis_0"), val = int32(0)]; bool concat_457_interleave_0 = const()[name = string("concat_457_interleave_0"), val = bool(false)]; tensor concat_457 = concat(axis = concat_457_axis_0, interleave = concat_457_interleave_0, values = (expand_dims_310, concat_457_values1_0, var_1746, concat_457_values3_0))[name = string("concat_457")]; tensor model_model_kv_cache_0_internal_tensor_assign_52_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_52_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_52_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_251 = transpose(perm = var_12983, x = var_12978)[name = string("transpose_23")]; tensor model_model_kv_cache_0_internal_tensor_assign_52_cast_fp16 = slice_update(begin = concat_456, begin_mask = model_model_kv_cache_0_internal_tensor_assign_52_begin_mask_0, end = concat_457, end_mask = model_model_kv_cache_0_internal_tensor_assign_52_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_52_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_52_stride_0, update = value_251, x = coreml_update_state_106)[name = string("model_model_kv_cache_0_internal_tensor_assign_52_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_52_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_219_write_state")]; tensor coreml_update_state_107 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_219")]; tensor var_13145_begin_0 = const()[name = string("op_13145_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_13145_end_0 = const()[name = string("op_13145_end_0"), val = tensor([26, 8, 1536, 128])]; tensor var_13145_end_mask_0 = const()[name = string("op_13145_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_13145_cast_fp16 = slice_by_index(begin = var_13145_begin_0, end = var_13145_end_0, end_mask = var_13145_end_mask_0, x = coreml_update_state_107)[name = string("op_13145_cast_fp16")]; tensor key_cache_51_axes_0 = const()[name = string("key_cache_51_axes_0"), val = tensor([0])]; tensor key_cache_51_cast_fp16 = squeeze(axes = key_cache_51_axes_0, x = var_13145_cast_fp16)[name = string("key_cache_51_cast_fp16")]; tensor var_13152_begin_0 = const()[name = string("op_13152_begin_0"), val = tensor([53, 0, 0, 0])]; tensor var_13152_end_0 = const()[name = string("op_13152_end_0"), val = tensor([54, 8, 1536, 128])]; tensor var_13152_end_mask_0 = const()[name = string("op_13152_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_13152_cast_fp16 = slice_by_index(begin = var_13152_begin_0, end = var_13152_end_0, end_mask = var_13152_end_mask_0, x = coreml_update_state_107)[name = string("op_13152_cast_fp16")]; tensor value_cache_51_axes_0 = const()[name = string("value_cache_51_axes_0"), val = tensor([0])]; tensor value_cache_51_cast_fp16 = squeeze(axes = value_cache_51_axes_0, x = var_13152_cast_fp16)[name = string("value_cache_51_cast_fp16")]; tensor var_13176_axes_0 = const()[name = string("op_13176_axes_0"), val = tensor([1])]; tensor var_13176_cast_fp16 = expand_dims(axes = var_13176_axes_0, x = key_cache_51_cast_fp16)[name = string("op_13176_cast_fp16")]; tensor var_13181 = const()[name = string("op_13181"), val = tensor([1, 2, 1, 1])]; tensor value_255_cast_fp16 = tile(reps = var_13181, x = var_13176_cast_fp16)[name = string("value_255_cast_fp16")]; tensor var_13187 = const()[name = string("op_13187"), val = tensor([1, 16, 1536, 128])]; tensor key_states_103_cast_fp16 = reshape(shape = var_13187, x = value_255_cast_fp16)[name = string("key_states_103_cast_fp16")]; tensor var_13190_axes_0 = const()[name = string("op_13190_axes_0"), val = tensor([1])]; tensor var_13190_cast_fp16 = expand_dims(axes = var_13190_axes_0, x = value_cache_51_cast_fp16)[name = string("op_13190_cast_fp16")]; tensor var_13195 = const()[name = string("op_13195"), val = tensor([1, 2, 1, 1])]; tensor value_259_cast_fp16 = tile(reps = var_13195, x = var_13190_cast_fp16)[name = string("value_259_cast_fp16")]; bool var_13216_transpose_x_0 = const()[name = string("op_13216_transpose_x_0"), val = bool(false)]; bool var_13216_transpose_y_0 = const()[name = string("op_13216_transpose_y_0"), val = bool(true)]; tensor var_13216 = matmul(transpose_x = var_13216_transpose_x_0, transpose_y = var_13216_transpose_y_0, x = query_51, y = key_states_103_cast_fp16)[name = string("op_13216")]; fp16 var_13217_to_fp16 = const()[name = string("op_13217_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_101_cast_fp16 = mul(x = var_13216, y = var_13217_to_fp16)[name = string("attention_101_cast_fp16")]; tensor attention_103_cast_fp16 = add(x = attention_101_cast_fp16, y = causal_mask)[name = string("attention_103_cast_fp16")]; int32 var_13226 = const()[name = string("op_13226"), val = int32(-1)]; tensor var_13228_cast_fp16 = softmax(axis = var_13226, x = attention_103_cast_fp16)[name = string("op_13228_cast_fp16")]; tensor concat_462 = const()[name = string("concat_462"), val = tensor([16, 64, 1536])]; tensor reshape_75_cast_fp16 = reshape(shape = concat_462, x = var_13228_cast_fp16)[name = string("reshape_75_cast_fp16")]; tensor concat_463 = const()[name = string("concat_463"), val = tensor([16, 1536, 128])]; tensor reshape_76_cast_fp16 = reshape(shape = concat_463, x = value_259_cast_fp16)[name = string("reshape_76_cast_fp16")]; bool matmul_25_transpose_x_0 = const()[name = string("matmul_25_transpose_x_0"), val = bool(false)]; bool matmul_25_transpose_y_0 = const()[name = string("matmul_25_transpose_y_0"), val = bool(false)]; tensor matmul_25_cast_fp16 = matmul(transpose_x = matmul_25_transpose_x_0, transpose_y = matmul_25_transpose_y_0, x = reshape_75_cast_fp16, y = reshape_76_cast_fp16)[name = string("matmul_25_cast_fp16")]; tensor concat_467 = const()[name = string("concat_467"), val = tensor([1, 16, 64, 128])]; tensor reshape_77_cast_fp16 = reshape(shape = concat_467, x = matmul_25_cast_fp16)[name = string("reshape_77_cast_fp16")]; tensor var_13240_perm_0 = const()[name = string("op_13240_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_13246 = const()[name = string("op_13246"), val = tensor([1, 64, 2048])]; tensor var_13240_cast_fp16 = transpose(perm = var_13240_perm_0, x = reshape_77_cast_fp16)[name = string("transpose_22")]; tensor output_153_cast_fp16 = reshape(shape = var_13246, x = var_13240_cast_fp16)[name = string("output_153_cast_fp16")]; tensor var_13251 = const()[name = string("op_13251"), val = tensor([0, 2, 1])]; string var_13267_pad_type_0 = const()[name = string("op_13267_pad_type_0"), val = string("valid")]; int32 var_13267_groups_0 = const()[name = string("op_13267_groups_0"), val = int32(1)]; tensor var_13267_strides_0 = const()[name = string("op_13267_strides_0"), val = tensor([1])]; tensor var_13267_pad_0 = const()[name = string("op_13267_pad_0"), val = tensor([0, 0])]; tensor var_13267_dilations_0 = const()[name = string("op_13267_dilations_0"), val = tensor([1])]; tensor squeeze_25_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332760832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334333760))))[name = string("squeeze_25_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_13252_cast_fp16 = transpose(perm = var_13251, x = output_153_cast_fp16)[name = string("transpose_21")]; tensor var_13267_cast_fp16 = conv(dilations = var_13267_dilations_0, groups = var_13267_groups_0, pad = var_13267_pad_0, pad_type = var_13267_pad_type_0, strides = var_13267_strides_0, weight = squeeze_25_cast_fp16_to_fp32_to_fp16_palettized, x = var_13252_cast_fp16)[name = string("op_13267_cast_fp16")]; tensor var_13271 = const()[name = string("op_13271"), val = tensor([0, 2, 1])]; tensor attn_output_51_cast_fp16 = transpose(perm = var_13271, x = var_13267_cast_fp16)[name = string("transpose_20")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_251_cast_fp16, y = attn_output_51_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; int32 var_13286 = const()[name = string("op_13286"), val = int32(-1)]; fp16 const_387_promoted_to_fp16 = const()[name = string("const_387_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13288_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_387_promoted_to_fp16)[name = string("op_13288_cast_fp16")]; bool input_461_interleave_0 = const()[name = string("input_461_interleave_0"), val = bool(false)]; tensor input_461_cast_fp16 = concat(axis = var_13286, interleave = input_461_interleave_0, values = (hidden_states_259_cast_fp16, var_13288_cast_fp16))[name = string("input_461_cast_fp16")]; tensor normed_413_axes_0 = const()[name = string("normed_413_axes_0"), val = tensor([-1])]; fp16 var_13283_to_fp16 = const()[name = string("op_13283_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_413_cast_fp16 = layer_norm(axes = normed_413_axes_0, epsilon = var_13283_to_fp16, x = input_461_cast_fp16)[name = string("normed_413_cast_fp16")]; tensor normed_415_begin_0 = const()[name = string("normed_415_begin_0"), val = tensor([0, 0, 0])]; tensor normed_415_end_0 = const()[name = string("normed_415_end_0"), val = tensor([1, 64, 1024])]; tensor normed_415_end_mask_0 = const()[name = string("normed_415_end_mask_0"), val = tensor([true, true, false])]; tensor normed_415_cast_fp16 = slice_by_index(begin = normed_415_begin_0, end = normed_415_end_0, end_mask = normed_415_end_mask_0, x = normed_413_cast_fp16)[name = string("normed_415_cast_fp16")]; tensor const_389_promoted_to_fp16 = const()[name = string("const_389_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334350208)))]; tensor x_101_cast_fp16 = mul(x = normed_415_cast_fp16, y = const_389_promoted_to_fp16)[name = string("x_101_cast_fp16")]; tensor var_13308 = const()[name = string("op_13308"), val = tensor([0, 2, 1])]; tensor input_463_axes_0 = const()[name = string("input_463_axes_0"), val = tensor([2])]; tensor var_13309 = transpose(perm = var_13308, x = x_101_cast_fp16)[name = string("transpose_19")]; tensor input_463 = expand_dims(axes = input_463_axes_0, x = var_13309)[name = string("input_463")]; string input_465_pad_type_0 = const()[name = string("input_465_pad_type_0"), val = string("valid")]; tensor input_465_strides_0 = const()[name = string("input_465_strides_0"), val = tensor([1, 1])]; tensor input_465_pad_0 = const()[name = string("input_465_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_465_dilations_0 = const()[name = string("input_465_dilations_0"), val = tensor([1, 1])]; int32 input_465_groups_0 = const()[name = string("input_465_groups_0"), val = int32(1)]; tensor input_465 = conv(dilations = input_465_dilations_0, groups = input_465_groups_0, pad = input_465_pad_0, pad_type = input_465_pad_type_0, strides = input_465_strides_0, weight = model_model_layers_25_mlp_gate_proj_weight_palettized, x = input_463)[name = string("input_465")]; string b_51_pad_type_0 = const()[name = string("b_51_pad_type_0"), val = string("valid")]; tensor b_51_strides_0 = const()[name = string("b_51_strides_0"), val = tensor([1, 1])]; tensor b_51_pad_0 = const()[name = string("b_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_51_dilations_0 = const()[name = string("b_51_dilations_0"), val = tensor([1, 1])]; int32 b_51_groups_0 = const()[name = string("b_51_groups_0"), val = int32(1)]; tensor b_51 = conv(dilations = b_51_dilations_0, groups = b_51_groups_0, pad = b_51_pad_0, pad_type = b_51_pad_type_0, strides = b_51_strides_0, weight = model_model_layers_25_mlp_up_proj_weight_palettized, x = input_463)[name = string("b_51")]; tensor c_51 = silu(x = input_465)[name = string("c_51")]; tensor input_467 = mul(x = c_51, y = b_51)[name = string("input_467")]; string e_51_pad_type_0 = const()[name = string("e_51_pad_type_0"), val = string("valid")]; tensor e_51_strides_0 = const()[name = string("e_51_strides_0"), val = tensor([1, 1])]; tensor e_51_pad_0 = const()[name = string("e_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_51_dilations_0 = const()[name = string("e_51_dilations_0"), val = tensor([1, 1])]; int32 e_51_groups_0 = const()[name = string("e_51_groups_0"), val = int32(1)]; tensor e_51 = conv(dilations = e_51_dilations_0, groups = e_51_groups_0, pad = e_51_pad_0, pad_type = e_51_pad_type_0, strides = e_51_strides_0, weight = model_model_layers_25_mlp_down_proj_weight_palettized, x = input_467)[name = string("e_51")]; tensor var_13331_axes_0 = const()[name = string("op_13331_axes_0"), val = tensor([2])]; tensor var_13331 = squeeze(axes = var_13331_axes_0, x = e_51)[name = string("op_13331")]; tensor var_13332 = const()[name = string("op_13332"), val = tensor([0, 2, 1])]; tensor var_13333 = transpose(perm = var_13332, x = var_13331)[name = string("transpose_18")]; tensor hidden_states_261_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = var_13333)[name = string("hidden_states_261_cast_fp16")]; int32 var_13347 = const()[name = string("op_13347"), val = int32(-1)]; fp16 const_390_promoted_to_fp16 = const()[name = string("const_390_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13349_cast_fp16 = mul(x = hidden_states_261_cast_fp16, y = const_390_promoted_to_fp16)[name = string("op_13349_cast_fp16")]; bool input_469_interleave_0 = const()[name = string("input_469_interleave_0"), val = bool(false)]; tensor input_469_cast_fp16 = concat(axis = var_13347, interleave = input_469_interleave_0, values = (hidden_states_261_cast_fp16, var_13349_cast_fp16))[name = string("input_469_cast_fp16")]; tensor normed_417_axes_0 = const()[name = string("normed_417_axes_0"), val = tensor([-1])]; fp16 var_13344_to_fp16 = const()[name = string("op_13344_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_417_cast_fp16 = layer_norm(axes = normed_417_axes_0, epsilon = var_13344_to_fp16, x = input_469_cast_fp16)[name = string("normed_417_cast_fp16")]; tensor normed_419_begin_0 = const()[name = string("normed_419_begin_0"), val = tensor([0, 0, 0])]; tensor normed_419_end_0 = const()[name = string("normed_419_end_0"), val = tensor([1, 64, 1024])]; tensor normed_419_end_mask_0 = const()[name = string("normed_419_end_mask_0"), val = tensor([true, true, false])]; tensor normed_419_cast_fp16 = slice_by_index(begin = normed_419_begin_0, end = normed_419_end_0, end_mask = normed_419_end_mask_0, x = normed_417_cast_fp16)[name = string("normed_419_cast_fp16")]; tensor const_392_promoted_to_fp16 = const()[name = string("const_392_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334352320)))]; tensor hidden_states_263_cast_fp16 = mul(x = normed_419_cast_fp16, y = const_392_promoted_to_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor var_13361 = const()[name = string("op_13361"), val = tensor([0, 2, 1])]; tensor var_13364_axes_0 = const()[name = string("op_13364_axes_0"), val = tensor([2])]; tensor var_13362_cast_fp16 = transpose(perm = var_13361, x = hidden_states_263_cast_fp16)[name = string("transpose_17")]; tensor var_13364_cast_fp16 = expand_dims(axes = var_13364_axes_0, x = var_13362_cast_fp16)[name = string("op_13364_cast_fp16")]; string var_13380_pad_type_0 = const()[name = string("op_13380_pad_type_0"), val = string("valid")]; tensor var_13380_strides_0 = const()[name = string("op_13380_strides_0"), val = tensor([1, 1])]; tensor var_13380_pad_0 = const()[name = string("op_13380_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13380_dilations_0 = const()[name = string("op_13380_dilations_0"), val = tensor([1, 1])]; int32 var_13380_groups_0 = const()[name = string("op_13380_groups_0"), val = int32(1)]; tensor var_13380 = conv(dilations = var_13380_dilations_0, groups = var_13380_groups_0, pad = var_13380_pad_0, pad_type = var_13380_pad_type_0, strides = var_13380_strides_0, weight = model_model_layers_26_self_attn_q_proj_weight_palettized, x = var_13364_cast_fp16)[name = string("op_13380")]; tensor var_13385 = const()[name = string("op_13385"), val = tensor([1, 16, 128, 64])]; tensor var_13386 = reshape(shape = var_13385, x = var_13380)[name = string("op_13386")]; tensor var_13391 = const()[name = string("op_13391"), val = tensor([0, 1, 3, 2])]; string var_13403_pad_type_0 = const()[name = string("op_13403_pad_type_0"), val = string("valid")]; tensor var_13403_strides_0 = const()[name = string("op_13403_strides_0"), val = tensor([1, 1])]; tensor var_13403_pad_0 = const()[name = string("op_13403_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13403_dilations_0 = const()[name = string("op_13403_dilations_0"), val = tensor([1, 1])]; int32 var_13403_groups_0 = const()[name = string("op_13403_groups_0"), val = int32(1)]; tensor var_13403 = conv(dilations = var_13403_dilations_0, groups = var_13403_groups_0, pad = var_13403_pad_0, pad_type = var_13403_pad_type_0, strides = var_13403_strides_0, weight = model_model_layers_26_self_attn_k_proj_weight_palettized, x = var_13364_cast_fp16)[name = string("op_13403")]; tensor var_13408 = const()[name = string("op_13408"), val = tensor([1, 8, 128, 64])]; tensor var_13409 = reshape(shape = var_13408, x = var_13403)[name = string("op_13409")]; tensor var_13414 = const()[name = string("op_13414"), val = tensor([0, 1, 3, 2])]; string var_13426_pad_type_0 = const()[name = string("op_13426_pad_type_0"), val = string("valid")]; tensor var_13426_strides_0 = const()[name = string("op_13426_strides_0"), val = tensor([1, 1])]; tensor var_13426_pad_0 = const()[name = string("op_13426_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13426_dilations_0 = const()[name = string("op_13426_dilations_0"), val = tensor([1, 1])]; int32 var_13426_groups_0 = const()[name = string("op_13426_groups_0"), val = int32(1)]; tensor var_13426 = conv(dilations = var_13426_dilations_0, groups = var_13426_groups_0, pad = var_13426_pad_0, pad_type = var_13426_pad_type_0, strides = var_13426_strides_0, weight = model_model_layers_26_self_attn_v_proj_weight_palettized, x = var_13364_cast_fp16)[name = string("op_13426")]; tensor var_13431 = const()[name = string("op_13431"), val = tensor([1, 8, 128, 64])]; tensor var_13432 = reshape(shape = var_13431, x = var_13426)[name = string("op_13432")]; tensor var_13437 = const()[name = string("op_13437"), val = tensor([0, 1, 3, 2])]; int32 var_13450 = const()[name = string("op_13450"), val = int32(-1)]; fp16 const_393_promoted = const()[name = string("const_393_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_265 = transpose(perm = var_13391, x = var_13386)[name = string("transpose_16")]; tensor var_13452 = mul(x = hidden_states_265, y = const_393_promoted)[name = string("op_13452")]; bool input_473_interleave_0 = const()[name = string("input_473_interleave_0"), val = bool(false)]; tensor input_473 = concat(axis = var_13450, interleave = input_473_interleave_0, values = (hidden_states_265, var_13452))[name = string("input_473")]; tensor normed_421_axes_0 = const()[name = string("normed_421_axes_0"), val = tensor([-1])]; fp16 var_13447_to_fp16 = const()[name = string("op_13447_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_421_cast_fp16 = layer_norm(axes = normed_421_axes_0, epsilon = var_13447_to_fp16, x = input_473)[name = string("normed_421_cast_fp16")]; tensor normed_423_begin_0 = const()[name = string("normed_423_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_423_end_0 = const()[name = string("normed_423_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_423_end_mask_0 = const()[name = string("normed_423_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_423 = slice_by_index(begin = normed_423_begin_0, end = normed_423_end_0, end_mask = normed_423_end_mask_0, x = normed_421_cast_fp16)[name = string("normed_423")]; tensor const_395 = const()[name = string("const_395"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334354432)))]; tensor q_53 = mul(x = normed_423, y = const_395)[name = string("q_53")]; int32 var_13472 = const()[name = string("op_13472"), val = int32(-1)]; fp16 const_396_promoted = const()[name = string("const_396_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_267 = transpose(perm = var_13414, x = var_13409)[name = string("transpose_15")]; tensor var_13474 = mul(x = hidden_states_267, y = const_396_promoted)[name = string("op_13474")]; bool input_475_interleave_0 = const()[name = string("input_475_interleave_0"), val = bool(false)]; tensor input_475 = concat(axis = var_13472, interleave = input_475_interleave_0, values = (hidden_states_267, var_13474))[name = string("input_475")]; tensor normed_425_axes_0 = const()[name = string("normed_425_axes_0"), val = tensor([-1])]; fp16 var_13469_to_fp16 = const()[name = string("op_13469_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_425_cast_fp16 = layer_norm(axes = normed_425_axes_0, epsilon = var_13469_to_fp16, x = input_475)[name = string("normed_425_cast_fp16")]; tensor normed_427_begin_0 = const()[name = string("normed_427_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_427_end_0 = const()[name = string("normed_427_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_427_end_mask_0 = const()[name = string("normed_427_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_427 = slice_by_index(begin = normed_427_begin_0, end = normed_427_end_0, end_mask = normed_427_end_mask_0, x = normed_425_cast_fp16)[name = string("normed_427")]; tensor const_398 = const()[name = string("const_398"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334354752)))]; tensor k_53 = mul(x = normed_427, y = const_398)[name = string("k_53")]; tensor var_13495 = mul(x = q_53, y = cos_1)[name = string("op_13495")]; tensor var_13500_begin_0 = const()[name = string("op_13500_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13500_end_0 = const()[name = string("op_13500_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_13500_end_mask_0 = const()[name = string("op_13500_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13500 = slice_by_index(begin = var_13500_begin_0, end = var_13500_end_0, end_mask = var_13500_end_mask_0, x = q_53)[name = string("op_13500")]; fp16 const_399_promoted = const()[name = string("const_399_promoted"), val = fp16(-0x1p+0)]; tensor var_13501 = mul(x = var_13500, y = const_399_promoted)[name = string("op_13501")]; tensor var_13506_begin_0 = const()[name = string("op_13506_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13506_end_0 = const()[name = string("op_13506_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_13506_end_mask_0 = const()[name = string("op_13506_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13506 = slice_by_index(begin = var_13506_begin_0, end = var_13506_end_0, end_mask = var_13506_end_mask_0, x = q_53)[name = string("op_13506")]; int32 var_13508 = const()[name = string("op_13508"), val = int32(-1)]; bool var_13509_interleave_0 = const()[name = string("op_13509_interleave_0"), val = bool(false)]; tensor var_13509 = concat(axis = var_13508, interleave = var_13509_interleave_0, values = (var_13501, var_13506))[name = string("op_13509")]; tensor var_13510 = mul(x = var_13509, y = sin_1)[name = string("op_13510")]; tensor query_53 = add(x = var_13495, y = var_13510)[name = string("query_53")]; tensor var_13513 = mul(x = k_53, y = cos_1)[name = string("op_13513")]; tensor var_13518_begin_0 = const()[name = string("op_13518_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13518_end_0 = const()[name = string("op_13518_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_13518_end_mask_0 = const()[name = string("op_13518_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13518 = slice_by_index(begin = var_13518_begin_0, end = var_13518_end_0, end_mask = var_13518_end_mask_0, x = k_53)[name = string("op_13518")]; fp16 const_400_promoted = const()[name = string("const_400_promoted"), val = fp16(-0x1p+0)]; tensor var_13519 = mul(x = var_13518, y = const_400_promoted)[name = string("op_13519")]; tensor var_13524_begin_0 = const()[name = string("op_13524_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13524_end_0 = const()[name = string("op_13524_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_13524_end_mask_0 = const()[name = string("op_13524_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13524 = slice_by_index(begin = var_13524_begin_0, end = var_13524_end_0, end_mask = var_13524_end_mask_0, x = k_53)[name = string("op_13524")]; int32 var_13526 = const()[name = string("op_13526"), val = int32(-1)]; bool var_13527_interleave_0 = const()[name = string("op_13527_interleave_0"), val = bool(false)]; tensor var_13527 = concat(axis = var_13526, interleave = var_13527_interleave_0, values = (var_13519, var_13524))[name = string("op_13527")]; tensor var_13528 = mul(x = var_13527, y = sin_1)[name = string("op_13528")]; tensor key_53 = add(x = var_13513, y = var_13528)[name = string("key_53")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; int32 concat_470_axis_0 = const()[name = string("concat_470_axis_0"), val = int32(0)]; bool concat_470_interleave_0 = const()[name = string("concat_470_interleave_0"), val = bool(false)]; tensor concat_470 = concat(axis = concat_470_axis_0, interleave = concat_470_interleave_0, values = (expand_dims_312, expand_dims_313, current_pos, expand_dims_315))[name = string("concat_470")]; tensor concat_471_values1_0 = const()[name = string("concat_471_values1_0"), val = tensor([0])]; tensor concat_471_values3_0 = const()[name = string("concat_471_values3_0"), val = tensor([0])]; int32 concat_471_axis_0 = const()[name = string("concat_471_axis_0"), val = int32(0)]; bool concat_471_interleave_0 = const()[name = string("concat_471_interleave_0"), val = bool(false)]; tensor concat_471 = concat(axis = concat_471_axis_0, interleave = concat_471_interleave_0, values = (expand_dims_316, concat_471_values1_0, var_1746, concat_471_values3_0))[name = string("concat_471")]; tensor model_model_kv_cache_0_internal_tensor_assign_53_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_53_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_53_cast_fp16 = slice_update(begin = concat_470, begin_mask = model_model_kv_cache_0_internal_tensor_assign_53_begin_mask_0, end = concat_471, end_mask = model_model_kv_cache_0_internal_tensor_assign_53_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_53_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_53_stride_0, update = key_53, x = coreml_update_state_107)[name = string("model_model_kv_cache_0_internal_tensor_assign_53_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_53_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_220_write_state")]; tensor coreml_update_state_108 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_220")]; tensor expand_dims_318 = const()[name = string("expand_dims_318"), val = tensor([54])]; tensor expand_dims_319 = const()[name = string("expand_dims_319"), val = tensor([0])]; tensor expand_dims_321 = const()[name = string("expand_dims_321"), val = tensor([0])]; tensor expand_dims_322 = const()[name = string("expand_dims_322"), val = tensor([55])]; int32 concat_474_axis_0 = const()[name = string("concat_474_axis_0"), val = int32(0)]; bool concat_474_interleave_0 = const()[name = string("concat_474_interleave_0"), val = bool(false)]; tensor concat_474 = concat(axis = concat_474_axis_0, interleave = concat_474_interleave_0, values = (expand_dims_318, expand_dims_319, current_pos, expand_dims_321))[name = string("concat_474")]; tensor concat_475_values1_0 = const()[name = string("concat_475_values1_0"), val = tensor([0])]; tensor concat_475_values3_0 = const()[name = string("concat_475_values3_0"), val = tensor([0])]; int32 concat_475_axis_0 = const()[name = string("concat_475_axis_0"), val = int32(0)]; bool concat_475_interleave_0 = const()[name = string("concat_475_interleave_0"), val = bool(false)]; tensor concat_475 = concat(axis = concat_475_axis_0, interleave = concat_475_interleave_0, values = (expand_dims_322, concat_475_values1_0, var_1746, concat_475_values3_0))[name = string("concat_475")]; tensor model_model_kv_cache_0_internal_tensor_assign_54_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_54_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_54_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_261 = transpose(perm = var_13437, x = var_13432)[name = string("transpose_14")]; tensor model_model_kv_cache_0_internal_tensor_assign_54_cast_fp16 = slice_update(begin = concat_474, begin_mask = model_model_kv_cache_0_internal_tensor_assign_54_begin_mask_0, end = concat_475, end_mask = model_model_kv_cache_0_internal_tensor_assign_54_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_54_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_54_stride_0, update = value_261, x = coreml_update_state_108)[name = string("model_model_kv_cache_0_internal_tensor_assign_54_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_54_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_221_write_state")]; tensor coreml_update_state_109 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_221")]; tensor var_13599_begin_0 = const()[name = string("op_13599_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_13599_end_0 = const()[name = string("op_13599_end_0"), val = tensor([27, 8, 1536, 128])]; tensor var_13599_end_mask_0 = const()[name = string("op_13599_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_13599_cast_fp16 = slice_by_index(begin = var_13599_begin_0, end = var_13599_end_0, end_mask = var_13599_end_mask_0, x = coreml_update_state_109)[name = string("op_13599_cast_fp16")]; tensor key_cache_53_axes_0 = const()[name = string("key_cache_53_axes_0"), val = tensor([0])]; tensor key_cache_53_cast_fp16 = squeeze(axes = key_cache_53_axes_0, x = var_13599_cast_fp16)[name = string("key_cache_53_cast_fp16")]; tensor var_13606_begin_0 = const()[name = string("op_13606_begin_0"), val = tensor([54, 0, 0, 0])]; tensor var_13606_end_0 = const()[name = string("op_13606_end_0"), val = tensor([55, 8, 1536, 128])]; tensor var_13606_end_mask_0 = const()[name = string("op_13606_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_13606_cast_fp16 = slice_by_index(begin = var_13606_begin_0, end = var_13606_end_0, end_mask = var_13606_end_mask_0, x = coreml_update_state_109)[name = string("op_13606_cast_fp16")]; tensor value_cache_53_axes_0 = const()[name = string("value_cache_53_axes_0"), val = tensor([0])]; tensor value_cache_53_cast_fp16 = squeeze(axes = value_cache_53_axes_0, x = var_13606_cast_fp16)[name = string("value_cache_53_cast_fp16")]; tensor var_13630_axes_0 = const()[name = string("op_13630_axes_0"), val = tensor([1])]; tensor var_13630_cast_fp16 = expand_dims(axes = var_13630_axes_0, x = key_cache_53_cast_fp16)[name = string("op_13630_cast_fp16")]; tensor var_13635 = const()[name = string("op_13635"), val = tensor([1, 2, 1, 1])]; tensor value_265_cast_fp16 = tile(reps = var_13635, x = var_13630_cast_fp16)[name = string("value_265_cast_fp16")]; tensor var_13641 = const()[name = string("op_13641"), val = tensor([1, 16, 1536, 128])]; tensor key_states_107_cast_fp16 = reshape(shape = var_13641, x = value_265_cast_fp16)[name = string("key_states_107_cast_fp16")]; tensor var_13644_axes_0 = const()[name = string("op_13644_axes_0"), val = tensor([1])]; tensor var_13644_cast_fp16 = expand_dims(axes = var_13644_axes_0, x = value_cache_53_cast_fp16)[name = string("op_13644_cast_fp16")]; tensor var_13649 = const()[name = string("op_13649"), val = tensor([1, 2, 1, 1])]; tensor value_269_cast_fp16 = tile(reps = var_13649, x = var_13644_cast_fp16)[name = string("value_269_cast_fp16")]; bool var_13670_transpose_x_0 = const()[name = string("op_13670_transpose_x_0"), val = bool(false)]; bool var_13670_transpose_y_0 = const()[name = string("op_13670_transpose_y_0"), val = bool(true)]; tensor var_13670 = matmul(transpose_x = var_13670_transpose_x_0, transpose_y = var_13670_transpose_y_0, x = query_53, y = key_states_107_cast_fp16)[name = string("op_13670")]; fp16 var_13671_to_fp16 = const()[name = string("op_13671_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_105_cast_fp16 = mul(x = var_13670, y = var_13671_to_fp16)[name = string("attention_105_cast_fp16")]; tensor attention_107_cast_fp16 = add(x = attention_105_cast_fp16, y = causal_mask)[name = string("attention_107_cast_fp16")]; int32 var_13680 = const()[name = string("op_13680"), val = int32(-1)]; tensor var_13682_cast_fp16 = softmax(axis = var_13680, x = attention_107_cast_fp16)[name = string("op_13682_cast_fp16")]; tensor concat_480 = const()[name = string("concat_480"), val = tensor([16, 64, 1536])]; tensor reshape_78_cast_fp16 = reshape(shape = concat_480, x = var_13682_cast_fp16)[name = string("reshape_78_cast_fp16")]; tensor concat_481 = const()[name = string("concat_481"), val = tensor([16, 1536, 128])]; tensor reshape_79_cast_fp16 = reshape(shape = concat_481, x = value_269_cast_fp16)[name = string("reshape_79_cast_fp16")]; bool matmul_26_transpose_x_0 = const()[name = string("matmul_26_transpose_x_0"), val = bool(false)]; bool matmul_26_transpose_y_0 = const()[name = string("matmul_26_transpose_y_0"), val = bool(false)]; tensor matmul_26_cast_fp16 = matmul(transpose_x = matmul_26_transpose_x_0, transpose_y = matmul_26_transpose_y_0, x = reshape_78_cast_fp16, y = reshape_79_cast_fp16)[name = string("matmul_26_cast_fp16")]; tensor concat_485 = const()[name = string("concat_485"), val = tensor([1, 16, 64, 128])]; tensor reshape_80_cast_fp16 = reshape(shape = concat_485, x = matmul_26_cast_fp16)[name = string("reshape_80_cast_fp16")]; tensor var_13694_perm_0 = const()[name = string("op_13694_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_13700 = const()[name = string("op_13700"), val = tensor([1, 64, 2048])]; tensor var_13694_cast_fp16 = transpose(perm = var_13694_perm_0, x = reshape_80_cast_fp16)[name = string("transpose_13")]; tensor output_159_cast_fp16 = reshape(shape = var_13700, x = var_13694_cast_fp16)[name = string("output_159_cast_fp16")]; tensor var_13705 = const()[name = string("op_13705"), val = tensor([0, 2, 1])]; string var_13721_pad_type_0 = const()[name = string("op_13721_pad_type_0"), val = string("valid")]; int32 var_13721_groups_0 = const()[name = string("op_13721_groups_0"), val = int32(1)]; tensor var_13721_strides_0 = const()[name = string("op_13721_strides_0"), val = tensor([1])]; tensor var_13721_pad_0 = const()[name = string("op_13721_pad_0"), val = tensor([0, 0])]; tensor var_13721_dilations_0 = const()[name = string("op_13721_dilations_0"), val = tensor([1])]; tensor squeeze_26_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(334355072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335928000))))[name = string("squeeze_26_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_13706_cast_fp16 = transpose(perm = var_13705, x = output_159_cast_fp16)[name = string("transpose_12")]; tensor var_13721_cast_fp16 = conv(dilations = var_13721_dilations_0, groups = var_13721_groups_0, pad = var_13721_pad_0, pad_type = var_13721_pad_type_0, strides = var_13721_strides_0, weight = squeeze_26_cast_fp16_to_fp32_to_fp16_palettized, x = var_13706_cast_fp16)[name = string("op_13721_cast_fp16")]; tensor var_13725 = const()[name = string("op_13725"), val = tensor([0, 2, 1])]; tensor attn_output_53_cast_fp16 = transpose(perm = var_13725, x = var_13721_cast_fp16)[name = string("transpose_11")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_261_cast_fp16, y = attn_output_53_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; int32 var_13740 = const()[name = string("op_13740"), val = int32(-1)]; fp16 const_402_promoted_to_fp16 = const()[name = string("const_402_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13742_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_402_promoted_to_fp16)[name = string("op_13742_cast_fp16")]; bool input_479_interleave_0 = const()[name = string("input_479_interleave_0"), val = bool(false)]; tensor input_479_cast_fp16 = concat(axis = var_13740, interleave = input_479_interleave_0, values = (hidden_states_269_cast_fp16, var_13742_cast_fp16))[name = string("input_479_cast_fp16")]; tensor normed_429_axes_0 = const()[name = string("normed_429_axes_0"), val = tensor([-1])]; fp16 var_13737_to_fp16 = const()[name = string("op_13737_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_429_cast_fp16 = layer_norm(axes = normed_429_axes_0, epsilon = var_13737_to_fp16, x = input_479_cast_fp16)[name = string("normed_429_cast_fp16")]; tensor normed_431_begin_0 = const()[name = string("normed_431_begin_0"), val = tensor([0, 0, 0])]; tensor normed_431_end_0 = const()[name = string("normed_431_end_0"), val = tensor([1, 64, 1024])]; tensor normed_431_end_mask_0 = const()[name = string("normed_431_end_mask_0"), val = tensor([true, true, false])]; tensor normed_431_cast_fp16 = slice_by_index(begin = normed_431_begin_0, end = normed_431_end_0, end_mask = normed_431_end_mask_0, x = normed_429_cast_fp16)[name = string("normed_431_cast_fp16")]; tensor const_404_promoted_to_fp16 = const()[name = string("const_404_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335944448)))]; tensor x_105_cast_fp16 = mul(x = normed_431_cast_fp16, y = const_404_promoted_to_fp16)[name = string("x_105_cast_fp16")]; tensor var_13762 = const()[name = string("op_13762"), val = tensor([0, 2, 1])]; tensor input_481_axes_0 = const()[name = string("input_481_axes_0"), val = tensor([2])]; tensor var_13763 = transpose(perm = var_13762, x = x_105_cast_fp16)[name = string("transpose_10")]; tensor input_481 = expand_dims(axes = input_481_axes_0, x = var_13763)[name = string("input_481")]; string input_483_pad_type_0 = const()[name = string("input_483_pad_type_0"), val = string("valid")]; tensor input_483_strides_0 = const()[name = string("input_483_strides_0"), val = tensor([1, 1])]; tensor input_483_pad_0 = const()[name = string("input_483_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_483_dilations_0 = const()[name = string("input_483_dilations_0"), val = tensor([1, 1])]; int32 input_483_groups_0 = const()[name = string("input_483_groups_0"), val = int32(1)]; tensor input_483 = conv(dilations = input_483_dilations_0, groups = input_483_groups_0, pad = input_483_pad_0, pad_type = input_483_pad_type_0, strides = input_483_strides_0, weight = model_model_layers_26_mlp_gate_proj_weight_palettized, x = input_481)[name = string("input_483")]; string b_53_pad_type_0 = const()[name = string("b_53_pad_type_0"), val = string("valid")]; tensor b_53_strides_0 = const()[name = string("b_53_strides_0"), val = tensor([1, 1])]; tensor b_53_pad_0 = const()[name = string("b_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_53_dilations_0 = const()[name = string("b_53_dilations_0"), val = tensor([1, 1])]; int32 b_53_groups_0 = const()[name = string("b_53_groups_0"), val = int32(1)]; tensor b_53 = conv(dilations = b_53_dilations_0, groups = b_53_groups_0, pad = b_53_pad_0, pad_type = b_53_pad_type_0, strides = b_53_strides_0, weight = model_model_layers_26_mlp_up_proj_weight_palettized, x = input_481)[name = string("b_53")]; tensor c_53 = silu(x = input_483)[name = string("c_53")]; tensor input_485 = mul(x = c_53, y = b_53)[name = string("input_485")]; string e_53_pad_type_0 = const()[name = string("e_53_pad_type_0"), val = string("valid")]; tensor e_53_strides_0 = const()[name = string("e_53_strides_0"), val = tensor([1, 1])]; tensor e_53_pad_0 = const()[name = string("e_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_53_dilations_0 = const()[name = string("e_53_dilations_0"), val = tensor([1, 1])]; int32 e_53_groups_0 = const()[name = string("e_53_groups_0"), val = int32(1)]; tensor e_53 = conv(dilations = e_53_dilations_0, groups = e_53_groups_0, pad = e_53_pad_0, pad_type = e_53_pad_type_0, strides = e_53_strides_0, weight = model_model_layers_26_mlp_down_proj_weight_palettized, x = input_485)[name = string("e_53")]; tensor var_13785_axes_0 = const()[name = string("op_13785_axes_0"), val = tensor([2])]; tensor var_13785 = squeeze(axes = var_13785_axes_0, x = e_53)[name = string("op_13785")]; tensor var_13786 = const()[name = string("op_13786"), val = tensor([0, 2, 1])]; tensor var_13787 = transpose(perm = var_13786, x = var_13785)[name = string("transpose_9")]; tensor hidden_states_271_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = var_13787)[name = string("hidden_states_271_cast_fp16")]; int32 var_13801 = const()[name = string("op_13801"), val = int32(-1)]; fp16 const_405_promoted_to_fp16 = const()[name = string("const_405_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_13803_cast_fp16 = mul(x = hidden_states_271_cast_fp16, y = const_405_promoted_to_fp16)[name = string("op_13803_cast_fp16")]; bool input_487_interleave_0 = const()[name = string("input_487_interleave_0"), val = bool(false)]; tensor input_487_cast_fp16 = concat(axis = var_13801, interleave = input_487_interleave_0, values = (hidden_states_271_cast_fp16, var_13803_cast_fp16))[name = string("input_487_cast_fp16")]; tensor normed_433_axes_0 = const()[name = string("normed_433_axes_0"), val = tensor([-1])]; fp16 var_13798_to_fp16 = const()[name = string("op_13798_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_433_cast_fp16 = layer_norm(axes = normed_433_axes_0, epsilon = var_13798_to_fp16, x = input_487_cast_fp16)[name = string("normed_433_cast_fp16")]; tensor normed_435_begin_0 = const()[name = string("normed_435_begin_0"), val = tensor([0, 0, 0])]; tensor normed_435_end_0 = const()[name = string("normed_435_end_0"), val = tensor([1, 64, 1024])]; tensor normed_435_end_mask_0 = const()[name = string("normed_435_end_mask_0"), val = tensor([true, true, false])]; tensor normed_435_cast_fp16 = slice_by_index(begin = normed_435_begin_0, end = normed_435_end_0, end_mask = normed_435_end_mask_0, x = normed_433_cast_fp16)[name = string("normed_435_cast_fp16")]; tensor const_407_promoted_to_fp16 = const()[name = string("const_407_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335946560)))]; tensor hidden_states_273_cast_fp16 = mul(x = normed_435_cast_fp16, y = const_407_promoted_to_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor var_13815 = const()[name = string("op_13815"), val = tensor([0, 2, 1])]; tensor var_13818_axes_0 = const()[name = string("op_13818_axes_0"), val = tensor([2])]; tensor var_13816_cast_fp16 = transpose(perm = var_13815, x = hidden_states_273_cast_fp16)[name = string("transpose_8")]; tensor var_13818_cast_fp16 = expand_dims(axes = var_13818_axes_0, x = var_13816_cast_fp16)[name = string("op_13818_cast_fp16")]; string var_13834_pad_type_0 = const()[name = string("op_13834_pad_type_0"), val = string("valid")]; tensor var_13834_strides_0 = const()[name = string("op_13834_strides_0"), val = tensor([1, 1])]; tensor var_13834_pad_0 = const()[name = string("op_13834_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13834_dilations_0 = const()[name = string("op_13834_dilations_0"), val = tensor([1, 1])]; int32 var_13834_groups_0 = const()[name = string("op_13834_groups_0"), val = int32(1)]; tensor var_13834 = conv(dilations = var_13834_dilations_0, groups = var_13834_groups_0, pad = var_13834_pad_0, pad_type = var_13834_pad_type_0, strides = var_13834_strides_0, weight = model_model_layers_27_self_attn_q_proj_weight_palettized, x = var_13818_cast_fp16)[name = string("op_13834")]; tensor var_13839 = const()[name = string("op_13839"), val = tensor([1, 16, 128, 64])]; tensor var_13840 = reshape(shape = var_13839, x = var_13834)[name = string("op_13840")]; tensor var_13845 = const()[name = string("op_13845"), val = tensor([0, 1, 3, 2])]; string var_13857_pad_type_0 = const()[name = string("op_13857_pad_type_0"), val = string("valid")]; tensor var_13857_strides_0 = const()[name = string("op_13857_strides_0"), val = tensor([1, 1])]; tensor var_13857_pad_0 = const()[name = string("op_13857_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13857_dilations_0 = const()[name = string("op_13857_dilations_0"), val = tensor([1, 1])]; int32 var_13857_groups_0 = const()[name = string("op_13857_groups_0"), val = int32(1)]; tensor var_13857 = conv(dilations = var_13857_dilations_0, groups = var_13857_groups_0, pad = var_13857_pad_0, pad_type = var_13857_pad_type_0, strides = var_13857_strides_0, weight = model_model_layers_27_self_attn_k_proj_weight_palettized, x = var_13818_cast_fp16)[name = string("op_13857")]; tensor var_13862 = const()[name = string("op_13862"), val = tensor([1, 8, 128, 64])]; tensor var_13863 = reshape(shape = var_13862, x = var_13857)[name = string("op_13863")]; tensor var_13868 = const()[name = string("op_13868"), val = tensor([0, 1, 3, 2])]; string var_13880_pad_type_0 = const()[name = string("op_13880_pad_type_0"), val = string("valid")]; tensor var_13880_strides_0 = const()[name = string("op_13880_strides_0"), val = tensor([1, 1])]; tensor var_13880_pad_0 = const()[name = string("op_13880_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_13880_dilations_0 = const()[name = string("op_13880_dilations_0"), val = tensor([1, 1])]; int32 var_13880_groups_0 = const()[name = string("op_13880_groups_0"), val = int32(1)]; tensor var_13880 = conv(dilations = var_13880_dilations_0, groups = var_13880_groups_0, pad = var_13880_pad_0, pad_type = var_13880_pad_type_0, strides = var_13880_strides_0, weight = model_model_layers_27_self_attn_v_proj_weight_palettized, x = var_13818_cast_fp16)[name = string("op_13880")]; tensor var_13885 = const()[name = string("op_13885"), val = tensor([1, 8, 128, 64])]; tensor var_13886 = reshape(shape = var_13885, x = var_13880)[name = string("op_13886")]; tensor var_13891 = const()[name = string("op_13891"), val = tensor([0, 1, 3, 2])]; int32 var_13904 = const()[name = string("op_13904"), val = int32(-1)]; fp16 const_408_promoted = const()[name = string("const_408_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_275 = transpose(perm = var_13845, x = var_13840)[name = string("transpose_7")]; tensor var_13906 = mul(x = hidden_states_275, y = const_408_promoted)[name = string("op_13906")]; bool input_491_interleave_0 = const()[name = string("input_491_interleave_0"), val = bool(false)]; tensor input_491 = concat(axis = var_13904, interleave = input_491_interleave_0, values = (hidden_states_275, var_13906))[name = string("input_491")]; tensor normed_437_axes_0 = const()[name = string("normed_437_axes_0"), val = tensor([-1])]; fp16 var_13901_to_fp16 = const()[name = string("op_13901_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_437_cast_fp16 = layer_norm(axes = normed_437_axes_0, epsilon = var_13901_to_fp16, x = input_491)[name = string("normed_437_cast_fp16")]; tensor normed_439_begin_0 = const()[name = string("normed_439_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_439_end_0 = const()[name = string("normed_439_end_0"), val = tensor([1, 16, 64, 128])]; tensor normed_439_end_mask_0 = const()[name = string("normed_439_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_439 = slice_by_index(begin = normed_439_begin_0, end = normed_439_end_0, end_mask = normed_439_end_mask_0, x = normed_437_cast_fp16)[name = string("normed_439")]; tensor const_410 = const()[name = string("const_410"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335948672)))]; tensor q = mul(x = normed_439, y = const_410)[name = string("q")]; int32 var_13926 = const()[name = string("op_13926"), val = int32(-1)]; fp16 const_411_promoted = const()[name = string("const_411_promoted"), val = fp16(-0x1p+0)]; tensor hidden_states_277 = transpose(perm = var_13868, x = var_13863)[name = string("transpose_6")]; tensor var_13928 = mul(x = hidden_states_277, y = const_411_promoted)[name = string("op_13928")]; bool input_493_interleave_0 = const()[name = string("input_493_interleave_0"), val = bool(false)]; tensor input_493 = concat(axis = var_13926, interleave = input_493_interleave_0, values = (hidden_states_277, var_13928))[name = string("input_493")]; tensor normed_441_axes_0 = const()[name = string("normed_441_axes_0"), val = tensor([-1])]; fp16 var_13923_to_fp16 = const()[name = string("op_13923_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_441_cast_fp16 = layer_norm(axes = normed_441_axes_0, epsilon = var_13923_to_fp16, x = input_493)[name = string("normed_441_cast_fp16")]; tensor normed_443_begin_0 = const()[name = string("normed_443_begin_0"), val = tensor([0, 0, 0, 0])]; tensor normed_443_end_0 = const()[name = string("normed_443_end_0"), val = tensor([1, 8, 64, 128])]; tensor normed_443_end_mask_0 = const()[name = string("normed_443_end_mask_0"), val = tensor([true, true, true, false])]; tensor normed_443 = slice_by_index(begin = normed_443_begin_0, end = normed_443_end_0, end_mask = normed_443_end_mask_0, x = normed_441_cast_fp16)[name = string("normed_443")]; tensor const_413 = const()[name = string("const_413"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335948992)))]; tensor k = mul(x = normed_443, y = const_413)[name = string("k")]; tensor var_13949 = mul(x = q, y = cos_1)[name = string("op_13949")]; tensor var_13954_begin_0 = const()[name = string("op_13954_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13954_end_0 = const()[name = string("op_13954_end_0"), val = tensor([1, 16, 64, 128])]; tensor var_13954_end_mask_0 = const()[name = string("op_13954_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13954 = slice_by_index(begin = var_13954_begin_0, end = var_13954_end_0, end_mask = var_13954_end_mask_0, x = q)[name = string("op_13954")]; fp16 const_414_promoted = const()[name = string("const_414_promoted"), val = fp16(-0x1p+0)]; tensor var_13955 = mul(x = var_13954, y = const_414_promoted)[name = string("op_13955")]; tensor var_13960_begin_0 = const()[name = string("op_13960_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13960_end_0 = const()[name = string("op_13960_end_0"), val = tensor([1, 16, 64, 64])]; tensor var_13960_end_mask_0 = const()[name = string("op_13960_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13960 = slice_by_index(begin = var_13960_begin_0, end = var_13960_end_0, end_mask = var_13960_end_mask_0, x = q)[name = string("op_13960")]; int32 var_13962 = const()[name = string("op_13962"), val = int32(-1)]; bool var_13963_interleave_0 = const()[name = string("op_13963_interleave_0"), val = bool(false)]; tensor var_13963 = concat(axis = var_13962, interleave = var_13963_interleave_0, values = (var_13955, var_13960))[name = string("op_13963")]; tensor var_13964 = mul(x = var_13963, y = sin_1)[name = string("op_13964")]; tensor query = add(x = var_13949, y = var_13964)[name = string("query")]; tensor var_13967 = mul(x = k, y = cos_1)[name = string("op_13967")]; tensor var_13972_begin_0 = const()[name = string("op_13972_begin_0"), val = tensor([0, 0, 0, 64])]; tensor var_13972_end_0 = const()[name = string("op_13972_end_0"), val = tensor([1, 8, 64, 128])]; tensor var_13972_end_mask_0 = const()[name = string("op_13972_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_13972 = slice_by_index(begin = var_13972_begin_0, end = var_13972_end_0, end_mask = var_13972_end_mask_0, x = k)[name = string("op_13972")]; fp16 const_415_promoted = const()[name = string("const_415_promoted"), val = fp16(-0x1p+0)]; tensor var_13973 = mul(x = var_13972, y = const_415_promoted)[name = string("op_13973")]; tensor var_13978_begin_0 = const()[name = string("op_13978_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_13978_end_0 = const()[name = string("op_13978_end_0"), val = tensor([1, 8, 64, 64])]; tensor var_13978_end_mask_0 = const()[name = string("op_13978_end_mask_0"), val = tensor([true, true, true, false])]; tensor var_13978 = slice_by_index(begin = var_13978_begin_0, end = var_13978_end_0, end_mask = var_13978_end_mask_0, x = k)[name = string("op_13978")]; int32 var_13980 = const()[name = string("op_13980"), val = int32(-1)]; bool var_13981_interleave_0 = const()[name = string("op_13981_interleave_0"), val = bool(false)]; tensor var_13981 = concat(axis = var_13980, interleave = var_13981_interleave_0, values = (var_13973, var_13978))[name = string("op_13981")]; tensor var_13982 = mul(x = var_13981, y = sin_1)[name = string("op_13982")]; tensor key = add(x = var_13967, y = var_13982)[name = string("key")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; int32 concat_488_axis_0 = const()[name = string("concat_488_axis_0"), val = int32(0)]; bool concat_488_interleave_0 = const()[name = string("concat_488_interleave_0"), val = bool(false)]; tensor concat_488 = concat(axis = concat_488_axis_0, interleave = concat_488_interleave_0, values = (expand_dims_324, expand_dims_325, current_pos, expand_dims_327))[name = string("concat_488")]; tensor concat_489_values1_0 = const()[name = string("concat_489_values1_0"), val = tensor([0])]; tensor concat_489_values3_0 = const()[name = string("concat_489_values3_0"), val = tensor([0])]; int32 concat_489_axis_0 = const()[name = string("concat_489_axis_0"), val = int32(0)]; bool concat_489_interleave_0 = const()[name = string("concat_489_interleave_0"), val = bool(false)]; tensor concat_489 = concat(axis = concat_489_axis_0, interleave = concat_489_interleave_0, values = (expand_dims_328, concat_489_values1_0, var_1746, concat_489_values3_0))[name = string("concat_489")]; tensor model_model_kv_cache_0_internal_tensor_assign_55_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_55_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_55_cast_fp16 = slice_update(begin = concat_488, begin_mask = model_model_kv_cache_0_internal_tensor_assign_55_begin_mask_0, end = concat_489, end_mask = model_model_kv_cache_0_internal_tensor_assign_55_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_55_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_55_stride_0, update = key, x = coreml_update_state_109)[name = string("model_model_kv_cache_0_internal_tensor_assign_55_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_55_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_222_write_state")]; tensor coreml_update_state_110 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_222")]; tensor expand_dims_330 = const()[name = string("expand_dims_330"), val = tensor([55])]; tensor expand_dims_331 = const()[name = string("expand_dims_331"), val = tensor([0])]; tensor expand_dims_333 = const()[name = string("expand_dims_333"), val = tensor([0])]; tensor expand_dims_334 = const()[name = string("expand_dims_334"), val = tensor([56])]; int32 concat_492_axis_0 = const()[name = string("concat_492_axis_0"), val = int32(0)]; bool concat_492_interleave_0 = const()[name = string("concat_492_interleave_0"), val = bool(false)]; tensor concat_492 = concat(axis = concat_492_axis_0, interleave = concat_492_interleave_0, values = (expand_dims_330, expand_dims_331, current_pos, expand_dims_333))[name = string("concat_492")]; tensor concat_493_values1_0 = const()[name = string("concat_493_values1_0"), val = tensor([0])]; tensor concat_493_values3_0 = const()[name = string("concat_493_values3_0"), val = tensor([0])]; int32 concat_493_axis_0 = const()[name = string("concat_493_axis_0"), val = int32(0)]; bool concat_493_interleave_0 = const()[name = string("concat_493_interleave_0"), val = bool(false)]; tensor concat_493 = concat(axis = concat_493_axis_0, interleave = concat_493_interleave_0, values = (expand_dims_334, concat_493_values1_0, var_1746, concat_493_values3_0))[name = string("concat_493")]; tensor model_model_kv_cache_0_internal_tensor_assign_56_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_56_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_56_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_271 = transpose(perm = var_13891, x = var_13886)[name = string("transpose_5")]; tensor model_model_kv_cache_0_internal_tensor_assign_56_cast_fp16 = slice_update(begin = concat_492, begin_mask = model_model_kv_cache_0_internal_tensor_assign_56_begin_mask_0, end = concat_493, end_mask = model_model_kv_cache_0_internal_tensor_assign_56_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_56_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_56_stride_0, update = value_271, x = coreml_update_state_110)[name = string("model_model_kv_cache_0_internal_tensor_assign_56_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_56_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_223_write_state")]; tensor coreml_update_state_111 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_223")]; tensor var_14053_begin_0 = const()[name = string("op_14053_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_14053_end_0 = const()[name = string("op_14053_end_0"), val = tensor([28, 8, 1536, 128])]; tensor var_14053_end_mask_0 = const()[name = string("op_14053_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_14053_cast_fp16 = slice_by_index(begin = var_14053_begin_0, end = var_14053_end_0, end_mask = var_14053_end_mask_0, x = coreml_update_state_111)[name = string("op_14053_cast_fp16")]; tensor key_cache_axes_0 = const()[name = string("key_cache_axes_0"), val = tensor([0])]; tensor key_cache_cast_fp16 = squeeze(axes = key_cache_axes_0, x = var_14053_cast_fp16)[name = string("key_cache_cast_fp16")]; tensor var_14060_begin_0 = const()[name = string("op_14060_begin_0"), val = tensor([55, 0, 0, 0])]; tensor var_14060_end_0 = const()[name = string("op_14060_end_0"), val = tensor([1, 8, 1536, 128])]; tensor var_14060_end_mask_0 = const()[name = string("op_14060_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_14060_cast_fp16 = slice_by_index(begin = var_14060_begin_0, end = var_14060_end_0, end_mask = var_14060_end_mask_0, x = coreml_update_state_111)[name = string("op_14060_cast_fp16")]; tensor value_cache_axes_0 = const()[name = string("value_cache_axes_0"), val = tensor([0])]; tensor value_cache_cast_fp16 = squeeze(axes = value_cache_axes_0, x = var_14060_cast_fp16)[name = string("value_cache_cast_fp16")]; tensor var_14084_axes_0 = const()[name = string("op_14084_axes_0"), val = tensor([1])]; tensor var_14084_cast_fp16 = expand_dims(axes = var_14084_axes_0, x = key_cache_cast_fp16)[name = string("op_14084_cast_fp16")]; tensor var_14089 = const()[name = string("op_14089"), val = tensor([1, 2, 1, 1])]; tensor value_275_cast_fp16 = tile(reps = var_14089, x = var_14084_cast_fp16)[name = string("value_275_cast_fp16")]; tensor var_14095 = const()[name = string("op_14095"), val = tensor([1, 16, 1536, 128])]; tensor key_states_cast_fp16 = reshape(shape = var_14095, x = value_275_cast_fp16)[name = string("key_states_cast_fp16")]; tensor var_14098_axes_0 = const()[name = string("op_14098_axes_0"), val = tensor([1])]; tensor var_14098_cast_fp16 = expand_dims(axes = var_14098_axes_0, x = value_cache_cast_fp16)[name = string("op_14098_cast_fp16")]; tensor var_14103 = const()[name = string("op_14103"), val = tensor([1, 2, 1, 1])]; tensor value_cast_fp16 = tile(reps = var_14103, x = var_14098_cast_fp16)[name = string("value_cast_fp16")]; bool var_14124_transpose_x_0 = const()[name = string("op_14124_transpose_x_0"), val = bool(false)]; bool var_14124_transpose_y_0 = const()[name = string("op_14124_transpose_y_0"), val = bool(true)]; tensor var_14124 = matmul(transpose_x = var_14124_transpose_x_0, transpose_y = var_14124_transpose_y_0, x = query, y = key_states_cast_fp16)[name = string("op_14124")]; fp16 var_14125_to_fp16 = const()[name = string("op_14125_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attention_109_cast_fp16 = mul(x = var_14124, y = var_14125_to_fp16)[name = string("attention_109_cast_fp16")]; tensor attention_cast_fp16 = add(x = attention_109_cast_fp16, y = causal_mask)[name = string("attention_cast_fp16")]; int32 var_14134 = const()[name = string("op_14134"), val = int32(-1)]; tensor var_14136_cast_fp16 = softmax(axis = var_14134, x = attention_cast_fp16)[name = string("op_14136_cast_fp16")]; tensor concat_498 = const()[name = string("concat_498"), val = tensor([16, 64, 1536])]; tensor reshape_81_cast_fp16 = reshape(shape = concat_498, x = var_14136_cast_fp16)[name = string("reshape_81_cast_fp16")]; tensor concat_499 = const()[name = string("concat_499"), val = tensor([16, 1536, 128])]; tensor reshape_82_cast_fp16 = reshape(shape = concat_499, x = value_cast_fp16)[name = string("reshape_82_cast_fp16")]; bool matmul_27_transpose_x_0 = const()[name = string("matmul_27_transpose_x_0"), val = bool(false)]; bool matmul_27_transpose_y_0 = const()[name = string("matmul_27_transpose_y_0"), val = bool(false)]; tensor matmul_27_cast_fp16 = matmul(transpose_x = matmul_27_transpose_x_0, transpose_y = matmul_27_transpose_y_0, x = reshape_81_cast_fp16, y = reshape_82_cast_fp16)[name = string("matmul_27_cast_fp16")]; tensor concat_503 = const()[name = string("concat_503"), val = tensor([1, 16, 64, 128])]; tensor reshape_83_cast_fp16 = reshape(shape = concat_503, x = matmul_27_cast_fp16)[name = string("reshape_83_cast_fp16")]; tensor var_14148_perm_0 = const()[name = string("op_14148_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_14154 = const()[name = string("op_14154"), val = tensor([1, 64, 2048])]; tensor var_14148_cast_fp16 = transpose(perm = var_14148_perm_0, x = reshape_83_cast_fp16)[name = string("transpose_4")]; tensor output_165_cast_fp16 = reshape(shape = var_14154, x = var_14148_cast_fp16)[name = string("output_165_cast_fp16")]; tensor var_14159 = const()[name = string("op_14159"), val = tensor([0, 2, 1])]; string var_14175_pad_type_0 = const()[name = string("op_14175_pad_type_0"), val = string("valid")]; int32 var_14175_groups_0 = const()[name = string("op_14175_groups_0"), val = int32(1)]; tensor var_14175_strides_0 = const()[name = string("op_14175_strides_0"), val = tensor([1])]; tensor var_14175_pad_0 = const()[name = string("op_14175_pad_0"), val = tensor([0, 0])]; tensor var_14175_dilations_0 = const()[name = string("op_14175_dilations_0"), val = tensor([1])]; tensor squeeze_27_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335949312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337522240))))[name = string("squeeze_27_cast_fp16_to_fp32_to_fp16_palettized")]; tensor var_14160_cast_fp16 = transpose(perm = var_14159, x = output_165_cast_fp16)[name = string("transpose_3")]; tensor var_14175_cast_fp16 = conv(dilations = var_14175_dilations_0, groups = var_14175_groups_0, pad = var_14175_pad_0, pad_type = var_14175_pad_type_0, strides = var_14175_strides_0, weight = squeeze_27_cast_fp16_to_fp32_to_fp16_palettized, x = var_14160_cast_fp16)[name = string("op_14175_cast_fp16")]; tensor var_14179 = const()[name = string("op_14179"), val = tensor([0, 2, 1])]; tensor attn_output_cast_fp16 = transpose(perm = var_14179, x = var_14175_cast_fp16)[name = string("transpose_2")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_271_cast_fp16, y = attn_output_cast_fp16)[name = string("hidden_states_cast_fp16")]; int32 var_14194 = const()[name = string("op_14194"), val = int32(-1)]; fp16 const_417_promoted_to_fp16 = const()[name = string("const_417_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_14196_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_417_promoted_to_fp16)[name = string("op_14196_cast_fp16")]; bool input_497_interleave_0 = const()[name = string("input_497_interleave_0"), val = bool(false)]; tensor input_497_cast_fp16 = concat(axis = var_14194, interleave = input_497_interleave_0, values = (hidden_states_cast_fp16, var_14196_cast_fp16))[name = string("input_497_cast_fp16")]; tensor normed_445_axes_0 = const()[name = string("normed_445_axes_0"), val = tensor([-1])]; fp16 var_14191_to_fp16 = const()[name = string("op_14191_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_445_cast_fp16 = layer_norm(axes = normed_445_axes_0, epsilon = var_14191_to_fp16, x = input_497_cast_fp16)[name = string("normed_445_cast_fp16")]; tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 64, 1024])]; tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_445_cast_fp16)[name = string("normed_cast_fp16")]; tensor const_419_promoted_to_fp16 = const()[name = string("const_419_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337538688)))]; tensor x_109_cast_fp16 = mul(x = normed_cast_fp16, y = const_419_promoted_to_fp16)[name = string("x_109_cast_fp16")]; tensor var_14216 = const()[name = string("op_14216"), val = tensor([0, 2, 1])]; tensor input_499_axes_0 = const()[name = string("input_499_axes_0"), val = tensor([2])]; tensor var_14217 = transpose(perm = var_14216, x = x_109_cast_fp16)[name = string("transpose_1")]; tensor input_499 = expand_dims(axes = input_499_axes_0, x = var_14217)[name = string("input_499")]; string input_501_pad_type_0 = const()[name = string("input_501_pad_type_0"), val = string("valid")]; tensor input_501_strides_0 = const()[name = string("input_501_strides_0"), val = tensor([1, 1])]; tensor input_501_pad_0 = const()[name = string("input_501_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_501_dilations_0 = const()[name = string("input_501_dilations_0"), val = tensor([1, 1])]; int32 input_501_groups_0 = const()[name = string("input_501_groups_0"), val = int32(1)]; tensor input_501 = conv(dilations = input_501_dilations_0, groups = input_501_groups_0, pad = input_501_pad_0, pad_type = input_501_pad_type_0, strides = input_501_strides_0, weight = model_model_layers_27_mlp_gate_proj_weight_palettized, x = input_499)[name = string("input_501")]; string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_27_mlp_up_proj_weight_palettized, x = input_499)[name = string("b")]; tensor c = silu(x = input_501)[name = string("c")]; tensor input = mul(x = c, y = b)[name = string("input")]; string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_27_mlp_down_proj_weight_palettized, x = input)[name = string("e")]; tensor var_14239_axes_0 = const()[name = string("op_14239_axes_0"), val = tensor([2])]; tensor var_14239 = squeeze(axes = var_14239_axes_0, x = e)[name = string("op_14239")]; tensor var_14240 = const()[name = string("op_14240"), val = tensor([0, 2, 1])]; tensor var_14241 = transpose(perm = var_14240, x = var_14239)[name = string("transpose_0")]; tensor out_cast_fp16 = add(x = hidden_states_cast_fp16, y = var_14241)[name = string("out_cast_fp16")]; tensor var_14253_begin_0 = const()[name = string("op_14253_begin_0"), val = tensor([0, 0, 0])]; tensor var_14253_end_0 = const()[name = string("op_14253_end_0"), val = tensor([1, 1, 1024])]; tensor var_14253_end_mask_0 = const()[name = string("op_14253_end_mask_0"), val = tensor([true, false, true])]; tensor output_hidden_states = slice_by_index(begin = var_14253_begin_0, end = var_14253_end_0, end_mask = var_14253_end_mask_0, x = out_cast_fp16)[name = string("op_14253_cast_fp16")]; } -> (output_hidden_states); }