program(1.0) [buildInfo = dict, tensor>({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}, {"coremltools-component-torch", "2.8.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})] { func main(tensor encoder_attention_mask, tensor encoder_hidden_states, tensor input_ids) [FlexibleShapeInformation = tuple, dict, tensor>>, tuple, dict, list, ?>>>>((("DefaultShapes", {{"encoder_attention_mask", [1, 1]}, {"encoder_hidden_states", [1, 1, 1024]}}), ("RangeDims", {{"encoder_attention_mask", [[1, 1], [1, 1024]]}, {"encoder_hidden_states", [[1, 1], [1, 1024], [1024, 1024]]}})))] { tensor decoder_embed_tokens_weight = const()[name = tensor("decoder_embed_tokens_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(64)))]; tensor decoder_embed_positions_weights = const()[name = tensor("decoder_embed_positions_weights"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1049419904)))]; tensor decoder_layers_0_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_0_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1053622464)))]; tensor decoder_layers_0_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_0_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1053626624)))]; tensor decoder_layers_0_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1053630784)))]; tensor decoder_layers_0_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_0_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1053634944)))]; tensor decoder_layers_0_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1057829312)))]; tensor decoder_layers_0_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_0_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1057833472)))]; tensor decoder_layers_0_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1062027840)))]; tensor decoder_layers_0_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_0_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1062032000)))]; tensor decoder_layers_0_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1066226368)))]; tensor decoder_layers_0_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_0_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1066230528)))]; tensor decoder_layers_0_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_0_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1070424896)))]; tensor decoder_layers_0_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_0_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1070429056)))]; tensor decoder_layers_0_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1070433216)))]; tensor decoder_layers_0_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_0_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1070437376)))]; tensor decoder_layers_0_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1074631744)))]; tensor decoder_layers_0_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_0_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1074635904)))]; tensor decoder_layers_0_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1078830272)))]; tensor decoder_layers_0_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_0_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1078834432)))]; tensor decoder_layers_0_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1083028800)))]; tensor decoder_layers_0_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_0_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1083032960)))]; tensor decoder_layers_0_final_layer_norm_bias = const()[name = tensor("decoder_layers_0_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1087227328)))]; tensor decoder_layers_0_final_layer_norm_weight = const()[name = tensor("decoder_layers_0_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1087231488)))]; tensor decoder_layers_0_fc1_bias = const()[name = tensor("decoder_layers_0_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1087235648)))]; tensor decoder_layers_0_fc1_weight = const()[name = tensor("decoder_layers_0_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1087252096)))]; tensor decoder_layers_0_fc2_bias = const()[name = tensor("decoder_layers_0_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1104029376)))]; tensor decoder_layers_0_fc2_weight = const()[name = tensor("decoder_layers_0_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1104033536)))]; tensor decoder_layers_1_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_1_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1120810816)))]; tensor decoder_layers_1_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_1_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1120814976)))]; tensor decoder_layers_1_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1120819136)))]; tensor decoder_layers_1_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_1_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1120823296)))]; tensor decoder_layers_1_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1125017664)))]; tensor decoder_layers_1_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_1_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1125021824)))]; tensor decoder_layers_1_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1129216192)))]; tensor decoder_layers_1_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_1_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1129220352)))]; tensor decoder_layers_1_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1133414720)))]; tensor decoder_layers_1_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_1_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1133418880)))]; tensor decoder_layers_1_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_1_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1137613248)))]; tensor decoder_layers_1_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_1_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1137617408)))]; tensor decoder_layers_1_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1137621568)))]; tensor decoder_layers_1_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_1_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1137625728)))]; tensor decoder_layers_1_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1141820096)))]; tensor decoder_layers_1_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_1_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1141824256)))]; tensor decoder_layers_1_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1146018624)))]; tensor decoder_layers_1_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_1_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1146022784)))]; tensor decoder_layers_1_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1150217152)))]; tensor decoder_layers_1_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_1_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1150221312)))]; tensor decoder_layers_1_final_layer_norm_bias = const()[name = tensor("decoder_layers_1_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1154415680)))]; tensor decoder_layers_1_final_layer_norm_weight = const()[name = tensor("decoder_layers_1_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1154419840)))]; tensor decoder_layers_1_fc1_bias = const()[name = tensor("decoder_layers_1_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1154424000)))]; tensor decoder_layers_1_fc1_weight = const()[name = tensor("decoder_layers_1_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1154440448)))]; tensor decoder_layers_1_fc2_bias = const()[name = tensor("decoder_layers_1_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1171217728)))]; tensor decoder_layers_1_fc2_weight = const()[name = tensor("decoder_layers_1_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1171221888)))]; tensor decoder_layers_2_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_2_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1187999168)))]; tensor decoder_layers_2_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_2_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1188003328)))]; tensor decoder_layers_2_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1188007488)))]; tensor decoder_layers_2_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_2_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1188011648)))]; tensor decoder_layers_2_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1192206016)))]; tensor decoder_layers_2_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_2_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1192210176)))]; tensor decoder_layers_2_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1196404544)))]; tensor decoder_layers_2_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_2_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1196408704)))]; tensor decoder_layers_2_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1200603072)))]; tensor decoder_layers_2_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_2_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1200607232)))]; tensor decoder_layers_2_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_2_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1204801600)))]; tensor decoder_layers_2_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_2_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1204805760)))]; tensor decoder_layers_2_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1204809920)))]; tensor decoder_layers_2_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_2_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1204814080)))]; tensor decoder_layers_2_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1209008448)))]; tensor decoder_layers_2_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_2_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1209012608)))]; tensor decoder_layers_2_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1213206976)))]; tensor decoder_layers_2_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_2_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1213211136)))]; tensor decoder_layers_2_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1217405504)))]; tensor decoder_layers_2_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_2_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1217409664)))]; tensor decoder_layers_2_final_layer_norm_bias = const()[name = tensor("decoder_layers_2_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1221604032)))]; tensor decoder_layers_2_final_layer_norm_weight = const()[name = tensor("decoder_layers_2_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1221608192)))]; tensor decoder_layers_2_fc1_bias = const()[name = tensor("decoder_layers_2_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1221612352)))]; tensor decoder_layers_2_fc1_weight = const()[name = tensor("decoder_layers_2_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1221628800)))]; tensor decoder_layers_2_fc2_bias = const()[name = tensor("decoder_layers_2_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1238406080)))]; tensor decoder_layers_2_fc2_weight = const()[name = tensor("decoder_layers_2_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1238410240)))]; tensor decoder_layers_3_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_3_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1255187520)))]; tensor decoder_layers_3_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_3_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1255191680)))]; tensor decoder_layers_3_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1255195840)))]; tensor decoder_layers_3_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_3_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1255200000)))]; tensor decoder_layers_3_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1259394368)))]; tensor decoder_layers_3_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_3_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1259398528)))]; tensor decoder_layers_3_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1263592896)))]; tensor decoder_layers_3_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_3_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1263597056)))]; tensor decoder_layers_3_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1267791424)))]; tensor decoder_layers_3_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_3_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1267795584)))]; tensor decoder_layers_3_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_3_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1271989952)))]; tensor decoder_layers_3_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_3_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1271994112)))]; tensor decoder_layers_3_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1271998272)))]; tensor decoder_layers_3_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_3_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1272002432)))]; tensor decoder_layers_3_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1276196800)))]; tensor decoder_layers_3_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_3_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1276200960)))]; tensor decoder_layers_3_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1280395328)))]; tensor decoder_layers_3_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_3_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1280399488)))]; tensor decoder_layers_3_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1284593856)))]; tensor decoder_layers_3_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_3_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1284598016)))]; tensor decoder_layers_3_final_layer_norm_bias = const()[name = tensor("decoder_layers_3_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1288792384)))]; tensor decoder_layers_3_final_layer_norm_weight = const()[name = tensor("decoder_layers_3_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1288796544)))]; tensor decoder_layers_3_fc1_bias = const()[name = tensor("decoder_layers_3_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1288800704)))]; tensor decoder_layers_3_fc1_weight = const()[name = tensor("decoder_layers_3_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1288817152)))]; tensor decoder_layers_3_fc2_bias = const()[name = tensor("decoder_layers_3_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1305594432)))]; tensor decoder_layers_3_fc2_weight = const()[name = tensor("decoder_layers_3_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1305598592)))]; tensor decoder_layers_4_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_4_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1322375872)))]; tensor decoder_layers_4_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_4_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1322380032)))]; tensor decoder_layers_4_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1322384192)))]; tensor decoder_layers_4_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_4_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1322388352)))]; tensor decoder_layers_4_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1326582720)))]; tensor decoder_layers_4_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_4_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1326586880)))]; tensor decoder_layers_4_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1330781248)))]; tensor decoder_layers_4_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_4_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1330785408)))]; tensor decoder_layers_4_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1334979776)))]; tensor decoder_layers_4_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_4_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1334983936)))]; tensor decoder_layers_4_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_4_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1339178304)))]; tensor decoder_layers_4_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_4_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1339182464)))]; tensor decoder_layers_4_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1339186624)))]; tensor decoder_layers_4_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_4_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1339190784)))]; tensor decoder_layers_4_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1343385152)))]; tensor decoder_layers_4_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_4_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1343389312)))]; tensor decoder_layers_4_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1347583680)))]; tensor decoder_layers_4_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_4_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1347587840)))]; tensor decoder_layers_4_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1351782208)))]; tensor decoder_layers_4_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_4_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1351786368)))]; tensor decoder_layers_4_final_layer_norm_bias = const()[name = tensor("decoder_layers_4_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1355980736)))]; tensor decoder_layers_4_final_layer_norm_weight = const()[name = tensor("decoder_layers_4_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1355984896)))]; tensor decoder_layers_4_fc1_bias = const()[name = tensor("decoder_layers_4_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1355989056)))]; tensor decoder_layers_4_fc1_weight = const()[name = tensor("decoder_layers_4_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1356005504)))]; tensor decoder_layers_4_fc2_bias = const()[name = tensor("decoder_layers_4_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1372782784)))]; tensor decoder_layers_4_fc2_weight = const()[name = tensor("decoder_layers_4_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1372786944)))]; tensor decoder_layers_5_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_5_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1389564224)))]; tensor decoder_layers_5_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_5_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1389568384)))]; tensor decoder_layers_5_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1389572544)))]; tensor decoder_layers_5_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_5_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1389576704)))]; tensor decoder_layers_5_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1393771072)))]; tensor decoder_layers_5_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_5_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1393775232)))]; tensor decoder_layers_5_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1397969600)))]; tensor decoder_layers_5_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_5_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1397973760)))]; tensor decoder_layers_5_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1402168128)))]; tensor decoder_layers_5_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_5_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1402172288)))]; tensor decoder_layers_5_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_5_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1406366656)))]; tensor decoder_layers_5_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_5_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1406370816)))]; tensor decoder_layers_5_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1406374976)))]; tensor decoder_layers_5_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_5_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1406379136)))]; tensor decoder_layers_5_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1410573504)))]; tensor decoder_layers_5_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_5_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1410577664)))]; tensor decoder_layers_5_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1414772032)))]; tensor decoder_layers_5_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_5_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1414776192)))]; tensor decoder_layers_5_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1418970560)))]; tensor decoder_layers_5_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_5_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1418974720)))]; tensor decoder_layers_5_final_layer_norm_bias = const()[name = tensor("decoder_layers_5_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1423169088)))]; tensor decoder_layers_5_final_layer_norm_weight = const()[name = tensor("decoder_layers_5_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1423173248)))]; tensor decoder_layers_5_fc1_bias = const()[name = tensor("decoder_layers_5_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1423177408)))]; tensor decoder_layers_5_fc1_weight = const()[name = tensor("decoder_layers_5_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1423193856)))]; tensor decoder_layers_5_fc2_bias = const()[name = tensor("decoder_layers_5_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1439971136)))]; tensor decoder_layers_5_fc2_weight = const()[name = tensor("decoder_layers_5_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1439975296)))]; tensor decoder_layers_6_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_6_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1456752576)))]; tensor decoder_layers_6_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_6_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1456756736)))]; tensor decoder_layers_6_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1456760896)))]; tensor decoder_layers_6_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_6_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1456765056)))]; tensor decoder_layers_6_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1460959424)))]; tensor decoder_layers_6_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_6_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1460963584)))]; tensor decoder_layers_6_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1465157952)))]; tensor decoder_layers_6_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_6_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1465162112)))]; tensor decoder_layers_6_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1469356480)))]; tensor decoder_layers_6_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_6_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1469360640)))]; tensor decoder_layers_6_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_6_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1473555008)))]; tensor decoder_layers_6_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_6_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1473559168)))]; tensor decoder_layers_6_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1473563328)))]; tensor decoder_layers_6_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_6_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1473567488)))]; tensor decoder_layers_6_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1477761856)))]; tensor decoder_layers_6_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_6_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1477766016)))]; tensor decoder_layers_6_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1481960384)))]; tensor decoder_layers_6_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_6_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1481964544)))]; tensor decoder_layers_6_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1486158912)))]; tensor decoder_layers_6_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_6_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1486163072)))]; tensor decoder_layers_6_final_layer_norm_bias = const()[name = tensor("decoder_layers_6_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1490357440)))]; tensor decoder_layers_6_final_layer_norm_weight = const()[name = tensor("decoder_layers_6_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1490361600)))]; tensor decoder_layers_6_fc1_bias = const()[name = tensor("decoder_layers_6_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1490365760)))]; tensor decoder_layers_6_fc1_weight = const()[name = tensor("decoder_layers_6_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1490382208)))]; tensor decoder_layers_6_fc2_bias = const()[name = tensor("decoder_layers_6_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1507159488)))]; tensor decoder_layers_6_fc2_weight = const()[name = tensor("decoder_layers_6_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1507163648)))]; tensor decoder_layers_7_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_7_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1523940928)))]; tensor decoder_layers_7_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_7_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1523945088)))]; tensor decoder_layers_7_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1523949248)))]; tensor decoder_layers_7_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_7_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1523953408)))]; tensor decoder_layers_7_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1528147776)))]; tensor decoder_layers_7_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_7_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1528151936)))]; tensor decoder_layers_7_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1532346304)))]; tensor decoder_layers_7_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_7_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1532350464)))]; tensor decoder_layers_7_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1536544832)))]; tensor decoder_layers_7_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_7_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1536548992)))]; tensor decoder_layers_7_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_7_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1540743360)))]; tensor decoder_layers_7_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_7_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1540747520)))]; tensor decoder_layers_7_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1540751680)))]; tensor decoder_layers_7_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_7_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1540755840)))]; tensor decoder_layers_7_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1544950208)))]; tensor decoder_layers_7_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_7_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1544954368)))]; tensor decoder_layers_7_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1549148736)))]; tensor decoder_layers_7_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_7_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1549152896)))]; tensor decoder_layers_7_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1553347264)))]; tensor decoder_layers_7_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_7_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1553351424)))]; tensor decoder_layers_7_final_layer_norm_bias = const()[name = tensor("decoder_layers_7_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1557545792)))]; tensor decoder_layers_7_final_layer_norm_weight = const()[name = tensor("decoder_layers_7_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1557549952)))]; tensor decoder_layers_7_fc1_bias = const()[name = tensor("decoder_layers_7_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1557554112)))]; tensor decoder_layers_7_fc1_weight = const()[name = tensor("decoder_layers_7_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1557570560)))]; tensor decoder_layers_7_fc2_bias = const()[name = tensor("decoder_layers_7_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1574347840)))]; tensor decoder_layers_7_fc2_weight = const()[name = tensor("decoder_layers_7_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1574352000)))]; tensor decoder_layers_8_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_8_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1591129280)))]; tensor decoder_layers_8_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_8_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1591133440)))]; tensor decoder_layers_8_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1591137600)))]; tensor decoder_layers_8_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_8_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1591141760)))]; tensor decoder_layers_8_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1595336128)))]; tensor decoder_layers_8_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_8_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1595340288)))]; tensor decoder_layers_8_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1599534656)))]; tensor decoder_layers_8_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_8_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1599538816)))]; tensor decoder_layers_8_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1603733184)))]; tensor decoder_layers_8_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_8_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1603737344)))]; tensor decoder_layers_8_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_8_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1607931712)))]; tensor decoder_layers_8_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_8_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1607935872)))]; tensor decoder_layers_8_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1607940032)))]; tensor decoder_layers_8_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_8_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1607944192)))]; tensor decoder_layers_8_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1612138560)))]; tensor decoder_layers_8_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_8_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1612142720)))]; tensor decoder_layers_8_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1616337088)))]; tensor decoder_layers_8_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_8_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1616341248)))]; tensor decoder_layers_8_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1620535616)))]; tensor decoder_layers_8_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_8_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1620539776)))]; tensor decoder_layers_8_final_layer_norm_bias = const()[name = tensor("decoder_layers_8_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1624734144)))]; tensor decoder_layers_8_final_layer_norm_weight = const()[name = tensor("decoder_layers_8_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1624738304)))]; tensor decoder_layers_8_fc1_bias = const()[name = tensor("decoder_layers_8_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1624742464)))]; tensor decoder_layers_8_fc1_weight = const()[name = tensor("decoder_layers_8_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1624758912)))]; tensor decoder_layers_8_fc2_bias = const()[name = tensor("decoder_layers_8_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1641536192)))]; tensor decoder_layers_8_fc2_weight = const()[name = tensor("decoder_layers_8_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1641540352)))]; tensor decoder_layers_9_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_9_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1658317632)))]; tensor decoder_layers_9_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_9_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1658321792)))]; tensor decoder_layers_9_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1658325952)))]; tensor decoder_layers_9_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_9_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1658330112)))]; tensor decoder_layers_9_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1662524480)))]; tensor decoder_layers_9_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_9_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1662528640)))]; tensor decoder_layers_9_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1666723008)))]; tensor decoder_layers_9_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_9_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1666727168)))]; tensor decoder_layers_9_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1670921536)))]; tensor decoder_layers_9_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_9_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1670925696)))]; tensor decoder_layers_9_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_9_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1675120064)))]; tensor decoder_layers_9_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_9_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1675124224)))]; tensor decoder_layers_9_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1675128384)))]; tensor decoder_layers_9_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_9_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1675132544)))]; tensor decoder_layers_9_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1679326912)))]; tensor decoder_layers_9_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_9_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1679331072)))]; tensor decoder_layers_9_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1683525440)))]; tensor decoder_layers_9_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_9_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1683529600)))]; tensor decoder_layers_9_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1687723968)))]; tensor decoder_layers_9_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_9_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1687728128)))]; tensor decoder_layers_9_final_layer_norm_bias = const()[name = tensor("decoder_layers_9_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1691922496)))]; tensor decoder_layers_9_final_layer_norm_weight = const()[name = tensor("decoder_layers_9_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1691926656)))]; tensor decoder_layers_9_fc1_bias = const()[name = tensor("decoder_layers_9_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1691930816)))]; tensor decoder_layers_9_fc1_weight = const()[name = tensor("decoder_layers_9_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1691947264)))]; tensor decoder_layers_9_fc2_bias = const()[name = tensor("decoder_layers_9_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1708724544)))]; tensor decoder_layers_9_fc2_weight = const()[name = tensor("decoder_layers_9_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1708728704)))]; tensor decoder_layers_10_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_10_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1725505984)))]; tensor decoder_layers_10_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_10_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1725510144)))]; tensor decoder_layers_10_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1725514304)))]; tensor decoder_layers_10_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_10_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1725518464)))]; tensor decoder_layers_10_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1729712832)))]; tensor decoder_layers_10_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_10_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1729716992)))]; tensor decoder_layers_10_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1733911360)))]; tensor decoder_layers_10_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_10_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1733915520)))]; tensor decoder_layers_10_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1738109888)))]; tensor decoder_layers_10_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_10_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1738114048)))]; tensor decoder_layers_10_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_10_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1742308416)))]; tensor decoder_layers_10_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_10_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1742312576)))]; tensor decoder_layers_10_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1742316736)))]; tensor decoder_layers_10_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_10_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1742320896)))]; tensor decoder_layers_10_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1746515264)))]; tensor decoder_layers_10_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_10_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1746519424)))]; tensor decoder_layers_10_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1750713792)))]; tensor decoder_layers_10_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_10_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1750717952)))]; tensor decoder_layers_10_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1754912320)))]; tensor decoder_layers_10_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_10_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1754916480)))]; tensor decoder_layers_10_final_layer_norm_bias = const()[name = tensor("decoder_layers_10_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1759110848)))]; tensor decoder_layers_10_final_layer_norm_weight = const()[name = tensor("decoder_layers_10_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1759115008)))]; tensor decoder_layers_10_fc1_bias = const()[name = tensor("decoder_layers_10_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1759119168)))]; tensor decoder_layers_10_fc1_weight = const()[name = tensor("decoder_layers_10_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1759135616)))]; tensor decoder_layers_10_fc2_bias = const()[name = tensor("decoder_layers_10_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1775912896)))]; tensor decoder_layers_10_fc2_weight = const()[name = tensor("decoder_layers_10_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1775917056)))]; tensor decoder_layers_11_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_11_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1792694336)))]; tensor decoder_layers_11_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_11_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1792698496)))]; tensor decoder_layers_11_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1792702656)))]; tensor decoder_layers_11_self_attn_q_proj_weight = const()[name = tensor("decoder_layers_11_self_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1792706816)))]; tensor decoder_layers_11_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1796901184)))]; tensor decoder_layers_11_self_attn_k_proj_weight = const()[name = tensor("decoder_layers_11_self_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1796905344)))]; tensor decoder_layers_11_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1801099712)))]; tensor decoder_layers_11_self_attn_v_proj_weight = const()[name = tensor("decoder_layers_11_self_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1801103872)))]; tensor decoder_layers_11_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1805298240)))]; tensor decoder_layers_11_self_attn_out_proj_weight = const()[name = tensor("decoder_layers_11_self_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1805302400)))]; tensor decoder_layers_11_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_11_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1809496768)))]; tensor decoder_layers_11_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_11_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1809500928)))]; tensor decoder_layers_11_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1809505088)))]; tensor decoder_layers_11_encoder_attn_q_proj_weight = const()[name = tensor("decoder_layers_11_encoder_attn_q_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1809509248)))]; tensor decoder_layers_11_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1813703616)))]; tensor decoder_layers_11_encoder_attn_k_proj_weight = const()[name = tensor("decoder_layers_11_encoder_attn_k_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1813707776)))]; tensor decoder_layers_11_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1817902144)))]; tensor decoder_layers_11_encoder_attn_v_proj_weight = const()[name = tensor("decoder_layers_11_encoder_attn_v_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1817906304)))]; tensor decoder_layers_11_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1822100672)))]; tensor decoder_layers_11_encoder_attn_out_proj_weight = const()[name = tensor("decoder_layers_11_encoder_attn_out_proj_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1822104832)))]; tensor decoder_layers_11_final_layer_norm_bias = const()[name = tensor("decoder_layers_11_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1826299200)))]; tensor decoder_layers_11_final_layer_norm_weight = const()[name = tensor("decoder_layers_11_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1826303360)))]; tensor decoder_layers_11_fc1_bias = const()[name = tensor("decoder_layers_11_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1826307520)))]; tensor decoder_layers_11_fc1_weight = const()[name = tensor("decoder_layers_11_fc1_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1826323968)))]; tensor decoder_layers_11_fc2_bias = const()[name = tensor("decoder_layers_11_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1843101248)))]; tensor decoder_layers_11_fc2_weight = const()[name = tensor("decoder_layers_11_fc2_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1843105408)))]; tensor decoder_layer_norm_bias = const()[name = tensor("decoder_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1859882688)))]; tensor decoder_layer_norm_weight = const()[name = tensor("decoder_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1859886848)))]; tensor var_9 = const()[name = tensor("op_9"), val = tensor(0x1.4f8b58p-17)]; tensor var_11 = const()[name = tensor("op_11"), val = tensor(0x1p-3)]; tensor var_13 = const()[name = tensor("op_13"), val = tensor(-2)]; tensor var_24 = const()[name = tensor("op_24"), val = tensor(-0x1.fffffep+127)]; tensor var_27 = const()[name = tensor("op_27"), val = tensor(0)]; tensor var_30 = const()[name = tensor("op_30"), val = tensor(1)]; tensor const_0 = const()[name = tensor("const_0"), val = tensor(2)]; tensor var_62_axis_0 = const()[name = tensor("op_62_axis_0"), val = tensor(0)]; tensor var_62_batch_dims_0 = const()[name = tensor("op_62_batch_dims_0"), val = tensor(0)]; tensor var_62 = gather(axis = var_62_axis_0, batch_dims = var_62_batch_dims_0, indices = input_ids, x = decoder_embed_tokens_weight)[name = tensor("op_62")]; tensor var_63 = const()[name = tensor("op_63"), val = tensor(0x1p+5)]; tensor inputs_embeds = mul(x = var_62, y = var_63)[name = tensor("inputs_embeds")]; tensor shape_1 = const()[name = tensor("shape_1"), val = tensor([1, 1, 2, 2])]; tensor reshape_1 = const()[name = tensor("reshape_1"), val = tensor([0, 1, 2, 3])]; tensor reshape_2 = const()[name = tensor("reshape_2"), val = tensor([0x0p+0, -0x1.fffffep+127, 0x0p+0, 0x0p+0])]; tensor reshape_3 = const()[name = tensor("reshape_3"), val = tensor([0x0p+0, -0x1.fffffep+127, 0x0p+0, 0x0p+0])]; tensor scatter_0_mode_0 = const()[name = tensor("scatter_0_mode_0"), val = tensor("update")]; tensor scatter_0_axis_0 = const()[name = tensor("scatter_0_axis_0"), val = tensor(0)]; tensor scatter_0 = scatter(axis = scatter_0_axis_0, data = reshape_3, indices = reshape_1, mode = scatter_0_mode_0, updates = reshape_2)[name = tensor("scatter_0")]; tensor reshape_4 = reshape(shape = shape_1, x = scatter_0)[name = tensor("reshape_4")]; tensor var_117_shape = shape(x = encoder_attention_mask)[name = tensor("op_117_shape")]; tensor gather_0 = const()[name = tensor("gather_0"), val = tensor(1)]; tensor gather_1_indices_0 = const()[name = tensor("gather_1_indices_0"), val = tensor(1)]; tensor gather_1_axis_0 = const()[name = tensor("gather_1_axis_0"), val = tensor(0)]; tensor gather_1_batch_dims_0 = const()[name = tensor("gather_1_batch_dims_0"), val = tensor(0)]; tensor gather_1 = gather(axis = gather_1_axis_0, batch_dims = gather_1_batch_dims_0, indices = gather_1_indices_0, x = var_117_shape)[name = tensor("gather_1")]; tensor var_120_axes_0 = const()[name = tensor("op_120_axes_0"), val = tensor([1])]; tensor var_120 = expand_dims(axes = var_120_axes_0, x = encoder_attention_mask)[name = tensor("op_120")]; tensor var_121_axes_0 = const()[name = tensor("op_121_axes_0"), val = tensor([2])]; tensor var_121 = expand_dims(axes = var_121_axes_0, x = var_120)[name = tensor("op_121")]; tensor concat_3_axis_0 = const()[name = tensor("concat_3_axis_0"), val = tensor(0)]; tensor concat_3_interleave_0 = const()[name = tensor("concat_3_interleave_0"), val = tensor(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (gather_0, var_30, const_0, gather_1))[name = tensor("concat_3")]; tensor shape_0 = shape(x = var_121)[name = tensor("shape_0")]; tensor equal_0_y_0 = const()[name = tensor("equal_0_y_0"), val = tensor(-1)]; tensor equal_0 = equal(x = concat_3, y = equal_0_y_0)[name = tensor("equal_0")]; tensor select_0 = select(a = shape_0, b = concat_3, cond = equal_0)[name = tensor("select_0")]; tensor real_div_0 = real_div(x = select_0, y = shape_0)[name = tensor("real_div_0")]; tensor var_124 = tile(reps = real_div_0, x = var_121)[name = tensor("op_124")]; tensor expanded_mask_dtype_0 = const()[name = tensor("expanded_mask_dtype_0"), val = tensor("fp32")]; tensor const_11 = const()[name = tensor("const_11"), val = tensor(0x1p+0)]; tensor expanded_mask = cast(dtype = expanded_mask_dtype_0, x = var_124)[name = tensor("cast_104")]; tensor inverted_mask = sub(x = const_11, y = expanded_mask)[name = tensor("inverted_mask")]; tensor var_129_dtype_0 = const()[name = tensor("op_129_dtype_0"), val = tensor("bool")]; tensor var_129 = cast(dtype = var_129_dtype_0, x = inverted_mask)[name = tensor("cast_103")]; tensor attention_mask_5 = select(a = var_24, b = inverted_mask, cond = var_129)[name = tensor("attention_mask_5")]; tensor var_134 = not_equal(x = input_ids, y = var_30)[name = tensor("op_134")]; tensor mask_dtype_0 = const()[name = tensor("mask_dtype_0"), val = tensor("int32")]; tensor var_136_exclusive_0 = const()[name = tensor("op_136_exclusive_0"), val = tensor(false)]; tensor var_136_reverse_0 = const()[name = tensor("op_136_reverse_0"), val = tensor(false)]; tensor mask = cast(dtype = mask_dtype_0, x = var_134)[name = tensor("cast_102")]; tensor var_136 = cumsum(axis = var_30, exclusive = var_136_exclusive_0, reverse = var_136_reverse_0, x = mask)[name = tensor("op_136")]; tensor incremental_indices = mul(x = var_136, y = mask)[name = tensor("incremental_indices")]; tensor var_142 = const()[name = tensor("op_142"), val = tensor(1)]; tensor var_143 = add(x = incremental_indices, y = var_142)[name = tensor("op_143")]; tensor var_145 = const()[name = tensor("op_145"), val = tensor([-1])]; tensor var_146 = reshape(shape = var_145, x = var_143)[name = tensor("op_146")]; tensor var_147_batch_dims_0 = const()[name = tensor("op_147_batch_dims_0"), val = tensor(0)]; tensor var_147 = gather(axis = var_27, batch_dims = var_147_batch_dims_0, indices = var_146, x = decoder_embed_positions_weights)[name = tensor("op_147")]; tensor var_149 = const()[name = tensor("op_149"), val = tensor([1, 2, 1024])]; tensor var_150 = reshape(shape = var_149, x = var_147)[name = tensor("op_150")]; tensor input_3 = add(x = inputs_embeds, y = var_150)[name = tensor("input_3")]; tensor hidden_states_1_axes_0 = const()[name = tensor("hidden_states_1_axes_0"), val = tensor([-1])]; tensor hidden_states_1 = layer_norm(axes = hidden_states_1_axes_0, beta = decoder_layers_0_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_0_self_attn_layer_norm_weight, x = input_3)[name = tensor("hidden_states_1")]; tensor var_174 = linear(bias = decoder_layers_0_self_attn_q_proj_bias, weight = decoder_layers_0_self_attn_q_proj_weight, x = hidden_states_1)[name = tensor("linear_0")]; tensor var_175 = const()[name = tensor("op_175"), val = tensor([1, 2, -1, 64])]; tensor var_176 = reshape(shape = var_175, x = var_174)[name = tensor("op_176")]; tensor key_states_1 = linear(bias = decoder_layers_0_self_attn_k_proj_bias, weight = decoder_layers_0_self_attn_k_proj_weight, x = hidden_states_1)[name = tensor("linear_1")]; tensor value_states_1 = linear(bias = decoder_layers_0_self_attn_v_proj_bias, weight = decoder_layers_0_self_attn_v_proj_weight, x = hidden_states_1)[name = tensor("linear_2")]; tensor var_184 = const()[name = tensor("op_184"), val = tensor([1, 2, -1, 64])]; tensor var_185 = reshape(shape = var_184, x = key_states_1)[name = tensor("op_185")]; tensor var_187 = const()[name = tensor("op_187"), val = tensor([1, 2, -1, 64])]; tensor var_188 = reshape(shape = var_187, x = value_states_1)[name = tensor("op_188")]; tensor value_states_3_perm_0 = const()[name = tensor("value_states_3_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_1_interleave_0 = const()[name = tensor("key_1_interleave_0"), val = tensor(false)]; tensor const_123 = const()[name = tensor("const_123"), val = tensor(1)]; tensor key_1 = concat(axis = const_123, interleave = key_1_interleave_0, values = var_185)[name = tensor("key_1")]; tensor value_1_interleave_0 = const()[name = tensor("value_1_interleave_0"), val = tensor(false)]; tensor value_states_3 = transpose(perm = value_states_3_perm_0, x = var_188)[name = tensor("transpose_215")]; tensor value_1 = concat(axis = var_13, interleave = value_1_interleave_0, values = value_states_3)[name = tensor("value_1")]; tensor mul_0 = mul(x = var_176, y = var_11)[name = tensor("mul_0")]; tensor matmul_0_transpose_y_0 = const()[name = tensor("matmul_0_transpose_y_0"), val = tensor(true)]; tensor matmul_0_transpose_x_0 = const()[name = tensor("matmul_0_transpose_x_0"), val = tensor(false)]; tensor transpose_72_perm_0 = const()[name = tensor("transpose_72_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_73_perm_0 = const()[name = tensor("transpose_73_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_73 = transpose(perm = transpose_73_perm_0, x = key_1)[name = tensor("transpose_213")]; tensor transpose_72 = transpose(perm = transpose_72_perm_0, x = mul_0)[name = tensor("transpose_214")]; tensor matmul_0 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = transpose_72, y = transpose_73)[name = tensor("matmul_0")]; tensor add_0 = add(x = matmul_0, y = reshape_4)[name = tensor("add_0")]; tensor softmax_0_axis_0 = const()[name = tensor("softmax_0_axis_0"), val = tensor(-1)]; tensor softmax_0 = softmax(axis = softmax_0_axis_0, x = add_0)[name = tensor("softmax_0")]; tensor attn_output_1_transpose_x_0 = const()[name = tensor("attn_output_1_transpose_x_0"), val = tensor(false)]; tensor attn_output_1_transpose_y_0 = const()[name = tensor("attn_output_1_transpose_y_0"), val = tensor(false)]; tensor attn_output_1 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = softmax_0, y = value_1)[name = tensor("attn_output_1")]; tensor var_204_perm_0 = const()[name = tensor("op_204_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_206 = const()[name = tensor("op_206"), val = tensor([1, 2, -1])]; tensor var_204 = transpose(perm = var_204_perm_0, x = attn_output_1)[name = tensor("transpose_212")]; tensor var_207 = reshape(shape = var_206, x = var_204)[name = tensor("op_207")]; tensor input_9 = linear(bias = decoder_layers_0_self_attn_out_proj_bias, weight = decoder_layers_0_self_attn_out_proj_weight, x = var_207)[name = tensor("linear_3")]; tensor input_11 = add(x = input_3, y = input_9)[name = tensor("input_11")]; tensor hidden_states_5_axes_0 = const()[name = tensor("hidden_states_5_axes_0"), val = tensor([-1])]; tensor hidden_states_5 = layer_norm(axes = hidden_states_5_axes_0, beta = decoder_layers_0_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_0_encoder_attn_layer_norm_weight, x = input_11)[name = tensor("hidden_states_5")]; tensor var_231 = linear(bias = decoder_layers_0_encoder_attn_q_proj_bias, weight = decoder_layers_0_encoder_attn_q_proj_weight, x = hidden_states_5)[name = tensor("linear_4")]; tensor var_232 = const()[name = tensor("op_232"), val = tensor([1, 2, -1, 64])]; tensor var_233 = reshape(shape = var_232, x = var_231)[name = tensor("op_233")]; tensor query_3_perm_0 = const()[name = tensor("query_3_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_5 = linear(bias = decoder_layers_0_encoder_attn_k_proj_bias, weight = decoder_layers_0_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_5")]; tensor value_states_5 = linear(bias = decoder_layers_0_encoder_attn_v_proj_bias, weight = decoder_layers_0_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_6")]; tensor concat_4x = const()[name = tensor("concat_4x"), val = tensor([1, -1, 16, 64])]; tensor var_242 = reshape(shape = concat_4x, x = key_states_5)[name = tensor("op_242")]; tensor key_states_7_perm_0 = const()[name = tensor("key_states_7_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_5x = const()[name = tensor("concat_5x"), val = tensor([1, -1, 16, 64])]; tensor var_245 = reshape(shape = concat_5x, x = value_states_5)[name = tensor("op_245")]; tensor value_states_7_perm_0 = const()[name = tensor("value_states_7_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_3_interleave_0 = const()[name = tensor("key_3_interleave_0"), val = tensor(false)]; tensor key_states_7 = transpose(perm = key_states_7_perm_0, x = var_242)[name = tensor("transpose_210")]; tensor key_3 = concat(axis = var_13, interleave = key_3_interleave_0, values = key_states_7)[name = tensor("key_3")]; tensor value_3_interleave_0 = const()[name = tensor("value_3_interleave_0"), val = tensor(false)]; tensor value_states_7 = transpose(perm = value_states_7_perm_0, x = var_245)[name = tensor("transpose_209")]; tensor value_3 = concat(axis = var_13, interleave = value_3_interleave_0, values = value_states_7)[name = tensor("value_3")]; tensor var_255_shape = shape(x = key_3)[name = tensor("op_255_shape")]; tensor gather_3_indices_0 = const()[name = tensor("gather_3_indices_0"), val = tensor(2)]; tensor gather_3_axis_0 = const()[name = tensor("gather_3_axis_0"), val = tensor(0)]; tensor gather_3_batch_dims_0 = const()[name = tensor("gather_3_batch_dims_0"), val = tensor(0)]; tensor gather_3 = gather(axis = gather_3_axis_0, batch_dims = gather_3_batch_dims_0, indices = gather_3_indices_0, x = var_255_shape)[name = tensor("gather_3")]; tensor concat_6_values0_0 = const()[name = tensor("concat_6_values0_0"), val = tensor(0)]; tensor concat_6_values1_0 = const()[name = tensor("concat_6_values1_0"), val = tensor(0)]; tensor concat_6_values2_0 = const()[name = tensor("concat_6_values2_0"), val = tensor(0)]; tensor concat_6_axis_0 = const()[name = tensor("concat_6_axis_0"), val = tensor(0)]; tensor concat_6_interleave_0 = const()[name = tensor("concat_6_interleave_0"), val = tensor(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (concat_6_values0_0, concat_6_values1_0, concat_6_values2_0, gather_3))[name = tensor("concat_6")]; tensor attention_mask_7_begin_0 = const()[name = tensor("attention_mask_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_7_end_mask_0 = const()[name = tensor("attention_mask_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_7 = slice_by_index(begin = attention_mask_7_begin_0, end = concat_6, end_mask = attention_mask_7_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_7")]; tensor query_3 = transpose(perm = query_3_perm_0, x = var_233)[name = tensor("transpose_211")]; tensor mul_1 = mul(x = query_3, y = var_11)[name = tensor("mul_1")]; tensor matmul_1_transpose_y_0 = const()[name = tensor("matmul_1_transpose_y_0"), val = tensor(true)]; tensor matmul_1_transpose_x_0 = const()[name = tensor("matmul_1_transpose_x_0"), val = tensor(false)]; tensor matmul_1 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = mul_1, y = key_3)[name = tensor("matmul_1")]; tensor add_1 = add(x = matmul_1, y = attention_mask_7)[name = tensor("add_1")]; tensor softmax_1_axis_0 = const()[name = tensor("softmax_1_axis_0"), val = tensor(-1)]; tensor softmax_1 = softmax(axis = softmax_1_axis_0, x = add_1)[name = tensor("softmax_1")]; tensor attn_output_5_transpose_x_0 = const()[name = tensor("attn_output_5_transpose_x_0"), val = tensor(false)]; tensor attn_output_5_transpose_y_0 = const()[name = tensor("attn_output_5_transpose_y_0"), val = tensor(false)]; tensor attn_output_5 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = softmax_1, y = value_3)[name = tensor("attn_output_5")]; tensor var_261_perm_0 = const()[name = tensor("op_261_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_263 = const()[name = tensor("op_263"), val = tensor([1, 2, -1])]; tensor var_261 = transpose(perm = var_261_perm_0, x = attn_output_5)[name = tensor("transpose_208")]; tensor var_264 = reshape(shape = var_263, x = var_261)[name = tensor("op_264")]; tensor input_15 = linear(bias = decoder_layers_0_encoder_attn_out_proj_bias, weight = decoder_layers_0_encoder_attn_out_proj_weight, x = var_264)[name = tensor("linear_7")]; tensor input_17 = add(x = input_11, y = input_15)[name = tensor("input_17")]; tensor input_19_axes_0 = const()[name = tensor("input_19_axes_0"), val = tensor([-1])]; tensor input_19 = layer_norm(axes = input_19_axes_0, beta = decoder_layers_0_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_0_final_layer_norm_weight, x = input_17)[name = tensor("input_19")]; tensor input_21 = linear(bias = decoder_layers_0_fc1_bias, weight = decoder_layers_0_fc1_weight, x = input_19)[name = tensor("linear_8")]; tensor input_23 = relu(x = input_21)[name = tensor("input_23")]; tensor input_27 = linear(bias = decoder_layers_0_fc2_bias, weight = decoder_layers_0_fc2_weight, x = input_23)[name = tensor("linear_9")]; tensor input_29 = add(x = input_17, y = input_27)[name = tensor("input_29")]; tensor hidden_states_11_axes_0 = const()[name = tensor("hidden_states_11_axes_0"), val = tensor([-1])]; tensor hidden_states_11 = layer_norm(axes = hidden_states_11_axes_0, beta = decoder_layers_1_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_1_self_attn_layer_norm_weight, x = input_29)[name = tensor("hidden_states_11")]; tensor var_314 = linear(bias = decoder_layers_1_self_attn_q_proj_bias, weight = decoder_layers_1_self_attn_q_proj_weight, x = hidden_states_11)[name = tensor("linear_10")]; tensor var_315 = const()[name = tensor("op_315"), val = tensor([1, 2, -1, 64])]; tensor var_316 = reshape(shape = var_315, x = var_314)[name = tensor("op_316")]; tensor key_states_9 = linear(bias = decoder_layers_1_self_attn_k_proj_bias, weight = decoder_layers_1_self_attn_k_proj_weight, x = hidden_states_11)[name = tensor("linear_11")]; tensor value_states_9 = linear(bias = decoder_layers_1_self_attn_v_proj_bias, weight = decoder_layers_1_self_attn_v_proj_weight, x = hidden_states_11)[name = tensor("linear_12")]; tensor var_324 = const()[name = tensor("op_324"), val = tensor([1, 2, -1, 64])]; tensor var_325 = reshape(shape = var_324, x = key_states_9)[name = tensor("op_325")]; tensor var_327 = const()[name = tensor("op_327"), val = tensor([1, 2, -1, 64])]; tensor var_328 = reshape(shape = var_327, x = value_states_9)[name = tensor("op_328")]; tensor value_states_11_perm_0 = const()[name = tensor("value_states_11_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_5_interleave_0 = const()[name = tensor("key_5_interleave_0"), val = tensor(false)]; tensor const_124 = const()[name = tensor("const_124"), val = tensor(1)]; tensor key_5 = concat(axis = const_124, interleave = key_5_interleave_0, values = var_325)[name = tensor("key_5")]; tensor value_5_interleave_0 = const()[name = tensor("value_5_interleave_0"), val = tensor(false)]; tensor value_states_11 = transpose(perm = value_states_11_perm_0, x = var_328)[name = tensor("transpose_207")]; tensor value_5 = concat(axis = var_13, interleave = value_5_interleave_0, values = value_states_11)[name = tensor("value_5")]; tensor mul_2 = mul(x = var_316, y = var_11)[name = tensor("mul_2")]; tensor matmul_2_transpose_y_0 = const()[name = tensor("matmul_2_transpose_y_0"), val = tensor(true)]; tensor matmul_2_transpose_x_0 = const()[name = tensor("matmul_2_transpose_x_0"), val = tensor(false)]; tensor transpose_74_perm_0 = const()[name = tensor("transpose_74_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_75_perm_0 = const()[name = tensor("transpose_75_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_75 = transpose(perm = transpose_75_perm_0, x = key_5)[name = tensor("transpose_205")]; tensor transpose_74 = transpose(perm = transpose_74_perm_0, x = mul_2)[name = tensor("transpose_206")]; tensor matmul_2 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = transpose_74, y = transpose_75)[name = tensor("matmul_2")]; tensor add_2 = add(x = matmul_2, y = reshape_4)[name = tensor("add_2")]; tensor softmax_2_axis_0 = const()[name = tensor("softmax_2_axis_0"), val = tensor(-1)]; tensor softmax_2 = softmax(axis = softmax_2_axis_0, x = add_2)[name = tensor("softmax_2")]; tensor attn_output_9_transpose_x_0 = const()[name = tensor("attn_output_9_transpose_x_0"), val = tensor(false)]; tensor attn_output_9_transpose_y_0 = const()[name = tensor("attn_output_9_transpose_y_0"), val = tensor(false)]; tensor attn_output_9 = matmul(transpose_x = attn_output_9_transpose_x_0, transpose_y = attn_output_9_transpose_y_0, x = softmax_2, y = value_5)[name = tensor("attn_output_9")]; tensor var_344_perm_0 = const()[name = tensor("op_344_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_346 = const()[name = tensor("op_346"), val = tensor([1, 2, -1])]; tensor var_344 = transpose(perm = var_344_perm_0, x = attn_output_9)[name = tensor("transpose_204")]; tensor var_347 = reshape(shape = var_346, x = var_344)[name = tensor("op_347")]; tensor input_33 = linear(bias = decoder_layers_1_self_attn_out_proj_bias, weight = decoder_layers_1_self_attn_out_proj_weight, x = var_347)[name = tensor("linear_13")]; tensor input_35 = add(x = input_29, y = input_33)[name = tensor("input_35")]; tensor hidden_states_15_axes_0 = const()[name = tensor("hidden_states_15_axes_0"), val = tensor([-1])]; tensor hidden_states_15 = layer_norm(axes = hidden_states_15_axes_0, beta = decoder_layers_1_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_1_encoder_attn_layer_norm_weight, x = input_35)[name = tensor("hidden_states_15")]; tensor var_371 = linear(bias = decoder_layers_1_encoder_attn_q_proj_bias, weight = decoder_layers_1_encoder_attn_q_proj_weight, x = hidden_states_15)[name = tensor("linear_14")]; tensor var_372 = const()[name = tensor("op_372"), val = tensor([1, 2, -1, 64])]; tensor var_373 = reshape(shape = var_372, x = var_371)[name = tensor("op_373")]; tensor query_7_perm_0 = const()[name = tensor("query_7_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_13 = linear(bias = decoder_layers_1_encoder_attn_k_proj_bias, weight = decoder_layers_1_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_15")]; tensor value_states_13 = linear(bias = decoder_layers_1_encoder_attn_v_proj_bias, weight = decoder_layers_1_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_16")]; tensor concat_7x = const()[name = tensor("concat_7x"), val = tensor([1, -1, 16, 64])]; tensor var_382 = reshape(shape = concat_7x, x = key_states_13)[name = tensor("op_382")]; tensor key_states_15_perm_0 = const()[name = tensor("key_states_15_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_8x = const()[name = tensor("concat_8x"), val = tensor([1, -1, 16, 64])]; tensor var_385 = reshape(shape = concat_8x, x = value_states_13)[name = tensor("op_385")]; tensor value_states_15_perm_0 = const()[name = tensor("value_states_15_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_7_interleave_0 = const()[name = tensor("key_7_interleave_0"), val = tensor(false)]; tensor key_states_15 = transpose(perm = key_states_15_perm_0, x = var_382)[name = tensor("transpose_202")]; tensor key_7 = concat(axis = var_13, interleave = key_7_interleave_0, values = key_states_15)[name = tensor("key_7")]; tensor value_7_interleave_0 = const()[name = tensor("value_7_interleave_0"), val = tensor(false)]; tensor value_states_15 = transpose(perm = value_states_15_perm_0, x = var_385)[name = tensor("transpose_201")]; tensor value_7 = concat(axis = var_13, interleave = value_7_interleave_0, values = value_states_15)[name = tensor("value_7")]; tensor var_395_shape = shape(x = key_7)[name = tensor("op_395_shape")]; tensor gather_5_indices_0 = const()[name = tensor("gather_5_indices_0"), val = tensor(2)]; tensor gather_5_axis_0 = const()[name = tensor("gather_5_axis_0"), val = tensor(0)]; tensor gather_5_batch_dims_0 = const()[name = tensor("gather_5_batch_dims_0"), val = tensor(0)]; tensor gather_5 = gather(axis = gather_5_axis_0, batch_dims = gather_5_batch_dims_0, indices = gather_5_indices_0, x = var_395_shape)[name = tensor("gather_5")]; tensor concat_9_values0_0 = const()[name = tensor("concat_9_values0_0"), val = tensor(0)]; tensor concat_9_values1_0 = const()[name = tensor("concat_9_values1_0"), val = tensor(0)]; tensor concat_9_values2_0 = const()[name = tensor("concat_9_values2_0"), val = tensor(0)]; tensor concat_9_axis_0 = const()[name = tensor("concat_9_axis_0"), val = tensor(0)]; tensor concat_9_interleave_0 = const()[name = tensor("concat_9_interleave_0"), val = tensor(false)]; tensor concat_9 = concat(axis = concat_9_axis_0, interleave = concat_9_interleave_0, values = (concat_9_values0_0, concat_9_values1_0, concat_9_values2_0, gather_5))[name = tensor("concat_9")]; tensor attention_mask_11_begin_0 = const()[name = tensor("attention_mask_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_11_end_mask_0 = const()[name = tensor("attention_mask_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_11 = slice_by_index(begin = attention_mask_11_begin_0, end = concat_9, end_mask = attention_mask_11_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_11")]; tensor query_7 = transpose(perm = query_7_perm_0, x = var_373)[name = tensor("transpose_203")]; tensor mul_3 = mul(x = query_7, y = var_11)[name = tensor("mul_3")]; tensor matmul_3_transpose_y_0 = const()[name = tensor("matmul_3_transpose_y_0"), val = tensor(true)]; tensor matmul_3_transpose_x_0 = const()[name = tensor("matmul_3_transpose_x_0"), val = tensor(false)]; tensor matmul_3 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = mul_3, y = key_7)[name = tensor("matmul_3")]; tensor add_3 = add(x = matmul_3, y = attention_mask_11)[name = tensor("add_3")]; tensor softmax_3_axis_0 = const()[name = tensor("softmax_3_axis_0"), val = tensor(-1)]; tensor softmax_3 = softmax(axis = softmax_3_axis_0, x = add_3)[name = tensor("softmax_3")]; tensor attn_output_13_transpose_x_0 = const()[name = tensor("attn_output_13_transpose_x_0"), val = tensor(false)]; tensor attn_output_13_transpose_y_0 = const()[name = tensor("attn_output_13_transpose_y_0"), val = tensor(false)]; tensor attn_output_13 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = softmax_3, y = value_7)[name = tensor("attn_output_13")]; tensor var_401_perm_0 = const()[name = tensor("op_401_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_403 = const()[name = tensor("op_403"), val = tensor([1, 2, -1])]; tensor var_401 = transpose(perm = var_401_perm_0, x = attn_output_13)[name = tensor("transpose_200")]; tensor var_404 = reshape(shape = var_403, x = var_401)[name = tensor("op_404")]; tensor input_39 = linear(bias = decoder_layers_1_encoder_attn_out_proj_bias, weight = decoder_layers_1_encoder_attn_out_proj_weight, x = var_404)[name = tensor("linear_17")]; tensor input_41 = add(x = input_35, y = input_39)[name = tensor("input_41")]; tensor input_43_axes_0 = const()[name = tensor("input_43_axes_0"), val = tensor([-1])]; tensor input_43 = layer_norm(axes = input_43_axes_0, beta = decoder_layers_1_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_1_final_layer_norm_weight, x = input_41)[name = tensor("input_43")]; tensor input_45 = linear(bias = decoder_layers_1_fc1_bias, weight = decoder_layers_1_fc1_weight, x = input_43)[name = tensor("linear_18")]; tensor input_47 = relu(x = input_45)[name = tensor("input_47")]; tensor input_51 = linear(bias = decoder_layers_1_fc2_bias, weight = decoder_layers_1_fc2_weight, x = input_47)[name = tensor("linear_19")]; tensor input_53 = add(x = input_41, y = input_51)[name = tensor("input_53")]; tensor hidden_states_21_axes_0 = const()[name = tensor("hidden_states_21_axes_0"), val = tensor([-1])]; tensor hidden_states_21 = layer_norm(axes = hidden_states_21_axes_0, beta = decoder_layers_2_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_2_self_attn_layer_norm_weight, x = input_53)[name = tensor("hidden_states_21")]; tensor var_454 = linear(bias = decoder_layers_2_self_attn_q_proj_bias, weight = decoder_layers_2_self_attn_q_proj_weight, x = hidden_states_21)[name = tensor("linear_20")]; tensor var_455 = const()[name = tensor("op_455"), val = tensor([1, 2, -1, 64])]; tensor var_456 = reshape(shape = var_455, x = var_454)[name = tensor("op_456")]; tensor key_states_17 = linear(bias = decoder_layers_2_self_attn_k_proj_bias, weight = decoder_layers_2_self_attn_k_proj_weight, x = hidden_states_21)[name = tensor("linear_21")]; tensor value_states_17 = linear(bias = decoder_layers_2_self_attn_v_proj_bias, weight = decoder_layers_2_self_attn_v_proj_weight, x = hidden_states_21)[name = tensor("linear_22")]; tensor var_464 = const()[name = tensor("op_464"), val = tensor([1, 2, -1, 64])]; tensor var_465 = reshape(shape = var_464, x = key_states_17)[name = tensor("op_465")]; tensor var_467 = const()[name = tensor("op_467"), val = tensor([1, 2, -1, 64])]; tensor var_468 = reshape(shape = var_467, x = value_states_17)[name = tensor("op_468")]; tensor value_states_19_perm_0 = const()[name = tensor("value_states_19_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_9_interleave_0 = const()[name = tensor("key_9_interleave_0"), val = tensor(false)]; tensor const_125 = const()[name = tensor("const_125"), val = tensor(1)]; tensor key_9 = concat(axis = const_125, interleave = key_9_interleave_0, values = var_465)[name = tensor("key_9")]; tensor value_9_interleave_0 = const()[name = tensor("value_9_interleave_0"), val = tensor(false)]; tensor value_states_19 = transpose(perm = value_states_19_perm_0, x = var_468)[name = tensor("transpose_199")]; tensor value_9 = concat(axis = var_13, interleave = value_9_interleave_0, values = value_states_19)[name = tensor("value_9")]; tensor mul_4 = mul(x = var_456, y = var_11)[name = tensor("mul_4")]; tensor matmul_4_transpose_y_0 = const()[name = tensor("matmul_4_transpose_y_0"), val = tensor(true)]; tensor matmul_4_transpose_x_0 = const()[name = tensor("matmul_4_transpose_x_0"), val = tensor(false)]; tensor transpose_76_perm_0 = const()[name = tensor("transpose_76_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_77_perm_0 = const()[name = tensor("transpose_77_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_77 = transpose(perm = transpose_77_perm_0, x = key_9)[name = tensor("transpose_197")]; tensor transpose_76 = transpose(perm = transpose_76_perm_0, x = mul_4)[name = tensor("transpose_198")]; tensor matmul_4 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = transpose_76, y = transpose_77)[name = tensor("matmul_4")]; tensor add_4 = add(x = matmul_4, y = reshape_4)[name = tensor("add_4")]; tensor softmax_4_axis_0 = const()[name = tensor("softmax_4_axis_0"), val = tensor(-1)]; tensor softmax_4 = softmax(axis = softmax_4_axis_0, x = add_4)[name = tensor("softmax_4")]; tensor attn_output_17_transpose_x_0 = const()[name = tensor("attn_output_17_transpose_x_0"), val = tensor(false)]; tensor attn_output_17_transpose_y_0 = const()[name = tensor("attn_output_17_transpose_y_0"), val = tensor(false)]; tensor attn_output_17 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = softmax_4, y = value_9)[name = tensor("attn_output_17")]; tensor var_484_perm_0 = const()[name = tensor("op_484_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_486 = const()[name = tensor("op_486"), val = tensor([1, 2, -1])]; tensor var_484 = transpose(perm = var_484_perm_0, x = attn_output_17)[name = tensor("transpose_196")]; tensor var_487 = reshape(shape = var_486, x = var_484)[name = tensor("op_487")]; tensor input_57 = linear(bias = decoder_layers_2_self_attn_out_proj_bias, weight = decoder_layers_2_self_attn_out_proj_weight, x = var_487)[name = tensor("linear_23")]; tensor input_59 = add(x = input_53, y = input_57)[name = tensor("input_59")]; tensor hidden_states_25_axes_0 = const()[name = tensor("hidden_states_25_axes_0"), val = tensor([-1])]; tensor hidden_states_25 = layer_norm(axes = hidden_states_25_axes_0, beta = decoder_layers_2_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_2_encoder_attn_layer_norm_weight, x = input_59)[name = tensor("hidden_states_25")]; tensor var_511 = linear(bias = decoder_layers_2_encoder_attn_q_proj_bias, weight = decoder_layers_2_encoder_attn_q_proj_weight, x = hidden_states_25)[name = tensor("linear_24")]; tensor var_512 = const()[name = tensor("op_512"), val = tensor([1, 2, -1, 64])]; tensor var_513 = reshape(shape = var_512, x = var_511)[name = tensor("op_513")]; tensor query_11_perm_0 = const()[name = tensor("query_11_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_21 = linear(bias = decoder_layers_2_encoder_attn_k_proj_bias, weight = decoder_layers_2_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_25")]; tensor value_states_21 = linear(bias = decoder_layers_2_encoder_attn_v_proj_bias, weight = decoder_layers_2_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_26")]; tensor concat_10x = const()[name = tensor("concat_10x"), val = tensor([1, -1, 16, 64])]; tensor var_522 = reshape(shape = concat_10x, x = key_states_21)[name = tensor("op_522")]; tensor key_states_23_perm_0 = const()[name = tensor("key_states_23_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_11x = const()[name = tensor("concat_11x"), val = tensor([1, -1, 16, 64])]; tensor var_525 = reshape(shape = concat_11x, x = value_states_21)[name = tensor("op_525")]; tensor value_states_23_perm_0 = const()[name = tensor("value_states_23_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_11_interleave_0 = const()[name = tensor("key_11_interleave_0"), val = tensor(false)]; tensor key_states_23 = transpose(perm = key_states_23_perm_0, x = var_522)[name = tensor("transpose_194")]; tensor key_11 = concat(axis = var_13, interleave = key_11_interleave_0, values = key_states_23)[name = tensor("key_11")]; tensor value_11_interleave_0 = const()[name = tensor("value_11_interleave_0"), val = tensor(false)]; tensor value_states_23 = transpose(perm = value_states_23_perm_0, x = var_525)[name = tensor("transpose_193")]; tensor value_11 = concat(axis = var_13, interleave = value_11_interleave_0, values = value_states_23)[name = tensor("value_11")]; tensor var_535_shape = shape(x = key_11)[name = tensor("op_535_shape")]; tensor gather_7_indices_0 = const()[name = tensor("gather_7_indices_0"), val = tensor(2)]; tensor gather_7_axis_0 = const()[name = tensor("gather_7_axis_0"), val = tensor(0)]; tensor gather_7_batch_dims_0 = const()[name = tensor("gather_7_batch_dims_0"), val = tensor(0)]; tensor gather_7 = gather(axis = gather_7_axis_0, batch_dims = gather_7_batch_dims_0, indices = gather_7_indices_0, x = var_535_shape)[name = tensor("gather_7")]; tensor concat_12_values0_0 = const()[name = tensor("concat_12_values0_0"), val = tensor(0)]; tensor concat_12_values1_0 = const()[name = tensor("concat_12_values1_0"), val = tensor(0)]; tensor concat_12_values2_0 = const()[name = tensor("concat_12_values2_0"), val = tensor(0)]; tensor concat_12_axis_0 = const()[name = tensor("concat_12_axis_0"), val = tensor(0)]; tensor concat_12_interleave_0 = const()[name = tensor("concat_12_interleave_0"), val = tensor(false)]; tensor concat_12 = concat(axis = concat_12_axis_0, interleave = concat_12_interleave_0, values = (concat_12_values0_0, concat_12_values1_0, concat_12_values2_0, gather_7))[name = tensor("concat_12")]; tensor attention_mask_15_begin_0 = const()[name = tensor("attention_mask_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_15_end_mask_0 = const()[name = tensor("attention_mask_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_15 = slice_by_index(begin = attention_mask_15_begin_0, end = concat_12, end_mask = attention_mask_15_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_15")]; tensor query_11 = transpose(perm = query_11_perm_0, x = var_513)[name = tensor("transpose_195")]; tensor mul_5 = mul(x = query_11, y = var_11)[name = tensor("mul_5")]; tensor matmul_5_transpose_y_0 = const()[name = tensor("matmul_5_transpose_y_0"), val = tensor(true)]; tensor matmul_5_transpose_x_0 = const()[name = tensor("matmul_5_transpose_x_0"), val = tensor(false)]; tensor matmul_5 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = mul_5, y = key_11)[name = tensor("matmul_5")]; tensor add_5 = add(x = matmul_5, y = attention_mask_15)[name = tensor("add_5")]; tensor softmax_5_axis_0 = const()[name = tensor("softmax_5_axis_0"), val = tensor(-1)]; tensor softmax_5 = softmax(axis = softmax_5_axis_0, x = add_5)[name = tensor("softmax_5")]; tensor attn_output_21_transpose_x_0 = const()[name = tensor("attn_output_21_transpose_x_0"), val = tensor(false)]; tensor attn_output_21_transpose_y_0 = const()[name = tensor("attn_output_21_transpose_y_0"), val = tensor(false)]; tensor attn_output_21 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = softmax_5, y = value_11)[name = tensor("attn_output_21")]; tensor var_541_perm_0 = const()[name = tensor("op_541_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_543 = const()[name = tensor("op_543"), val = tensor([1, 2, -1])]; tensor var_541 = transpose(perm = var_541_perm_0, x = attn_output_21)[name = tensor("transpose_192")]; tensor var_544 = reshape(shape = var_543, x = var_541)[name = tensor("op_544")]; tensor input_63 = linear(bias = decoder_layers_2_encoder_attn_out_proj_bias, weight = decoder_layers_2_encoder_attn_out_proj_weight, x = var_544)[name = tensor("linear_27")]; tensor input_65 = add(x = input_59, y = input_63)[name = tensor("input_65")]; tensor input_67_axes_0 = const()[name = tensor("input_67_axes_0"), val = tensor([-1])]; tensor input_67 = layer_norm(axes = input_67_axes_0, beta = decoder_layers_2_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_2_final_layer_norm_weight, x = input_65)[name = tensor("input_67")]; tensor input_69 = linear(bias = decoder_layers_2_fc1_bias, weight = decoder_layers_2_fc1_weight, x = input_67)[name = tensor("linear_28")]; tensor input_71 = relu(x = input_69)[name = tensor("input_71")]; tensor input_75 = linear(bias = decoder_layers_2_fc2_bias, weight = decoder_layers_2_fc2_weight, x = input_71)[name = tensor("linear_29")]; tensor input_77 = add(x = input_65, y = input_75)[name = tensor("input_77")]; tensor hidden_states_31_axes_0 = const()[name = tensor("hidden_states_31_axes_0"), val = tensor([-1])]; tensor hidden_states_31 = layer_norm(axes = hidden_states_31_axes_0, beta = decoder_layers_3_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_3_self_attn_layer_norm_weight, x = input_77)[name = tensor("hidden_states_31")]; tensor var_594 = linear(bias = decoder_layers_3_self_attn_q_proj_bias, weight = decoder_layers_3_self_attn_q_proj_weight, x = hidden_states_31)[name = tensor("linear_30")]; tensor var_595 = const()[name = tensor("op_595"), val = tensor([1, 2, -1, 64])]; tensor var_596 = reshape(shape = var_595, x = var_594)[name = tensor("op_596")]; tensor key_states_25 = linear(bias = decoder_layers_3_self_attn_k_proj_bias, weight = decoder_layers_3_self_attn_k_proj_weight, x = hidden_states_31)[name = tensor("linear_31")]; tensor value_states_25 = linear(bias = decoder_layers_3_self_attn_v_proj_bias, weight = decoder_layers_3_self_attn_v_proj_weight, x = hidden_states_31)[name = tensor("linear_32")]; tensor var_604 = const()[name = tensor("op_604"), val = tensor([1, 2, -1, 64])]; tensor var_605 = reshape(shape = var_604, x = key_states_25)[name = tensor("op_605")]; tensor var_607 = const()[name = tensor("op_607"), val = tensor([1, 2, -1, 64])]; tensor var_608 = reshape(shape = var_607, x = value_states_25)[name = tensor("op_608")]; tensor value_states_27_perm_0 = const()[name = tensor("value_states_27_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_13_interleave_0 = const()[name = tensor("key_13_interleave_0"), val = tensor(false)]; tensor const_126 = const()[name = tensor("const_126"), val = tensor(1)]; tensor key_13 = concat(axis = const_126, interleave = key_13_interleave_0, values = var_605)[name = tensor("key_13")]; tensor value_13_interleave_0 = const()[name = tensor("value_13_interleave_0"), val = tensor(false)]; tensor value_states_27 = transpose(perm = value_states_27_perm_0, x = var_608)[name = tensor("transpose_191")]; tensor value_13 = concat(axis = var_13, interleave = value_13_interleave_0, values = value_states_27)[name = tensor("value_13")]; tensor mul_6 = mul(x = var_596, y = var_11)[name = tensor("mul_6")]; tensor matmul_6_transpose_y_0 = const()[name = tensor("matmul_6_transpose_y_0"), val = tensor(true)]; tensor matmul_6_transpose_x_0 = const()[name = tensor("matmul_6_transpose_x_0"), val = tensor(false)]; tensor transpose_78_perm_0 = const()[name = tensor("transpose_78_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_79_perm_0 = const()[name = tensor("transpose_79_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_79 = transpose(perm = transpose_79_perm_0, x = key_13)[name = tensor("transpose_189")]; tensor transpose_78 = transpose(perm = transpose_78_perm_0, x = mul_6)[name = tensor("transpose_190")]; tensor matmul_6 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = transpose_78, y = transpose_79)[name = tensor("matmul_6")]; tensor add_6 = add(x = matmul_6, y = reshape_4)[name = tensor("add_6")]; tensor softmax_6_axis_0 = const()[name = tensor("softmax_6_axis_0"), val = tensor(-1)]; tensor softmax_6 = softmax(axis = softmax_6_axis_0, x = add_6)[name = tensor("softmax_6")]; tensor attn_output_25_transpose_x_0 = const()[name = tensor("attn_output_25_transpose_x_0"), val = tensor(false)]; tensor attn_output_25_transpose_y_0 = const()[name = tensor("attn_output_25_transpose_y_0"), val = tensor(false)]; tensor attn_output_25 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = softmax_6, y = value_13)[name = tensor("attn_output_25")]; tensor var_624_perm_0 = const()[name = tensor("op_624_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_626 = const()[name = tensor("op_626"), val = tensor([1, 2, -1])]; tensor var_624 = transpose(perm = var_624_perm_0, x = attn_output_25)[name = tensor("transpose_188")]; tensor var_627 = reshape(shape = var_626, x = var_624)[name = tensor("op_627")]; tensor input_81 = linear(bias = decoder_layers_3_self_attn_out_proj_bias, weight = decoder_layers_3_self_attn_out_proj_weight, x = var_627)[name = tensor("linear_33")]; tensor input_83 = add(x = input_77, y = input_81)[name = tensor("input_83")]; tensor hidden_states_35_axes_0 = const()[name = tensor("hidden_states_35_axes_0"), val = tensor([-1])]; tensor hidden_states_35 = layer_norm(axes = hidden_states_35_axes_0, beta = decoder_layers_3_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_3_encoder_attn_layer_norm_weight, x = input_83)[name = tensor("hidden_states_35")]; tensor var_651 = linear(bias = decoder_layers_3_encoder_attn_q_proj_bias, weight = decoder_layers_3_encoder_attn_q_proj_weight, x = hidden_states_35)[name = tensor("linear_34")]; tensor var_652 = const()[name = tensor("op_652"), val = tensor([1, 2, -1, 64])]; tensor var_653 = reshape(shape = var_652, x = var_651)[name = tensor("op_653")]; tensor query_15_perm_0 = const()[name = tensor("query_15_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_29 = linear(bias = decoder_layers_3_encoder_attn_k_proj_bias, weight = decoder_layers_3_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_35")]; tensor value_states_29 = linear(bias = decoder_layers_3_encoder_attn_v_proj_bias, weight = decoder_layers_3_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_36")]; tensor concat_13x = const()[name = tensor("concat_13x"), val = tensor([1, -1, 16, 64])]; tensor var_662 = reshape(shape = concat_13x, x = key_states_29)[name = tensor("op_662")]; tensor key_states_31_perm_0 = const()[name = tensor("key_states_31_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_14x = const()[name = tensor("concat_14x"), val = tensor([1, -1, 16, 64])]; tensor var_665 = reshape(shape = concat_14x, x = value_states_29)[name = tensor("op_665")]; tensor value_states_31_perm_0 = const()[name = tensor("value_states_31_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_15_interleave_0 = const()[name = tensor("key_15_interleave_0"), val = tensor(false)]; tensor key_states_31 = transpose(perm = key_states_31_perm_0, x = var_662)[name = tensor("transpose_186")]; tensor key_15 = concat(axis = var_13, interleave = key_15_interleave_0, values = key_states_31)[name = tensor("key_15")]; tensor value_15_interleave_0 = const()[name = tensor("value_15_interleave_0"), val = tensor(false)]; tensor value_states_31 = transpose(perm = value_states_31_perm_0, x = var_665)[name = tensor("transpose_185")]; tensor value_15 = concat(axis = var_13, interleave = value_15_interleave_0, values = value_states_31)[name = tensor("value_15")]; tensor var_675_shape = shape(x = key_15)[name = tensor("op_675_shape")]; tensor gather_9_indices_0 = const()[name = tensor("gather_9_indices_0"), val = tensor(2)]; tensor gather_9_axis_0 = const()[name = tensor("gather_9_axis_0"), val = tensor(0)]; tensor gather_9_batch_dims_0 = const()[name = tensor("gather_9_batch_dims_0"), val = tensor(0)]; tensor gather_9 = gather(axis = gather_9_axis_0, batch_dims = gather_9_batch_dims_0, indices = gather_9_indices_0, x = var_675_shape)[name = tensor("gather_9")]; tensor concat_15_values0_0 = const()[name = tensor("concat_15_values0_0"), val = tensor(0)]; tensor concat_15_values1_0 = const()[name = tensor("concat_15_values1_0"), val = tensor(0)]; tensor concat_15_values2_0 = const()[name = tensor("concat_15_values2_0"), val = tensor(0)]; tensor concat_15_axis_0 = const()[name = tensor("concat_15_axis_0"), val = tensor(0)]; tensor concat_15_interleave_0 = const()[name = tensor("concat_15_interleave_0"), val = tensor(false)]; tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (concat_15_values0_0, concat_15_values1_0, concat_15_values2_0, gather_9))[name = tensor("concat_15")]; tensor attention_mask_19_begin_0 = const()[name = tensor("attention_mask_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_19_end_mask_0 = const()[name = tensor("attention_mask_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_19 = slice_by_index(begin = attention_mask_19_begin_0, end = concat_15, end_mask = attention_mask_19_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_19")]; tensor query_15 = transpose(perm = query_15_perm_0, x = var_653)[name = tensor("transpose_187")]; tensor mul_7 = mul(x = query_15, y = var_11)[name = tensor("mul_7")]; tensor matmul_7_transpose_y_0 = const()[name = tensor("matmul_7_transpose_y_0"), val = tensor(true)]; tensor matmul_7_transpose_x_0 = const()[name = tensor("matmul_7_transpose_x_0"), val = tensor(false)]; tensor matmul_7 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = mul_7, y = key_15)[name = tensor("matmul_7")]; tensor add_7 = add(x = matmul_7, y = attention_mask_19)[name = tensor("add_7")]; tensor softmax_7_axis_0 = const()[name = tensor("softmax_7_axis_0"), val = tensor(-1)]; tensor softmax_7 = softmax(axis = softmax_7_axis_0, x = add_7)[name = tensor("softmax_7")]; tensor attn_output_29_transpose_x_0 = const()[name = tensor("attn_output_29_transpose_x_0"), val = tensor(false)]; tensor attn_output_29_transpose_y_0 = const()[name = tensor("attn_output_29_transpose_y_0"), val = tensor(false)]; tensor attn_output_29 = matmul(transpose_x = attn_output_29_transpose_x_0, transpose_y = attn_output_29_transpose_y_0, x = softmax_7, y = value_15)[name = tensor("attn_output_29")]; tensor var_681_perm_0 = const()[name = tensor("op_681_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_683 = const()[name = tensor("op_683"), val = tensor([1, 2, -1])]; tensor var_681 = transpose(perm = var_681_perm_0, x = attn_output_29)[name = tensor("transpose_184")]; tensor var_684 = reshape(shape = var_683, x = var_681)[name = tensor("op_684")]; tensor input_87 = linear(bias = decoder_layers_3_encoder_attn_out_proj_bias, weight = decoder_layers_3_encoder_attn_out_proj_weight, x = var_684)[name = tensor("linear_37")]; tensor input_89 = add(x = input_83, y = input_87)[name = tensor("input_89")]; tensor input_91_axes_0 = const()[name = tensor("input_91_axes_0"), val = tensor([-1])]; tensor input_91 = layer_norm(axes = input_91_axes_0, beta = decoder_layers_3_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_3_final_layer_norm_weight, x = input_89)[name = tensor("input_91")]; tensor input_93 = linear(bias = decoder_layers_3_fc1_bias, weight = decoder_layers_3_fc1_weight, x = input_91)[name = tensor("linear_38")]; tensor input_95 = relu(x = input_93)[name = tensor("input_95")]; tensor input_99 = linear(bias = decoder_layers_3_fc2_bias, weight = decoder_layers_3_fc2_weight, x = input_95)[name = tensor("linear_39")]; tensor input_101 = add(x = input_89, y = input_99)[name = tensor("input_101")]; tensor hidden_states_41_axes_0 = const()[name = tensor("hidden_states_41_axes_0"), val = tensor([-1])]; tensor hidden_states_41 = layer_norm(axes = hidden_states_41_axes_0, beta = decoder_layers_4_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_4_self_attn_layer_norm_weight, x = input_101)[name = tensor("hidden_states_41")]; tensor var_734 = linear(bias = decoder_layers_4_self_attn_q_proj_bias, weight = decoder_layers_4_self_attn_q_proj_weight, x = hidden_states_41)[name = tensor("linear_40")]; tensor var_735 = const()[name = tensor("op_735"), val = tensor([1, 2, -1, 64])]; tensor var_736 = reshape(shape = var_735, x = var_734)[name = tensor("op_736")]; tensor key_states_33 = linear(bias = decoder_layers_4_self_attn_k_proj_bias, weight = decoder_layers_4_self_attn_k_proj_weight, x = hidden_states_41)[name = tensor("linear_41")]; tensor value_states_33 = linear(bias = decoder_layers_4_self_attn_v_proj_bias, weight = decoder_layers_4_self_attn_v_proj_weight, x = hidden_states_41)[name = tensor("linear_42")]; tensor var_744 = const()[name = tensor("op_744"), val = tensor([1, 2, -1, 64])]; tensor var_745 = reshape(shape = var_744, x = key_states_33)[name = tensor("op_745")]; tensor var_747 = const()[name = tensor("op_747"), val = tensor([1, 2, -1, 64])]; tensor var_748 = reshape(shape = var_747, x = value_states_33)[name = tensor("op_748")]; tensor value_states_35_perm_0 = const()[name = tensor("value_states_35_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_17_interleave_0 = const()[name = tensor("key_17_interleave_0"), val = tensor(false)]; tensor const_127 = const()[name = tensor("const_127"), val = tensor(1)]; tensor key_17 = concat(axis = const_127, interleave = key_17_interleave_0, values = var_745)[name = tensor("key_17")]; tensor value_17_interleave_0 = const()[name = tensor("value_17_interleave_0"), val = tensor(false)]; tensor value_states_35 = transpose(perm = value_states_35_perm_0, x = var_748)[name = tensor("transpose_183")]; tensor value_17 = concat(axis = var_13, interleave = value_17_interleave_0, values = value_states_35)[name = tensor("value_17")]; tensor mul_8 = mul(x = var_736, y = var_11)[name = tensor("mul_8")]; tensor matmul_8_transpose_y_0 = const()[name = tensor("matmul_8_transpose_y_0"), val = tensor(true)]; tensor matmul_8_transpose_x_0 = const()[name = tensor("matmul_8_transpose_x_0"), val = tensor(false)]; tensor transpose_80_perm_0 = const()[name = tensor("transpose_80_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_81_perm_0 = const()[name = tensor("transpose_81_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_81 = transpose(perm = transpose_81_perm_0, x = key_17)[name = tensor("transpose_181")]; tensor transpose_80 = transpose(perm = transpose_80_perm_0, x = mul_8)[name = tensor("transpose_182")]; tensor matmul_8 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = transpose_80, y = transpose_81)[name = tensor("matmul_8")]; tensor add_8 = add(x = matmul_8, y = reshape_4)[name = tensor("add_8")]; tensor softmax_8_axis_0 = const()[name = tensor("softmax_8_axis_0"), val = tensor(-1)]; tensor softmax_8 = softmax(axis = softmax_8_axis_0, x = add_8)[name = tensor("softmax_8")]; tensor attn_output_33_transpose_x_0 = const()[name = tensor("attn_output_33_transpose_x_0"), val = tensor(false)]; tensor attn_output_33_transpose_y_0 = const()[name = tensor("attn_output_33_transpose_y_0"), val = tensor(false)]; tensor attn_output_33 = matmul(transpose_x = attn_output_33_transpose_x_0, transpose_y = attn_output_33_transpose_y_0, x = softmax_8, y = value_17)[name = tensor("attn_output_33")]; tensor var_764_perm_0 = const()[name = tensor("op_764_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_766 = const()[name = tensor("op_766"), val = tensor([1, 2, -1])]; tensor var_764 = transpose(perm = var_764_perm_0, x = attn_output_33)[name = tensor("transpose_180")]; tensor var_767 = reshape(shape = var_766, x = var_764)[name = tensor("op_767")]; tensor input_105 = linear(bias = decoder_layers_4_self_attn_out_proj_bias, weight = decoder_layers_4_self_attn_out_proj_weight, x = var_767)[name = tensor("linear_43")]; tensor input_107 = add(x = input_101, y = input_105)[name = tensor("input_107")]; tensor hidden_states_45_axes_0 = const()[name = tensor("hidden_states_45_axes_0"), val = tensor([-1])]; tensor hidden_states_45 = layer_norm(axes = hidden_states_45_axes_0, beta = decoder_layers_4_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_4_encoder_attn_layer_norm_weight, x = input_107)[name = tensor("hidden_states_45")]; tensor var_791 = linear(bias = decoder_layers_4_encoder_attn_q_proj_bias, weight = decoder_layers_4_encoder_attn_q_proj_weight, x = hidden_states_45)[name = tensor("linear_44")]; tensor var_792 = const()[name = tensor("op_792"), val = tensor([1, 2, -1, 64])]; tensor var_793 = reshape(shape = var_792, x = var_791)[name = tensor("op_793")]; tensor query_19_perm_0 = const()[name = tensor("query_19_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_37 = linear(bias = decoder_layers_4_encoder_attn_k_proj_bias, weight = decoder_layers_4_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_45")]; tensor value_states_37 = linear(bias = decoder_layers_4_encoder_attn_v_proj_bias, weight = decoder_layers_4_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_46")]; tensor concat_16x = const()[name = tensor("concat_16x"), val = tensor([1, -1, 16, 64])]; tensor var_802 = reshape(shape = concat_16x, x = key_states_37)[name = tensor("op_802")]; tensor key_states_39_perm_0 = const()[name = tensor("key_states_39_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_17x = const()[name = tensor("concat_17x"), val = tensor([1, -1, 16, 64])]; tensor var_805 = reshape(shape = concat_17x, x = value_states_37)[name = tensor("op_805")]; tensor value_states_39_perm_0 = const()[name = tensor("value_states_39_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_19_interleave_0 = const()[name = tensor("key_19_interleave_0"), val = tensor(false)]; tensor key_states_39 = transpose(perm = key_states_39_perm_0, x = var_802)[name = tensor("transpose_178")]; tensor key_19 = concat(axis = var_13, interleave = key_19_interleave_0, values = key_states_39)[name = tensor("key_19")]; tensor value_19_interleave_0 = const()[name = tensor("value_19_interleave_0"), val = tensor(false)]; tensor value_states_39 = transpose(perm = value_states_39_perm_0, x = var_805)[name = tensor("transpose_177")]; tensor value_19 = concat(axis = var_13, interleave = value_19_interleave_0, values = value_states_39)[name = tensor("value_19")]; tensor var_815_shape = shape(x = key_19)[name = tensor("op_815_shape")]; tensor gather_11_indices_0 = const()[name = tensor("gather_11_indices_0"), val = tensor(2)]; tensor gather_11_axis_0 = const()[name = tensor("gather_11_axis_0"), val = tensor(0)]; tensor gather_11_batch_dims_0 = const()[name = tensor("gather_11_batch_dims_0"), val = tensor(0)]; tensor gather_11 = gather(axis = gather_11_axis_0, batch_dims = gather_11_batch_dims_0, indices = gather_11_indices_0, x = var_815_shape)[name = tensor("gather_11")]; tensor concat_18_values0_0 = const()[name = tensor("concat_18_values0_0"), val = tensor(0)]; tensor concat_18_values1_0 = const()[name = tensor("concat_18_values1_0"), val = tensor(0)]; tensor concat_18_values2_0 = const()[name = tensor("concat_18_values2_0"), val = tensor(0)]; tensor concat_18_axis_0 = const()[name = tensor("concat_18_axis_0"), val = tensor(0)]; tensor concat_18_interleave_0 = const()[name = tensor("concat_18_interleave_0"), val = tensor(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (concat_18_values0_0, concat_18_values1_0, concat_18_values2_0, gather_11))[name = tensor("concat_18")]; tensor attention_mask_23_begin_0 = const()[name = tensor("attention_mask_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_23_end_mask_0 = const()[name = tensor("attention_mask_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_23 = slice_by_index(begin = attention_mask_23_begin_0, end = concat_18, end_mask = attention_mask_23_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_23")]; tensor query_19 = transpose(perm = query_19_perm_0, x = var_793)[name = tensor("transpose_179")]; tensor mul_9 = mul(x = query_19, y = var_11)[name = tensor("mul_9")]; tensor matmul_9_transpose_y_0 = const()[name = tensor("matmul_9_transpose_y_0"), val = tensor(true)]; tensor matmul_9_transpose_x_0 = const()[name = tensor("matmul_9_transpose_x_0"), val = tensor(false)]; tensor matmul_9 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = mul_9, y = key_19)[name = tensor("matmul_9")]; tensor add_9 = add(x = matmul_9, y = attention_mask_23)[name = tensor("add_9")]; tensor softmax_9_axis_0 = const()[name = tensor("softmax_9_axis_0"), val = tensor(-1)]; tensor softmax_9 = softmax(axis = softmax_9_axis_0, x = add_9)[name = tensor("softmax_9")]; tensor attn_output_37_transpose_x_0 = const()[name = tensor("attn_output_37_transpose_x_0"), val = tensor(false)]; tensor attn_output_37_transpose_y_0 = const()[name = tensor("attn_output_37_transpose_y_0"), val = tensor(false)]; tensor attn_output_37 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = softmax_9, y = value_19)[name = tensor("attn_output_37")]; tensor var_821_perm_0 = const()[name = tensor("op_821_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_823 = const()[name = tensor("op_823"), val = tensor([1, 2, -1])]; tensor var_821 = transpose(perm = var_821_perm_0, x = attn_output_37)[name = tensor("transpose_176")]; tensor var_824 = reshape(shape = var_823, x = var_821)[name = tensor("op_824")]; tensor input_111 = linear(bias = decoder_layers_4_encoder_attn_out_proj_bias, weight = decoder_layers_4_encoder_attn_out_proj_weight, x = var_824)[name = tensor("linear_47")]; tensor input_113 = add(x = input_107, y = input_111)[name = tensor("input_113")]; tensor input_115_axes_0 = const()[name = tensor("input_115_axes_0"), val = tensor([-1])]; tensor input_115 = layer_norm(axes = input_115_axes_0, beta = decoder_layers_4_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_4_final_layer_norm_weight, x = input_113)[name = tensor("input_115")]; tensor input_117 = linear(bias = decoder_layers_4_fc1_bias, weight = decoder_layers_4_fc1_weight, x = input_115)[name = tensor("linear_48")]; tensor input_119 = relu(x = input_117)[name = tensor("input_119")]; tensor input_123 = linear(bias = decoder_layers_4_fc2_bias, weight = decoder_layers_4_fc2_weight, x = input_119)[name = tensor("linear_49")]; tensor input_125 = add(x = input_113, y = input_123)[name = tensor("input_125")]; tensor hidden_states_51_axes_0 = const()[name = tensor("hidden_states_51_axes_0"), val = tensor([-1])]; tensor hidden_states_51 = layer_norm(axes = hidden_states_51_axes_0, beta = decoder_layers_5_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_5_self_attn_layer_norm_weight, x = input_125)[name = tensor("hidden_states_51")]; tensor var_874 = linear(bias = decoder_layers_5_self_attn_q_proj_bias, weight = decoder_layers_5_self_attn_q_proj_weight, x = hidden_states_51)[name = tensor("linear_50")]; tensor var_875 = const()[name = tensor("op_875"), val = tensor([1, 2, -1, 64])]; tensor var_876 = reshape(shape = var_875, x = var_874)[name = tensor("op_876")]; tensor key_states_41 = linear(bias = decoder_layers_5_self_attn_k_proj_bias, weight = decoder_layers_5_self_attn_k_proj_weight, x = hidden_states_51)[name = tensor("linear_51")]; tensor value_states_41 = linear(bias = decoder_layers_5_self_attn_v_proj_bias, weight = decoder_layers_5_self_attn_v_proj_weight, x = hidden_states_51)[name = tensor("linear_52")]; tensor var_884 = const()[name = tensor("op_884"), val = tensor([1, 2, -1, 64])]; tensor var_885 = reshape(shape = var_884, x = key_states_41)[name = tensor("op_885")]; tensor var_887 = const()[name = tensor("op_887"), val = tensor([1, 2, -1, 64])]; tensor var_888 = reshape(shape = var_887, x = value_states_41)[name = tensor("op_888")]; tensor value_states_43_perm_0 = const()[name = tensor("value_states_43_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_21_interleave_0 = const()[name = tensor("key_21_interleave_0"), val = tensor(false)]; tensor const_128 = const()[name = tensor("const_128"), val = tensor(1)]; tensor key_21 = concat(axis = const_128, interleave = key_21_interleave_0, values = var_885)[name = tensor("key_21")]; tensor value_21_interleave_0 = const()[name = tensor("value_21_interleave_0"), val = tensor(false)]; tensor value_states_43 = transpose(perm = value_states_43_perm_0, x = var_888)[name = tensor("transpose_175")]; tensor value_21 = concat(axis = var_13, interleave = value_21_interleave_0, values = value_states_43)[name = tensor("value_21")]; tensor mul_10 = mul(x = var_876, y = var_11)[name = tensor("mul_10")]; tensor matmul_10_transpose_y_0 = const()[name = tensor("matmul_10_transpose_y_0"), val = tensor(true)]; tensor matmul_10_transpose_x_0 = const()[name = tensor("matmul_10_transpose_x_0"), val = tensor(false)]; tensor transpose_82_perm_0 = const()[name = tensor("transpose_82_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_83_perm_0 = const()[name = tensor("transpose_83_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_83 = transpose(perm = transpose_83_perm_0, x = key_21)[name = tensor("transpose_173")]; tensor transpose_82 = transpose(perm = transpose_82_perm_0, x = mul_10)[name = tensor("transpose_174")]; tensor matmul_10 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = transpose_82, y = transpose_83)[name = tensor("matmul_10")]; tensor add_10 = add(x = matmul_10, y = reshape_4)[name = tensor("add_10")]; tensor softmax_10_axis_0 = const()[name = tensor("softmax_10_axis_0"), val = tensor(-1)]; tensor softmax_10 = softmax(axis = softmax_10_axis_0, x = add_10)[name = tensor("softmax_10")]; tensor attn_output_41_transpose_x_0 = const()[name = tensor("attn_output_41_transpose_x_0"), val = tensor(false)]; tensor attn_output_41_transpose_y_0 = const()[name = tensor("attn_output_41_transpose_y_0"), val = tensor(false)]; tensor attn_output_41 = matmul(transpose_x = attn_output_41_transpose_x_0, transpose_y = attn_output_41_transpose_y_0, x = softmax_10, y = value_21)[name = tensor("attn_output_41")]; tensor var_904_perm_0 = const()[name = tensor("op_904_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_906 = const()[name = tensor("op_906"), val = tensor([1, 2, -1])]; tensor var_904 = transpose(perm = var_904_perm_0, x = attn_output_41)[name = tensor("transpose_172")]; tensor var_907 = reshape(shape = var_906, x = var_904)[name = tensor("op_907")]; tensor input_129 = linear(bias = decoder_layers_5_self_attn_out_proj_bias, weight = decoder_layers_5_self_attn_out_proj_weight, x = var_907)[name = tensor("linear_53")]; tensor input_131 = add(x = input_125, y = input_129)[name = tensor("input_131")]; tensor hidden_states_55_axes_0 = const()[name = tensor("hidden_states_55_axes_0"), val = tensor([-1])]; tensor hidden_states_55 = layer_norm(axes = hidden_states_55_axes_0, beta = decoder_layers_5_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_5_encoder_attn_layer_norm_weight, x = input_131)[name = tensor("hidden_states_55")]; tensor var_931 = linear(bias = decoder_layers_5_encoder_attn_q_proj_bias, weight = decoder_layers_5_encoder_attn_q_proj_weight, x = hidden_states_55)[name = tensor("linear_54")]; tensor var_932 = const()[name = tensor("op_932"), val = tensor([1, 2, -1, 64])]; tensor var_933 = reshape(shape = var_932, x = var_931)[name = tensor("op_933")]; tensor query_23_perm_0 = const()[name = tensor("query_23_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_45 = linear(bias = decoder_layers_5_encoder_attn_k_proj_bias, weight = decoder_layers_5_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_55")]; tensor value_states_45 = linear(bias = decoder_layers_5_encoder_attn_v_proj_bias, weight = decoder_layers_5_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_56")]; tensor concat_19x = const()[name = tensor("concat_19x"), val = tensor([1, -1, 16, 64])]; tensor var_942 = reshape(shape = concat_19x, x = key_states_45)[name = tensor("op_942")]; tensor key_states_47_perm_0 = const()[name = tensor("key_states_47_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_20x = const()[name = tensor("concat_20x"), val = tensor([1, -1, 16, 64])]; tensor var_945 = reshape(shape = concat_20x, x = value_states_45)[name = tensor("op_945")]; tensor value_states_47_perm_0 = const()[name = tensor("value_states_47_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_23_interleave_0 = const()[name = tensor("key_23_interleave_0"), val = tensor(false)]; tensor key_states_47 = transpose(perm = key_states_47_perm_0, x = var_942)[name = tensor("transpose_170")]; tensor key_23 = concat(axis = var_13, interleave = key_23_interleave_0, values = key_states_47)[name = tensor("key_23")]; tensor value_23_interleave_0 = const()[name = tensor("value_23_interleave_0"), val = tensor(false)]; tensor value_states_47 = transpose(perm = value_states_47_perm_0, x = var_945)[name = tensor("transpose_169")]; tensor value_23 = concat(axis = var_13, interleave = value_23_interleave_0, values = value_states_47)[name = tensor("value_23")]; tensor var_955_shape = shape(x = key_23)[name = tensor("op_955_shape")]; tensor gather_13_indices_0 = const()[name = tensor("gather_13_indices_0"), val = tensor(2)]; tensor gather_13_axis_0 = const()[name = tensor("gather_13_axis_0"), val = tensor(0)]; tensor gather_13_batch_dims_0 = const()[name = tensor("gather_13_batch_dims_0"), val = tensor(0)]; tensor gather_13 = gather(axis = gather_13_axis_0, batch_dims = gather_13_batch_dims_0, indices = gather_13_indices_0, x = var_955_shape)[name = tensor("gather_13")]; tensor concat_21_values0_0 = const()[name = tensor("concat_21_values0_0"), val = tensor(0)]; tensor concat_21_values1_0 = const()[name = tensor("concat_21_values1_0"), val = tensor(0)]; tensor concat_21_values2_0 = const()[name = tensor("concat_21_values2_0"), val = tensor(0)]; tensor concat_21_axis_0 = const()[name = tensor("concat_21_axis_0"), val = tensor(0)]; tensor concat_21_interleave_0 = const()[name = tensor("concat_21_interleave_0"), val = tensor(false)]; tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (concat_21_values0_0, concat_21_values1_0, concat_21_values2_0, gather_13))[name = tensor("concat_21")]; tensor attention_mask_27_begin_0 = const()[name = tensor("attention_mask_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_27_end_mask_0 = const()[name = tensor("attention_mask_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_27 = slice_by_index(begin = attention_mask_27_begin_0, end = concat_21, end_mask = attention_mask_27_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_27")]; tensor query_23 = transpose(perm = query_23_perm_0, x = var_933)[name = tensor("transpose_171")]; tensor mul_11 = mul(x = query_23, y = var_11)[name = tensor("mul_11")]; tensor matmul_11_transpose_y_0 = const()[name = tensor("matmul_11_transpose_y_0"), val = tensor(true)]; tensor matmul_11_transpose_x_0 = const()[name = tensor("matmul_11_transpose_x_0"), val = tensor(false)]; tensor matmul_11 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = mul_11, y = key_23)[name = tensor("matmul_11")]; tensor add_11 = add(x = matmul_11, y = attention_mask_27)[name = tensor("add_11")]; tensor softmax_11_axis_0 = const()[name = tensor("softmax_11_axis_0"), val = tensor(-1)]; tensor softmax_11 = softmax(axis = softmax_11_axis_0, x = add_11)[name = tensor("softmax_11")]; tensor attn_output_45_transpose_x_0 = const()[name = tensor("attn_output_45_transpose_x_0"), val = tensor(false)]; tensor attn_output_45_transpose_y_0 = const()[name = tensor("attn_output_45_transpose_y_0"), val = tensor(false)]; tensor attn_output_45 = matmul(transpose_x = attn_output_45_transpose_x_0, transpose_y = attn_output_45_transpose_y_0, x = softmax_11, y = value_23)[name = tensor("attn_output_45")]; tensor var_961_perm_0 = const()[name = tensor("op_961_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_963 = const()[name = tensor("op_963"), val = tensor([1, 2, -1])]; tensor var_961 = transpose(perm = var_961_perm_0, x = attn_output_45)[name = tensor("transpose_168")]; tensor var_964 = reshape(shape = var_963, x = var_961)[name = tensor("op_964")]; tensor input_135 = linear(bias = decoder_layers_5_encoder_attn_out_proj_bias, weight = decoder_layers_5_encoder_attn_out_proj_weight, x = var_964)[name = tensor("linear_57")]; tensor input_137 = add(x = input_131, y = input_135)[name = tensor("input_137")]; tensor input_139_axes_0 = const()[name = tensor("input_139_axes_0"), val = tensor([-1])]; tensor input_139 = layer_norm(axes = input_139_axes_0, beta = decoder_layers_5_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_5_final_layer_norm_weight, x = input_137)[name = tensor("input_139")]; tensor input_141 = linear(bias = decoder_layers_5_fc1_bias, weight = decoder_layers_5_fc1_weight, x = input_139)[name = tensor("linear_58")]; tensor input_143 = relu(x = input_141)[name = tensor("input_143")]; tensor input_147 = linear(bias = decoder_layers_5_fc2_bias, weight = decoder_layers_5_fc2_weight, x = input_143)[name = tensor("linear_59")]; tensor input_149 = add(x = input_137, y = input_147)[name = tensor("input_149")]; tensor hidden_states_61_axes_0 = const()[name = tensor("hidden_states_61_axes_0"), val = tensor([-1])]; tensor hidden_states_61 = layer_norm(axes = hidden_states_61_axes_0, beta = decoder_layers_6_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_6_self_attn_layer_norm_weight, x = input_149)[name = tensor("hidden_states_61")]; tensor var_1014 = linear(bias = decoder_layers_6_self_attn_q_proj_bias, weight = decoder_layers_6_self_attn_q_proj_weight, x = hidden_states_61)[name = tensor("linear_60")]; tensor var_1015 = const()[name = tensor("op_1015"), val = tensor([1, 2, -1, 64])]; tensor var_1016 = reshape(shape = var_1015, x = var_1014)[name = tensor("op_1016")]; tensor key_states_49 = linear(bias = decoder_layers_6_self_attn_k_proj_bias, weight = decoder_layers_6_self_attn_k_proj_weight, x = hidden_states_61)[name = tensor("linear_61")]; tensor value_states_49 = linear(bias = decoder_layers_6_self_attn_v_proj_bias, weight = decoder_layers_6_self_attn_v_proj_weight, x = hidden_states_61)[name = tensor("linear_62")]; tensor var_1024 = const()[name = tensor("op_1024"), val = tensor([1, 2, -1, 64])]; tensor var_1025 = reshape(shape = var_1024, x = key_states_49)[name = tensor("op_1025")]; tensor var_1027 = const()[name = tensor("op_1027"), val = tensor([1, 2, -1, 64])]; tensor var_1028 = reshape(shape = var_1027, x = value_states_49)[name = tensor("op_1028")]; tensor value_states_51_perm_0 = const()[name = tensor("value_states_51_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_25_interleave_0 = const()[name = tensor("key_25_interleave_0"), val = tensor(false)]; tensor const_129 = const()[name = tensor("const_129"), val = tensor(1)]; tensor key_25 = concat(axis = const_129, interleave = key_25_interleave_0, values = var_1025)[name = tensor("key_25")]; tensor value_25_interleave_0 = const()[name = tensor("value_25_interleave_0"), val = tensor(false)]; tensor value_states_51 = transpose(perm = value_states_51_perm_0, x = var_1028)[name = tensor("transpose_167")]; tensor value_25 = concat(axis = var_13, interleave = value_25_interleave_0, values = value_states_51)[name = tensor("value_25")]; tensor mul_12 = mul(x = var_1016, y = var_11)[name = tensor("mul_12")]; tensor matmul_12_transpose_y_0 = const()[name = tensor("matmul_12_transpose_y_0"), val = tensor(true)]; tensor matmul_12_transpose_x_0 = const()[name = tensor("matmul_12_transpose_x_0"), val = tensor(false)]; tensor transpose_84_perm_0 = const()[name = tensor("transpose_84_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_85_perm_0 = const()[name = tensor("transpose_85_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_85 = transpose(perm = transpose_85_perm_0, x = key_25)[name = tensor("transpose_165")]; tensor transpose_84 = transpose(perm = transpose_84_perm_0, x = mul_12)[name = tensor("transpose_166")]; tensor matmul_12 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = transpose_84, y = transpose_85)[name = tensor("matmul_12")]; tensor add_12 = add(x = matmul_12, y = reshape_4)[name = tensor("add_12")]; tensor softmax_12_axis_0 = const()[name = tensor("softmax_12_axis_0"), val = tensor(-1)]; tensor softmax_12 = softmax(axis = softmax_12_axis_0, x = add_12)[name = tensor("softmax_12")]; tensor attn_output_49_transpose_x_0 = const()[name = tensor("attn_output_49_transpose_x_0"), val = tensor(false)]; tensor attn_output_49_transpose_y_0 = const()[name = tensor("attn_output_49_transpose_y_0"), val = tensor(false)]; tensor attn_output_49 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = softmax_12, y = value_25)[name = tensor("attn_output_49")]; tensor var_1044_perm_0 = const()[name = tensor("op_1044_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1046 = const()[name = tensor("op_1046"), val = tensor([1, 2, -1])]; tensor var_1044 = transpose(perm = var_1044_perm_0, x = attn_output_49)[name = tensor("transpose_164")]; tensor var_1047 = reshape(shape = var_1046, x = var_1044)[name = tensor("op_1047")]; tensor input_153 = linear(bias = decoder_layers_6_self_attn_out_proj_bias, weight = decoder_layers_6_self_attn_out_proj_weight, x = var_1047)[name = tensor("linear_63")]; tensor input_155 = add(x = input_149, y = input_153)[name = tensor("input_155")]; tensor hidden_states_65_axes_0 = const()[name = tensor("hidden_states_65_axes_0"), val = tensor([-1])]; tensor hidden_states_65 = layer_norm(axes = hidden_states_65_axes_0, beta = decoder_layers_6_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_6_encoder_attn_layer_norm_weight, x = input_155)[name = tensor("hidden_states_65")]; tensor var_1071 = linear(bias = decoder_layers_6_encoder_attn_q_proj_bias, weight = decoder_layers_6_encoder_attn_q_proj_weight, x = hidden_states_65)[name = tensor("linear_64")]; tensor var_1072 = const()[name = tensor("op_1072"), val = tensor([1, 2, -1, 64])]; tensor var_1073 = reshape(shape = var_1072, x = var_1071)[name = tensor("op_1073")]; tensor query_27_perm_0 = const()[name = tensor("query_27_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_53 = linear(bias = decoder_layers_6_encoder_attn_k_proj_bias, weight = decoder_layers_6_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_65")]; tensor value_states_53 = linear(bias = decoder_layers_6_encoder_attn_v_proj_bias, weight = decoder_layers_6_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_66")]; tensor concat_22x = const()[name = tensor("concat_22x"), val = tensor([1, -1, 16, 64])]; tensor var_1082 = reshape(shape = concat_22x, x = key_states_53)[name = tensor("op_1082")]; tensor key_states_55_perm_0 = const()[name = tensor("key_states_55_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_23x = const()[name = tensor("concat_23x"), val = tensor([1, -1, 16, 64])]; tensor var_1085 = reshape(shape = concat_23x, x = value_states_53)[name = tensor("op_1085")]; tensor value_states_55_perm_0 = const()[name = tensor("value_states_55_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_27_interleave_0 = const()[name = tensor("key_27_interleave_0"), val = tensor(false)]; tensor key_states_55 = transpose(perm = key_states_55_perm_0, x = var_1082)[name = tensor("transpose_162")]; tensor key_27 = concat(axis = var_13, interleave = key_27_interleave_0, values = key_states_55)[name = tensor("key_27")]; tensor value_27_interleave_0 = const()[name = tensor("value_27_interleave_0"), val = tensor(false)]; tensor value_states_55 = transpose(perm = value_states_55_perm_0, x = var_1085)[name = tensor("transpose_161")]; tensor value_27 = concat(axis = var_13, interleave = value_27_interleave_0, values = value_states_55)[name = tensor("value_27")]; tensor var_1095_shape = shape(x = key_27)[name = tensor("op_1095_shape")]; tensor gather_15_indices_0 = const()[name = tensor("gather_15_indices_0"), val = tensor(2)]; tensor gather_15_axis_0 = const()[name = tensor("gather_15_axis_0"), val = tensor(0)]; tensor gather_15_batch_dims_0 = const()[name = tensor("gather_15_batch_dims_0"), val = tensor(0)]; tensor gather_15 = gather(axis = gather_15_axis_0, batch_dims = gather_15_batch_dims_0, indices = gather_15_indices_0, x = var_1095_shape)[name = tensor("gather_15")]; tensor concat_24_values0_0 = const()[name = tensor("concat_24_values0_0"), val = tensor(0)]; tensor concat_24_values1_0 = const()[name = tensor("concat_24_values1_0"), val = tensor(0)]; tensor concat_24_values2_0 = const()[name = tensor("concat_24_values2_0"), val = tensor(0)]; tensor concat_24_axis_0 = const()[name = tensor("concat_24_axis_0"), val = tensor(0)]; tensor concat_24_interleave_0 = const()[name = tensor("concat_24_interleave_0"), val = tensor(false)]; tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (concat_24_values0_0, concat_24_values1_0, concat_24_values2_0, gather_15))[name = tensor("concat_24")]; tensor attention_mask_31_begin_0 = const()[name = tensor("attention_mask_31_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_31_end_mask_0 = const()[name = tensor("attention_mask_31_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_31 = slice_by_index(begin = attention_mask_31_begin_0, end = concat_24, end_mask = attention_mask_31_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_31")]; tensor query_27 = transpose(perm = query_27_perm_0, x = var_1073)[name = tensor("transpose_163")]; tensor mul_13 = mul(x = query_27, y = var_11)[name = tensor("mul_13")]; tensor matmul_13_transpose_y_0 = const()[name = tensor("matmul_13_transpose_y_0"), val = tensor(true)]; tensor matmul_13_transpose_x_0 = const()[name = tensor("matmul_13_transpose_x_0"), val = tensor(false)]; tensor matmul_13 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = mul_13, y = key_27)[name = tensor("matmul_13")]; tensor add_13 = add(x = matmul_13, y = attention_mask_31)[name = tensor("add_13")]; tensor softmax_13_axis_0 = const()[name = tensor("softmax_13_axis_0"), val = tensor(-1)]; tensor softmax_13 = softmax(axis = softmax_13_axis_0, x = add_13)[name = tensor("softmax_13")]; tensor attn_output_53_transpose_x_0 = const()[name = tensor("attn_output_53_transpose_x_0"), val = tensor(false)]; tensor attn_output_53_transpose_y_0 = const()[name = tensor("attn_output_53_transpose_y_0"), val = tensor(false)]; tensor attn_output_53 = matmul(transpose_x = attn_output_53_transpose_x_0, transpose_y = attn_output_53_transpose_y_0, x = softmax_13, y = value_27)[name = tensor("attn_output_53")]; tensor var_1101_perm_0 = const()[name = tensor("op_1101_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1103 = const()[name = tensor("op_1103"), val = tensor([1, 2, -1])]; tensor var_1101 = transpose(perm = var_1101_perm_0, x = attn_output_53)[name = tensor("transpose_160")]; tensor var_1104 = reshape(shape = var_1103, x = var_1101)[name = tensor("op_1104")]; tensor input_159 = linear(bias = decoder_layers_6_encoder_attn_out_proj_bias, weight = decoder_layers_6_encoder_attn_out_proj_weight, x = var_1104)[name = tensor("linear_67")]; tensor input_161 = add(x = input_155, y = input_159)[name = tensor("input_161")]; tensor input_163_axes_0 = const()[name = tensor("input_163_axes_0"), val = tensor([-1])]; tensor input_163 = layer_norm(axes = input_163_axes_0, beta = decoder_layers_6_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_6_final_layer_norm_weight, x = input_161)[name = tensor("input_163")]; tensor input_165 = linear(bias = decoder_layers_6_fc1_bias, weight = decoder_layers_6_fc1_weight, x = input_163)[name = tensor("linear_68")]; tensor input_167 = relu(x = input_165)[name = tensor("input_167")]; tensor input_171 = linear(bias = decoder_layers_6_fc2_bias, weight = decoder_layers_6_fc2_weight, x = input_167)[name = tensor("linear_69")]; tensor input_173 = add(x = input_161, y = input_171)[name = tensor("input_173")]; tensor hidden_states_71_axes_0 = const()[name = tensor("hidden_states_71_axes_0"), val = tensor([-1])]; tensor hidden_states_71 = layer_norm(axes = hidden_states_71_axes_0, beta = decoder_layers_7_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_7_self_attn_layer_norm_weight, x = input_173)[name = tensor("hidden_states_71")]; tensor var_1154 = linear(bias = decoder_layers_7_self_attn_q_proj_bias, weight = decoder_layers_7_self_attn_q_proj_weight, x = hidden_states_71)[name = tensor("linear_70")]; tensor var_1155 = const()[name = tensor("op_1155"), val = tensor([1, 2, -1, 64])]; tensor var_1156 = reshape(shape = var_1155, x = var_1154)[name = tensor("op_1156")]; tensor key_states_57 = linear(bias = decoder_layers_7_self_attn_k_proj_bias, weight = decoder_layers_7_self_attn_k_proj_weight, x = hidden_states_71)[name = tensor("linear_71")]; tensor value_states_57 = linear(bias = decoder_layers_7_self_attn_v_proj_bias, weight = decoder_layers_7_self_attn_v_proj_weight, x = hidden_states_71)[name = tensor("linear_72")]; tensor var_1164 = const()[name = tensor("op_1164"), val = tensor([1, 2, -1, 64])]; tensor var_1165 = reshape(shape = var_1164, x = key_states_57)[name = tensor("op_1165")]; tensor var_1167 = const()[name = tensor("op_1167"), val = tensor([1, 2, -1, 64])]; tensor var_1168 = reshape(shape = var_1167, x = value_states_57)[name = tensor("op_1168")]; tensor value_states_59_perm_0 = const()[name = tensor("value_states_59_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_29_interleave_0 = const()[name = tensor("key_29_interleave_0"), val = tensor(false)]; tensor const_130 = const()[name = tensor("const_130"), val = tensor(1)]; tensor key_29 = concat(axis = const_130, interleave = key_29_interleave_0, values = var_1165)[name = tensor("key_29")]; tensor value_29_interleave_0 = const()[name = tensor("value_29_interleave_0"), val = tensor(false)]; tensor value_states_59 = transpose(perm = value_states_59_perm_0, x = var_1168)[name = tensor("transpose_159")]; tensor value_29 = concat(axis = var_13, interleave = value_29_interleave_0, values = value_states_59)[name = tensor("value_29")]; tensor mul_14 = mul(x = var_1156, y = var_11)[name = tensor("mul_14")]; tensor matmul_14_transpose_y_0 = const()[name = tensor("matmul_14_transpose_y_0"), val = tensor(true)]; tensor matmul_14_transpose_x_0 = const()[name = tensor("matmul_14_transpose_x_0"), val = tensor(false)]; tensor transpose_86_perm_0 = const()[name = tensor("transpose_86_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_87_perm_0 = const()[name = tensor("transpose_87_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_87 = transpose(perm = transpose_87_perm_0, x = key_29)[name = tensor("transpose_157")]; tensor transpose_86 = transpose(perm = transpose_86_perm_0, x = mul_14)[name = tensor("transpose_158")]; tensor matmul_14 = matmul(transpose_x = matmul_14_transpose_x_0, transpose_y = matmul_14_transpose_y_0, x = transpose_86, y = transpose_87)[name = tensor("matmul_14")]; tensor add_14 = add(x = matmul_14, y = reshape_4)[name = tensor("add_14")]; tensor softmax_14_axis_0 = const()[name = tensor("softmax_14_axis_0"), val = tensor(-1)]; tensor softmax_14 = softmax(axis = softmax_14_axis_0, x = add_14)[name = tensor("softmax_14")]; tensor attn_output_57_transpose_x_0 = const()[name = tensor("attn_output_57_transpose_x_0"), val = tensor(false)]; tensor attn_output_57_transpose_y_0 = const()[name = tensor("attn_output_57_transpose_y_0"), val = tensor(false)]; tensor attn_output_57 = matmul(transpose_x = attn_output_57_transpose_x_0, transpose_y = attn_output_57_transpose_y_0, x = softmax_14, y = value_29)[name = tensor("attn_output_57")]; tensor var_1184_perm_0 = const()[name = tensor("op_1184_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1186 = const()[name = tensor("op_1186"), val = tensor([1, 2, -1])]; tensor var_1184 = transpose(perm = var_1184_perm_0, x = attn_output_57)[name = tensor("transpose_156")]; tensor var_1187 = reshape(shape = var_1186, x = var_1184)[name = tensor("op_1187")]; tensor input_177 = linear(bias = decoder_layers_7_self_attn_out_proj_bias, weight = decoder_layers_7_self_attn_out_proj_weight, x = var_1187)[name = tensor("linear_73")]; tensor input_179 = add(x = input_173, y = input_177)[name = tensor("input_179")]; tensor hidden_states_75_axes_0 = const()[name = tensor("hidden_states_75_axes_0"), val = tensor([-1])]; tensor hidden_states_75 = layer_norm(axes = hidden_states_75_axes_0, beta = decoder_layers_7_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_7_encoder_attn_layer_norm_weight, x = input_179)[name = tensor("hidden_states_75")]; tensor var_1211 = linear(bias = decoder_layers_7_encoder_attn_q_proj_bias, weight = decoder_layers_7_encoder_attn_q_proj_weight, x = hidden_states_75)[name = tensor("linear_74")]; tensor var_1212 = const()[name = tensor("op_1212"), val = tensor([1, 2, -1, 64])]; tensor var_1213 = reshape(shape = var_1212, x = var_1211)[name = tensor("op_1213")]; tensor query_31_perm_0 = const()[name = tensor("query_31_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_61 = linear(bias = decoder_layers_7_encoder_attn_k_proj_bias, weight = decoder_layers_7_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_75")]; tensor value_states_61 = linear(bias = decoder_layers_7_encoder_attn_v_proj_bias, weight = decoder_layers_7_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_76")]; tensor concat_25x = const()[name = tensor("concat_25x"), val = tensor([1, -1, 16, 64])]; tensor var_1222 = reshape(shape = concat_25x, x = key_states_61)[name = tensor("op_1222")]; tensor key_states_63_perm_0 = const()[name = tensor("key_states_63_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_26x = const()[name = tensor("concat_26x"), val = tensor([1, -1, 16, 64])]; tensor var_1225 = reshape(shape = concat_26x, x = value_states_61)[name = tensor("op_1225")]; tensor value_states_63_perm_0 = const()[name = tensor("value_states_63_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_31_interleave_0 = const()[name = tensor("key_31_interleave_0"), val = tensor(false)]; tensor key_states_63 = transpose(perm = key_states_63_perm_0, x = var_1222)[name = tensor("transpose_154")]; tensor key_31 = concat(axis = var_13, interleave = key_31_interleave_0, values = key_states_63)[name = tensor("key_31")]; tensor value_31_interleave_0 = const()[name = tensor("value_31_interleave_0"), val = tensor(false)]; tensor value_states_63 = transpose(perm = value_states_63_perm_0, x = var_1225)[name = tensor("transpose_153")]; tensor value_31 = concat(axis = var_13, interleave = value_31_interleave_0, values = value_states_63)[name = tensor("value_31")]; tensor var_1235_shape = shape(x = key_31)[name = tensor("op_1235_shape")]; tensor gather_17_indices_0 = const()[name = tensor("gather_17_indices_0"), val = tensor(2)]; tensor gather_17_axis_0 = const()[name = tensor("gather_17_axis_0"), val = tensor(0)]; tensor gather_17_batch_dims_0 = const()[name = tensor("gather_17_batch_dims_0"), val = tensor(0)]; tensor gather_17 = gather(axis = gather_17_axis_0, batch_dims = gather_17_batch_dims_0, indices = gather_17_indices_0, x = var_1235_shape)[name = tensor("gather_17")]; tensor concat_27_values0_0 = const()[name = tensor("concat_27_values0_0"), val = tensor(0)]; tensor concat_27_values1_0 = const()[name = tensor("concat_27_values1_0"), val = tensor(0)]; tensor concat_27_values2_0 = const()[name = tensor("concat_27_values2_0"), val = tensor(0)]; tensor concat_27_axis_0 = const()[name = tensor("concat_27_axis_0"), val = tensor(0)]; tensor concat_27_interleave_0 = const()[name = tensor("concat_27_interleave_0"), val = tensor(false)]; tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (concat_27_values0_0, concat_27_values1_0, concat_27_values2_0, gather_17))[name = tensor("concat_27")]; tensor attention_mask_35_begin_0 = const()[name = tensor("attention_mask_35_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_35_end_mask_0 = const()[name = tensor("attention_mask_35_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_35 = slice_by_index(begin = attention_mask_35_begin_0, end = concat_27, end_mask = attention_mask_35_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_35")]; tensor query_31 = transpose(perm = query_31_perm_0, x = var_1213)[name = tensor("transpose_155")]; tensor mul_15 = mul(x = query_31, y = var_11)[name = tensor("mul_15")]; tensor matmul_15_transpose_y_0 = const()[name = tensor("matmul_15_transpose_y_0"), val = tensor(true)]; tensor matmul_15_transpose_x_0 = const()[name = tensor("matmul_15_transpose_x_0"), val = tensor(false)]; tensor matmul_15 = matmul(transpose_x = matmul_15_transpose_x_0, transpose_y = matmul_15_transpose_y_0, x = mul_15, y = key_31)[name = tensor("matmul_15")]; tensor add_15 = add(x = matmul_15, y = attention_mask_35)[name = tensor("add_15")]; tensor softmax_15_axis_0 = const()[name = tensor("softmax_15_axis_0"), val = tensor(-1)]; tensor softmax_15 = softmax(axis = softmax_15_axis_0, x = add_15)[name = tensor("softmax_15")]; tensor attn_output_61_transpose_x_0 = const()[name = tensor("attn_output_61_transpose_x_0"), val = tensor(false)]; tensor attn_output_61_transpose_y_0 = const()[name = tensor("attn_output_61_transpose_y_0"), val = tensor(false)]; tensor attn_output_61 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = softmax_15, y = value_31)[name = tensor("attn_output_61")]; tensor var_1241_perm_0 = const()[name = tensor("op_1241_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1243 = const()[name = tensor("op_1243"), val = tensor([1, 2, -1])]; tensor var_1241 = transpose(perm = var_1241_perm_0, x = attn_output_61)[name = tensor("transpose_152")]; tensor var_1244 = reshape(shape = var_1243, x = var_1241)[name = tensor("op_1244")]; tensor input_183 = linear(bias = decoder_layers_7_encoder_attn_out_proj_bias, weight = decoder_layers_7_encoder_attn_out_proj_weight, x = var_1244)[name = tensor("linear_77")]; tensor input_185 = add(x = input_179, y = input_183)[name = tensor("input_185")]; tensor input_187_axes_0 = const()[name = tensor("input_187_axes_0"), val = tensor([-1])]; tensor input_187 = layer_norm(axes = input_187_axes_0, beta = decoder_layers_7_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_7_final_layer_norm_weight, x = input_185)[name = tensor("input_187")]; tensor input_189 = linear(bias = decoder_layers_7_fc1_bias, weight = decoder_layers_7_fc1_weight, x = input_187)[name = tensor("linear_78")]; tensor input_191 = relu(x = input_189)[name = tensor("input_191")]; tensor input_195 = linear(bias = decoder_layers_7_fc2_bias, weight = decoder_layers_7_fc2_weight, x = input_191)[name = tensor("linear_79")]; tensor input_197 = add(x = input_185, y = input_195)[name = tensor("input_197")]; tensor hidden_states_81_axes_0 = const()[name = tensor("hidden_states_81_axes_0"), val = tensor([-1])]; tensor hidden_states_81 = layer_norm(axes = hidden_states_81_axes_0, beta = decoder_layers_8_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_8_self_attn_layer_norm_weight, x = input_197)[name = tensor("hidden_states_81")]; tensor var_1294 = linear(bias = decoder_layers_8_self_attn_q_proj_bias, weight = decoder_layers_8_self_attn_q_proj_weight, x = hidden_states_81)[name = tensor("linear_80")]; tensor var_1295 = const()[name = tensor("op_1295"), val = tensor([1, 2, -1, 64])]; tensor var_1296 = reshape(shape = var_1295, x = var_1294)[name = tensor("op_1296")]; tensor key_states_65 = linear(bias = decoder_layers_8_self_attn_k_proj_bias, weight = decoder_layers_8_self_attn_k_proj_weight, x = hidden_states_81)[name = tensor("linear_81")]; tensor value_states_65 = linear(bias = decoder_layers_8_self_attn_v_proj_bias, weight = decoder_layers_8_self_attn_v_proj_weight, x = hidden_states_81)[name = tensor("linear_82")]; tensor var_1304 = const()[name = tensor("op_1304"), val = tensor([1, 2, -1, 64])]; tensor var_1305 = reshape(shape = var_1304, x = key_states_65)[name = tensor("op_1305")]; tensor var_1307 = const()[name = tensor("op_1307"), val = tensor([1, 2, -1, 64])]; tensor var_1308 = reshape(shape = var_1307, x = value_states_65)[name = tensor("op_1308")]; tensor value_states_67_perm_0 = const()[name = tensor("value_states_67_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_33_interleave_0 = const()[name = tensor("key_33_interleave_0"), val = tensor(false)]; tensor const_131 = const()[name = tensor("const_131"), val = tensor(1)]; tensor key_33 = concat(axis = const_131, interleave = key_33_interleave_0, values = var_1305)[name = tensor("key_33")]; tensor value_33_interleave_0 = const()[name = tensor("value_33_interleave_0"), val = tensor(false)]; tensor value_states_67 = transpose(perm = value_states_67_perm_0, x = var_1308)[name = tensor("transpose_151")]; tensor value_33 = concat(axis = var_13, interleave = value_33_interleave_0, values = value_states_67)[name = tensor("value_33")]; tensor mul_16 = mul(x = var_1296, y = var_11)[name = tensor("mul_16")]; tensor matmul_16_transpose_y_0 = const()[name = tensor("matmul_16_transpose_y_0"), val = tensor(true)]; tensor matmul_16_transpose_x_0 = const()[name = tensor("matmul_16_transpose_x_0"), val = tensor(false)]; tensor transpose_88_perm_0 = const()[name = tensor("transpose_88_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_89_perm_0 = const()[name = tensor("transpose_89_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_89 = transpose(perm = transpose_89_perm_0, x = key_33)[name = tensor("transpose_149")]; tensor transpose_88 = transpose(perm = transpose_88_perm_0, x = mul_16)[name = tensor("transpose_150")]; tensor matmul_16 = matmul(transpose_x = matmul_16_transpose_x_0, transpose_y = matmul_16_transpose_y_0, x = transpose_88, y = transpose_89)[name = tensor("matmul_16")]; tensor add_16 = add(x = matmul_16, y = reshape_4)[name = tensor("add_16")]; tensor softmax_16_axis_0 = const()[name = tensor("softmax_16_axis_0"), val = tensor(-1)]; tensor softmax_16 = softmax(axis = softmax_16_axis_0, x = add_16)[name = tensor("softmax_16")]; tensor attn_output_65_transpose_x_0 = const()[name = tensor("attn_output_65_transpose_x_0"), val = tensor(false)]; tensor attn_output_65_transpose_y_0 = const()[name = tensor("attn_output_65_transpose_y_0"), val = tensor(false)]; tensor attn_output_65 = matmul(transpose_x = attn_output_65_transpose_x_0, transpose_y = attn_output_65_transpose_y_0, x = softmax_16, y = value_33)[name = tensor("attn_output_65")]; tensor var_1324_perm_0 = const()[name = tensor("op_1324_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1326 = const()[name = tensor("op_1326"), val = tensor([1, 2, -1])]; tensor var_1324 = transpose(perm = var_1324_perm_0, x = attn_output_65)[name = tensor("transpose_148")]; tensor var_1327 = reshape(shape = var_1326, x = var_1324)[name = tensor("op_1327")]; tensor input_201 = linear(bias = decoder_layers_8_self_attn_out_proj_bias, weight = decoder_layers_8_self_attn_out_proj_weight, x = var_1327)[name = tensor("linear_83")]; tensor input_203 = add(x = input_197, y = input_201)[name = tensor("input_203")]; tensor hidden_states_85_axes_0 = const()[name = tensor("hidden_states_85_axes_0"), val = tensor([-1])]; tensor hidden_states_85 = layer_norm(axes = hidden_states_85_axes_0, beta = decoder_layers_8_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_8_encoder_attn_layer_norm_weight, x = input_203)[name = tensor("hidden_states_85")]; tensor var_1351 = linear(bias = decoder_layers_8_encoder_attn_q_proj_bias, weight = decoder_layers_8_encoder_attn_q_proj_weight, x = hidden_states_85)[name = tensor("linear_84")]; tensor var_1352 = const()[name = tensor("op_1352"), val = tensor([1, 2, -1, 64])]; tensor var_1353 = reshape(shape = var_1352, x = var_1351)[name = tensor("op_1353")]; tensor query_35_perm_0 = const()[name = tensor("query_35_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_69 = linear(bias = decoder_layers_8_encoder_attn_k_proj_bias, weight = decoder_layers_8_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_85")]; tensor value_states_69 = linear(bias = decoder_layers_8_encoder_attn_v_proj_bias, weight = decoder_layers_8_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_86")]; tensor concat_28x = const()[name = tensor("concat_28x"), val = tensor([1, -1, 16, 64])]; tensor var_1362 = reshape(shape = concat_28x, x = key_states_69)[name = tensor("op_1362")]; tensor key_states_71_perm_0 = const()[name = tensor("key_states_71_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_29x = const()[name = tensor("concat_29x"), val = tensor([1, -1, 16, 64])]; tensor var_1365 = reshape(shape = concat_29x, x = value_states_69)[name = tensor("op_1365")]; tensor value_states_71_perm_0 = const()[name = tensor("value_states_71_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_35_interleave_0 = const()[name = tensor("key_35_interleave_0"), val = tensor(false)]; tensor key_states_71 = transpose(perm = key_states_71_perm_0, x = var_1362)[name = tensor("transpose_146")]; tensor key_35 = concat(axis = var_13, interleave = key_35_interleave_0, values = key_states_71)[name = tensor("key_35")]; tensor value_35_interleave_0 = const()[name = tensor("value_35_interleave_0"), val = tensor(false)]; tensor value_states_71 = transpose(perm = value_states_71_perm_0, x = var_1365)[name = tensor("transpose_145")]; tensor value_35 = concat(axis = var_13, interleave = value_35_interleave_0, values = value_states_71)[name = tensor("value_35")]; tensor var_1375_shape = shape(x = key_35)[name = tensor("op_1375_shape")]; tensor gather_19_indices_0 = const()[name = tensor("gather_19_indices_0"), val = tensor(2)]; tensor gather_19_axis_0 = const()[name = tensor("gather_19_axis_0"), val = tensor(0)]; tensor gather_19_batch_dims_0 = const()[name = tensor("gather_19_batch_dims_0"), val = tensor(0)]; tensor gather_19 = gather(axis = gather_19_axis_0, batch_dims = gather_19_batch_dims_0, indices = gather_19_indices_0, x = var_1375_shape)[name = tensor("gather_19")]; tensor concat_30_values0_0 = const()[name = tensor("concat_30_values0_0"), val = tensor(0)]; tensor concat_30_values1_0 = const()[name = tensor("concat_30_values1_0"), val = tensor(0)]; tensor concat_30_values2_0 = const()[name = tensor("concat_30_values2_0"), val = tensor(0)]; tensor concat_30_axis_0 = const()[name = tensor("concat_30_axis_0"), val = tensor(0)]; tensor concat_30_interleave_0 = const()[name = tensor("concat_30_interleave_0"), val = tensor(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (concat_30_values0_0, concat_30_values1_0, concat_30_values2_0, gather_19))[name = tensor("concat_30")]; tensor attention_mask_39_begin_0 = const()[name = tensor("attention_mask_39_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_39_end_mask_0 = const()[name = tensor("attention_mask_39_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_39 = slice_by_index(begin = attention_mask_39_begin_0, end = concat_30, end_mask = attention_mask_39_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_39")]; tensor query_35 = transpose(perm = query_35_perm_0, x = var_1353)[name = tensor("transpose_147")]; tensor mul_17 = mul(x = query_35, y = var_11)[name = tensor("mul_17")]; tensor matmul_17_transpose_y_0 = const()[name = tensor("matmul_17_transpose_y_0"), val = tensor(true)]; tensor matmul_17_transpose_x_0 = const()[name = tensor("matmul_17_transpose_x_0"), val = tensor(false)]; tensor matmul_17 = matmul(transpose_x = matmul_17_transpose_x_0, transpose_y = matmul_17_transpose_y_0, x = mul_17, y = key_35)[name = tensor("matmul_17")]; tensor add_17 = add(x = matmul_17, y = attention_mask_39)[name = tensor("add_17")]; tensor softmax_17_axis_0 = const()[name = tensor("softmax_17_axis_0"), val = tensor(-1)]; tensor softmax_17 = softmax(axis = softmax_17_axis_0, x = add_17)[name = tensor("softmax_17")]; tensor attn_output_69_transpose_x_0 = const()[name = tensor("attn_output_69_transpose_x_0"), val = tensor(false)]; tensor attn_output_69_transpose_y_0 = const()[name = tensor("attn_output_69_transpose_y_0"), val = tensor(false)]; tensor attn_output_69 = matmul(transpose_x = attn_output_69_transpose_x_0, transpose_y = attn_output_69_transpose_y_0, x = softmax_17, y = value_35)[name = tensor("attn_output_69")]; tensor var_1381_perm_0 = const()[name = tensor("op_1381_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1383 = const()[name = tensor("op_1383"), val = tensor([1, 2, -1])]; tensor var_1381 = transpose(perm = var_1381_perm_0, x = attn_output_69)[name = tensor("transpose_144")]; tensor var_1384 = reshape(shape = var_1383, x = var_1381)[name = tensor("op_1384")]; tensor input_207 = linear(bias = decoder_layers_8_encoder_attn_out_proj_bias, weight = decoder_layers_8_encoder_attn_out_proj_weight, x = var_1384)[name = tensor("linear_87")]; tensor input_209 = add(x = input_203, y = input_207)[name = tensor("input_209")]; tensor input_211_axes_0 = const()[name = tensor("input_211_axes_0"), val = tensor([-1])]; tensor input_211 = layer_norm(axes = input_211_axes_0, beta = decoder_layers_8_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_8_final_layer_norm_weight, x = input_209)[name = tensor("input_211")]; tensor input_213 = linear(bias = decoder_layers_8_fc1_bias, weight = decoder_layers_8_fc1_weight, x = input_211)[name = tensor("linear_88")]; tensor input_215 = relu(x = input_213)[name = tensor("input_215")]; tensor input_219 = linear(bias = decoder_layers_8_fc2_bias, weight = decoder_layers_8_fc2_weight, x = input_215)[name = tensor("linear_89")]; tensor input_221 = add(x = input_209, y = input_219)[name = tensor("input_221")]; tensor hidden_states_91_axes_0 = const()[name = tensor("hidden_states_91_axes_0"), val = tensor([-1])]; tensor hidden_states_91 = layer_norm(axes = hidden_states_91_axes_0, beta = decoder_layers_9_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_9_self_attn_layer_norm_weight, x = input_221)[name = tensor("hidden_states_91")]; tensor var_1434 = linear(bias = decoder_layers_9_self_attn_q_proj_bias, weight = decoder_layers_9_self_attn_q_proj_weight, x = hidden_states_91)[name = tensor("linear_90")]; tensor var_1435 = const()[name = tensor("op_1435"), val = tensor([1, 2, -1, 64])]; tensor var_1436 = reshape(shape = var_1435, x = var_1434)[name = tensor("op_1436")]; tensor key_states_73 = linear(bias = decoder_layers_9_self_attn_k_proj_bias, weight = decoder_layers_9_self_attn_k_proj_weight, x = hidden_states_91)[name = tensor("linear_91")]; tensor value_states_73 = linear(bias = decoder_layers_9_self_attn_v_proj_bias, weight = decoder_layers_9_self_attn_v_proj_weight, x = hidden_states_91)[name = tensor("linear_92")]; tensor var_1444 = const()[name = tensor("op_1444"), val = tensor([1, 2, -1, 64])]; tensor var_1445 = reshape(shape = var_1444, x = key_states_73)[name = tensor("op_1445")]; tensor var_1447 = const()[name = tensor("op_1447"), val = tensor([1, 2, -1, 64])]; tensor var_1448 = reshape(shape = var_1447, x = value_states_73)[name = tensor("op_1448")]; tensor value_states_75_perm_0 = const()[name = tensor("value_states_75_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_37_interleave_0 = const()[name = tensor("key_37_interleave_0"), val = tensor(false)]; tensor const_132 = const()[name = tensor("const_132"), val = tensor(1)]; tensor key_37 = concat(axis = const_132, interleave = key_37_interleave_0, values = var_1445)[name = tensor("key_37")]; tensor value_37_interleave_0 = const()[name = tensor("value_37_interleave_0"), val = tensor(false)]; tensor value_states_75 = transpose(perm = value_states_75_perm_0, x = var_1448)[name = tensor("transpose_143")]; tensor value_37 = concat(axis = var_13, interleave = value_37_interleave_0, values = value_states_75)[name = tensor("value_37")]; tensor mul_18 = mul(x = var_1436, y = var_11)[name = tensor("mul_18")]; tensor matmul_18_transpose_y_0 = const()[name = tensor("matmul_18_transpose_y_0"), val = tensor(true)]; tensor matmul_18_transpose_x_0 = const()[name = tensor("matmul_18_transpose_x_0"), val = tensor(false)]; tensor transpose_90_perm_0 = const()[name = tensor("transpose_90_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_91_perm_0 = const()[name = tensor("transpose_91_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_91 = transpose(perm = transpose_91_perm_0, x = key_37)[name = tensor("transpose_141")]; tensor transpose_90 = transpose(perm = transpose_90_perm_0, x = mul_18)[name = tensor("transpose_142")]; tensor matmul_18 = matmul(transpose_x = matmul_18_transpose_x_0, transpose_y = matmul_18_transpose_y_0, x = transpose_90, y = transpose_91)[name = tensor("matmul_18")]; tensor add_18 = add(x = matmul_18, y = reshape_4)[name = tensor("add_18")]; tensor softmax_18_axis_0 = const()[name = tensor("softmax_18_axis_0"), val = tensor(-1)]; tensor softmax_18 = softmax(axis = softmax_18_axis_0, x = add_18)[name = tensor("softmax_18")]; tensor attn_output_73_transpose_x_0 = const()[name = tensor("attn_output_73_transpose_x_0"), val = tensor(false)]; tensor attn_output_73_transpose_y_0 = const()[name = tensor("attn_output_73_transpose_y_0"), val = tensor(false)]; tensor attn_output_73 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = softmax_18, y = value_37)[name = tensor("attn_output_73")]; tensor var_1464_perm_0 = const()[name = tensor("op_1464_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1466 = const()[name = tensor("op_1466"), val = tensor([1, 2, -1])]; tensor var_1464 = transpose(perm = var_1464_perm_0, x = attn_output_73)[name = tensor("transpose_140")]; tensor var_1467 = reshape(shape = var_1466, x = var_1464)[name = tensor("op_1467")]; tensor input_225 = linear(bias = decoder_layers_9_self_attn_out_proj_bias, weight = decoder_layers_9_self_attn_out_proj_weight, x = var_1467)[name = tensor("linear_93")]; tensor input_227 = add(x = input_221, y = input_225)[name = tensor("input_227")]; tensor hidden_states_95_axes_0 = const()[name = tensor("hidden_states_95_axes_0"), val = tensor([-1])]; tensor hidden_states_95 = layer_norm(axes = hidden_states_95_axes_0, beta = decoder_layers_9_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_9_encoder_attn_layer_norm_weight, x = input_227)[name = tensor("hidden_states_95")]; tensor var_1491 = linear(bias = decoder_layers_9_encoder_attn_q_proj_bias, weight = decoder_layers_9_encoder_attn_q_proj_weight, x = hidden_states_95)[name = tensor("linear_94")]; tensor var_1492 = const()[name = tensor("op_1492"), val = tensor([1, 2, -1, 64])]; tensor var_1493 = reshape(shape = var_1492, x = var_1491)[name = tensor("op_1493")]; tensor query_39_perm_0 = const()[name = tensor("query_39_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_77 = linear(bias = decoder_layers_9_encoder_attn_k_proj_bias, weight = decoder_layers_9_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_95")]; tensor value_states_77 = linear(bias = decoder_layers_9_encoder_attn_v_proj_bias, weight = decoder_layers_9_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_96")]; tensor concat_31x = const()[name = tensor("concat_31x"), val = tensor([1, -1, 16, 64])]; tensor var_1502 = reshape(shape = concat_31x, x = key_states_77)[name = tensor("op_1502")]; tensor key_states_79_perm_0 = const()[name = tensor("key_states_79_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_32x = const()[name = tensor("concat_32x"), val = tensor([1, -1, 16, 64])]; tensor var_1505 = reshape(shape = concat_32x, x = value_states_77)[name = tensor("op_1505")]; tensor value_states_79_perm_0 = const()[name = tensor("value_states_79_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_39_interleave_0 = const()[name = tensor("key_39_interleave_0"), val = tensor(false)]; tensor key_states_79 = transpose(perm = key_states_79_perm_0, x = var_1502)[name = tensor("transpose_138")]; tensor key_39 = concat(axis = var_13, interleave = key_39_interleave_0, values = key_states_79)[name = tensor("key_39")]; tensor value_39_interleave_0 = const()[name = tensor("value_39_interleave_0"), val = tensor(false)]; tensor value_states_79 = transpose(perm = value_states_79_perm_0, x = var_1505)[name = tensor("transpose_137")]; tensor value_39 = concat(axis = var_13, interleave = value_39_interleave_0, values = value_states_79)[name = tensor("value_39")]; tensor var_1515_shape = shape(x = key_39)[name = tensor("op_1515_shape")]; tensor gather_21_indices_0 = const()[name = tensor("gather_21_indices_0"), val = tensor(2)]; tensor gather_21_axis_0 = const()[name = tensor("gather_21_axis_0"), val = tensor(0)]; tensor gather_21_batch_dims_0 = const()[name = tensor("gather_21_batch_dims_0"), val = tensor(0)]; tensor gather_21 = gather(axis = gather_21_axis_0, batch_dims = gather_21_batch_dims_0, indices = gather_21_indices_0, x = var_1515_shape)[name = tensor("gather_21")]; tensor concat_33_values0_0 = const()[name = tensor("concat_33_values0_0"), val = tensor(0)]; tensor concat_33_values1_0 = const()[name = tensor("concat_33_values1_0"), val = tensor(0)]; tensor concat_33_values2_0 = const()[name = tensor("concat_33_values2_0"), val = tensor(0)]; tensor concat_33_axis_0 = const()[name = tensor("concat_33_axis_0"), val = tensor(0)]; tensor concat_33_interleave_0 = const()[name = tensor("concat_33_interleave_0"), val = tensor(false)]; tensor concat_33 = concat(axis = concat_33_axis_0, interleave = concat_33_interleave_0, values = (concat_33_values0_0, concat_33_values1_0, concat_33_values2_0, gather_21))[name = tensor("concat_33")]; tensor attention_mask_43_begin_0 = const()[name = tensor("attention_mask_43_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_43_end_mask_0 = const()[name = tensor("attention_mask_43_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_43 = slice_by_index(begin = attention_mask_43_begin_0, end = concat_33, end_mask = attention_mask_43_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_43")]; tensor query_39 = transpose(perm = query_39_perm_0, x = var_1493)[name = tensor("transpose_139")]; tensor mul_19 = mul(x = query_39, y = var_11)[name = tensor("mul_19")]; tensor matmul_19_transpose_y_0 = const()[name = tensor("matmul_19_transpose_y_0"), val = tensor(true)]; tensor matmul_19_transpose_x_0 = const()[name = tensor("matmul_19_transpose_x_0"), val = tensor(false)]; tensor matmul_19 = matmul(transpose_x = matmul_19_transpose_x_0, transpose_y = matmul_19_transpose_y_0, x = mul_19, y = key_39)[name = tensor("matmul_19")]; tensor add_19 = add(x = matmul_19, y = attention_mask_43)[name = tensor("add_19")]; tensor softmax_19_axis_0 = const()[name = tensor("softmax_19_axis_0"), val = tensor(-1)]; tensor softmax_19 = softmax(axis = softmax_19_axis_0, x = add_19)[name = tensor("softmax_19")]; tensor attn_output_77_transpose_x_0 = const()[name = tensor("attn_output_77_transpose_x_0"), val = tensor(false)]; tensor attn_output_77_transpose_y_0 = const()[name = tensor("attn_output_77_transpose_y_0"), val = tensor(false)]; tensor attn_output_77 = matmul(transpose_x = attn_output_77_transpose_x_0, transpose_y = attn_output_77_transpose_y_0, x = softmax_19, y = value_39)[name = tensor("attn_output_77")]; tensor var_1521_perm_0 = const()[name = tensor("op_1521_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1523 = const()[name = tensor("op_1523"), val = tensor([1, 2, -1])]; tensor var_1521 = transpose(perm = var_1521_perm_0, x = attn_output_77)[name = tensor("transpose_136")]; tensor var_1524 = reshape(shape = var_1523, x = var_1521)[name = tensor("op_1524")]; tensor input_231 = linear(bias = decoder_layers_9_encoder_attn_out_proj_bias, weight = decoder_layers_9_encoder_attn_out_proj_weight, x = var_1524)[name = tensor("linear_97")]; tensor input_233 = add(x = input_227, y = input_231)[name = tensor("input_233")]; tensor input_235_axes_0 = const()[name = tensor("input_235_axes_0"), val = tensor([-1])]; tensor input_235 = layer_norm(axes = input_235_axes_0, beta = decoder_layers_9_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_9_final_layer_norm_weight, x = input_233)[name = tensor("input_235")]; tensor input_237 = linear(bias = decoder_layers_9_fc1_bias, weight = decoder_layers_9_fc1_weight, x = input_235)[name = tensor("linear_98")]; tensor input_239 = relu(x = input_237)[name = tensor("input_239")]; tensor input_243 = linear(bias = decoder_layers_9_fc2_bias, weight = decoder_layers_9_fc2_weight, x = input_239)[name = tensor("linear_99")]; tensor input_245 = add(x = input_233, y = input_243)[name = tensor("input_245")]; tensor hidden_states_101_axes_0 = const()[name = tensor("hidden_states_101_axes_0"), val = tensor([-1])]; tensor hidden_states_101 = layer_norm(axes = hidden_states_101_axes_0, beta = decoder_layers_10_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_10_self_attn_layer_norm_weight, x = input_245)[name = tensor("hidden_states_101")]; tensor var_1574 = linear(bias = decoder_layers_10_self_attn_q_proj_bias, weight = decoder_layers_10_self_attn_q_proj_weight, x = hidden_states_101)[name = tensor("linear_100")]; tensor var_1575 = const()[name = tensor("op_1575"), val = tensor([1, 2, -1, 64])]; tensor var_1576 = reshape(shape = var_1575, x = var_1574)[name = tensor("op_1576")]; tensor key_states_81 = linear(bias = decoder_layers_10_self_attn_k_proj_bias, weight = decoder_layers_10_self_attn_k_proj_weight, x = hidden_states_101)[name = tensor("linear_101")]; tensor value_states_81 = linear(bias = decoder_layers_10_self_attn_v_proj_bias, weight = decoder_layers_10_self_attn_v_proj_weight, x = hidden_states_101)[name = tensor("linear_102")]; tensor var_1584 = const()[name = tensor("op_1584"), val = tensor([1, 2, -1, 64])]; tensor var_1585 = reshape(shape = var_1584, x = key_states_81)[name = tensor("op_1585")]; tensor var_1587 = const()[name = tensor("op_1587"), val = tensor([1, 2, -1, 64])]; tensor var_1588 = reshape(shape = var_1587, x = value_states_81)[name = tensor("op_1588")]; tensor value_states_83_perm_0 = const()[name = tensor("value_states_83_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_41_interleave_0 = const()[name = tensor("key_41_interleave_0"), val = tensor(false)]; tensor const_133 = const()[name = tensor("const_133"), val = tensor(1)]; tensor key_41 = concat(axis = const_133, interleave = key_41_interleave_0, values = var_1585)[name = tensor("key_41")]; tensor value_41_interleave_0 = const()[name = tensor("value_41_interleave_0"), val = tensor(false)]; tensor value_states_83 = transpose(perm = value_states_83_perm_0, x = var_1588)[name = tensor("transpose_135")]; tensor value_41 = concat(axis = var_13, interleave = value_41_interleave_0, values = value_states_83)[name = tensor("value_41")]; tensor mul_20 = mul(x = var_1576, y = var_11)[name = tensor("mul_20")]; tensor matmul_20_transpose_y_0 = const()[name = tensor("matmul_20_transpose_y_0"), val = tensor(true)]; tensor matmul_20_transpose_x_0 = const()[name = tensor("matmul_20_transpose_x_0"), val = tensor(false)]; tensor transpose_92_perm_0 = const()[name = tensor("transpose_92_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_93_perm_0 = const()[name = tensor("transpose_93_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_93 = transpose(perm = transpose_93_perm_0, x = key_41)[name = tensor("transpose_133")]; tensor transpose_92 = transpose(perm = transpose_92_perm_0, x = mul_20)[name = tensor("transpose_134")]; tensor matmul_20 = matmul(transpose_x = matmul_20_transpose_x_0, transpose_y = matmul_20_transpose_y_0, x = transpose_92, y = transpose_93)[name = tensor("matmul_20")]; tensor add_20 = add(x = matmul_20, y = reshape_4)[name = tensor("add_20")]; tensor softmax_20_axis_0 = const()[name = tensor("softmax_20_axis_0"), val = tensor(-1)]; tensor softmax_20 = softmax(axis = softmax_20_axis_0, x = add_20)[name = tensor("softmax_20")]; tensor attn_output_81_transpose_x_0 = const()[name = tensor("attn_output_81_transpose_x_0"), val = tensor(false)]; tensor attn_output_81_transpose_y_0 = const()[name = tensor("attn_output_81_transpose_y_0"), val = tensor(false)]; tensor attn_output_81 = matmul(transpose_x = attn_output_81_transpose_x_0, transpose_y = attn_output_81_transpose_y_0, x = softmax_20, y = value_41)[name = tensor("attn_output_81")]; tensor var_1604_perm_0 = const()[name = tensor("op_1604_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1606 = const()[name = tensor("op_1606"), val = tensor([1, 2, -1])]; tensor var_1604 = transpose(perm = var_1604_perm_0, x = attn_output_81)[name = tensor("transpose_132")]; tensor var_1607 = reshape(shape = var_1606, x = var_1604)[name = tensor("op_1607")]; tensor input_249 = linear(bias = decoder_layers_10_self_attn_out_proj_bias, weight = decoder_layers_10_self_attn_out_proj_weight, x = var_1607)[name = tensor("linear_103")]; tensor input_251 = add(x = input_245, y = input_249)[name = tensor("input_251")]; tensor hidden_states_105_axes_0 = const()[name = tensor("hidden_states_105_axes_0"), val = tensor([-1])]; tensor hidden_states_105 = layer_norm(axes = hidden_states_105_axes_0, beta = decoder_layers_10_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_10_encoder_attn_layer_norm_weight, x = input_251)[name = tensor("hidden_states_105")]; tensor var_1631 = linear(bias = decoder_layers_10_encoder_attn_q_proj_bias, weight = decoder_layers_10_encoder_attn_q_proj_weight, x = hidden_states_105)[name = tensor("linear_104")]; tensor var_1632 = const()[name = tensor("op_1632"), val = tensor([1, 2, -1, 64])]; tensor var_1633 = reshape(shape = var_1632, x = var_1631)[name = tensor("op_1633")]; tensor query_43_perm_0 = const()[name = tensor("query_43_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_85 = linear(bias = decoder_layers_10_encoder_attn_k_proj_bias, weight = decoder_layers_10_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_105")]; tensor value_states_85 = linear(bias = decoder_layers_10_encoder_attn_v_proj_bias, weight = decoder_layers_10_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_106")]; tensor concat_34x = const()[name = tensor("concat_34x"), val = tensor([1, -1, 16, 64])]; tensor var_1642 = reshape(shape = concat_34x, x = key_states_85)[name = tensor("op_1642")]; tensor key_states_87_perm_0 = const()[name = tensor("key_states_87_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_35x = const()[name = tensor("concat_35x"), val = tensor([1, -1, 16, 64])]; tensor var_1645 = reshape(shape = concat_35x, x = value_states_85)[name = tensor("op_1645")]; tensor value_states_87_perm_0 = const()[name = tensor("value_states_87_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_43_interleave_0 = const()[name = tensor("key_43_interleave_0"), val = tensor(false)]; tensor key_states_87 = transpose(perm = key_states_87_perm_0, x = var_1642)[name = tensor("transpose_130")]; tensor key_43 = concat(axis = var_13, interleave = key_43_interleave_0, values = key_states_87)[name = tensor("key_43")]; tensor value_43_interleave_0 = const()[name = tensor("value_43_interleave_0"), val = tensor(false)]; tensor value_states_87 = transpose(perm = value_states_87_perm_0, x = var_1645)[name = tensor("transpose_129")]; tensor value_43 = concat(axis = var_13, interleave = value_43_interleave_0, values = value_states_87)[name = tensor("value_43")]; tensor var_1655_shape = shape(x = key_43)[name = tensor("op_1655_shape")]; tensor gather_23_indices_0 = const()[name = tensor("gather_23_indices_0"), val = tensor(2)]; tensor gather_23_axis_0 = const()[name = tensor("gather_23_axis_0"), val = tensor(0)]; tensor gather_23_batch_dims_0 = const()[name = tensor("gather_23_batch_dims_0"), val = tensor(0)]; tensor gather_23 = gather(axis = gather_23_axis_0, batch_dims = gather_23_batch_dims_0, indices = gather_23_indices_0, x = var_1655_shape)[name = tensor("gather_23")]; tensor concat_36_values0_0 = const()[name = tensor("concat_36_values0_0"), val = tensor(0)]; tensor concat_36_values1_0 = const()[name = tensor("concat_36_values1_0"), val = tensor(0)]; tensor concat_36_values2_0 = const()[name = tensor("concat_36_values2_0"), val = tensor(0)]; tensor concat_36_axis_0 = const()[name = tensor("concat_36_axis_0"), val = tensor(0)]; tensor concat_36_interleave_0 = const()[name = tensor("concat_36_interleave_0"), val = tensor(false)]; tensor concat_36 = concat(axis = concat_36_axis_0, interleave = concat_36_interleave_0, values = (concat_36_values0_0, concat_36_values1_0, concat_36_values2_0, gather_23))[name = tensor("concat_36")]; tensor attention_mask_47_begin_0 = const()[name = tensor("attention_mask_47_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_47_end_mask_0 = const()[name = tensor("attention_mask_47_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask_47 = slice_by_index(begin = attention_mask_47_begin_0, end = concat_36, end_mask = attention_mask_47_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_47")]; tensor query_43 = transpose(perm = query_43_perm_0, x = var_1633)[name = tensor("transpose_131")]; tensor mul_21 = mul(x = query_43, y = var_11)[name = tensor("mul_21")]; tensor matmul_21_transpose_y_0 = const()[name = tensor("matmul_21_transpose_y_0"), val = tensor(true)]; tensor matmul_21_transpose_x_0 = const()[name = tensor("matmul_21_transpose_x_0"), val = tensor(false)]; tensor matmul_21 = matmul(transpose_x = matmul_21_transpose_x_0, transpose_y = matmul_21_transpose_y_0, x = mul_21, y = key_43)[name = tensor("matmul_21")]; tensor add_21 = add(x = matmul_21, y = attention_mask_47)[name = tensor("add_21")]; tensor softmax_21_axis_0 = const()[name = tensor("softmax_21_axis_0"), val = tensor(-1)]; tensor softmax_21 = softmax(axis = softmax_21_axis_0, x = add_21)[name = tensor("softmax_21")]; tensor attn_output_85_transpose_x_0 = const()[name = tensor("attn_output_85_transpose_x_0"), val = tensor(false)]; tensor attn_output_85_transpose_y_0 = const()[name = tensor("attn_output_85_transpose_y_0"), val = tensor(false)]; tensor attn_output_85 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = softmax_21, y = value_43)[name = tensor("attn_output_85")]; tensor var_1661_perm_0 = const()[name = tensor("op_1661_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1663 = const()[name = tensor("op_1663"), val = tensor([1, 2, -1])]; tensor var_1661 = transpose(perm = var_1661_perm_0, x = attn_output_85)[name = tensor("transpose_128")]; tensor var_1664 = reshape(shape = var_1663, x = var_1661)[name = tensor("op_1664")]; tensor input_255 = linear(bias = decoder_layers_10_encoder_attn_out_proj_bias, weight = decoder_layers_10_encoder_attn_out_proj_weight, x = var_1664)[name = tensor("linear_107")]; tensor input_257 = add(x = input_251, y = input_255)[name = tensor("input_257")]; tensor input_259_axes_0 = const()[name = tensor("input_259_axes_0"), val = tensor([-1])]; tensor input_259 = layer_norm(axes = input_259_axes_0, beta = decoder_layers_10_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_10_final_layer_norm_weight, x = input_257)[name = tensor("input_259")]; tensor input_261 = linear(bias = decoder_layers_10_fc1_bias, weight = decoder_layers_10_fc1_weight, x = input_259)[name = tensor("linear_108")]; tensor input_263 = relu(x = input_261)[name = tensor("input_263")]; tensor input_267 = linear(bias = decoder_layers_10_fc2_bias, weight = decoder_layers_10_fc2_weight, x = input_263)[name = tensor("linear_109")]; tensor input_269 = add(x = input_257, y = input_267)[name = tensor("input_269")]; tensor hidden_states_111_axes_0 = const()[name = tensor("hidden_states_111_axes_0"), val = tensor([-1])]; tensor hidden_states_111 = layer_norm(axes = hidden_states_111_axes_0, beta = decoder_layers_11_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_11_self_attn_layer_norm_weight, x = input_269)[name = tensor("hidden_states_111")]; tensor var_1714 = linear(bias = decoder_layers_11_self_attn_q_proj_bias, weight = decoder_layers_11_self_attn_q_proj_weight, x = hidden_states_111)[name = tensor("linear_110")]; tensor var_1715 = const()[name = tensor("op_1715"), val = tensor([1, 2, -1, 64])]; tensor var_1716 = reshape(shape = var_1715, x = var_1714)[name = tensor("op_1716")]; tensor key_states_89 = linear(bias = decoder_layers_11_self_attn_k_proj_bias, weight = decoder_layers_11_self_attn_k_proj_weight, x = hidden_states_111)[name = tensor("linear_111")]; tensor value_states_89 = linear(bias = decoder_layers_11_self_attn_v_proj_bias, weight = decoder_layers_11_self_attn_v_proj_weight, x = hidden_states_111)[name = tensor("linear_112")]; tensor var_1724 = const()[name = tensor("op_1724"), val = tensor([1, 2, -1, 64])]; tensor var_1725 = reshape(shape = var_1724, x = key_states_89)[name = tensor("op_1725")]; tensor var_1727 = const()[name = tensor("op_1727"), val = tensor([1, 2, -1, 64])]; tensor var_1728 = reshape(shape = var_1727, x = value_states_89)[name = tensor("op_1728")]; tensor value_states_91_perm_0 = const()[name = tensor("value_states_91_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_45_interleave_0 = const()[name = tensor("key_45_interleave_0"), val = tensor(false)]; tensor const_134 = const()[name = tensor("const_134"), val = tensor(1)]; tensor key_45 = concat(axis = const_134, interleave = key_45_interleave_0, values = var_1725)[name = tensor("key_45")]; tensor value_45_interleave_0 = const()[name = tensor("value_45_interleave_0"), val = tensor(false)]; tensor value_states_91 = transpose(perm = value_states_91_perm_0, x = var_1728)[name = tensor("transpose_127")]; tensor value_45 = concat(axis = var_13, interleave = value_45_interleave_0, values = value_states_91)[name = tensor("value_45")]; tensor mul_22 = mul(x = var_1716, y = var_11)[name = tensor("mul_22")]; tensor matmul_22_transpose_y_0 = const()[name = tensor("matmul_22_transpose_y_0"), val = tensor(true)]; tensor matmul_22_transpose_x_0 = const()[name = tensor("matmul_22_transpose_x_0"), val = tensor(false)]; tensor transpose_94_perm_0 = const()[name = tensor("transpose_94_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_95_perm_0 = const()[name = tensor("transpose_95_perm_0"), val = tensor([0, 2, -3, -1])]; tensor transpose_95 = transpose(perm = transpose_95_perm_0, x = key_45)[name = tensor("transpose_125")]; tensor transpose_94 = transpose(perm = transpose_94_perm_0, x = mul_22)[name = tensor("transpose_126")]; tensor matmul_22 = matmul(transpose_x = matmul_22_transpose_x_0, transpose_y = matmul_22_transpose_y_0, x = transpose_94, y = transpose_95)[name = tensor("matmul_22")]; tensor add_22 = add(x = matmul_22, y = reshape_4)[name = tensor("add_22")]; tensor softmax_22_axis_0 = const()[name = tensor("softmax_22_axis_0"), val = tensor(-1)]; tensor softmax_22 = softmax(axis = softmax_22_axis_0, x = add_22)[name = tensor("softmax_22")]; tensor attn_output_89_transpose_x_0 = const()[name = tensor("attn_output_89_transpose_x_0"), val = tensor(false)]; tensor attn_output_89_transpose_y_0 = const()[name = tensor("attn_output_89_transpose_y_0"), val = tensor(false)]; tensor attn_output_89 = matmul(transpose_x = attn_output_89_transpose_x_0, transpose_y = attn_output_89_transpose_y_0, x = softmax_22, y = value_45)[name = tensor("attn_output_89")]; tensor var_1744_perm_0 = const()[name = tensor("op_1744_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1746 = const()[name = tensor("op_1746"), val = tensor([1, 2, -1])]; tensor var_1744 = transpose(perm = var_1744_perm_0, x = attn_output_89)[name = tensor("transpose_124")]; tensor var_1747 = reshape(shape = var_1746, x = var_1744)[name = tensor("op_1747")]; tensor input_273 = linear(bias = decoder_layers_11_self_attn_out_proj_bias, weight = decoder_layers_11_self_attn_out_proj_weight, x = var_1747)[name = tensor("linear_113")]; tensor input_275 = add(x = input_269, y = input_273)[name = tensor("input_275")]; tensor hidden_states_115_axes_0 = const()[name = tensor("hidden_states_115_axes_0"), val = tensor([-1])]; tensor hidden_states_115 = layer_norm(axes = hidden_states_115_axes_0, beta = decoder_layers_11_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_11_encoder_attn_layer_norm_weight, x = input_275)[name = tensor("hidden_states_115")]; tensor var_1771 = linear(bias = decoder_layers_11_encoder_attn_q_proj_bias, weight = decoder_layers_11_encoder_attn_q_proj_weight, x = hidden_states_115)[name = tensor("linear_114")]; tensor var_1772 = const()[name = tensor("op_1772"), val = tensor([1, 2, -1, 64])]; tensor var_1773 = reshape(shape = var_1772, x = var_1771)[name = tensor("op_1773")]; tensor query_perm_0 = const()[name = tensor("query_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_states_93 = linear(bias = decoder_layers_11_encoder_attn_k_proj_bias, weight = decoder_layers_11_encoder_attn_k_proj_weight, x = encoder_hidden_states)[name = tensor("linear_115")]; tensor value_states_93 = linear(bias = decoder_layers_11_encoder_attn_v_proj_bias, weight = decoder_layers_11_encoder_attn_v_proj_weight, x = encoder_hidden_states)[name = tensor("linear_116")]; tensor concat_37x = const()[name = tensor("concat_37x"), val = tensor([1, -1, 16, 64])]; tensor var_1782 = reshape(shape = concat_37x, x = key_states_93)[name = tensor("op_1782")]; tensor key_states_perm_0 = const()[name = tensor("key_states_perm_0"), val = tensor([0, 2, 1, 3])]; tensor concat_38x = const()[name = tensor("concat_38x"), val = tensor([1, -1, 16, 64])]; tensor var_1785 = reshape(shape = concat_38x, x = value_states_93)[name = tensor("op_1785")]; tensor value_states_perm_0 = const()[name = tensor("value_states_perm_0"), val = tensor([0, 2, 1, 3])]; tensor key_interleave_0 = const()[name = tensor("key_interleave_0"), val = tensor(false)]; tensor key_states = transpose(perm = key_states_perm_0, x = var_1782)[name = tensor("transpose_122")]; tensor key = concat(axis = var_13, interleave = key_interleave_0, values = key_states)[name = tensor("key")]; tensor value_interleave_0 = const()[name = tensor("value_interleave_0"), val = tensor(false)]; tensor value_states = transpose(perm = value_states_perm_0, x = var_1785)[name = tensor("transpose_121")]; tensor value = concat(axis = var_13, interleave = value_interleave_0, values = value_states)[name = tensor("value")]; tensor var_1795_shape = shape(x = key)[name = tensor("op_1795_shape")]; tensor gather_25_indices_0 = const()[name = tensor("gather_25_indices_0"), val = tensor(2)]; tensor gather_25_axis_0 = const()[name = tensor("gather_25_axis_0"), val = tensor(0)]; tensor gather_25_batch_dims_0 = const()[name = tensor("gather_25_batch_dims_0"), val = tensor(0)]; tensor gather_25 = gather(axis = gather_25_axis_0, batch_dims = gather_25_batch_dims_0, indices = gather_25_indices_0, x = var_1795_shape)[name = tensor("gather_25")]; tensor concat_39_values0_0 = const()[name = tensor("concat_39_values0_0"), val = tensor(0)]; tensor concat_39_values1_0 = const()[name = tensor("concat_39_values1_0"), val = tensor(0)]; tensor concat_39_values2_0 = const()[name = tensor("concat_39_values2_0"), val = tensor(0)]; tensor concat_39_axis_0 = const()[name = tensor("concat_39_axis_0"), val = tensor(0)]; tensor concat_39_interleave_0 = const()[name = tensor("concat_39_interleave_0"), val = tensor(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (concat_39_values0_0, concat_39_values1_0, concat_39_values2_0, gather_25))[name = tensor("concat_39")]; tensor attention_mask_begin_0 = const()[name = tensor("attention_mask_begin_0"), val = tensor([0, 0, 0, 0])]; tensor attention_mask_end_mask_0 = const()[name = tensor("attention_mask_end_mask_0"), val = tensor([true, true, true, false])]; tensor attention_mask = slice_by_index(begin = attention_mask_begin_0, end = concat_39, end_mask = attention_mask_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask")]; tensor query = transpose(perm = query_perm_0, x = var_1773)[name = tensor("transpose_123")]; tensor mul_23 = mul(x = query, y = var_11)[name = tensor("mul_23")]; tensor matmul_23_transpose_y_0 = const()[name = tensor("matmul_23_transpose_y_0"), val = tensor(true)]; tensor matmul_23_transpose_x_0 = const()[name = tensor("matmul_23_transpose_x_0"), val = tensor(false)]; tensor matmul_23 = matmul(transpose_x = matmul_23_transpose_x_0, transpose_y = matmul_23_transpose_y_0, x = mul_23, y = key)[name = tensor("matmul_23")]; tensor add_23 = add(x = matmul_23, y = attention_mask)[name = tensor("add_23")]; tensor softmax_23_axis_0 = const()[name = tensor("softmax_23_axis_0"), val = tensor(-1)]; tensor softmax_23 = softmax(axis = softmax_23_axis_0, x = add_23)[name = tensor("softmax_23")]; tensor attn_output_93_transpose_x_0 = const()[name = tensor("attn_output_93_transpose_x_0"), val = tensor(false)]; tensor attn_output_93_transpose_y_0 = const()[name = tensor("attn_output_93_transpose_y_0"), val = tensor(false)]; tensor attn_output_93 = matmul(transpose_x = attn_output_93_transpose_x_0, transpose_y = attn_output_93_transpose_y_0, x = softmax_23, y = value)[name = tensor("attn_output_93")]; tensor var_1801_perm_0 = const()[name = tensor("op_1801_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1803 = const()[name = tensor("op_1803"), val = tensor([1, 2, -1])]; tensor var_1801 = transpose(perm = var_1801_perm_0, x = attn_output_93)[name = tensor("transpose_120")]; tensor var_1804 = reshape(shape = var_1803, x = var_1801)[name = tensor("op_1804")]; tensor input_279 = linear(bias = decoder_layers_11_encoder_attn_out_proj_bias, weight = decoder_layers_11_encoder_attn_out_proj_weight, x = var_1804)[name = tensor("linear_117")]; tensor input_281 = add(x = input_275, y = input_279)[name = tensor("input_281")]; tensor input_283_axes_0 = const()[name = tensor("input_283_axes_0"), val = tensor([-1])]; tensor input_283 = layer_norm(axes = input_283_axes_0, beta = decoder_layers_11_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_11_final_layer_norm_weight, x = input_281)[name = tensor("input_283")]; tensor input_285 = linear(bias = decoder_layers_11_fc1_bias, weight = decoder_layers_11_fc1_weight, x = input_283)[name = tensor("linear_118")]; tensor input_287 = relu(x = input_285)[name = tensor("input_287")]; tensor input_291 = linear(bias = decoder_layers_11_fc2_bias, weight = decoder_layers_11_fc2_weight, x = input_287)[name = tensor("linear_119")]; tensor input_293 = add(x = input_281, y = input_291)[name = tensor("input_293")]; tensor var_1838_axes_0 = const()[name = tensor("op_1838_axes_0"), val = tensor([-1])]; tensor var_1838 = layer_norm(axes = var_1838_axes_0, beta = decoder_layer_norm_bias, epsilon = var_9, gamma = decoder_layer_norm_weight, x = input_293)[name = tensor("op_1838")]; tensor var_1898_begin_0 = const()[name = tensor("op_1898_begin_0"), val = tensor([0, -1, 0])]; tensor var_1898_end_0 = const()[name = tensor("op_1898_end_0"), val = tensor([1, 2, 1024])]; tensor var_1898_end_mask_0 = const()[name = tensor("op_1898_end_mask_0"), val = tensor([true, true, true])]; tensor var_1898 = slice_by_index(begin = var_1898_begin_0, end = var_1898_end_0, end_mask = var_1898_end_mask_0, x = var_1838)[name = tensor("op_1898")]; tensor linear_120_bias_0 = const()[name = tensor("linear_120_bias_0"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(1859891008)))]; tensor logits = linear(bias = linear_120_bias_0, weight = decoder_embed_tokens_weight, x = var_1898)[name = tensor("linear_120")]; tensor var_1908_axis_0 = const()[name = tensor("op_1908_axis_0"), val = tensor(0)]; tensor transpose_96_perm_0 = const()[name = tensor("transpose_96_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_97_perm_0 = const()[name = tensor("transpose_97_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_98_perm_0 = const()[name = tensor("transpose_98_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_99_perm_0 = const()[name = tensor("transpose_99_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_100_perm_0 = const()[name = tensor("transpose_100_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_101_perm_0 = const()[name = tensor("transpose_101_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_102_perm_0 = const()[name = tensor("transpose_102_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_103_perm_0 = const()[name = tensor("transpose_103_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_104_perm_0 = const()[name = tensor("transpose_104_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_105_perm_0 = const()[name = tensor("transpose_105_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_106_perm_0 = const()[name = tensor("transpose_106_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_107_perm_0 = const()[name = tensor("transpose_107_perm_0"), val = tensor([0, 2, 1, 3])]; tensor transpose_107 = transpose(perm = transpose_107_perm_0, x = key_45)[name = tensor("transpose_108")]; tensor transpose_106 = transpose(perm = transpose_106_perm_0, x = key_41)[name = tensor("transpose_109")]; tensor transpose_105 = transpose(perm = transpose_105_perm_0, x = key_37)[name = tensor("transpose_110")]; tensor transpose_104 = transpose(perm = transpose_104_perm_0, x = key_33)[name = tensor("transpose_111")]; tensor transpose_103 = transpose(perm = transpose_103_perm_0, x = key_29)[name = tensor("transpose_112")]; tensor transpose_102 = transpose(perm = transpose_102_perm_0, x = key_25)[name = tensor("transpose_113")]; tensor transpose_101 = transpose(perm = transpose_101_perm_0, x = key_21)[name = tensor("transpose_114")]; tensor transpose_100 = transpose(perm = transpose_100_perm_0, x = key_17)[name = tensor("transpose_115")]; tensor transpose_99 = transpose(perm = transpose_99_perm_0, x = key_13)[name = tensor("transpose_116")]; tensor transpose_98 = transpose(perm = transpose_98_perm_0, x = key_9)[name = tensor("transpose_117")]; tensor transpose_97 = transpose(perm = transpose_97_perm_0, x = key_5)[name = tensor("transpose_118")]; tensor transpose_96 = transpose(perm = transpose_96_perm_0, x = key_1)[name = tensor("transpose_119")]; tensor past_self_key = stack(axis = var_1908_axis_0, values = (transpose_96, transpose_97, transpose_98, transpose_99, transpose_100, transpose_101, transpose_102, transpose_103, transpose_104, transpose_105, transpose_106, transpose_107))[name = tensor("op_1908")]; tensor var_1911_axis_0 = const()[name = tensor("op_1911_axis_0"), val = tensor(0)]; tensor past_self_value = stack(axis = var_1911_axis_0, values = (value_1, value_5, value_9, value_13, value_17, value_21, value_25, value_29, value_33, value_37, value_41, value_45))[name = tensor("op_1911")]; tensor var_1914_axis_0 = const()[name = tensor("op_1914_axis_0"), val = tensor(0)]; tensor past_cross_key = stack(axis = var_1914_axis_0, values = (key_3, key_7, key_11, key_15, key_19, key_23, key_27, key_31, key_35, key_39, key_43, key))[name = tensor("op_1914")]; tensor var_1917_axis_0 = const()[name = tensor("op_1917_axis_0"), val = tensor(0)]; tensor past_cross_value = stack(axis = var_1917_axis_0, values = (value_3, value_7, value_11, value_15, value_19, value_23, value_27, value_31, value_35, value_39, value_43, value))[name = tensor("op_1917")]; } -> (logits, past_self_key, past_self_value, past_cross_key, past_cross_value); }